[MODEL] LoRA support for Jamba model (#11209)
Signed-off-by: Erez Schwartz <erezs@ai21.com>
This commit is contained in:
@@ -4,6 +4,7 @@ from typing import Dict, List, TypedDict
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
import safetensors
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
from huggingface_hub import snapshot_download
|
||||
@@ -169,6 +170,29 @@ def mixtral_lora_files_all_target_modules():
|
||||
return snapshot_download(repo_id="dyang415/mixtral-lora-v0")
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def jamba_lora_files():
|
||||
# some of the adapters have unnecessary weights for serving,
|
||||
# hence we remove them
|
||||
def remove_unnecessary_weights(path):
|
||||
lora_path = f"{adapter_path}/adapter_model.safetensors"
|
||||
tensors = safetensors.torch.load_file(lora_path)
|
||||
nonlora_keys = []
|
||||
for k in list(tensors.keys()):
|
||||
if "lora" not in k:
|
||||
nonlora_keys.append(k)
|
||||
for k in nonlora_keys:
|
||||
del tensors[k]
|
||||
safetensors.torch.save_file(tensors, lora_path)
|
||||
|
||||
adapter_path = snapshot_download(
|
||||
repo_id=
|
||||
"hf-100/Jamba-1.5-mini-Spellbound-StoryWriter-0.1-6583896-ckpt53-lora")
|
||||
|
||||
remove_unnecessary_weights(adapter_path)
|
||||
return adapter_path
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def gemma_lora_files():
|
||||
return snapshot_download(repo_id="wskwon/gemma-7b-test-lora")
|
||||
|
||||
54
tests/lora/test_jamba.py
Normal file
54
tests/lora/test_jamba.py
Normal file
@@ -0,0 +1,54 @@
|
||||
from typing import List
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
|
||||
import vllm
|
||||
from vllm.lora.request import LoRARequest
|
||||
|
||||
MODEL_PATH = "ai21labs/AI21-Jamba-1.5-Mini"
|
||||
|
||||
MAX_TOKENS = 40
|
||||
|
||||
|
||||
def do_sample(llm: vllm.LLM, lora_path: str, lora_id: int,
|
||||
prompts: List[str]) -> List[str]:
|
||||
|
||||
sampling_params = vllm.SamplingParams(temperature=0, max_tokens=MAX_TOKENS)
|
||||
outputs = llm.generate(
|
||||
prompts,
|
||||
sampling_params,
|
||||
lora_request=LoRARequest(str(lora_id), lora_id, lora_path)
|
||||
if lora_id else None)
|
||||
# Print the outputs.
|
||||
generated_texts: List[str] = []
|
||||
for output in outputs:
|
||||
prompt = output.prompt
|
||||
generated_text = output.outputs[0].text.strip()
|
||||
generated_texts.append(generated_text)
|
||||
print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
|
||||
return generated_texts
|
||||
|
||||
|
||||
@pytest.mark.parametrize("tp_size", [4])
|
||||
def test_jamba_lora(jamba_lora_files, tp_size):
|
||||
"""Original test, the LoRA model has the common target modules, not all"""
|
||||
if torch.cuda.device_count() < tp_size:
|
||||
pytest.skip(f"Not enough GPUs for tensor parallelism {tp_size}")
|
||||
|
||||
prompts = ["Write a story about a sheep and a goat."]
|
||||
|
||||
llm = vllm.LLM(
|
||||
MODEL_PATH,
|
||||
enable_lora=True,
|
||||
max_num_seqs=16,
|
||||
max_loras=4,
|
||||
distributed_executor_backend="ray",
|
||||
tensor_parallel_size=tp_size,
|
||||
)
|
||||
|
||||
expected_jamba_output = [
|
||||
"""Once upon a time, in a lush green meadow, there lived a sheep named Clara and a goat named Billy. Clara was a gentle creature, always nibbling on the soft grass and humming""" # noqa: E501
|
||||
]
|
||||
assert do_sample(llm, jamba_lora_files, lora_id=1,
|
||||
prompts=prompts) == expected_jamba_output
|
||||
Reference in New Issue
Block a user