[Bugfix] Correct adapter usage for cohere and jamba (#8292)

f9b4a2d4 · Vladislav Kruglikov · GitHub · 58fcc854 · f9b4a2d4 · f9b4a2d4
Unverified Commit f9b4a2d4 authored Sep 09, 2024 by Vladislav Kruglikov Committed by GitHub Sep 09, 2024
Hide whitespace changes
Inline Side-by-side

Showing with 6 additions and 3 deletions

vllm/model_executor/models/commandr.py vllm/model_executor/models/commandr.py +3 -2

vllm/model_executor/models/jamba.py vllm/model_executor/models/jamba.py +3 -1

No files found.
--- a/vllm/model_executor/models/commandr.py
+++ b/vllm/model_executor/models/commandr.py
@@ -47,6 +47,8 @@ from vllm.model_executor.sampling_metadata import SamplingMetadata
 from vllm.model_executor.utils import set_weight_attrs
 from vllm.sequence import IntermediateTensors
+from .interfaces import SupportsLoRA
 @torch.compile
 def layer_norm_func(hidden_states, weight, variance_epsilon):
@@ -292,8 +294,7 @@ class CohereModel(nn.Module):
        return hidden_states
-class CohereForCausalLM(nn.Module):
+class CohereForCausalLM(nn.Module, SupportsLoRA):
    packed_modules_mapping = {
        "qkv_proj": [
            "q_proj",

--- a/vllm/model_executor/models/jamba.py
+++ b/vllm/model_executor/models/jamba.py
@@ -38,6 +38,8 @@ from vllm.sequence import IntermediateTensors
 from vllm.worker.model_runner import (_BATCH_SIZES_TO_CAPTURE,
                                      _get_graph_batch_size)
+from .interfaces import SupportsLoRA
 KVCache = Tuple[torch.Tensor, torch.Tensor]
@@ -539,7 +541,7 @@ class JambaModel(nn.Module):
        return hidden_states
-class JambaForCausalLM(nn.Module, HasInnerState):
+class JambaForCausalLM(nn.Module, HasInnerState, SupportsLoRA):
    packed_modules_mapping = {
        "qkv_proj": [
            "q_proj",