qwen2_cls.py 3.73 KB
Newer Older
1
2
3
4
5
6
# Adapted from
# https://huggingface.co/Qwen/Qwen2.5-Math-RM-72B/blob/main/modeling_qwen2_rm.py
# Copyright 2024 Kakao Corp. (Kanana-X Team)
# Copyright 2024 The Qwen team.
# Copyright 2023 The vLLM team.
"""Inference-only Qwen2-Classification model compatible with HF weights."""
7
from typing import Iterable, List, Optional, Set, Tuple
8
9
10
11
12

import torch
from torch import nn

from vllm.attention import AttentionMetadata
13
from vllm.config import VllmConfig
14
15
16
17
18
19
from vllm.model_executor.layers.linear import RowParallelLinear
from vllm.model_executor.layers.pooler import Pooler, PoolingType
from vllm.model_executor.models.qwen2 import Qwen2Model
from vllm.model_executor.pooling_metadata import PoolingMetadata
from vllm.sequence import IntermediateTensors, PoolerOutput

20
from .interfaces import SupportsLoRA, SupportsPP
21
from .utils import AutoWeightsLoader, maybe_prefix
22
23


24
class Qwen2ForSequenceClassification(nn.Module, SupportsLoRA, SupportsPP):
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
    packed_modules_mapping = {
        "qkv_proj": [
            "q_proj",
            "k_proj",
            "v_proj",
        ],
        "gate_up_proj": [
            "gate_proj",
            "up_proj",
        ],
    }

    # LoRA specific attributes
    supported_lora_modules = [
        "qkv_proj",
        "o_proj",
        "gate_up_proj",
        "down_proj",
    ]
    embedding_modules = {}
    embedding_padding_modules = []

47
    def __init__(self, *, vllm_config: VllmConfig, prefix: str = ""):
48
49
50
51
52
        super().__init__()
        config = vllm_config.model_config.hf_config
        quant_config = vllm_config.quant_config
        lora_config = vllm_config.lora_config
        pooler_config = vllm_config.model_config.pooler_config
53
54
55
56
57

        self.config = config
        self.lora_config = lora_config

        self.quant_config = quant_config
58
59
        self.model = Qwen2Model(vllm_config=vllm_config,
                                prefix=maybe_prefix(prefix, "model"))
60

61
62
        # hidden_states from Qwen2Model has been reduced,
        # the input of score layer is not parallelized.
63
64
        self.score = RowParallelLinear(config.hidden_size,
                                       config.num_labels,
65
66
67
68
                                       quant_config=quant_config,
                                       input_is_parallel=False,
                                       bias=False,
                                       prefix=maybe_prefix(prefix, "score"))
69
70
71
72
73
        self._pooler = Pooler.from_config_with_defaults(
            pooler_config,
            pooling_type=PoolingType.LAST,
            normalize=False,
            softmax=True)
74

75
76
77
    def get_input_embeddings(self, input_ids: torch.Tensor) -> torch.Tensor:
        return self.model.get_input_embeddings(input_ids)

78
79
80
81
82
83
84
    def forward(
        self,
        input_ids: torch.Tensor,
        positions: torch.Tensor,
        kv_caches: List[torch.Tensor],
        attn_metadata: AttentionMetadata,
        intermediate_tensors: Optional[IntermediateTensors] = None,
85
        inputs_embeds: Optional[torch.Tensor] = None,
86
87
    ) -> torch.Tensor:
        hidden_states = self.model(input_ids, positions, kv_caches,
88
89
                                   attn_metadata, intermediate_tensors,
                                   inputs_embeds)
90
91
92
93
94
95
96
97
98
99
        logits, _ = self.score(hidden_states)
        return logits

    def pooler(
        self,
        hidden_states: torch.Tensor,
        pooling_metadata: PoolingMetadata,
    ) -> Optional[PoolerOutput]:
        return self._pooler(hidden_states, pooling_metadata)

100
101
    def load_weights(self, weights: Iterable[Tuple[str,
                                                   torch.Tensor]]) -> Set[str]:
102
103
        loader = AutoWeightsLoader(self,
                                   ignore_unexpected_prefixes=["lm_head."])
104
        return loader.load_weights(weights)