"tests/vscode:/vscode.git/clone" did not exist on "3fd1fb63efb6c96f30237b12e2816b4f2c5323d0"
glm.py 1.01 KB
Newer Older
1
# SPDX-License-Identifier: Apache-2.0
2
3
4
5
"""Inference-only HF format GLM-4 model compatible with THUDM weights."""
from vllm.config import VllmConfig
from vllm.model_executor.models.llama import LlamaForCausalLM

6
from .interfaces import SupportsV0Only
7
8
9
from .utils import PPMissingLayer


10
class GlmForCausalLM(LlamaForCausalLM, SupportsV0Only):
11
12
13
14
15
16
17
18
19
20
21
22
23

    def __init__(self, *, vllm_config: VllmConfig, prefix: str = ""):
        super().__init__(vllm_config=vllm_config, prefix=prefix)
        # Hack Llama model to fit HF format GLM implementation
        # Attention difference between GLM and Llama:
        # 1. Half partial rotary_dim and no Neox style.
        # 2. There is no bias for o_proj in attention
        for layer in self.model.layers:
            if not isinstance(layer, PPMissingLayer):
                layer.self_attn.rotary_emb.rotary_dim //= 2
                layer.self_attn.rotary_emb.is_neox_style = False
                layer.self_attn.o_proj.bias = None
                layer.self_attn.o_proj.skip_bias_add = True