[Bugfix] Fix GLM rotary_dim issue and support v1

Fix glm4.py residual bug

[Bugfix] Fix GLM rotary_dim issue and support v1
Fix glm4.py residual bug
7f3dec12 · zhuwenwen · b956dfd6 · 7f3dec12 · 7f3dec12
Commit 7f3dec12 authored Apr 22, 2025 by zhuwenwen
Hide whitespace changes
Inline Side-by-side

Showing with 4 additions and 5 deletions

vllm/model_executor/models/glm.py vllm/model_executor/models/glm.py +2 -3

vllm/model_executor/models/glm4.py vllm/model_executor/models/glm4.py +2 -2

No files found.
--- a/vllm/model_executor/models/glm.py
+++ b/vllm/model_executor/models/glm.py
@@ -3,13 +3,13 @@
 from vllm.config import VllmConfig
 from vllm.model_executor.models.llama import LlamaForCausalLM
-from .interfaces import SupportsV0Only
 from .utils import PPMissingLayer
-class GlmForCausalLM(LlamaForCausalLM, SupportsV0Only):
+class GlmForCausalLM(LlamaForCausalLM):
    def __init__(self, *, vllm_config: VllmConfig, prefix: str = ""):
+        vllm_config.model_config.hf_config.partial_rotary_factor = 0.5
        super().__init__(vllm_config=vllm_config, prefix=prefix)
        # Hack Llama model to fit HF format GLM implementation
        # Attention difference between GLM and Llama:
@@ -17,7 +17,6 @@ class GlmForCausalLM(LlamaForCausalLM, SupportsV0Only):
        # 2. There is no bias for o_proj in attention
        for layer in self.model.layers:
            if not isinstance(layer, PPMissingLayer):
-                layer.self_attn.rotary_emb.rotary_dim //= 2
                layer.self_attn.rotary_emb.is_neox_style = False
                layer.self_attn.o_proj.bias = None
                layer.self_attn.o_proj.skip_bias_add = True
--- a/vllm/model_executor/models/glm4.py
+++ b/vllm/model_executor/models/glm4.py
@@ -200,8 +200,8 @@ class Glm4DecoderLayer(nn.Module):
        hidden_states = self.post_self_attn_layernorm(hidden_states)
        # Fully Connected
-        hidden_states, residual = self.post_attention_layernorm(
+        residual = hidden_states
-             hidden_states, residual)
+        hidden_states = self.post_attention_layernorm(hidden_states)
        hidden_states = self.mlp(hidden_states)
        hidden_states = self.post_mlp_layernorm(hidden_states)