fix: use weights from base_layer (#2141)

2025-09-12 04:44:52 +00:00 · 2024-07-01 06:58:40 -04:00 · 2024-07-01 06:58:40 -04:00 · 3e02d4fdbf
commit 3e02d4fdbf
parent 03691f6d34
1 changed files with 3 additions and 1 deletions
--- a/server/text_generation_server/models/custom_modeling/flash_llama_modeling.py
+++ b/server/text_generation_server/models/custom_modeling/flash_llama_modeling.py
@ -309,7 +309,9 @@ class LlamaMLP(nn.Module):
                dtype=hidden_states.dtype,
                device="cuda",
            )
-            _custom_C.LLMM_Silu(self.gate_up_proj.linear.weight, hidden_states, out, 8)
+            _custom_C.LLMM_Silu(
+                self.gate_up_proj.base_layer.linear.weight, hidden_states, out, 8
+            )
            return self.down_proj(out, adapter_data)
        else:
            gate_up_states = self.gate_up_proj(hidden_states, adapter_data)