Add support for Deepseek V2 (#2224)

Deepseek V2 is a MoE model from Deepseek. Relevant variations compared to other models: - Grouped top-K in expert selection. - mscale in yarn is calculated using the `mscale` and `mscale_all_dim` configuration options. - `mscale_all_dim` is also used in scaling attention softmax. - Permuting of the query/key representations before applying rotary embeddings. - Some projections cannot be sharded (`q_a_proj`, `kv_a_proj_with_mqa`). So, we need weight loads that supports quantized weights. To this end `{Weights,WeightLoader}.get_weight` was added. - The query/key head dimensionality differs from that of the value, so we need to pad during attention. - Heads with size 192, needs an extension to our paged attention fork and we need to ensure that the KV cache is allocated with the correct size. - Shared experts.

Add support for Deepseek V2 (#2224)
Deepseek V2 is a MoE model from Deepseek. Relevant variations compared to other models: - Grouped top-K in expert selection. - mscale in yarn is calculated using the `mscale` and `mscale_all_dim` configuration options. - `mscale_all_dim` is also used in scaling attention softmax. - Permuting of the query/key representations before applying rotary embeddings. - Some projections cannot be sharded (`q_a_proj`, `kv_a_proj_with_mqa`). So, we need weight loads that supports quantized weights. To this end `{Weights,WeightLoader}.get_weight` was added. - The query/key head dimensionality differs from that of the value, so we need to pad during attention. - Heads with size 192, needs an extension to our paged attention fork and we need to ensure that the KV cache is allocated with the correct size. - Shared experts.
e52be9bb · Daniël de Kok · GitHub · 68a9685f · e52be9bb · e52be9bb
Unverified Commit e52be9bb authored Jul 19, 2024 by Daniël de Kok Committed by GitHub Jul 19, 2024
14 changed files
--- a/docs/source/supported_models.md
+++ b/docs/source/supported_models.md
@@ -5,6 +5,7 @@ Text Generation Inference enables serving optimized models on specific hardware
 ## Supported Models
+- [Deepseek V2](https://huggingface.co/deepseek-ai/DeepSeek-V2)
 - [Idefics 2](https://huggingface.co/HuggingFaceM4/idefics2-8b) (Multimodal)
 - [Llava Next (1.6)](https://huggingface.co/llava-hf/llava-v1.6-vicuna-13b-hf) (Multimodal)
 - [Llama](https://huggingface.co/meta-llama/Meta-Llama-3-8B-Instruct)

--- a/integration-tests/models/__snapshots__/test_flash_deepseek_v2/test_flash_deepseek_v2.json
+++ b/integration-tests/models/__snapshots__/test_flash_deepseek_v2/test_flash_deepseek_v2.json
+{
+  "details": {
+    "best_of_sequences": null,
+    "finish_reason": "length",
+    "generated_tokens": 10,
+    "prefill": [
+      {
+        "id": 100000,
+        "logprob": null,
+        "text": "<｜begin▁of▁sentence｜>"
+      },
+      {
+        "id": 3533,
+        "logprob": -9.625,
+        "text": "Test"
+      },
+      {
+        "id": 3102,
+        "logprob": -11.1875,
+        "text": " request"
+      }
+    ],
+    "seed": null,
+    "tokens": [
+      {
+        "id": 185,
+        "logprob": -1.5546875,
+        "special": false,
+        "text": "\n"
+      },
+      {
+        "id": 549,
+        "logprob": -2.84375,
+        "special": false,
+        "text": "The"
+      },
+      {
+        "id": 1727,
+        "logprob": -2.34375,
+        "special": false,
+        "text": " test"
+      },
+      {
+        "id": 3102,
+        "logprob": -0.8359375,
+        "special": false,
+        "text": " request"
+      },
+      {
+        "id": 317,
+        "logprob": -1.0859375,
+        "special": false,
+        "text": " is"
+      },
+      {
+        "id": 254,
+        "logprob": -1.5390625,
+        "special": false,
+        "text": " the"
+      },
+      {
+        "id": 1022,
+        "logprob": -1.1875,
+        "special": false,
+        "text": " first"
+      },
+      {
+        "id": 3458,
+        "logprob": -0.35546875,
+        "special": false,
+        "text": " step"
+      },
+      {
+        "id": 279,
+        "logprob": -0.8828125,
+        "special": false,
+        "text": " in"
+      },
+      {
+        "id": 254,
+        "logprob": -0.71484375,
+        "special": false,
+        "text": " the"
+      }
+    ],
+    "top_tokens": null
+  },
+  "generated_text": "\nThe test request is the first step in the"
+}
--- a/integration-tests/models/__snapshots__/test_flash_deepseek_v2/test_flash_deepseek_v2_all_params.json
+++ b/integration-tests/models/__snapshots__/test_flash_deepseek_v2/test_flash_deepseek_v2_all_params.json
+{
+  "details": {
+    "best_of_sequences": null,
+    "finish_reason": "length",
+    "generated_tokens": 10,
+    "prefill": [
+      {
+        "id": 100000,
+        "logprob": null,
+        "text": "<｜begin▁of▁sentence｜>"
+      },
+      {
+        "id": 3533,
+        "logprob": -9.625,
+        "text": "Test"
+      },
+      {
+        "id": 3102,
+        "logprob": -11.1875,
+        "text": " request"
+      }
+    ],
+    "seed": 0,
+    "tokens": [
+      {
+        "id": 2143,
+        "logprob": -1.828125,
+        "special": false,
+        "text": " sent"
+      },
+      {
+        "id": 10081,
+        "logprob": -0.36914062,
+        "special": false,
+        "text": " successfully"
+      },
+      {
+        "id": 13,
+        "logprob": 0.0,
+        "special": false,
+        "text": "."
+      },
+      {
+        "id": 185,
+        "logprob": 0.0,
+        "special": false,
+        "text": "\n"
+      },
+      {
+        "id": 1380,
+        "logprob": -0.38671875,
+        "special": false,
+        "text": "We"
+      },
+      {
+        "id": 543,
+        "logprob": -0.12695312,
+        "special": false,
+        "text": " will"
+      },
+      {
+        "id": 752,
+        "logprob": -0.20117188,
+        "special": false,
+        "text": " get"
+      },
+      {
+        "id": 279,
+        "logprob": 0.0,
+        "special": false,
+        "text": " in"
+      },
+      {
+        "id": 5402,
+        "logprob": 0.0,
+        "special": false,
+        "text": " touch"
+      },
+      {
+        "id": 366,
+        "logprob": 0.0,
+        "special": false,
+        "text": " with"
+      }
+    ],
+    "top_tokens": null
+  },
+  "generated_text": "Test request sent successfully.\nWe will get in touch with"
+}
--- a/integration-tests/models/__snapshots__/test_flash_deepseek_v2/test_flash_deepseek_v2_load.json
+++ b/integration-tests/models/__snapshots__/test_flash_deepseek_v2/test_flash_deepseek_v2_load.json
+[
+  {
+    "details": {
+      "best_of_sequences": null,
+      "finish_reason": "length",
+      "generated_tokens": 10,
+      "prefill": [
+        {
+          "id": 100000,
+          "logprob": null,
+          "text": "<｜begin▁of▁sentence｜>"
+        },
+        {
+          "id": 3533,
+          "logprob": -9.625,
+          "text": "Test"
+        },
+        {
+          "id": 3102,
+          "logprob": -11.1875,
+          "text": " request"
+        }
+      ],
+      "seed": null,
+      "tokens": [
+        {
+          "id": 185,
+          "logprob": -1.5546875,
+          "special": false,
+          "text": "\n"
+        },
+        {
+          "id": 549,
+          "logprob": -2.8125,
+          "special": false,
+          "text": "The"
+        },
+        {
+          "id": 1727,
+          "logprob": -2.375,
+          "special": false,
+          "text": " test"
+        },
+        {
+          "id": 3102,
+          "logprob": -0.890625,
+          "special": false,
+          "text": " request"
+        },
+        {
+          "id": 317,
+          "logprob": -1.1484375,
+          "special": false,
+          "text": " is"
+        },
+        {
+          "id": 245,
+          "logprob": -1.5390625,
+          "special": false,
+          "text": " a"
+        },
+        {
+          "id": 3102,
+          "logprob": -2.609375,
+          "special": false,
+          "text": " request"
+        },
+        {
+          "id": 327,
+          "logprob": -0.75,
+          "special": false,
+          "text": " for"
+        },
+        {
+          "id": 245,
+          "logprob": -1.1171875,
+          "special": false,
+          "text": " a"
+        },
+        {
+          "id": 1727,
+          "logprob": -0.90625,
+          "special": false,
+          "text": " test"
+        }
+      ],
+      "top_tokens": null
+    },
+    "generated_text": "\nThe test request is a request for a test"
+  },
+  {
+    "details": {
+      "best_of_sequences": null,
+      "finish_reason": "length",
+      "generated_tokens": 10,
+      "prefill": [
+        {
+          "id": 100000,
+          "logprob": null,
+          "text": "<｜begin▁of▁sentence｜>"
+        },
+        {
+          "id": 3533,
+          "logprob": -9.625,
+          "text": "Test"
+        },
+        {
+          "id": 3102,
+          "logprob": -11.25,
+          "text": " request"
+        }
+      ],
+      "seed": null,
+      "tokens": [
+        {
+          "id": 185,
+          "logprob": -1.5546875,
+          "special": false,
+          "text": "\n"
+        },
+        {
+          "id": 549,
+          "logprob": -2.8125,
+          "special": false,
+          "text": "The"
+        },
+        {
+          "id": 1727,
+          "logprob": -2.375,
+          "special": false,
+          "text": " test"
+        },
+        {
+          "id": 3102,
+          "logprob": -0.890625,
+          "special": false,
+          "text": " request"
+        },
+        {
+          "id": 317,
+          "logprob": -1.1484375,
+          "special": false,
+          "text": " is"
+        },
+        {
+          "id": 245,
+          "logprob": -1.5390625,
+          "special": false,
+          "text": " a"
+        },
+        {
+          "id": 3102,
+          "logprob": -2.609375,
+          "special": false,
+          "text": " request"
+        },
+        {
+          "id": 327,
+          "logprob": -0.75,
+          "special": false,
+          "text": " for"
+        },
+        {
+          "id": 245,
+          "logprob": -1.1171875,
+          "special": false,
+          "text": " a"
+        },
+        {
+          "id": 1727,
+          "logprob": -0.90625,
+          "special": false,
+          "text": " test"
+        }
+      ],
+      "top_tokens": null
+    },
+    "generated_text": "\nThe test request is a request for a test"
+  },
+  {
+    "details": {
+      "best_of_sequences": null,
+      "finish_reason": "length",
+      "generated_tokens": 10,
+      "prefill": [
+        {
+          "id": 100000,
+          "logprob": null,
+          "text": "<｜begin▁of▁sentence｜>"
+        },
+        {
+          "id": 3533,
+          "logprob": -9.625,
+          "text": "Test"
+        },
+        {
+          "id": 3102,
+          "logprob": -11.25,
+          "text": " request"
+        }
+      ],
+      "seed": null,
+      "tokens": [
+        {
+          "id": 185,
+          "logprob": -1.5546875,
+          "special": false,
+          "text": "\n"
+        },
+        {
+          "id": 549,
+          "logprob": -2.8125,
+          "special": false,
+          "text": "The"
+        },
+        {
+          "id": 1727,
+          "logprob": -2.375,
+          "special": false,
+          "text": " test"
+        },
+        {
+          "id": 3102,
+          "logprob": -0.890625,
+          "special": false,
+          "text": " request"
+        },
+        {
+          "id": 317,
+          "logprob": -1.1484375,
+          "special": false,
+          "text": " is"
+        },
+        {
+          "id": 245,
+          "logprob": -1.5390625,
+          "special": false,
+          "text": " a"
+        },
+        {
+          "id": 3102,
+          "logprob": -2.609375,
+          "special": false,
+          "text": " request"
+        },
+        {
+          "id": 327,
+          "logprob": -0.75,
+          "special": false,
+          "text": " for"
+        },
+        {
+          "id": 245,
+          "logprob": -1.1171875,
+          "special": false,
+          "text": " a"
+        },
+        {
+          "id": 1727,
+          "logprob": -0.90625,
+          "special": false,
+          "text": " test"
+        }
+      ],
+      "top_tokens": null
+    },
+    "generated_text": "\nThe test request is a request for a test"
+  },
+  {
+    "details": {
+      "best_of_sequences": null,
+      "finish_reason": "length",
+      "generated_tokens": 10,
+      "prefill": [
+        {
+          "id": 100000,
+          "logprob": null,
+          "text": "<｜begin▁of▁sentence｜>"
+        },
+        {
+          "id": 3533,
+          "logprob": -9.625,
+          "text": "Test"
+        },
+        {
+          "id": 3102,
+          "logprob": -11.25,
+          "text": " request"
+        }
+      ],
+      "seed": null,
+      "tokens": [
+        {
+          "id": 185,
+          "logprob": -1.5546875,
+          "special": false,
+          "text": "\n"
+        },
+        {
+          "id": 549,
+          "logprob": -2.8125,
+          "special": false,
+          "text": "The"
+        },
+        {
+          "id": 1727,
+          "logprob": -2.375,
+          "special": false,
+          "text": " test"
+        },
+        {
+          "id": 3102,
+          "logprob": -0.890625,
+          "special": false,
+          "text": " request"
+        },
+        {
+          "id": 317,
+          "logprob": -1.1484375,
+          "special": false,
+          "text": " is"
+        },
+        {
+          "id": 245,
+          "logprob": -1.5390625,
+          "special": false,
+          "text": " a"
+        },
+        {
+          "id": 3102,
+          "logprob": -2.609375,
+          "special": false,
+          "text": " request"
+        },
+        {
+          "id": 327,
+          "logprob": -0.75,
+          "special": false,
+          "text": " for"
+        },
+        {
+          "id": 245,
+          "logprob": -1.1171875,
+          "special": false,
+          "text": " a"
+        },
+        {
+          "id": 1727,
+          "logprob": -0.90625,
+          "special": false,
+          "text": " test"
+        }
+      ],
+      "top_tokens": null
+    },
+    "generated_text": "\nThe test request is a request for a test"
+  }
+]
--- a/integration-tests/models/test_flash_deepseek_v2.py
+++ b/integration-tests/models/test_flash_deepseek_v2.py
+import pytest
+@pytest.fixture(scope="module")
+def flash_deepseek_v2_handle(launcher):
+    with launcher("deepseek-ai/DeepSeek-V2-Lite", num_shard=2) as handle:
+        yield handle
+@pytest.fixture(scope="module")
+async def flash_deepseek_v2(flash_deepseek_v2_handle):
+    await flash_deepseek_v2_handle.health(300)
+    return flash_deepseek_v2_handle.client
+@pytest.mark.release
+@pytest.mark.asyncio
+@pytest.mark.private
+async def test_flash_deepseek_v2(flash_deepseek_v2, response_snapshot):
+    response = await flash_deepseek_v2.generate(
+        "Test request", max_new_tokens=10, decoder_input_details=True
+    )
+    assert response == response_snapshot
+@pytest.mark.release
+@pytest.mark.asyncio
+@pytest.mark.private
+async def test_flash_deepseek_v2_all_params(flash_deepseek_v2, response_snapshot):
+    response = await flash_deepseek_v2.generate(
+        "Test request",
+        max_new_tokens=10,
+        repetition_penalty=1.2,
+        return_full_text=True,
+        stop_sequences=["test"],
+        temperature=0.5,
+        top_p=0.9,
+        top_k=10,
+        truncate=5,
+        typical_p=0.9,
+        watermark=True,
+        decoder_input_details=True,
+        seed=0,
+    )
+    assert response == response_snapshot
+@pytest.mark.release
+@pytest.mark.asyncio
+@pytest.mark.private
+async def test_flash_deepseek_v2_load(
+    flash_deepseek_v2, generate_load, response_snapshot
+):
+    responses = await generate_load(
+        flash_deepseek_v2, "Test request", max_new_tokens=10, n=4
+    )
+    assert len(responses) == 4
+    assert all([r.generated_text == responses[0].generated_text for r in responses])
+    assert responses == response_snapshot
--- a/server/Makefile-vllm
+++ b/server/Makefile-vllm
-commit_cuda := b5dfc61db88a81069e45b44f7cc99bd9e62a60fa
+commit_cuda := d243e9dc7e2c9c2e36a4150ec8e64809cb55c01b
 commit_rocm := c6ee53b1be97e3bbc791b95f22827501297f8921
 build-vllm-cuda:
 	if [ ! -d 'vllm' ]; then \
 		pip install -U ninja packaging --no-cache-dir && \
 		git clone https://github.com/Narsil/vllm.git vllm; \
 	fi
-	cd vllm  && git fetch && git checkout $(commit_cuda) && python setup.py build
+	cd vllm  && git fetch origin && git checkout $(commit_cuda) && python setup.py build
 install-vllm-cuda: build-vllm-cuda
-	cd vllm  && git fetch && git checkout $(commit_cuda) && pip install -e .
+	cd vllm  && git fetch origin && git checkout $(commit_cuda) && pip install -e .
 build-vllm-rocm:
 	if [ ! -d 'vllm' ]; then \

--- a/server/text_generation_server/layers/exl2.py
+++ b/server/text_generation_server/layers/exl2.py
@@ -34,15 +34,10 @@ class Exl2Weight(Weight):
 class Exl2WeightsLoader(WeightsLoader):
    """Loader for exl2-quantized weights."""
-    def get_weights_col_packed(
+    def get_weights(self, weights: "Weights", prefix: str):
-        self,
+        """
-        weights: Weights,
+        Get weights at the given prefix and apply without tensor paralllism.
-        prefix: str,
+        """
-        block_sizes: Union[int, List[int]],
-    ):
-        raise RuntimeError("Column-packed weights are not supported for exl")
-    def get_weights_col(self, weights: Weights, prefix: str):
        try:
            q_weight = weights.get_tensor(f"{prefix}.q_weight")
        except RuntimeError:
@@ -63,26 +58,21 @@ class Exl2WeightsLoader(WeightsLoader):
            q_groups=q_groups,
        )
+    def get_weights_col_packed(
+        self,
+        weights: Weights,
+        prefix: str,
+        block_sizes: Union[int, List[int]],
+    ):
+        raise RuntimeError("Column-packed weights are not supported for exl")
+    def get_weights_col(self, weights: Weights, prefix: str):
+        # Sharding is not yet supported, so we return the weights as-is.
+        return self.get_weights(weights, prefix)
    def get_multi_weights_col(self, weights: Weights, prefixes: List[str], dim: int):
        raise ValueError("get_multi_weights_col is not supported for exl2")
    def get_weights_row(self, weights: Weights, prefix: str):
-        try:
+        # Sharding is not yet supported, so we return the weights as-is.
-            q_weight = weights.get_tensor(f"{prefix}.q_weight")
+        return self.get_weights(weights, prefix)
-        except RuntimeError:
-            raise RuntimeError(
-                "Cannot load `exl2`-quantized weight, make sure the model is already quantized."
-            )
-        q_scale = weights.get_tensor(f"{prefix}.q_scale")
-        q_invperm = weights.get_tensor(f"{prefix}.q_invperm")
-        q_scale_max = weights.get_tensor(f"{prefix}.q_scale_max")
-        q_groups = weights.get_tensor(f"{prefix}.q_groups")
-        return Exl2Weight(
-            q_weight=q_weight,
-            q_scale=q_scale,
-            q_invperm=q_invperm,
-            q_scale_max=q_scale_max,
-            q_groups=q_groups,
-        )
--- a/server/text_generation_server/layers/gptq/__init__.py
+++ b/server/text_generation_server/layers/gptq/__init__.py
@@ -134,6 +134,115 @@ class GPTQWeightsLoader(WeightsLoader):
        self.quantize = quantize
        self.sym = sym
+    def get_weights(self, weights: Weights, prefix: str):
+        from text_generation_server.layers.marlin import (
+            can_use_gptq_marlin,
+            repack_gptq_for_marlin,
+        )
+        self._get_gptq_params(weights)
+        if can_use_gptq_marlin(
+            bits=self.bits,
+            groupsize=self.groupsize,
+            quant_method=self.quant_method,
+            quantize=self.quantize,
+            sym=self.sym,
+        ):
+            log_once(logger.info, "Using GPTQ-Marlin kernels")
+            try:
+                qweight = weights.get_tensor(f"{prefix}.qweight")
+            except RuntimeError:
+                raise RuntimeError(
+                    f"Cannot load `{self.quantize}` weight for GPTQ -> Marlin repacking, make sure the model is already quantized"
+                )
+            g_idx = weights.get_tensor(f"{prefix}.g_idx")
+            scales = weights.get_tensor(f"{prefix}.scales")
+            return repack_gptq_for_marlin(
+                qweight=qweight,
+                scales=scales,
+                g_idx=g_idx,
+                bits=self.bits,
+                desc_act=self.desc_act,
+                groupsize=self.groupsize,
+                sym=self.sym,
+                sharded_infeatures=False,
+            )
+        use_exllama = True
+        if self.bits != 4:
+            use_exllama = False
+        if self.desc_act:
+            log_once(logger.warning, "Disabling exllama because desc_act=True")
+            use_exllama = False
+        try:
+            qweight = weights.get_tensor(f"{prefix}.qweight")
+        except RuntimeError:
+            raise RuntimeError(
+                "Cannot load `gptq` weight, make sure the model is already quantized, or quantize it with `text-generation-server quantize ORIGINAL_MODEL_ID NEW_MODEL_ID`"
+            )
+        if self.quantize == "gptq" and self.quant_method == "gptq":
+            g_idx = weights.get_tensor(f"{prefix}.g_idx")
+        else:
+            g_idx = None
+        from text_generation_server.layers.gptq import (
+            HAS_EXLLAMA,
+            CAN_EXLLAMA,
+            GPTQWeight,
+        )
+        if use_exllama:
+            if not HAS_EXLLAMA:
+                if CAN_EXLLAMA:
+                    log_once(
+                        logger.warning,
+                        "Exllama GPTQ cuda kernels (which are faster) could have been used, but are not currently installed, try using BUILD_EXTENSIONS=True",
+                    )
+                use_exllama = False
+            else:
+                log_once(logger.info, f"Using exllama kernels v{HAS_EXLLAMA}")
+        qzeros = weights.get_tensor(f"{prefix}.qzeros")
+        scales = weights.get_tensor(f"{prefix}.scales")
+        if use_exllama and g_idx is not None:
+            g_idx = g_idx - g_idx[0]
+        if self.quantize == "gptq" and self.quant_method == "awq":
+            log_once(
+                logger.info, "Converting AWQ model to Exllama/GPTQ packing format."
+            )
+            from text_generation_server.layers.awq.conversion_utils import (
+                fast_awq_to_gptq,
+            )
+            qweight, qzeros = fast_awq_to_gptq(qweight, qzeros)
+            if use_exllama:
+                g_idx = None
+            else:
+                g_idx = (
+                    torch.arange(
+                        qweight.shape[0] * (32 // self.bits),
+                        device=qweight.device,
+                    )
+                    // self.groupsize
+                ).to(dtype=torch.int32)
+        return GPTQWeight(
+            qweight=qweight,
+            qzeros=qzeros,
+            scales=scales,
+            g_idx=g_idx,
+            bits=self.bits,
+            groupsize=self.groupsize,
+            use_exllama=use_exllama,
+        )
    def get_weights_col_packed(
        self,
        weights: Weights,

--- a/server/text_generation_server/layers/marlin.py
+++ b/server/text_generation_server/layers/marlin.py
@@ -33,6 +33,35 @@ class MarlinWeightsLoader(WeightsLoader):
        self.bits = bits
        self.is_marlin_24 = is_marlin_24
+    def get_weights(self, weights: "Weights", prefix: str):
+        """
+        Get weights at the given prefix and apply without tensor paralllism.
+        """
+        is_marlin_24 = getattr(self, "gptq_checkpoint_format", None) == "marlin_24"
+        if is_marlin_24:
+            try:
+                B = weights.get_tensor(f"{prefix}.B_24")
+            except RuntimeError:
+                raise RuntimeError(
+                    "Cannot load `marlin` 2:4 sparsity weight, make sure the model is already quantized."
+                )
+            B_meta = weights.get_tensor(f"{prefix}.B_meta")
+            s = weights.get_tensor(f"{prefix}.s")
+            weight = GPTQMarlin24Weight(B=B, B_meta=B_meta, s=s, bits=self.bits)
+        else:
+            try:
+                B = weights.get_tensor(f"{prefix}.B")
+            except RuntimeError:
+                raise RuntimeError(
+                    "Cannot load `marlin` weight, make sure the model is already quantized."
+                )
+            s = weights.get_tensor(f"{prefix}.s")
+            weight = MarlinWeight(B=B, s=s)
+        return weight
    def get_weights_col_packed(
        self,
        weights: Weights,

--- a/server/text_generation_server/layers/rotary.py
+++ b/server/text_generation_server/layers/rotary.py
 import os
 import torch
 from torch import nn
+from loguru import logger
 from text_generation_server.utils.import_utils import SYSTEM
@@ -97,6 +98,8 @@ class PositionRotaryEmbedding(nn.Module):
                )
            elif rope_scaling["type"] == "yarn":
                scaling_factor = rope_scaling["factor"]
+                mscale = rope_scaling.get("mscale", 1.0)
+                mscale_all_dim = rope_scaling.get("mscale_all_dim", 0.0)
                return YarnPositionRotaryEmbedding(
                    dim=2 * inv_freq.shape[0],
                    max_position_embeddings=rope_scaling[
@@ -109,6 +112,8 @@ class PositionRotaryEmbedding(nn.Module):
                    attn_factor=1,
                    beta_fast=32,
                    beta_slow=1,
+                    mscale=mscale,
+                    mscale_all_dim=mscale_all_dim,
                )
            elif rope_scaling["type"] in ["su", "longrope"]:
                short_factor = torch.tensor(
@@ -181,6 +186,8 @@ class PositionRotaryEmbedding(nn.Module):
                    scaling_factor=scaling_factor,
                )
            elif rope_scaling["type"] == "yarn":
+                mscale = rope_scaling.get("mscale", 1.0)
+                mscale_all_dim = rope_scaling.get("mscale_all_dim", 0.0)
                return YarnPositionRotaryEmbedding(
                    dim=2 * inv_freq.shape[0],
                    max_position_embeddings=rope_scaling[
@@ -193,6 +200,8 @@ class PositionRotaryEmbedding(nn.Module):
                    attn_factor=1,
                    beta_fast=32,
                    beta_slow=1,
+                    mscale=mscale,
+                    mscale_all_dim=mscale_all_dim,
                )
            else:
                raise NotImplementedError(
@@ -346,10 +355,10 @@ def linear_ramp_mask(min, max, dim):
    return ramp_func
-def get_mscale(scale=1):
+def get_mscale(scale: float = 1.0, mscale: float = 1.0):
    if scale <= 1:
        return 1.0
-    return 0.1 * math.log(scale) + 1.0
+    return 0.1 * mscale * math.log(scale) + 1.0
 class YarnPositionRotaryEmbedding(PositionRotaryEmbedding):
@@ -365,6 +374,8 @@ class YarnPositionRotaryEmbedding(PositionRotaryEmbedding):
        attn_factor,
        beta_fast,
        beta_slow,
+        mscale: float,
+        mscale_all_dim: float,
    ):
        inv_freq = _create_inv_freq(dim, base, device)
        super().__init__(inv_freq, scaling_factor)
@@ -375,8 +386,12 @@ class YarnPositionRotaryEmbedding(PositionRotaryEmbedding):
        self.attn_factor = attn_factor
        self.beta_fast = beta_fast
        self.beta_slow = beta_slow
+        self.mscale_all_dim = mscale_all_dim
+        self.scaling_factor = scaling_factor
        self.mscale = float(
-            get_mscale(self.scaling_factor) * self.attn_factor
+            get_mscale(self.scaling_factor, mscale)
+            / get_mscale(self.scaling_factor, mscale_all_dim)
+            * self.attn_factor
        )  # Get n-d magnitude scaling corrected for interpolation
    def _update_cos_sin_cache(self, dtype, device, seqlen):
@@ -387,7 +402,7 @@ class YarnPositionRotaryEmbedding(PositionRotaryEmbedding):
            or self._cos_cached.device != device
            or self._cos_cached.dtype != dtype
        ):
-            if seqlen > self.max_position_embeddings:
+            if seqlen > self.max_position_embeddings or True:
                inv_freq_extrapolation = _create_inv_freq(
                    self.dim, self.base, self.inv_freq.device
                )
@@ -400,6 +415,7 @@ class YarnPositionRotaryEmbedding(PositionRotaryEmbedding):
                    self.base,
                    self.max_position_embeddings,
                )
                inv_freq_mask = (
                    1 - linear_ramp_mask(low, high, self.dim // 2).float().to(device)
                ) * self.extrapolation_factor  # Get n-d rotational scaling corrected for extrapolation
@@ -409,9 +425,6 @@ class YarnPositionRotaryEmbedding(PositionRotaryEmbedding):
                )
                self.inv_freq = inv_freq
-                self.mscale = float(
-                    get_mscale(self.scaling_factor) * self.attn_factor
-                )  # Get n-d magnitude scaling corrected for interpolation
            self._seq_len_cached = seqlen
            t = torch.arange(seqlen, device=device, dtype=self.inv_freq.dtype)

--- a/server/text_generation_server/models/__init__.py
+++ b/server/text_generation_server/models/__init__.py
@@ -61,6 +61,10 @@ FLASH_ATTENTION = True
 try:
    from text_generation_server.models.flash_causal_lm import FlashCausalLM
    from text_generation_server.models.vlm_causal_lm import VlmCausalLM
+    from text_generation_server.models.custom_modeling.flash_deepseek_v2_modeling import (
+        FlashDeepseekV2ForCausalLM,
+        DeepseekV2Config,
+    )
    from text_generation_server.models.custom_modeling.flash_llama_modeling import (
        FlashLlamaForCausalLM,
    )
@@ -141,6 +145,11 @@ if MAMBA_AVAILABLE:
 class ModelType(enum.Enum):
+    DEEPSEEK_V2 = {
+        "type": "deepseek_v2",
+        "name": "Deepseek V2",
+        "url": "https://huggingface.co/deepseek-ai/DeepSeek-V2",
+    }
    IDEFICS2 = {
        "type": "idefics2",
        "name": "Idefics 2",
@@ -459,7 +468,40 @@ def get_model(
            f"The backend {SYSTEM} does not support sliding window attention that is used by the model type {model_type}. To use this model nonetheless with the {SYSTEM} backend, please launch TGI with the argument `--max-input-tokens` smaller than sliding_window={sliding_window} (got here max_input_tokens={max_input_tokens})."
        )
-    if model_type == MAMBA:
+    if model_type == DEEPSEEK_V2:
+        if FLASH_ATTENTION:
+            head_size = max(
+                config_dict.get("qk_nope_dim", 128)
+                + config_dict.get("qk_rope_dim", 64),
+                config_dict.get("v_head_dim", 128),
+            )
+            return FlashCausalLM(
+                model_id=model_id,
+                model_class=FlashDeepseekV2ForCausalLM,
+                revision=revision,
+                quantize=quantize,
+                speculator=speculator,
+                default_dtype=torch.bfloat16,
+                dtype=dtype,
+                trust_remote_code=trust_remote_code,
+                lora_adapter_ids=lora_adapter_ids,
+                config_class=DeepseekV2Config,
+                head_size=head_size,
+            )
+        elif sharded:
+            raise NotImplementedError(
+                FLASH_ATT_ERROR_MESSAGE.format("Sharded Deepseek V2")
+            )
+        else:
+            return CausalLM.fallback(
+                model_id,
+                revision,
+                quantize=quantize,
+                speculator=speculator,
+                dtype=dtype,
+                trust_remote_code=trust_remote_code,
+            )
+    elif model_type == MAMBA:
        return Mamba(
            model_id,
            revision,

--- a/server/text_generation_server/models/custom_modeling/flash_deepseek_v2_modeling.py
+++ b/server/text_generation_server/models/custom_modeling/flash_deepseek_v2_modeling.py
--- a/server/text_generation_server/models/flash_causal_lm.py
+++ b/server/text_generation_server/models/flash_causal_lm.py
@@ -839,7 +839,9 @@ class FlashCausalLM(Model):
        default_dtype=torch.float16,
        aliases=None,
        # Used for Santacoder override of config
-        num_kv_heads=None,
+        num_kv_heads: Optional[int] = None,
+        # Deepseek V2 uses different QK and V dims.
+        head_size: Optional[int] = None,
        skip_special_tokens: bool = True,
    ):
        self.process_group, rank, world_size = initialize_torch_distributed()
@@ -922,7 +924,11 @@ class FlashCausalLM(Model):
            else num_kv_heads
        )
        assert self.num_kv_heads > 0
+        if head_size is None:
            self.head_size = config.hidden_size // config.num_attention_heads
+        else:
+            self.head_size = head_size
        self.cuda_graphs = {}
        self.kv_cache = []

--- a/server/text_generation_server/utils/weights.py
+++ b/server/text_generation_server/utils/weights.py
@@ -21,6 +21,13 @@ class WeightsLoader(ABC):
    with the format, etc.
    """
+    @abstractmethod
+    def get_weights(self, weights: "Weights", prefix: str):
+        """
+        Get weights at the given prefix and apply without tensor paralllism.
+        """
+        ...
    @abstractmethod
    def get_weights_col_packed(
        self,
@@ -104,6 +111,9 @@ class DefaultWeightsLoader(WeightsLoader):
    and/or concatenation.
    """
+    def get_weights(self, weights: "Weights", prefix: str):
+        return weights.get_tensor(f"{prefix}.weight")
    def get_weights_col_packed(
        self,
        weights: "Weights",
@@ -299,6 +309,9 @@ class Weights:
        return tensor
+    def get_weights(self, prefix: str):
+        return self.weights_loader.get_weights(self, prefix)
    def get_weights_col_packed_qkv(
        self,
        prefix: str,