vllm-project
diff --git a/‎.github/workflows/vllm_ascend_test_long_term.yaml‎
Lines changed: 21 additions & 6 deletions b/‎.github/workflows/vllm_ascend_test_long_term.yaml‎
Lines changed: 21 additions & 6 deletions
diff --git a/‎docs/source/developer_guide/evaluation/profile_execute_duration.md‎
Lines changed: 1 addition & 1 deletion b/‎docs/source/developer_guide/evaluation/profile_execute_duration.md‎
Lines changed: 1 addition & 1 deletion
diff --git a/‎tests/conftest.py‎
Lines changed: 1 addition & 1 deletion b/‎tests/conftest.py‎
Lines changed: 1 addition & 1 deletion
diff --git a/‎tests/long_term/test_deepseek_v2_lite_tp2_accuracy.py‎
Lines changed: 72 additions & 0 deletions b/‎tests/long_term/test_deepseek_v2_lite_tp2_accuracy.py‎
Lines changed: 72 additions & 0 deletions
diff --git a/‎tests/multicard/test_offline_inference_distributed.py‎
Lines changed: 0 additions & 2 deletions b/‎tests/multicard/test_offline_inference_distributed.py‎
Lines changed: 0 additions & 2 deletions
diff --git a/‎tests/singlecard/test_profile_execute_duration.py‎
Lines changed: 3 additions & 4 deletions b/‎tests/singlecard/test_profile_execute_duration.py‎
Lines changed: 3 additions & 4 deletions
diff --git a/‎vllm_ascend/attention/attention.py‎
Lines changed: 2 additions & 0 deletions b/‎vllm_ascend/attention/attention.py‎
Lines changed: 2 additions & 0 deletions
diff --git a/‎vllm_ascend/attention/attention_v1.py‎
Lines changed: 1 addition & 0 deletions b/‎vllm_ascend/attention/attention_v1.py‎
Lines changed: 1 addition & 0 deletions
diff --git a/‎vllm_ascend/attention/mla_v1.py‎
Lines changed: 37 additions & 43 deletions b/‎vllm_ascend/attention/mla_v1.py‎
Lines changed: 37 additions & 43 deletions
diff --git a/‎vllm_ascend/envs.py‎
Lines changed: 2 additions & 2 deletions b/‎vllm_ascend/envs.py‎
Lines changed: 2 additions & 2 deletions
@@ -41,9 +41,19 @@ jobs:
     strategy:
       max-parallel: 2
       matrix:
+        os: [linux-arm64-npu-1, linux-arm64-npu-4]
         vllm_version: [main, v0.9.0]
+    concurrency:
+      group: >
+        ${{
+        matrix.os == 'linux-arm64-npu-4'
+          && github.event.pull_request.number
+          && format('pr-{0}-limit-npu-4-long-term', github.event.pull_request.number)
+        || format('job-{0}-{1}-{2}-long-term', matrix.os, matrix.vllm_version, github.event.pull_request.number)
+        }}
+      cancel-in-progress: false
     name: vLLM Ascend long term test
-    runs-on: linux-arm64-npu-1
+    runs-on: ${{ matrix.os }}
     container:
       # TODO(yikun): Remove m.daocloud.io prefix when infra proxy ready
       image: m.daocloud.io/quay.io/ascend/cann:8.1.rc1-910b-ubuntu22.04-py3.10
@@ -92,8 +102,13 @@ jobs:
 
       - name: Run vllm-project/vllm-ascend long term test
         run: |
-          # spec decode test
-          VLLM_USE_MODELSCOPE=True pytest -sv tests/long_term/spec_decode/e2e/test_v1_mtp_correctness.py
-          VLLM_USE_MODELSCOPE=true pytest -sv tests/long_term/spec_decode/e2e/test_v1_spec_decode.py
-          VLLM_USE_MODELSCOPE=True pytest -sv tests/long_term/spec_decode/e2e/test_mtp_correctness.py  # it needs a clean process
-          pytest -sv tests/long_term/spec_decode --ignore=tests/long_term/spec_decode/e2e/test_mtp_correctness.py --ignore=tests/long_term/spec_decode/e2e/test_v1_spec_decode.py --ignore=tests/long_term/spec_decode/e2e/test_v1_mtp_correctness.py
+          if [[ "${{ matrix.os }}" == "linux-arm64-npu-1" ]]; then
+            # spec decode test
+            VLLM_USE_MODELSCOPE=True pytest -sv tests/long_term/spec_decode/e2e/test_v1_mtp_correctness.py
+            VLLM_USE_MODELSCOPE=True pytest -sv tests/long_term/spec_decode/e2e/test_v1_spec_decode.py
+            VLLM_USE_MODELSCOPE=True pytest -sv tests/long_term/spec_decode/e2e/test_mtp_correctness.py  # it needs a clean process
+            pytest -sv tests/long_term/spec_decode --ignore=tests/long_term/spec_decode/e2e/test_mtp_correctness.py --ignore=tests/long_term/spec_decode/e2e/test_v1_spec_decode.py --ignore=tests/long_term/spec_decode/e2e/test_v1_mtp_correctness.py
+            pytest -sv tests/long_term/test_accuracy.py
+          else
+            VLLM_USE_MODELSCOPE=True pytest -sv tests/long_term/test_deepseek_v2_lite_tp2_accuracy.py
+          fi
@@ -5,7 +5,7 @@ The execution duration of each stage (including pre/post-processing, model forwa
 **To reduce the performance overhead, we add this feature, using the NPU event timestamp mechanism to observe the device execution time asynchronously.**
 
 ## Usage
-* Use the environment variable `VLLM_MODEL_EXECUTE_TIME_OBSERVE` to enable this feature.
+* Use the environment variable `VLLM_ASCEND_MODEL_EXECUTE_TIME_OBSERVE` to enable this feature.
 * Use the non-blocking API `ProfileExecuteDuration().capture_async` to set observation points asynchronously when you need to observe the execution duration.
 * Use the blocking API `ProfileExecuteDuration().pop_captured_sync` at an appropriate time to get and print the execution durations of all observed stages.
 
 
@@ -354,4 +354,4 @@ def prompt_template(request):
 
 @pytest.fixture(scope="session")
 def ilama_lora_files():
-    return snapshot_download(repo_id="jeeejeee/ilama-text2sql-spider")
+    return snapshot_download(repo_id="jeeejeee/ilama-text2sql-spider")
@@ -0,0 +1,72 @@
+#
+# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
+# Copyright 2023 The vLLM team.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# This file is a part of the vllm-ascend project.
+# Adapted from vllm-project/blob/main/tests/entrypoints/llm/test_accuracy.py
+#
+
+import gc
+import multiprocessing
+from multiprocessing import Queue
+
+import lm_eval
+import pytest
+import torch
+
+# pre-trained model path on Hugging Face.
+MODELS = ["deepseek-ai/DeepSeek-V2-Lite"]
+# Math reasoning benchmark (Grade School Math 8K).
+TASK = "gsm8k"
+# Answer validation requiring format consistency.
+FILTER = "exact_match,strict-match"
+# 3% relative tolerance for numerical accuracy.
+RTOL = 0.03
+# Baseline accuracy after VLLM optimization.
+# FIXME: fix the accuracy issue
+EXPECTED_VALUE = 0.000758150113722517
+
+
+def run_test(model_name, queue, more_args=None):
+    model_args = f"pretrained={model_name},max_model_len=4096,trust_remote_code=True,tensor_parallel_size=4"
+    if more_args is not None:
+        model_args = f"{model_args},{more_args}"
+    results = lm_eval.simple_evaluate(
+        model="vllm",
+        model_args=model_args,
+        tasks=TASK,
+        batch_size="auto",
+    )
+    result = results["results"][TASK][FILTER]
+    print(100 * "*", "\nThe accuracy test result:", result)
+    queue.put(result)
+    del results
+    torch.npu.empty_cache()
+    gc.collect()
+
+
+@pytest.mark.parametrize("model", MODELS)
+def test_lm_eval_accuracy(model, monkeypatch: pytest.MonkeyPatch):
+    with monkeypatch.context():
+        result_queue: Queue[float] = multiprocessing.Queue()
+        p = multiprocessing.Process(target=run_test,
+                                    args=(
+                                        model,
+                                        result_queue,
+                                    ))
+        p.start()
+        p.join()
+        result = result_queue.get()
+        assert (EXPECTED_VALUE - RTOL < result < EXPECTED_VALUE + RTOL), \
+            f"Expected: {EXPECTED_VALUE}±{RTOL} | Measured: {result}"
@@ -22,7 +22,6 @@
 """
 import os
 
-import pytest
 import vllm  # noqa: F401
 
 from tests.conftest import VllmRunner
@@ -47,7 +46,6 @@ def test_models_distributed_QwQ():
         vllm_model.generate_greedy(example_prompts, max_tokens)
 
 
-@pytest.mark.skipif(True, reason="wait for mla issue fixed on v1")
 def test_models_distributed_DeepSeek():
     example_prompts = [
         "vLLM is a high-throughput and memory-efficient inference and serving engine for LLMs.",
 
@@ -16,15 +16,16 @@
 # See the License for the specific language governing permissions and
 # limitations under the License.
 #
+import os
 import time
+from unittest.mock import patch
 
 import torch
 import vllm  # noqa: F401
-
 import vllm_ascend.envs as envs
 from vllm_ascend.utils import ProfileExecuteDuration
 
-
+@patch.dict(os.environ, {"VLLM_ASCEND_MODEL_EXECUTE_TIME_OBSERVE": "1"})
 def test_execue_duration_enabled_discrepancy():
     a = torch.randn(10000, 10000).npu()
     b = torch.randn(10000, 10000).npu()
@@ -33,7 +34,6 @@ def test_execue_duration_enabled_discrepancy():
     torch.matmul(a, b)
     torch.npu.synchronize()
 
-    envs.VLLM_MODEL_EXECUTE_TIME_OBSERVE = True
     cpu_start = time.perf_counter()
     with ProfileExecuteDuration().capture_async("forward"):
         torch.matmul(a, b)
@@ -54,7 +54,6 @@ def test_execue_duration_disabled():
     a = torch.randn(100, 100).npu()
     b = torch.randn(100, 100).npu()
 
-    envs.VLLM_MODEL_EXECUTE_TIME_OBSERVE = False
     with ProfileExecuteDuration().capture_async("forward"):
         torch.matmul(a, b)
         torch.npu.synchronize()
 
@@ -720,6 +720,7 @@ def __init__(
         blocksparse_params: Optional[Dict[str, Any]] = None,
         logits_soft_cap: Optional[float] = None,
         attn_type: str = AttentionType.DECODER,
+        kv_sharing_target_layer_name: Optional[str] = None,
         use_irope: bool = False,
     ) -> None:
         self.num_heads = num_heads
@@ -961,6 +962,7 @@ def __init__(
         blocksparse_params: Optional[Dict[str, Any]] = None,
         logits_soft_cap: Optional[float] = None,
         attn_type: str = AttentionType.DECODER,
+        kv_sharing_target_layer_name: Optional[str] = None,
         **extra_impl_args,
     ) -> None:
         self.num_heads = num_heads
 
@@ -186,6 +186,7 @@ def __init__(
         blocksparse_params: Optional[Dict[str, Any]] = None,
         logits_soft_cap: Optional[float] = None,
         attn_type: str = AttentionType.DECODER,
+        kv_sharing_target_layer_name: Optional[str] = None,
         use_irope: bool = False,
     ) -> None:
         self.num_heads = num_heads
 
@@ -9,10 +9,8 @@
                                               MLAAttentionImpl)
 from vllm.attention.backends.utils import PAD_SLOT_ID
 from vllm.config import get_current_vllm_config
-from vllm.model_executor.layers.linear import (ColumnParallelLinear,
-                                               LinearBase, RowParallelLinear,
+from vllm.model_executor.layers.linear import (LinearBase,
                                                UnquantizedLinearMethod)
-from vllm.model_executor.layers.rotary_embedding import RotaryEmbedding
 
 from vllm_ascend.attention.attention_v1 import AscendAttentionState
 from vllm_ascend.ops.attention import vanilla_chunked_prefill_mla
@@ -117,6 +115,8 @@ class AscendMLAMetadata:
     # For logging.
     num_input_tokens: int = 0  # Number of tokens including padding.
 
+    with_prefill_across_dp: bool = False
+
     # The dimension of the attention heads
     head_dim: Optional[int] = None
     attn_mask: torch.Tensor = None
@@ -260,6 +260,10 @@ def build_dummy(self, num_reqs: int,
                                   PAD_SLOT_ID,
                                   dtype=torch.int32,
                                   device=device)
+        query_start_loc = torch.full((num_reqs, ),
+                                     -1,
+                                     dtype=torch.int32,
+                                     device=device)
         decode_metadata = AscendMLADecodeMetadata(
             input_positions=input_positions,
             block_table=block_table,
@@ -278,15 +282,21 @@ def build_dummy(self, num_reqs: int,
             attn_state=AscendAttentionState.DecodeOnly,
             prefill=None,
             decode=decode_metadata,
+            query_start_loc=query_start_loc,
+            seq_lens=seq_lens,
+            block_tables=block_table,
         )
 
-    def build(self,
-              num_reqs: int,
-              num_actual_tokens: int,
-              max_query_len: int,
-              common_attn_metadata: CommonAttentionMetadata,
-              common_prefix_len: Optional[int] = None,
-              graph_pad_size: int = -1) -> AscendMLAMetadata:
+    def build(
+        self,
+        num_reqs: int,
+        num_actual_tokens: int,
+        max_query_len: int,
+        common_attn_metadata: CommonAttentionMetadata,
+        common_prefix_len: Optional[int] = None,
+        graph_pad_size: int = -1,
+        with_prefill_across_dp: bool = False,
+    ) -> AscendMLAMetadata:
         assert self._num_decodes + self._num_prefills == num_reqs
 
         # Note(simon): be careful about the CPU <> GPU memory movement in this
@@ -388,6 +398,7 @@ def build(self,
             query_start_loc=query_start_loc,
             block_tables=block_table,
             seq_lens=seq_lens,
+            with_prefill_across_dp=with_prefill_across_dp,
         )
 
 
@@ -409,20 +420,7 @@ def __init__(
         blocksparse_params: Optional[dict[str, Any]],
         logits_soft_cap: Optional[float],
         attn_type: str,
-        # MLA Specific Arguments
-        q_lora_rank: Optional[int],
-        kv_lora_rank: int,
-        qk_nope_head_dim: int,
-        qk_rope_head_dim: int,
-        qk_head_dim: int,
-        v_head_dim: int,
-        rotary_emb: RotaryEmbedding,
-        # q_proj should be q_b_proj if q_lora_rank is not None, but from an
-        # attention backend perspective we rely on the layer to pass in the
-        # correct matrix
-        q_proj: ColumnParallelLinear,
-        kv_b_proj: ColumnParallelLinear,
-        o_proj: RowParallelLinear,
+        kv_sharing_target_layer_name: Optional[str] = None,
         **kwargs,
     ) -> None:
         self.num_heads = num_heads
@@ -431,25 +429,20 @@ def __init__(
         self.num_kv_heads = num_kv_heads
         self.kv_cache_dtype = kv_cache_dtype
 
-        self.q_lora_rank = q_lora_rank
-        self.kv_lora_rank = kv_lora_rank
-        self.qk_nope_head_dim = qk_nope_head_dim
-        self.qk_rope_head_dim = qk_rope_head_dim
-        self.qk_head_dim = qk_head_dim
-        self.v_head_dim = v_head_dim
-
-        # Hack for V1 for now to avoid torch library overhead (since we are
-        # already inside an attention custom op), pull out the forward
-        # method from the rotary embedding and call it directly
-        # TODO(lucas): we should probably find a cleaner way to do this
-        self.rotary_emb = rotary_emb
-
-        self.q_proj = q_proj
-        self.kv_b_proj = kv_b_proj
-        self.o_proj = o_proj
-
+        # MLA Args
+        self.q_lora_rank = kwargs['q_lora_rank']
+        self.kv_lora_rank = kwargs['kv_lora_rank']
+        self.qk_nope_head_dim = kwargs['qk_nope_head_dim']
+        self.qk_rope_head_dim = kwargs['qk_rope_head_dim']
+        self.qk_head_dim = kwargs['qk_head_dim']
+        self.v_head_dim = kwargs['v_head_dim']
+        self.rotary_emb = kwargs['rotary_emb']
+        self.q_proj = kwargs['q_proj']
+        self.kv_b_proj = kwargs['kv_b_proj']
+        self.o_proj = kwargs['o_proj']
         self.kv_a_proj_with_mqa = kwargs.get('kv_a_proj_with_mqa', None)
         self.kv_a_layernorm = kwargs.get('kv_a_layernorm', None)
+
         # Handle the differences between the flash_attn_varlen from flash_attn
         # and the one from vllm_flash_attn. The former is used on RoCM and the
         # latter has an additional parameter to control FA2 vs FA3
@@ -621,7 +614,7 @@ def exec_kv(
         kv = self.kv_a_proj_with_mqa(hidden_states)[0]
         # npu_kv_rmsnorm_rope_cache needs [B, N, S, D]
         kv = kv.view(B, N, S, self.kv_lora_rank + self.qk_rope_head_dim)
-        k_pe, k_nope, _, _ = torch.ops.npu_inference.npu_kv_rmsnorm_rope_cache(
+        k_pe, k_nope, _, _ = torch_npu.npu_kv_rmsnorm_rope_cache(
             kv,
             self.kv_a_layernorm.weight,
             cos,
@@ -643,7 +636,7 @@ def rope_single(
         B, N, D = x.shape
         S = 1
         x = x.view(B, N, S, D)
-        x = torch.ops.npu_inference.npu_interleave_rope(x, cos, sin)
+        x = torch_npu.npu_interleave_rope(x, cos, sin)
         return x.view(B, N, D)
 
     def _forward_decode(
@@ -766,6 +759,7 @@ def forward(
                 sin = sin[attn_metadata.decode.input_positions]
                 cos = cos[:, None, None, :]
                 sin = sin[:, None, None, :]
+
                 decode_q_pe = self.rope_single(decode_q_pe, cos, sin)
                 decode_k_pe, decode_k_nope = self.exec_kv(
                     hidden_states_or_kv_c_normed, cos, sin, kv_cache,
 
@@ -36,8 +36,6 @@
     lambda: bool(int(os.getenv("COMPILE_CUSTOM_KERNELS", "1"))),
     "VLLM_ENABLE_MC2":
     lambda: bool(int(os.getenv("VLLM_ENABLE_MC2", '0'))),
-    "VLLM_MODEL_EXECUTE_TIME_OBSERVE":
-    lambda: bool(int(os.getenv("VLLM_MODEL_EXECUTE_TIME_OBSERVE", '0'))),
     "USING_LCCL_COM":
     lambda: bool(int(os.getenv("USING_LCCL_COM", '0'))),
     "SOC_VERSION":
@@ -70,6 +68,8 @@
     lambda: os.getenv("VLLM_VERSION", None),
     "VLLM_ASCEND_TRACE_RECOMPILES":
     lambda: bool(int(os.getenv("VLLM_ASCEND_TRACE_RECOMPILES", '0'))),
+    "VLLM_ASCEND_MODEL_EXECUTE_TIME_OBSERVE":
+    lambda: bool(int(os.getenv("VLLM_ASCEND_MODEL_EXECUTE_TIME_OBSERVE", '0'))),
 }
 
 # end-env-vars-definition