vllm-project
diff --git a/‎.github/format_pr_body.sh‎
Lines changed: 52 additions & 0 deletions b/‎.github/format_pr_body.sh‎
Lines changed: 52 additions & 0 deletions
diff --git a/‎.github/workflows/accuracy_test.yaml‎
Lines changed: 23 additions & 30 deletions b/‎.github/workflows/accuracy_test.yaml‎
Lines changed: 23 additions & 30 deletions
diff --git a/‎.github/workflows/doc_codespell.yaml‎
Lines changed: 1 addition & 1 deletion b/‎.github/workflows/doc_codespell.yaml‎
Lines changed: 1 addition & 1 deletion
diff --git a/‎.github/workflows/format_pr_body.yaml‎
Lines changed: 63 additions & 0 deletions b/‎.github/workflows/format_pr_body.yaml‎
Lines changed: 63 additions & 0 deletions
diff --git a/‎.github/workflows/nightly_benchmarks.yaml‎
Lines changed: 3 additions & 3 deletions b/‎.github/workflows/nightly_benchmarks.yaml‎
Lines changed: 3 additions & 3 deletions
diff --git a/‎.github/workflows/release_whl.yml‎
Lines changed: 8 additions & 1 deletion b/‎.github/workflows/release_whl.yml‎
Lines changed: 8 additions & 1 deletion
@@ -0,0 +1,52 @@
+#
+# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# This file is a part of the vllm-ascend project.
+#
+
+#!/bin/bash
+
+set -eux
+
+# ensure 2 argument is passed
+if [ "$#" -ne 3 ]; then
+    echo "Usage: $0 <pr_number> <vllm_version> <vllm_commit>"
+    exit 1
+fi
+
+PR_NUMBER=$1
+VLLM_VERSION=$2
+VLLM_COMMIT=$3
+OLD=/tmp/orig_pr_body.txt
+NEW=/tmp/new_pr_body.txt
+
+gh pr view --json body --template "{{.body}}" "${PR_NUMBER}" > "${OLD}"
+cp "${OLD}" "${NEW}"
+
+# Remove "FIX #xxxx (*link existing issues this PR will resolve*)"
+sed -i '/<!--/,/-->/d' "${NEW}"
+sed -i '/- vLLM .*$/d' "${NEW}"
+echo "- vLLM version: $VLLM_VERSION" >> "${NEW}"
+echo "- vLLM main: $VLLM_COMMIT" >> "${NEW}"
+
+# Run this only if ${NEW} is different than ${OLD}
+if ! cmp -s "${OLD}" "${NEW}"; then
+    echo
+    echo "Updating PR body:"
+    echo
+    cat "${NEW}"
+    gh pr edit --body-file "${NEW}" "${PR_NUMBER}"
+else
+    echo "No changes needed"
+fi
@@ -53,9 +53,9 @@ on:
         type: choice
         options:
           - all
-          - Qwen/Qwen2.5-7B-Instruct
           - Qwen/Qwen2.5-VL-7B-Instruct
           - Qwen/Qwen3-8B-Base
+          - Qwen/Qwen3-30B-A3B
         default: 'all'
 
 # Bash shells do not use ~/.profile or ~/.bashrc so these shells need to be explicitly
@@ -77,58 +77,57 @@ jobs:
       ${{
       (contains(github.event.pull_request.labels.*.name, 'accuracy-test') ||
       contains(github.event.pull_request.labels.*.name, 'vl-accuracy-test') ||
+      contains(github.event.pull_request.labels.*.name, 'moe-accuracy-test') ||
       contains(github.event.pull_request.labels.*.name, 'dense-accuracy-test')) &&
       contains(github.event.pull_request.labels.*.name, 'ready-for-test') ||
       github.event_name == 'workflow_dispatch' || github.event_name == 'schedule'
       }}
     runs-on: >-
       ${{
-          (matrix.model_name == 'Qwen/Qwen2.5-VL-7B-Instruct' && 'linux-arm64-npu-4') ||
+          (matrix.model_name == 'Qwen/Qwen3-30B-A3B' && 'linux-arm64-npu-4') ||
           'linux-arm64-npu-2'
       }}
     strategy:
       matrix:
-        vllm_use_version: [0, 1]
+        vllm_use_version: [1]
         # the accuracy test will run:
         # 1. workflow_dispatch with models input
-        #   - all: Qwen/Qwen2.5-7B-Instruct, Qwen/Qwen2.5-VL-7B-Instruct, Qwen/Qwen3-8B-Base
-        #   - specified but not all: Qwen/Qwen2.5-7B-Instruct, Qwen/Qwen2.5-VL-7B-Instruct, Qwen/Qwen3-8B-Base
+        #   - all: Qwen/Qwen3-30B-A3B, Qwen/Qwen2.5-VL-7B-Instruct, Qwen/Qwen3-8B-Base
+        #   - specified but not all: Qwen/Qwen3-30B-A3B, Qwen/Qwen2.5-VL-7B-Instruct, Qwen/Qwen3-8B-Base
         # 2. PR labeled with "*-accuracy-test"
-        #   - accuracy-test: Qwen/Qwen2.5-7B-Instruct, Qwen/Qwen2.5-VL-7B-Instruct
-        #   - dense-accuracy-test: Qwen/Qwen2.5-7B-Instruct
+        #   - accuracy-test: Qwen/Qwen3-8B-Base, Qwen/Qwen2.5-VL-7B-Instruct, Qwen/Qwen3-30B-A3B
+        #   - dense-accuracy-test: Qwen/Qwen3-8B-Base
         #   - vl-accuracy-test: Qwen/Qwen2.5-VL-7B-Instruct
+        #   - moe-accuracy-test: Qwen/Qwen3-30B-A3B
         model_name: ${{ fromJSON(
           (github.event_name == 'schedule' &&
-            '["Qwen/Qwen2.5-7B-Instruct","Qwen/Qwen2.5-VL-7B-Instruct","Qwen/Qwen3-8B-Base"]') ||
+            '["Qwen/Qwen3-30B-A3B","Qwen/Qwen2.5-VL-7B-Instruct","Qwen/Qwen3-8B-Base"]') ||
           (github.event.inputs.models == 'all' &&
-            '["Qwen/Qwen2.5-7B-Instruct","Qwen/Qwen2.5-VL-7B-Instruct","Qwen/Qwen3-8B-Base"]') ||
-          (github.event.inputs.models == 'Qwen/Qwen2.5-7B-Instruct' &&
-            '["Qwen/Qwen2.5-7B-Instruct"]') ||
+            '["Qwen/Qwen3-30B-A3B","Qwen/Qwen2.5-VL-7B-Instruct","Qwen/Qwen3-8B-Base"]') ||
+          (github.event.inputs.models == 'Qwen/Qwen3-30B-A3B' &&
+            '["Qwen/Qwen3-30B-A3B"]') ||
           (github.event.inputs.models == 'Qwen/Qwen2.5-VL-7B-Instruct' &&
             '["Qwen/Qwen2.5-VL-7B-Instruct"]') ||
           (github.event.inputs.models == 'Qwen/Qwen3-8B-Base' &&
             '["Qwen/Qwen3-8B-Base"]') ||
           contains(github.event.pull_request.labels.*.name, 'accuracy-test') &&
-            '["Qwen/Qwen3-8B-Base","Qwen/Qwen2.5-VL-7B-Instruct"]' ||
+            '["Qwen/Qwen3-8B-Base","Qwen/Qwen2.5-VL-7B-Instruct", "Qwen/Qwen3-30B-A3B"]' ||
           contains(github.event.pull_request.labels.*.name, 'dense-accuracy-test') &&
             '["Qwen/Qwen3-8B-Base"]' ||
           contains(github.event.pull_request.labels.*.name, 'vl-accuracy-test') &&
-            '["Qwen/Qwen2.5-VL-7B-Instruct"]'
+            '["Qwen/Qwen2.5-VL-7B-Instruct"]' ||
+          contains(github.event.pull_request.labels.*.name, 'moe-accuracy-test') &&
+            '["Qwen/Qwen3-30B-A3B"]'
          ) }}
-        # Remove exclude after https://github.com/vllm-project/vllm-ascend/issues/1044 resolved
-        exclude:
-          - model_name: Qwen/Qwen2.5-VL-7B-Instruct
-            vllm_use_version: 1
 
       fail-fast: false
     name: ${{ matrix.model_name }} accuracy V${{ matrix.vllm_use_version }}
     container:
       image: m.daocloud.io/quay.io/ascend/cann:8.1.rc1-910b-ubuntu22.04-py3.10
       env:
-        HF_ENDPOINT: https://hf-mirror.com
-        HF_TOKEN: ${{ secrets.HF_TOKEN }}
         DATASET_SOURCE: ModelScope
         VLLM_USE_MODELSCOPE: True
+        USE_MODELSCOPE_HUB: 1
         # 1. If version specified (work_dispatch), do specified branch accuracy test
         # 2. If no version (labeled PR), do accuracy test by default ref:
         # The branch, tag or SHA to checkout. When checking out the repository that
@@ -188,23 +187,19 @@ jobs:
       - name: Get vLLM commit hash and URL
         working-directory: ./vllm-empty
         run: |
-          VLLM_COMMIT=$(git rev-parse HEAD)
+          VLLM_COMMIT=$(git rev-parse --short=7 HEAD)
           echo "VLLM_COMMIT=$VLLM_COMMIT" >> $GITHUB_ENV
-          echo "VLLM_COMMIT_URL=https://github.com/vllm-project/vllm/commit/$VLLM_COMMIT" >> $GITHUB_ENV
 
       - name: Get vLLM-Ascend commit hash and URL
         working-directory: ./vllm-ascend
         run: |
-          VLLM_ASCEND_COMMIT=$(git rev-parse HEAD)
+          VLLM_ASCEND_COMMIT=$(git rev-parse --short=7 HEAD)
           echo "VLLM_ASCEND_COMMIT=$VLLM_ASCEND_COMMIT" >> $GITHUB_ENV
-          echo "VLLM_ASCEND_COMMIT_URL=https://github.com/vllm-project/vllm-ascend/commit/$VLLM_ASCEND_COMMIT" >> $GITHUB_ENV
 
-      - name: Print resolved hashes and URLs
+      - name: Print resolved hashes
         run: |
           echo "vLLM       : ${{ env.VLLM_COMMIT }}"
-          echo "vLLM link  : ${{ env.VLLM_COMMIT_URL }}"
           echo "vLLM-Ascend: ${{ env.VLLM_ASCEND_COMMIT }}"
-          echo "Ascend link: ${{ env.VLLM_ASCEND_COMMIT_URL }}" 
 
       - name: Install lm-eval, ray, and datasets
         run: |
@@ -263,8 +258,6 @@ jobs:
             --vllm_version "${{ env.GHA_VLLM_VERSION }}" \
             --vllm_commit "${{ env.VLLM_COMMIT }}" \
             --vllm_ascend_commit "${{ env.VLLM_ASCEND_COMMIT }}" \
-            --vllm_commit_url "${{ env.VLLM_COMMIT_URL }}" \
-            --vllm_ascend_commit_url "${{ env.VLLM_ASCEND_COMMIT_URL }}" \
             --vllm_use_v1 "$VLLM_USE_V1"
 
       - name: Generate step summary
@@ -373,7 +366,7 @@ jobs:
           git push -f origin "${{ env.BRANCH_NAME }}"
 
       - name: Create PR in upstream via API
-        uses: actions/github-script@v6
+        uses: actions/github-script@v7
         with:
           github-token: ${{ secrets.PAT_TOKEN }}
           script: |
@@ -386,7 +379,7 @@ jobs:
               body: `The accuracy results running on NPU Altlas A2 have changed, updating reports for:
             ${{ 
               github.event.inputs.models == 'all' 
-                && 'All models (Qwen2.5-7B-Instruct, Qwen2.5-VL-7B-Instruct, Qwen3-8B-Base)' 
+                && 'All models (Qwen/Qwen3-30B-A3B, Qwen2.5-VL-7B-Instruct, Qwen3-8B-Base)' 
                 || github.event.inputs.models 
             }}
             
 
@@ -28,6 +28,6 @@ jobs:
       - name: Run codespell check
         run: |
           CODESPELL_EXCLUDES=('--skip' 'tests/prompts/**,./benchmarks/sonnet.txt,*tests/lora/data/**,build/**,./vllm_ascend.egg-info/**')
-          CODESPELL_IGNORE_WORDS=('-L' 'CANN,cann,NNAL,nnal,ASCEND,ascend,EnQue,CopyIn')
+          CODESPELL_IGNORE_WORDS=('-L' 'CANN,cann,NNAL,nnal,ASCEND,ascend,EnQue,CopyIn,assertIn,rever')
 
           codespell --toml pyproject.toml "${CODESPELL_EXCLUDES[@]}" "${CODESPELL_IGNORE_WORDS[@]}"
@@ -0,0 +1,63 @@
+#
+# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+# This file is a part of the vllm-ascend project.
+#
+
+name: format / pr body
+
+on:
+  # The PR updated when PR opened and push new commits
+  pull_request_target:
+    types: [opened, synchronize]
+    branches:
+      - 'main'
+
+permissions:
+  pull-requests: write
+
+jobs:
+  update-description:
+    name: update vLLM version
+    runs-on: ubuntu-latest
+
+    steps:
+      - name: Checkout vllm-project/vllm repo
+        uses: actions/checkout@v4
+        with:
+          repository: vllm-project/vllm
+          path: ./vllm-empty
+
+      - name: Get vLLM version
+        working-directory: ./vllm-empty
+        run: |
+          VLLM_COMMIT=$(git rev-parse HEAD)
+          echo "VLLM_COMMIT=https://github.com/vllm-project/vllm/commit/$VLLM_COMMIT" >> $GITHUB_ENV
+
+      - name: Checkout repository
+        uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2
+
+      - name: Set up Python
+        uses: actions/setup-python@42375524e23c412d93fb67b49958b491fce71c38 # v5.4.0
+
+      - name: Get vLLM release version
+        run: |
+          VLLM_VERSION=$(python3 docs/source/conf.py | jq .vllm_version | tr -d '"')
+          echo "VLLM_VERSION=$VLLM_VERSION" >> $GITHUB_ENV
+
+      - name: Update PR description
+        env:
+          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+        run: |
+          bash .github/format_pr_body.sh "${{ github.event.number }}" "${{ env.VLLM_VERSION }}" "${{ env.VLLM_COMMIT }}"
@@ -69,8 +69,7 @@ jobs:
         --device /dev/devmm_svm
         --device /dev/hisi_hdc
       env:
-        HF_ENDPOINT: https://hf-mirror.com
-        HF_TOKEN: ${{ secrets.HF_TOKEN }}
+        VLLM_USE_MODELSCOPE: True
         ES_OM_DOMAIN: ${{ secrets.ES_OM_DOMAIN }}
         ES_OM_AUTHORIZATION: ${{ secrets.ES_OM_AUTHORIZATION }}
         VLLM_USE_V1: ${{ matrix.vllm_use_v1 }}
@@ -115,6 +114,7 @@ jobs:
         env:
           PIP_EXTRA_INDEX_URL: https://mirrors.huaweicloud.com/ascend/repos/pypi
         run: |
+          pip install "transformers<=4.52.4"
           pip install -e .
           pip install -r benchmarks/requirements-bench.txt
 
@@ -197,7 +197,7 @@ jobs:
             --commit_title "$commit_title" \
             --created_at "$commit_time_no_tz" \
             --res_dir ./benchmarks/results \
-            --error $ERROR_MSG \
+            --error "$ERROR_MSG" \
             --extra_feat '{"VLLM_USE_V1": "${{ matrix.vllm_use_v1 }}"}'
             rm -rf ./benchmarks/results
             cd -
 
@@ -18,6 +18,9 @@
 name: build / wheel
 
 on:
+  schedule:
+    # Runs at 23:00 UTC (7:00 AM Beijing) every day
+    - cron: '0 23 * * *'
   pull_request:
     branches:
       - 'main'
@@ -55,7 +58,11 @@ jobs:
     strategy:
       matrix:
         os: [ubuntu-24.04, ubuntu-24.04-arm]
-        python-version: ['3.9', '3.10', '3.11']
+        # PR only trigger latest version
+        python-version: ${{ fromJSON(
+          (github.event_name == 'pull_request' && '["3.11"]') ||
+          '["3.9", "3.10", "3.11"]'
+         ) }}
     runs-on: ${{ matrix.os }}
     steps:
     - uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2