vllm-project · wangxiyuan · Jun 16, 2025 · Jun 13, 2025
diff --git a/.github/workflows/vllm_ascend_test.yaml b/.github/workflows/vllm_ascend_test.yaml
@@ -15,7 +15,7 @@
 # This file is a part of the vllm-ascend project.
 #
 
-name: 'e2e test / basic'
+name: 'test'
 
 on:
   schedule:
@@ -114,6 +114,56 @@ jobs:
           echo "::add-matcher::.github/workflows/matchers/mypy.json"
           tools/mypy.sh 1 ${{ matrix.python-version }}
 
+  ut:
+    needs: [lint]
+    name: unit test
+    if: ${{ needs.lint.result == 'success' }}
+    runs-on: ubuntu-latest
+    container:
+      image: m.daocloud.io/quay.io/ascend/cann:8.1.rc1-910b-ubuntu22.04-py3.10
+      env:
+        VLLM_LOGGING_LEVEL: ERROR
+        VLLM_USE_MODELSCOPE: True
+    strategy:
+      matrix:
+        vllm_version: [main, v0.9.1]
+    steps:
+      - name: Install packages
+        run: |
+          apt-get update -y
+          apt-get install -y python3-pip git vim wget net-tools gcc g++ cmake libnuma-dev
+
+      - name: Checkout vllm-project/vllm repo
+        uses: actions/checkout@v4
+        with:
+          repository: vllm-project/vllm
+          ref: ${{ matrix.vllm_version }}
+          path: ./vllm-empty
+
+      - name: Install vllm-project/vllm from source
+        working-directory: ./vllm-empty
+        run: |
+          VLLM_TARGET_DEVICE=empty python3 -m pip install . --extra-index https://download.pytorch.org/whl/cpu/
+          python3 -m pip uninstall -y triton
+
+      - name: Checkout vllm-project/vllm-ascend repo
+        uses: actions/checkout@v4
+
+      - name: Install vllm-project/vllm-ascend
+        run: |
+          export LD_LIBRARY_PATH=$LD_LIBRARY_PATH:/usr/local/Ascend/ascend-toolkit/latest/x86_64-linux/devlib
+          python3 -m pip install -r requirements-dev.txt --extra-index https://download.pytorch.org/whl/cpu/
+          python3 -m pip install -v . --extra-index https://download.pytorch.org/whl/cpu/
+
+      - name: Run unit test for V1 Engine
+        env:
+          VLLM_USE_V1: 1
+          VLLM_WORKER_MULTIPROC_METHOD: spawn
+          TORCH_DEVICE_BACKEND_AUTOLOAD: 0
+        run: |
+          export LD_LIBRARY_PATH=$LD_LIBRARY_PATH:/usr/local/Ascend/ascend-toolkit/latest/x86_64-linux/devlib
+          pytest -sv tests/ut
+
   e2e:
     needs: [lint]
     if: ${{ needs.lint.result == 'success' }}
@@ -122,7 +172,7 @@ jobs:
       matrix:
         os: [linux-arm64-npu-1]
         vllm_version: [main, v0.9.1]
-    name: vLLM Ascend test
+    name: singlecard e2e test
     runs-on: ${{ matrix.os }}
     container:
       # TODO(yikun): Remove m.daocloud.io prefix when infra proxy ready
@@ -168,53 +218,47 @@ jobs:
           pip install -r requirements-dev.txt
           pip install -v -e .
 
-      - name: Run vllm-project/vllm-ascend test for V1 Engine
+      - name: Run e2e test for V1 Engine
         env:
           VLLM_USE_V1: 1
           VLLM_WORKER_MULTIPROC_METHOD: spawn
           VLLM_USE_MODELSCOPE: True
         run: |
-          pytest -sv tests/singlecard/test_offline_inference.py
+          pytest -sv tests/e2e/singlecard/test_offline_inference.py
           # TODO: switch hf to modelscope
           VLLM_USE_MODELSCOPE=False HF_ENDPOINT=https://hf-mirror.com \
-            pytest -sv tests/singlecard/test_ilama_lora.py
+            pytest -sv tests/e2e/singlecard/test_ilama_lora.py
           # TODO(sss): guided decoding doesn't work, fix it later
-          # pytest -sv tests/singlecard/test_guided_decoding.py
-          # test_ascend_config.py should be ran separately because it will regenerate the global config many times.
-          pytest -sv tests/singlecard/test_ascend_config.py
-          pytest -sv tests/singlecard/test_camem.py
-          pytest -sv tests/singlecard/ \
-          --ignore=tests/singlecard/test_offline_inference.py \
-          --ignore=tests/singlecard/test_ilama_lora.py \
-          --ignore=tests/singlecard/test_guided_decoding.py \
-          --ignore=tests/singlecard/test_ascend_config.py \
-          --ignore=tests/singlecard/test_camem.py
+          # pytest -sv tests/e2e/singlecard/test_guided_decoding.py
+          pytest -sv tests/e2e/singlecard/test_camem.py
+          pytest -sv tests/e2e/singlecard/ \
+          --ignore=tests/e2e/singlecard/test_offline_inference.py \
+          --ignore=tests/e2e/singlecard/test_ilama_lora.py \
+          --ignore=tests/e2e/singlecard/test_guided_decoding.py \
+          --ignore=tests/e2e/singlecard/test_camem.py
 
-      - name: Run vllm-project/vllm-ascend test on V0 engine
+      - name: Run e2e test on V0 engine
         if: ${{ github.event_name == 'schedule' }}
         env:
           VLLM_USE_V1: 0
           VLLM_USE_MODELSCOPE: True
         run: |
-          pytest -sv tests/singlecard/test_offline_inference.py
+          pytest -sv tests/e2e/singlecard/test_offline_inference.py
           # TODO: switch hf to modelscope
           VLLM_USE_MODELSCOPE=False HF_ENDPOINT=https://hf-mirror.com \
-            pytest -sv tests/singlecard/test_ilama_lora.py
+            pytest -sv tests/e2e/singlecard/test_ilama_lora.py
           # guided decoding doesn't work, fix it later
-          # pytest -sv tests/singlecard/test_guided_decoding.py
-          pytest -sv tests/singlecard/test_camem.py
-          # test_ascend_config.py should be ran separately because it will regenerate the global config many times.
-          pytest -sv tests/singlecard/test_ascend_config.py
-          pytest -sv tests/singlecard/test_prompt_embedding.py
-          pytest -sv tests/singlecard/ \
-            --ignore=tests/singlecard/test_offline_inference.py \
-            --ignore=tests/singlecard/test_ilama_lora.py \
-            --ignore=tests/singlecard/test_guided_decoding.py \
-            --ignore=tests/singlecard/test_camem.py \
-            --ignore=tests/singlecard/test_ascend_config.py \
-            --ignore=tests/singlecard/test_prompt_embedding.py \
-            --ignore=tests/singlecard/core/test_ascend_scheduler.py \
-            --ignore=tests/singlecard/core/test_ascend_scheduler_e2e.py
+          # pytest -sv tests/e2e/singlecard/test_guided_decoding.py
+          pytest -sv tests/e2e/singlecard/test_camem.py
+          pytest -sv tests/e2e/singlecard/test_prompt_embedding.py
+          pytest -sv tests/e2e/singlecard/ \
+            --ignore=tests/e2e/singlecard/test_offline_inference.py \
+            --ignore=tests/e2e/singlecard/test_ilama_lora.py \
+            --ignore=tests/e2e/singlecard/test_guided_decoding.py \
+            --ignore=tests/e2e/singlecard/test_camem.py \
+            --ignore=tests/e2e/singlecard/test_prompt_embedding.py \
+            --ignore=tests/e2e/singlecard/core/test_ascend_scheduler.py \
+            --ignore=tests/e2e/singlecard/core/test_ascend_scheduler_e2e.py
 
   e2e-4-cards:
     needs: [e2e]
@@ -224,7 +268,7 @@ jobs:
       matrix:
         os: [linux-arm64-npu-4]
         vllm_version: [main, v0.9.1]
-    name: vLLM Ascend test
+    name: multicard e2e test
     runs-on: ${{ matrix.os }}
     container:
       # TODO(yikun): Remove m.daocloud.io prefix when infra proxy ready
@@ -279,14 +323,14 @@ jobs:
         run: |
           # TODO: switch hf to modelscope
           VLLM_USE_MODELSCOPE=False HF_ENDPOINT=https://hf-mirror.com \
-            pytest -sv tests/multicard/test_ilama_lora_tp2.py
-          # Fixme: run VLLM_USE_MODELSCOPE=True pytest -sv tests/multicard/test_offline_inference_distributed.py will raise error.
+            pytest -sv tests/e2e/multicard/test_ilama_lora_tp2.py
+          # Fixme: run VLLM_USE_MODELSCOPE=True pytest -sv tests/e2e/multicard/test_offline_inference_distributed.py will raise error.
           # To avoid oom, we need to run the test in a single process.
-          pytest -sv tests/multicard/test_offline_inference_distributed.py::test_models_distributed_QwQ
-          pytest -sv tests/multicard/test_offline_inference_distributed.py::test_models_distributed_DeepSeek
-          pytest -sv tests/multicard/test_offline_inference_distributed.py::test_models_distributed_topk
-          pytest -sv tests/multicard/test_offline_inference_distributed.py::test_models_distributed_DeepSeek_W8A8
-          pytest -sv tests/multicard/ --ignore=tests/multicard/test_ilama_lora_tp2.py --ignore=tests/multicard/test_offline_inference_distributed.py
+          pytest -sv tests/e2e/multicard/test_offline_inference_distributed.py::test_models_distributed_QwQ
+          pytest -sv tests/e2e/multicard/test_offline_inference_distributed.py::test_models_distributed_DeepSeek
+          pytest -sv tests/e2e/multicard/test_offline_inference_distributed.py::test_models_distributed_topk
+          pytest -sv tests/e2e/multicard/test_offline_inference_distributed.py::test_models_distributed_DeepSeek_W8A8
+          pytest -sv tests/e2e/multicard/ --ignore=tests/e2e/multicard/test_ilama_lora_tp2.py --ignore=tests/e2e/multicard/test_offline_inference_distributed.py
 
       - name: Run vllm-project/vllm-ascend test on V0 engine
         if: ${{ github.event_name == 'schedule' }}
@@ -296,11 +340,11 @@ jobs:
         run: |
           # TODO: switch hf to modelscope
           VLLM_USE_MODELSCOPE=False HF_ENDPOINT=https://hf-mirror.com \
-            pytest -sv tests/multicard/test_ilama_lora_tp2.py
-          # Fixme: run VLLM_USE_MODELSCOPE=True pytest -sv tests/multicard/test_offline_inference_distributed.py will raise error.
+            pytest -sv tests/e2e/multicard/test_ilama_lora_tp2.py
+          # Fixme: run VLLM_USE_MODELSCOPE=True pytest -sv tests/e2e/multicard/test_offline_inference_distributed.py will raise error.
           # To avoid oom, we need to run the test in a single process.
-          pytest -sv tests/multicard/test_offline_inference_distributed.py::test_models_distributed_QwQ
-          pytest -sv tests/multicard/test_offline_inference_distributed.py::test_models_distributed_DeepSeek
-          pytest -sv tests/multicard/test_offline_inference_distributed.py::test_models_distributed_topk
-          pytest -sv tests/multicard/test_offline_inference_distributed.py::test_models_distributed_DeepSeek_W8A8
-          pytest -sv tests/multicard/ --ignore=tests/multicard/test_ilama_lora_tp2.py --ignore=tests/multicard/test_offline_inference_distributed.py
+          pytest -sv tests/e2e/multicard/test_offline_inference_distributed.py::test_models_distributed_QwQ
+          pytest -sv tests/e2e/multicard/test_offline_inference_distributed.py::test_models_distributed_DeepSeek
+          pytest -sv tests/e2e/multicard/test_offline_inference_distributed.py::test_models_distributed_topk
+          pytest -sv tests/e2e/multicard/test_offline_inference_distributed.py::test_models_distributed_DeepSeek_W8A8
+          pytest -sv tests/e2e/multicard/ --ignore=tests/e2e/multicard/test_ilama_lora_tp2.py --ignore=tests/e2e/multicard/test_offline_inference_distributed.py
diff --git a/.github/workflows/vllm_ascend_test_long_term.yaml b/.github/workflows/vllm_ascend_test_long_term.yaml
@@ -96,12 +96,12 @@ jobs:
         run: |
           if [[ "${{ matrix.os }}" == "linux-arm64-npu-1" ]]; then
             # spec decode test
-            VLLM_USE_MODELSCOPE=True pytest -sv tests/long_term/spec_decode/e2e/test_v1_mtp_correctness.py
+            VLLM_USE_MODELSCOPE=True pytest -sv tests/e2e/long_term/spec_decode/e2e/test_v1_mtp_correctness.py
             # TODO: revert me when test_v1_spec_decode.py::test_ngram_correctness is fixed
-            # VLLM_USE_MODELSCOPE=True pytest -sv tests/long_term/spec_decode/e2e/test_v1_spec_decode.py
-            VLLM_USE_MODELSCOPE=True pytest -sv tests/long_term/spec_decode/e2e/test_mtp_correctness.py  # it needs a clean process
-            pytest -sv tests/long_term/spec_decode --ignore=tests/long_term/spec_decode/e2e/test_mtp_correctness.py --ignore=tests/long_term/spec_decode/e2e/test_v1_spec_decode.py --ignore=tests/long_term/spec_decode/e2e/test_v1_mtp_correctness.py
-            pytest -sv tests/long_term/test_accuracy.py
+            # VLLM_USE_MODELSCOPE=True pytest -sv tests/e2e/long_term/spec_decode/e2e/test_v1_spec_decode.py
+            VLLM_USE_MODELSCOPE=True pytest -sv tests/e2e/long_term/spec_decode/e2e/test_mtp_correctness.py  # it needs a clean process
+            pytest -sv tests/e2e/long_term/spec_decode --ignore=tests/e2e/long_term/spec_decode/e2e/test_mtp_correctness.py --ignore=tests/e2e/long_term/spec_decode/e2e/test_v1_spec_decode.py --ignore=tests/e2e/long_term/spec_decode/e2e/test_v1_mtp_correctness.py
+            pytest -sv tests/e2e/long_term/test_accuracy.py
           else
-            VLLM_USE_MODELSCOPE=True pytest -sv tests/long_term/test_deepseek_v2_lite_tp2_accuracy.py
+            VLLM_USE_MODELSCOPE=True pytest -sv tests/e2e/long_term/test_deepseek_v2_lite_tp2_accuracy.py
           fi
diff --git a/format.sh b/format.sh
@@ -273,7 +273,7 @@ echo 'vllm-ascend isort: Done'
 # Clang-format section
 # Exclude some files for formatting because they are vendored
 CLANG_FORMAT_EXCLUDES=(
-    'csrc/kernels/pos_encoding_kernels.cpp' 'csrc/kernels/advance_step.cpp' 'csrc/torch_binding.cpp' 'csrc/ops.h'
+    'csrc/kernels/pos_encoding_kernels.cpp' 'csrc/kernels/advance_step.cpp' 'csrc/kernels/get_masked_input_and_mask_kernel.cpp' 'csrc/torch_binding.cpp' 'csrc/ops.h'
 )
 
 # Format specified files with clang-format

diff --git a/tests/long_term/spec_decode/__init__.py → tests/e2e/long_term/spec_decode/__init__.py b/tests/long_term/spec_decode/__init__.py → tests/e2e/long_term/spec_decode/__init__.py
diff --git a/tests/long_term/spec_decode/conftest.py → tests/e2e/long_term/spec_decode/conftest.py b/tests/long_term/spec_decode/conftest.py → tests/e2e/long_term/spec_decode/conftest.py
diff --git a/tests/long_term/spec_decode/e2e/__init__.py → ...e2e/long_term/spec_decode/e2e/__init__.py b/tests/long_term/spec_decode/e2e/__init__.py → ...e2e/long_term/spec_decode/e2e/__init__.py
diff --git a/tests/long_term/spec_decode/e2e/conftest.py → ...e2e/long_term/spec_decode/e2e/conftest.py b/tests/long_term/spec_decode/e2e/conftest.py → ...e2e/long_term/spec_decode/e2e/conftest.py
diff --git a/...pec_decode/e2e/test_medusa_correctness.py → ...pec_decode/e2e/test_medusa_correctness.py b/...pec_decode/e2e/test_medusa_correctness.py → ...pec_decode/e2e/test_medusa_correctness.py
@@ -41,9 +41,9 @@
 
 import pytest
 
-from tests.long_term.spec_decode.e2e.conftest import \
+from tests.e2e.long_term.spec_decode.e2e.conftest import \
     run_equality_correctness_test
-from tests.long_term.spec_decode.utils import maybe_enable_chunked_prefill
+from tests.e2e.long_term.spec_decode.utils import maybe_enable_chunked_prefill
 
 # main model
 # lmsys/vicuna-7b-v1.3 was to be used but it's causing

diff --git a/...m/spec_decode/e2e/test_mlp_correctness.py → ...m/spec_decode/e2e/test_mlp_correctness.py b/...m/spec_decode/e2e/test_mlp_correctness.py → ...m/spec_decode/e2e/test_mlp_correctness.py
@@ -41,9 +41,9 @@
 from vllm.model_executor.layers.vocab_parallel_embedding import \
     pad_vocab_size  # noqa: F401
 
-from tests.long_term.spec_decode.e2e.conftest import \
+from tests.e2e.long_term.spec_decode.e2e.conftest import \
     run_equality_correctness_test
-from tests.long_term.spec_decode.utils import maybe_enable_chunked_prefill
+from tests.e2e.long_term.spec_decode.utils import maybe_enable_chunked_prefill
 
 # main model
 MAIN_MODEL = "JackFram/llama-160m"

diff --git a/...m/spec_decode/e2e/test_mtp_correctness.py → ...m/spec_decode/e2e/test_mtp_correctness.py b/...m/spec_decode/e2e/test_mtp_correctness.py → ...m/spec_decode/e2e/test_mtp_correctness.py
diff --git a/...spec_decode/e2e/test_ngram_correctness.py → ...spec_decode/e2e/test_ngram_correctness.py b/...spec_decode/e2e/test_ngram_correctness.py → ...spec_decode/e2e/test_ngram_correctness.py
@@ -44,9 +44,9 @@
 
 import pytest
 
-from tests.long_term.spec_decode.e2e.conftest import \
+from tests.e2e.long_term.spec_decode.e2e.conftest import \
     run_equality_correctness_test
-from tests.long_term.spec_decode.utils import maybe_enable_chunked_prefill
+from tests.e2e.long_term.spec_decode.utils import maybe_enable_chunked_prefill
 
 
 @pytest.mark.parametrize(

diff --git a/...pec_decode/e2e/test_v1_mtp_correctness.py → ...pec_decode/e2e/test_v1_mtp_correctness.py b/...pec_decode/e2e/test_v1_mtp_correctness.py → ...pec_decode/e2e/test_v1_mtp_correctness.py
diff --git a/...rm/spec_decode/e2e/test_v1_spec_decode.py → ...rm/spec_decode/e2e/test_v1_spec_decode.py b/...rm/spec_decode/e2e/test_v1_spec_decode.py → ...rm/spec_decode/e2e/test_v1_spec_decode.py
diff --git a/...m/spec_decode/test_dynamic_spec_decode.py → ...m/spec_decode/test_dynamic_spec_decode.py b/...m/spec_decode/test_dynamic_spec_decode.py → ...m/spec_decode/test_dynamic_spec_decode.py
@@ -27,8 +27,8 @@
 from vllm.spec_decode.spec_decode_worker import SpecDecodeWorker
 from vllm.spec_decode.top1_proposer import Top1Proposer
 
-from tests.long_term.spec_decode.test_utils import mock_spec_decode_sampler
-from tests.long_term.spec_decode.utils import create_batch, mock_worker
+from tests.e2e.long_term.spec_decode.test_utils import mock_spec_decode_sampler
+from tests.e2e.long_term.spec_decode.utils import create_batch, mock_worker
 
 
 @pytest.mark.parametrize('queue_size', [4])

diff --git a/...erm/spec_decode/test_multi_step_worker.py → ...erm/spec_decode/test_multi_step_worker.py b/...erm/spec_decode/test_multi_step_worker.py → ...erm/spec_decode/test_multi_step_worker.py
@@ -29,7 +29,7 @@
 from vllm.spec_decode.multi_step_worker import MultiStepWorker
 from vllm.spec_decode.top1_proposer import Top1Proposer
 
-from tests.long_term.spec_decode.utils import (
+from tests.e2e.long_term.spec_decode.utils import (
     assert_logprobs_dict_allclose, create_batch,
     create_seq_group_metadata_from_prompts, create_worker,
     patch_execute_model_with_seeds, zero_kv_cache)

diff --git a/...ong_term/spec_decode/test_ngram_worker.py → ...ong_term/spec_decode/test_ngram_worker.py b/...ong_term/spec_decode/test_ngram_worker.py → ...ong_term/spec_decode/test_ngram_worker.py
@@ -22,7 +22,7 @@
 from vllm.spec_decode.ngram_worker import NGramWorker
 from vllm.spec_decode.top1_proposer import Top1Proposer
 
-from tests.long_term.spec_decode.utils import (
+from tests.e2e.long_term.spec_decode.utils import (
     create_seq_group_metadata_from_prompts, create_worker)
 
 

diff --git a/...rm/spec_decode/test_spec_decode_worker.py → ...rm/spec_decode/test_spec_decode_worker.py b/...rm/spec_decode/test_spec_decode_worker.py → ...rm/spec_decode/test_spec_decode_worker.py
@@ -35,10 +35,10 @@
 from vllm.spec_decode.spec_decode_worker import (SpecDecodeWorker,
                                                  split_num_cache_blocks_evenly)
 
-from tests.long_term.spec_decode.test_utils import mock_spec_decode_sampler
-from tests.long_term.spec_decode.utils import (create_batch,
-                                               create_sampler_output_list,
-                                               create_worker, mock_worker)
+from tests.e2e.long_term.spec_decode.test_utils import mock_spec_decode_sampler
+from tests.e2e.long_term.spec_decode.utils import (create_batch,
+                                                   create_sampler_output_list,
+                                                   create_worker, mock_worker)
 from vllm_ascend.worker.draft_model_runner import TP1DraftModelRunner
 from vllm_ascend.worker.worker import NPUWorker
 

diff --git a/tests/long_term/spec_decode/test_utils.py → ...s/e2e/long_term/spec_decode/test_utils.py b/tests/long_term/spec_decode/test_utils.py → ...s/e2e/long_term/spec_decode/test_utils.py
diff --git a/tests/long_term/spec_decode/utils.py → tests/e2e/long_term/spec_decode/utils.py b/tests/long_term/spec_decode/utils.py → tests/e2e/long_term/spec_decode/utils.py
diff --git a/tests/long_term/test_accuracy.py → tests/e2e/long_term/test_accuracy.py b/tests/long_term/test_accuracy.py → tests/e2e/long_term/test_accuracy.py
diff --git a/...erm/test_deepseek_v2_lite_tp2_accuracy.py → ...erm/test_deepseek_v2_lite_tp2_accuracy.py b/...erm/test_deepseek_v2_lite_tp2_accuracy.py → ...erm/test_deepseek_v2_lite_tp2_accuracy.py
diff --git a/...ticard/test_dynamic_npugraph_batchsize.py → ...ticard/test_dynamic_npugraph_batchsize.py b/...ticard/test_dynamic_npugraph_batchsize.py → ...ticard/test_dynamic_npugraph_batchsize.py
diff --git a/tests/multicard/test_ilama_lora_tp2.py → tests/e2e/multicard/test_ilama_lora_tp2.py b/tests/multicard/test_ilama_lora_tp2.py → tests/e2e/multicard/test_ilama_lora_tp2.py
@@ -1,8 +1,8 @@
 import pytest
 
 from tests.conftest import VllmRunner
-from tests.singlecard.test_ilama_lora import (EXPECTED_LORA_OUTPUT, MODEL_PATH,
-                                              do_sample)
+from tests.e2e.singlecard.test_ilama_lora import (EXPECTED_LORA_OUTPUT,
+                                                  MODEL_PATH, do_sample)
 
 
 @pytest.mark.parametrize("distributed_executor_backend", ["mp"])

diff --git a/...ard/test_offline_inference_distributed.py → ...ard/test_offline_inference_distributed.py b/...ard/test_offline_inference_distributed.py → ...ard/test_offline_inference_distributed.py
diff --git a/tests/multicard/test_pyhccl_distributed.py → .../e2e/multicard/test_pyhccl_distributed.py b/tests/multicard/test_pyhccl_distributed.py → .../e2e/multicard/test_pyhccl_distributed.py
diff --git a/tests/multicard/test_torchair_graph_mode.py → ...e2e/multicard/test_torchair_graph_mode.py b/tests/multicard/test_torchair_graph_mode.py → ...e2e/multicard/test_torchair_graph_mode.py
diff --git a/tests/singlecard/__init__.py → tests/e2e/singlecard/__init__.py b/tests/singlecard/__init__.py → tests/e2e/singlecard/__init__.py
diff --git a/tests/singlecard/compile/__init__.py → tests/e2e/singlecard/compile/__init__.py b/tests/singlecard/compile/__init__.py → tests/e2e/singlecard/compile/__init__.py
diff --git a/tests/singlecard/compile/test_simple.py → tests/e2e/singlecard/compile/test_simple.py b/tests/singlecard/compile/test_simple.py → tests/e2e/singlecard/compile/test_simple.py
diff --git a/tests/singlecard/core/__init__.py → tests/e2e/singlecard/core/__init__.py b/tests/singlecard/core/__init__.py → tests/e2e/singlecard/core/__init__.py
diff --git a/.../singlecard/core/test_ascend_scheduler.py → .../singlecard/core/test_ascend_scheduler.py b/.../singlecard/core/test_ascend_scheduler.py → .../singlecard/core/test_ascend_scheduler.py
diff --git a/...glecard/core/test_ascend_scheduler_e2e.py → ...glecard/core/test_ascend_scheduler_e2e.py b/...glecard/core/test_ascend_scheduler_e2e.py → ...glecard/core/test_ascend_scheduler_e2e.py
diff --git a/tests/singlecard/ops/__init__.py → tests/e2e/singlecard/ops/__init__.py b/tests/singlecard/ops/__init__.py → tests/e2e/singlecard/ops/__init__.py
diff --git a/tests/singlecard/ops/test_fused_moe.py → tests/e2e/singlecard/ops/test_fused_moe.py b/tests/singlecard/ops/test_fused_moe.py → tests/e2e/singlecard/ops/test_fused_moe.py
diff --git a/tests/singlecard/ops/test_multi_step.py → tests/e2e/singlecard/ops/test_multi_step.py b/tests/singlecard/ops/test_multi_step.py → tests/e2e/singlecard/ops/test_multi_step.py
diff --git a/...s/singlecard/ops/test_rotary_embedding.py → ...e/singlecard/ops/test_rotary_embedding.py b/...s/singlecard/ops/test_rotary_embedding.py → ...e/singlecard/ops/test_rotary_embedding.py
diff --git a/tests/ops/test_vocabparallelembedding.py → ...lecard/ops/test_vocabparallelembedding.py b/tests/ops/test_vocabparallelembedding.py → ...lecard/ops/test_vocabparallelembedding.py
diff --git a/tests/singlecard/sample/__init__.py → tests/e2e/singlecard/sample/__init__.py b/tests/singlecard/sample/__init__.py → tests/e2e/singlecard/sample/__init__.py
diff --git a/...nglecard/sample/test_rejection_sampler.py → ...nglecard/sample/test_rejection_sampler.py b/...nglecard/sample/test_rejection_sampler.py → ...nglecard/sample/test_rejection_sampler.py
diff --git a/tests/singlecard/test_aclgraph.py → tests/e2e/singlecard/test_aclgraph.py b/tests/singlecard/test_aclgraph.py → tests/e2e/singlecard/test_aclgraph.py
diff --git a/tests/singlecard/test_camem.py → tests/e2e/singlecard/test_camem.py b/tests/singlecard/test_camem.py → tests/e2e/singlecard/test_camem.py
diff --git a/tests/singlecard/test_chunked.py → tests/e2e/singlecard/test_chunked.py b/tests/singlecard/test_chunked.py → tests/e2e/singlecard/test_chunked.py
diff --git a/tests/singlecard/test_guided_decoding.py → tests/e2e/singlecard/test_guided_decoding.py b/tests/singlecard/test_guided_decoding.py → tests/e2e/singlecard/test_guided_decoding.py
diff --git a/tests/singlecard/test_ilama_lora.py → tests/e2e/singlecard/test_ilama_lora.py b/tests/singlecard/test_ilama_lora.py → tests/e2e/singlecard/test_ilama_lora.py
diff --git a/tests/singlecard/test_offline_inference.py → .../e2e/singlecard/test_offline_inference.py b/tests/singlecard/test_offline_inference.py → .../e2e/singlecard/test_offline_inference.py
diff --git a/...nglecard/test_profile_execute_duration.py → ...nglecard/test_profile_execute_duration.py b/...nglecard/test_profile_execute_duration.py → ...nglecard/test_profile_execute_duration.py
diff --git a/tests/singlecard/test_prompt_embedding.py → ...s/e2e/singlecard/test_prompt_embedding.py b/tests/singlecard/test_prompt_embedding.py → ...s/e2e/singlecard/test_prompt_embedding.py
diff --git a/tests/singlecard/test_pyhccl.py → tests/e2e/singlecard/test_pyhccl.py b/tests/singlecard/test_pyhccl.py → tests/e2e/singlecard/test_pyhccl.py
diff --git a/tests/singlecard/test_sampler.py → tests/e2e/singlecard/test_sampler.py b/tests/singlecard/test_sampler.py → tests/e2e/singlecard/test_sampler.py
diff --git a/tests/singlecard/test_scheduler.py → tests/e2e/singlecard/test_scheduler.py b/tests/singlecard/test_scheduler.py → tests/e2e/singlecard/test_scheduler.py