add model model usability test

Potabk · Potabk · commit 3889968df6ca · 2025-04-21T09:22:55.000+08:00
Signed-off-by: wangli &lt;wangli858794774@gmail.com&gt;
diff --git a/.github/workflows/vllm_ascend_test.yaml b/.github/workflows/vllm_ascend_test.yaml
@@ -120,6 +120,7 @@ jobs:
       - name: Run vllm-project/vllm-ascend test on V0 engine
         env:
           VLLM_USE_V1: 0
+          VLLM_WORKER_MULTIPROC_METHOD: spawn
         run: |
           if [[ "${{ matrix.os }}" == "linux-arm64-npu-1" ]]; then
             pytest -sv tests/singlecard/test_offline_inference.py
diff --git a/tests/conftest.py b/tests/conftest.py
@@ -17,6 +17,7 @@
 # Adapted from vllm-project/vllm/blob/main/tests/conftest.py
 #
 
+import contextlib
 import gc
 from typing import List, Optional, Tuple, TypeVar, Union
 
@@ -31,7 +32,7 @@
 from vllm.sampling_params import BeamSearchParams
 from vllm.utils import is_list_of
 
-from tests.model_utils import (TokensTextLogprobs,
+from tests.model_utils import (PROMPT_TEMPLATES, TokensTextLogprobs,
                                TokensTextLogprobsPromptLogprobs)
 # TODO: remove this part after the patch merged into vllm, if
 # we not explicitly patch here, some of them might be effectiveless
@@ -55,6 +56,8 @@
 def cleanup_dist_env_and_memory():
     destroy_model_parallel()
     destroy_distributed_environment()
+    with contextlib.suppress(AssertionError):
+        torch.distributed.destroy_process_group()
     gc.collect()
     torch.npu.empty_cache()
 
@@ -344,3 +347,8 @@ def __exit__(self, exc_type, exc_value, traceback):
 @pytest.fixture(scope="session")
 def vllm_runner():
     return VllmRunner
+
+
+@pytest.fixture(params=list(PROMPT_TEMPLATES.keys()))
+def prompt_template(request):
+    return PROMPT_TEMPLATES[request.param]
diff --git a/tests/model_utils.py b/tests/model_utils.py
@@ -18,7 +18,7 @@
 #
 
 import warnings
-from typing import Dict, List, Optional, Sequence, Tuple, Union
+from typing import Callable, Dict, List, Optional, Sequence, Tuple, Union
 
 import torch
 from vllm.config import ModelConfig, TaskOption
@@ -301,3 +301,16 @@ def build_model_context(model_name: str,
         limit_mm_per_prompt=limit_mm_per_prompt,
     )
     return InputContext(model_config)
+
+
+def qwen_prompt(questions: List[str]) -> List[str]:
+    placeholder = "<|image_pad|>"
+    return [("<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n"
+             f"<|im_start|>user\n<|vision_start|>{placeholder}<|vision_end|>"
+             f"{q}<|im_end|>\n<|im_start|>assistant\n") for q in questions]
+
+
+# Map of prompt templates for different models.
+PROMPT_TEMPLATES: dict[str, Callable] = {
+    "qwen2.5vl": qwen_prompt,
+}
diff --git a/tests/multicard/test_offline_inference_distributed.py b/tests/multicard/test_offline_inference_distributed.py
@@ -26,12 +26,14 @@
 import vllm  # noqa: F401
 
 from tests.conftest import VllmRunner
+from vllm.assets.image import ImageAsset
 
 os.environ["PYTORCH_NPU_ALLOC_CONF"] = "max_split_size_mb:256"
 
 
 @pytest.mark.parametrize("model, distributed_executor_backend", [
     ("Qwen/QwQ-32B", "mp"),
+    ("deepseek-ai/DeepSeek-V2-Lite", "mp"),
 ])
 def test_models_distributed(model: str,
                             distributed_executor_backend: str) -> None:
@@ -51,6 +53,34 @@ def test_models_distributed(model: str,
         vllm_model.generate_greedy(example_prompts, max_tokens)
 
 
+@pytest.mark.parametrize("model", ["Qwen/Qwen2.5-VL-32B-Instruct"])
+@pytest.mark.skipif(os.getenv("VLLM_USE_V1") == "1",
+                    reason="qwen2.5_vl is not supported on v1")
+def test_multimodal(model: str, prompt_template, vllm_runner):
+    image = ImageAsset("cherry_blossom") \
+        .pil_image.convert("RGB")
+    img_questions = [
+        "What is the content of this image?",
+        "Describe the content of this image in detail.",
+        "What's in the image?",
+        "Where is this image taken?",
+    ]
+    images = [image] * len(img_questions)
+    prompts = prompt_template(img_questions)
+    with vllm_runner(model,
+                     max_model_len=4096,
+                     tensor_parallel_size=4,
+                     distributed_executor_backend="mp",
+                     mm_processor_kwargs={
+                         "min_pixels": 28 * 28,
+                         "max_pixels": 1280 * 28 * 28,
+                         "fps": 1,
+                     }) as vllm_model:
+        vllm_model.generate_greedy(prompts=prompts,
+                                   images=images,
+                                   max_tokens=64)
+
+
 if __name__ == "__main__":
     import pytest
     pytest.main([__file__])
diff --git a/tests/ops/test_rotary_embedding.py b/tests/ops/test_rotary_embedding.py
@@ -202,3 +202,4 @@ def test_rotary_embedding_quant_with_leading_dim(
                                ref_key,
                                atol=DEFAULT_ATOL,
                                rtol=DEFAULT_RTOL)
+    torch.npu.empty_cache()
diff --git a/tests/singlecard/test_offline_inference.py b/tests/singlecard/test_offline_inference.py
@@ -24,14 +24,17 @@
 
 import pytest
 import vllm  # noqa: F401
+from vllm.assets.image import ImageAsset
 
 import vllm_ascend  # noqa: F401
 from tests.conftest import VllmRunner
+from vllm.assets.image import ImageAsset
 
 MODELS = [
     "Qwen/Qwen2.5-0.5B-Instruct",
     "vllm-ascend/Qwen2.5-0.5B-Instruct-w8a8",
 ]
+MULTIMODALITY_MODELS = ["Qwen/Qwen2.5-VL-3B-Instruct"]
 os.environ["VLLM_USE_MODELSCOPE"] = "True"
 os.environ["PYTORCH_NPU_ALLOC_CONF"] = "max_split_size_mb:256"
 
@@ -55,6 +58,32 @@ def test_models(model: str, dtype: str, max_tokens: int) -> None:
         vllm_model.generate_greedy(example_prompts, max_tokens)
 
 
+@pytest.mark.parametrize("model", MULTIMODALITY_MODELS)
+@pytest.mark.skipif(os.getenv("VLLM_USE_V1") == "1",
+                    reason="qwen2.5_vl is not supported on v1")
+def test_multimodal(model: str, prompt_template, vllm_runner):
+    image = ImageAsset("cherry_blossom") \
+        .pil_image.convert("RGB")
+    img_questions = [
+        "What is the content of this image?",
+        "Describe the content of this image in detail.",
+        "What's in the image?",
+        "Where is this image taken?",
+    ]
+    images = [image] * len(img_questions)
+    prompts = prompt_template(img_questions)
+    with vllm_runner(model,
+                     max_model_len=4096,
+                     mm_processor_kwargs={
+                         "min_pixels": 28 * 28,
+                         "max_pixels": 1280 * 28 * 28,
+                         "fps": 1,
+                     }) as vllm_model:
+        vllm_model.generate_greedy(prompts=prompts,
+                                   images=images,
+                                   max_tokens=64)
+
+
 if __name__ == "__main__":
     import pytest
     pytest.main([__file__])