use modelscope

MengqingCao · MengqingCao · commit 8c983091c3fe · 2025-04-21T12:36:55.000Z
Signed-off-by: MengqingCao &lt;cmq0113@163.com&gt;
diff --git a/tests/singlecard/embedding/test_embedding.py b/tests/singlecard/embedding/test_embedding.py
@@ -60,34 +60,36 @@ def test_models(
     example_prompts,
     model,
     dtype: str,
+    monkeypatch: pytest.MonkeyPatch,
 ) -> None:
+    with monkeypatch.context() as m:
+        m.setenv("VLLM_USE_MODELSCOPE", "True")
+        vllm_extra_kwargs: Dict[str, Any] = {}
 
-    vllm_extra_kwargs: Dict[str, Any] = {}
+        # The example_prompts has ending "\n", for example:
+        # "Write a short story about a robot that dreams for the first time.\n"
+        # sentence_transformers will strip the input texts, see:
+        # https://github.com/UKPLab/sentence-transformers/blob/v3.1.1/sentence_transformers/models/Transformer.py#L159
+        # This makes the input_ids different between hf_model and vllm_model.
+        # So we need to strip the input texts to avoid test failing.
+        example_prompts = [str(s).strip() for s in example_prompts]
 
-    # The example_prompts has ending "\n", for example:
-    # "Write a short story about a robot that dreams for the first time.\n"
-    # sentence_transformers will strip the input texts, see:
-    # https://github.com/UKPLab/sentence-transformers/blob/v3.1.1/sentence_transformers/models/Transformer.py#L159
-    # This makes the input_ids different between hf_model and vllm_model.
-    # So we need to strip the input texts to avoid test failing.
-    example_prompts = [str(s).strip() for s in example_prompts]
+        with vllm_runner(model,
+                        task="embed",
+                        dtype=dtype,
+                        max_model_len=None,
+                        **vllm_extra_kwargs) as vllm_model:
+            vllm_outputs = vllm_model.encode(example_prompts)
 
-    with vllm_runner(model,
-                     task="embed",
-                     dtype=dtype,
-                     max_model_len=None,
-                     **vllm_extra_kwargs) as vllm_model:
-        vllm_outputs = vllm_model.encode(example_prompts)
+        with hf_runner(MODELSCOPE_CACHE + model,
+                    dtype=dtype,
+                    is_sentence_transformer=True) as hf_model:
+            hf_outputs = hf_model.encode(example_prompts)
 
-    with hf_runner(MODELSCOPE_CACHE + model,
-                   dtype=dtype,
-                   is_sentence_transformer=True) as hf_model:
-        hf_outputs = hf_model.encode(example_prompts)
-
-    check_embeddings_close(
-        embeddings_0_lst=hf_outputs,
-        embeddings_1_lst=vllm_outputs,
-        name_0="hf",
-        name_1="vllm",
-        tol=1e-2,
-    )
+        check_embeddings_close(
+            embeddings_0_lst=hf_outputs,
+            embeddings_1_lst=vllm_outputs,
+            name_0="hf",
+            name_1="vllm",
+            tol=1e-2,
+        )