add guided decoding

Potabk · Potabk · commit 31e8138e7f4a · 2025-04-02T07:19:05.000Z
Signed-off-by: wangli &lt;wangli858794774@gmail.com&gt;
diff --git a/fusion_result.json b/fusion_result.json
@@ -0,0 +1 @@
+null
diff --git a/requirements-dev.txt b/requirements-dev.txt
@@ -2,3 +2,4 @@
 modelscope
 pytest >= 6.0
 pytest-asyncio
+types-jsonschema
diff --git a/tests/__init__.py b/tests/__init__.py
diff --git a/tests/entrypoints/__init__.py b/tests/entrypoints/__init__.py
diff --git a/tests/entrypoints/conftest.py b/tests/entrypoints/conftest.py
@@ -0,0 +1,50 @@
+# Inspired from https://github.com/vllm-project/vllm/blob/main/tests/entrypoints/llm/test_guided_generate.py
+import pytest
+
+
+@pytest.fixture
+def sample_regex():
+    return (r"((25[0-5]|(2[0-4]|1\d|[1-9]|)\d)\.){3}"
+            r"(25[0-5]|(2[0-4]|1\d|[1-9]|)\d)")
+
+
+@pytest.fixture
+def sample_json_schema():
+    return {
+        "type": "object",
+        "properties": {
+            "name": {
+                "type": "string"
+            },
+            "age": {
+                "type": "integer"
+            },
+            "skills": {
+                "type": "array",
+                "items": {
+                    "type": "string",
+                    "maxLength": 10
+                },
+                "minItems": 3
+            },
+            "work_history": {
+                "type": "array",
+                "items": {
+                    "type": "object",
+                    "properties": {
+                        "company": {
+                            "type": "string"
+                        },
+                        "duration": {
+                            "type": "number"
+                        },
+                        "position": {
+                            "type": "string"
+                        }
+                    },
+                    "required": ["company", "position"]
+                }
+            }
+        },
+        "required": ["name", "age", "skills", "work_history"]
+    }
diff --git a/tests/entrypoints/test_guided_decoding.py b/tests/entrypoints/test_guided_decoding.py
@@ -0,0 +1,91 @@
+# Inspired from https://github.com/vllm-project/vllm/blob/main/tests/entrypoints/llm/test_guided_generate.py
+import gc
+import json
+import os
+import re
+import weakref
+
+import jsonschema
+import pytest
+import torch
+from vllm.entrypoints.llm import LLM
+from vllm.outputs import RequestOutput
+from vllm.sampling_params import GuidedDecodingParams, SamplingParams
+
+os.environ["PYTORCH_NPU_ALLOC_CONF"] = "max_split_size_mb:256"
+MODEL_NAME = "Qwen/Qwen2.5-1.5B-Instruct"
+GUIDED_DECODING_BACKENDS = [
+    "outlines",
+    "lm-format-enforcer",
+    "xgrammar",
+]
+
+
+def clean_up():
+    gc.collect()
+    torch.npu.empty_cache()
+
+
+@pytest.fixture(scope="module")
+def llm():
+    # pytest caches the fixture so we use weakref.proxy to
+    # enable garbage collection
+    llm = LLM(model=MODEL_NAME, max_model_len=1024, seed=0)
+    with llm.deprecate_legacy_api():
+        yield weakref.proxy(llm)
+        del llm
+    clean_up()
+
+
+@pytest.mark.parametrize("guided_decoding_backend", GUIDED_DECODING_BACKENDS)
+def test_guided_regex(sample_regex, llm, guided_decoding_backend: str):
+    sampling_params = SamplingParams(temperature=0.8,
+                                     top_p=0.95,
+                                     guided_decoding=GuidedDecodingParams(
+                                         regex=sample_regex,
+                                         backend=guided_decoding_backend))
+    outputs = llm.generate(prompts=[
+        f"Give an example IPv4 address with this regex: {sample_regex}"
+    ] * 2,
+                           sampling_params=sampling_params,
+                           use_tqdm=True)
+
+    assert outputs is not None
+    for output in outputs:
+        assert output is not None
+        assert isinstance(output, RequestOutput)
+        prompt = output.prompt
+        generated_text = output.outputs[0].text
+        print(generated_text)
+        assert generated_text is not None
+        assert re.fullmatch(sample_regex, generated_text) is not None
+        print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
+
+
+@pytest.mark.parametrize("guided_decoding_backend", GUIDED_DECODING_BACKENDS)
+def test_guided_json_completion(sample_json_schema, llm,
+                                guided_decoding_backend: str):
+    sampling_params = SamplingParams(temperature=1.0,
+                                     max_tokens=1000,
+                                     guided_decoding=GuidedDecodingParams(
+                                         json=sample_json_schema,
+                                         backend=guided_decoding_backend))
+    outputs = llm.generate(prompts=[
+        f"Give an example JSON for an employee profile "
+        f"that fits this schema: {sample_json_schema}"
+    ] * 2,
+                           sampling_params=sampling_params,
+                           use_tqdm=True)
+
+    assert outputs is not None
+
+    for output in outputs:
+        assert output is not None
+        assert isinstance(output, RequestOutput)
+        prompt = output.prompt
+
+        generated_text = output.outputs[0].text
+        assert generated_text is not None
+        print(f"Prompt: {prompt!r}, Generated text: {generated_text!r}")
+        output_json = json.loads(generated_text)
+        jsonschema.validate(instance=output_json, schema=sample_json_schema)
diff --git a/tests/test_offline_inference.py b/tests/test_offline_inference.py
@@ -24,9 +24,9 @@
 
 import pytest
 import vllm  # noqa: F401
-from conftest import VllmRunner
 
 import vllm_ascend  # noqa: F401
+from tests.conftest import VllmRunner
 
 MODELS = [
     "Qwen/Qwen2.5-0.5B-Instruct",