Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion tests/standalone_tests/lazy_imports.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,9 @@
# too many processes before we set the number of compiler threads.
# Lazy import `cv2` to avoid bothering users who only use text models.
# `cv2` can easily mess up the environment.
module_names = ["torch._inductor.async_compile", "cv2"]
# Lazy import `xgrammar` because it has no wheels for some platforms
# (e.g. s390x) and is only needed once structured output is requested.
module_names = ["torch._inductor.async_compile", "cv2", "xgrammar"]

# set all modules in `module_names` to be None.
# if we import any modules during `import vllm`, there would be a
Expand Down
23 changes: 23 additions & 0 deletions tests/tool_parsers/test_structural_tag_registry.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,8 @@
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project

import json
import subprocess
import sys
from types import SimpleNamespace
from unittest.mock import MagicMock

Expand Down Expand Up @@ -1030,3 +1032,24 @@ def test_tool_strict_level_from_name():
ToolStrictLevel.from_name("strict")
with pytest.raises(ValueError, match="expected one of auto, function, parameter"):
ToolStrictLevel.from_name("off")


def test_import_without_xgrammar():
"""Importing vLLM must not require xgrammar (it has no wheels on some
platforms); structural tag builders fail only when actually used."""
code = """
import sys

sys.modules["xgrammar"] = None

from vllm import LLM # noqa: F401
from vllm.tool_parsers.structural_tag_registry import get_hermes_structural_tag

try:
get_hermes_structural_tag([], [], "auto", False)
except ImportError as error:
assert "xgrammar" in str(error), error
else:
raise AssertionError("expected ImportError when xgrammar is unavailable")
"""
subprocess.run([sys.executable, "-c", code], check=True)
64 changes: 43 additions & 21 deletions vllm/parser/harmony.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,21 +10,6 @@
from typing import TYPE_CHECKING, NamedTuple

from openai_harmony import HarmonyError, Message, Role
from xgrammar import StructuralTag
from xgrammar.openai_tool_call_schema import BuiltinToolParam, FunctionToolParam
from xgrammar.structural_tag import (
AnyTextFormat,
ConstStringFormat,
Format,
GrammarFormat,
JSONSchemaFormat,
OptionalFormat,
OrFormat,
RegexFormat,
SequenceFormat,
TagFormat,
TriggeredTagsFormat,
)

from vllm.entrypoints.chat_utils import make_tool_call_id
from vllm.entrypoints.generate.base.protocol import (
Expand Down Expand Up @@ -52,10 +37,49 @@
get_function_parameters,
register_vllm_structural_tag,
)
from vllm.utils.import_utils import PlaceholderModule

if TYPE_CHECKING:
from openai_harmony import Message, StreamableParser

try:
from xgrammar import StructuralTag
from xgrammar.openai_tool_call_schema import BuiltinToolParam, FunctionToolParam
from xgrammar.structural_tag import (
AnyTextFormat,
ConstStringFormat,
Format,
GrammarFormat,
JSONSchemaFormat,
OptionalFormat,
OrFormat,
RegexFormat,
SequenceFormat,
TagFormat,
TriggeredTagsFormat,
)
except ImportError:
# xgrammar has no wheels for some platforms (e.g. s390x). Only the
# structural tag builders below need it, so fail on first use instead
# of at import time.
_xgrammar = PlaceholderModule("xgrammar")
_xgr_tool_schema = _xgrammar.placeholder_attr("openai_tool_call_schema")
_xgr_structural_tag = _xgrammar.placeholder_attr("structural_tag")
StructuralTag = _xgrammar.placeholder_attr("StructuralTag")
BuiltinToolParam = _xgr_tool_schema.placeholder_attr("BuiltinToolParam")
FunctionToolParam = _xgr_tool_schema.placeholder_attr("FunctionToolParam")
AnyTextFormat = _xgr_structural_tag.placeholder_attr("AnyTextFormat")
ConstStringFormat = _xgr_structural_tag.placeholder_attr("ConstStringFormat")
Format = _xgr_structural_tag.placeholder_attr("Format")
GrammarFormat = _xgr_structural_tag.placeholder_attr("GrammarFormat")
JSONSchemaFormat = _xgr_structural_tag.placeholder_attr("JSONSchemaFormat")
OptionalFormat = _xgr_structural_tag.placeholder_attr("OptionalFormat")
OrFormat = _xgr_structural_tag.placeholder_attr("OrFormat")
RegexFormat = _xgr_structural_tag.placeholder_attr("RegexFormat")
SequenceFormat = _xgr_structural_tag.placeholder_attr("SequenceFormat")
TagFormat = _xgr_structural_tag.placeholder_attr("TagFormat")
TriggeredTagsFormat = _xgr_structural_tag.placeholder_attr("TriggeredTagsFormat")


logger = init_logger(__name__)

Expand Down Expand Up @@ -402,8 +426,6 @@ def _normalize_recipient(recipient: str | None) -> str | None:
" to=functions.{name}{channel}{constrain}<|message|>",
"{channel} to=functions.{name}{constrain}<|message|>",
]
_JSON_CONTENT = JSONSchemaFormat(json_schema={"type": "object"})
_ANY_CONTENT = AnyTextFormat()


def _assemble_tag(
Expand All @@ -416,7 +438,7 @@ def _assemble_tag(
elements=[
TagFormat(
begin="<|channel|>analysis<|message|>",
content=_ANY_CONTENT,
content=AnyTextFormat(),
end="<|end|>",
),
ConstStringFormat(value="<|start|>assistant"),
Expand All @@ -431,7 +453,7 @@ def _assemble_tag(
elements=[
TagFormat(
begin="<|channel|>commentary<|message|>",
content=_ANY_CONTENT,
content=AnyTextFormat(),
end="<|end|>",
),
ConstStringFormat(value="<|start|>assistant"),
Expand Down Expand Up @@ -494,7 +516,7 @@ def get_harmony_structural_tag(
tags.extend(
TagFormat(
begin=_FINAL_BEGIN.format(constrain=constrain),
content=_ANY_CONTENT,
content=AnyTextFormat(),
end=_END_TAG,
)
for constrain in _JSON_CONSTRAINS + [""]
Expand All @@ -508,7 +530,7 @@ def get_harmony_structural_tag(
def _params_to_final_content(params: StructuredOutputsParams) -> Format | None:
"""Map StructuredOutputsParams in a XGrammar Format."""
if params.json_object:
return _JSON_CONTENT
return JSONSchemaFormat(json_schema={"type": "object"})
if params.json is not None:
schema = params.json
if isinstance(schema, str):
Expand Down
88 changes: 58 additions & 30 deletions vllm/tool_parsers/structural_tag_registry.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project

from __future__ import annotations

from collections.abc import Callable, Sequence
from typing import Any, Literal, TypeAlias

Expand All @@ -9,32 +11,63 @@
from openai.types.responses.tool import Tool as ResponsesTool
from openai.types.responses.tool_choice_allowed import ToolChoiceAllowed
from openai.types.responses.tool_choice_function import ToolChoiceFunction
from xgrammar import StructuralTag, normalize_tool_choice
from xgrammar import get_model_structural_tag as get_xgrammar_model_structural_tag
from xgrammar.openai_tool_call_schema import (
BuiltinToolParam,
FunctionToolParam,
)
from xgrammar.structural_tag import (
AnyTextFormat,
ConstStringFormat,
JSONSchemaFormat,
OptionalFormat,
OrFormat,
PlusFormat,
RegexFormat,
SequenceFormat,
StarFormat,
TagFormat,
TagsWithSeparatorFormat,
TriggeredTagsFormat,
)

from vllm.entrypoints.openai.chat_completion.protocol import (
ChatCompletionNamedToolChoiceParam,
ChatCompletionToolsParam,
)
from vllm.tool_parsers.tool_strict_level import ToolStrictLevel
from vllm.utils.import_utils import PlaceholderModule

try:
from xgrammar import StructuralTag, normalize_tool_choice
from xgrammar import get_model_structural_tag as get_xgrammar_model_structural_tag
from xgrammar.openai_tool_call_schema import (
BuiltinToolParam,
FunctionToolParam,
)
from xgrammar.structural_tag import (
AnyTextFormat,
ConstStringFormat,
JSONSchemaFormat,
OptionalFormat,
OrFormat,
PlusFormat,
RegexFormat,
SequenceFormat,
StarFormat,
TagFormat,
TagsWithSeparatorFormat,
TriggeredTagsFormat,
)
except ImportError:
# xgrammar has no wheels for some platforms (e.g. s390x). Only the
# structural tag builders below need it, so fail on first use instead
# of at import time.
_xgrammar = PlaceholderModule("xgrammar")
_xgr_tool_schema = _xgrammar.placeholder_attr("openai_tool_call_schema")
_xgr_structural_tag = _xgrammar.placeholder_attr("structural_tag")
StructuralTag = _xgrammar.placeholder_attr("StructuralTag")
normalize_tool_choice = _xgrammar.placeholder_attr("normalize_tool_choice")
get_xgrammar_model_structural_tag = _xgrammar.placeholder_attr(
"get_model_structural_tag"
)
BuiltinToolParam = _xgr_tool_schema.placeholder_attr("BuiltinToolParam")
FunctionToolParam = _xgr_tool_schema.placeholder_attr("FunctionToolParam")
AnyTextFormat = _xgr_structural_tag.placeholder_attr("AnyTextFormat")
ConstStringFormat = _xgr_structural_tag.placeholder_attr("ConstStringFormat")
JSONSchemaFormat = _xgr_structural_tag.placeholder_attr("JSONSchemaFormat")
OptionalFormat = _xgr_structural_tag.placeholder_attr("OptionalFormat")
OrFormat = _xgr_structural_tag.placeholder_attr("OrFormat")
PlusFormat = _xgr_structural_tag.placeholder_attr("PlusFormat")
RegexFormat = _xgr_structural_tag.placeholder_attr("RegexFormat")
SequenceFormat = _xgr_structural_tag.placeholder_attr("SequenceFormat")
StarFormat = _xgr_structural_tag.placeholder_attr("StarFormat")
TagFormat = _xgr_structural_tag.placeholder_attr("TagFormat")
TagsWithSeparatorFormat = _xgr_structural_tag.placeholder_attr(
"TagsWithSeparatorFormat"
)
TriggeredTagsFormat = _xgr_structural_tag.placeholder_attr("TriggeredTagsFormat")

ToolChoice: TypeAlias = (
Literal["none", "auto", "required"]
Expand All @@ -44,16 +77,11 @@
)
AllowedToolRef: TypeAlias = dict[str, object]
SimplifiedToolChoice: TypeAlias = Literal["auto", "required", "forced"]
StructuralTagBuilder: TypeAlias = Callable[
[
list[FunctionToolParam],
list[BuiltinToolParam],
SimplifiedToolChoice,
bool,
str,
],
StructuralTag,
]
# Quoted so the placeholders above are never evaluated at import time.
StructuralTagBuilder: TypeAlias = (
"Callable[[list[FunctionToolParam], list[BuiltinToolParam], "
"SimplifiedToolChoice, bool, str], StructuralTag]"
)

# Keep this list in sync with xgrammar.builtin_structural_tag. It is used for
# vLLM-side validation and for documenting the xgrammar builtin surface that
Expand Down
2 changes: 2 additions & 0 deletions vllm/v1/structured_output/backend_xgrammar.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project

from __future__ import annotations

import json
from dataclasses import dataclass, field
from typing import TYPE_CHECKING, Any
Expand Down
Loading