Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -82,6 +82,7 @@ jobs:
tests/test_request.py \
tests/test_anthropic_models.py \
tests/test_anthropic_adapter.py \
tests/test_anthropic_adapter_canonicalization.py \
tests/test_dependency_floors.py \
tests/test_ssd_cache_shutdown.py \
tests/test_chat_template_safety.py \
Expand Down
112 changes: 112 additions & 0 deletions tests/test_anthropic_adapter_canonicalization.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,112 @@
# SPDX-License-Identifier: Apache-2.0
"""The Anthropic adapter must canonicalize system prompts the shared way.

`anthropic_to_openai()` used to carry its own unanchored `re.sub` for the
billing header. It runs *before* `canonicalize_system_messages()`, so anything
it removed was gone by the time the shared, line-anchored stripper saw the
messages — including the middle of a user's own sentence. These tests pin both
stages: what the adapter alone does, and what the prepared-message path
produces end to end.
"""

import pytest

from vllm_mlx.api.anthropic_adapter import anthropic_to_openai
from vllm_mlx.api.anthropic_models import AnthropicRequest
from vllm_mlx.api.prompt_canonicalize import canonicalize_system_messages

HEADER = "x-anthropic-billing-header: cch=9f2a1b;"
BODY = "You are helpful.\nBe concise."


def _system_text(system) -> str:
request = AnthropicRequest(
model="test-model",
max_tokens=16,
system=system,
messages=[{"role": "user", "content": "hi"}],
)
converted = anthropic_to_openai(request)
systems = [m for m in converted.messages if m.role == "system"]
assert len(systems) == 1, f"expected one system message, got {len(systems)}"
return systems[0].content


def _two_stage(system) -> str:
"""Adapter output after the shared pass the server applies next."""
text = _system_text(system)
canonicalized = canonicalize_system_messages([{"role": "system", "content": text}])
return canonicalized[0]["content"]


class TestStandaloneHeaderIsRemoved:
@pytest.mark.parametrize(
"header",
[
"x-anthropic-billing-header: cch=9f2a1b;",
"X-Anthropic-Billing-Header: cch=9f2a1b;",
"X-ANTHROPIC-BILLING-HEADER: cch=9f2a1b;",
],
ids=["lower", "mixed", "upper"],
)
def test_case_insensitive_on_its_own_line(self, header):
text = _system_text(f"You are helpful.\n{header}\nBe concise.")
assert "cch=" not in text, f"volatile hash survived: {text!r}"
assert "You are helpful." in text
assert "Be concise." in text

def test_header_as_the_entire_prompt(self):
assert _system_text(HEADER).strip() == ""

def test_trailing_header_without_newline(self):
text = _system_text(f"{BODY}\n{HEADER}")
assert "cch=" not in text
assert "Be concise." in text


class TestUserTextIsPreserved:
"""The regression this replaces: an unanchored strip ate real prompt text."""

def test_mid_line_mention_survives(self):
prompt = "Explain what x-anthropic-billing-header: means in HTTP terms."
assert _system_text(prompt) == prompt

def test_mid_line_mention_survives_the_second_pass_too(self):
prompt = "Explain what x-anthropic-billing-header: means in HTTP terms."
assert _two_stage(prompt) == prompt

@pytest.mark.parametrize(
"prompt",
[
BODY,
"Timestamps look like 2026-05-10T13:42:18.123Z in our logs.",
"Use the header: Authorization: Bearer <token> when calling the API.",
"Discuss anthropic billing headers in general.",
],
ids=["plain", "timestamp", "other-header", "prose"],
)
def test_unrelated_content_is_untouched(self, prompt):
assert _system_text(prompt) == prompt
assert _two_stage(prompt) == prompt


class TestBothStagesAgree:
"""Adapter and shared pass must not disagree about what to remove."""

@pytest.mark.parametrize(
"prompt",
[
f"You are helpful.\n{HEADER}\nBe concise.",
"You are helpful.\nX-Anthropic-Billing-Header: cch=abc;\nBe concise.",
"Explain what x-anthropic-billing-header: means in HTTP terms.",
BODY,
],
ids=["lower", "mixed", "mid-line", "clean"],
)
def test_second_pass_is_a_no_op(self, prompt):
"""Whatever the adapter returns is already canonical."""
once = _system_text(prompt)
twice = canonicalize_system_messages([{"role": "system", "content": once}])[0][
"content"
]
assert twice == once, f"shared pass still changed {once!r} -> {twice!r}"
10 changes: 8 additions & 2 deletions vllm_mlx/api/anthropic_adapter.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,6 @@
"""

import json
import re
import uuid

from .anthropic_models import (
Expand All @@ -26,6 +25,7 @@
Message,
ToolDefinition,
)
from .prompt_canonicalize import canonicalize_system_prompt


def anthropic_to_openai(request: AnthropicRequest) -> ChatCompletionRequest:
Expand Down Expand Up @@ -64,7 +64,13 @@ def anthropic_to_openai(request: AnthropicRequest) -> ChatCompletionRequest:
# Strip per-request billing/tracking headers injected by some
# clients (e.g. Claude Code). These contain a per-request hash
# that prevents prefix-cache reuse across turn boundaries.
system_text = re.sub(r"x-anthropic-billing-header:[^\n]*\n?", "", system_text)
#
# Use the shared canonicalizer rather than an inline pattern. The
# local one was unanchored, so it deleted the header text wherever
# it appeared -- including mid-sentence in a user's own prompt --
# and this pass runs before canonicalize_system_messages(), which
# cannot restore what was already removed.
system_text = canonicalize_system_prompt(system_text) or ""
messages.append(Message(role="system", content=system_text))

# Convert each message
Expand Down
Loading