Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
52 changes: 52 additions & 0 deletions agent/model_metadata.py
Original file line number Diff line number Diff line change
Expand Up @@ -1140,6 +1140,33 @@ def _model_name_suggests_minimax_m3(model: str) -> bool:
return "minimax-m3" in model.lower()


def _model_name_suggests_qwen3_6_plus(model: str) -> bool:
"""Return True if the model name looks like a Qwen 3.6-Plus variant.

Catches ``qwen3.6-plus``, ``qwen/qwen3.6-plus``, ``dashscope/qwen3.6-plus``,
and similar prefixed forms. Used as a guard against stale cache entries
seeded by pre-catalog builds that resolved qwen3.6-plus via the generic
``qwen`` catch-all (131,072) before the ``qwen3.6-plus`` (1M) entry was
added to DEFAULT_CONTEXT_LENGTHS on 2026-05-17.
"""
return "qwen3.6-plus" in model.lower()


def _model_name_suggests_grok_4_fast_or_4_20(model: str) -> bool:
"""Return True if the model name looks like a Grok 4-fast or 4.20 variant.

Catches ``grok-4-fast``, ``grok-4-fast-reasoning``,
``grok-4.20-0309-reasoning``, ``grok-4.20-multi-agent-0309``, and similar
slugs. Used as a guard against stale cache entries seeded before the
``grok-4-fast`` (2M) and ``grok-4.20`` (2M) entries were added to
DEFAULT_CONTEXT_LENGTHS on 2026-04-10. Prior to that, these slugs fell
through to the DEFAULT_FALLBACK_CONTEXT (256,000) or, via a custom
endpoint probe, could have been persisted at an even lower tier value.
"""
lower = model.lower()
return "grok-4-fast" in lower or "grok-4.20" in lower


def _query_local_context_length(model: str, base_url: str, api_key: str = "") -> Optional[int]:
"""Query a local server for the model's context length."""
import httpx
Expand Down Expand Up @@ -1564,6 +1591,31 @@ def get_model_context_length(
model, base_url, f"{cached:,}",
)
_invalidate_cached_context_length(model, base_url)
# Invalidate stale ≤131,072 cache entries for qwen3.6-plus. The
# ``qwen3.6-plus`` (1M) entry was added to DEFAULT_CONTEXT_LENGTHS on
# 2026-05-17; prior to that, qwen3.6-plus slugs resolved via the
# generic ``qwen`` catch-all (131,072) and that value was persisted.
# qwen3.6-plus is 1M, so any sub-262K cached value is a pre-catalog
# leftover — drop it and fall through to the hardcoded default.
elif cached <= 131_072 and _model_name_suggests_qwen3_6_plus(model):
logger.info(
"Dropping stale qwen3.6-plus cache entry %s@%s -> %s (pre-catalog value); "
"re-resolving via hardcoded defaults",
model, base_url, f"{cached:,}",
)
_invalidate_cached_context_length(model, base_url)
# Invalidate stale ≤256,000 cache entries for grok-4-fast and grok-4.20.
# These 2M-context models were added to DEFAULT_CONTEXT_LENGTHS on
# 2026-04-10; prior to that, they fell through to the DEFAULT_FALLBACK_CONTEXT
# (256,000) or lower probe tiers, and that value could have been persisted.
# Any sub-512K cached value for these slugs is a pre-catalog leftover.
elif cached <= 256_000 and _model_name_suggests_grok_4_fast_or_4_20(model):
logger.info(
"Dropping stale grok-4-fast/grok-4.20 cache entry %s@%s -> %s (pre-catalog value); "
"re-resolving via hardcoded defaults",
model, base_url, f"{cached:,}",
)
_invalidate_cached_context_length(model, base_url)
# Nous Portal: the portal /v1/models endpoint is authoritative.
# Bypass the persistent cache so step 5b can always reconcile
# against it — this corrects pre-fix entries seeded from the
Expand Down
162 changes: 162 additions & 0 deletions tests/agent/test_model_metadata.py
Original file line number Diff line number Diff line change
Expand Up @@ -1247,3 +1247,165 @@ def test_special_chars_in_model_name(self, tmp_path):
with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file):
save_context_length(model, url, 200000)
assert get_cached_context_length(model, url) == 200000


class TestQwen36PlusStaleCacheGuard:
"""Pre-catalog builds resolved qwen3.6-plus via the generic 'qwen' catch-all
(131,072) and persisted it before the 'qwen3.6-plus' (1M) entry was added on
2026-05-17. The step-1 cache guard must drop that stale value so the correct
1M entry from DEFAULT_CONTEXT_LENGTHS is used instead."""

def test_helper_catches_bare_slug(self):
from agent.model_metadata import _model_name_suggests_qwen3_6_plus
assert _model_name_suggests_qwen3_6_plus("qwen3.6-plus") is True
assert _model_name_suggests_qwen3_6_plus("qwen/qwen3.6-plus") is True
assert _model_name_suggests_qwen3_6_plus("dashscope/qwen3.6-plus") is True

def test_helper_rejects_other_qwen(self):
from agent.model_metadata import _model_name_suggests_qwen3_6_plus
assert _model_name_suggests_qwen3_6_plus("qwen3-coder-plus") is False
assert _model_name_suggests_qwen3_6_plus("qwen") is False
assert _model_name_suggests_qwen3_6_plus("qwen3.5-72b") is False

@patch("agent.models_dev.lookup_models_dev_context", return_value=None)
@patch("agent.model_metadata.fetch_model_metadata")
def test_stale_131k_cache_dropped_and_re_resolved(self, mock_fetch, _mock_mdev, tmp_path):
"""A cached value of 131,072 (pre-catalog qwen catch-all) for qwen3.6-plus
must be dropped so the lookup falls through to the 1M hardcoded default
in DEFAULT_CONTEXT_LENGTHS."""
from agent.model_metadata import (
get_model_context_length,
save_context_length,
)
mock_fetch.return_value = {}
cache_file = tmp_path / "cache.yaml"
url = "https://dashscope.aliyuncs.com/compatible-mode/v1"
with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file):
save_context_length("qwen3.6-plus", url, 131_072)
result = get_model_context_length("qwen3.6-plus", base_url=url)
assert result == 1_048_576

@patch("agent.models_dev.lookup_models_dev_context", return_value=None)
@patch("agent.model_metadata.fetch_model_metadata")
def test_correct_1m_cache_preserved(self, mock_fetch, _mock_mdev, tmp_path):
"""A cached value of 1,048,576 (already correct) must not be evicted."""
from agent.model_metadata import (
get_model_context_length,
save_context_length,
)
mock_fetch.return_value = {}
cache_file = tmp_path / "cache.yaml"
url = "https://dashscope.aliyuncs.com/compatible-mode/v1"
with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file):
save_context_length("qwen3.6-plus", url, 1_048_576)
result = get_model_context_length("qwen3.6-plus", base_url=url)
assert result == 1_048_576

@patch("agent.models_dev.lookup_models_dev_context", return_value=None)
@patch("agent.model_metadata.fetch_model_metadata")
def test_plain_qwen_256k_cache_not_clobbered(self, mock_fetch, _mock_mdev, tmp_path):
"""A generic 'qwen' slug with a cached 131,072 must not be evicted by
the qwen3.6-plus guard (no false-positive widening of the guard)."""
from agent.model_metadata import (
get_model_context_length,
save_context_length,
)
mock_fetch.return_value = {}
cache_file = tmp_path / "cache.yaml"
url = "https://dashscope.aliyuncs.com/compatible-mode/v1"
with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file):
save_context_length("qwen", url, 131_072)
result = get_model_context_length("qwen", base_url=url)
assert result == 131_072


class TestGrokFastAndGrok420StaleCacheGuard:
"""grok-4-fast (2M) and grok-4.20 (2M) entries were added to
DEFAULT_CONTEXT_LENGTHS on 2026-04-10. Before that, these slugs had no
explicit entry and could be cached at DEFAULT_FALLBACK_CONTEXT (256,000) or
lower via a context-overflow probe. The step-1 guard must drop those stale
values so the 2M default is used."""

def test_helper_catches_grok_4_fast_variants(self):
from agent.model_metadata import _model_name_suggests_grok_4_fast_or_4_20
assert _model_name_suggests_grok_4_fast_or_4_20("grok-4-fast") is True
assert _model_name_suggests_grok_4_fast_or_4_20("grok-4-fast-reasoning") is True
assert _model_name_suggests_grok_4_fast_or_4_20("grok-4-fast-non-reasoning") is True

def test_helper_catches_grok_4_20_variants(self):
from agent.model_metadata import _model_name_suggests_grok_4_fast_or_4_20
assert _model_name_suggests_grok_4_fast_or_4_20("grok-4.20-0309-reasoning") is True
assert _model_name_suggests_grok_4_fast_or_4_20("grok-4.20-multi-agent-0309") is True
assert _model_name_suggests_grok_4_fast_or_4_20("grok-4.20") is True

def test_helper_rejects_plain_grok_4(self):
from agent.model_metadata import _model_name_suggests_grok_4_fast_or_4_20
assert _model_name_suggests_grok_4_fast_or_4_20("grok-4") is False
assert _model_name_suggests_grok_4_fast_or_4_20("grok-4-0709") is False
assert _model_name_suggests_grok_4_fast_or_4_20("grok-4.3") is False
assert _model_name_suggests_grok_4_fast_or_4_20("grok-3") is False

@patch("agent.models_dev.lookup_models_dev_context", return_value=None)
@patch("agent.model_metadata.fetch_model_metadata")
def test_stale_256k_cache_dropped_grok_4_fast(self, mock_fetch, _mock_mdev, tmp_path):
"""A cached 256,000 (DEFAULT_FALLBACK_CONTEXT, pre-catalog) for grok-4-fast
must be dropped so the 2M hardcoded default is used."""
from agent.model_metadata import (
get_model_context_length,
save_context_length,
)
mock_fetch.return_value = {}
cache_file = tmp_path / "cache.yaml"
url = "https://api.x.ai/v1"
with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file):
save_context_length("grok-4-fast", url, 256_000)
result = get_model_context_length("grok-4-fast", base_url=url)
assert result == 2_000_000

@patch("agent.models_dev.lookup_models_dev_context", return_value=None)
@patch("agent.model_metadata.fetch_model_metadata")
def test_stale_256k_cache_dropped_grok_4_20(self, mock_fetch, _mock_mdev, tmp_path):
"""A cached 256,000 for grok-4.20-0309-reasoning must be dropped."""
from agent.model_metadata import (
get_model_context_length,
save_context_length,
)
mock_fetch.return_value = {}
cache_file = tmp_path / "cache.yaml"
url = "https://api.x.ai/v1"
with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file):
save_context_length("grok-4.20-0309-reasoning", url, 256_000)
result = get_model_context_length("grok-4.20-0309-reasoning", base_url=url)
assert result == 2_000_000

@patch("agent.models_dev.lookup_models_dev_context", return_value=None)
@patch("agent.model_metadata.fetch_model_metadata")
def test_correct_2m_cache_preserved(self, mock_fetch, _mock_mdev, tmp_path):
"""A cached 2,000,000 (already correct) must not be evicted."""
from agent.model_metadata import (
get_model_context_length,
save_context_length,
)
mock_fetch.return_value = {}
cache_file = tmp_path / "cache.yaml"
url = "https://api.x.ai/v1"
with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file):
save_context_length("grok-4-fast", url, 2_000_000)
result = get_model_context_length("grok-4-fast", base_url=url)
assert result == 2_000_000

@patch("agent.models_dev.lookup_models_dev_context", return_value=None)
@patch("agent.model_metadata.fetch_model_metadata")
def test_plain_grok_4_256k_cache_not_clobbered(self, mock_fetch, _mock_mdev, tmp_path):
"""A plain 'grok-4' (256K is correct) cached value must not be evicted."""
from agent.model_metadata import (
get_model_context_length,
save_context_length,
)
mock_fetch.return_value = {}
cache_file = tmp_path / "cache.yaml"
url = "https://api.x.ai/v1"
with patch("agent.model_metadata._get_context_cache_path", return_value=cache_file):
save_context_length("grok-4", url, 256_000)
result = get_model_context_length("grok-4", base_url=url)
assert result == 256_000
Loading