From 6b658ef11f42ff3812220c23bf864dc3b8c73a66 Mon Sep 17 00:00:00 2001 From: "ang-fleet-workers[bot]" <333956806+ang-fleet-workers[bot]@users.noreply.github.com> Date: Wed, 30 Sep 2026 11:17:03 -0700 Subject: [PATCH] feat(codex): gpt-6.1-sol joins the -900k opt-in (measured 922,000 ceiling) openai-codex/gpt-6.1-sol-900k now strips to gpt-6.1-sol and resolves to 900,000. Ceiling measured 2026-09-30 by bisection on the Codex responses endpoint: 921,028 ok / 921,998 context_length_exceeded. Bare slug keeps 272K; EXACT entry, no prefix leak. Picker curates + forward-compats the slug; -900k pricing alias added. Verified: tests/agent/test_gpt61_sol_900k.py 5 fail on fork/main, pass after; model_metadata/usage_pricing/codex_models/picker files green. --- agent/model_metadata.py | 12 +++- agent/usage_pricing.py | 2 +- hermes_cli/codex_models.py | 5 ++ tests/agent/test_gpt61_sol_900k.py | 99 ++++++++++++++++++++++++++ tests/agent/test_gpt61_sol_prestage.py | 7 +- 5 files changed, 119 insertions(+), 6 deletions(-) create mode 100644 tests/agent/test_gpt61_sol_900k.py diff --git a/agent/model_metadata.py b/agent/model_metadata.py index fe1fde217e9bd..2edf65837aa95 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -2918,7 +2918,15 @@ def _query_anthropic_context_length(model: str, base_url: str, api_key: str) -> # unprobed future gpt-6 descendant can never inherit this cap. "gpt-6-sol": 872_000, "gpt-6-luna": 872_000, - # gpt-6.1-sol: add once the codex lane serves it and max_context_window is measured (2026-09-29) + # GPT-6.1 Sol: the codex-sub catalog did NOT list the slug at measurement + # time, so there is no catalog max_context_window. Measured instead by + # request bisection against chatgpt.com/backend-api/codex/responses on + # 2026-09-30: 921,028 input tokens accepted, 921,998 rejected with + # context_length_exceeded -> 922,000 hard input ceiling (models.dev: + # 1,050,000 = 922,000 input + 128,000 output). 900K keeps the >=11K + # margin rule used for gpt-5.6. EXACT: no prefix, so -pro/-fast/other + # 6.1 descendants never inherit it. + "gpt-6.1-sol": 900_000, } # The advertised value the verified-above table is allowed to override. @@ -2938,7 +2946,7 @@ def _query_anthropic_context_length(model: str, base_url: str, api_key: str) -> "gpt-6-astra", # measured max_context_window 872,000 "gpt-6-sol", # measured max_context_window 872,000 "gpt-6-luna", # measured max_context_window 872,000 - # gpt-6.1-sol: add once the codex lane serves it and max_context_window is measured (2026-09-29) + "gpt-6.1-sol", # measured ceiling 922,000 (2026-09-30 bisection) "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna", diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index d8d510309dc78..0a4584b416169 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -1833,7 +1833,7 @@ class CostResult: # GPT-6 Sol/Luna have no "-pro" variant (the 2026-09-22 launch shipped the # base slugs only), so alias ONLY the Hermes-side "-900k" Codex picker # variant — the suffix is stripped on the wire, so it is the same model. -for _base_6 in ("gpt-6-sol", "gpt-6-luna"): +for _base_6 in ("gpt-6-sol", "gpt-6.1-sol", "gpt-6-luna"): _OFFICIAL_DOCS_PRICING[("openai", f"{_base_6}-900k")] = _OFFICIAL_DOCS_PRICING[ ("openai", _base_6) ] diff --git a/hermes_cli/codex_models.py b/hermes_cli/codex_models.py index b1af2ad0193b0..1422fbe0eb6a8 100644 --- a/hermes_cli/codex_models.py +++ b/hermes_cli/codex_models.py @@ -20,6 +20,10 @@ # they shipped (catalog visibility=list for both). Sol is the mid-tier # coding/agentic slug, Luna the cheap high-volume one. "gpt-6-sol", + # GPT-6.1 Sol (2026-09-29). Not yet listed by the account catalog + # (2026-09-30), but the Codex responses endpoint serves it (measured + # 921,028-token request OK), so it is curated + forward-compat'd here. + "gpt-6.1-sol", "gpt-6-luna", # GPT-5.6 series (Sol/Terra/Luna). The public API exposes "-pro" # variants, but the ChatGPT Codex OAuth backend rejects them with HTTP 400, @@ -61,6 +65,7 @@ _FORWARD_COMPAT_TEMPLATE_MODELS: List[tuple[str, tuple[str, ...]]] = [ ("gpt-6-astra", ("gpt-5.6-sol", "gpt-5.5", "gpt-5.4")), ("gpt-6-sol", ("gpt-5.6-sol", "gpt-5.5", "gpt-5.4")), + ("gpt-6.1-sol", ("gpt-6-sol", "gpt-5.6-sol")), ("gpt-6-luna", ("gpt-5.6-luna", "gpt-5.6-sol", "gpt-5.5", "gpt-5.4")), ("gpt-5.6-sol", ("gpt-5.5", "gpt-5.4")), ("gpt-5.6-terra", ("gpt-5.5", "gpt-5.4")), diff --git a/tests/agent/test_gpt61_sol_900k.py b/tests/agent/test_gpt61_sol_900k.py new file mode 100644 index 0000000000000..35e6ff705e079 --- /dev/null +++ b/tests/agent/test_gpt61_sol_900k.py @@ -0,0 +1,99 @@ +"""GPT-6.1 Sol joins the Codex ``-900k`` opt-in (2026-09-30). + +Measured 2026-09-30 by request bisection against +chatgpt.com/backend-api/codex/responses: 921,028 input tokens accepted, +921,998 rejected with ``context_length_exceeded`` -> 922,000 hard input +ceiling. The variant grants 900,000 (>=11K margin, same rule as gpt-5.6). + +Behaviour contracts: +- ``openai-codex/gpt-6.1-sol-900k`` strips to the wire slug ``gpt-6.1-sol`` + and resolves to a 900,000 window; +- bare ``gpt-6.1-sol`` keeps the advertised 272,000 (opt-in policy); +- no prefix leak: ``gpt-6.1-sol-pro`` / ``-fast`` are not eligible; +- picker surfaces ``gpt-6.1-sol-900k`` right after its base; +- the ``-900k`` alias prices exactly like the base. +""" + +from unittest.mock import MagicMock, patch + +import pytest + +from agent.model_metadata import ( + _verified_codex_ctx_for_slug, + get_model_context_length, + is_codex_900k_base, + is_codex_context_variant, + strip_codex_context_variant_suffix, +) +from agent.usage_pricing import get_pricing_entry + + +def _codex_ctx(model: str) -> int: + fake_response = MagicMock() + fake_response.status_code = 401 + fake_response.json.return_value = {} + import agent.model_metadata as mm + + mm._codex_oauth_context_cache = {} + with patch("agent.model_metadata.requests.get", return_value=fake_response), \ + patch("agent.model_metadata.get_cached_context_length", return_value=None), \ + patch("agent.model_metadata.save_context_length"): + return get_model_context_length( + model=model, + base_url="https://chatgpt.com/backend-api/codex", + api_key="expired-token", + provider="openai-codex", + ) + + +def test_gpt61_sol_900k_strips_to_wire_slug(): + assert is_codex_900k_base("gpt-6.1-sol") is True + assert is_codex_context_variant("openai-codex/gpt-6.1-sol-900k") is True + assert strip_codex_context_variant_suffix("gpt-6.1-sol-900k") == "gpt-6.1-sol" + assert ( + strip_codex_context_variant_suffix("openai-codex/gpt-6.1-sol-900k") + == "openai-codex/gpt-6.1-sol" + ) + + +def test_gpt61_sol_900k_resolves_to_900k_window(): + assert _verified_codex_ctx_for_slug("openai-codex/gpt-6.1-sol-900k") == 900_000 + assert _codex_ctx("gpt-6.1-sol-900k") == 900_000 + # Measured hard input ceiling is 922,000; the grant stays under it. + assert _verified_codex_ctx_for_slug("gpt-6.1-sol-900k") < 922_000 + + +def test_bare_gpt61_sol_keeps_advertised_272k(): + assert _verified_codex_ctx_for_slug("gpt-6.1-sol") is None + assert _codex_ctx("gpt-6.1-sol") == 272_000 + + +@pytest.mark.parametrize("slug", ["gpt-6.1-sol-pro", "gpt-6.1-sol-fast", "gpt-6.1", "gpt-6.1-luna"]) +def test_no_prefix_leak(slug): + assert is_codex_900k_base(slug) is False + assert is_codex_context_variant(f"{slug}-900k") is False + assert strip_codex_context_variant_suffix(f"{slug}-900k") == f"{slug}-900k" + + +def test_picker_surfaces_gpt61_sol_900k(): + from hermes_cli.codex_models import ( + DEFAULT_CODEX_MODELS, + _finalize_codex_models, + get_codex_model_ids, + ) + + assert "gpt-6.1-sol" in DEFAULT_CODEX_MODELS + ids = get_codex_model_ids() # offline curated path + assert ids.index("gpt-6.1-sol-900k") == ids.index("gpt-6.1-sol") + 1 + # A live catalog that only lists gpt-6-sol (true on 2026-09-30) still + # surfaces 6.1 via forward-compat. + out = _finalize_codex_models(["gpt-6-sol"]) + assert "gpt-6.1-sol" in out and "gpt-6.1-sol-900k" in out + + +@pytest.mark.parametrize("provider", ["openai", "openai-codex"]) +def test_gpt61_sol_900k_prices_like_base(provider): + base = get_pricing_entry("gpt-6.1-sol", provider=provider) + variant = get_pricing_entry("gpt-6.1-sol-900k", provider=provider) + assert base is not None and variant is not None + assert variant == base diff --git a/tests/agent/test_gpt61_sol_prestage.py b/tests/agent/test_gpt61_sol_prestage.py index b5fa4845950c6..df6d041c3c5e2 100644 --- a/tests/agent/test_gpt61_sol_prestage.py +++ b/tests/agent/test_gpt61_sol_prestage.py @@ -4,7 +4,7 @@ Behaviour contracts, not snapshots: - context resolves to the documented 1.05M window on the direct API and to the advertised 272K on the Codex OAuth offline fallback; -- the ``-900k`` opt-in is NOT granted (no measured max_context_window yet); +- the ``-900k`` opt-in is granted since 2026-09-30 (measured 922,000 ceiling); - pricing resolves to 6.1's own rates (cache read $0.10) while gpt-6-sol keeps its own ($0.20), on both the ``openai`` and ``openai-codex`` routes; - ``openai-codex/gpt-6.1-sol`` normalizes to the bare slug. @@ -47,9 +47,10 @@ def test_gpt61_sol_codex_offline_fallback_is_advertised_272k(): assert ctx == 272_000 -def test_gpt61_sol_not_900k_eligible_until_measured(): +def test_gpt61_sol_900k_eligible_once_measured(): + # Flipped 2026-09-30: ceiling measured at 922,000 (see test_gpt61_sol_900k.py). assert is_codex_900k_base("gpt-6-sol") is True - assert is_codex_900k_base("gpt-6.1-sol") is False + assert is_codex_900k_base("gpt-6.1-sol") is True def test_gpt61_sol_prices_distinct_from_gpt6_sol():