Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 10 additions & 2 deletions agent/model_metadata.py
Original file line number Diff line number Diff line change
Expand Up @@ -2918,7 +2918,15 @@ def _query_anthropic_context_length(model: str, base_url: str, api_key: str) ->
# unprobed future gpt-6 descendant can never inherit this cap.
"gpt-6-sol": 872_000,
"gpt-6-luna": 872_000,
# gpt-6.1-sol: add once the codex lane serves it and max_context_window is measured (2026-09-29)
# GPT-6.1 Sol: the codex-sub catalog did NOT list the slug at measurement
# time, so there is no catalog max_context_window. Measured instead by
# request bisection against chatgpt.com/backend-api/codex/responses on
# 2026-09-30: 921,028 input tokens accepted, 921,998 rejected with
# context_length_exceeded -> 922,000 hard input ceiling (models.dev:
# 1,050,000 = 922,000 input + 128,000 output). 900K keeps the >=11K
# margin rule used for gpt-5.6. EXACT: no prefix, so -pro/-fast/other
# 6.1 descendants never inherit it.
"gpt-6.1-sol": 900_000,
}

# The advertised value the verified-above table is allowed to override.
Expand All @@ -2938,7 +2946,7 @@ def _query_anthropic_context_length(model: str, base_url: str, api_key: str) ->
"gpt-6-astra", # measured max_context_window 872,000
"gpt-6-sol", # measured max_context_window 872,000
"gpt-6-luna", # measured max_context_window 872,000
# gpt-6.1-sol: add once the codex lane serves it and max_context_window is measured (2026-09-29)
"gpt-6.1-sol", # measured ceiling 922,000 (2026-09-30 bisection)
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-5.6-luna",
Expand Down
2 changes: 1 addition & 1 deletion agent/usage_pricing.py
Original file line number Diff line number Diff line change
Expand Up @@ -1833,7 +1833,7 @@ class CostResult:
# GPT-6 Sol/Luna have no "-pro" variant (the 2026-09-22 launch shipped the
# base slugs only), so alias ONLY the Hermes-side "-900k" Codex picker
# variant — the suffix is stripped on the wire, so it is the same model.
for _base_6 in ("gpt-6-sol", "gpt-6-luna"):
for _base_6 in ("gpt-6-sol", "gpt-6.1-sol", "gpt-6-luna"):
_OFFICIAL_DOCS_PRICING[("openai", f"{_base_6}-900k")] = _OFFICIAL_DOCS_PRICING[
("openai", _base_6)
]
Expand Down
5 changes: 5 additions & 0 deletions hermes_cli/codex_models.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,10 @@
# they shipped (catalog visibility=list for both). Sol is the mid-tier
# coding/agentic slug, Luna the cheap high-volume one.
"gpt-6-sol",
# GPT-6.1 Sol (2026-09-29). Not yet listed by the account catalog
# (2026-09-30), but the Codex responses endpoint serves it (measured
# 921,028-token request OK), so it is curated + forward-compat'd here.
"gpt-6.1-sol",
"gpt-6-luna",
# GPT-5.6 series (Sol/Terra/Luna). The public API exposes "-pro"
# variants, but the ChatGPT Codex OAuth backend rejects them with HTTP 400,
Expand Down Expand Up @@ -61,6 +65,7 @@
_FORWARD_COMPAT_TEMPLATE_MODELS: List[tuple[str, tuple[str, ...]]] = [
("gpt-6-astra", ("gpt-5.6-sol", "gpt-5.5", "gpt-5.4")),
("gpt-6-sol", ("gpt-5.6-sol", "gpt-5.5", "gpt-5.4")),
("gpt-6.1-sol", ("gpt-6-sol", "gpt-5.6-sol")),
("gpt-6-luna", ("gpt-5.6-luna", "gpt-5.6-sol", "gpt-5.5", "gpt-5.4")),
("gpt-5.6-sol", ("gpt-5.5", "gpt-5.4")),
("gpt-5.6-terra", ("gpt-5.5", "gpt-5.4")),
Expand Down
99 changes: 99 additions & 0 deletions tests/agent/test_gpt61_sol_900k.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,99 @@
"""GPT-6.1 Sol joins the Codex ``-900k`` opt-in (2026-09-30).

Measured 2026-09-30 by request bisection against
chatgpt.com/backend-api/codex/responses: 921,028 input tokens accepted,
921,998 rejected with ``context_length_exceeded`` -> 922,000 hard input
ceiling. The variant grants 900,000 (>=11K margin, same rule as gpt-5.6).

Behaviour contracts:
- ``openai-codex/gpt-6.1-sol-900k`` strips to the wire slug ``gpt-6.1-sol``
and resolves to a 900,000 window;
- bare ``gpt-6.1-sol`` keeps the advertised 272,000 (opt-in policy);
- no prefix leak: ``gpt-6.1-sol-pro`` / ``-fast`` are not eligible;
- picker surfaces ``gpt-6.1-sol-900k`` right after its base;
- the ``-900k`` alias prices exactly like the base.
"""

from unittest.mock import MagicMock, patch

import pytest

from agent.model_metadata import (
_verified_codex_ctx_for_slug,
get_model_context_length,
is_codex_900k_base,
is_codex_context_variant,
strip_codex_context_variant_suffix,
)
from agent.usage_pricing import get_pricing_entry


def _codex_ctx(model: str) -> int:
fake_response = MagicMock()
fake_response.status_code = 401
fake_response.json.return_value = {}
import agent.model_metadata as mm

mm._codex_oauth_context_cache = {}
with patch("agent.model_metadata.requests.get", return_value=fake_response), \
patch("agent.model_metadata.get_cached_context_length", return_value=None), \
patch("agent.model_metadata.save_context_length"):
return get_model_context_length(
model=model,
base_url="https://chatgpt.com/backend-api/codex",
api_key="expired-token",
provider="openai-codex",
)


def test_gpt61_sol_900k_strips_to_wire_slug():
assert is_codex_900k_base("gpt-6.1-sol") is True
assert is_codex_context_variant("openai-codex/gpt-6.1-sol-900k") is True
assert strip_codex_context_variant_suffix("gpt-6.1-sol-900k") == "gpt-6.1-sol"
assert (
strip_codex_context_variant_suffix("openai-codex/gpt-6.1-sol-900k")
== "openai-codex/gpt-6.1-sol"
)


def test_gpt61_sol_900k_resolves_to_900k_window():
assert _verified_codex_ctx_for_slug("openai-codex/gpt-6.1-sol-900k") == 900_000
assert _codex_ctx("gpt-6.1-sol-900k") == 900_000
# Measured hard input ceiling is 922,000; the grant stays under it.
assert _verified_codex_ctx_for_slug("gpt-6.1-sol-900k") < 922_000


def test_bare_gpt61_sol_keeps_advertised_272k():
assert _verified_codex_ctx_for_slug("gpt-6.1-sol") is None
assert _codex_ctx("gpt-6.1-sol") == 272_000


@pytest.mark.parametrize("slug", ["gpt-6.1-sol-pro", "gpt-6.1-sol-fast", "gpt-6.1", "gpt-6.1-luna"])
def test_no_prefix_leak(slug):
assert is_codex_900k_base(slug) is False
assert is_codex_context_variant(f"{slug}-900k") is False
assert strip_codex_context_variant_suffix(f"{slug}-900k") == f"{slug}-900k"


def test_picker_surfaces_gpt61_sol_900k():
from hermes_cli.codex_models import (
DEFAULT_CODEX_MODELS,
_finalize_codex_models,
get_codex_model_ids,
)

assert "gpt-6.1-sol" in DEFAULT_CODEX_MODELS
ids = get_codex_model_ids() # offline curated path
assert ids.index("gpt-6.1-sol-900k") == ids.index("gpt-6.1-sol") + 1
# A live catalog that only lists gpt-6-sol (true on 2026-09-30) still
# surfaces 6.1 via forward-compat.
out = _finalize_codex_models(["gpt-6-sol"])
assert "gpt-6.1-sol" in out and "gpt-6.1-sol-900k" in out


@pytest.mark.parametrize("provider", ["openai", "openai-codex"])
def test_gpt61_sol_900k_prices_like_base(provider):
base = get_pricing_entry("gpt-6.1-sol", provider=provider)
variant = get_pricing_entry("gpt-6.1-sol-900k", provider=provider)
assert base is not None and variant is not None
assert variant == base
7 changes: 4 additions & 3 deletions tests/agent/test_gpt61_sol_prestage.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@
Behaviour contracts, not snapshots:
- context resolves to the documented 1.05M window on the direct API and to
the advertised 272K on the Codex OAuth offline fallback;
- the ``-900k`` opt-in is NOT granted (no measured max_context_window yet);
- the ``-900k`` opt-in is granted since 2026-09-30 (measured 922,000 ceiling);
- pricing resolves to 6.1's own rates (cache read $0.10) while gpt-6-sol keeps
its own ($0.20), on both the ``openai`` and ``openai-codex`` routes;
- ``openai-codex/gpt-6.1-sol`` normalizes to the bare slug.
Expand Down Expand Up @@ -47,9 +47,10 @@ def test_gpt61_sol_codex_offline_fallback_is_advertised_272k():
assert ctx == 272_000


def test_gpt61_sol_not_900k_eligible_until_measured():
def test_gpt61_sol_900k_eligible_once_measured():
# Flipped 2026-09-30: ceiling measured at 922,000 (see test_gpt61_sol_900k.py).
assert is_codex_900k_base("gpt-6-sol") is True
assert is_codex_900k_base("gpt-6.1-sol") is False
assert is_codex_900k_base("gpt-6.1-sol") is True


def test_gpt61_sol_prices_distinct_from_gpt6_sol():
Expand Down
Loading