Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 12 additions & 1 deletion agent/auxiliary_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -1316,6 +1316,7 @@ def create(self, **kwargs) -> Any:
# out of cache-key routing entirely — for those hosts, skip it here.
try:
from agent.transports.codex import (
_cache_scope_from_session_id,
_content_cache_key,
_default_prompt_cache_retention_for_request,
)
Expand All @@ -1328,7 +1329,15 @@ def create(self, **kwargs) -> Any:
or base_url_host_matches(_host_src, "models.github.ai")
)
if not _is_xai and not _is_github and "prompt_cache_key" not in resp_kwargs:
_cache_key = _content_cache_key(instructions, resp_kwargs.get("tools"))
# Scope by the owning turn's session so two unrelated sessions
# with the same instructions/tools (e.g. compression, MoA,
# flush_memories firing back-to-back on different sessions)
# don't bucket-share a prompt cache slot (#78941). The main
# transport (agent/transports/codex.py::build_kwargs) does the
# same; this adapter had no session handle before
# set_runtime_main() started threading one through.
_scope = _cache_scope_from_session_id(_runtime_main_value("session_id"))
_cache_key = _content_cache_key(instructions, resp_kwargs.get("tools"), _scope)
if _cache_key:
resp_kwargs["prompt_cache_key"] = _cache_key
if "prompt_cache_retention" not in resp_kwargs:
Expand Down Expand Up @@ -3049,6 +3058,7 @@ def set_runtime_main(
api_key: Any = "",
api_mode: str = "",
auth_mode: str = "",
session_id: str = "",
) -> contextvars.Token:
"""Record the current context's live main runtime for auxiliary routing.

Expand All @@ -3070,6 +3080,7 @@ def set_runtime_main(
),
"api_mode": (api_mode or "").strip(),
"auth_mode": (auth_mode or "").strip().lower(),
"session_id": (session_id or "").strip(),
}
# Publish authoritative context before updating locked compatibility
# mirrors; concurrent sessions never read those mirrors at runtime.
Expand Down
1 change: 1 addition & 0 deletions agent/turn_context.py
Original file line number Diff line number Diff line change
Expand Up @@ -393,6 +393,7 @@ def build_turn_context(
api_key=getattr(agent, "api_key", "") or "",
api_mode=getattr(agent, "api_mode", "") or "",
auth_mode=getattr(agent, "auth_mode", "") or "",
session_id=getattr(agent, "session_id", "") or "",
)
except Exception:
pass
Expand Down
63 changes: 63 additions & 0 deletions tests/agent/test_auxiliary_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -3187,6 +3187,69 @@ def create(self, **kwargs):
assert time.monotonic() - started < 0.14


class TestCodexAuxiliaryAdapterCacheScope:
"""Regression for issue #78941: auxiliary Codex calls (compression,
flush_memories, MoA, session_search) must not bucket-share a prompt
cache slot across unrelated sessions just because their instructions
and tools happen to match.
"""

def _create_and_capture(self, *, session_id):
import agent.auxiliary_client as aux

class _FakeCreateStream:
def __iter__(self):
return iter([
SimpleNamespace(
type="response.output_item.done",
item=SimpleNamespace(
type="message",
content=[SimpleNamespace(type="output_text", text="ok")],
),
),
SimpleNamespace(type="response.completed", response=SimpleNamespace(
status="completed", id="r1", usage=None,
)),
])

def close(self):
pass

class FakeResponses:
def __init__(self):
self.kwargs = None

def create(self, **kwargs):
self.kwargs = kwargs
return _FakeCreateStream()

fake_client = SimpleNamespace(responses=FakeResponses(), base_url="")
adapter = aux._CodexCompletionsAdapter(fake_client, "gpt-5.5")
token = aux.set_runtime_main("openai", "gpt-5.5", session_id=session_id)
try:
adapter.create(
messages=[
{"role": "system", "content": "You are a memory summarizer."},
{"role": "user", "content": "Summarize the last turn."},
],
)
finally:
aux.reset_runtime_main(token)
return fake_client.responses.kwargs["prompt_cache_key"]

def test_different_sessions_get_different_cache_keys(self):
key_a = self._create_and_capture(session_id="session-A")
key_b = self._create_and_capture(session_id="session-B")
assert key_a != key_b

def test_cron_refires_of_the_same_job_share_a_cache_key(self):
first = self._create_and_capture(session_id="cron_job42_20260801_090000")
second = self._create_and_capture(session_id="cron_job42_20260802_090000")
other_job = self._create_and_capture(session_id="cron_job99_20260801_090000")
assert first == second
assert first != other_job


class TestCodexAuxiliaryToolMessageConversion:
"""Regression for issue #5709.

Expand Down
Loading