diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index d66d5d81acc15..a4b531209fe77 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1316,6 +1316,7 @@ def create(self, **kwargs) -> Any: # out of cache-key routing entirely — for those hosts, skip it here. try: from agent.transports.codex import ( + _cache_scope_from_session_id, _content_cache_key, _default_prompt_cache_retention_for_request, ) @@ -1328,7 +1329,15 @@ def create(self, **kwargs) -> Any: or base_url_host_matches(_host_src, "models.github.ai") ) if not _is_xai and not _is_github and "prompt_cache_key" not in resp_kwargs: - _cache_key = _content_cache_key(instructions, resp_kwargs.get("tools")) + # Scope by the owning turn's session so two unrelated sessions + # with the same instructions/tools (e.g. compression, MoA, + # flush_memories firing back-to-back on different sessions) + # don't bucket-share a prompt cache slot (#78941). The main + # transport (agent/transports/codex.py::build_kwargs) does the + # same; this adapter had no session handle before + # set_runtime_main() started threading one through. + _scope = _cache_scope_from_session_id(_runtime_main_value("session_id")) + _cache_key = _content_cache_key(instructions, resp_kwargs.get("tools"), _scope) if _cache_key: resp_kwargs["prompt_cache_key"] = _cache_key if "prompt_cache_retention" not in resp_kwargs: @@ -3049,6 +3058,7 @@ def set_runtime_main( api_key: Any = "", api_mode: str = "", auth_mode: str = "", + session_id: str = "", ) -> contextvars.Token: """Record the current context's live main runtime for auxiliary routing. @@ -3070,6 +3080,7 @@ def set_runtime_main( ), "api_mode": (api_mode or "").strip(), "auth_mode": (auth_mode or "").strip().lower(), + "session_id": (session_id or "").strip(), } # Publish authoritative context before updating locked compatibility # mirrors; concurrent sessions never read those mirrors at runtime. diff --git a/agent/turn_context.py b/agent/turn_context.py index def497b158983..45c18cec10032 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -393,6 +393,7 @@ def build_turn_context( api_key=getattr(agent, "api_key", "") or "", api_mode=getattr(agent, "api_mode", "") or "", auth_mode=getattr(agent, "auth_mode", "") or "", + session_id=getattr(agent, "session_id", "") or "", ) except Exception: pass diff --git a/tests/agent/test_auxiliary_client.py b/tests/agent/test_auxiliary_client.py index efbd2e2b88dcd..65ed2f54c7d50 100644 --- a/tests/agent/test_auxiliary_client.py +++ b/tests/agent/test_auxiliary_client.py @@ -3187,6 +3187,69 @@ def create(self, **kwargs): assert time.monotonic() - started < 0.14 +class TestCodexAuxiliaryAdapterCacheScope: + """Regression for issue #78941: auxiliary Codex calls (compression, + flush_memories, MoA, session_search) must not bucket-share a prompt + cache slot across unrelated sessions just because their instructions + and tools happen to match. + """ + + def _create_and_capture(self, *, session_id): + import agent.auxiliary_client as aux + + class _FakeCreateStream: + def __iter__(self): + return iter([ + SimpleNamespace( + type="response.output_item.done", + item=SimpleNamespace( + type="message", + content=[SimpleNamespace(type="output_text", text="ok")], + ), + ), + SimpleNamespace(type="response.completed", response=SimpleNamespace( + status="completed", id="r1", usage=None, + )), + ]) + + def close(self): + pass + + class FakeResponses: + def __init__(self): + self.kwargs = None + + def create(self, **kwargs): + self.kwargs = kwargs + return _FakeCreateStream() + + fake_client = SimpleNamespace(responses=FakeResponses(), base_url="") + adapter = aux._CodexCompletionsAdapter(fake_client, "gpt-5.5") + token = aux.set_runtime_main("openai", "gpt-5.5", session_id=session_id) + try: + adapter.create( + messages=[ + {"role": "system", "content": "You are a memory summarizer."}, + {"role": "user", "content": "Summarize the last turn."}, + ], + ) + finally: + aux.reset_runtime_main(token) + return fake_client.responses.kwargs["prompt_cache_key"] + + def test_different_sessions_get_different_cache_keys(self): + key_a = self._create_and_capture(session_id="session-A") + key_b = self._create_and_capture(session_id="session-B") + assert key_a != key_b + + def test_cron_refires_of_the_same_job_share_a_cache_key(self): + first = self._create_and_capture(session_id="cron_job42_20260801_090000") + second = self._create_and_capture(session_id="cron_job42_20260802_090000") + other_job = self._create_and_capture(session_id="cron_job99_20260801_090000") + assert first == second + assert first != other_job + + class TestCodexAuxiliaryToolMessageConversion: """Regression for issue #5709.