From cee244622236d97ea90012c7134d456455827553 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 15 Aug 2026 10:45:52 +0530 Subject: [PATCH] fix(cache): keep prompt_cache_key warm across compression session rotation Legacy compaction mode (compression.in_place: false) rotates the physical session_id mid-conversation. The prompt-cache scope introduced in #79161 was derived from that physical id, so every rotation moved the same conversation into a fresh cache bucket - the prompt cache went cold at every rotation boundary (#79017). Fix: resolve a rotation-stable logical scope - the compression-lineage ROOT of the current session (SessionDB.get_compression_lineage, fork-aware post-#79193) - once per turn, memoized per transcript segment, and prefer it over the physical session_id at every prompt_cache_key derivation site: - agent/prompt_cache_scope.py (new): resolve_prompt_cache_scope(agent) - lineage-root walk with per-segment memo; falls back to the physical id when no DB is attached or the walk fails, degrading to pre-fix behavior. - transports/codex.py: build_kwargs accepts cache_scope_id and prefers it for the body prompt_cache_key, the xAI x-grok-conv-id header, and the Codex x-client-request-id routing header. The Codex session_id header keeps the raw physical id (transcript identity, #57012 contract). - transports/chat_completions.py: _add_prompt_cache_key accepts cache_scope_id with the same precedence. - chat_completion_helpers.py: build_api_kwargs threads the resolved scope into all three build_kwargs call sites (codex, profile, legacy). - auxiliary_client.py: set_runtime_main carries cache_scope; the aux Responses cache-key site prefers it over the physical session_id. - turn_context.py: resolves the scope once per turn and threads it through set_runtime_main (no DB walk on the per-API-call hot path). Scope semantics preserved from #79161: /new starts a fresh scope (new lineage), /branch children, delegate subagents, and tool children stay isolated (explicit-fork exclusion in get_compression_lineage), unrelated sessions keep distinct buckets, and cron per-fire timestamps still normalize via _cache_scope_from_session_id. Default installs compact in place (session_id never rotates), so they hit the memo and produce byte-identical keys to before. Fixes #79017 --- agent/auxiliary_client.py | 27 +- agent/chat_completion_helpers.py | 20 ++ agent/prompt_cache_scope.py | 92 +++++++ agent/transports/chat_completions.py | 13 +- agent/transports/codex.py | 29 ++- agent/turn_context.py | 6 + tests/agent/test_prompt_cache_scope.py | 346 +++++++++++++++++++++++++ 7 files changed, 515 insertions(+), 18 deletions(-) create mode 100644 agent/prompt_cache_scope.py create mode 100644 tests/agent/test_prompt_cache_scope.py diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 0c8253d3e5..7e409eb8e3 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1627,14 +1627,18 @@ class _CodexCompletionsAdapter: or base_url_host_matches(_host_src, "models.github.ai") ) if not _is_xai and not _is_github and "prompt_cache_key" not in resp_kwargs: - # Scope by the owning turn's session so two unrelated sessions - # with the same instructions/tools (e.g. compression, MoA, - # flush_memories firing back-to-back on different sessions) - # don't bucket-share a prompt cache slot (#78941). The main - # transport (agent/transports/codex.py::build_kwargs) does the - # same; this adapter had no session handle before - # set_runtime_main() started threading one through. - _scope = _cache_scope_from_session_id(_runtime_main_value("session_id")) + # Scope by the owning turn's conversation so two unrelated + # sessions with the same instructions/tools (e.g. compression, + # MoA, flush_memories firing back-to-back on different + # sessions) don't bucket-share a prompt cache slot (#78941). + # Prefer the rotation-stable logical scope threaded through + # set_runtime_main() (compression-lineage root, #79017) and + # fall back to the physical session id, mirroring the main + # transport (agent/transports/codex.py::build_kwargs). + _scope = _cache_scope_from_session_id( + _runtime_main_value("cache_scope") + or _runtime_main_value("session_id") + ) _cache_key = _content_cache_key(instructions, resp_kwargs.get("tools"), _scope) if _cache_key: resp_kwargs["prompt_cache_key"] = _cache_key @@ -3377,11 +3381,17 @@ def set_runtime_main( api_mode: str = "", auth_mode: str = "", session_id: str = "", + cache_scope: str = "", ) -> contextvars.Token: """Record the current context's live main runtime for auxiliary routing. Context-local state prevents concurrent gateway sessions from overwriting one another while retaining compatibility mirrors for legacy readers. + + ``cache_scope`` is the rotation-stable logical cache scope (compression- + lineage root — agent/prompt_cache_scope.py) resolved once per turn by + turn_context; auxiliary Responses calls prefer it over ``session_id`` + for prompt_cache_key derivation (#79017). """ global _RUNTIME_MAIN_PROVIDER, _RUNTIME_MAIN_MODEL global _RUNTIME_MAIN_BASE_URL, _RUNTIME_MAIN_API_KEY, _RUNTIME_MAIN_API_MODE @@ -3399,6 +3409,7 @@ def set_runtime_main( "api_mode": (api_mode or "").strip(), "auth_mode": (auth_mode or "").strip().lower(), "session_id": (session_id or "").strip(), + "cache_scope": (cache_scope or "").strip(), } # Publish authoritative context before updating locked compatibility # mirrors; concurrent sessions never read those mirrors at runtime. diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 160b65b704..4dd5dbed95 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -578,6 +578,23 @@ def _provider_preferences_for_agent(agent) -> Dict[str, Any]: return preferences +def _prompt_cache_scope_for_agent(agent) -> "str | None": + """Rotation-stable logical cache scope for *agent*, or None. + + Thin guard around ``agent.prompt_cache_scope.resolve_prompt_cache_scope`` + — the transports treat a None/empty value as "fall back to the physical + session_id", so any resolution failure degrades to pre-#79017 behavior + instead of blocking the request build. + """ + try: + from agent.prompt_cache_scope import resolve_prompt_cache_scope + + return resolve_prompt_cache_scope(agent) or None + except Exception: + logger.debug("prompt-cache scope resolution failed", exc_info=True) + return None + + def _merge_nous_portal_messages_extra_body(agent, anthropic_kwargs: dict) -> dict: """Merge Portal ``tags`` / ``session_id`` onto an Anthropic Messages kwargs dict. @@ -1880,6 +1897,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non tools=tools_for_api, reasoning_config=agent.reasoning_config, session_id=getattr(agent, "session_id", None), + cache_scope_id=_prompt_cache_scope_for_agent(agent), base_url=agent.base_url, max_tokens=agent.max_tokens, timeout=agent._resolved_api_call_timeout(), @@ -1989,6 +2007,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non reasoning_config=agent.reasoning_config, request_overrides=agent.request_overrides, session_id=getattr(agent, "session_id", None), + cache_scope_id=_prompt_cache_scope_for_agent(agent), provider_profile=_profile, ollama_num_ctx=agent._ollama_num_ctx, # Context forwarded to profile hooks: @@ -2021,6 +2040,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non reasoning_config=agent.reasoning_config, request_overrides=agent.request_overrides, session_id=getattr(agent, "session_id", None), + cache_scope_id=_prompt_cache_scope_for_agent(agent), model_lower=(agent.model or "").lower(), is_openrouter=_is_or, is_nous=_is_nous, diff --git a/agent/prompt_cache_scope.py b/agent/prompt_cache_scope.py new file mode 100644 index 0000000000..b9c8e86bc9 --- /dev/null +++ b/agent/prompt_cache_scope.py @@ -0,0 +1,92 @@ +"""Rotation-stable logical cache scope for prompt_cache_key derivation. + +Context-compression rotation (legacy ``compression.in_place: false`` mode) +mints a new physical ``session_id`` mid-conversation to segment the +transcript. The prompt-cache scope introduced by #79161 was derived from that +physical id, so every rotation moved the conversation into a fresh cache +bucket even though it is logically the same conversation continuing +(issue #79017). + +``resolve_prompt_cache_scope()`` maps the physical session id to the ROOT of +its *compression lineage* — the pre-rotation session id — using +``SessionDB.get_compression_lineage()``, whose fork-aware semantics +(hardened in #79193) give exactly the scope boundaries the cache key needs: + +- compression-rotation children walk back to the original segment + (rotation-stable scope — the fix); +- ``/new`` starts a lineage-less session (fresh scope); +- ``/branch`` children (``_branched_from``), delegate subagents + (``_delegate_from``), and tool-tagged children (``source="tool"``) are + explicit fork children and keep their own isolated scope, preserving the + sibling/subagent isolation #79161 established; +- cron fires keep their physical ``cron__`` id here — the per-fire + timestamp is stripped later by ``_cache_scope_from_session_id`` exactly as + before. + +The resolution is memoized per (agent, session_id): the lineage walk runs +once per transcript segment — NOT per API call — and re-runs only when +rotation actually changes ``agent.session_id`` (per the no-DB-on-the-hot-path +constraint recorded on #79017). Default installs compact in place and never +rotate, so they hit the memo forever and behave byte-identically to before. +""" + +import logging +from typing import Any, Optional + +logger = logging.getLogger(__name__) + +_MEMO_ATTR = "_prompt_cache_scope_memo" + + +def _lineage_root(session_id: str, session_db: Any) -> Optional[str]: + """Return the compression-lineage root of *session_id*, or None. + + Defensive about the DB handle: test doubles and partially constructed + agents can hand back non-list results — anything that is not a non-empty + list/tuple whose first element is a non-empty string is ignored. + """ + if session_db is None: + return None + try: + lineage = session_db.get_compression_lineage(session_id) + except Exception: + logger.debug("prompt-cache scope lineage walk failed", exc_info=True) + return None + if isinstance(lineage, (list, tuple)) and lineage: + root = lineage[0] + if isinstance(root, str) and root: + return root + return None + + +def resolve_prompt_cache_scope(agent: Any) -> str: + """Resolve the rotation-stable cache-scope id for *agent*'s conversation. + + Returns the compression-lineage ROOT of ``agent.session_id`` (the + physical id itself when the session has no compression ancestry, no DB + is attached, or the walk fails). The result is memoized on the agent + keyed by the current session id, so the DB walk happens once per + transcript segment rather than once per API call. + """ + sid = str(getattr(agent, "session_id", None) or "") + if not sid: + return "" + memo = getattr(agent, _MEMO_ATTR, None) + if isinstance(memo, tuple) and len(memo) == 2 and memo[0] == sid: + return memo[1] + db = getattr(agent, "_session_db", None) + root = _lineage_root(sid, db) if db is not None else None + scope = root or sid + # Memoize on a successful walk, or when there is no DB to consult at all. + # A failed/empty walk (row not persisted yet, transient DB error) is NOT + # memoized: falling back to the physical id is the correct degraded + # answer right now, but pinning it for the whole segment would keep the + # scope wrong after the session row lands. + if root is not None or db is None: + try: + setattr(agent, _MEMO_ATTR, (sid, scope)) + except Exception: + # Frozen/slotted test doubles — resolution still works, just + # unmemoized. + pass + return scope diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index 88a5d2ea17..f6f0cd5184 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -48,8 +48,15 @@ def _add_prompt_cache_key( tools: list[dict[str, Any]] | None, supports_prompt_cache_key: bool, session_id: str | None = None, + cache_scope_id: str | None = None, ) -> None: - """Add a content-addressed key only for an explicitly capable endpoint.""" + """Add a content-addressed key only for an explicitly capable endpoint. + + ``cache_scope_id``, when provided, is the rotation-stable logical scope + (compression-lineage root — agent/prompt_cache_scope.py) and takes + precedence over the physical ``session_id`` so the key survives + context-compression session rotation (#79017). + """ if not supports_prompt_cache_key: return @@ -70,7 +77,7 @@ def _add_prompt_cache_key( cache_key = _content_cache_key( _static_prompt_instructions(messages), tools, - _cache_scope_from_session_id(session_id), + _cache_scope_from_session_id(cache_scope_id or session_id), ) if cache_key: api_kwargs["prompt_cache_key"] = cache_key @@ -641,6 +648,7 @@ class ChatCompletionsTransport(ProviderTransport): supports_prompt_cache_key=bool(params.get("supports_prompt_cache_key")) or _is_openai_api_base_url(params.get("base_url")), session_id=params.get("session_id"), + cache_scope_id=params.get("cache_scope_id"), ) return api_kwargs @@ -791,6 +799,7 @@ class ChatCompletionsTransport(ProviderTransport): tools=api_kwargs.get("tools"), supports_prompt_cache_key=bool(getattr(profile, "supports_prompt_cache_key", False)), session_id=params.get("session_id"), + cache_scope_id=params.get("cache_scope_id"), ) return api_kwargs diff --git a/agent/transports/codex.py b/agent/transports/codex.py index 919e6e6fdd..1370bdabcc 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -342,10 +342,15 @@ class ResponsesApiTransport(ProviderTransport): params: instructions: str — system prompt (extracted from messages[0] if not given) reasoning_config: dict | None — {effort, enabled} - session_id: str | None — transcript/session id; drives the xAI - x-grok-conv-id header and the Codex cache-scope headers, and is - the fallback prompt_cache_key when there is no static prefix to - content-address + session_id: str | None — transcript/session id; drives the Codex + ``session_id`` header, and is the cache-scope fallback when no + ``cache_scope_id`` is given + cache_scope_id: str | None — rotation-stable logical scope id + (compression-lineage root; see agent/prompt_cache_scope.py). + Preferred over session_id for prompt_cache_key derivation and + the xAI x-grok-conv-id / Codex x-client-request-id routing + headers, so the cache stays warm across context-compression + session rotation (#79017) max_tokens: int | None — max_output_tokens timeout: float | None — per-request timeout forwarded to the SDK request_overrides: dict | None — extra kwargs merged in @@ -512,10 +517,18 @@ class ResponsesApiTransport(ProviderTransport): # recurring cron jobs carry a per-fire timestamp in session_id # (cron__) that made every run cache-cold, so the scope strips # that suffix (see _cache_scope_from_session_id). session_id is left - # untouched for transcript isolation and the cache-scope routing - # headers below. Falls back to session_id when there is no static - # content to hash. - _cache_scope = _cache_scope_from_session_id(session_id) + # untouched for transcript isolation (the Codex ``session_id`` header + # below). Falls back to session_id when there is no static content to + # hash. + # + # cache_scope_id, when provided, is the rotation-stable logical scope + # (compression-lineage root — agent/prompt_cache_scope.py): legacy + # ``compression.in_place: false`` compaction rotates session_id + # mid-conversation, and scoping by the physical id went cache-cold at + # every rotation boundary (#79017). + _cache_scope = _cache_scope_from_session_id( + params.get("cache_scope_id") or session_id + ) cache_key = _content_cache_key( instructions, response_tools, _cache_scope ) or _cache_scope diff --git a/agent/turn_context.py b/agent/turn_context.py index dfa5fbbd8e..60c9279a79 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -485,6 +485,7 @@ def build_turn_context( # after primary restoration has settled the runtime. try: from agent.auxiliary_client import set_runtime_main + from agent.prompt_cache_scope import resolve_prompt_cache_scope set_runtime_main( getattr(agent, "provider", "") or "", getattr(agent, "model", "") or "", @@ -494,6 +495,11 @@ def build_turn_context( api_mode=getattr(agent, "api_mode", "") or "", auth_mode=getattr(agent, "auth_mode", "") or "", session_id=getattr(agent, "session_id", "") or "", + # Rotation-stable prompt-cache scope, resolved once per turn here + # (memoized per segment — no DB walk on the per-call hot path). + # Stays valid through a mid-turn compression rotation because the + # lineage root is by definition rotation-invariant (#79017). + cache_scope=resolve_prompt_cache_scope(agent), ) except Exception: pass diff --git a/tests/agent/test_prompt_cache_scope.py b/tests/agent/test_prompt_cache_scope.py new file mode 100644 index 0000000000..4088769b70 --- /dev/null +++ b/tests/agent/test_prompt_cache_scope.py @@ -0,0 +1,346 @@ +"""Tests for the rotation-stable prompt-cache scope (issue #79017). + +Legacy ``compression.in_place: false`` compaction rotates the physical +session_id mid-conversation. The prompt_cache_key scope (#79161) was derived +from that physical id, so every rotation went cache-cold. The fix resolves +the compression-lineage ROOT once per turn and threads it to the key +derivation sites, while preserving #79161's isolation semantics for /new, +/branch, delegate subagents, tool children, and unrelated sessions. +""" + +from __future__ import annotations + +from types import SimpleNamespace + +import pytest + +from agent.prompt_cache_scope import resolve_prompt_cache_scope +from agent.transports.codex import _cache_scope_from_session_id, _content_cache_key +from hermes_state import SessionDB + + +@pytest.fixture() +def db(tmp_path): + session_db = SessionDB(db_path=tmp_path / "state.db") + try: + yield session_db + finally: + session_db.close() + + +def _agent(session_id, session_db=None): + return SimpleNamespace(session_id=session_id, _session_db=session_db) + + +def _rotate(db, parent_id: str, child_id: str) -> None: + """Simulate a legacy-mode compression rotation parent -> child.""" + db.end_session(parent_id, "compression") + db.create_session(child_id, source="webui", parent_session_id=parent_id) + + +class TestResolvePromptCacheScope: + def test_no_session_id_returns_empty(self): + assert resolve_prompt_cache_scope(_agent(None)) == "" + assert resolve_prompt_cache_scope(_agent("")) == "" + + def test_no_db_falls_back_to_physical_id(self): + assert resolve_prompt_cache_scope(_agent("root-sess")) == "root-sess" + + def test_unrotated_session_is_its_own_scope(self, db): + db.create_session("root-sess", source="webui") + assert resolve_prompt_cache_scope(_agent("root-sess", db)) == "root-sess" + + def test_rotation_child_inherits_root_scope(self, db): + """THE fix: scope survives a compression rotation boundary.""" + db.create_session("root-sess", source="webui") + _rotate(db, "root-sess", "rotated-1") + + assert resolve_prompt_cache_scope(_agent("rotated-1", db)) == "root-sess" + + def test_chained_rotations_share_one_scope(self, db): + db.create_session("root-sess", source="webui") + _rotate(db, "root-sess", "rotated-1") + _rotate(db, "rotated-1", "rotated-2") + + assert resolve_prompt_cache_scope(_agent("rotated-2", db)) == "root-sess" + + def test_new_session_gets_fresh_scope(self, db): + """/new starts a lineage-less session — never inherits an old scope.""" + db.create_session("old-conv", source="webui") + _rotate(db, "old-conv", "old-rotated") + db.create_session("new-conv", source="webui") # /new: no parent link + + assert resolve_prompt_cache_scope(_agent("new-conv", db)) == "new-conv" + + def test_branch_child_stays_isolated(self, db): + """/branch children are explicit forks — own scope, not the root's.""" + db.create_session("root-sess", source="webui") + db.end_session("root-sess", "compression") + db.create_session( + "branch-child", + source="webui", + parent_session_id="root-sess", + model_config={"_branched_from": "root-sess"}, + ) + + assert ( + resolve_prompt_cache_scope(_agent("branch-child", db)) == "branch-child" + ) + + def test_delegate_child_stays_isolated(self, db): + """Delegate subagents keep per-child scopes (matches #79161 semantics).""" + db.create_session("parent-sess", source="webui") + db.end_session("parent-sess", "compression") + db.create_session( + "delegate-child", + source="webui", + parent_session_id="parent-sess", + model_config={"_delegate_from": "parent-sess"}, + ) + + assert ( + resolve_prompt_cache_scope(_agent("delegate-child", db)) + == "delegate-child" + ) + + def test_tool_child_stays_isolated(self, db): + db.create_session("parent-sess", source="webui") + db.end_session("parent-sess", "compression") + db.create_session( + "tool-child", source="tool", parent_session_id="parent-sess" + ) + + assert resolve_prompt_cache_scope(_agent("tool-child", db)) == "tool-child" + + def test_memoized_per_segment(self, db): + """The lineage walk runs once per (agent, session_id) — hot-path rule.""" + db.create_session("root-sess", source="webui") + _rotate(db, "root-sess", "rotated-1") + agent = _agent("rotated-1", db) + + assert resolve_prompt_cache_scope(agent) == "root-sess" + + calls = [] + original = db.get_compression_lineage + db.get_compression_lineage = lambda sid: calls.append(sid) or original(sid) + try: + assert resolve_prompt_cache_scope(agent) == "root-sess" + assert calls == [] # memo hit — no second walk + # Rotation changes the physical id -> memo invalidates, one re-walk. + _rotate(db, "rotated-1", "rotated-2") + agent.session_id = "rotated-2" + assert resolve_prompt_cache_scope(agent) == "root-sess" + assert calls == ["rotated-2"] + finally: + db.get_compression_lineage = original + + def test_db_failure_falls_back_to_physical_id(self): + class BoomDB: + def get_compression_lineage(self, sid): + raise RuntimeError("db exploded") + + assert resolve_prompt_cache_scope(_agent("sess-x", BoomDB())) == "sess-x" + + def test_failed_walk_is_not_pinned(self, db): + """A pre-persist miss must not memoize the physical id for the segment. + + turn_context resolves the scope before _ensure_db_session persists the + row on a brand-new agent; once the row (and any rotation ancestry) + lands, the next resolution must see it. + """ + agent = _agent("late-row", db) + # Row doesn't exist yet -> degraded fallback, unmemoized. + assert resolve_prompt_cache_scope(agent) == "late-row" + # Row lands with rotation ancestry. + db.create_session("late-root", source="webui") + db.end_session("late-root", "compression") + db.create_session("late-row", source="webui", parent_session_id="late-root") + assert resolve_prompt_cache_scope(agent) == "late-root" + + def test_bogus_lineage_shape_falls_back(self): + class WeirdDB: + def get_compression_lineage(self, sid): + return "not-a-list" + + assert resolve_prompt_cache_scope(_agent("sess-y", WeirdDB())) == "sess-y" + + +class TestRotationContinuityEndToEnd: + """The acceptance shape from #79017: same conversation, same key.""" + + INSTRUCTIONS = "You are a helpful assistant." + TOOLS = [{"type": "function", "name": "terminal"}] + + def _key_for(self, agent): + scope = _cache_scope_from_session_id(resolve_prompt_cache_scope(agent)) + return _content_cache_key(self.INSTRUCTIONS, self.TOOLS, scope) + + def test_rotation_keeps_prompt_cache_key_stable(self, db): + db.create_session("root-sess", source="webui") + key_before = self._key_for(_agent("root-sess", db)) + + _rotate(db, "root-sess", "rotated-1") + key_after = self._key_for(_agent("rotated-1", db)) + + assert key_before == key_after + + def test_unrelated_sessions_keep_distinct_keys(self, db): + db.create_session("conv-a", source="webui") + db.create_session("conv-b", source="webui") + + assert self._key_for(_agent("conv-a", db)) != self._key_for( + _agent("conv-b", db) + ) + + def test_sibling_forks_keep_distinct_keys(self, db): + db.create_session("parent-sess", source="webui") + db.end_session("parent-sess", "compression") + for child in ("delegate-a", "delegate-b"): + db.create_session( + child, + source="webui", + parent_session_id="parent-sess", + model_config={"_delegate_from": "parent-sess"}, + ) + + key_a = self._key_for(_agent("delegate-a", db)) + key_b = self._key_for(_agent("delegate-b", db)) + assert key_a != key_b + + +class TestTransportWiring: + """cache_scope_id reaches the key derivation on both transports.""" + + def test_codex_build_kwargs_prefers_cache_scope_id(self): + from agent.transports.codex import ResponsesApiTransport + + transport = ResponsesApiTransport() + base = dict( + model="gpt-5.5", + messages=[ + {"role": "system", "content": "sys"}, + {"role": "user", "content": "hi"}, + ], + tools=[], + ) + # Rotation: different physical ids, same logical scope -> same key. + k1 = transport.build_kwargs( + **base, session_id="root-sess", cache_scope_id="root-sess" + ) + k2 = transport.build_kwargs( + **base, session_id="rotated-1", cache_scope_id="root-sess" + ) + assert k1["prompt_cache_key"] == k2["prompt_cache_key"] + # Without the logical scope, rotation used to change the key. + k3 = transport.build_kwargs(**base, session_id="rotated-1") + assert k3["prompt_cache_key"] != k1["prompt_cache_key"] + + def test_codex_session_header_keeps_physical_id(self): + """Transcript identity (#57012 contract) must NOT be rewritten.""" + from agent.transports.codex import ResponsesApiTransport + + transport = ResponsesApiTransport() + kwargs = transport.build_kwargs( + model="gpt-5.5", + messages=[ + {"role": "system", "content": "sys"}, + {"role": "user", "content": "hi"}, + ], + tools=[], + session_id="rotated-1", + cache_scope_id="root-sess", + is_codex_backend=True, + ) + assert kwargs["extra_headers"]["session_id"] == "rotated-1" + # Routing header mirrors the body's scoped cache key. + assert kwargs["extra_headers"]["x-client-request-id"] == kwargs[ + "prompt_cache_key" + ] + + def test_xai_conv_id_uses_logical_scope(self): + from agent.transports.codex import ResponsesApiTransport + + transport = ResponsesApiTransport() + kwargs = transport.build_kwargs( + model="grok-4", + messages=[ + {"role": "system", "content": "sys"}, + {"role": "user", "content": "hi"}, + ], + tools=[], + session_id="rotated-1", + cache_scope_id="root-sess", + is_xai_responses=True, + ) + assert kwargs["extra_headers"]["x-grok-conv-id"] == "root-sess" + + def test_chat_completions_prefers_cache_scope_id(self): + from agent.transports.chat_completions import _add_prompt_cache_key + + messages = [{"role": "system", "content": "sys"}] + + def key(session_id, cache_scope_id=None): + kwargs = {} + _add_prompt_cache_key( + kwargs, + messages=messages, + tools=None, + supports_prompt_cache_key=True, + session_id=session_id, + cache_scope_id=cache_scope_id, + ) + return kwargs.get("prompt_cache_key") + + assert key("root-sess", "root-sess") == key("rotated-1", "root-sess") + assert key("rotated-1") != key("rotated-1", "root-sess") + + def test_cron_normalization_still_applies_to_scope(self): + """cron__ scopes still normalize per-fire timestamps away.""" + from agent.transports.codex import ResponsesApiTransport + + transport = ResponsesApiTransport() + base = dict( + model="gpt-5.5", + messages=[ + {"role": "system", "content": "sys"}, + {"role": "user", "content": "hi"}, + ], + tools=[], + ) + k1 = transport.build_kwargs( + **base, + session_id="cron_backup_20260814_120000", + cache_scope_id="cron_backup_20260814_120000", + ) + k2 = transport.build_kwargs( + **base, + session_id="cron_backup_20260815_120000", + cache_scope_id="cron_backup_20260815_120000", + ) + assert k1["prompt_cache_key"] == k2["prompt_cache_key"] + + +class TestAuxiliaryRuntimeThreading: + def test_set_runtime_main_carries_cache_scope(self): + import agent.auxiliary_client as aux + + token = aux.set_runtime_main( + "openrouter", + "gpt-5.5", + session_id="rotated-1", + cache_scope="root-sess", + ) + try: + assert aux._runtime_main_value("cache_scope") == "root-sess" + assert aux._runtime_main_value("session_id") == "rotated-1" + finally: + aux.reset_runtime_main(token) + + def test_cache_scope_defaults_empty(self): + import agent.auxiliary_client as aux + + token = aux.set_runtime_main("openrouter", "gpt-5.5", session_id="s-1") + try: + assert aux._runtime_main_value("cache_scope") == "" + finally: + aux.reset_runtime_main(token)