fix(cache): keep prompt_cache_key warm across compression session rotation

Legacy compaction mode (compression.in_place: false) rotates the physical
session_id mid-conversation. The prompt-cache scope introduced in #79161 was
derived from that physical id, so every rotation moved the same conversation
into a fresh cache bucket - the prompt cache went cold at every rotation
boundary (#79017).

Fix: resolve a rotation-stable logical scope - the compression-lineage ROOT
of the current session (SessionDB.get_compression_lineage, fork-aware
post-#79193) - once per turn, memoized per transcript segment, and prefer it
over the physical session_id at every prompt_cache_key derivation site:

- agent/prompt_cache_scope.py (new): resolve_prompt_cache_scope(agent) -
  lineage-root walk with per-segment memo; falls back to the physical id
  when no DB is attached or the walk fails, degrading to pre-fix behavior.
- transports/codex.py: build_kwargs accepts cache_scope_id and prefers it
  for the body prompt_cache_key, the xAI x-grok-conv-id header, and the
  Codex x-client-request-id routing header. The Codex session_id header
  keeps the raw physical id (transcript identity, #57012 contract).
- transports/chat_completions.py: _add_prompt_cache_key accepts
  cache_scope_id with the same precedence.
- chat_completion_helpers.py: build_api_kwargs threads the resolved scope
  into all three build_kwargs call sites (codex, profile, legacy).
- auxiliary_client.py: set_runtime_main carries cache_scope; the aux
  Responses cache-key site prefers it over the physical session_id.
- turn_context.py: resolves the scope once per turn and threads it through
  set_runtime_main (no DB walk on the per-API-call hot path).

Scope semantics preserved from #79161: /new starts a fresh scope (new
lineage), /branch children, delegate subagents, and tool children stay
isolated (explicit-fork exclusion in get_compression_lineage), unrelated
sessions keep distinct buckets, and cron per-fire timestamps still
normalize via _cache_scope_from_session_id.

Default installs compact in place (session_id never rotates), so they hit
the memo and produce byte-identical keys to before.

Fixes #79017
This commit is contained in:
kshitij
2026-08-15 10:45:52 +05:30
parent 471c687c2b
commit cee2446222
7 changed files with 515 additions and 18 deletions
+19 -8
View File
@@ -1627,14 +1627,18 @@ class _CodexCompletionsAdapter:
or base_url_host_matches(_host_src, "models.github.ai")
)
if not _is_xai and not _is_github and "prompt_cache_key" not in resp_kwargs:
# Scope by the owning turn's session so two unrelated sessions
# with the same instructions/tools (e.g. compression, MoA,
# flush_memories firing back-to-back on different sessions)
# don't bucket-share a prompt cache slot (#78941). The main
# transport (agent/transports/codex.py::build_kwargs) does the
# same; this adapter had no session handle before
# set_runtime_main() started threading one through.
_scope = _cache_scope_from_session_id(_runtime_main_value("session_id"))
# Scope by the owning turn's conversation so two unrelated
# sessions with the same instructions/tools (e.g. compression,
# MoA, flush_memories firing back-to-back on different
# sessions) don't bucket-share a prompt cache slot (#78941).
# Prefer the rotation-stable logical scope threaded through
# set_runtime_main() (compression-lineage root, #79017) and
# fall back to the physical session id, mirroring the main
# transport (agent/transports/codex.py::build_kwargs).
_scope = _cache_scope_from_session_id(
_runtime_main_value("cache_scope")
or _runtime_main_value("session_id")
)
_cache_key = _content_cache_key(instructions, resp_kwargs.get("tools"), _scope)
if _cache_key:
resp_kwargs["prompt_cache_key"] = _cache_key
@@ -3377,11 +3381,17 @@ def set_runtime_main(
api_mode: str = "",
auth_mode: str = "",
session_id: str = "",
cache_scope: str = "",
) -> contextvars.Token:
"""Record the current context's live main runtime for auxiliary routing.
Context-local state prevents concurrent gateway sessions from overwriting
one another while retaining compatibility mirrors for legacy readers.
``cache_scope`` is the rotation-stable logical cache scope (compression-
lineage root — agent/prompt_cache_scope.py) resolved once per turn by
turn_context; auxiliary Responses calls prefer it over ``session_id``
for prompt_cache_key derivation (#79017).
"""
global _RUNTIME_MAIN_PROVIDER, _RUNTIME_MAIN_MODEL
global _RUNTIME_MAIN_BASE_URL, _RUNTIME_MAIN_API_KEY, _RUNTIME_MAIN_API_MODE
@@ -3399,6 +3409,7 @@ def set_runtime_main(
"api_mode": (api_mode or "").strip(),
"auth_mode": (auth_mode or "").strip().lower(),
"session_id": (session_id or "").strip(),
"cache_scope": (cache_scope or "").strip(),
}
# Publish authoritative context before updating locked compatibility
# mirrors; concurrent sessions never read those mirrors at runtime.
+20
View File
@@ -578,6 +578,23 @@ def _provider_preferences_for_agent(agent) -> Dict[str, Any]:
return preferences
def _prompt_cache_scope_for_agent(agent) -> "str | None":
"""Rotation-stable logical cache scope for *agent*, or None.
Thin guard around ``agent.prompt_cache_scope.resolve_prompt_cache_scope``
— the transports treat a None/empty value as "fall back to the physical
session_id", so any resolution failure degrades to pre-#79017 behavior
instead of blocking the request build.
"""
try:
from agent.prompt_cache_scope import resolve_prompt_cache_scope
return resolve_prompt_cache_scope(agent) or None
except Exception:
logger.debug("prompt-cache scope resolution failed", exc_info=True)
return None
def _merge_nous_portal_messages_extra_body(agent, anthropic_kwargs: dict) -> dict:
"""Merge Portal ``tags`` / ``session_id`` onto an Anthropic Messages kwargs dict.
@@ -1880,6 +1897,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
tools=tools_for_api,
reasoning_config=agent.reasoning_config,
session_id=getattr(agent, "session_id", None),
cache_scope_id=_prompt_cache_scope_for_agent(agent),
base_url=agent.base_url,
max_tokens=agent.max_tokens,
timeout=agent._resolved_api_call_timeout(),
@@ -1989,6 +2007,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
reasoning_config=agent.reasoning_config,
request_overrides=agent.request_overrides,
session_id=getattr(agent, "session_id", None),
cache_scope_id=_prompt_cache_scope_for_agent(agent),
provider_profile=_profile,
ollama_num_ctx=agent._ollama_num_ctx,
# Context forwarded to profile hooks:
@@ -2021,6 +2040,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
reasoning_config=agent.reasoning_config,
request_overrides=agent.request_overrides,
session_id=getattr(agent, "session_id", None),
cache_scope_id=_prompt_cache_scope_for_agent(agent),
model_lower=(agent.model or "").lower(),
is_openrouter=_is_or,
is_nous=_is_nous,
+92
View File
@@ -0,0 +1,92 @@
"""Rotation-stable logical cache scope for prompt_cache_key derivation.
Context-compression rotation (legacy ``compression.in_place: false`` mode)
mints a new physical ``session_id`` mid-conversation to segment the
transcript. The prompt-cache scope introduced by #79161 was derived from that
physical id, so every rotation moved the conversation into a fresh cache
bucket even though it is logically the same conversation continuing
(issue #79017).
``resolve_prompt_cache_scope()`` maps the physical session id to the ROOT of
its *compression lineage* — the pre-rotation session id — using
``SessionDB.get_compression_lineage()``, whose fork-aware semantics
(hardened in #79193) give exactly the scope boundaries the cache key needs:
- compression-rotation children walk back to the original segment
(rotation-stable scope — the fix);
- ``/new`` starts a lineage-less session (fresh scope);
- ``/branch`` children (``_branched_from``), delegate subagents
(``_delegate_from``), and tool-tagged children (``source="tool"``) are
explicit fork children and keep their own isolated scope, preserving the
sibling/subagent isolation #79161 established;
- cron fires keep their physical ``cron_<job>_<ts>`` id here — the per-fire
timestamp is stripped later by ``_cache_scope_from_session_id`` exactly as
before.
The resolution is memoized per (agent, session_id): the lineage walk runs
once per transcript segment — NOT per API call — and re-runs only when
rotation actually changes ``agent.session_id`` (per the no-DB-on-the-hot-path
constraint recorded on #79017). Default installs compact in place and never
rotate, so they hit the memo forever and behave byte-identically to before.
"""
import logging
from typing import Any, Optional
logger = logging.getLogger(__name__)
_MEMO_ATTR = "_prompt_cache_scope_memo"
def _lineage_root(session_id: str, session_db: Any) -> Optional[str]:
"""Return the compression-lineage root of *session_id*, or None.
Defensive about the DB handle: test doubles and partially constructed
agents can hand back non-list results — anything that is not a non-empty
list/tuple whose first element is a non-empty string is ignored.
"""
if session_db is None:
return None
try:
lineage = session_db.get_compression_lineage(session_id)
except Exception:
logger.debug("prompt-cache scope lineage walk failed", exc_info=True)
return None
if isinstance(lineage, (list, tuple)) and lineage:
root = lineage[0]
if isinstance(root, str) and root:
return root
return None
def resolve_prompt_cache_scope(agent: Any) -> str:
"""Resolve the rotation-stable cache-scope id for *agent*'s conversation.
Returns the compression-lineage ROOT of ``agent.session_id`` (the
physical id itself when the session has no compression ancestry, no DB
is attached, or the walk fails). The result is memoized on the agent
keyed by the current session id, so the DB walk happens once per
transcript segment rather than once per API call.
"""
sid = str(getattr(agent, "session_id", None) or "")
if not sid:
return ""
memo = getattr(agent, _MEMO_ATTR, None)
if isinstance(memo, tuple) and len(memo) == 2 and memo[0] == sid:
return memo[1]
db = getattr(agent, "_session_db", None)
root = _lineage_root(sid, db) if db is not None else None
scope = root or sid
# Memoize on a successful walk, or when there is no DB to consult at all.
# A failed/empty walk (row not persisted yet, transient DB error) is NOT
# memoized: falling back to the physical id is the correct degraded
# answer right now, but pinning it for the whole segment would keep the
# scope wrong after the session row lands.
if root is not None or db is None:
try:
setattr(agent, _MEMO_ATTR, (sid, scope))
except Exception:
# Frozen/slotted test doubles — resolution still works, just
# unmemoized.
pass
return scope
+11 -2
View File
@@ -48,8 +48,15 @@ def _add_prompt_cache_key(
tools: list[dict[str, Any]] | None,
supports_prompt_cache_key: bool,
session_id: str | None = None,
cache_scope_id: str | None = None,
) -> None:
"""Add a content-addressed key only for an explicitly capable endpoint."""
"""Add a content-addressed key only for an explicitly capable endpoint.
``cache_scope_id``, when provided, is the rotation-stable logical scope
(compression-lineage root — agent/prompt_cache_scope.py) and takes
precedence over the physical ``session_id`` so the key survives
context-compression session rotation (#79017).
"""
if not supports_prompt_cache_key:
return
@@ -70,7 +77,7 @@ def _add_prompt_cache_key(
cache_key = _content_cache_key(
_static_prompt_instructions(messages),
tools,
_cache_scope_from_session_id(session_id),
_cache_scope_from_session_id(cache_scope_id or session_id),
)
if cache_key:
api_kwargs["prompt_cache_key"] = cache_key
@@ -641,6 +648,7 @@ class ChatCompletionsTransport(ProviderTransport):
supports_prompt_cache_key=bool(params.get("supports_prompt_cache_key"))
or _is_openai_api_base_url(params.get("base_url")),
session_id=params.get("session_id"),
cache_scope_id=params.get("cache_scope_id"),
)
return api_kwargs
@@ -791,6 +799,7 @@ class ChatCompletionsTransport(ProviderTransport):
tools=api_kwargs.get("tools"),
supports_prompt_cache_key=bool(getattr(profile, "supports_prompt_cache_key", False)),
session_id=params.get("session_id"),
cache_scope_id=params.get("cache_scope_id"),
)
return api_kwargs
+21 -8
View File
@@ -342,10 +342,15 @@ class ResponsesApiTransport(ProviderTransport):
params:
instructions: str — system prompt (extracted from messages[0] if not given)
reasoning_config: dict | None — {effort, enabled}
session_id: str | None — transcript/session id; drives the xAI
x-grok-conv-id header and the Codex cache-scope headers, and is
the fallback prompt_cache_key when there is no static prefix to
content-address
session_id: str | None — transcript/session id; drives the Codex
``session_id`` header, and is the cache-scope fallback when no
``cache_scope_id`` is given
cache_scope_id: str | None — rotation-stable logical scope id
(compression-lineage root; see agent/prompt_cache_scope.py).
Preferred over session_id for prompt_cache_key derivation and
the xAI x-grok-conv-id / Codex x-client-request-id routing
headers, so the cache stays warm across context-compression
session rotation (#79017)
max_tokens: int | None — max_output_tokens
timeout: float | None — per-request timeout forwarded to the SDK
request_overrides: dict | None — extra kwargs merged in
@@ -512,10 +517,18 @@ class ResponsesApiTransport(ProviderTransport):
# recurring cron jobs carry a per-fire timestamp in session_id
# (cron_<id>_<ts>) that made every run cache-cold, so the scope strips
# that suffix (see _cache_scope_from_session_id). session_id is left
# untouched for transcript isolation and the cache-scope routing
# headers below. Falls back to session_id when there is no static
# content to hash.
_cache_scope = _cache_scope_from_session_id(session_id)
# untouched for transcript isolation (the Codex ``session_id`` header
# below). Falls back to session_id when there is no static content to
# hash.
#
# cache_scope_id, when provided, is the rotation-stable logical scope
# (compression-lineage root — agent/prompt_cache_scope.py): legacy
# ``compression.in_place: false`` compaction rotates session_id
# mid-conversation, and scoping by the physical id went cache-cold at
# every rotation boundary (#79017).
_cache_scope = _cache_scope_from_session_id(
params.get("cache_scope_id") or session_id
)
cache_key = _content_cache_key(
instructions, response_tools, _cache_scope
) or _cache_scope
+6
View File
@@ -485,6 +485,7 @@ def build_turn_context(
# after primary restoration has settled the runtime.
try:
from agent.auxiliary_client import set_runtime_main
from agent.prompt_cache_scope import resolve_prompt_cache_scope
set_runtime_main(
getattr(agent, "provider", "") or "",
getattr(agent, "model", "") or "",
@@ -494,6 +495,11 @@ def build_turn_context(
api_mode=getattr(agent, "api_mode", "") or "",
auth_mode=getattr(agent, "auth_mode", "") or "",
session_id=getattr(agent, "session_id", "") or "",
# Rotation-stable prompt-cache scope, resolved once per turn here
# (memoized per segment — no DB walk on the per-call hot path).
# Stays valid through a mid-turn compression rotation because the
# lineage root is by definition rotation-invariant (#79017).
cache_scope=resolve_prompt_cache_scope(agent),
)
except Exception:
pass
+346
View File
@@ -0,0 +1,346 @@
"""Tests for the rotation-stable prompt-cache scope (issue #79017).
Legacy ``compression.in_place: false`` compaction rotates the physical
session_id mid-conversation. The prompt_cache_key scope (#79161) was derived
from that physical id, so every rotation went cache-cold. The fix resolves
the compression-lineage ROOT once per turn and threads it to the key
derivation sites, while preserving #79161's isolation semantics for /new,
/branch, delegate subagents, tool children, and unrelated sessions.
"""
from __future__ import annotations
from types import SimpleNamespace
import pytest
from agent.prompt_cache_scope import resolve_prompt_cache_scope
from agent.transports.codex import _cache_scope_from_session_id, _content_cache_key
from hermes_state import SessionDB
@pytest.fixture()
def db(tmp_path):
session_db = SessionDB(db_path=tmp_path / "state.db")
try:
yield session_db
finally:
session_db.close()
def _agent(session_id, session_db=None):
return SimpleNamespace(session_id=session_id, _session_db=session_db)
def _rotate(db, parent_id: str, child_id: str) -> None:
"""Simulate a legacy-mode compression rotation parent -> child."""
db.end_session(parent_id, "compression")
db.create_session(child_id, source="webui", parent_session_id=parent_id)
class TestResolvePromptCacheScope:
def test_no_session_id_returns_empty(self):
assert resolve_prompt_cache_scope(_agent(None)) == ""
assert resolve_prompt_cache_scope(_agent("")) == ""
def test_no_db_falls_back_to_physical_id(self):
assert resolve_prompt_cache_scope(_agent("root-sess")) == "root-sess"
def test_unrotated_session_is_its_own_scope(self, db):
db.create_session("root-sess", source="webui")
assert resolve_prompt_cache_scope(_agent("root-sess", db)) == "root-sess"
def test_rotation_child_inherits_root_scope(self, db):
"""THE fix: scope survives a compression rotation boundary."""
db.create_session("root-sess", source="webui")
_rotate(db, "root-sess", "rotated-1")
assert resolve_prompt_cache_scope(_agent("rotated-1", db)) == "root-sess"
def test_chained_rotations_share_one_scope(self, db):
db.create_session("root-sess", source="webui")
_rotate(db, "root-sess", "rotated-1")
_rotate(db, "rotated-1", "rotated-2")
assert resolve_prompt_cache_scope(_agent("rotated-2", db)) == "root-sess"
def test_new_session_gets_fresh_scope(self, db):
"""/new starts a lineage-less session — never inherits an old scope."""
db.create_session("old-conv", source="webui")
_rotate(db, "old-conv", "old-rotated")
db.create_session("new-conv", source="webui") # /new: no parent link
assert resolve_prompt_cache_scope(_agent("new-conv", db)) == "new-conv"
def test_branch_child_stays_isolated(self, db):
"""/branch children are explicit forks — own scope, not the root's."""
db.create_session("root-sess", source="webui")
db.end_session("root-sess", "compression")
db.create_session(
"branch-child",
source="webui",
parent_session_id="root-sess",
model_config={"_branched_from": "root-sess"},
)
assert (
resolve_prompt_cache_scope(_agent("branch-child", db)) == "branch-child"
)
def test_delegate_child_stays_isolated(self, db):
"""Delegate subagents keep per-child scopes (matches #79161 semantics)."""
db.create_session("parent-sess", source="webui")
db.end_session("parent-sess", "compression")
db.create_session(
"delegate-child",
source="webui",
parent_session_id="parent-sess",
model_config={"_delegate_from": "parent-sess"},
)
assert (
resolve_prompt_cache_scope(_agent("delegate-child", db))
== "delegate-child"
)
def test_tool_child_stays_isolated(self, db):
db.create_session("parent-sess", source="webui")
db.end_session("parent-sess", "compression")
db.create_session(
"tool-child", source="tool", parent_session_id="parent-sess"
)
assert resolve_prompt_cache_scope(_agent("tool-child", db)) == "tool-child"
def test_memoized_per_segment(self, db):
"""The lineage walk runs once per (agent, session_id) — hot-path rule."""
db.create_session("root-sess", source="webui")
_rotate(db, "root-sess", "rotated-1")
agent = _agent("rotated-1", db)
assert resolve_prompt_cache_scope(agent) == "root-sess"
calls = []
original = db.get_compression_lineage
db.get_compression_lineage = lambda sid: calls.append(sid) or original(sid)
try:
assert resolve_prompt_cache_scope(agent) == "root-sess"
assert calls == [] # memo hit — no second walk
# Rotation changes the physical id -> memo invalidates, one re-walk.
_rotate(db, "rotated-1", "rotated-2")
agent.session_id = "rotated-2"
assert resolve_prompt_cache_scope(agent) == "root-sess"
assert calls == ["rotated-2"]
finally:
db.get_compression_lineage = original
def test_db_failure_falls_back_to_physical_id(self):
class BoomDB:
def get_compression_lineage(self, sid):
raise RuntimeError("db exploded")
assert resolve_prompt_cache_scope(_agent("sess-x", BoomDB())) == "sess-x"
def test_failed_walk_is_not_pinned(self, db):
"""A pre-persist miss must not memoize the physical id for the segment.
turn_context resolves the scope before _ensure_db_session persists the
row on a brand-new agent; once the row (and any rotation ancestry)
lands, the next resolution must see it.
"""
agent = _agent("late-row", db)
# Row doesn't exist yet -> degraded fallback, unmemoized.
assert resolve_prompt_cache_scope(agent) == "late-row"
# Row lands with rotation ancestry.
db.create_session("late-root", source="webui")
db.end_session("late-root", "compression")
db.create_session("late-row", source="webui", parent_session_id="late-root")
assert resolve_prompt_cache_scope(agent) == "late-root"
def test_bogus_lineage_shape_falls_back(self):
class WeirdDB:
def get_compression_lineage(self, sid):
return "not-a-list"
assert resolve_prompt_cache_scope(_agent("sess-y", WeirdDB())) == "sess-y"
class TestRotationContinuityEndToEnd:
"""The acceptance shape from #79017: same conversation, same key."""
INSTRUCTIONS = "You are a helpful assistant."
TOOLS = [{"type": "function", "name": "terminal"}]
def _key_for(self, agent):
scope = _cache_scope_from_session_id(resolve_prompt_cache_scope(agent))
return _content_cache_key(self.INSTRUCTIONS, self.TOOLS, scope)
def test_rotation_keeps_prompt_cache_key_stable(self, db):
db.create_session("root-sess", source="webui")
key_before = self._key_for(_agent("root-sess", db))
_rotate(db, "root-sess", "rotated-1")
key_after = self._key_for(_agent("rotated-1", db))
assert key_before == key_after
def test_unrelated_sessions_keep_distinct_keys(self, db):
db.create_session("conv-a", source="webui")
db.create_session("conv-b", source="webui")
assert self._key_for(_agent("conv-a", db)) != self._key_for(
_agent("conv-b", db)
)
def test_sibling_forks_keep_distinct_keys(self, db):
db.create_session("parent-sess", source="webui")
db.end_session("parent-sess", "compression")
for child in ("delegate-a", "delegate-b"):
db.create_session(
child,
source="webui",
parent_session_id="parent-sess",
model_config={"_delegate_from": "parent-sess"},
)
key_a = self._key_for(_agent("delegate-a", db))
key_b = self._key_for(_agent("delegate-b", db))
assert key_a != key_b
class TestTransportWiring:
"""cache_scope_id reaches the key derivation on both transports."""
def test_codex_build_kwargs_prefers_cache_scope_id(self):
from agent.transports.codex import ResponsesApiTransport
transport = ResponsesApiTransport()
base = dict(
model="gpt-5.5",
messages=[
{"role": "system", "content": "sys"},
{"role": "user", "content": "hi"},
],
tools=[],
)
# Rotation: different physical ids, same logical scope -> same key.
k1 = transport.build_kwargs(
**base, session_id="root-sess", cache_scope_id="root-sess"
)
k2 = transport.build_kwargs(
**base, session_id="rotated-1", cache_scope_id="root-sess"
)
assert k1["prompt_cache_key"] == k2["prompt_cache_key"]
# Without the logical scope, rotation used to change the key.
k3 = transport.build_kwargs(**base, session_id="rotated-1")
assert k3["prompt_cache_key"] != k1["prompt_cache_key"]
def test_codex_session_header_keeps_physical_id(self):
"""Transcript identity (#57012 contract) must NOT be rewritten."""
from agent.transports.codex import ResponsesApiTransport
transport = ResponsesApiTransport()
kwargs = transport.build_kwargs(
model="gpt-5.5",
messages=[
{"role": "system", "content": "sys"},
{"role": "user", "content": "hi"},
],
tools=[],
session_id="rotated-1",
cache_scope_id="root-sess",
is_codex_backend=True,
)
assert kwargs["extra_headers"]["session_id"] == "rotated-1"
# Routing header mirrors the body's scoped cache key.
assert kwargs["extra_headers"]["x-client-request-id"] == kwargs[
"prompt_cache_key"
]
def test_xai_conv_id_uses_logical_scope(self):
from agent.transports.codex import ResponsesApiTransport
transport = ResponsesApiTransport()
kwargs = transport.build_kwargs(
model="grok-4",
messages=[
{"role": "system", "content": "sys"},
{"role": "user", "content": "hi"},
],
tools=[],
session_id="rotated-1",
cache_scope_id="root-sess",
is_xai_responses=True,
)
assert kwargs["extra_headers"]["x-grok-conv-id"] == "root-sess"
def test_chat_completions_prefers_cache_scope_id(self):
from agent.transports.chat_completions import _add_prompt_cache_key
messages = [{"role": "system", "content": "sys"}]
def key(session_id, cache_scope_id=None):
kwargs = {}
_add_prompt_cache_key(
kwargs,
messages=messages,
tools=None,
supports_prompt_cache_key=True,
session_id=session_id,
cache_scope_id=cache_scope_id,
)
return kwargs.get("prompt_cache_key")
assert key("root-sess", "root-sess") == key("rotated-1", "root-sess")
assert key("rotated-1") != key("rotated-1", "root-sess")
def test_cron_normalization_still_applies_to_scope(self):
"""cron_<job>_<ts> scopes still normalize per-fire timestamps away."""
from agent.transports.codex import ResponsesApiTransport
transport = ResponsesApiTransport()
base = dict(
model="gpt-5.5",
messages=[
{"role": "system", "content": "sys"},
{"role": "user", "content": "hi"},
],
tools=[],
)
k1 = transport.build_kwargs(
**base,
session_id="cron_backup_20260814_120000",
cache_scope_id="cron_backup_20260814_120000",
)
k2 = transport.build_kwargs(
**base,
session_id="cron_backup_20260815_120000",
cache_scope_id="cron_backup_20260815_120000",
)
assert k1["prompt_cache_key"] == k2["prompt_cache_key"]
class TestAuxiliaryRuntimeThreading:
def test_set_runtime_main_carries_cache_scope(self):
import agent.auxiliary_client as aux
token = aux.set_runtime_main(
"openrouter",
"gpt-5.5",
session_id="rotated-1",
cache_scope="root-sess",
)
try:
assert aux._runtime_main_value("cache_scope") == "root-sess"
assert aux._runtime_main_value("session_id") == "rotated-1"
finally:
aux.reset_runtime_main(token)
def test_cache_scope_defaults_empty(self):
import agent.auxiliary_client as aux
token = aux.set_runtime_main("openrouter", "gpt-5.5", session_id="s-1")
try:
assert aux._runtime_main_value("cache_scope") == ""
finally:
aux.reset_runtime_main(token)