diff --git a/agent/codex_headers.py b/agent/codex_headers.py index 607ba4055b..b7e72f8ff1 100644 --- a/agent/codex_headers.py +++ b/agent/codex_headers.py @@ -1,8 +1,8 @@ """Codex request identity helpers shared by agent client builders. -This leaf module intentionally has no dependency on the large auxiliary-client -router. Long-lived processes can therefore import a newly added client builder -without resolving a new symbol from an older cached ``auxiliary_client`` module. +Leaf module with no dependency on the large auxiliary-client router, so a +long-lived process can import a newly added client builder without resolving a +new symbol from an older cached ``auxiliary_client`` module. """ from __future__ import annotations @@ -36,27 +36,19 @@ def codex_cloudflare_headers( ) -> Dict[str, str]: """Identity and account headers for chatgpt.com/backend-api/codex. - OpenAI requires third-party harnesses to identify themselves. Requests to - the official endpoint always send Hermes' originator and version. Custom - endpoints retain the existing compatibility identity. In either case, - preserve ``ChatGPT-Account-ID`` from the OAuth JWT's - ``chatgpt_account_id`` claim. - - Malformed tokens are tolerated — we drop the account-ID header rather than - raise, so a bad token still surfaces as an auth error (401) instead of a - crash at client construction. + OpenAI requires third-party harnesses to identify themselves: the official + endpoint gets Hermes' originator and version, custom endpoints keep the + codex_cli_rs compatibility identity. ``ChatGPT-Account-ID`` comes from the + OAuth JWT's ``chatgpt_account_id`` claim; a malformed token drops the header + rather than raising, so it surfaces as a 401 instead of a crash at client + construction. """ - headers = { - "User-Agent": "codex_cli_rs/0.0.0 (Hermes Agent)", - "originator": "codex_cli_rs", - } if is_official_codex_base_url(base_url): from hermes_cli import __version__ - headers.update({ - "User-Agent": f"HermesAgent/{__version__}", - "originator": "hermes-agent", - }) + headers = {"User-Agent": f"HermesAgent/{__version__}", "originator": "hermes-agent"} + else: + headers = {"User-Agent": "codex_cli_rs/0.0.0 (Hermes Agent)", "originator": "codex_cli_rs"} if not isinstance(access_token, str) or not access_token.strip(): return headers try: @@ -83,10 +75,6 @@ def apply_required_codex_headers( required_names = {name.lower() for name in required} existing = client_kwargs.get("default_headers") or {} client_kwargs["default_headers"] = { - **{ - name: value - for name, value in existing.items() - if str(name).lower() not in required_names - }, + **{name: value for name, value in existing.items() if str(name).lower() not in required_names}, **required, } diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index 909437a6ab..83a40906e1 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -1,11 +1,7 @@ """Codex Responses API adapter. -Pure format-conversion and normalization logic for the OpenAI Responses API -(used by OpenAI Codex, xAI, GitHub Models, and other Responses-compatible endpoints). - -Extracted from run_agent.py to isolate Responses API-specific logic from the -core agent loop. All functions are stateless — they operate on the data passed -in and return transformed results. +Stateless format-conversion and normalization for the OpenAI Responses API +(OpenAI Codex, xAI, GitHub Models and other Responses-compatible endpoints). """ from __future__ import annotations @@ -17,7 +13,7 @@ import re import unicodedata import uuid from types import SimpleNamespace -from typing import Any, Dict, List, NamedTuple, Optional +from typing import Any, Callable, Dict, Iterator, List, NamedTuple, Optional, TypeGuard from agent.message_sanitization import deterministic_call_id from agent.prompt_builder import DEFAULT_AGENT_IDENTITY @@ -26,20 +22,14 @@ logger = logging.getLogger(__name__) def _classify_responses_issuer( - *, - is_xai_responses: bool = False, - is_github_responses: bool = False, - is_codex_backend: bool = False, + *, is_xai_responses: bool = False, is_github_responses: bool = False, is_codex_backend: bool = False, base_url: Optional[str] = None, ) -> str: - """Stable identifier for the Responses endpoint that mints encrypted_content. + """Stable identifier for the endpoint that mints ``reasoning.encrypted_content``. - ``reasoning.encrypted_content`` is sealed to the endpoint that issued it: - replaying a Codex-minted blob against xAI (or vice versa) deterministically - returns HTTP 400 ``invalid_encrypted_content``. Stamping the issuer on - persisted reasoning items and filtering at replay time lets a single - conversation switch models without poisoning history with un-decryptable - reasoning blocks. + Blobs are sealed to their issuer (replaying across endpoints yields HTTP 400 + ``invalid_encrypted_content``); stamping items with the issuer lets replay + drop foreign blobs after a mid-conversation model switch. """ if is_xai_responses: return "xai_responses" @@ -47,44 +37,88 @@ def _classify_responses_issuer( return "github_responses" if is_codex_backend: return "codex_backend" - if base_url: - return f"other:{base_url}" - return "other" + return f"other:{base_url}" if base_url else "other" -# Throttle the per-process cross-issuer skip warning so we don't flood logs -# when a long history contains many stale-issuer reasoning blocks. +# Per-process throttle for the cross-issuer skip warning (long histories can +# carry many stale-issuer reasoning blocks). _CROSS_ISSUER_WARN_EMITTED = False +# Codex/Harmony tool-call serialization leaked into assistant text when the +# model fails to emit a structured ``function_call`` (``to=functions.``, +# optionally prefixed by ``assistant`` or a Harmony channel marker). +_TOOL_CALL_LEAK_PATTERN = re.compile(r"(?:^|[\s>|])to=functions\.[A-Za-z_][\w.]*", re.IGNORECASE) -# Matches Codex/Harmony tool-call serialization that occasionally leaks into -# assistant-message content when the model fails to emit a structured -# ``function_call`` item. Accepts the common forms: -# -# to=functions.exec_command -# assistant to=functions.exec_command -# <|channel|>commentary to=functions.exec_command -# -# ``to=functions.`` is the stable marker — the optional ``assistant`` or -# Harmony channel prefix varies by degeneration mode. Case-insensitive to -# cover lowercase/uppercase ``assistant`` variants. -_TOOL_CALL_LEAK_PATTERN = re.compile( - r"(?:^|[\s>|])to=functions\.[A-Za-z_][\w.]*", - re.IGNORECASE, -) - - -# The ChatGPT Codex backend reserves these Harmony wire tokens. If their -# literal spellings are replayed anywhere in request text, the backend rejects -# the request before inference with ``invalid_prompt: Request blocked.``. -# Category-Cf handling covers persisted sessions from an earlier U+200B weak -# defang; fullwidth bars survive format-character stripping while keeping the -# inspected source legible. -_HARMONY_CONTROL_TOKEN_RE = re.compile( - r"<\|(start|end|channel|message|constrain|return|call)\|>" -) +# The Codex backend rejects requests containing these literal Harmony wire +# tokens (``invalid_prompt: Request blocked.``). Fullwidth bars survive +# format-character stripping while keeping the text legible. +_HARMONY_CONTROL_TOKEN_RE = re.compile(r"<\|(start|end|channel|message|constrain|return|call)\|>") _FULLWIDTH_PIPE = "\uff5c" +_TEXT_PART_TYPES = {"text", "input_text", "output_text"} +_IMAGE_PART_TYPES = {"image_url", "input_image"} +_ASSISTANT_IMAGE_PLACEHOLDER = "[Assistant image omitted during replay]" +_INCOMPLETE_STATUSES = {"queued", "in_progress", "incomplete"} +_RESPONSE_MESSAGE_STATUSES = {"completed", "incomplete", "in_progress"} + +# input[].id longer than this is a non-retryable 400 ("string too long"). +# Codex-issued assistant message ids can run 400+ chars; Hermes-minted +# ``msg_...`` ids stay under the cap and are kept for prefix-cache hits. +# Function names share the same cap (same non-retryable 400). +_MAX_RESPONSES_ITEM_ID_LENGTH = 64 +_VALID_RESPONSES_FN_NAME_RE = re.compile(r"[a-zA-Z0-9_-]{1,64}") + +# Provider-executed built-in tools: declared on ``tools`` by ``type`` alone +# (no name/parameters) and run server-side, reporting via the ``*_call`` +# output items below. Hermes injects xAI's ``web_search`` in +# agent/transports/codex.py; the rest are listed so preflight passes them +# through instead of rejecting them as "unsupported type". +_RESPONSES_BUILTIN_TOOL_TYPES = { + "web_search", "web_search_preview", "file_search", "code_interpreter", + "image_generation", "computer_use_preview", "local_shell", +} + +# Server-side ``*_call`` output items. xAI routinely leaves these at +# ``status="in_progress"`` even when the response is ``completed`` (the search +# finished server-side; the per-item status is never reconciled), so they must +# NOT flip the incomplete verdict — otherwise every server-search turn burns 3 +# fruitless continuation retries. Client-side function/custom tool calls keep +# their own in_progress handling. +_SERVER_SIDE_TOOL_CALL_TYPES = { + "web_search_call", "file_search_call", "code_interpreter_call", + "image_generation_call", "computer_call", "local_shell_call", "mcp_call", +} + + +def _nonblank(value: Any) -> TypeGuard[str]: + return isinstance(value, str) and bool(value.strip()) + + +def _nonempty_str(value: Any) -> TypeGuard[str]: + return isinstance(value, str) and bool(value) + + +def _str_or_empty(value: Any) -> str: + return "" if value is None else str(value) + + +def _lower_or_none(value: Any) -> Optional[str]: + return value.strip().lower() if isinstance(value, str) else None + + +def _field(obj: Any, name: str, default: Any = None) -> Any: + """Read ``name`` from a dict or an attribute-style (SDK/SimpleNamespace) object.""" + return obj.get(name) if isinstance(obj, dict) else getattr(obj, name, default) + + +def _coerce_arguments(arguments: Any) -> str: + """Normalize replayed tool-call arguments to a non-empty JSON string.""" + if isinstance(arguments, dict): + arguments = json.dumps(arguments, ensure_ascii=False) + elif not isinstance(arguments, str): + arguments = str(arguments) + return arguments.strip() or "{}" + def _neutralize_harmony_tokens(text: str) -> str: """Keep Harmony source readable without emitting reserved wire tokens.""" @@ -95,40 +129,26 @@ def _neutralize_harmony_tokens(text: str) -> str: if not any(unicodedata.category(char) == "Cf" for char in text): return _HARMONY_CONTROL_TOKEN_RE.sub(replacement, text) - # U+200B is confirmed to be stripped by the Codex backend before its - # reserved-token check. Treat every Unicode format control equivalently so - # moving the character elsewhere in the token (or swapping in another Cf) - # cannot recreate the same visually hidden form. - visible_chars: List[str] = [] - original_positions: List[int] = [] - for index, char in enumerate(text): - if unicodedata.category(char) == "Cf": - continue - visible_chars.append(char) - original_positions.append(index) - - visible_text = "".join(visible_chars) - matches = list(_HARMONY_CONTROL_TOKEN_RE.finditer(visible_text)) - if not matches: - return text - + # The backend strips Unicode format controls (e.g. U+200B) before its + # reserved-token check, so match on the visible text and rewrite the + # original spans — any Cf-hidden variant is neutralized the same way. + original_positions = [i for i, char in enumerate(text) if unicodedata.category(char) != "Cf"] + visible_text = "".join(text[i] for i in original_positions) result: List[str] = [] - original_cursor = 0 - for match in matches: - original_start = original_positions[match.start()] - original_end = original_positions[match.end() - 1] + 1 - result.append(text[original_cursor:original_start]) - result.append(f"<{_FULLWIDTH_PIPE}{match.group(1)}{_FULLWIDTH_PIPE}>") - original_cursor = original_end - result.append(text[original_cursor:]) + cursor = 0 + for match in _HARMONY_CONTROL_TOKEN_RE.finditer(visible_text): + start, end = original_positions[match.start()], original_positions[match.end() - 1] + 1 + result += [text[cursor:start], f"<{_FULLWIDTH_PIPE}{match.group(1)}{_FULLWIDTH_PIPE}>"] + cursor = end + result.append(text[cursor:]) return "".join(result) def _neutralize_harmony_structure(value: Any) -> Any: - """Neutralize JSON-like values; normalize tuples and reject unsafe keys. + """Neutralize JSON-like values; normalize tuples to lists. - Rewriting an object key could desynchronize a tool schema from the executor - contract, so a reserved token there is rejected explicitly instead. + A reserved token in an object *key* is rejected rather than rewritten — + renaming a key could desynchronize a tool schema from the executor contract. """ if isinstance(value, str): return _neutralize_harmony_tokens(value) @@ -147,114 +167,80 @@ def _neutralize_harmony_structure(value: Any) -> Any: return value -# --------------------------------------------------------------------------- -# Multimodal content helpers -# --------------------------------------------------------------------------- +# --- Multimodal content helpers --------------------------------------------- + +def _iter_content_parts(content: list) -> Iterator[tuple[str, Any]]: + """Yield ``("text", str)`` / ``("image", part)`` for recognized chat parts.""" + for part in content: + if isinstance(part, str): + if part: + yield "text", part + elif isinstance(part, dict): + ptype = str(part.get("type") or "").strip().lower() + if ptype in _TEXT_PART_TYPES and _nonempty_str(part.get("text")): + yield "text", part["text"] + elif ptype in _IMAGE_PART_TYPES: + yield "image", part + + +def _resolve_image_ref(part: Dict[str, Any]) -> tuple[Any, Any]: + """Return ``(url, detail)`` from either ``image_url: str`` or ``{url, detail}``.""" + image_ref = part.get("image_url") + detail = part.get("detail") + if isinstance(image_ref, dict): + return image_ref.get("url"), image_ref.get("detail", detail) + return image_ref, detail + + +def _input_image_part(url: str, detail: Any) -> Dict[str, Any]: + image_part: Dict[str, Any] = {"type": "input_image", "image_url": url} + if _nonblank(detail): + image_part["detail"] = detail.strip() + return image_part + def _chat_content_to_responses_parts(content: Any, *, role: str = "user") -> List[Dict[str, Any]]: """Convert chat-style multimodal content to Responses API input parts. - Input: ``[{"type":"text"|"image_url", ...}]`` (native OpenAI Chat format) - Output: ``[{"type":"input_text"|"output_text"|"input_image", ...}]`` (Responses format) + Text becomes ``input_text`` (user) or ``output_text`` (assistant) — the API + rejects the wrong type per role, so callers MUST pass the right role. + ``input_image`` is only legal on user messages; on assistant messages the + image is replaced by a text marker (an assistant ``input_image`` 400s on + every replay and bricks the session, and the wire cannot carry it anyway). - The ``role`` parameter controls the text content type: - - ``"user"`` (default) → ``"input_text"`` - - ``"assistant"`` → ``"output_text"`` - - The Responses API rejects ``input_text`` inside assistant messages and - ``output_text`` inside user messages, so callers MUST pass the correct - role for the message being converted. - - Image parts are likewise role-restricted: the API only accepts - ``input_image`` on user-role messages. An assistant message carrying - ``input_image`` is rejected with HTTP 400 on every history replay, which - permanently bricks the session (#96816), so image parts are dropped for - the assistant role here (the API cannot carry them in any form, so the - drop is lossless w.r.t. what would survive the wire). - - Returns an empty list when ``content`` is not a list or contains no - recognized parts — callers fall back to the string path. + Returns an empty list when ``content`` is not a list or has no recognized + parts — callers fall back to the string path. """ text_type = "output_text" if role == "assistant" else "input_text" - if not isinstance(content, list): - return [] converted: List[Dict[str, Any]] = [] - for part in content: - if isinstance(part, str): - if part: - converted.append({"type": text_type, "text": part}) - continue - if not isinstance(part, dict): - continue - ptype = str(part.get("type") or "").strip().lower() - if ptype in {"text", "input_text", "output_text"}: - text = part.get("text") - if isinstance(text, str) and text: - converted.append({"type": text_type, "text": text}) - continue - if ptype in {"image_url", "input_image"}: - if role == "assistant": - # Responses output messages cannot carry input_image. Keep a - # text marker so image-only assistant turns still survive in - # replay and later references retain their conversational slot. - converted.append({ - "type": "output_text", - "text": "[Assistant image omitted during replay]", - }) - continue - image_ref = part.get("image_url") - detail = part.get("detail") - if isinstance(image_ref, dict): - url = image_ref.get("url") - detail = image_ref.get("detail", detail) - else: - url = image_ref - if not isinstance(url, str) or not url: - continue - image_part: Dict[str, Any] = {"type": "input_image", "image_url": url} - if isinstance(detail, str) and detail.strip(): - image_part["detail"] = detail.strip() - converted.append(image_part) + for kind, payload in _iter_content_parts(content if isinstance(content, list) else []): + if kind == "text": + converted.append({"type": text_type, "text": payload}) + elif role == "assistant": + converted.append({"type": "output_text", "text": _ASSISTANT_IMAGE_PLACEHOLDER}) + else: + url, detail = _resolve_image_ref(payload) + if _nonempty_str(url): + converted.append(_input_image_part(url, detail)) return converted def _summarize_user_message_for_log(content: Any, *, sep: str = " ") -> str: - """Flatten message content to a plain-text summary. + """Flatten message content to plain text. - Multimodal messages arrive as a list of ``{type:"text"|"image_url", ...}`` - parts from the API server. Several consumers want a plain string: - - - Logging, spinner previews, and trajectory files (the default ``sep=" "``). - - External memory providers, which feed the text to regexes - (``sanitize_context``) and text APIs — a raw list crashes the sync with - ``expected string or bytes-like object, got 'list'`` (use ``sep="\\n"``). - - Text parts are joined with ``sep``; images become a ``[N image(s)]`` marker - so the turn isn't recorded as if the attachment never existed. Returns an - empty string for empty lists and ``str(content)`` for unexpected scalar - types. + Text parts are joined with ``sep`` (``" "`` for logs/spinner/trajectories; + ``"\\n"`` for memory providers that feed the text to regexes and text APIs); + images become a ``[N image(s)]`` marker so the attachment is not erased. + Returns ``""`` for None/empty lists and ``str(content)`` for other scalars. """ if content is None: return "" if isinstance(content, str): return content if isinstance(content, list): - text_bits: List[str] = [] - image_count = 0 - for part in content: - if isinstance(part, str): - if part: - text_bits.append(part) - continue - if not isinstance(part, dict): - continue - ptype = str(part.get("type") or "").strip().lower() - if ptype in {"text", "input_text", "output_text"}: - text = part.get("text") - if isinstance(text, str) and text: - text_bits.append(text) - elif ptype in {"image_url", "input_image"}: - image_count += 1 + parts = list(_iter_content_parts(content)) + text_bits = [payload for kind, payload in parts if kind == "text"] + image_count = len(parts) - len(text_bits) summary = sep.join(text_bits).strip() if image_count: note = f"[{image_count} image{'s' if image_count != 1 else ''}]" @@ -266,37 +252,24 @@ def _summarize_user_message_for_log(content: Any, *, sep: str = " ") -> str: return "" -# --------------------------------------------------------------------------- -# ID helpers -# --------------------------------------------------------------------------- +# --- ID helpers --------------------------------------------------------------- def _deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str: - """Generate a deterministic call_id from tool call content. + """Deterministic call_id (random ids would break the prompt-cache prefix). - Thin wrapper over the single policy owner - ``agent.message_sanitization.deterministic_call_id`` (audit F4) — kept - as a module-level name because run_agent and tests import it from here. - Deterministic IDs prevent cache invalidation — random UUIDs would - make every API call's prefix unique, breaking OpenAI's prompt cache. + Thin wrapper over ``agent.message_sanitization.deterministic_call_id``; + kept here because run_agent and tests import it from this module. """ return deterministic_call_id(fn_name, arguments, index) def _clamp_responses_call_id(call_id: str) -> str: - """Keep a ``call_id`` within the Responses API's 64-char limit (#73492). + """Keep ``call_id`` within the API's 64-char cap. - The codex app-server namespaces MCP tool call ids as - ``codex_mcp_____``; with an ``exec-`` - component the built-in ``hermes-tools`` server already overflows 64 chars, - and the Responses API rejects the whole payload with a non-retryable HTTP - 400 that then replays every turn — permanently bricking the session. - - Sibling defect to #10788 (which clamped ``input[*].id``), applied here to - ``call_id``. The surrogate is a pure, deterministic function of the - original, so the ``function_call`` and its matching ``function_call_output`` - — which carry the same original id — map to the same surrogate and stay - paired without correlating the two items. Short ids pass through unchanged, - preserving prompt-cache prefixes. + The codex app-server namespaces MCP call ids (``codex_mcp_____ + ``) past the cap, and the resulting 400 replays on every turn. The + surrogate is a pure function of the original so a ``function_call`` and its + ``function_call_output`` map to the same value; short ids pass through. """ if len(call_id) <= _MAX_RESPONSES_ITEM_ID_LENGTH: return call_id @@ -304,33 +277,15 @@ def _clamp_responses_call_id(call_id: str) -> str: return f"call_{digest}" -# The Responses API enforces the same 64-char cap on function names as on -# input item ids (_MAX_RESPONSES_ITEM_ID_LENGTH) — names over the cap are -# rejected with the same non-retryable 400 as pattern violations. -_VALID_RESPONSES_FN_NAME_RE = re.compile(r"[a-zA-Z0-9_-]{1,64}") - - def _sanitize_replayed_fn_name(name: str) -> str: - """Coerce a *replayed* function_call name to the Responses API contract. + """Coerce a *replayed* ``function_call.name`` to ``^[a-zA-Z0-9_-]{1,64}$``. - The Responses API requires ``function_call.name`` to match - ``^[a-zA-Z0-9_-]+$`` and rejects the whole request with a non-retryable - HTTP 400 otherwise (issue #31666). A name with invalid characters (dots, - spaces, unicode — e.g. from an earlier model degeneration) stored in - conversation history therefore bricks every subsequent turn of the - session: the 400 replays forever until the user manually starts a new - conversation. - - Invalid characters are replaced with ``_`` (runs collapsed) rather than - stripped, so an all-invalid name degrades to the ``"fn"`` placeholder - instead of an empty string — an empty name would just trade one - non-retryable 400 for a preflight ValueError. Valid names pass through - unchanged, preserving prompt-cache prefixes. - - Apply this ONLY to replayed function_call input items, never to live - tool definitions: tool schema names must match the dispatch registry - exactly. Pairing with function_call_output is by call_id, so renaming - a replayed function_call is safe. + An invalid name stored in history (dots, spaces, unicode from a model + degeneration) otherwise 400s every later turn. Invalid runs collapse to + ``_``; an all-invalid name degrades to ``"fn"`` rather than an empty string + (which would just trade the 400 for a preflight ValueError). Apply ONLY to + replayed items, never to live tool definitions (schema names must match the + dispatch registry); pairing with the output is by call_id, so renaming is safe. """ if not isinstance(name, str): return "fn" @@ -342,72 +297,62 @@ def _sanitize_replayed_fn_name(name: str) -> str: def _canonical_call_id_from_fc(response_item_id: Any) -> Optional[str]: - """Map an ``fc_…`` response-item id to its canonical ``call_``. + """Map an ``fc_…`` item id to its canonical ``call_``. - Both sides of a replayed pair — the assistant ``function_call`` and the - tool ``function_call_output`` — must derive the SAME call_id from an + Both sides of a replayed pair must derive the SAME call_id from an fc_-only stored id, or an oversized pair clamps to two different - surrogates and the API rejects the output as unmatched. Keep every - caller on this single helper. + surrogates. Every caller must go through this helper. """ - if ( - isinstance(response_item_id, str) - and response_item_id.startswith("fc_") - and len(response_item_id) > len("fc_") - ): - return f"call_{response_item_id[len('fc_'):]}" + if isinstance(response_item_id, str) and response_item_id.startswith("fc_") and len(response_item_id) > 3: + return f"call_{response_item_id[3:]}" return None def _split_responses_tool_id(raw_id: Any) -> tuple[Optional[str], Optional[str]]: """Split a stored tool id into (call_id, response_item_id).""" - if not isinstance(raw_id, str): - return None, None - value = raw_id.strip() + value = raw_id.strip() if isinstance(raw_id, str) else "" if not value: return None, None if "|" in value: call_id, response_item_id = value.split("|", 1) - call_id = call_id.strip() or None - response_item_id = response_item_id.strip() or None - return call_id, response_item_id - if value.startswith("fc_"): - return None, value - return value, None + return call_id.strip() or None, response_item_id.strip() or None + return (None, value) if value.startswith("fc_") else (value, None) -def _derive_responses_function_call_id( - call_id: str, - response_item_id: Optional[str] = None, +def _resolve_call_id( + raw_call_id: Any, raw_item_id: Any, fn_name: str, arguments: Any, index: int, *, canonicalize_fc: bool, ) -> str: + """Pick a non-blank call_id: explicit -> embedded in ``call|fc`` id -> (replay only) + canonical ``call_`` -> deterministic hash of name/arguments/index.""" + embedded_call_id, embedded_response_item_id = _split_responses_tool_id(raw_item_id) + call_id = raw_call_id if _nonblank(raw_call_id) else embedded_call_id + if not _nonblank(call_id) and canonicalize_fc: + call_id = _canonical_call_id_from_fc(embedded_response_item_id) + if not _nonblank(call_id): + call_id = _deterministic_call_id(fn_name, arguments, index) + return call_id.strip() + + +def _derive_responses_function_call_id(call_id: str, response_item_id: Optional[str] = None) -> str: """Build a valid Responses `function_call.id` (must start with `fc_`).""" - if isinstance(response_item_id, str): - candidate = response_item_id.strip() - if candidate.startswith("fc_"): - return candidate + if isinstance(response_item_id, str) and response_item_id.strip().startswith("fc_"): + return response_item_id.strip() source = (call_id or "").strip() - if source.startswith("fc_"): - return source - if source.startswith("call_") and len(source) > len("call_"): - return f"fc_{source[len('call_'):]}" - sanitized = re.sub(r"[^A-Za-z0-9_-]", "", source) - if sanitized.startswith("fc_"): - return sanitized - if sanitized.startswith("call_") and len(sanitized) > len("call_"): - return f"fc_{sanitized[len('call_'):]}" + for candidate in (source, sanitized): + if candidate.startswith("fc_"): + return candidate + if candidate.startswith("call_") and len(candidate) > len("call_"): + return f"fc_{candidate[len('call_'):]}" if sanitized: return f"fc_{sanitized[:48]}" seed = source or str(response_item_id or "") or uuid.uuid4().hex - digest = hashlib.sha1(seed.encode("utf-8")).hexdigest()[:24] - return f"fc_{digest}" + return f"fc_{hashlib.sha1(seed.encode('utf-8')).hexdigest()[:24]}" -# --------------------------------------------------------------------------- -# Schema conversion -# --------------------------------------------------------------------------- +# --- Schema conversion -------------------------------------------------------- def _responses_tools(tools: Optional[List[Dict[str, Any]]] = None) -> Optional[List[Dict[str, Any]]]: """Convert chat-completions tool schemas to Responses function-tool schemas.""" @@ -418,58 +363,22 @@ def _responses_tools(tools: Optional[List[Dict[str, Any]]] = None) -> Optional[L for item in tools: fn = item.get("function", {}) if isinstance(item, dict) else {} name = fn.get("name") - if not isinstance(name, str) or not name.strip(): + if not _nonblank(name): continue converted.append({ - "type": "function", - "name": name, - "description": fn.get("description", ""), - "strict": False, + "type": "function", "name": name, "description": fn.get("description", ""), "strict": False, "parameters": fn.get("parameters", {"type": "object", "properties": {}}), }) return converted or None -# Provider-executed built-in tool *declaration* types accepted on the -# Responses ``tools`` array. These are declared by ``type`` alone (no -# client-side name/parameters schema) and run server-side — the provider -# owns the implementation and reports progress via the matching ``*_call`` -# output items. Hermes injects xAI's native ``web_search`` for the xAI -# transport (see agent/transports/codex.py); the rest are listed so the -# preflight validator passes them through rather than rejecting them as -# "unsupported type". Mirrors the ``*_call`` item-type set used in -# _normalize_codex_response. -_RESPONSES_BUILTIN_TOOL_TYPES = { - "web_search", - "web_search_preview", - "file_search", - "code_interpreter", - "image_generation", - "computer_use_preview", - "local_shell", -} - - -# --------------------------------------------------------------------------- -# Message format conversion -# --------------------------------------------------------------------------- - -_RESPONSE_MESSAGE_STATUSES = {"completed", "incomplete", "in_progress"} - -# The Responses API rejects input[].id longer than this with a non-retryable -# HTTP 400 ("string too long"). Codex-issued assistant message ids are -# server-assigned base64 blobs that can run 400+ chars, while Hermes-minted -# ids (msg_...) stay well under this cap and are worth keeping for -# prefix-cache hits. Drop only the oversized ones on replay. -_MAX_RESPONSES_ITEM_ID_LENGTH = 64 - +# --- Message format conversion (chat history -> Responses input) -------------- def _normalize_responses_message_status(value: Any, *, default: str = "completed") -> str: - """Normalize a Responses assistant message status for replay. + """Normalize a replayed assistant message status (completed/incomplete/in_progress). - The API accepts completed/incomplete/in_progress on replayed assistant - output messages. Preserve those exactly (modulo case/hyphen spelling) so - incomplete Codex continuation turns don't get falsely marked completed. + Preserved modulo case/hyphen spelling so incomplete Codex continuation + turns are not falsely marked completed. """ if isinstance(value, str): status = value.strip().lower().replace("-", "_").replace(" ", "_") @@ -478,6 +387,145 @@ def _normalize_responses_message_status(value: Any, *, default: str = "completed return default +def _message_item( + content: List[Dict[str, Any]], *, status: str, item_id: Optional[str] = None, phase: Optional[str] = None, +) -> Dict[str, Any]: + """Assistant ``message`` item; ``id``/``phase`` are added only when non-empty.""" + item: Dict[str, Any] = {"type": "message", "role": "assistant", "status": status, "content": content} + if item_id: + item["id"] = item_id + if phase: + item["phase"] = phase + return item + + +def _assistant_message_item( + raw: Dict[str, Any], content: List[Dict[str, Any]], *, is_github_responses: bool, +) -> Dict[str, Any]: + """Build a replayable assistant ``message`` item from a stored one. + + ``id`` is kept only when short enough (see _MAX_RESPONSES_ITEM_ID_LENGTH) + and never for GitHub Copilot, which binds ids to a backend connection and + 401s on a stale one; ``phase`` is preserved per OpenAI's cache guidance. + """ + item_id, phase = raw.get("id"), raw.get("phase") + keep_id = not is_github_responses and _nonblank(item_id) and len(item_id.strip()) <= _MAX_RESPONSES_ITEM_ID_LENGTH + return _message_item( + content, status=_normalize_responses_message_status(raw.get("status")), + item_id=item_id.strip() if keep_id else None, phase=phase.strip() if _nonblank(phase) else None, + ) + + +def _replay_reasoning_items( + msg: Dict[str, Any], *, seen_item_ids: set, current_issuer_kind: Optional[str], native_compaction_eligible: bool, +) -> List[Dict[str, Any]]: + """Replay persisted encrypted reasoning/compaction items for one assistant turn. + + Skips: duplicate ids; ``compaction`` checkpoints unless THIS request carries + ``context_management`` (a persisted checkpoint outlives the gate and would + otherwise erase pre-checkpoint history on a model that cannot decrypt it); + items stamped by a different issuer (undecryptable → HTTP 400). Unstamped + legacy items pass through. ``id`` is stripped (store=False lookups 404) along + with the Hermes-only ``_issuer_kind`` stamp. + """ + global _CROSS_ISSUER_WARN_EMITTED + codex_reasoning = msg.get("codex_reasoning_items") + if not isinstance(codex_reasoning, list): + return [] + replayed: List[Dict[str, Any]] = [] + for ri in codex_reasoning: + if not (isinstance(ri, dict) and ri.get("encrypted_content")): + continue + item_id = ri.get("id") + if item_id and item_id in seen_item_ids: + continue + if ri.get("type") == "compaction" and not native_compaction_eligible: + continue + item_issuer = ri.get("_issuer_kind") + if current_issuer_kind is not None and item_issuer is not None and item_issuer != current_issuer_kind: + if not _CROSS_ISSUER_WARN_EMITTED: + logger.warning( + "Dropping reasoning item minted by %s while calling %s — encrypted_content is sealed to " + "its issuer. This happens when a session switches model providers mid-conversation.", + item_issuer, current_issuer_kind, + ) + _CROSS_ISSUER_WARN_EMITTED = True + continue + replayed.append({k: v for k, v in ri.items() if k not in ("id", "_issuer_kind")}) + if item_id: + seen_item_ids.add(item_id) + return replayed + + +def _replay_message_items(msg: Dict[str, Any], *, is_github_responses: bool) -> List[Dict[str, Any]]: + """Replay exact assistant message items (id/phase) for prefix-cache hits.""" + codex_message_items = msg.get("codex_message_items") + if not isinstance(codex_message_items, list): + return [] + replayed: List[Dict[str, Any]] = [] + for raw_item in codex_message_items: + if not ( + isinstance(raw_item, dict) and raw_item.get("type") == "message" + and raw_item.get("role") == "assistant" and isinstance(raw_item.get("content"), list) + ): + continue + content = [ + {"type": "output_text", "text": _str_or_empty(part.get("text", ""))} + for part in raw_item["content"] + if isinstance(part, dict) and str(part.get("type") or "").strip() in {"output_text", "text"} + ] + if content: + replayed.append(_assistant_message_item(raw_item, content, is_github_responses=is_github_responses)) + return replayed + + +def _replay_tool_call_items(msg: Dict[str, Any], *, start_index: int) -> List[Dict[str, Any]]: + """Convert an assistant message's ``tool_calls`` into ``function_call`` items.""" + tool_calls = msg.get("tool_calls") + if not isinstance(tool_calls, list): + return [] + replayed: List[Dict[str, Any]] = [] + for tc in tool_calls: + if not isinstance(tc, dict): + continue + fn = tc.get("function", {}) + fn_name = fn.get("name") + if not _nonblank(fn_name): + continue + + call_id = _resolve_call_id( + tc.get("call_id"), tc.get("id"), fn_name, str(fn.get("arguments", "{}")), start_index + len(replayed), + canonicalize_fc=True, + ) + replayed.append({ + "type": "function_call", "call_id": _clamp_responses_call_id(call_id), + "name": _sanitize_replayed_fn_name(fn_name), "arguments": _coerce_arguments(fn.get("arguments", "{}")), + }) + return replayed + + +def _tool_output_item(msg: Dict[str, Any]) -> Optional[Dict[str, Any]]: + """Convert a tool-role message to ``function_call_output`` (None if unpairable).""" + raw_tool_call_id = msg.get("tool_call_id") + call_id, tool_response_item_id = _split_responses_tool_id(raw_tool_call_id) + if not _nonblank(call_id): + # Legacy fc_-only ids canonicalize to the same ``call_`` the + # assistant side synthesizes, so a >64-char pair clamps identically. + call_id = _canonical_call_id_from_fc(tool_response_item_id) + if call_id is None and _nonblank(raw_tool_call_id): + call_id = raw_tool_call_id.strip() + if not _nonblank(call_id): + return None + + # ``output`` may be a string or an ``input_text``/``input_image`` array. + tool_content = msg.get("content") + output_value: Any = ( + (_chat_content_to_responses_parts(tool_content, role="user") or "") + if isinstance(tool_content, list) else str(tool_content or "") + ) + return {"type": "function_call_output", "call_id": _clamp_responses_call_id(call_id), "output": output_value} + + def _chat_messages_to_responses_input( messages: List[Dict[str, Any]], *, @@ -489,348 +537,94 @@ def _chat_messages_to_responses_input( ) -> List[Dict[str, Any]]: """Convert internal chat-style messages to Responses input items. - ``is_xai_responses`` is kept for transport signature compatibility but - no longer suppresses encrypted reasoning replay. Earlier (PR #26644, - May 2026) we believed xAI's OAuth/SuperGrok ``/v1/responses`` surface - rejected replayed ``encrypted_content`` reasoning items minted by - prior turns, and we stripped them. That decision was wrong — xAI - explicitly relies on Hermes threading encrypted reasoning back across - turns for cross-turn coherence (the whole point of their partnership - integration). We now replay encrypted reasoning on every Responses - transport (xAI, native Codex, custom relays) and let xAI tell us - explicitly if a specific surface ever rejects a payload. + ``is_xai_responses``: kept for transport signature compatibility; encrypted + reasoning IS replayed on xAI (it relies on cross-turn reasoning threading). - ``replay_encrypted_reasoning`` is the per-session kill switch. Some - OpenAI-compatible relays accept the request but later reject the - replayed encrypted blob with HTTP 400 ``invalid_encrypted_content``; - when that happens the retry loop calls - ``AIAgent._disable_codex_reasoning_replay`` which both strips cached - items from the conversation history and threads ``replay_enabled=False`` - through this converter so subsequent turns send no reasoning items. + ``replay_encrypted_reasoning``: per-session kill switch. Relays that reject + a replayed blob with HTTP 400 ``invalid_encrypted_content`` trigger + ``AIAgent._disable_codex_reasoning_replay``, which strips cached items and + threads False here so later turns send no reasoning items. - ``is_github_responses`` drops the ``id`` field from replayed - ``codex_message_items`` regardless of length. The Copilot backend - (api.githubcopilot.com/responses) binds these ids to a specific - backend "connection" — credential-pool rotation, a gateway restart, - or routine load-balancer churn between turns all invalidate it — and - rejects a stale id with HTTP 401 "input item ID does not belong to - this connection" even for short ids (see #32716). ``phase``/ - ``status``/``content`` are still replayed; only ``id`` is unsafe to - reuse across a Copilot connection. + ``is_github_responses``: drops ``id`` from replayed message items regardless + of length — Copilot binds ids to a backend connection and rejects a stale + one with HTTP 401 even for short ids. phase/status/content still replay. - ``current_issuer_kind`` enables a per-item cross-issuer guard. The - Responses API's ``encrypted_content`` blob is decryptable only by the - endpoint that minted it — replaying a Codex-issued blob against xAI - (or vice versa) always yields HTTP 400 ``invalid_encrypted_content`` - and breaks every subsequent turn in the same session. When this - argument is provided and a reasoning item carries an ``_issuer_kind`` - stamp from a different endpoint, the item is dropped from the replayed - input. Legacy items without a stamp are still replayed - (backwards-compatible). The two guards compose: - ``replay_encrypted_reasoning=False`` is the session-wide kill switch - (drops ALL replay); ``current_issuer_kind`` is the per-item filter - that runs only when replay is still enabled. + ``current_issuer_kind``: per-item cross-issuer guard (runs only while replay + is enabled); items stamped by another endpoint are dropped, unstamped legacy + items replay. - ``native_compaction_eligible`` mirrors, for THIS request, the decision - made by ``native_compaction.native_compaction_context_management`` — it - is True only when that gate returned a payload, i.e. when the request - actually carries ``context_management``. It controls two things that - must never outlive the gate: replaying ``type: "compaction"`` checkpoint - items, and restructuring the wire around them - (``prune_pre_checkpoint_items``). Checkpoints are persisted in the - ``codex_reasoning_items`` sidecar and survive a mid-session model swap, - a ``compression.enabled: false`` flip, the rejection kill switch and a - resumed session; without this flag a single captured checkpoint would - keep deleting every pre-checkpoint item from every later request, on a - model that cannot decrypt the blob (#85914). Default False = pre-feature - wire, which is also correct for every caller that never sends - ``context_management`` (auxiliary/compression client, ad-hoc - ``convert_messages``). Dropping the checkpoint costs nothing: Hermes' - local history is never truncated by native compaction, so the full - conversation is still on the wire. + ``native_compaction_eligible``: mirrors, for THIS request, whether + ``native_compaction_context_management`` produced a payload. Gates both + replaying ``compaction`` checkpoints and restructuring the wire around them + (``prune_pre_checkpoint_items``). Checkpoints persist in the reasoning + sidecar across model swaps / compression flips / resume; without this gate + one captured checkpoint would delete pre-checkpoint history from every later + request on a model that cannot decrypt it. Default False = pre-feature wire, + correct for callers that never send ``context_management``. Dropping the + checkpoint is lossless: local history is never truncated by native compaction. """ items: List[Dict[str, Any]] = [] - # Parallel to `items`: the raw chat message each converted item came - # from. Pruning needs this to read a canonical summary carrier's - # up-to-date, provenance-tagged content directly — the converted `item` - # can be a lossy shape (stale exact-replay, or a typed - # `function_call_output` wrapper) that no longer carries it (#90976). + # Parallel to ``items``: the raw chat message each item came from. Pruning + # reads a summary carrier's up-to-date, provenance-tagged content from the + # source — the converted item may be a lossy shape (stale exact replay, or + # a typed ``function_call_output`` wrapper) that no longer carries it. item_sources: List[Optional[Dict[str, Any]]] = [] seen_item_ids: set = set() + def emit(new_items: List[Dict[str, Any]], msg: Dict[str, Any]) -> None: + items.extend(new_items) + item_sources.extend([msg] * len(new_items)) + for msg in messages: if not isinstance(msg, dict): continue role = msg.get("role") - if role == "system": - continue - - if role in {"user", "assistant"}: - content = msg.get("content", "") - if isinstance(content, list): - content_parts = _chat_content_to_responses_parts(content, role=role) - text_type = "output_text" if role == "assistant" else "input_text" - content_text = "".join( - p.get("text", "") for p in content_parts if p.get("type") == text_type - ) - else: - content_parts = [] - content_text = str(content) if content is not None else "" - - if role == "assistant": - # Replay encrypted reasoning items from previous turns - # so the API can maintain coherent reasoning chains. - # This applies to every Responses transport including - # xAI — see _chat_messages_to_responses_input docstring - # for the May 2026 reversal of the earlier xAI gate. - codex_reasoning = ( - msg.get("codex_reasoning_items") - if replay_encrypted_reasoning - else None - ) - has_codex_reasoning = False - if isinstance(codex_reasoning, list): - for ri in codex_reasoning: - if isinstance(ri, dict) and ri.get("encrypted_content"): - item_id = ri.get("id") - if item_id and item_id in seen_item_ids: - continue - # Native-compaction gate: a checkpoint is only - # meaningful to the endpoint/model that minted it - # AND only while this request still asks for - # server-side compaction. Once the gate closes - # (model swapped out of the gpt-5.6 family, - # compression disabled, rejection kill switch), - # the persisted checkpoint must not be replayed — - # replaying it is what makes the wire restructure - # below erase pre-checkpoint history forever. - if ( - ri.get("type") == "compaction" - and not native_compaction_eligible - ): - continue - # Cross-issuer guard: drop reasoning blocks that - # were minted by a different Responses endpoint. - # The current endpoint cannot decrypt foreign - # encrypted_content and would reject the whole - # request with HTTP 400 invalid_encrypted_content. - # Unstamped (legacy) items pass through. - item_issuer = ri.get("_issuer_kind") - if ( - current_issuer_kind is not None - and item_issuer is not None - and item_issuer != current_issuer_kind - ): - global _CROSS_ISSUER_WARN_EMITTED - if not _CROSS_ISSUER_WARN_EMITTED: - logger.warning( - "Dropping reasoning item minted by %s while " - "calling %s — encrypted_content is sealed to " - "its issuer. This happens when a session " - "switches model providers mid-conversation.", - item_issuer, current_issuer_kind, - ) - _CROSS_ISSUER_WARN_EMITTED = True - continue - # Strip the "id" field — with store=False the - # Responses API cannot look up items by ID and - # returns 404. The encrypted_content blob is - # self-contained for reasoning chain continuity. - # Also strip the internal "_issuer_kind" stamp; - # it is a Hermes-side metadata key and not part - # of the Responses API schema. - replay_item = { - k: v for k, v in ri.items() - if k not in ("id", "_issuer_kind") - } - items.append(replay_item) - item_sources.append(msg) - if item_id: - seen_item_ids.add(item_id) - has_codex_reasoning = True - - # Replay exact assistant message items (with id/phase) from - # previous turns so the API can maintain prefix-cache hits. - # OpenAI docs: "preserve and resend phase on all assistant - # messages — dropping it can degrade performance." - codex_message_items = msg.get("codex_message_items") - replayed_message_items = 0 - if isinstance(codex_message_items, list): - for raw_item in codex_message_items: - if not isinstance(raw_item, dict): - continue - if raw_item.get("type") != "message" or raw_item.get("role") != "assistant": - continue - raw_content_parts = raw_item.get("content") - if not isinstance(raw_content_parts, list): - continue - - normalized_content_parts = [] - for part in raw_content_parts: - if not isinstance(part, dict): - continue - part_type = str(part.get("type") or "").strip() - if part_type not in {"output_text", "text"}: - continue - text = part.get("text", "") - if text is None: - text = "" - if not isinstance(text, str): - text = str(text) - normalized_content_parts.append({"type": "output_text", "text": text}) - - if not normalized_content_parts: - continue - - replay_item = { - "type": "message", - "role": "assistant", - "status": _normalize_responses_message_status(raw_item.get("status")), - "content": normalized_content_parts, - } - item_id = raw_item.get("id") - if ( - not is_github_responses - and isinstance(item_id, str) - and item_id.strip() - ): - stripped_id = item_id.strip() - if len(stripped_id) <= _MAX_RESPONSES_ITEM_ID_LENGTH: - replay_item["id"] = stripped_id - phase = raw_item.get("phase") - if isinstance(phase, str) and phase.strip(): - replay_item["phase"] = phase.strip() - items.append(replay_item) - item_sources.append(msg) - replayed_message_items += 1 - - if replayed_message_items > 0: - pass - elif content_parts: - items.append({"role": "assistant", "content": content_parts}) - item_sources.append(msg) - elif content_text.strip(): - items.append({"role": "assistant", "content": content_text}) - item_sources.append(msg) - elif has_codex_reasoning: - # The Responses API requires a following item after each - # reasoning item (otherwise: missing_following_item error). - # When the assistant produced only reasoning with no visible - # content, emit an empty assistant message as the required - # following item. - items.append({"role": "assistant", "content": ""}) - item_sources.append(msg) - - tool_calls = msg.get("tool_calls") - if isinstance(tool_calls, list): - for tc in tool_calls: - if not isinstance(tc, dict): - continue - fn = tc.get("function", {}) - fn_name = fn.get("name") - if not isinstance(fn_name, str) or not fn_name.strip(): - continue - - embedded_call_id, embedded_response_item_id = _split_responses_tool_id( - tc.get("id") - ) - call_id = tc.get("call_id") - if not isinstance(call_id, str) or not call_id.strip(): - call_id = embedded_call_id - if not isinstance(call_id, str) or not call_id.strip(): - call_id = _canonical_call_id_from_fc(embedded_response_item_id) - if call_id is None: - _raw_args = str(fn.get("arguments", "{}")) - call_id = _deterministic_call_id(fn_name, _raw_args, len(items)) - call_id = call_id.strip() - - arguments = fn.get("arguments", "{}") - if isinstance(arguments, dict): - arguments = json.dumps(arguments, ensure_ascii=False) - elif not isinstance(arguments, str): - arguments = str(arguments) - arguments = arguments.strip() or "{}" - - items.append({ - "type": "function_call", - "call_id": _clamp_responses_call_id(call_id), - "name": _sanitize_replayed_fn_name(fn_name), - "arguments": arguments, - }) - item_sources.append(msg) - continue - - # Non-assistant (user) role: emit multimodal parts when present, - # otherwise fall back to the text payload. - if content_parts: - items.append({"role": role, "content": content_parts}) - else: - items.append({"role": role, "content": content_text}) - item_sources.append(msg) - continue - if role == "tool": - raw_tool_call_id = msg.get("tool_call_id") - call_id, tool_response_item_id = _split_responses_tool_id(raw_tool_call_id) - if not isinstance(call_id, str) or not call_id.strip(): - # Legacy fc_-only stored ids: canonicalize to the same - # ``call_`` the assistant branch synthesizes above, so - # a >64-char pair clamps to the SAME surrogate on both sides. - call_id = _canonical_call_id_from_fc(tool_response_item_id) - if call_id is None and isinstance(raw_tool_call_id, str) and raw_tool_call_id.strip(): - call_id = raw_tool_call_id.strip() - if not isinstance(call_id, str) or not call_id.strip(): - continue + tool_item = _tool_output_item(msg) + if tool_item is not None: + emit([tool_item], msg) + continue + if role not in {"user", "assistant"}: + continue - # Multimodal tool result: convert OpenAI-style content list into - # Responses ``function_call_output.output`` array. The Responses - # API accepts ``output`` as either a string or an array of - # ``input_text``/``input_image`` items. See - # https://developers.openai.com/api/reference/python/resources/responses/. - tool_content = msg.get("content") - output_value: Any - if isinstance(tool_content, list): - converted = _chat_content_to_responses_parts( - tool_content, role="user", - ) - if converted: - output_value = converted - else: - output_value = "" - else: - output_value = str(tool_content or "") + content = msg.get("content", "") + content_parts = _chat_content_to_responses_parts(content, role=role) # [] unless a list + if isinstance(content, list): + text_type = "output_text" if role == "assistant" else "input_text" + content_text = "".join(p["text"] for p in content_parts if p["type"] == text_type) + else: + content_text = "" if content is None else str(content) - items.append({ - "type": "function_call_output", - "call_id": _clamp_responses_call_id(call_id), - "output": output_value, - }) - item_sources.append(msg) + if role == "user": + emit([{"role": role, "content": content_parts or content_text}], msg) + continue - # Native server-side compaction: when a replayed checkpoint is present, - # restructure the wire around it. The server renders nothing placed - # before a compaction item (live-verified Aug 2026), so pre-checkpoint - # history is dead upload weight and — worse — the user's plaintext asks, - # and any local-compression summary already merged into that history, - # silently vanish from the model's view. Keep the newest checkpoint - # first, retain pre-checkpoint USER messages and compression-SUMMARY - # messages (whole, never byte-sliced) verbatim within a token budget - # each (Codex CLI parity for the user side), and leave the - # post-checkpoint tail untouched. Gated on the CURRENT request's native - # eligibility, not merely on the presence of a checkpoint: a persisted - # checkpoint outlives the gate, and pruning for a request that carries no - # ``context_management`` deletes history the server never compacted. - # - # ``item_sources`` (parallel to ``items``) carries the raw chat message - # each converted item came from. A canonical summary carrier's content - # can be lost or gone stale by the time it becomes a Responses item — a - # merge-into-tail tool-result carrier becomes a typed - # ``function_call_output`` (no ``content``/``role`` at all), and a - # merge-into-tail assistant carrier can be shadowed by a stale exact - # ``codex_message_items`` replay from before the merge rewrote its - # content. Pruning reads the source message's own up-to-date, - # provenance-tagged content directly instead of trying to recover it - # from whatever shape the conversion produced (#90976). + reasoning_items = [] if not replay_encrypted_reasoning else _replay_reasoning_items( + msg, seen_item_ids=seen_item_ids, current_issuer_kind=current_issuer_kind, + native_compaction_eligible=native_compaction_eligible, + ) + emit(reasoning_items, msg) + message_items = _replay_message_items(msg, is_github_responses=is_github_responses) + emit(message_items, msg) + + if not message_items: + if content_parts: + emit([{"role": "assistant", "content": content_parts}], msg) + elif content_text.strip(): + emit([{"role": "assistant", "content": content_text}], msg) + elif reasoning_items: + # Every reasoning item needs a following item (else + # missing_following_item); emit an empty message. + emit([{"role": "assistant", "content": ""}], msg) + + emit(_replay_tool_call_items(msg, start_index=len(items)), msg) + + # Native server-side compaction: the server renders nothing placed before a + # compaction item, so pre-checkpoint history is dead upload weight and the + # user's plaintext asks / merged local summaries silently vanish. Keep the + # newest checkpoint first, retain pre-checkpoint USER and compression-SUMMARY + # messages verbatim within a token budget, leave the tail untouched. Gated + # on THIS request's eligibility, not merely on a checkpoint being present. if not native_compaction_eligible: return items @@ -842,11 +636,9 @@ def _chat_messages_to_responses_input( class ResponsesRouteFlags(NamedTuple): """Which special Responses-API route an agent is talking to. - Single owner of the codex/xai/github route predicates. Every site that - needs these flags (request kwargs build, preflight estimation, silent- - reject hints) must call :func:`classify_responses_route` instead of - re-implementing the string comparisons inline — inline copies drift - (backend-identity class: #22548/#70893/#59561/#72468). + Single owner of the codex/xai/github predicates: every site (request + kwargs, preflight estimation, silent-reject hints) must call + :func:`classify_responses_route` — inline string comparisons drift. """ is_codex_backend: bool @@ -857,9 +649,8 @@ class ResponsesRouteFlags(NamedTuple): def classify_responses_route(agent: Any) -> ResponsesRouteFlags: """Classify the agent's Responses route from provider + base URL. - Host checks are exact-host-or-subdomain (``base_url_hostname`` - semantics), never substring matching — ``https://evil.com/models.github.ai`` - must not classify as a GitHub route. + Host checks are exact-host-or-subdomain, never substring — + ``https://evil.com/models.github.ai`` must not classify as GitHub. """ from utils import base_url_hostname @@ -873,15 +664,12 @@ def classify_responses_route(agent: Any) -> ResponsesRouteFlags: def _host_is(domain: str) -> bool: return hostname == domain or hostname.endswith("." + domain) - is_codex_backend = provider == "openai-codex" or ( - _host_is("chatgpt.com") and "/backend-api/codex" in lower - ) - is_github_responses = _host_is("models.github.ai") or _host_is("githubcopilot.com") - is_xai_responses = provider in {"xai", "xai-oauth"} or hostname == "api.x.ai" return ResponsesRouteFlags( - is_codex_backend=is_codex_backend, - is_xai_responses=is_xai_responses, - is_github_responses=is_github_responses, + is_codex_backend=( + provider == "openai-codex" or (_host_is("chatgpt.com") and "/backend-api/codex" in lower) + ), + is_xai_responses=provider in {"xai", "xai-oauth"} or hostname == "api.x.ai", + is_github_responses=_host_is("models.github.ai") or _host_is("githubcopilot.com"), ) @@ -894,54 +682,37 @@ def estimate_native_responses_preflight_tokens( ) -> Optional[int]: """Estimate tokens for the checkpoint-pruned Responses payload. - Automatic preflight previously counted the full durable transcript. - On a natively compacted Codex session that overstates the wire by - several times and fires local compression against history the main - request will never send (#96155). - - Returns None when native compaction is not proven eligible for this - request, or when conversion fails — the caller must then use the - generic durable-transcript estimate (conservative). + Counting the full durable transcript overstates a natively compacted + session several times over and fires local compression against history + the request will never send. Returns None when native compaction is not + proven eligible or conversion fails — caller falls back to the generic + (conservative) estimate. """ - if getattr(agent, "api_mode", None) != "codex_responses": - return None - if not isinstance(messages, list): + if getattr(agent, "api_mode", None) != "codex_responses" or not isinstance(messages, list): return None is_codex_backend, is_xai_responses, is_github_responses = classify_responses_route(agent) from agent.native_compaction import native_compaction_context_management - context_management = native_compaction_context_management( - agent, - is_codex_backend=is_codex_backend, - is_xai_responses=is_xai_responses, + if not native_compaction_context_management( + agent, is_codex_backend=is_codex_backend, is_xai_responses=is_xai_responses, is_github_responses=is_github_responses, - ) - if not context_management: + ): return None try: items = _chat_messages_to_responses_input( - messages, - is_xai_responses=is_xai_responses, - is_github_responses=is_github_responses, - replay_encrypted_reasoning=bool( - getattr(agent, "_codex_reasoning_replay_enabled", True) - ), + messages, is_xai_responses=is_xai_responses, is_github_responses=is_github_responses, + replay_encrypted_reasoning=bool(getattr(agent, "_codex_reasoning_replay_enabled", True)), current_issuer_kind=_classify_responses_issuer( - is_xai_responses=is_xai_responses, - is_github_responses=is_github_responses, - is_codex_backend=is_codex_backend, - base_url=getattr(agent, "base_url", None), + is_xai_responses=is_xai_responses, is_github_responses=is_github_responses, + is_codex_backend=is_codex_backend, base_url=getattr(agent, "base_url", None), ), native_compaction_eligible=True, ) except Exception: - logger.debug( - "native Responses preflight conversion failed; falling back to generic estimate", - exc_info=True, - ) + logger.debug("native Responses preflight conversion failed; falling back to generic estimate", exc_info=True) return None if not isinstance(items, list): @@ -949,269 +720,228 @@ def estimate_native_responses_preflight_tokens( from agent.model_metadata import estimate_request_tokens_rough - return estimate_request_tokens_rough( - items, - system_prompt=system_prompt or "", - tools=tools, - ) + return estimate_request_tokens_rough(items, system_prompt=system_prompt or "", tools=tools) -# --------------------------------------------------------------------------- -# Input preflight / validation -# --------------------------------------------------------------------------- +# --- Input preflight / validation -------------------------------------------- + +class _PreflightCtx(NamedTuple): + sanitize_text: Callable[[str], str] + sanitize_harmony_tokens: bool + is_github_responses: bool + seen_ids: set + + +def _preflight_function_call(item: Dict[str, Any], idx: int, ctx: _PreflightCtx) -> Dict[str, Any]: + call_id = item.get("call_id") + name = item.get("name") + if not _nonblank(call_id): + raise ValueError(f"Codex Responses input[{idx}] function_call is missing call_id.") + if not _nonblank(name): + raise ValueError(f"Codex Responses input[{idx}] function_call is missing name.") + return { + "type": "function_call", "call_id": call_id.strip(), "name": _sanitize_replayed_fn_name(name), + "arguments": ctx.sanitize_text(_coerce_arguments(item.get("arguments", "{}"))), + } + + +def _preflight_function_call_output(item: Dict[str, Any], idx: int, ctx: _PreflightCtx) -> Dict[str, Any]: + call_id = item.get("call_id") + if not _nonblank(call_id): + raise ValueError(f"Codex Responses input[{idx}] function_call_output is missing call_id.") + output = item.get("output", "") + if isinstance(output, list): + # Multimodal tool result: keep recognised input_text/input_image + # parts, drop anything else to avoid a 4xx. + cleaned: List[Dict[str, Any]] = [] + for part in output: + ptype = part.get("type") if isinstance(part, dict) else None + if ptype == "input_text" and _nonempty_str(part.get("text")): + cleaned.append({"type": "input_text", "text": ctx.sanitize_text(part["text"])}) + elif ptype == "input_image" and _nonempty_str(part.get("image_url")): + cleaned.append(_input_image_part(part["image_url"], part.get("detail"))) + output_value: Any = cleaned or "" + else: + output_value = ctx.sanitize_text(_str_or_empty(output)) + return {"type": "function_call_output", "call_id": call_id.strip(), "output": output_value} + + +def _preflight_reasoning(item: Dict[str, Any], idx: int, ctx: _PreflightCtx) -> Optional[Dict[str, Any]]: + encrypted = item.get("encrypted_content") + if not _nonempty_str(encrypted): + return None + # ``id`` is used only for local dedup and NOT forwarded: with store=False + # the API resolves ids server-side and 404s. + item_id = item.get("id") + if _nonempty_str(item_id): + if item_id in ctx.seen_ids: + return None + ctx.seen_ids.add(item_id) + summary = item.get("summary") + if not isinstance(summary, list): + summary = [] + return { + "type": "reasoning", "encrypted_content": encrypted, + "summary": _neutralize_harmony_structure(summary) if ctx.sanitize_harmony_tokens else summary, + } + + +def _preflight_compaction(item: Dict[str, Any], idx: int, ctx: _PreflightCtx) -> Optional[Dict[str, Any]]: + # Opaque, issuer-sealed checkpoint; forward only the fields the API defines. + encrypted = item.get("encrypted_content") + return {"type": "compaction", "encrypted_content": encrypted} if _nonempty_str(encrypted) else None + + +def _preflight_message(item: Dict[str, Any], idx: int, ctx: _PreflightCtx) -> Dict[str, Any]: + if item.get("role") != "assistant": + raise ValueError(f"Codex Responses input[{idx}] message items must have role='assistant'.") + content = item.get("content") + if not isinstance(content, list): + raise ValueError(f"Codex Responses input[{idx}] message item must have content list.") + normalized_content = [] + for part_idx, part in enumerate(content): + if not isinstance(part, dict): + raise ValueError(f"Codex Responses input[{idx}] message content[{part_idx}] must be an object.") + part_type = part.get("type") + if part_type not in {"output_text", "text"}: + raise ValueError( + f"Codex Responses input[{idx}] message content[{part_idx}] has unsupported type {part_type!r}." + ) + text = _str_or_empty(part.get("text", "")) + normalized_content.append({"type": "output_text", "text": ctx.sanitize_text(text)}) + if not normalized_content: + raise ValueError(f"Codex Responses input[{idx}] message item must contain at least one text part.") + return _assistant_message_item(item, normalized_content, is_github_responses=ctx.is_github_responses) + + +def _preflight_role_message(item: Dict[str, Any], idx: int, role: str, ctx: _PreflightCtx) -> Dict[str, Any]: + content = item.get("content", "") + if not isinstance(content, list): + return {"role": role, "content": ctx.sanitize_text(_str_or_empty(content))} + + # Parts are already Responses-shaped; validate and re-type text for the + # role (``output_text`` for assistant, ``input_text`` for user). Unlike + # history conversion, empty text / empty image urls are kept, not dropped. + text_type = "output_text" if role == "assistant" else "input_text" + validated: List[Dict[str, Any]] = [] + for part_idx, part in enumerate(content): + if isinstance(part, str): + if part: + validated.append({"type": text_type, "text": ctx.sanitize_text(part)}) + continue + if not isinstance(part, dict): + raise ValueError(f"Codex Responses input[{idx}].content[{part_idx}] must be an object or string.") + ptype = str(part.get("type") or "").strip().lower() + if ptype in _TEXT_PART_TYPES: + text = part.get("text", "") + text = text if isinstance(text, str) else str(text or "") + validated.append({"type": text_type, "text": ctx.sanitize_text(text)}) + elif ptype in _IMAGE_PART_TYPES: + if role == "assistant": + # Same output-message invariant as normal history replay. + validated.append({"type": "output_text", "text": _ASSISTANT_IMAGE_PLACEHOLDER}) + else: + url, detail = _resolve_image_ref(part) + validated.append(_input_image_part(url if isinstance(url, str) else str(url or ""), detail)) + else: + raise ValueError( + f"Codex Responses input[{idx}].content[{part_idx}] has unsupported type {part.get('type')!r}." + ) + return {"role": role, "content": validated} + + +_PREFLIGHT_ITEM_HANDLERS: Dict[str, Callable[..., Optional[Dict[str, Any]]]] = { + "function_call": _preflight_function_call, + "function_call_output": _preflight_function_call_output, + "reasoning": _preflight_reasoning, + "compaction": _preflight_compaction, + "message": _preflight_message, +} + def _preflight_codex_input_items( - raw_items: Any, - *, - is_github_responses: bool = False, - sanitize_harmony_tokens: bool = False, + raw_items: Any, *, is_github_responses: bool = False, sanitize_harmony_tokens: bool = False, ) -> List[Dict[str, Any]]: if not isinstance(raw_items, list): raise ValueError("Codex Responses input must be a list of input items.") - sanitize_text = ( - _neutralize_harmony_tokens - if sanitize_harmony_tokens - else lambda text: text + ctx = _PreflightCtx( + sanitize_text=_neutralize_harmony_tokens if sanitize_harmony_tokens else (lambda text: text), + sanitize_harmony_tokens=sanitize_harmony_tokens, + is_github_responses=is_github_responses, + seen_ids=set(), ) normalized: List[Dict[str, Any]] = [] - seen_ids: set = set() for idx, item in enumerate(raw_items): if not isinstance(item, dict): raise ValueError(f"Codex Responses input[{idx}] must be an object.") item_type = item.get("type") - if item_type == "function_call": - call_id = item.get("call_id") - name = item.get("name") - if not isinstance(call_id, str) or not call_id.strip(): - raise ValueError(f"Codex Responses input[{idx}] function_call is missing call_id.") - if not isinstance(name, str) or not name.strip(): - raise ValueError(f"Codex Responses input[{idx}] function_call is missing name.") - - arguments = item.get("arguments", "{}") - if isinstance(arguments, dict): - arguments = json.dumps(arguments, ensure_ascii=False) - elif not isinstance(arguments, str): - arguments = str(arguments) - arguments = sanitize_text(arguments.strip() or "{}") - - normalized.append( - { - "type": "function_call", - "call_id": call_id.strip(), - "name": _sanitize_replayed_fn_name(name), - "arguments": arguments, - } - ) - continue - - if item_type == "function_call_output": - call_id = item.get("call_id") - if not isinstance(call_id, str) or not call_id.strip(): - raise ValueError(f"Codex Responses input[{idx}] function_call_output is missing call_id.") - output = item.get("output", "") - if output is None: - output = "" - # Output may be a string OR an array of structured content - # items (input_text / input_image) for multimodal tool results. - # Both shapes are accepted by the Responses API. We preserve - # the array form when present. - if isinstance(output, list): - # Validate each item is a recognised content shape; drop - # anything else to avoid 4xx from the API. - cleaned: List[Dict[str, Any]] = [] - for part in output: - if not isinstance(part, dict): - continue - ptype = part.get("type") - if ptype == "input_text": - text = part.get("text") - if isinstance(text, str) and text: - cleaned.append({"type": "input_text", "text": sanitize_text(text)}) - elif ptype == "input_image": - url = part.get("image_url") - if isinstance(url, str) and url: - entry: Dict[str, Any] = {"type": "input_image", "image_url": url} - detail = part.get("detail") - if isinstance(detail, str) and detail.strip(): - entry["detail"] = detail.strip() - cleaned.append(entry) - normalized.append( - { - "type": "function_call_output", - "call_id": call_id.strip(), - "output": cleaned if cleaned else "", - } - ) - continue - if not isinstance(output, str): - output = str(output) - - normalized.append( - { - "type": "function_call_output", - "call_id": call_id.strip(), - "output": sanitize_text(output), - } - ) - continue - - if item_type == "reasoning": - encrypted = item.get("encrypted_content") - if isinstance(encrypted, str) and encrypted: - item_id = item.get("id") - if isinstance(item_id, str) and item_id: - if item_id in seen_ids: - continue - seen_ids.add(item_id) - reasoning_item: Dict[str, Any] = { - "type": "reasoning", - "encrypted_content": encrypted, - } - # Do NOT include the "id" in the outgoing item — with - # store=False (our default) the API tries to resolve the - # id server-side and returns 404. The id is still used - # above for local deduplication via seen_ids. - summary = item.get("summary") - if isinstance(summary, list): - reasoning_item["summary"] = ( - _neutralize_harmony_structure(summary) - if sanitize_harmony_tokens - else summary - ) - else: - reasoning_item["summary"] = [] - normalized.append(reasoning_item) - continue - - if item_type == "compaction": - # Replayed native server-side compaction checkpoint (gpt-5.6, - # direct OpenAI/Codex routes). Opaque, issuer-sealed; forward - # only the fields the API defines. - encrypted = item.get("encrypted_content") - if isinstance(encrypted, str) and encrypted: - normalized.append( - {"type": "compaction", "encrypted_content": encrypted} - ) - continue - - if item_type == "message": + handler = _PREFLIGHT_ITEM_HANDLERS.get(item_type) if isinstance(item_type, str) else None + if handler is not None: + normalized_item = handler(item, idx, ctx) + else: + # Untyped role messages (user/assistant) are the only other legal shape. role = item.get("role") - if role != "assistant": - raise ValueError(f"Codex Responses input[{idx}] message items must have role='assistant'.") - content = item.get("content") - if not isinstance(content, list): - raise ValueError(f"Codex Responses input[{idx}] message item must have content list.") - normalized_content = [] - for part_idx, part in enumerate(content): - if not isinstance(part, dict): - raise ValueError( - f"Codex Responses input[{idx}] message content[{part_idx}] must be an object." - ) - part_type = part.get("type") - if part_type not in {"output_text", "text"}: - raise ValueError( - f"Codex Responses input[{idx}] message content[{part_idx}] has unsupported type {part_type!r}." - ) - text = part.get("text", "") - if text is None: - text = "" - if not isinstance(text, str): - text = str(text) - normalized_content.append({"type": "output_text", "text": sanitize_text(text)}) - if not normalized_content: - raise ValueError(f"Codex Responses input[{idx}] message item must contain at least one text part.") - normalized_item: Dict[str, Any] = { - "type": "message", - "role": "assistant", - "status": _normalize_responses_message_status(item.get("status")), - "content": normalized_content, - } - item_id = item.get("id") - if ( - not is_github_responses - and isinstance(item_id, str) - and item_id.strip() - ): - stripped_id = item_id.strip() - if len(stripped_id) <= _MAX_RESPONSES_ITEM_ID_LENGTH: - normalized_item["id"] = stripped_id - phase = item.get("phase") - if isinstance(phase, str) and phase.strip(): - normalized_item["phase"] = phase.strip() + if role not in {"user", "assistant"}: + raise ValueError( + f"Codex Responses input[{idx}] has unsupported item shape (type={item_type!r}, role={role!r})." + ) + normalized_item = _preflight_role_message(item, idx, role, ctx) + if normalized_item is not None: normalized.append(normalized_item) - continue - - role = item.get("role") - if role in {"user", "assistant"}: - content = item.get("content", "") - if content is None: - content = "" - if isinstance(content, list): - # Multimodal content from ``_chat_messages_to_responses_input`` - # is already in Responses format (``input_text`` / ``output_text`` - # / ``input_image``). Validate each part and pass through. - # Use the correct text type for the role — ``output_text`` for - # assistant messages, ``input_text`` for user messages. - text_type = "output_text" if role == "assistant" else "input_text" - validated: List[Dict[str, Any]] = [] - for part_idx, part in enumerate(content): - if isinstance(part, str): - if part: - validated.append({"type": text_type, "text": sanitize_text(part)}) - continue - if not isinstance(part, dict): - raise ValueError( - f"Codex Responses input[{idx}].content[{part_idx}] must be an object or string." - ) - ptype = str(part.get("type") or "").strip().lower() - if ptype in {"input_text", "text", "output_text"}: - text = part.get("text", "") - if not isinstance(text, str): - text = str(text or "") - validated.append({"type": text_type, "text": sanitize_text(text)}) - elif ptype in {"input_image", "image_url"}: - if role == "assistant": - # Enforce the same output-message invariant for - # raw request overrides as for normal history. - validated.append({ - "type": "output_text", - "text": "[Assistant image omitted during replay]", - }) - continue - image_ref = part.get("image_url", "") - detail = part.get("detail") - if isinstance(image_ref, dict): - url = image_ref.get("url", "") - detail = image_ref.get("detail", detail) - else: - url = image_ref - if not isinstance(url, str): - url = str(url or "") - image_part: Dict[str, Any] = {"type": "input_image", "image_url": url} - if isinstance(detail, str) and detail.strip(): - image_part["detail"] = detail.strip() - validated.append(image_part) - else: - raise ValueError( - f"Codex Responses input[{idx}].content[{part_idx}] has unsupported type {part.get('type')!r}." - ) - normalized.append({"role": role, "content": validated}) - continue - if not isinstance(content, str): - content = str(content) - - normalized.append({"role": role, "content": sanitize_text(content)}) - continue - - raise ValueError( - f"Codex Responses input[{idx}] has unsupported item shape (type={item_type!r}, role={role!r})." - ) return normalized +def _preflight_tool(tool: Any, idx: int) -> Dict[str, Any]: + if not isinstance(tool, dict): + raise ValueError(f"Codex Responses tools[{idx}] must be an object.") + tool_type = tool.get("type") + # Provider-executed built-ins carry no name/parameters; pass them through + # verbatim rather than rejecting them below. + if tool_type in _RESPONSES_BUILTIN_TOOL_TYPES: + return dict(tool) + if tool_type != "function": + raise ValueError(f"Codex Responses tools[{idx}] has unsupported type {tool.get('type')!r}.") + + name = tool.get("name") + parameters = tool.get("parameters") + if not _nonblank(name): + raise ValueError(f"Codex Responses tools[{idx}] is missing a valid name.") + if not isinstance(parameters, dict): + raise ValueError(f"Codex Responses tools[{idx}] is missing valid parameters.") + return { + "type": "function", "name": name.strip(), "description": _str_or_empty(tool.get("description", "")), + "strict": bool(tool.get("strict", False)), "parameters": parameters, + } + + +# Optional scalar request fields, in wire order: (key, accept(value), coerce). +# Values failing ``accept`` are silently dropped (never an error). +_PREFLIGHT_OPTIONAL_FIELDS: tuple[tuple[str, Callable[[Any], bool], Optional[Callable[[Any], Any]]], ...] = ( + ("reasoning", lambda v: isinstance(v, dict), None), + ("include", lambda v: isinstance(v, list), None), + ("service_tier", _nonblank, str.strip), + ("max_output_tokens", lambda v: isinstance(v, (int, float)) and v > 0, int), + ("timeout", lambda v: isinstance(v, (int, float)) and not isinstance(v, bool) and 0 < v < float("inf"), float), + ("temperature", lambda v: isinstance(v, (int, float)), float), + # Cache routing/retention and tool-dispatch hints pass through as-is. + ("tool_choice", lambda v: v is not None, None), + ("parallel_tool_calls", lambda v: v is not None, None), + ("prompt_cache_key", lambda v: v is not None, None), + ("prompt_cache_retention", lambda v: v is not None, None), + # Native compaction directive; eligibility is resolved upstream in + # agent/native_compaction.py — preflight only preserves the shape. + ("context_management", lambda v: isinstance(v, list) and bool(v), None), +) + +_PREFLIGHT_ALLOWED_KEYS = { + "model", "instructions", "input", "tools", "store", "extra_headers", "extra_body", + *(key for key, _, _ in _PREFLIGHT_OPTIONAL_FIELDS), +} + + def _preflight_codex_api_kwargs( api_kwargs: Any, *, @@ -1222,163 +952,53 @@ def _preflight_codex_api_kwargs( if not isinstance(api_kwargs, dict): raise ValueError("Codex Responses request must be a dict.") - required = {"model", "instructions", "input"} - missing = [key for key in required if key not in api_kwargs] + missing = [key for key in ("model", "instructions", "input") if key not in api_kwargs] if missing: raise ValueError(f"Codex Responses request missing required field(s): {', '.join(sorted(missing))}.") model = api_kwargs.get("model") - if not isinstance(model, str) or not model.strip(): + if not _nonblank(model): raise ValueError("Codex Responses request 'model' must be a non-empty string.") - model = model.strip() - instructions = api_kwargs.get("instructions") - if instructions is None: - instructions = "" - if not isinstance(instructions, str): - instructions = str(instructions) - instructions = instructions.strip() or DEFAULT_AGENT_IDENTITY + instructions = _str_or_empty(api_kwargs.get("instructions")).strip() or DEFAULT_AGENT_IDENTITY if sanitize_harmony_tokens: instructions = _neutralize_harmony_tokens(instructions) - normalized_input = _preflight_codex_input_items( - api_kwargs.get("input"), - is_github_responses=is_github_responses, - sanitize_harmony_tokens=sanitize_harmony_tokens, - ) + normalized: Dict[str, Any] = { + "model": model.strip(), + "instructions": instructions, + "input": _preflight_codex_input_items( + api_kwargs.get("input"), + is_github_responses=is_github_responses, + sanitize_harmony_tokens=sanitize_harmony_tokens, + ), + "store": False, + } tools = api_kwargs.get("tools") - normalized_tools = None if tools is not None: if not isinstance(tools, list): raise ValueError("Codex Responses request 'tools' must be a list when provided.") - normalized_tools = [] - for idx, tool in enumerate(tools): - if not isinstance(tool, dict): - raise ValueError(f"Codex Responses tools[{idx}] must be an object.") - - tool_type = tool.get("type") - - # Provider-executed built-in tools (xAI native web_search, code - # interpreter, etc.) are declared by ``type`` alone and carry no - # ``name``/``parameters`` schema — the provider owns the - # implementation. Pass them through verbatim instead of forcing - # them through the function-tool validation below (which would - # otherwise reject them with "unsupported type"). See - # agent/transports/codex.py for where xAI's native web_search is - # injected. - if tool_type in _RESPONSES_BUILTIN_TOOL_TYPES: - normalized_tools.append(dict(tool)) - continue - - if tool_type != "function": - raise ValueError(f"Codex Responses tools[{idx}] has unsupported type {tool.get('type')!r}.") - - name = tool.get("name") - parameters = tool.get("parameters") - if not isinstance(name, str) or not name.strip(): - raise ValueError(f"Codex Responses tools[{idx}] is missing a valid name.") - if not isinstance(parameters, dict): - raise ValueError(f"Codex Responses tools[{idx}] is missing valid parameters.") - - description = tool.get("description", "") - if description is None: - description = "" - if not isinstance(description, str): - description = str(description) - - strict = tool.get("strict", False) - if not isinstance(strict, bool): - strict = bool(strict) - - normalized_tools.append( - { - "type": "function", - "name": name.strip(), - "description": description, - "strict": strict, - "parameters": parameters, - } - ) - - if sanitize_harmony_tokens and normalized_tools is not None: - normalized_tools = _neutralize_harmony_structure(normalized_tools) - - store = api_kwargs.get("store", False) - if store is not False: - raise ValueError("Codex Responses contract requires 'store' to be false.") - - allowed_keys = { - "model", "instructions", "input", "tools", "store", - "reasoning", "include", "max_output_tokens", "temperature", - "tool_choice", "parallel_tool_calls", "prompt_cache_key", - "prompt_cache_retention", "service_tier", "context_management", - "extra_headers", "extra_body", "timeout", - } - normalized: Dict[str, Any] = { - "model": model, - "instructions": instructions, - "input": normalized_input, - "store": False, - } - if normalized_tools is not None: + normalized_tools = [_preflight_tool(tool, idx) for idx, tool in enumerate(tools)] + if sanitize_harmony_tokens: + normalized_tools = _neutralize_harmony_structure(normalized_tools) normalized["tools"] = normalized_tools - # Pass through reasoning config - reasoning = api_kwargs.get("reasoning") - if isinstance(reasoning, dict): - normalized["reasoning"] = reasoning - include = api_kwargs.get("include") - if isinstance(include, list): - normalized["include"] = include - service_tier = api_kwargs.get("service_tier") - if isinstance(service_tier, str) and service_tier.strip(): - normalized["service_tier"] = service_tier.strip() + if api_kwargs.get("store", False) is not False: + raise ValueError("Codex Responses contract requires 'store' to be false.") - # Pass through max_output_tokens and temperature - max_output_tokens = api_kwargs.get("max_output_tokens") - if isinstance(max_output_tokens, (int, float)) and max_output_tokens > 0: - normalized["max_output_tokens"] = int(max_output_tokens) - timeout = api_kwargs.get("timeout") - if ( - isinstance(timeout, (int, float)) - and not isinstance(timeout, bool) - and 0 < float(timeout) < float("inf") - ): - normalized["timeout"] = float(timeout) - temperature = api_kwargs.get("temperature") - if isinstance(temperature, (int, float)): - normalized["temperature"] = float(temperature) - - # Pass through cache routing/retention and tool-dispatch hints. - for passthrough_key in ( - "tool_choice", - "parallel_tool_calls", - "prompt_cache_key", - "prompt_cache_retention", - ): - val = api_kwargs.get(passthrough_key) - if val is not None: - normalized[passthrough_key] = val - - # Native server-side compaction directive (gpt-5.6 on direct OpenAI / - # Codex routes — eligibility already resolved upstream in - # agent/native_compaction.py; the preflight only preserves the shape). - context_management = api_kwargs.get("context_management") - if isinstance(context_management, list) and context_management: - normalized["context_management"] = context_management + for key, accept, coerce in _PREFLIGHT_OPTIONAL_FIELDS: + value = api_kwargs.get(key) + if accept(value): + normalized[key] = coerce(value) if coerce else value extra_headers = api_kwargs.get("extra_headers") if extra_headers is not None: if not isinstance(extra_headers, dict): raise ValueError("Codex Responses request 'extra_headers' must be an object.") - normalized_headers: Dict[str, str] = {} - for key, value in extra_headers.items(): - if not isinstance(key, str) or not key.strip(): - raise ValueError("Codex Responses request 'extra_headers' keys must be non-empty strings.") - if value is None: - continue - normalized_headers[key.strip()] = str(value) + if not all(_nonblank(key) for key in extra_headers): + raise ValueError("Codex Responses request 'extra_headers' keys must be non-empty strings.") + normalized_headers = {key.strip(): str(value) for key, value in extra_headers.items() if value is not None} if normalized_headers: normalized["extra_headers"] = normalized_headers @@ -1386,15 +1006,13 @@ def _preflight_codex_api_kwargs( if extra_body is not None: if not isinstance(extra_body, dict): raise ValueError("Codex Responses request 'extra_body' must be an object.") - # Pass extra_body through verbatim — used by xAI Responses to - # carry `prompt_cache_key` as a body-level field (the documented - # cache-routing surface on /v1/responses). The openai SDK - # serializes extra_body into the JSON body without per-field - # type checks, so it survives Responses.stream() kwarg-signature - # changes that would otherwise raise TypeError before the wire. + # Verbatim: xAI carries ``prompt_cache_key`` as a body-level field. The + # SDK serializes extra_body without per-field checks, so it survives + # Responses.stream() kwarg-signature changes. if extra_body: normalized["extra_body"] = dict(extra_body) + allowed_keys = set(_PREFLIGHT_ALLOWED_KEYS) if allow_stream: stream = api_kwargs.get("stream") if stream is not None and stream is not True: @@ -1405,19 +1023,11 @@ def _preflight_codex_api_kwargs( elif "stream" in api_kwargs: raise ValueError("Codex Responses stream flag is only allowed in fallback streaming requests.") - # Safety-net sanitization for xAI Responses (#28490): defense-in-depth - # for the same slash-enum strip that ``chat_completion_helpers`` and - # ``auxiliary_client`` apply at request-build time. If a future code - # path forgets to sanitize before calling us, this catches the bypass - # so xAI doesn't 400 with ``Invalid arguments passed to the model`` - # (HuggingFace IDs like ``Qwen/Qwen3.5-0.8B`` from MCP tool schemas). - # - # Gated on the model name pattern because native Codex (OpenAI) DOES - # accept slash-containing enum values — stripping them there would - # silently degrade tool-schema constraints. xAI is the only - # Responses-API surface that rejects the shape. - model_name_for_provider_check = str(api_kwargs.get("model") or "").lower() - is_xai_model = model_name_for_provider_check.startswith(("grok-", "x-ai/grok-")) + # Defense-in-depth slash-enum strip for xAI (rejects ``Qwen/Qwen3.5`` style + # enum values with "Invalid arguments passed to the model"). Gated on the + # model name because native Codex accepts slashes — stripping there would + # silently degrade tool-schema constraints. + is_xai_model = str(api_kwargs.get("model") or "").lower().startswith(("grok-", "x-ai/grok-")) if is_xai_model and normalized.get("tools"): try: from tools.schema_sanitizer import strip_slash_enum @@ -1427,421 +1037,248 @@ def _preflight_codex_api_kwargs( unexpected = sorted(key for key in api_kwargs if key not in allowed_keys) if unexpected: - raise ValueError( - f"Codex Responses request has unsupported field(s): {', '.join(unexpected)}." - ) + raise ValueError(f"Codex Responses request has unsupported field(s): {', '.join(unexpected)}.") return normalized -# --------------------------------------------------------------------------- -# Response extraction helpers -# --------------------------------------------------------------------------- +# --- Response extraction helpers ---------------------------------------------- + +def _text_chunks(parts: Any, types: Optional[set] = None) -> List[str]: + """Non-empty ``.text`` of each part (optionally filtered by ``.type``); [] if not a list.""" + if not isinstance(parts, list): + return [] + return [ + text for text in ( + getattr(part, "text", None) for part in parts if types is None or getattr(part, "type", None) in types + ) + if _nonempty_str(text) + ] + def _extract_responses_message_text(item: Any) -> str: """Extract assistant text from a Responses message output item.""" - content = getattr(item, "content", None) - if not isinstance(content, list): - return "" - - chunks: List[str] = [] - for part in content: - ptype = getattr(part, "type", None) - if ptype not in {"output_text", "text"}: - continue - text = getattr(part, "text", None) - if isinstance(text, str) and text: - chunks.append(text) - return "".join(chunks).strip() + return "".join(_text_chunks(getattr(item, "content", None), {"output_text", "text"})).strip() def _extract_responses_reasoning_text(item: Any) -> str: - """Extract a compact reasoning text from a Responses reasoning item.""" - summary = getattr(item, "summary", None) - if isinstance(summary, list): - chunks: List[str] = [] - for part in summary: - text = getattr(part, "text", None) - if isinstance(text, str) and text: - chunks.append(text) - if chunks: - return "\n".join(chunks).strip() + """Extract a compact reasoning text from a Responses reasoning item (summary, else ``text``).""" + chunks = _text_chunks(getattr(item, "summary", None)) text = getattr(item, "text", None) - if isinstance(text, str) and text: - return text.strip() - return "" + return "\n".join(chunks).strip() if chunks else (text.strip() if isinstance(text, str) else "") def _format_responses_error(error_obj: Any, response_status: str) -> str: - """Build a human-readable error string from a Responses ``response.error`` payload. + """Human-readable string for a ``response.error`` payload (dict or object). - The OpenAI Responses API carries failure details under ``response.error`` - on terminal ``response.failed`` events, in the shape - ``{"code": "rate_limit_exceeded", "message": "Slow down", "param": ...}``. - Earlier code only surfaced ``message``, which left users staring at bare - strings like ``"Slow down"`` while the failure mode (rate limit vs - context-length vs internal_error vs model-overloaded) was hidden in - ``code``. We now prefix ``code`` when both are present so consumers can - distinguish failure modes without parsing the bare message. - - Falls back to ``code`` alone when ``message`` is empty, and to a stable - default referencing the response status when no error payload is - available at all. Adapted from anomalyco/opencode#28757. + Prefers ``": "`` so failure modes (rate limit vs context + length vs overloaded) are distinguishable; falls back to whichever is + present, then ``str(error_obj)``, then a status-based default. """ - # Pull code and message from either dict or attribute-style payloads. - code: Any = None - message: Any = None - if isinstance(error_obj, dict): - code = error_obj.get("code") - message = error_obj.get("message") - elif error_obj is not None: - code = getattr(error_obj, "code", None) - message = getattr(error_obj, "message", None) - - code_str = str(code).strip() if isinstance(code, str) else (str(code).strip() if code else "") - message_str = str(message).strip() if isinstance(message, str) else (str(message).strip() if message else "") + def field(name: str) -> str: + value = _field(error_obj, name) + return str(value).strip() if isinstance(value, str) or value else "" + code_str, message_str = field("code"), field("message") if code_str and message_str: return f"{code_str}: {message_str}" - if message_str: - return message_str - if code_str: - return code_str - if error_obj: - # Last-resort: stringify whatever the provider sent so it's at least - # visible in logs/UI rather than silently swallowed. - return str(error_obj) + if message_str or code_str or error_obj: + return message_str or code_str or str(error_obj) return f"Responses API returned status '{response_status}'" -# --------------------------------------------------------------------------- -# Full response normalization -# --------------------------------------------------------------------------- +# --- Full response normalization ---------------------------------------------- -def _normalize_codex_response( - response: Any, - *, - issuer_kind: Optional[str] = None, -) -> tuple[Any, str]: +def _synthetic_message(content: List[Any]) -> SimpleNamespace: + return SimpleNamespace(type="message", role="assistant", status="completed", content=content) + + +def _response_tool_call(item: Any, item_type: str, index: int) -> SimpleNamespace: + """Build a chat-style tool_call from a ``function_call``/``custom_tool_call`` item.""" + fn_name = getattr(item, "name", "") or "" + arguments = getattr(item, "arguments" if item_type == "function_call" else "input", "{}") + if not isinstance(arguments, str): + arguments = json.dumps(arguments, ensure_ascii=False) + raw_item_id = getattr(item, "id", None) + call_id = _resolve_call_id( + getattr(item, "call_id", None), raw_item_id, fn_name, arguments, index, canonicalize_fc=False, + ) + fc_id = _derive_responses_function_call_id(call_id, raw_item_id if isinstance(raw_item_id, str) else None) + return SimpleNamespace( + id=call_id, call_id=call_id, response_item_id=fc_id, type="function", + function=SimpleNamespace(name=fn_name, arguments=arguments), + ) + + +def _stamped_encrypted_item(item: Any, item_type: str, issuer_kind: Optional[str]) -> Optional[Dict[str, Any]]: + """``{type, encrypted_content[, _issuer_kind]}`` for replay, or None without a blob. + + ``_issuer_kind`` is stamped so a later model swap can detect an endpoint + that cannot decrypt the blob. + """ + encrypted = getattr(item, "encrypted_content", None) + if not _nonempty_str(encrypted): + return None + raw_item: Dict[str, Any] = {"type": item_type, "encrypted_content": encrypted} + if issuer_kind: + raw_item["_issuer_kind"] = issuer_kind + return raw_item + + +def _capture_reasoning_item(item: Any, issuer_kind: Optional[str]) -> Optional[Dict[str, Any]]: + """Capture a reasoning item (blob + summary) for replay; transient ``rs_tmp_`` items are skipped.""" + raw_item = _stamped_encrypted_item(item, "reasoning", issuer_kind) + if raw_item is None: + return None + item_id = getattr(item, "id", None) + if isinstance(item_id, str) and item_id.startswith("rs_tmp_"): + logger.debug("Skipping transient Codex reasoning item during normalization: %s", item_id) + return None + if _nonempty_str(item_id): + raw_item["id"] = item_id + # Summary is required by the API when replaying reasoning items. + summary = getattr(item, "summary", None) + if isinstance(summary, list): + raw_item["summary"] = [ + {"type": "summary_text", "text": text} + for text in (getattr(part, "text", None) for part in summary) + if isinstance(text, str) + ] + return raw_item + + +def _normalize_codex_response(response: Any, *, issuer_kind: Optional[str] = None) -> tuple[Any, str]: """Normalize a Responses API object to an assistant_message-like object. - ``issuer_kind`` (when provided) is stamped onto each reasoning item the - response yields, so future replays can detect when the active endpoint - differs from the one that minted the encrypted_content blob and drop - the item instead of triggering HTTP 400 invalid_encrypted_content. + ``issuer_kind`` is stamped onto captured reasoning items so replay can drop + them once the active endpoint differs from the one that minted the blob. """ - response_status = getattr(response, "status", None) - if isinstance(response_status, str): - response_status = response_status.strip().lower() - else: - response_status = None - - incomplete_details = getattr(response, "incomplete_details", None) - incomplete_reason = "" - if isinstance(incomplete_details, dict): - incomplete_reason = str(incomplete_details.get("reason") or "").strip().lower() - elif incomplete_details is not None: - incomplete_reason = str(getattr(incomplete_details, "reason", "") or "").strip().lower() + response_status = _lower_or_none(getattr(response, "status", None)) + incomplete_reason = _field(getattr(response, "incomplete_details", None), "reason", "") response_incomplete_content_filter = ( - response_status == "incomplete" and incomplete_reason == "content_filter" + response_status == "incomplete" and str(incomplete_reason or "").strip().lower() == "content_filter" ) output = getattr(response, "output", None) if not isinstance(output, list) or not output: - # The Codex backend can return empty output when the answer was - # delivered entirely via stream events. Check output_text as a - # last-resort fallback before raising. + # Codex can deliver the whole answer via stream events and return an + # empty output; fall back to output_text before raising. out_text = getattr(response, "output_text", None) if isinstance(out_text, str) and out_text.strip(): logger.debug( - "Codex response has empty output but output_text is present (%d chars); " - "synthesizing output item.", len(out_text.strip()), + "Codex response has empty output but output_text is present (%d chars); synthesizing output item.", + len(out_text.strip()), ) - output = [SimpleNamespace( - type="message", role="assistant", status="completed", - content=[SimpleNamespace(type="output_text", text=out_text.strip())], - )] - response.output = output + output = [_synthetic_message([SimpleNamespace(type="output_text", text=out_text.strip())])] elif response_incomplete_content_filter: - # This is a deterministic provider safety block, not a partial - # answer. Synthesize an empty message so finish_reason below becomes - # content_filter and the conversation loop can fallback/surface it - # instead of burning three continuation attempts. - output = [SimpleNamespace( - type="message", role="assistant", status="completed", content=[] - )] - response.output = output + # Deterministic provider safety block, not a partial answer: + # synthesize an empty message so finish_reason becomes + # content_filter instead of burning continuation attempts. + output = [_synthetic_message([])] else: raise RuntimeError("Responses API returned no output items") + response.output = output if response_status in {"failed", "cancelled"}: - error_obj = getattr(response, "error", None) - error_msg = _format_responses_error(error_obj, response_status) - raise RuntimeError(error_msg) + raise RuntimeError(_format_responses_error(getattr(response, "error", None), response_status)) content_parts: List[str] = [] reasoning_parts: List[str] = [] reasoning_items_raw: List[Dict[str, Any]] = [] message_items_raw: List[Dict[str, Any]] = [] tool_calls: List[Any] = [] - has_incomplete_items = response_status in {"queued", "in_progress", "incomplete"} + has_incomplete_items = response_status in _INCOMPLETE_STATUSES saw_streaming_or_item_incomplete = response_status in {"queued", "in_progress"} saw_commentary_phase = False saw_final_answer_phase = False saw_reasoning_item = False - # Server-side built-in tool calls (xAI's native web_search, code - # interpreter, etc.) are executed by the provider and reported as - # discrete ``*_call`` output items. xAI's /v1/responses surface - # (e.g. grok-composer-2.5-fast on SuperGrok OAuth) routinely leaves - # these items at ``status="in_progress"`` even when the overall - # ``response.status == "completed"`` — the search ran to completion - # server-side, the per-item status simply isn't reconciled. These - # are NOT a signal that the model's turn is unfinished, so they must - # not flip ``has_incomplete_items``. Only the response-level status - # and genuine model output items (message/reasoning/function_call) - # govern the incomplete verdict. Without this guard, any turn where - # grok-composer invokes server-side search is misclassified as - # ``finish_reason="incomplete"`` and burns 3 fruitless continuation - # retries before failing with "Codex response remained incomplete - # after 3 continuation attempts". client-side function/custom tool - # calls keep their own in_progress handling below (they are skipped, - # not awaited). - _SERVER_SIDE_TOOL_CALL_TYPES = { - "web_search_call", - "file_search_call", - "code_interpreter_call", - "image_generation_call", - "computer_call", - "local_shell_call", - "mcp_call", - } - for item in output: item_type = getattr(item, "type", None) - item_status = getattr(item, "status", None) - if isinstance(item_status, str): - item_status = item_status.strip().lower() - else: - item_status = None + item_status = _lower_or_none(getattr(item, "status", None)) - if ( - item_status in {"queued", "in_progress", "incomplete"} - and item_type not in _SERVER_SIDE_TOOL_CALL_TYPES - ): + if item_status in _INCOMPLETE_STATUSES and item_type not in _SERVER_SIDE_TOOL_CALL_TYPES: has_incomplete_items = True saw_streaming_or_item_incomplete = True if item_type == "message": - item_phase = getattr(item, "phase", None) - normalized_phase = None - is_commentary_phase = False - if isinstance(item_phase, str): - normalized_phase = item_phase.strip().lower() - if normalized_phase in {"commentary", "analysis"}: - saw_commentary_phase = True - is_commentary_phase = True - elif normalized_phase in {"final_answer", "final"}: - saw_final_answer_phase = True + normalized_phase = _lower_or_none(getattr(item, "phase", None)) + is_commentary_phase = normalized_phase in {"commentary", "analysis"} + saw_commentary_phase = saw_commentary_phase or is_commentary_phase + saw_final_answer_phase = saw_final_answer_phase or normalized_phase in {"final_answer", "final"} message_text = _extract_responses_message_text(item) if message_text: - # Responses ``commentary``/``analysis`` phase text is mid-turn - # preamble/progress narration, never the turn's final answer - # (Codex CLI excludes it from last-message extraction; issues - # #24933 / #41293). Keep it out of assistant content so it - # can't be concatenated into — or leak as — the final response, - # but surface it through the reasoning channel so the CLI/ - # gateway display it like thinking text. The exact message - # item is still preserved below for replay/cache continuity. - if is_commentary_phase: - reasoning_parts.append(message_text) - else: - content_parts.append(message_text) - raw_message_item: Dict[str, Any] = { - "type": "message", - "role": "assistant", - "status": _normalize_responses_message_status(item_status), - "content": [{"type": "output_text", "text": message_text}], - } + # commentary/analysis phase text is mid-turn narration, never + # the final answer: keep it out of content (so it cannot leak + # into the response) but surface it via the reasoning channel. + # The exact item is still preserved for replay/cache continuity. + (reasoning_parts if is_commentary_phase else content_parts).append(message_text) item_id = getattr(item, "id", None) - if isinstance(item_id, str) and item_id: - raw_message_item["id"] = item_id - if normalized_phase: - raw_message_item["phase"] = normalized_phase - message_items_raw.append(raw_message_item) + message_items_raw.append(_message_item( + [{"type": "output_text", "text": message_text}], + status=_normalize_responses_message_status(item_status), + item_id=item_id if isinstance(item_id, str) else None, phase=normalized_phase, + )) elif item_type == "reasoning": saw_reasoning_item = True reasoning_text = _extract_responses_reasoning_text(item) if reasoning_text: reasoning_parts.append(reasoning_text) - # Capture the full reasoning item for multi-turn continuity. - # encrypted_content is an opaque blob the API needs back on - # subsequent turns to maintain coherent reasoning chains. - encrypted = getattr(item, "encrypted_content", None) - if isinstance(encrypted, str) and encrypted: - raw_item = {"type": "reasoning", "encrypted_content": encrypted} - # Stamp the issuer so future turns can detect when a - # model swap moved the conversation to an endpoint that - # cannot decrypt this blob — see _chat_messages_to_responses_input - # cross-issuer guard. - if issuer_kind: - raw_item["_issuer_kind"] = issuer_kind - item_id = getattr(item, "id", None) - if isinstance(item_id, str) and item_id.startswith("rs_tmp_"): - logger.debug( - "Skipping transient Codex reasoning item during normalization: %s", - item_id, - ) - continue - if isinstance(item_id, str) and item_id: - raw_item["id"] = item_id - # Capture summary — required by the API when replaying reasoning items - summary = getattr(item, "summary", None) - if isinstance(summary, list): - raw_summary = [] - for part in summary: - text = getattr(part, "text", None) - if isinstance(text, str): - raw_summary.append({"type": "summary_text", "text": text}) - raw_item["summary"] = raw_summary + raw_item = _capture_reasoning_item(item, issuer_kind) + if raw_item is not None: reasoning_items_raw.append(raw_item) elif item_type == "compaction": - # Native server-side compaction checkpoint (gpt-5.6 on direct - # OpenAI/Codex routes). The encrypted blob stands in for the - # pruned older context on subsequent requests. It rides the - # codex_reasoning_items sidecar so it inherits persistence - # (state.db), session replay, the cross-issuer guard, and the - # invalid-encrypted-content kill switch without new state. - encrypted = getattr(item, "encrypted_content", None) - if isinstance(encrypted, str) and encrypted: - raw_item = {"type": "compaction", "encrypted_content": encrypted} - if issuer_kind: - raw_item["_issuer_kind"] = issuer_kind + # Native compaction checkpoint: rides the codex_reasoning_items + # sidecar so it inherits persistence, replay, the cross-issuer + # guard and the kill switch without new state. + raw_item = _stamped_encrypted_item(item, "compaction", issuer_kind) + if raw_item is not None: reasoning_items_raw.append(raw_item) logger.info( "Native Responses compaction item captured (%d chars encrypted).", - len(encrypted), + len(raw_item["encrypted_content"]), ) - elif item_type == "function_call": - if item_status in {"queued", "in_progress", "incomplete"}: + elif item_type in {"function_call", "custom_tool_call"}: + if item_type == "function_call" and item_status in _INCOMPLETE_STATUSES: continue - fn_name = getattr(item, "name", "") or "" - arguments = getattr(item, "arguments", "{}") - if not isinstance(arguments, str): - arguments = json.dumps(arguments, ensure_ascii=False) - raw_call_id = getattr(item, "call_id", None) - raw_item_id = getattr(item, "id", None) - embedded_call_id, _ = _split_responses_tool_id(raw_item_id) - call_id = raw_call_id if isinstance(raw_call_id, str) and raw_call_id.strip() else embedded_call_id - if not isinstance(call_id, str) or not call_id.strip(): - call_id = _deterministic_call_id(fn_name, arguments, len(tool_calls)) - call_id = call_id.strip() - response_item_id = raw_item_id if isinstance(raw_item_id, str) else None - response_item_id = _derive_responses_function_call_id(call_id, response_item_id) - tool_calls.append(SimpleNamespace( - id=call_id, - call_id=call_id, - response_item_id=response_item_id, - type="function", - function=SimpleNamespace(name=fn_name, arguments=arguments), - )) - elif item_type == "custom_tool_call": - fn_name = getattr(item, "name", "") or "" - arguments = getattr(item, "input", "{}") - if not isinstance(arguments, str): - arguments = json.dumps(arguments, ensure_ascii=False) - raw_call_id = getattr(item, "call_id", None) - raw_item_id = getattr(item, "id", None) - embedded_call_id, _ = _split_responses_tool_id(raw_item_id) - call_id = raw_call_id if isinstance(raw_call_id, str) and raw_call_id.strip() else embedded_call_id - if not isinstance(call_id, str) or not call_id.strip(): - call_id = _deterministic_call_id(fn_name, arguments, len(tool_calls)) - call_id = call_id.strip() - response_item_id = raw_item_id if isinstance(raw_item_id, str) else None - response_item_id = _derive_responses_function_call_id(call_id, response_item_id) - tool_calls.append(SimpleNamespace( - id=call_id, - call_id=call_id, - response_item_id=response_item_id, - type="function", - function=SimpleNamespace(name=fn_name, arguments=arguments), - )) + tool_calls.append(_response_tool_call(item, item_type, len(tool_calls))) - final_text = "\n".join([p for p in content_parts if p]).strip() - if ( - not final_text - and hasattr(response, "output_text") - and not (saw_commentary_phase and not saw_final_answer_phase) - ): + final_text = "\n".join(content_parts).strip() + if not final_text and hasattr(response, "output_text") and (saw_final_answer_phase or not saw_commentary_phase): out_text = getattr(response, "output_text", "") if isinstance(out_text, str): final_text = out_text.strip() - # ── Tool-call leak recovery ────────────────────────────────── - # gpt-5.x on the Codex Responses API sometimes degenerates and emits - # what should be a structured `function_call` item as plain assistant - # text using the Harmony/Codex serialization (``to=functions.foo - # {json}`` or ``assistant to=functions.foo {json}``). The model - # intended to call a tool, but the intent never made it into - # ``response.output`` as a ``function_call`` item, so ``tool_calls`` - # is empty here. If we pass this through, the parent sees a - # confident-looking summary with no audit trail (empty ``tool_trace``) - # and no tools actually ran — the Taiwan-embassy-email incident. - # - # Detection: leaked tokens always contain ``to=functions.`` and - # the assistant message has no real tool calls. Treat it as incomplete - # so the existing Codex-incomplete continuation path (3 retries, - # handled in run_agent.py) gets a chance to re-elicit a proper - # ``function_call`` item. The existing loop already handles message - # append, dedup, and retry budget. - leaked_tool_call_text = False - if final_text and not tool_calls and _TOOL_CALL_LEAK_PATTERN.search(final_text): - leaked_tool_call_text = True + # Tool-call leak recovery: gpt-5.x sometimes emits the intended + # ``function_call`` as plain Harmony text (``to=functions.foo {json}``) + # with no structured item, so ``tool_calls`` is empty and no tool ran. + # Treat as incomplete so the continuation path re-elicits a real call; + # clear the text so the garbage is not surfaced as a summary (encrypted + # reasoning is preserved for the retry). + leaked_tool_call_text = bool(final_text and not tool_calls and _TOOL_CALL_LEAK_PATTERN.search(final_text)) + if leaked_tool_call_text: logger.warning( - "Codex response contains leaked tool-call text in assistant content " - "(no structured function_call items). Treating as incomplete so the " - "continuation path can re-elicit a proper tool call. Leaked snippet: %r", - final_text[:300], + "Codex response contains leaked tool-call text in assistant content (no structured function_call " + "items). Treating as incomplete so the continuation path can re-elicit a proper tool call. " + "Leaked snippet: %r", final_text[:300], ) - # Clear the text so downstream code doesn't surface the garbage as - # a summary. The encrypted reasoning items (if any) are preserved - # so the model keeps its chain-of-thought on the retry. final_text = "" - # ── Reasoning-channel answer salvage (xAI grok) ────────────── - # grok-4.x on the xAI /v1/responses surface sometimes emits its final - # answer inside the reasoning item instead of as a ``message`` output - # item, marking where the answer starts with grok's internal - # ```` delimiter. Without salvage, the reasoning-only rule - # below classifies the turn ``incomplete`` — and because reasoning - # items on this surface carry no ``encrypted_content``, the interim - # message replays as nothing, so every continuation request is - # byte-identical to the one that just failed. The turn burns its 3 - # retries and dies with "Codex response remained incomplete after 3 - # continuation attempts" even though the answer was produced on the - # first attempt. Observed live with grok-4.20 on xai-oauth - # (2026-07-13). Promote the delimited tail to assistant content and - # keep the untagged prefix as thinking text. - if ( - issuer_kind == "xai_responses" - and not final_text - and not tool_calls - and reasoning_parts - ): + # Reasoning-channel answer salvage (xAI grok): grok-4.x sometimes puts the + # final answer inside the reasoning item after its ```` delimiter. + # Without salvage the reasoning-only rule marks the turn incomplete, and + # since these items carry no encrypted_content every continuation request + # is byte-identical to the failed one. Promote the delimited tail to + # content and keep the untagged prefix as thinking text. + if issuer_kind == "xai_responses" and not final_text and not tool_calls and reasoning_parts: joined_reasoning = "\n\n".join(reasoning_parts) marker = joined_reasoning.rfind("") if marker != -1: - salvaged = joined_reasoning[marker + len(""):] - closing = salvaged.find("") - if closing != -1: - salvaged = salvaged[:closing] - salvaged = salvaged.strip() + salvaged = joined_reasoning[marker + len(""):].split("", 1)[0].strip() if salvaged: logger.warning( - "xAI response delivered its final answer inside the " - "reasoning channel ( delimiter); promoting " - "%d chars to assistant content.", - len(salvaged), + "xAI response delivered its final answer inside the reasoning channel " + "( delimiter); promoting %d chars to assistant content.", len(salvaged), ) final_text = salvaged reasoning_prefix = joined_reasoning[:marker].strip() @@ -1861,36 +1298,22 @@ def _normalize_codex_response( finish_reason = "tool_calls" elif response_incomplete_content_filter: finish_reason = "content_filter" - elif leaked_tool_call_text: - finish_reason = "incomplete" - elif saw_streaming_or_item_incomplete: - finish_reason = "incomplete" - elif (has_incomplete_items or saw_commentary_phase) and not saw_final_answer_phase: + elif ( + leaked_tool_call_text + or saw_streaming_or_item_incomplete + or ((has_incomplete_items or saw_commentary_phase) and not saw_final_answer_phase) + ): finish_reason = "incomplete" elif (reasoning_items_raw or reasoning_parts or saw_reasoning_item) and not final_text: - # Response contains only reasoning (encrypted thinking state and/or - # human-readable summary) with no visible content or tool calls. - # - # For the specially-handled backends (Codex, xAI, GitHub/Copilot), - # reasoning-only with status="completed" means "the model is still - # thinking and needs another turn" — treat it as incomplete so the - # Codex continuation path retries instead of falling into the - # empty-content retry loop. - # - # For all other backends (other:, etc.), trust the provider's - # own response.status signal. When status == "completed" and no items - # are queued/in_progress/incomplete, reasoning alone is a valid final - # state — forcing "incomplete" causes multi-minute stalls as the - # continuation path re-issues calls (3 retries × up to 240s each). - # See https://github.com/NousResearch/hermes-agent/issues/64434 - if response_status == "completed" and issuer_kind not in ( - "codex_backend", - "xai_responses", - "github_responses", - ): - finish_reason = "stop" - else: - finish_reason = "incomplete" + # Reasoning-only response. For Codex/xAI/GitHub, reasoning-only with + # status=completed means "still thinking, needs another turn" → + # incomplete so the continuation path retries. Other backends: trust + # response.status — forcing incomplete there stalls for minutes + # (3 retries × up to 240s) on a legitimately final state. + trusted_final = ( + response_status == "completed" and issuer_kind not in ("codex_backend", "xai_responses", "github_responses") + ) + finish_reason = "stop" if trusted_final else "incomplete" else: finish_reason = "stop" return assistant_message, finish_reason diff --git a/agent/codex_runtime.py b/agent/codex_runtime.py index 30b86c6171..ba8cea2426 100644 --- a/agent/codex_runtime.py +++ b/agent/codex_runtime.py @@ -1,17 +1,10 @@ """Codex API runtime — App Server and Responses-API streaming paths. -Extracted from :class:`AIAgent` to keep the agent loop file focused. -Each function takes the parent ``AIAgent`` as its first argument -(``agent``). AIAgent keeps thin forwarder methods for backward -compatibility. - -* ``run_codex_app_server_turn`` — drives one turn through the - ``codex_app_server`` subprocess client (used when a Codex CLI install - is the active provider). -* ``run_codex_stream`` — streams a Codex Responses API call (the - ``codex_responses`` api_mode). -* ``run_codex_create_stream_fallback`` — recovery path when the - Responses ``stream=True`` initial create fails. +Extracted from :class:`AIAgent`; every entry point takes the parent agent as its +first argument and AIAgent keeps thin forwarders. ``run_codex_app_server_turn`` +drives one ``codex app-server`` subprocess turn (``codex_app_server`` api_mode); +``run_codex_stream`` runs one streaming Codex Responses call (``codex_responses``); +``run_codex_create_stream_fallback`` is a legacy alias of the latter. """ from __future__ import annotations @@ -28,14 +21,21 @@ from agent.stream_single_writer import claim_stream_writer, stream_writer_is_cur logger = logging.getLogger(__name__) -def _codex_request_failure_details(error: BaseException) -> tuple[int | None, str]: - """Return the serialized request size and exception class chain. +def _call_guarded(fn: Callable | None, fail_msg: str, *fail_args: Any, args: tuple = (), kwargs: dict | None = None): + """Invoke an optional display/debug callback; a buggy hook must never tear down the turn.""" + if fn is None: + return + try: + fn(*args, **(kwargs or {})) + except Exception: + logger.debug(fail_msg, *fail_args, exc_info=True) - OpenAI connection exceptions retain the final ``httpx.Request``. Reading - its already-buffered content gives us the exact byte count handed to the - transport without logging any request content. The class-only chain keeps - the underlying transport failure visible without exposing URLs or payloads - from exception messages. + +def _codex_request_failure_details(error: BaseException) -> tuple[int | None, str]: + """Return (serialized request bytes, exception class chain) for a failed request. + + OpenAI connection exceptions retain the final ``httpx.Request``; its buffered + content gives the exact byte count without logging payloads or URLs. """ request_body_bytes: int | None = None exception_classes: list[str] = [] @@ -45,143 +45,101 @@ def _codex_request_failure_details(error: BaseException) -> tuple[int | None, st while current is not None and id(current) not in seen and len(seen) < 8: seen.add(id(current)) exception_classes.append(type(current).__name__) - if request_body_bytes is None: try: request = getattr(current, "request", None) + content = request.content if request is not None else None except Exception: - request = None - if request is not None: - try: - content = request.content - except Exception: - content = None - if isinstance(content, str): - request_body_bytes = len(content.encode("utf-8")) - elif isinstance(content, (bytes, bytearray, memoryview)): - request_body_bytes = len(content) - - cause = current.__cause__ - if cause is None and not current.__suppress_context__: - cause = current.__context__ - current = cause - + content = None + if isinstance(content, str): + request_body_bytes = len(content.encode("utf-8")) + elif isinstance(content, (bytes, bytearray, memoryview)): + request_body_bytes = len(content) + if current.__cause__ is None and not current.__suppress_context__: + current = current.__context__ + else: + current = current.__cause__ return request_body_bytes, " <- ".join(exception_classes) -def _log_codex_request_failure( - agent: Any, - error: BaseException, - *, - stream_opened: bool, -) -> None: +def _log_codex_request_failure(agent: Any, error: BaseException, *, stream_opened: bool) -> None: request_body_bytes, exception_chain = _codex_request_failure_details(error) logger.warning( - "Codex Responses request failed: " - "serialized_request_body_bytes=%s stream_opened=%s " + "Codex Responses request failed: serialized_request_body_bytes=%s stream_opened=%s " "exception_chain=%s model=%s", request_body_bytes if request_body_bytes is not None else "unknown", - str(stream_opened).lower(), - exception_chain, - getattr(agent, "model", "unknown"), + str(stream_opened).lower(), exception_chain, getattr(agent, "model", "unknown"), ) def _coerce_usage_int(value: Any) -> int: - if isinstance(value, bool): + if isinstance(value, bool) or not isinstance(value, (int, float, str)): return 0 - if isinstance(value, int): - return max(value, 0) - if isinstance(value, float): + try: return max(int(value), 0) - if isinstance(value, str): - try: - return max(int(value), 0) - except ValueError: - return 0 - return 0 + except ValueError: + return 0 + + +def _queue_token_counts(agent, fail_msg: str, *fail_extra: Any, counts: Callable[[], dict]) -> None: + """Enqueue per-call accounting for the SessionDB background writer (off the turn thread). + + ``counts`` is built lazily inside the guarded try so a stub agent without a + session DB never has its accounting attributes touched.""" + if not (agent._session_db and agent.session_id): + return + try: + if not agent._session_db_created: + agent._ensure_db_session() + agent._session_db.queue_token_counts(agent.session_id, **counts()) + except Exception as exc: + logger.debug(fail_msg, agent.session_id, *fail_extra, exc) def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]: - """Translate Codex app-server token usage into Hermes accounting. + """Translate Codex app-server token usage (thread/tokenUsage/updated) into Hermes accounting. - Codex app-server reports usage via thread/tokenUsage/updated as: - inputTokens, cachedInputTokens, outputTokens, reasoningOutputTokens, - totalTokens. - - Hermes' canonical prompt bucket includes uncached input + cached input. - The Codex app-server protocol does not currently expose cache-write tokens, - so that bucket remains zero on this runtime. - - Even when Codex omits usage for a turn, Hermes should still count that turn - as one API call for session/status accounting. + Hermes' prompt bucket = uncached + cached input. The app-server protocol + exposes no cache-write tokens, so that bucket stays zero here. A turn with + no usage still counts as one API call for session/status accounting. """ agent.session_api_calls += 1 usage = getattr(turn, "token_usage_last", None) + compressor = getattr(agent, "context_compressor", None) if not isinstance(usage, dict) or not usage: - compressor = getattr(agent, "context_compressor", None) - if ( - compressor is not None - and getattr(compressor, "awaiting_real_usage_after_compression", False) - ): - # No usage means this turn cannot adjudicate the pending compaction. - # Consume the marker so a later unrelated reading is not charged to - # it and preflight deferral cannot stay latched indefinitely. + if compressor is not None and getattr(compressor, "awaiting_real_usage_after_compression", False): + # No usage cannot adjudicate the pending compaction; consume the marker so + # preflight deferral cannot stay latched. compressor.update_from_response({}) - if agent._session_db and agent.session_id: - try: - if not agent._session_db_created: - agent._ensure_db_session() - # Enqueued for the SessionDB background writer — keeps the - # per-call accounting write off the turn thread (see - # conversation_loop's queue_token_counts call). - agent._session_db.queue_token_counts( - agent.session_id, - model=agent.model, - billing_provider=agent.provider, - billing_base_url=agent.base_url, - billing_mode="subscription_included", - api_call_count=1, - ) - except Exception as exc: - logger.debug( - "Codex app-server api-call persistence failed (session=%s): %s", - agent.session_id, exc, - ) + _queue_token_counts( + agent, "Codex app-server api-call persistence failed (session=%s): %s", + counts=lambda: dict( + model=agent.model, billing_provider=agent.provider, billing_base_url=agent.base_url, + billing_mode="subscription_included", api_call_count=1, + ), + ) return {} from agent.usage_pricing import CanonicalUsage, estimate_usage_cost - input_tokens = _coerce_usage_int(usage.get("inputTokens")) - cache_read_tokens = _coerce_usage_int(usage.get("cachedInputTokens")) - output_tokens = _coerce_usage_int(usage.get("outputTokens")) - reasoning_tokens = _coerce_usage_int(usage.get("reasoningOutputTokens")) - reported_total = _coerce_usage_int(usage.get("totalTokens")) - canonical_usage = CanonicalUsage( - input_tokens=input_tokens, - output_tokens=output_tokens, - cache_read_tokens=cache_read_tokens, + input_tokens=_coerce_usage_int(usage.get("inputTokens")), + output_tokens=_coerce_usage_int(usage.get("outputTokens")), + cache_read_tokens=_coerce_usage_int(usage.get("cachedInputTokens")), cache_write_tokens=0, - reasoning_tokens=reasoning_tokens, + reasoning_tokens=_coerce_usage_int(usage.get("reasoningOutputTokens")), raw_usage=usage, ) prompt_tokens = canonical_usage.prompt_tokens - completion_tokens = canonical_usage.output_tokens - total_tokens = reported_total or canonical_usage.total_tokens - usage_dict = { - "prompt_tokens": prompt_tokens, - "completion_tokens": completion_tokens, - "total_tokens": total_tokens, - "input_tokens": canonical_usage.input_tokens, - "output_tokens": canonical_usage.output_tokens, - "cache_read_tokens": canonical_usage.cache_read_tokens, - "cache_write_tokens": canonical_usage.cache_write_tokens, - "reasoning_tokens": canonical_usage.reasoning_tokens, + total_tokens = _coerce_usage_int(usage.get("totalTokens")) or canonical_usage.total_tokens + token_counts = { + field: getattr(canonical_usage, field) + for field in ("input_tokens", "output_tokens", "cache_read_tokens", "cache_write_tokens", "reasoning_tokens") } + usage_dict = {"prompt_tokens": prompt_tokens, "completion_tokens": canonical_usage.output_tokens, + "total_tokens": total_tokens, **token_counts} - compressor = getattr(agent, "context_compressor", None) if compressor is not None: try: compressor.update_from_response(usage_dict) @@ -191,64 +149,30 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]: except Exception: logger.debug("codex app-server usage update failed", exc_info=True) - agent.session_prompt_tokens += prompt_tokens - agent.session_completion_tokens += completion_tokens - agent.session_total_tokens += total_tokens - agent.session_input_tokens += canonical_usage.input_tokens - agent.session_output_tokens += canonical_usage.output_tokens - agent.session_cache_read_tokens += canonical_usage.cache_read_tokens - agent.session_cache_write_tokens += canonical_usage.cache_write_tokens - agent.session_reasoning_tokens += canonical_usage.reasoning_tokens + for key, value in usage_dict.items(): + setattr(agent, f"session_{key}", getattr(agent, f"session_{key}") + value) cost_result = estimate_usage_cost( - agent.model, - canonical_usage, - provider=agent.provider, - base_url=agent.base_url, - api_key=getattr(agent, "api_key", ""), + agent.model, canonical_usage, + provider=agent.provider, base_url=agent.base_url, api_key=getattr(agent, "api_key", ""), ) - if cost_result.amount_usd is not None: - agent.session_estimated_cost_usd += float(cost_result.amount_usd) - agent.session_cost_status = cost_result.status - agent.session_cost_source = cost_result.source + cost_usd = float(cost_result.amount_usd) if cost_result.amount_usd is not None else None + if cost_usd is not None: + agent.session_estimated_cost_usd += cost_usd + agent.session_cost_status, agent.session_cost_source = cost_result.status, cost_result.source + cost_fields = {"estimated_cost_usd": cost_usd, "cost_status": cost_result.status, "cost_source": cost_result.source} - if agent._session_db and agent.session_id: - try: - if not agent._session_db_created: - agent._ensure_db_session() - # Enqueued for the SessionDB background writer (see above). - agent._session_db.queue_token_counts( - agent.session_id, - input_tokens=canonical_usage.input_tokens, - output_tokens=canonical_usage.output_tokens, - cache_read_tokens=canonical_usage.cache_read_tokens, - cache_write_tokens=canonical_usage.cache_write_tokens, - reasoning_tokens=canonical_usage.reasoning_tokens, - estimated_cost_usd=float(cost_result.amount_usd) - if cost_result.amount_usd is not None else None, - cost_status=cost_result.status, - cost_source=cost_result.source, - billing_provider=agent.provider, - billing_base_url=agent.base_url, - billing_mode="subscription_included" - if cost_result.status == "included" else None, - model=agent.model, - api_call_count=1, - ) - except Exception as exc: - logger.debug( - "Codex app-server token persistence failed (session=%s, tokens=%d): %s", - agent.session_id, total_tokens, exc, - ) + _queue_token_counts( + agent, "Codex app-server token persistence failed (session=%s, tokens=%d): %s", total_tokens, + counts=lambda: dict( + **token_counts, **cost_fields, + billing_provider=agent.provider, billing_base_url=agent.base_url, + billing_mode="subscription_included" if cost_result.status == "included" else None, + model=agent.model, api_call_count=1, + ), + ) - return { - **usage_dict, - "last_prompt_tokens": prompt_tokens, - "estimated_cost_usd": float(cost_result.amount_usd) - if cost_result.amount_usd is not None else None, - "cost_status": cost_result.status, - "cost_source": cost_result.source, - } + return {**usage_dict, "last_prompt_tokens": prompt_tokens, **cost_fields} def _record_codex_app_server_compaction( @@ -258,11 +182,10 @@ def _record_codex_app_server_compaction( approx_tokens: int | None = None, force: bool = False, ) -> bool: - """Record a Codex-native context compaction boundary in Hermes state. + """Record a Codex-native compaction boundary in Hermes state. - The app-server owns the compacted thread context, so Hermes should not - rewrite local transcript rows here; state.db records the boundary via the - session event/usage counters while preserving the visible transcript. + The app-server owns the compacted thread, so local transcript rows are NOT + rewritten; only session event/usage counters record the boundary. """ if not force and not getattr(turn, "compacted", False): return False @@ -271,34 +194,24 @@ def _record_codex_app_server_compaction( turn_id = getattr(turn, "turn_id", None) or "" logger.info( "codex app-server compaction observed: session=%s thread=%s turn=%s force=%s", - getattr(agent, "session_id", None) or "none", - thread_id, - turn_id, - force, + getattr(agent, "session_id", None) or "none", thread_id, turn_id, force, ) if not force: try: from agent.conversation_compression import COMPACTION_STATUS - agent._emit_status(COMPACTION_STATUS) except Exception: pass compressor = getattr(agent, "context_compressor", None) if compressor is not None: - compressor.compression_count = getattr( - compressor, "compression_count", 0 - ) + 1 + compressor.compression_count = getattr(compressor, "compression_count", 0) + 1 compressor.last_compression_rough_tokens = approx_tokens or 0 - # The app server has already completed a real compaction boundary. Its - # usage update (when supplied) is therefore the same real-vs-real - # effectiveness verdict used by the normal compression path. - record_boundary = getattr( - type(compressor), "record_completed_compaction", None - ) + # The boundary already happened server-side; its usage update (when supplied) + # is the same real-vs-real effectiveness verdict the normal path uses. Codex owns + # this summary, so a prior Hermes deterministic-fallback flag must not leak into it. + record_boundary = getattr(type(compressor), "record_completed_compaction", None) if callable(record_boundary): - # Codex owns this summary. A prior Hermes deterministic-fallback - # flag must not leak into the native boundary's quality verdict. record_boundary(compressor, used_fallback=False) elif hasattr(compressor, "_verify_compaction_cleared_threshold"): compressor._verify_compaction_cleared_threshold = True @@ -307,111 +220,71 @@ def _record_codex_app_server_compaction( compressor.last_completion_tokens = 0 compressor.awaiting_real_usage_after_compression = True - # Native compaction rewrote the provider-side context; the usage anchor's - # transcript snapshot no longer matches what will be sent. Invalidate it. + # Provider-side context was rewritten; the usage anchor's transcript snapshot no longer matches. agent._usage_anchor = None agent._turn_base_usage_anchor = None - agent._last_compaction_in_place = False - try: - if getattr(agent, "event_callback", None): - agent.event_callback( - "session:compress", - { - "platform": getattr(agent, "platform", None) or "", - "session_id": getattr(agent, "session_id", None) or "", - "old_session_id": "", - "in_place": False, - "compression_count": getattr( - compressor, "compression_count", 0 - ) - if compressor is not None - else 0, - "runtime": "codex_app_server", - "thread_id": thread_id, - "turn_id": turn_id, - }, - ) - except Exception: - logger.debug("event_callback error on codex session:compress", exc_info=True) - + _call_guarded(getattr(agent, "event_callback", None) or None, "event_callback error on codex session:compress", + args=("session:compress", { + "platform": getattr(agent, "platform", None) or "", + "session_id": getattr(agent, "session_id", None) or "", + "old_session_id": "", + "in_place": False, + "compression_count": getattr(compressor, "compression_count", 0) if compressor is not None else 0, + "runtime": "codex_app_server", + "thread_id": thread_id, + "turn_id": turn_id, + })) return True -# --------------------------------------------------------------------------- -# Codex app-server → Hermes UI bridge (#33200) -# -# The codex_app_server runtime hands the entire turn to a subprocess and -# bypasses the normal Hermes tool loop. Without this bridge gateway -# adapters (Discord, Telegram, TUI) never see live tool-progress bubbles -# or interim assistant commentary while codex is working — the user just -# stares at a quiet channel until the final answer lands. The bridge -# translates raw codex JSON-RPC notifications into the same three agent -# callbacks the standard runtime fires: -# - tool_progress_callback("tool.started"|"tool.completed", name, ...) -# - _fire_stream_delta(text) for streaming agentMessage chunks -# - _emit_interim_assistant_message({...}) for completed agentMessages -# --------------------------------------------------------------------------- +# --- Codex app-server → Hermes UI bridge ------------------------------------- +# The app-server runtime hands the whole turn to a subprocess and bypasses the +# Hermes tool loop, so gateway adapters would see nothing until the final answer. +# The bridge translates JSON-RPC notifications into the callbacks the standard +# runtime fires: tool_progress_callback("tool.started"|"tool.completed"), +# _fire_stream_delta(text), _emit_interim_assistant_message. -# Codex item types that map to a Hermes tool_call in the projector (and -# therefore deserve a tool_progress bubble pair). The projector lives in -# agent/transports/codex_event_projector.py — keep these in sync so the -# tool name shown in the UI matches the name recorded in messages. -# webSearch is codex's built-in web search tool — it has no projector -# entry (codex handles it internally) but still deserves a bubble. -_CODEX_TOOL_ITEM_TYPES = frozenset( - {"commandExecution", "fileChange", "mcpToolCall", "dynamicToolCall", "webSearch"} -) +# Item types that project to a Hermes tool_call (keep in sync with +# agent/transports/codex_event_projector.py so UI names match recorded names). +# webSearch is codex's built-in tool: no projector entry, still gets a bubble. +_CODEX_TOOL_ITEM_TYPES = frozenset({"commandExecution", "fileChange", "mcpToolCall", "dynamicToolCall", "webSearch"}) -# Internal MCP server that wraps Hermes' native tools for codex. When -# codex calls back through it, the inner dispatch runs in a SEPARATE -# hermes-tools-mcp-server subprocess that has no access to the parent -# agent's tool_progress_callback — so the inner call can never surface -# its own native progress event. The codex-level mcpToolCall event IS -# the display event for those calls; we strip the mcp.hermes-tools.* -# namespacing and emit the bare tool name (web_search, browser_navigate, -# vision_analyze, ...) since the user thinks of these as Hermes tools, -# not as MCP calls. +# Internal MCP server wrapping Hermes' native tools. Its inner dispatch runs in a +# separate subprocess with no tool_progress_callback, so the codex-level mcpToolCall +# IS the display event; the mcp.hermes-tools.* prefix is stripped because the +# user thinks of these as Hermes tools. _INTERNAL_MCP_SERVER = "hermes-tools" +_STATIC_TOOL_NAMES = {"commandExecution": "exec_command", "fileChange": "apply_patch", "webSearch": "web_search"} +_STABLE_ID_PREFIXES = {"commandExecution": "exec", "fileChange": "apply_patch"} +_MCP_LIKE_ITEM_TYPES = {"mcpToolCall", "dynamicToolCall"} +# Item types whose preview is the first 120 chars of one string field. +_PREVIEW_FIELDS = {"commandExecution": "command", "webSearch": "query"} + def _codex_item_to_tool_name(item: dict) -> str: - """Synthetic Hermes tool name for a codex item. Mirrors - CodexEventProjector so the progress bubble and the projected - tool_calls entry use the same identifier.""" + """Synthetic Hermes tool name for a codex item (mirrors CodexEventProjector).""" item_type = item.get("type") or "" - if item_type == "commandExecution": - return "exec_command" - if item_type == "fileChange": - return "apply_patch" if item_type == "mcpToolCall": - server = item.get("server") or "mcp" - tool = item.get("tool") or "unknown" - if server == _INTERNAL_MCP_SERVER: - return tool - return f"mcp.{server}.{tool}" + server, tool = item.get("server") or "mcp", item.get("tool") or "unknown" + return tool if server == _INTERNAL_MCP_SERVER else f"mcp.{server}.{tool}" if item_type == "dynamicToolCall": return item.get("tool") or "dynamic" - if item_type == "webSearch": - return "web_search" - return item_type or "unknown" + return _STATIC_TOOL_NAMES.get(item_type) or item_type or "unknown" def _codex_item_to_args(item: dict) -> dict: - """Args dict surfaced to tool_progress_callback("tool.started", ...). - Mirrors the projector's _project_command / _project_file_change / - _project_mcp_tool_call / _project_dynamic_tool_call shapes.""" + """Args dict for tool_progress_callback("tool.started"); mirrors the projector shapes.""" item_type = item.get("type") or "" if item_type == "commandExecution": - return {"command": item.get("command") or "", - "cwd": item.get("cwd") or ""} + return {"command": item.get("command") or "", "cwd": item.get("cwd") or ""} if item_type == "fileChange": return {"changes": [ - {"kind": (c.get("kind") or {}).get("type") or "update", - "path": c.get("path") or ""} + {"kind": (c.get("kind") or {}).get("type") or "update", "path": c.get("path") or ""} for c in (item.get("changes") or []) if isinstance(c, dict) ]} - if item_type in {"mcpToolCall", "dynamicToolCall"}: + if item_type in _MCP_LIKE_ITEM_TYPES: args = item.get("arguments") or {} return args if isinstance(args, dict) else {"arguments": args} if item_type == "webSearch": @@ -420,22 +293,16 @@ def _codex_item_to_args(item: dict) -> dict: def _codex_item_to_preview(item: dict) -> Any: - """Short human-readable preview for the tool.started bubble. Returns - None when no useful preview is available (Hermes' UI tolerates None).""" + """Short preview for the tool.started bubble; None when nothing useful (UI tolerates None).""" item_type = item.get("type") or "" - if item_type == "commandExecution": - cmd = item.get("command") or "" - return cmd[:120] if cmd else None + if item_type in _PREVIEW_FIELDS: + return (item.get(_PREVIEW_FIELDS[item_type]) or "")[:120] or None if item_type == "fileChange": - paths = [c.get("path") for c in (item.get("changes") or []) - if isinstance(c, dict) and c.get("path")] + paths = [c.get("path") for c in (item.get("changes") or []) if isinstance(c, dict) and c.get("path")] if not paths: return None - preview = ", ".join(paths[:3]) - if len(paths) > 3: - preview += f", +{len(paths) - 3} more" - return preview - if item_type in {"mcpToolCall", "dynamicToolCall"}: + return ", ".join(paths[:3]) + (f", +{len(paths) - 3} more" if len(paths) > 3 else "") + if item_type in _MCP_LIKE_ITEM_TYPES: args = item.get("arguments") or {} if not isinstance(args, dict) or not args: return None @@ -443,16 +310,11 @@ def _codex_item_to_preview(item: dict) -> Any: return json.dumps(args, ensure_ascii=False)[:120] except (TypeError, ValueError): return None - if item_type == "webSearch": - query = item.get("query") or "" - return query[:120] if query else None return None def _codex_item_completion_payload(item: dict) -> tuple[str, bool]: - """Return (result_text, is_error) for a completed codex tool item. - Mirrors the projector's tool-result content so the bubble shows the - same outcome string that ends up in the messages list.""" + """(result_text, is_error) for a completed tool item; mirrors the projector's tool-result content.""" item_type = item.get("type") or "" if item_type == "commandExecution": out = item.get("aggregatedOutput") or "" @@ -464,123 +326,70 @@ def _codex_item_completion_payload(item: dict) -> tuple[str, bool]: if item_type == "fileChange": status = item.get("status") or "unknown" n = len(item.get("changes") or []) - return ( - f"apply_patch status={status}, {n} change(s)", - status not in {"completed", "applied", "success"}, - ) + return f"apply_patch status={status}, {n} change(s)", status not in {"completed", "applied", "success"} if item_type == "mcpToolCall": error = item.get("error") if error: - return ( - f"[error] {json.dumps(error, ensure_ascii=False)[:1000]}", - True, - ) + return f"[error] {json.dumps(error, ensure_ascii=False)[:1000]}", True result = item.get("result") - return ( - json.dumps(result, ensure_ascii=False)[:4000] - if result is not None else "", - False, - ) + return (json.dumps(result, ensure_ascii=False)[:4000] if result is not None else ""), False if item_type == "dynamicToolCall": content_items = item.get("contentItems") or [] - if isinstance(content_items, list) and content_items: - return ( - json.dumps(content_items, ensure_ascii=False)[:4000], - not bool(item.get("success", True)), - ) success = item.get("success", True) + if isinstance(content_items, list) and content_items: + return json.dumps(content_items, ensure_ascii=False)[:4000], not bool(success) return f"success={success}", not bool(success) return "", False +def _stable_call_id(item: dict, name: str) -> str: + """Deterministic tool_call id mirroring CodexEventProjector (live TUI card correlates with projected history).""" + from agent.transports.codex_event_projector import _deterministic_call_id + + item_type = item.get("type") or "" + tool = item.get("tool") or "unknown" + if item_type == "mcpToolCall": + prefix = f"mcp__{item.get('server') or 'mcp'}__{tool}" + elif item_type == "dynamicToolCall": + prefix = f"dyn_{tool}" + else: + prefix = _STABLE_ID_PREFIXES.get(item_type, name) + return _deterministic_call_id(prefix, item.get("id") or "") + + def make_codex_app_server_event_bridge(agent) -> Callable[[dict], None]: - """Build an ``on_event`` callback that wires codex app-server JSON-RPC - notifications into Hermes' gateway UI callbacks. + """Build the ``on_event`` callback for ``CodexAppServerSession(on_event=...)``. - Returns a single-argument callable suitable for - ``CodexAppServerSession(on_event=...)``. - - Translation map: - * ``item/started`` for tool-shaped items → ``tool_progress_callback( - "tool.started", name, preview, args)`` - * ``item/completed`` for tool-shaped items → ``tool_progress_callback( - "tool.completed", name, None, None, duration=..., is_error=..., - result=...)`` - * ``item/agentMessage/delta`` → ``_fire_stream_delta(text)`` so chat - adapters can render the assistant's reply as it streams. - * ``item/reasoning/delta`` → ``_fire_reasoning_delta(text)`` - * ``item/completed`` for ``agentMessage`` → - ``_emit_interim_assistant_message({"role": "assistant", - "content": text})``. The gateway's ``already_streamed`` check - dedupes against any text the stream-delta callback already - rendered for the same message. - - All callback invocations are guarded — a buggy display callback must - not tear down the codex turn loop. Errors are logged at DEBUG so the - notification stream keeps flowing regardless. + Tool items fire ``tool_progress_callback`` ("tool.started" / "tool.completed" + with duration=, is_error=, result=) plus the stable-ID ``tool_start_callback`` + / ``tool_complete_callback`` card hooks; deltas go to ``_fire_stream_delta`` / + ``_fire_reasoning_delta``; a completed agentMessage goes to + ``_emit_interim_assistant_message`` (the gateway's ``already_streamed`` check + dedupes against streamed deltas). Every callback is guarded (DEBUG log) so a + buggy display hook cannot tear down the turn loop. """ - # item_id -> (tool_name, args, started_wall_time). Populated on - # item/started and consumed on item/completed so duration is correct - # even when codex doesn't report durationMs. + # item_id -> (tool_name, args, started_monotonic); duration even when codex omits durationMs. started: dict[str, tuple[str, dict, float]] = {} - def _stable_call_id(item: dict, name: str) -> str: - """Deterministic tool_call id mirroring CodexEventProjector, so a - live TUI tool card correlates with the same tool call after the - session is resumed and history is projected.""" - from agent.transports.codex_event_projector import _deterministic_call_id - - item_id = item.get("id") or "" - item_type = item.get("type") or "" - if item_type == "commandExecution": - return _deterministic_call_id("exec", item_id) - if item_type == "fileChange": - return _deterministic_call_id("apply_patch", item_id) - if item_type == "mcpToolCall": - server = item.get("server") or "mcp" - tool = item.get("tool") or "unknown" - return _deterministic_call_id(f"mcp__{server}__{tool}", item_id) - if item_type == "dynamicToolCall": - tool = item.get("tool") or "unknown" - return _deterministic_call_id(f"dyn_{tool}", item_id) - return _deterministic_call_id(name, item_id) - def _fire_tool_started(item: dict) -> None: item_id = item.get("id") or "" name = _codex_item_to_tool_name(item) args = _codex_item_to_args(item) if item_id: started[item_id] = (name, args, time.monotonic()) - cb = getattr(agent, "tool_progress_callback", None) - if cb is not None: - try: - cb("tool.started", name, _codex_item_to_preview(item), args) - except Exception: - logger.debug( - "tool_progress_callback raised on tool.started for %s", - name, exc_info=True, - ) - # Authoritative stable-ID tool card (TUI / desktop). Fires - # alongside tool_progress so surfaces that render structured tool - # cards (not just progress bubbles) stay correlated with the - # projected history entry after a resume. - start_cb = getattr(agent, "tool_start_callback", None) - if start_cb is not None: - try: - start_cb(_stable_call_id(item, name), name, args) - except Exception: - logger.debug( - "tool_start_callback raised for %s", name, exc_info=True, - ) + _call_guarded(getattr(agent, "tool_progress_callback", None), + "tool_progress_callback raised on tool.started for %s", name, + args=("tool.started", name, _codex_item_to_preview(item), args)) + # Stable-ID tool card (TUI/desktop) fires alongside the progress bubble. + _call_guarded(getattr(agent, "tool_start_callback", None), "tool_start_callback raised for %s", name, + args=(_stable_call_id(item, name), name, args)) def _fire_tool_completed(item: dict) -> None: item_id = item.get("id") or "" name = _codex_item_to_tool_name(item) prior = started.pop(item_id, None) - # Prefer codex's own durationMs when present so the bubble shows - # exact tool wall-time; fall back to our started timestamp; fall - # back to None if we never saw an item/started (some codex - # versions only emit completed for fast items). + # Prefer codex's durationMs; else our started timestamp; else None + # (some codex versions only emit completed for fast items). duration: Any = None codex_ms = item.get("durationMs") if isinstance(codex_ms, (int, float)) and codex_ms >= 0: @@ -588,331 +397,167 @@ def make_codex_app_server_event_bridge(agent) -> Callable[[dict], None]: elif prior is not None: duration = time.monotonic() - prior[2] result, is_error = _codex_item_completion_payload(item) - cb = getattr(agent, "tool_progress_callback", None) - if cb is not None: - try: - cb("tool.completed", name, None, None, - duration=duration, is_error=is_error, result=result) - except Exception: - logger.debug( - "tool_progress_callback raised on tool.completed for %s", - name, exc_info=True, - ) - complete_cb = getattr(agent, "tool_complete_callback", None) - if complete_cb is not None: - args = prior[1] if prior is not None else _codex_item_to_args(item) - try: - complete_cb(_stable_call_id(item, name), name, args, result) - except Exception: - logger.debug( - "tool_complete_callback raised for %s", name, exc_info=True, - ) + _call_guarded(getattr(agent, "tool_progress_callback", None), + "tool_progress_callback raised on tool.completed for %s", name, + args=("tool.completed", name, None, None), + kwargs={"duration": duration, "is_error": is_error, "result": result}) + args = prior[1] if prior is not None else _codex_item_to_args(item) + _call_guarded(getattr(agent, "tool_complete_callback", None), "tool_complete_callback raised for %s", name, + args=(_stable_call_id(item, name), name, args, result)) - def _fire_text_delta(params: dict) -> None: + def _fire_delta(params: dict, attr: str) -> None: text = params.get("delta") or params.get("text") or "" - if not isinstance(text, str) or not text: - return - fn = getattr(agent, "_fire_stream_delta", None) - if fn is None: - return - try: - fn(text) - except Exception: - logger.debug("_fire_stream_delta raised", exc_info=True) - - def _fire_reasoning_delta(params: dict) -> None: - text = params.get("delta") or params.get("text") or "" - if not isinstance(text, str) or not text: - return - fn = getattr(agent, "_fire_reasoning_delta", None) - if fn is None: - return - try: - fn(text) - except Exception: - logger.debug("_fire_reasoning_delta raised", exc_info=True) + if isinstance(text, str) and text: + _call_guarded(getattr(agent, attr, None), f"{attr} raised", args=(text,)) def _fire_agent_message_completed(item: dict) -> None: text = item.get("text") or "" if not isinstance(text, str) or not text.strip(): return - # display.show_commentary=false — mid-turn narration stays off the - # visible interim path on this runtime too (same contract as the - # codex_responses commentary channel). + # display.show_commentary=false keeps mid-turn narration off the + # interim path here too (same contract as codex_responses commentary). if not getattr(agent, "show_commentary", True): return - emit = getattr(agent, "_emit_interim_assistant_message", None) - if emit is None: - return - try: - emit({"role": "assistant", "content": text}) - except Exception: - logger.debug( - "_emit_interim_assistant_message raised", exc_info=True, - ) + _call_guarded(getattr(agent, "_emit_interim_assistant_message", None), + "_emit_interim_assistant_message raised", + args=({"role": "assistant", "content": text},)) - def on_event(note: dict) -> None: - if not isinstance(note, dict): - return - method = note.get("method") or "" - params = note.get("params") or {} - if not isinstance(params, dict): - params = {} - if method == "item/agentMessage/delta": - _fire_text_delta(params) - return - if method in {"item/reasoning/delta", "item/reasoning/summaryDelta"}: - _fire_reasoning_delta(params) - return + def _on_item(params: dict, completed: bool) -> None: item = params.get("item") if not isinstance(item, dict): return item_type = item.get("type") or "" - if method == "item/started" and item_type in _CODEX_TOOL_ITEM_TYPES: - _fire_tool_started(item) - return - if method == "item/completed": - if item_type in _CODEX_TOOL_ITEM_TYPES: - _fire_tool_completed(item) - elif item_type == "agentMessage": - _fire_agent_message_completed(item) + if item_type in _CODEX_TOOL_ITEM_TYPES: + (_fire_tool_completed if completed else _fire_tool_started)(item) + elif completed and item_type == "agentMessage": + _fire_agent_message_completed(item) + + handlers: dict[str, Callable[[dict], None]] = { + "item/agentMessage/delta": lambda p: _fire_delta(p, "_fire_stream_delta"), + "item/reasoning/delta": lambda p: _fire_delta(p, "_fire_reasoning_delta"), + "item/reasoning/summaryDelta": lambda p: _fire_delta(p, "_fire_reasoning_delta"), + "item/started": lambda p: _on_item(p, completed=False), + "item/completed": lambda p: _on_item(p, completed=True), + } + + def on_event(note: dict) -> None: + handler = handlers.get(note.get("method") or "") if isinstance(note, dict) else None + if handler is not None: + params = note.get("params") or {} + handler(params if isinstance(params, dict) else {}) return on_event -def run_codex_app_server_turn( - agent, - *, - user_message: str, - original_user_message: Any, - messages: List[Dict[str, Any]], - effective_task_id: str, - should_review_memory: bool = False, -) -> Dict[str, Any]: - """Codex app-server runtime path. Hands the entire turn to a `codex - app-server` subprocess and projects its events back into Hermes' - messages list so memory/skill review keep working. +# --- Codex app-server turn ---------------------------------------------------- - Called from run_conversation() when agent.api_mode == "codex_app_server". - Returns the same dict shape as the chat_completions path. - """ - # Defense in depth for compression.checkpoint_required: agent init - # already refuses this combination, but api_mode is a plain attribute a - # future code path could mutate on a live agent. Fail closed before the - # codex agent can compact its thread — once run_turn() executes, a - # codex-owned compaction may already have happened with no pre-compress - # checkpoint. Explicit-True check matches the compress_context() gate. - if getattr(agent, "compression_checkpoint_required", False) is True: - from agent.conversation_compression import _checkpoint_blocked - - raise _checkpoint_blocked( - "codex_app_server owns the authoritative thread and compacts it " - "without a truthful pre-compaction transcript boundary" - ) - - from agent.transports.codex_app_server_session import ( - CodexAppServerSession, - _ServerRequestRouting, - ) - - # Lazy session: one CodexAppServerSession per AIAgent instance. - # Spawned on first turn, reused across turns, closed at AIAgent - # shutdown (see _cleanup hook). - if not hasattr(agent, "_codex_session") or agent._codex_session is None: - from agent.runtime_cwd import resolve_agent_cwd - - cwd = getattr(agent, "session_cwd", None) or str(resolve_agent_cwd()) - # Approval callback: defer to Hermes' standard prompt flow if a - # CLI thread has installed one. Gateway / cron contexts get the - # codex-side fail-closed default. - try: - from tools.terminal_tool import _get_approval_callback - approval_callback = _get_approval_callback() - except Exception: - approval_callback = None - - # Gateway / cron contexts have no UI to surface codex's approval - # requests through, so codex app-server exec / apply_patch requests - # fail closed (silently decline) by default. When the user has - # explicitly opted out of Hermes approvals — via `approvals.mode: off` - # in config, the /yolo session toggle, or --yolo / HERMES_YOLO_MODE — - # honor that and let codex's own sandbox permission profile - # (~/.codex/config.toml) be the policy gate instead of double-gating - # with a missing Hermes UI. Defaults (manual/smart/unset) preserve the - # current fail-closed behavior — this is a no-op for those users. - auto_approve_requests = False - try: - from tools.approval import is_approval_bypass_active - - auto_approve_requests = is_approval_bypass_active() - except Exception: - logger.debug( - "codex app-server: approval-bypass lookup failed; " - "keeping fail-closed default", - exc_info=True, - ) - - # Bridge codex JSON-RPC notifications (item/started, item/completed, - # item/agentMessage/delta, ...) into Hermes' gateway UI callbacks - # (tool_progress_callback, _fire_stream_delta, - # _emit_interim_assistant_message). Without this, Discord/Telegram - # users see no live tool-progress or interim commentary while - # codex_app_server is running — only the final answer (#33200). - # Supersedes the narrower item/started-only bridge from #38835. - agent._codex_session = CodexAppServerSession( - cwd=cwd, - approval_callback=approval_callback, - request_routing=_ServerRequestRouting( - auto_approve_exec=auto_approve_requests, - auto_approve_apply_patch=auto_approve_requests, - ), - on_event=make_codex_app_server_event_bridge(agent), - ) - - # NOTE: the user message is ALREADY appended to messages by the - # standard run_conversation() flow (line ~11823) before the early - # return reaches us. Do NOT append again — that would duplicate. +def _close_codex_session(agent) -> None: + """Drop the session so the next turn respawns codex instead of reusing a dead client.""" try: - turn = agent._codex_session.run_turn(user_input=user_message) - except Exception as exc: - logger.exception("codex app-server turn failed") - # Crash → unconditionally drop the session so the next turn - # respawns from scratch instead of reusing a dead client. - try: - agent._codex_session.close() - except Exception: - pass - agent._codex_session = None - _user_interrupted = bool( - getattr(agent, "_interrupt_requested", False) - ) - _interrupt_message = ( - getattr(agent, "_interrupt_message", None) - if _user_interrupted - else None - ) - if _user_interrupted: - agent.clear_interrupt() - return { - "final_response": ( - f"Codex app-server turn failed: {exc}. " - f"Fall back to default runtime with `/codex-runtime auto`." - ), - "messages": messages, - "api_calls": 0, - "completed": False, - "partial": True, - "interrupted": _user_interrupted, - **( - {"interrupt_message": _interrupt_message} - if _interrupt_message - else {} - ), - "error": str(exc), - } + agent._codex_session.close() + except Exception: + pass + agent._codex_session = None - # This runtime bypasses the normal conversation-loop finalizer. Mirror its - # interrupt handoff/cleanup so a hard stop cannot poison the next turn and a - # message-bearing compatibility interrupt can still be replayed by callers. - _user_interrupted = bool( - turn.interrupted and getattr(agent, "_interrupt_requested", False) - ) - _interrupt_message = ( - getattr(agent, "_interrupt_message", None) if _user_interrupted else None - ) - if _user_interrupted: + +def _consume_user_interrupt(agent, active: bool = True) -> tuple[bool, Any]: + """Mirror the conversation-loop finalizer's interrupt handoff: returns + (user_interrupted, interrupt_message) and clears the agent-level interrupt so a + hard stop cannot poison the next turn.""" + interrupted = bool(active and getattr(agent, "_interrupt_requested", False)) + message = getattr(agent, "_interrupt_message", None) if interrupted else None + if interrupted: agent.clear_interrupt() - - # If the turn signalled the underlying client is wedged (deadline - # blown, post-tool watchdog tripped, OAuth refresh died, subprocess - # exited), retire the session so the next turn respawns codex - # rather than riding the broken process. Mirrors openclaw beta.8's - # "retire timed-out app-server clients" fix. - if getattr(turn, "should_retire", False): - logger.warning( - "codex app-server session retired (turn error: %s)", - turn.error, - ) - try: - agent._codex_session.close() - except Exception: - pass - agent._codex_session = None - - # Splice projected messages into the conversation. The projector emits - # standard {role, content, tool_calls, tool_call_id} entries, which - # is exactly what curator.py / sessions DB expect. - if turn.projected_messages: - from agent.message_metadata import append_message - - for projected_message in turn.projected_messages: - append_message(messages, projected_message) - - # Persist the newly-projected assistant/tool messages ourselves. - # This path is an early return that bypasses conversation_loop, whose - # normal per-step _persist_session() calls would otherwise flush them. - # The inbound user turn was already flushed at turn start - # (turn_context.py _persist_session), and _flush_messages_to_session_db - # is idempotent via the intrinsic _DB_PERSISTED_MARKER — so this writes - # ONLY the new codex projected rows and does NOT re-write the user turn. - # Keeping the agent as the sole persister lets us return - # agent_persisted=True below, so the gateway skips its own DB write and - # we avoid the #860/#42039 duplicate user-message write (append_message - # is a raw INSERT with no dedup, so a gateway re-write would duplicate - # the already-flushed user turn). See gateway/run.py agent_persisted. - if getattr(agent, "_session_db", None) is not None: - try: - _codex_flush_ok = agent._flush_messages_to_session_db(messages) - except Exception: - _codex_flush_ok = False - logger.warning( - "codex app-server projected-message flush failed", - exc_info=True, - ) - if _codex_flush_ok is False: - # Unlike the chat-completions loop (which fails closed BEFORE - # projection — see conversation_loop session_persistence_failed), - # codex output has already streamed to the user by the time this - # flush runs, so there is nothing left to withhold. We cannot - # flip agent_persisted=False either: the gateway fallback write - # would re-INSERT the already-flushed user turn (#860/#42039). - # Surface the durability gap loudly instead of a silent debug. - logger.warning( - "codex app-server turn was delivered but could NOT be " - "persisted to the session DB (session=%s) — this turn " - "will be missing after restart/resume", - getattr(agent, "session_id", None), - ) + return interrupted, message - # Counter ticks for the agent-improvement loop. - # _turns_since_memory and _user_turn_count are ALREADY incremented - # in the run_conversation() pre-loop block (lines ~11793-11817) so we - # do NOT touch them here — that would double-count. - # Only _iters_since_skill needs explicit increment, since the - # chat_completions loop bumps it per tool iteration (line ~12110) - # and that loop is bypassed on this path. - agent._iters_since_skill = ( - getattr(agent, "_iters_since_skill", 0) + turn.tool_iterations +def _ensure_codex_session(agent) -> None: + """Lazily spawn one CodexAppServerSession per AIAgent (reused across turns, closed by the _cleanup hook).""" + if getattr(agent, "_codex_session", None) is not None: + return + from agent.runtime_cwd import resolve_agent_cwd + from agent.transports.codex_app_server_session import CodexAppServerSession, _ServerRequestRouting + + # Approval callback: Hermes' standard prompt flow when a CLI thread installed one. + try: + from tools.terminal_tool import _get_approval_callback + approval_callback = _get_approval_callback() + except Exception: + approval_callback = None + # Gateway/cron have no UI for codex approval requests, so exec/apply_patch fail + # closed (silently decline) by default. Only an explicit approval bypass + # (approvals.mode: off, /yolo, --yolo, HERMES_YOLO_MODE) hands policy to codex's + # own sandbox profile (~/.codex/config.toml). + auto_approve_requests = False + try: + from tools.approval import is_approval_bypass_active + auto_approve_requests = is_approval_bypass_active() + except Exception: + logger.debug("codex app-server: approval-bypass lookup failed; keeping fail-closed default", exc_info=True) + + agent._codex_session = CodexAppServerSession( + cwd=getattr(agent, "session_cwd", None) or str(resolve_agent_cwd()), + approval_callback=approval_callback, + request_routing=_ServerRequestRouting( + auto_approve_exec=auto_approve_requests, auto_approve_apply_patch=auto_approve_requests, + ), + on_event=make_codex_app_server_event_bridge(agent), ) + + +def _persist_projected_messages(agent, turn, messages: List[Dict[str, Any]]) -> None: + """Splice the projected {role, content, tool_calls, tool_call_id} entries into + ``messages`` and flush them to the session DB. + + Bypasses conversation_loop's per-step _persist_session(). The user turn was + flushed at turn start and the flush dedups via _DB_PERSISTED_MARKER, so only the + new codex rows are written. The agent stays the sole persister + (agent_persisted=True): a gateway re-write would re-INSERT the user turn. + """ + if not turn.projected_messages: + return + from agent.message_metadata import append_message + + for projected_message in turn.projected_messages: + append_message(messages, projected_message) + + if getattr(agent, "_session_db", None) is None: + return + try: + flush_ok = agent._flush_messages_to_session_db(messages) + except Exception: + flush_ok = False + logger.warning("codex app-server projected-message flush failed", exc_info=True) + if flush_ok is False: + # Output already streamed, and agent_persisted cannot flip to False (gateway + # fallback would duplicate the user turn): surface the durability gap loudly. + logger.warning( + "codex app-server turn was delivered but could NOT be persisted to the session DB " + "(session=%s) — this turn will be missing after restart/resume", + getattr(agent, "session_id", None), + ) + + +def _finish_codex_turn( + agent, turn, messages: List[Dict[str, Any]], *, original_user_message: Any, should_review_memory: bool, +) -> dict[str, Any]: + """Post-turn bookkeeping mirroring the chat_completions loop; returns usage fields.""" + # run_conversation()'s pre-loop block already bumped _turns_since_memory / + # _user_turn_count; only _iters_since_skill (per tool iteration in the bypassed loop) is ours. + agent._iters_since_skill = getattr(agent, "_iters_since_skill", 0) + turn.tool_iterations _record_codex_app_server_compaction(agent, turn) usage_result = _record_codex_app_server_usage(agent, turn) - api_calls = 1 - # Now check the skill nudge AFTER iters were incremented — same - # pattern the chat_completions path uses (line ~15432). - should_review_skills = False - if ( - agent._skill_nudge_interval > 0 - and agent._iters_since_skill >= agent._skill_nudge_interval + # Skill nudge check AFTER iters were incremented (same as chat_completions). + should_review_skills = ( + agent._skill_nudge_interval > 0 and agent._iters_since_skill >= agent._skill_nudge_interval and "skill_manage" in agent.valid_tool_names - ): - should_review_skills = True + ) + if should_review_skills: agent._iters_since_skill = 0 - # External memory provider sync (mirrors line ~15439). Skipped on - # interrupt/error to avoid feeding partial transcripts to memory. + # External memory sync skipped on interrupt/error (no partial transcripts). if not turn.interrupted and turn.error is None: try: agent._sync_external_memory_for_turn( @@ -924,14 +569,8 @@ def run_codex_app_server_turn( except Exception: logger.debug("external memory sync raised", exc_info=True) - # Background review fork — same cadence + signature as the default - # path (line ~15449). Only fires when a trigger actually tripped AND - # we have a real final response. - if ( - turn.final_text - and not turn.interrupted - and (should_review_memory or should_review_skills) - ): + # Background review fork: only when a trigger tripped AND a real final response exists. + if turn.final_text and not turn.interrupted and (should_review_memory or should_review_skills): try: agent._spawn_background_review( messages_snapshot=list(messages), @@ -941,98 +580,116 @@ def run_codex_app_server_turn( except Exception: logger.debug("background review spawn raised", exc_info=True) + return usage_result + + +def run_codex_app_server_turn( + agent, + *, + user_message: str, + original_user_message: Any, + messages: List[Dict[str, Any]], + effective_task_id: str, + should_review_memory: bool = False, +) -> Dict[str, Any]: + """Hand the turn to a ``codex app-server`` subprocess and project its events into ``messages``. + + Called from run_conversation() when agent.api_mode == "codex_app_server"; returns + the chat_completions result shape. The user message is ALREADY in ``messages`` — never append it again. + """ + # Defense in depth for compression.checkpoint_required: agent init refuses the + # combination, but api_mode is mutable. Fail closed before run_turn() can trigger a + # codex-owned compaction with no checkpoint. Explicit-True check matches compress_context(). + if getattr(agent, "compression_checkpoint_required", False) is True: + from agent.conversation_compression import _checkpoint_blocked + raise _checkpoint_blocked( + "codex_app_server owns the authoritative thread and compacts it " + "without a truthful pre-compaction transcript boundary" + ) + + _ensure_codex_session(agent) + + try: + turn = agent._codex_session.run_turn(user_input=user_message) + except Exception as exc: + logger.exception("codex app-server turn failed") + _close_codex_session(agent) + return _turn_result( + _consume_user_interrupt(agent), messages, api_calls=0, completed=False, error=str(exc), + final_response=( + f"Codex app-server turn failed: {exc}. Fall back to default runtime with `/codex-runtime auto`." + ), + ) + + interrupt = _consume_user_interrupt(agent, turn.interrupted) + + # Wedged client (deadline blown, watchdog tripped, OAuth refresh died, + # subprocess exited): retire the session so the next turn respawns codex. + if getattr(turn, "should_retire", False): + logger.warning("codex app-server session retired (turn error: %s)", turn.error) + _close_codex_session(agent) + + _persist_projected_messages(agent, turn, messages) + usage_result = _finish_codex_turn( + agent, turn, messages, original_user_message=original_user_message, should_review_memory=should_review_memory, + ) + + return _turn_result( + interrupt, messages, api_calls=1, completed=not turn.interrupted and turn.error is None, error=turn.error, + final_response=turn.final_text, + # We flushed the projected rows ourselves (see _persist_projected_messages); + # True makes the gateway skip its own DB write, which would duplicate + # the already-flushed user turn. + agent_persisted=True, + codex_thread_id=turn.thread_id, + codex_turn_id=turn.turn_id, + **usage_result, + ) + + +def _turn_result( + interrupt: tuple[bool, Any], messages: List[Dict[str, Any]], *, + api_calls: int, completed: bool, error: Any, final_response: Any, **extra: Any, +) -> Dict[str, Any]: + """Result shape shared with the chat_completions path (``partial`` == ``not completed``).""" + user_interrupted, interrupt_message = interrupt return { - "final_response": turn.final_text, + "final_response": final_response, "messages": messages, "api_calls": api_calls, - "completed": not turn.interrupted and turn.error is None, - "partial": turn.interrupted or turn.error is not None, - "interrupted": _user_interrupted, - **( - {"interrupt_message": _interrupt_message} - if _interrupt_message - else {} - ), - "error": turn.error, - # The codex app-server runtime IS an early-return path that bypasses - # conversation_loop, but we flush the projected assistant/tool messages - # ourselves above (see the _flush_messages_to_session_db call after - # messages.extend). The inbound user turn was already flushed at turn - # start (turn_context._persist_session) and the flush dedups via - # _DB_PERSISTED_MARKER, so state.db ends up with each real message - # exactly once and session_search / conversation-distill see the full - # gateway conversation. Report agent_persisted=True so the gateway - # skips its own append_to_transcript DB write — writing again there - # would re-INSERT the already-flushed user turn (append_message has no - # dedup), reintroducing the #860 / #42039 duplicate-write bug. - "agent_persisted": True, - "codex_thread_id": turn.thread_id, - "codex_turn_id": turn.turn_id, - **usage_result, + "completed": completed, + "partial": not completed, + "interrupted": user_interrupted, + **({"interrupt_message": interrupt_message} if interrupt_message else {}), + "error": error, + **extra, } -# --------------------------------------------------------------------------- -# Event-driven Responses streaming -# -# OpenAI ships its consumer Codex backend (chatgpt.com/backend-api/codex) on -# a different schedule from the openai Python SDK. The high-level -# ``client.responses.stream(...)`` helper reconstructs a typed Response from -# the terminal ``response.completed`` event's ``response.output`` field, and -# when that field drifts to ``null`` (gpt-5.5, May 2026) the SDK raises -# ``TypeError: 'NoneType' object is not iterable`` mid-iteration. -# -# We sidestep the whole class of failure by going one level lower: -# ``client.responses.create(stream=True)`` returns the raw AsyncIterable of -# SSE events, and we assemble the final response object purely from -# ``response.output_item.done`` events as they arrive. We never read -# ``response.completed.response.output`` for content reconstruction, so the -# backend can return ``null``, ``[]``, a string, or omit the field entirely -# and we don't care. -# -# This mirrors what the OpenClaw TS implementation does for the same backend -# and is structurally immune to the bug class rather than patched. -# --------------------------------------------------------------------------- - - -_TERMINAL_EVENT_TYPES = frozenset({ - "response.completed", - "response.incomplete", - "response.failed", -}) +# --- Event-driven Responses streaming ----------------------------------------- +# The consumer Codex backend drifts independently of the openai SDK: the high-level +# ``responses.stream(...)`` helper rebuilds a typed Response from +# ``response.completed.response.output`` and crashes when that field is null. We go +# one level lower (``responses.create(stream=True)`` raw SSE events) and assemble the +# final response from ``response.output_item.done``, so the terminal ``output`` may +# be null / [] / a string / absent. def _event_field(event: Any, name: str, default: Any = None) -> Any: - """Field access that handles both attr-style (SDK objects) and dict (raw JSON) events.""" + """Field access for attr-style (SDK objects) and dict (raw JSON) events/items.""" value = getattr(event, name, None) if value is None and isinstance(event, dict): value = event.get(name, default) return value if value is not None else default -def _item_field(item: Any, name: str, default: Any = None) -> Any: - """Field access for nested Response items (attr-style SDK object or dict).""" - value = getattr(item, name, None) - if value is None and isinstance(item, dict): - value = item.get(name, default) - return value if value is not None else default - - def _raise_stream_error(event: Any) -> None: - """Raise a ``_StreamErrorEvent`` from a ``type=error`` SSE frame. + """Raise ``_StreamErrorEvent`` from a ``type=error`` SSE frame. - The Responses spec puts the failure details at the top level of the - frame (``{"type": "error", "code": ..., "message": ..., "param": ...}``), - but the official OpenAI SDK and several OpenAI-compatible proxies wrap - them in an HTTP-style nested envelope instead - (``{"type": "error", "error": {"code": ..., "message": ..., "param": ...}}``). - Read the top-level fields first, then fall back to the nested envelope so - the error classifier sees the provider's real code/message (rate-limit vs - context-overflow vs entitlement) rather than the generic placeholder. - Port of anomalyco/opencode#36130. - - Imported lazily so this module stays importable from places that don't - pull in ``run_agent`` (e.g. plugin code, doc tools). + The spec puts code/message/param at the top level, but the OpenAI SDK and + several proxies nest them under ``error``. Read top-level first, then the + envelope, so the classifier sees the provider's real code/message. + ``run_agent`` is imported lazily to keep this module importable standalone. """ from run_agent import _StreamErrorEvent @@ -1040,95 +697,35 @@ def _raise_stream_error(event: Any) -> None: def _error_field(name: str) -> Any: value = _event_field(event, name) - if value is None and nested is not None: - value = _item_field(nested, name) - return value + return _event_field(nested, name) if value is None and nested is not None else value raw_message = _error_field("message") if raw_message is not None and not isinstance(raw_message, str): raw_message = str(raw_message) message = (raw_message or "stream emitted error event").strip() or "stream emitted error event" - raise _StreamErrorEvent( - message, - code=_error_field("code"), - param=_error_field("param"), - ) + raise _StreamErrorEvent(message, code=_error_field("code"), param=_error_field("param")) -def _consume_codex_event_stream( - event_iter: Any, - *, - model: str, - on_text_delta=None, - on_reasoning_delta=None, - on_commentary_message=None, - on_first_delta=None, - on_event=None, - interrupt_check=None, -) -> SimpleNamespace: - """Consume a Codex Responses SSE event stream and return a final response. +def _message_phase(item: Any) -> str | None: + phase = _event_field(item, "phase", None) + return phase.strip().lower() if isinstance(phase, str) else None - The returned object is a ``SimpleNamespace`` shaped like the SDK's typed - ``Response`` for the fields downstream code actually reads: - * ``output``: list of output items, assembled from ``response.output_item.done``. - For tool-call turns this contains the function_call items; for plain-text - turns it contains a synthesized ``message`` item built from streamed deltas - if no message item was emitted directly. - * ``output_text``: assembled text from ``response.output_text.delta`` deltas. - * ``usage``: copied from the terminal event's ``response.usage`` (when present). - * ``status``: ``completed`` / ``incomplete`` / ``failed`` (or ``completed`` if - the stream ended without a terminal frame but produced content). - * ``id``: ``response.id`` when present. - * ``incomplete_details``: passed through for ``response.incomplete`` frames. - * ``error``: passed through for ``response.failed`` frames. - * ``model``: from kwargs (the wire model name is not authoritative). +class _CodexResponseAssembler: + """Assemble a Response-shaped ``SimpleNamespace`` from raw Responses SSE events. - Critically, we never read ``response.output`` from the terminal event for - content reconstruction — only ``usage``, ``status``, ``id``. That field - being ``null`` / ``[]`` / missing is fine. - - Callbacks: - - * ``on_text_delta(str)`` — fires per ``response.output_text.delta``, suppressed - once a function_call event is seen (so tool-call turns don't bleed text - into the chat). - * ``on_reasoning_delta(str)`` — fires per ``response.reasoning.*.delta`` and - ``phase=analysis`` message deltas. When no dedicated commentary callback - is supplied, commentary also uses this legacy fallback. - * ``on_commentary_message(str)`` — fires once per completed - ``phase=commentary`` message, before any following tool item executes. - * ``on_first_delta()`` — one-shot, fires on the first text delta only. - * ``on_event(event)`` — fires for every event before any other processing. - Used for watchdog activity, debug logging, anything wire-shape-agnostic. - * ``interrupt_check()`` — returns True to break the loop early, or raises - ``TimeoutError`` / ``InterruptedError`` for request-retirement control - flow that must not be converted into a partial final response. + Only ``usage`` / ``status`` / ``id`` are read from the terminal frame — never + ``response.output``. Output items come from ``output_item.done``, or are + synthesized from text deltas, or settled from function calls announced via + ``output_item.added`` but never confirmed (some compatible backends omit + per-item done events on success). """ - collected_output_items: List[Any] = [] - # output_index of each collected_output_items entry, appended in lockstep - # so settled pending calls can be merged back in stream order. - collected_output_indexes: List[Any] = [] - collected_output_sequences: List[int] = [] - collected_text_deltas: List[str] = [] + has_tool_calls = False - # Function calls announced via output_item.added but not yet confirmed by - # output_item.done, keyed by item id. Some OpenAI-compatible backends omit - # per-item done events on a successful completion (upstream evidence: - # anomalyco/opencode#37159); these are settled from accumulated stream - # state at the terminal event so the tool call executes instead of being - # silently dropped. - pending_function_calls: Dict[str, Dict[str, Any]] = {} - # First-observed (sequence, output_index) per announced item id, so items - # confirmed later via output_item.done keep their announced stream - # position when merged with settled pending calls. - announced_output_order: Dict[str, tuple] = {} + next_output_sequence = 0 first_delta_fired = False active_message_phase: str | None = None - commentary_text_deltas: List[str] = [] - # Last reasoning summary_index seen. The Responses stream delimits summary - # parts by this index and gives each part no separator of its own, so a - # change of index is where the blank line belongs. + # Reasoning summary parts carry no separator; a summary_index change is where the blank line belongs. active_summary_index: Any = None terminal_status: str = "completed" terminal_usage: Any = None @@ -1136,373 +733,300 @@ def _consume_codex_event_stream( terminal_incomplete_details: Any = None terminal_error: Any = None saw_terminal = False - # Settlement of pending calls requires an actually observed successful - # terminal frame. ``terminal_status`` defaults to "completed", so it - # cannot distinguish a real response.completed from EOF/interruption. + # terminal_status defaults to "completed", so settlement needs an + # explicitly observed response.completed frame (not EOF/interrupt). saw_response_completed = False - next_output_sequence = 0 + def __init__(self, *, model, on_text_delta, on_reasoning_delta, on_commentary_message, on_first_delta): + self.model = model + self.on_text_delta = on_text_delta + self.on_reasoning_delta = on_reasoning_delta + self.on_commentary_message = on_commentary_message + self.on_first_delta = on_first_delta + self.output_items: List[Any] = [] + # output_index / first-observed sequence per output item, in lockstep, so + # settled pending calls merge back in stream order. + self.output_indexes: List[Any] = [] + self.output_sequences: List[int] = [] + self.text_deltas: List[str] = [] + self.commentary_text_deltas: List[str] = [] + # Announced-but-unconfirmed function calls keyed by item id. + self.pending_function_calls: Dict[str, Dict[str, Any]] = {} + # First-observed (sequence, output_index) per announced item id so a later + # .done keeps its announced position when merged with settled calls. + self.announced_output_order: Dict[str, tuple] = {} + + def _safe(self, cb: Callable | None, label: str, *args: Any) -> None: + _call_guarded(cb, f"Codex stream {label} raised", args=args) + + def _on_error(self, event: Any, event_type: str) -> None: + # ``error`` frames carry the provider's real failure reason (quota / model + # unavailable / rejected reasoning replay); surface them so the credential + # pool + error classifier see the body. + _raise_stream_error(event) + + def _on_item_added(self, event: Any, event_type: str) -> None: + item = _event_field(event, "item") + item_type = _event_field(item, "type", "") + if item_type == "message": + self.active_message_phase = _message_phase(item) + if self.active_message_phase == "commentary": + self.commentary_text_deltas = [] + else: + self.active_message_phase = None + # Record first-observed ordering for EVERY announced item; the .done path must + # reuse it, or a mixed announced/pending stream without output_index values reorders the calls. + item_id = str(_event_field(item, "id", "")) + if item_id and item_id not in self.announced_output_order: + self.announced_output_order[item_id] = (self.next_output_sequence, _event_field(event, "output_index")) + self.next_output_sequence += 1 + if "function_call" in str(item_type): + self.has_tool_calls = True + if item_id: + announced_sequence, announced_index = self.announced_output_order[item_id] + self.pending_function_calls[item_id] = { + "item": item, + "arguments": str(_event_field(item, "arguments", "") or ""), + "output_index": announced_index, + "sequence": announced_sequence, + } + + def _on_text_delta(self, event: Any, event_type: str) -> None: + delta_text = _event_field(event, "delta", "") + if not delta_text: + return + # Harmony commentary/analysis text is mid-turn narration, never the final + # answer: route to the reasoning callback, keep only the item for replay. + if self.active_message_phase == "commentary": + self.commentary_text_deltas.append(delta_text) + # Legacy fallback when no first-class commentary consumer is installed. + if self.on_commentary_message is None: + self._safe(self.on_reasoning_delta, "on_reasoning_delta", delta_text) + elif self.active_message_phase == "analysis": + self._safe(self.on_reasoning_delta, "on_reasoning_delta", delta_text) + else: + self.text_deltas.append(delta_text) + if not self.has_tool_calls: + if not self.first_delta_fired: + self.first_delta_fired = True + self._safe(self.on_first_delta, "on_first_delta") + self._safe(self.on_text_delta, "on_text_delta", delta_text) + + def _on_function_call(self, event: Any, event_type: str) -> None: + self.has_tool_calls = True + pending = self.pending_function_calls.get(str(_event_field(event, "item_id", ""))) + if "delta" in event_type: + delta_args = _event_field(event, "delta", "") + if pending is not None and delta_args: + pending["arguments"] += delta_args + elif event_type.endswith("function_call_arguments.done"): + # Authoritative for the accumulated string; an explicit "" (zero-arg + # call) counts, only a missing field keeps the streamed deltas. + done_args = _event_field(event, "arguments", None) + if pending is not None and done_args is not None: + pending["arguments"] = str(done_args) + # Other function_call frames: the item itself lands on output_item.done. + + def _on_reasoning_delta(self, event: Any, event_type: str) -> None: + reasoning_text = _event_field(event, "delta", "") + if not reasoning_text or self.on_reasoning_delta is None: + return + summary_index = _event_field(event, "summary_index") + if summary_index is not None: + if self.active_summary_index is not None and summary_index != self.active_summary_index: + reasoning_text = f"\n\n{reasoning_text}" + self.active_summary_index = summary_index + self._safe(self.on_reasoning_delta, "on_reasoning_delta", reasoning_text) + + def _on_item_done(self, event: Any, event_type: str) -> None: + done_item = _event_field(event, "item") + if done_item is None: + return + self.output_items.append(done_item) + # Reuse the announced position when known; fresh tail sequence only for + # unannounced items. The .done event's own output_index wins over the announced one. + done_id = str(_event_field(done_item, "id", "")) + announced_sequence, announced_index = self.announced_output_order.get(done_id, (None, None)) + if announced_sequence is None: + announced_sequence = self.next_output_sequence + self.next_output_sequence += 1 + self.output_indexes.append(_event_field(event, "output_index", announced_index)) + self.output_sequences.append(announced_sequence) + # Confirmed by the authoritative done event; never settle it twice. + self.pending_function_calls.pop(done_id, None) + if _message_phase(done_item) == "commentary" and self.on_commentary_message is not None: + commentary_text = "".join(self.commentary_text_deltas).strip() + if not commentary_text: + content_parts = _event_field(done_item, "content", []) + if isinstance(content_parts, list): + commentary_text = "".join( + str(_event_field(part, "text", "") or "") + for part in content_parts + if _event_field(part, "type", "") == "output_text" + ).strip() + if commentary_text: + self._safe(self.on_commentary_message, "on_commentary_message", commentary_text) + self.commentary_text_deltas = [] + + def _on_terminal(self, event: Any, event_type: str) -> bool: + self.saw_terminal = True + resp_obj = _event_field(event, "response") + if resp_obj is not None: + self.terminal_usage = _event_field(resp_obj, "usage") + self.terminal_response_id = _event_field(resp_obj, "id") + rstatus = _event_field(resp_obj, "status") + if isinstance(rstatus, str): + self.terminal_status = rstatus + if event_type == "response.incomplete": + self.terminal_incomplete_details = _event_field(resp_obj, "incomplete_details") + elif event_type == "response.failed": + self.terminal_error = _event_field(resp_obj, "error") + if event_type == "response.completed": + self.saw_response_completed = True + self.terminal_status = self.terminal_status or event_type.removeprefix("response.") + return True + + # Exact-type handlers first, then substring-matched ones in priority order. + _EXACT_HANDLERS = { + "error": _on_error, "response.output_item.added": _on_item_added, "response.output_item.done": _on_item_done, + "response.completed": _on_terminal, "response.incomplete": _on_terminal, "response.failed": _on_terminal, + } + _FUZZY_HANDLERS = ( + (lambda t: "output_text.delta" in t, _on_text_delta), + (lambda t: "function_call" in t, _on_function_call), + (lambda t: "reasoning" in t and "delta" in t, _on_reasoning_delta), + ) + + def feed(self, event: Any) -> bool: + """Process one event; True when the stream hit a terminal frame.""" + event_type = _event_field(event, "type", "") + if not isinstance(event_type, str): + event_type = "" + handler = self._EXACT_HANDLERS.get(event_type) or next( + (h for matches, h in self._FUZZY_HANDLERS if matches(event_type)), None + ) + return bool(handler(self, event, event_type)) if handler is not None else False + + def _settled_output(self) -> List[Any]: + """Merge .done items with settled pending calls, keeping stream order.""" + indexed = list(zip(self.output_indexes, self.output_sequences, self.output_items)) + for pending in self.pending_function_calls.values(): + item = pending["item"] + indexed.append((pending.get("output_index"), pending["sequence"], SimpleNamespace( + type="function_call", + id=_event_field(item, "id", None), + call_id=_event_field(item, "call_id", None), + name=_event_field(item, "name", None), + # Empty/whitespace arguments become "{}" so zero-delta calls stay + # executable; malformed non-empty JSON passes through untouched. + arguments=(pending["arguments"] or "").strip() or "{}", + status="completed", + ))) + + # output_index is optional and a partial ordering over mixed indexed/unindexed + # entries is ill-defined: protocol order only when every entry has an index, else wire order. + if all(entry[0] is not None for entry in indexed): + try: + indexed.sort(key=lambda entry: entry[0]) + except TypeError: + pass # non-comparable index values: keep wire order + else: + indexed.sort(key=lambda entry: entry[1]) + return [entry[2] for entry in indexed] + + def result(self) -> SimpleNamespace: + # Prefer .done items; with only plain text deltas (no tool calls), + # synthesize a single message item for downstream normalization. + output: List[Any] = list(self.output_items) + if not output and self.text_deltas and not self.has_tool_calls: + output = [SimpleNamespace( + type="message", + role="assistant", + status="completed", + content=[SimpleNamespace(type="output_text", text="".join(self.text_deltas))], + )] + + # Done items stay authoritative; settlement only fills the gap left by + # backends that omit per-item done events on a successful completion. + if self.pending_function_calls and self.saw_response_completed: + output = self._settled_output() + + # No terminal frame AND no usable content = truncated / rejected stream, + # distinct from "completed with empty body" (what the SDK helper raised as RuntimeError). + if not self.saw_terminal and not output: + raise RuntimeError("Codex Responses stream did not emit a terminal response") + + return SimpleNamespace( + output=output, + output_text="".join(self.text_deltas), + usage=self.terminal_usage, + status=self.terminal_status, + id=self.terminal_response_id, + model=self.model, + incomplete_details=self.terminal_incomplete_details, + error=self.terminal_error, + ) + + +def _consume_codex_event_stream( + event_iter: Any, *, model: str, on_text_delta=None, on_reasoning_delta=None, + on_commentary_message=None, on_first_delta=None, on_event=None, interrupt_check=None, +) -> SimpleNamespace: + """Consume a Codex Responses SSE stream into a Response-shaped ``SimpleNamespace``. + + Result fields: ``output`` (items from ``output_item.done``, or a synthesized + message for plain-text turns), ``output_text``, ``usage``, ``status`` + (``completed`` when the stream ended with content but no terminal frame), + ``id``, ``incomplete_details``, ``error``, ``model`` (from kwargs; the wire + model name is not authoritative). + + Callbacks: ``on_text_delta(str)`` per output_text delta, suppressed once a + function_call is seen so tool-call turns don't bleed text into chat; + ``on_reasoning_delta(str)`` for reasoning and ``phase=analysis`` deltas (also + commentary when no commentary callback is given); ``on_commentary_message(str)`` + once per completed ``phase=commentary`` message, before any following tool item + executes; ``on_first_delta()`` one-shot on the first text delta; ``on_event(event)`` + every event, before any other processing; ``interrupt_check()`` True breaks the + loop early and may raise ``TimeoutError`` / ``InterruptedError`` for request + retirement that must not become a partial final response. + """ + assembler = _CodexResponseAssembler( + model=model, on_text_delta=on_text_delta, on_reasoning_delta=on_reasoning_delta, + on_commentary_message=on_commentary_message, on_first_delta=on_first_delta, + ) for event in event_iter: if on_event is not None: try: on_event(event) except (TimeoutError, InterruptedError): - # Control-flow signals from watchdog/cancellation hooks must - # propagate, not get swallowed as "debug noise". - raise + raise # watchdog / cancellation control flow must propagate except Exception: - # Genuine bugs in third-party debug/log hooks shouldn't break - # stream consumption. logger.debug("Codex stream on_event hook raised", exc_info=True) - if interrupt_check is not None and interrupt_check(): + if (interrupt_check is not None and interrupt_check()) or assembler.feed(event): break - - event_type = _event_field(event, "type", "") - if not isinstance(event_type, str): - event_type = "" - - # ``error`` SSE frames carry the provider's real failure reason - # (subscription / quota / model-not-available / rejected-reasoning-replay) - # but never appear in the terminal set. Surface them as a structured - # exception so the credential pool + error classifier see the body. - if event_type == "error": - _raise_stream_error(event) - - # Track the phase of the active streamed message item. Codex/Harmony - # ``commentary``/``analysis`` text is mid-turn preamble/progress - # narration, never the final answer. We still collect completed output - # items for replay, but route those deltas to the reasoning callback so - # they display like thinking text instead of assistant content. - if event_type == "response.output_item.added": - item = _event_field(event, "item") - item_type = _item_field(item, "type", "") - if item_type == "message": - phase = _item_field(item, "phase", None) - active_message_phase = phase.strip().lower() if isinstance(phase, str) else None - if active_message_phase == "commentary": - commentary_text_deltas = [] - else: - active_message_phase = None - # First-observed ordering metadata for EVERY announced item (not - # just function calls): when this item later lands via - # output_item.done, the done path must reuse the announced - # sequence/index instead of allocating a fresh tail position, or - # a mixed announced/pending stream without output_index values - # reorders the calls (review P1 on PR #92767). - item_id = str(_item_field(item, "id", "")) - if item_id and item_id not in announced_output_order: - announced_output_order[item_id] = ( - next_output_sequence, - _event_field(event, "output_index", None), - ) - next_output_sequence += 1 - if "function_call" in str(item_type): - has_tool_calls = True - if item_id: - announced_sequence, announced_index = announced_output_order[item_id] - # Seed from the announced item's own arguments when the - # backend attaches them up front, and remember the stream - # position so a settled call keeps its place in the output. - pending_function_calls[item_id] = { - "item": item, - "arguments": str(_item_field(item, "arguments", "") or ""), - "output_index": announced_index, - "sequence": announced_sequence, - } - continue - - if "output_text.delta" in event_type or event_type == "response.output_text.delta": - delta_text = _event_field(event, "delta", "") - if delta_text and active_message_phase == "commentary": - commentary_text_deltas.append(delta_text) - # Preserve CLI/backward compatibility when no first-class - # commentary consumer is installed. - if on_commentary_message is None and on_reasoning_delta is not None: - try: - on_reasoning_delta(delta_text) - except Exception: - logger.debug("Codex stream on_reasoning_delta raised", exc_info=True) - elif delta_text and active_message_phase == "analysis": - if on_reasoning_delta is not None: - try: - on_reasoning_delta(delta_text) - except Exception: - logger.debug("Codex stream on_reasoning_delta raised", exc_info=True) - elif delta_text: - collected_text_deltas.append(delta_text) - if not has_tool_calls: - if not first_delta_fired: - first_delta_fired = True - if on_first_delta is not None: - try: - on_first_delta() - except Exception: - logger.debug("Codex stream on_first_delta raised", exc_info=True) - if on_text_delta is not None: - try: - on_text_delta(delta_text) - except Exception: - logger.debug("Codex stream on_text_delta raised", exc_info=True) - continue - - if "function_call" in event_type: - has_tool_calls = True - # Accumulate streamed argument deltas for calls announced via - # output_item.added, so a stream that completes without per-item - # done events can still be settled from accumulated state. - if "delta" in event_type: - delta_args = _event_field(event, "delta", "") - pending = pending_function_calls.get(str(_event_field(event, "item_id", ""))) - if pending is not None and delta_args: - pending["arguments"] += delta_args - continue - if event_type.endswith("function_call_arguments.done"): - done_args = _event_field(event, "arguments", None) - pending = pending_function_calls.get(str(_event_field(event, "item_id", ""))) - if pending is not None and done_args is not None: - # Per-item arguments.done is authoritative for the - # accumulated string when the item itself never lands. - # An explicit empty string (zero-argument call) counts as - # authoritative; only a missing field leaves the streamed - # deltas in place. - pending["arguments"] = str(done_args) - continue - # other function_call frames fall through — function_call items still get added on output_item.done - - if "reasoning" in event_type and "delta" in event_type: - reasoning_text = _event_field(event, "delta", "") - if reasoning_text and on_reasoning_delta is not None: - # Summary parts stream one after another with no separator of - # their own; summary_index is the boundary the wire gives us. - summary_index = _event_field(event, "summary_index") - if ( - summary_index is not None - and active_summary_index is not None - and summary_index != active_summary_index - ): - reasoning_text = f"\n\n{reasoning_text}" - if summary_index is not None: - active_summary_index = summary_index - try: - on_reasoning_delta(reasoning_text) - except Exception: - logger.debug("Codex stream on_reasoning_delta raised", exc_info=True) - continue - - if event_type == "response.output_item.done": - done_item = _event_field(event, "item") - if done_item is not None: - collected_output_items.append(done_item) - # Reuse the first-observed position when this item was - # announced earlier via output_item.added; a fresh tail - # sequence is allocated only for genuinely unannounced items. - # The .done event's own output_index wins when present, with - # the announced index as its fallback. - done_id = str(_item_field(done_item, "id", "")) - announced_sequence, announced_index = announced_output_order.get( - done_id, (None, None) - ) - done_index = _event_field(event, "output_index", None) - if done_index is None: - done_index = announced_index - if announced_sequence is None: - announced_sequence = next_output_sequence - next_output_sequence += 1 - collected_output_indexes.append(done_index) - collected_output_sequences.append(announced_sequence) - # Confirmed by the authoritative per-item done event; remove - # from pending so it is not settled twice. - pending_function_calls.pop(done_id, None) - done_phase = _item_field(done_item, "phase", None) - done_phase = done_phase.strip().lower() if isinstance(done_phase, str) else None - if done_phase == "commentary" and on_commentary_message is not None: - commentary_text = "".join(commentary_text_deltas).strip() - if not commentary_text: - content_parts = _item_field(done_item, "content", []) - if isinstance(content_parts, list): - commentary_text = "".join( - str(_item_field(part, "text", "") or "") - for part in content_parts - if _item_field(part, "type", "") == "output_text" - ).strip() - if commentary_text: - try: - on_commentary_message(commentary_text) - except Exception: - logger.debug( - "Codex stream on_commentary_message raised", - exc_info=True, - ) - commentary_text_deltas = [] - continue - - if event_type in _TERMINAL_EVENT_TYPES: - saw_terminal = True - resp_obj = _event_field(event, "response") - if resp_obj is not None: - terminal_usage = getattr(resp_obj, "usage", None) - if terminal_usage is None and isinstance(resp_obj, dict): - terminal_usage = resp_obj.get("usage") - rid = getattr(resp_obj, "id", None) - if rid is None and isinstance(resp_obj, dict): - rid = resp_obj.get("id") - terminal_response_id = rid - rstatus = getattr(resp_obj, "status", None) - if rstatus is None and isinstance(resp_obj, dict): - rstatus = resp_obj.get("status") - if isinstance(rstatus, str): - terminal_status = rstatus - if event_type == "response.incomplete": - terminal_incomplete_details = getattr(resp_obj, "incomplete_details", None) - if terminal_incomplete_details is None and isinstance(resp_obj, dict): - terminal_incomplete_details = resp_obj.get("incomplete_details") - if event_type == "response.failed": - terminal_error = getattr(resp_obj, "error", None) - if terminal_error is None and isinstance(resp_obj, dict): - terminal_error = resp_obj.get("error") - if event_type == "response.completed": - saw_response_completed = True - terminal_status = terminal_status or "completed" - elif event_type == "response.incomplete": - terminal_status = terminal_status or "incomplete" - elif event_type == "response.failed": - terminal_status = terminal_status or "failed" - # Stop on terminal event. - break - - # Build the final output list. Prefer items observed via output_item.done; - # if none arrived but we streamed plain text deltas (no tool calls), synthesize - # a single message item so downstream normalization has something to work with. - if collected_output_items: - output = list(collected_output_items) - elif collected_text_deltas and not has_tool_calls: - assembled = "".join(collected_text_deltas) - output = [SimpleNamespace( - type="message", - role="assistant", - status="completed", - content=[SimpleNamespace(type="output_text", text=assembled)], - )] - else: - output = [] - - # Settle function calls that were announced via output_item.added and - # streamed argument deltas but never confirmed by output_item.done: some - # OpenAI-compatible backends omit per-item done events on a successful - # completion (anomalyco/opencode#37159). Done items stay authoritative; - # this only fills the gap so the call executes instead of vanishing. - if pending_function_calls and saw_response_completed: - # Assemble settled calls and .done items in output_index order instead - # of appending at the tail: a pending call that streamed before a later - # .done item must keep its position, or dependent side effects invert. - indexed = [ - (index, sequence, position, item) - for position, (index, sequence, item) in enumerate( - zip( - collected_output_indexes, - collected_output_sequences, - collected_output_items, - ) - ) - ] - for position, pending in enumerate(pending_function_calls.values(), start=len(indexed)): - item = pending["item"] - # Canonicalize empty/whitespace arguments so zero-delta calls stay - # executable; malformed non-empty JSON passes through untouched and - # stays rejected by downstream argument parsing. - arguments = (pending["arguments"] or "").strip() or "{}" - indexed.append((pending.get("output_index"), pending["sequence"], position, SimpleNamespace( - type="function_call", - id=_item_field(item, "id", None), - call_id=_item_field(item, "call_id", None), - name=_item_field(item, "name", None), - arguments=arguments, - status="completed", - ))) - - # output_index is optional in compatible Responses streams. A partial - # ordering (sorting indexed entries while interleaving unindexed ones) - # is not well-defined and can produce contradictory comparisons. Keep - # the observed wire order whenever any index is missing; use the - # protocol ordering only when every entry provides an index. - if all(entry[0] is not None for entry in indexed): - try: - indexed.sort(key=lambda entry: entry[0]) - except TypeError: - # Preserve wire order if a backend sends non-comparable index - # values instead of integers. - pass - else: - indexed.sort(key=lambda entry: entry[1]) - output = [entry[3] for entry in indexed] - - # If the stream ended without any terminal event AND produced no usable - # content (no items, no text deltas), surface that as a RuntimeError so - # callers can distinguish "stream truncated mid-flight / provider rejected - # the call" from "stream completed with empty body". This preserves the - # signal the SDK's high-level helper used to raise as - # ``RuntimeError("Didn't receive a `response.completed` event.")``. - if not saw_terminal and not output: - raise RuntimeError( - "Codex Responses stream did not emit a terminal response" - ) - - assembled_text = "".join(collected_text_deltas) - - final = SimpleNamespace( - output=output, - output_text=assembled_text, - usage=terminal_usage, - status=terminal_status, - id=terminal_response_id, - model=model, - incomplete_details=terminal_incomplete_details, - error=terminal_error, - ) - return final + return assembler.result() -def _sanitize_consumer_codex_request( - agent: Any, - request: dict[str, Any], -) -> dict[str, Any]: - """Drop fields the ChatGPT OAuth Codex endpoint does not accept. +def _sanitize_consumer_codex_request(agent: Any, request: dict[str, Any]) -> dict[str, Any]: + """Drop fields the ChatGPT OAuth Codex endpoint rejects, at the final wire boundary. - This guard intentionally lives at the final wire boundary, after Relay or - other request middleware has had a chance to transform the request. The - normal transport builder already omits ``prompt_cache_retention`` for this - endpoint, but a late mutation must not be allowed to turn a valid tool - follow-up into a non-retryable HTTP 400. - - Explicit ``request_overrides`` are subject to the same endpoint contract: - unsupported retention is dropped with a warning instead of being sent and - rejected by the provider. The check covers both the top-level kwarg and a - nested ``extra_body`` entry — the OpenAI SDK merges ``extra_body`` into - the outgoing JSON body, so either shape reaches the endpoint. + Runs after Relay / request middleware and explicit ``request_overrides`` so a + late ``prompt_cache_retention`` (top-level or nested in ``extra_body``, which + the SDK merges into the body) cannot turn a valid follow-up into an HTTP 400. """ sanitized = dict(request) - # Resolved defensively on purpose: run_codex_stream is also driven with - # lightweight stand-in agents that carry only the attributes a given path - # needs (see tests/agent/test_codex_request_transport_diagnostics.py), so a - # bare agent._is_codex_backend() here would raise AttributeError on them. + # getattr: run_codex_stream is also driven with stand-in agents carrying only the attrs a path needs. backend_predicate = getattr(agent, "_is_codex_backend", None) - is_consumer_codex = ( - bool(backend_predicate()) if callable(backend_predicate) else False - ) - if not is_consumer_codex: + if not (callable(backend_predicate) and bool(backend_predicate())): return sanitized dropped_from: list[str] = [] if "prompt_cache_retention" in sanitized: - sanitized.pop("prompt_cache_retention") + del sanitized["prompt_cache_retention"] dropped_from.append("top-level") - # The OpenAI SDK merges ``extra_body`` into the outgoing JSON body, so a - # nested ``extra_body.prompt_cache_retention`` reaches the endpoint just - # like the top-level field would. Copy before editing — the caller's - # mapping must not be mutated — and drop the mapping when it empties. + # Copy before editing (caller's mapping must not mutate); drop when emptied. extra_body = sanitized.get("extra_body") if isinstance(extra_body, dict) and "prompt_cache_retention" in extra_body: - extra_body = dict(extra_body) - extra_body.pop("prompt_cache_retention") + extra_body = {k: v for k, v in extra_body.items() if k != "prompt_cache_retention"} if extra_body: sanitized["extra_body"] = extra_body else: @@ -1510,62 +1034,44 @@ def _sanitize_consumer_codex_request( dropped_from.append("extra_body") if dropped_from: logger.warning( - "Dropped unsupported prompt_cache_retention at consumer Codex " - "wire boundary (model=%s, via %s).", - sanitized.get("model", getattr(agent, "model", "unknown")), - ", ".join(dropped_from), + "Dropped unsupported prompt_cache_retention at consumer Codex wire boundary (model=%s, via %s).", + sanitized.get("model", getattr(agent, "model", "unknown")), ", ".join(dropped_from), ) return sanitized -# Bulk request fields that carry the conversation payload. Everything else in -# the request is scalar configuration the SDK transform handles in microseconds. +# Bulk request fields carrying the conversation payload; the rest is scalar +# config the SDK transform handles in microseconds. _SDK_TRANSFORM_BYPASS_FIELDS = ("input", "tools") def _is_plain_json_data(value: Any) -> bool: """True when ``value`` is composed purely of JSON wire types. - The SDK's request transform exists to convert typed params (TypedDict - key aliases, pydantic models, ``PropertyInfo`` formats) into wire - format. Hermes assembles Codex payloads from JSON round-trips, so they - are already wire format — but that is only provable when every node is - a plain JSON type. Anything else must keep the typed SDK path. + Hermes builds Codex payloads from JSON round-trips, so they are provably wire + format only when every node is plain JSON; anything else (pydantic models, + generators) must keep the typed SDK path. """ if value is None or isinstance(value, (str, int, float, bool)): return True if isinstance(value, dict): - return all( - isinstance(key, str) and _is_plain_json_data(item) - for key, item in value.items() - ) + return all(isinstance(key, str) and _is_plain_json_data(item) for key, item in value.items()) if isinstance(value, list): return all(_is_plain_json_data(item) for item in value) return False def _bypass_sdk_request_transform(stream_kwargs: dict) -> dict: - """Route bulk payload fields around the SDK's ``maybe_transform`` (#93650). + """Route bulk payload fields around the SDK's ``maybe_transform``. - ``responses.create`` re-walks the entire request body against the - ``ResponseCreateParams`` union graph before any byte leaves the process. - That walk runs with the GIL held, and #93650 documents it wedging for - 12+ hours on a ~1.4 MB conversation — starving every other thread, - including the TTFB/stale watchdogs whose job is to rescue this exact - call. Because the hang is client-side and pre-network, no socket kill - can unblock it. - - The SDK merges ``extra_body`` into the JSON body *after* the transform - (``_base_client._build_request``), so moving the already-wire-format - bulk fields there skips the walk entirely and produces a byte-identical - request. Fields containing anything that is not plain JSON data (e.g. - pydantic models, generators) stay on the typed path, which still needs - the transform. Set HERMES_CODEX_SDK_TRANSFORM=1 to restore the pre-fix - behavior. + ``responses.create`` re-walks the whole body against the ResponseCreateParams + union graph with the GIL held — multi-MB conversations can wedge for hours and + starve the watchdogs (client-side, pre-network: no socket kill helps). The SDK + merges ``extra_body`` AFTER the transform, so moving already-wire-format bulk + fields there skips the walk and yields a byte-identical request. + HERMES_CODEX_SDK_TRANSFORM=1 disables. """ - if os.environ.get("HERMES_CODEX_SDK_TRANSFORM", "").strip().lower() in { - "1", "true", "yes", "on" - }: + if os.environ.get("HERMES_CODEX_SDK_TRANSFORM", "").strip().lower() in {"1", "true", "yes", "on"}: return stream_kwargs moved = { @@ -1577,14 +1083,11 @@ def _bypass_sdk_request_transform(stream_kwargs: dict) -> dict: if not moved: return stream_kwargs - bypassed = { - key: value for key, value in stream_kwargs.items() if key not in moved - } + bypassed = {key: value for key, value in stream_kwargs.items() if key not in moved} extra_body = bypassed.get("extra_body") merged = dict(extra_body) if isinstance(extra_body, dict) else {} for field, value in moved.items(): - # An explicit caller-provided extra_body entry keeps precedence, - # matching what the SDK's post-transform merge would have done. + # An explicit caller-provided extra_body entry keeps precedence (SDK post-transform merge). merged.setdefault(field, value) bypassed["extra_body"] = merged return bypassed @@ -1593,249 +1096,167 @@ def _bypass_sdk_request_transform(stream_kwargs: dict) -> dict: def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta=None): """Execute one streaming Responses API request and return the final response. - Uses ``responses.create(stream=True)`` (low-level raw event iteration) - rather than the high-level ``responses.stream(...)`` helper. This makes - us structurally immune to backend drift in the ``response.completed`` - payload shape — we never let the SDK reconstruct a typed object from - the terminal event's ``output`` field. + Uses ``responses.create(stream=True)`` raw event iteration rather than the + ``responses.stream(...)`` helper, so the SDK never reconstructs a typed + object from the terminal event's ``output`` field. """ import httpx as _httpx from openai import APIConnectionError as _APIConnectionError from agent import relay_llm + transport_errors = (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) active_client = client or agent._ensure_primary_openai_client(reason="codex_stream_direct") max_stream_retries = 1 + model = api_kwargs.get("model") # Accumulate streamed text so callers / compat shims can read it. agent._codex_streamed_text_parts: list = [] - # Retirement token for THIS request, installed by - # ``interruptible_api_call`` before it hands off to the worker thread. When - # a watchdog (TTFB / stream-idle / stale-call) kills the connection it - # clears the agent-level token, so a worker that is still draining frames - # can tell it has been retired. ``None`` means no watchdog owns this call - # (auxiliary callers drive this function directly) — then every check - # passes and behavior is unchanged. + # Retirement token for THIS request, installed by ``interruptible_api_call``. + # A watchdog (TTFB / stream-idle / stale-call) that kills the connection + # clears the agent-level token, so a worker still draining frames can tell + # it was retired. ``None`` = no watchdog owns this call; every check passes. request_token = getattr(agent, "_active_codex_stream_request_token", None) + # Delta-sink claim for the CURRENT physical attempt (None until the stream opens). + writer_token = {"value": None} def _request_is_current() -> bool: - if request_token is None: - return True - return getattr(agent, "_active_codex_stream_request_token", None) is request_token + return request_token is None or getattr(agent, "_active_codex_stream_request_token", None) is request_token + + def _fenced(fn: Callable[[Any], None]) -> Callable[[Any], None]: + """Wrap a callback so a retired request's late frames never reach the agent.""" + return lambda value: fn(value) if _request_is_current() else None def _on_text_delta(text: str) -> None: - if not _request_is_current(): - return agent._codex_streamed_text_parts.append(text) agent._fire_stream_delta(text) - def _on_reasoning_delta(text: str) -> None: - if not _request_is_current(): - return - agent._fire_reasoning_delta(text) - - def _on_commentary_message(text: str) -> None: - if not _request_is_current(): - return - agent._fire_streamed_codex_commentary(text) - def _on_event(event: Any) -> None: - if not _request_is_current(): - return - # TTFB watchdog and activity touch — runs once per SSE event. + # TTFB watchdog and activity touch — once per SSE event. agent._codex_stream_last_event_ts = time.time() agent._touch_activity("receiving stream response") + def _interrupt_or_superseded() -> bool: + # A retired request must NOT break out of the consume loop: that returns a + # partial ``final`` (status defaults to "completed") the caller would persist + # as finished. Raise so the watchdog's own TimeoutError is what the retry path sees. + if not _request_is_current(): + raise TimeoutError("Codex Responses stream request retired before terminal response") + return bool(agent._interrupt_requested) + + def _open_codex_stream(next_api_kwargs: dict[str, Any]): + stream_kwargs = _sanitize_consumer_codex_request(agent, next_api_kwargs) + stream_kwargs["stream"] = True + return active_client.responses.create(**_bypass_sdk_request_transform(stream_kwargs)) + + def _log_failure(exc: BaseException) -> None: + _log_codex_request_failure(agent, exc, stream_opened=writer_token["value"] is not None) + + def _codex_stream_created(_raw_stream: Any) -> None: + # Claim the delta sink for THIS physical attempt; a newer attempt + # supersedes this token and fences late deltas out of the turn. + writer_token["value"] = claim_stream_writer(agent) + + def _accept_codex_chunk(_chunk: Any) -> bool: + token = writer_token["value"] + if token is None or stream_writer_is_current(agent, token): + return True + logger.warning( + "Codex streaming attempt superseded by a newer stream; stopping consumption to preserve " + "the single-writer invariant (model=%s).", + api_kwargs.get("model", "unknown"), + ) + return False + + def _drain_for_finalizer(event_stream: Any) -> None: + # ``final`` is already assembled; draining the rest of the iterator + # only lets Relay run its response finalizer. A transport error here + # must NOT discard the completed, already-billed response or start + # a new physical request — warn and return it. + try: + for _ignored in event_stream: + pass + except (*transport_errors, _APIConnectionError) as exc: + if not isinstance(exc, transport_errors): + _log_failure(exc) + logger.warning( + "Codex Responses stream transport finalization failed after a terminal response was already " + "received; returning the completed response instead of retrying. %s error=%s", + agent._client_log_context(), exc, + ) + + on_commentary_message = ( + _fenced(lambda text: agent._fire_streamed_codex_commentary(text)) + if getattr(agent, "interim_assistant_callback", None) is not None and getattr(agent, "show_commentary", True) + else None + ) + call_role = ( + "delegated" if getattr(agent, "is_subagent", False) + else "fallback" if int(getattr(agent, "_fallback_index", 0) or 0) > 0 + else "primary" + ) + for attempt in range(max_stream_retries + 1): if agent._interrupt_requested: raise InterruptedError("Agent interrupted before Codex stream retry") - intercepted_events = [] - writer_token = {"value": None} - - def _open_codex_stream(next_api_kwargs: dict[str, Any]): - stream_kwargs = _sanitize_consumer_codex_request( - agent, - next_api_kwargs, - ) - stream_kwargs["stream"] = True - stream_kwargs = _bypass_sdk_request_transform(stream_kwargs) - return active_client.responses.create(**stream_kwargs) - - def _codex_stream_created(_raw_stream: Any) -> None: - # Claim the delta sink for THIS physical attempt. A newer attempt - # supersedes this token and fences late deltas out of the turn. - writer_token["value"] = claim_stream_writer(agent) - - def _accept_codex_chunk(_chunk: Any) -> bool: - token = writer_token["value"] - if token is None or stream_writer_is_current(agent, token): - return True - logger.warning( - "Codex streaming attempt superseded by a newer stream; " - "stopping consumption to preserve the single-writer " - "invariant (model=%s).", - api_kwargs.get("model", "unknown"), - ) - return False - - def _finalize_codex_stream() -> Any: - return _consume_codex_event_stream( - list(intercepted_events), - model=api_kwargs.get("model"), - ) - - try: - event_stream = relay_llm.stream( - dict(api_kwargs), - _open_codex_stream, - session_id=str(getattr(agent, "session_id", "") or ""), - name=str(getattr(agent, "provider", "") or "codex"), - model_name=str(api_kwargs.get("model") or ""), - finalizer=_finalize_codex_stream, - on_stream_created=_codex_stream_created, - on_chunk=intercepted_events.append, - chunk_adapter=lambda chunk: chunk, - accept_chunk=_accept_codex_chunk, - completed_response_predicate=lambda response: bool( - hasattr(response, "output") and not hasattr(response, "__iter__") - ), - metadata={ - "api_mode": "codex_responses", - "api_request_id": getattr(agent, "_current_api_request_id", None), - "call_role": ( - "delegated" - if getattr(agent, "is_subagent", False) - else "fallback" - if int(getattr(agent, "_fallback_index", 0) or 0) > 0 - else "primary" - ), - "retry_count": attempt, - }, - defer_logical_completion=True, - ) - except ( - _httpx.RemoteProtocolError, - _httpx.ReadTimeout, - _httpx.ConnectError, - ConnectionError, - ) as exc: - if attempt < max_stream_retries: - logger.debug( - "Codex Responses stream connect failed (attempt %s/%s); " - "retrying. %s error=%s", - attempt + 1, - max_stream_retries + 1, - agent._client_log_context(), - exc, - ) - continue - _log_codex_request_failure( - agent, - exc, - stream_opened=writer_token["value"] is not None, - ) - raise - except _APIConnectionError as exc: - _log_codex_request_failure( - agent, - exc, - stream_opened=writer_token["value"] is not None, - ) - raise - - def _interrupt_or_superseded() -> bool: - # A retired request must NOT break out of the consume loop: breaking - # returns the partial `final` (status defaults to "completed"), which - # the caller persists as a finished assistant turn. Raise so the - # watchdog's own TimeoutError is what the retry path sees. - if not _request_is_current(): - raise TimeoutError( - "Codex Responses stream request retired before terminal response" - ) - return bool(agent._interrupt_requested) - + intercepted_events: list = [] + writer_token["value"] = None + event_stream = None try: try: + event_stream = relay_llm.stream( + dict(api_kwargs), + _open_codex_stream, + session_id=str(getattr(agent, "session_id", "") or ""), + name=str(getattr(agent, "provider", "") or "codex"), + model_name=str(model or ""), + finalizer=lambda: _consume_codex_event_stream(list(intercepted_events), model=model), + on_stream_created=_codex_stream_created, + on_chunk=intercepted_events.append, + chunk_adapter=lambda chunk: chunk, + accept_chunk=_accept_codex_chunk, + completed_response_predicate=lambda r: bool(hasattr(r, "output") and not hasattr(r, "__iter__")), + metadata={ + "api_mode": "codex_responses", + "api_request_id": getattr(agent, "_current_api_request_id", None), + "call_role": call_role, + "retry_count": attempt, + }, + defer_logical_completion=True, + ) final = _consume_codex_event_stream( event_stream, - model=api_kwargs.get("model"), - on_text_delta=_on_text_delta, - on_reasoning_delta=_on_reasoning_delta, - on_commentary_message=( - _on_commentary_message - if ( - getattr(agent, "interim_assistant_callback", None) is not None - and getattr(agent, "show_commentary", True) - ) - else None - ), + model=model, + on_text_delta=_fenced(_on_text_delta), + on_reasoning_delta=_fenced(lambda text: agent._fire_reasoning_delta(text)), + on_commentary_message=on_commentary_message, on_first_delta=on_first_delta, - on_event=_on_event, + on_event=_fenced(_on_event), interrupt_check=_interrupt_or_superseded, ) - except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc: - if attempt < max_stream_retries: - logger.debug( - "Codex Responses stream transport failed mid-iteration " - "(attempt %s/%s); retrying. %s error=%s", - attempt + 1, max_stream_retries + 1, - agent._client_log_context(), exc, - ) - continue - _log_codex_request_failure( - agent, - exc, - stream_opened=writer_token["value"] is not None, + except transport_errors as exc: + if attempt >= max_stream_retries: + _log_failure(exc) + raise + logger.debug( + "Codex Responses stream connect failed (attempt %s/%s); retrying. %s error=%s" + if event_stream is None + else "Codex Responses stream transport failed mid-iteration (attempt %s/%s); retrying. %s error=%s", + attempt + 1, max_stream_retries + 1, agent._client_log_context(), exc, ) - raise + continue except RuntimeError: - if event_stream.final_response is not None: + # The consumer's "no terminal response" signal; Relay may still + # hold a completed response assembled by its finalizer. + if event_stream is not None and event_stream.final_response is not None: return event_stream.final_response raise except _APIConnectionError as exc: - _log_codex_request_failure( - agent, - exc, - stream_opened=writer_token["value"] is not None, - ) + _log_failure(exc) raise - # A terminal response has already been assembled at this point - # (``final`` is built), so a transport error while draining the - # rest of the iterator — done only to let Relay run its response - # finalizer — must NOT discard it or trigger a new physical - # request. Record it as a non-fatal finalization warning and - # still return the already-completed, already-billed response. if not agent._interrupt_requested: - try: - for _ignored in event_stream: - pass - except ( - _httpx.RemoteProtocolError, - _httpx.ReadTimeout, - _httpx.ConnectError, - ConnectionError, - ) as exc: - logger.warning( - "Codex Responses stream transport finalization failed " - "after a terminal response was already received; " - "returning the completed response instead of " - "retrying. %s error=%s", - agent._client_log_context(), exc, - ) - except _APIConnectionError as exc: - _log_codex_request_failure( - agent, - exc, - stream_opened=writer_token["value"] is not None, - ) - logger.warning( - "Codex Responses stream transport finalization failed " - "after a terminal response was already received; " - "returning the completed response instead of " - "retrying. %s error=%s", - agent._client_log_context(), exc, - ) + _drain_for_finalizer(event_stream) if final.status in {"incomplete", "failed"}: logger.warning( @@ -1848,36 +1269,24 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta return final finally: - close_fn = getattr(event_stream, "close", None) + close_fn = getattr(event_stream, "close", None) # None while connect never succeeded if callable(close_fn): try: close_fn() except Exception: - # A failed close can leave this response's connection - # checked out of the httpx pool while the caller's finally - # reports a reuse-reason close (e.g. interrupt_check broke - # the event loop with collected output) — caching the - # client with the leaked connection. Poison the slot so - # that close really closes the pool (owner-thread abort; - # mirrors the chat-streaming interrupt-break handling). - # ``client is None`` means the shared primary client, - # which is never reuse-cached and must not have its - # sockets force-shut here. + # A failed close can leave this response's connection checked + # out of the httpx pool while the caller's finally reports a + # reuse-reason close — caching a client with a leaked + # connection. Poison the slot so close really closes the pool. + # ``client is None`` is the shared primary client, which is + # never reuse-cached and must not be force-shut here. if client is not None: - agent._abort_request_openai_client( - active_client, reason="codex_stream_close_failed" - ) + agent._abort_request_openai_client(active_client, reason="codex_stream_close_failed") def run_codex_create_stream_fallback(agent, api_kwargs: dict, client: Any = None): - """Backward-compatible alias for the unified event-driven path. - - Historically this was the fallback when the SDK's high-level - ``responses.stream(...)`` helper raised on shape drift. The primary - path now does exactly what the fallback did, so this just forwards. - Kept as a public symbol because tests and a small number of call sites - still reference it by name. - """ + """Backward-compatible alias: the primary path now does what this fallback did. + Kept public because tests and a few call sites reference it by name.""" return run_codex_stream(agent, api_kwargs, client=client)