refactor(agent/transports): compact ACP/SSL/stream helper modules (copilot_acp_client, acp_openai_bridge, ssl_*, stream_*, jiter_preload)

This commit is contained in:
Teknium
2026-09-02 13:18:33 -07:00
parent d2b3e545cb
commit a53d886584
7 changed files with 235 additions and 615 deletions
+27 -86
View File
@@ -1,15 +1,10 @@
"""Stream diagnostics — per-attempt counters, exception chains, retry logging.
When a streaming chat-completions request dies mid-response, we want to
know why: which Cloudflare edge served the request, which OpenRouter
downstream provider answered, how many bytes/chunks we got before the
drop, the HTTP status, the underlying httpx error class. These helpers
collect that info and emit it both to ``agent.log`` (full detail) and to
the user-facing status line (compact).
All helpers are extracted from :class:`AIAgent` for cleanliness.
``run_agent`` keeps thin forwarder methods so existing call sites and
tests that patch ``run_agent.<helper>`` keep working.
When a streaming request dies mid-response these helpers record WHY (which CF
edge / OpenRouter downstream served it, bytes+chunks before the drop, HTTP
status, underlying httpx error class) to ``agent.log`` in full and to the
user-facing status line compactly. ``run_agent`` keeps thin forwarders so
existing call sites and tests patching ``run_agent.<helper>`` keep working.
"""
from __future__ import annotations
@@ -20,10 +15,7 @@ from typing import Any, Dict, List, Optional
logger = logging.getLogger(__name__)
# Per-attempt stream diagnostic headers. Lowercased; httpx returns
# CIMultiDict so case-insensitive lookups already work, but we read .get()
# on the dict from agent.log for free-form post-hoc analysis.
# Lowercased upstream headers captured per attempt for post-hoc analysis.
STREAM_DIAG_HEADERS = (
"cf-ray",
"cf-cache-status",
@@ -39,12 +31,7 @@ STREAM_DIAG_HEADERS = (
def stream_diag_init() -> Dict[str, Any]:
"""Return a fresh per-attempt diagnostic dict.
Mutated in-place by the streaming functions and read from the retry
block when a stream dies. Lives on ``request_client_holder`` so it
survives across the closure boundary.
"""
"""Fresh per-attempt diagnostic dict; mutated in place by the streaming functions and read by the retry block."""
return {
"started_at": time.time(),
"first_chunk_at": None,
@@ -56,12 +43,7 @@ def stream_diag_init() -> Dict[str, Any]:
def stream_diag_capture_response(agent: Any, diag: Dict[str, Any], http_response: Any) -> None:
"""Snapshot interesting headers + HTTP status from the live stream.
Called once at stream open (before iterating chunks) so the metadata
survives even if the stream dies before any chunk arrives. Failures
are swallowed — diag is best-effort.
"""
"""Snapshot headers + HTTP status at stream open (so they survive a drop before the first chunk). Best-effort."""
if http_response is None or not isinstance(diag, dict):
return
try:
@@ -71,14 +53,12 @@ def stream_diag_capture_response(agent: Any, diag: Dict[str, Any], http_response
try:
headers = getattr(http_response, "headers", None) or {}
captured: Dict[str, str] = {}
# Allow per-agent override of the headers list (back-compat).
target_headers = getattr(agent, "_STREAM_DIAG_HEADERS", STREAM_DIAG_HEADERS)
for name in target_headers:
# Per-agent override of the headers list (back-compat).
for name in getattr(agent, "_STREAM_DIAG_HEADERS", STREAM_DIAG_HEADERS):
try:
val = headers.get(name)
if val:
# Truncate single-value to keep log lines bounded.
captured[name] = str(val)[:120]
captured[name] = str(val)[:120] # keep log lines bounded
except Exception:
continue
diag["headers"] = captured
@@ -87,14 +67,11 @@ def stream_diag_capture_response(agent: Any, diag: Dict[str, Any], http_response
def flatten_exception_chain(error: BaseException) -> str:
"""Return a compact ``Outer(msg) <- Inner(msg) <- ...`` rendering.
"""Compact ``Outer(msg) <- Inner(msg) <- ...`` rendering.
OpenAI SDK wraps httpx errors as ``APIConnectionError`` /
``APIError`` and only the wrapper's class is visible at the catch
site — but the underlying ``RemoteProtocolError`` /
``ConnectError`` / ``ReadError`` is what tells us WHY the stream
died. Walks ``__cause__`` then ``__context__`` (deduped, max 4
deep) to surface the chain in one line.
The OpenAI SDK wraps httpx errors so only the wrapper class is visible at the
catch site; the inner RemoteProtocolError/ConnectError/ReadError says WHY the
stream died. Walks ``__cause__`` then ``__context__`` (deduped, max 4 deep).
"""
seen: List[BaseException] = []
link: Optional[BaseException] = error
@@ -102,9 +79,7 @@ def flatten_exception_chain(error: BaseException) -> str:
if link in seen:
break
seen.append(link)
nxt = getattr(link, "__cause__", None) or getattr(
link, "__context__", None
)
nxt = getattr(link, "__cause__", None) or getattr(link, "__context__", None)
if nxt is None or nxt is link:
break
link = nxt
@@ -127,19 +102,12 @@ def log_stream_retry(
mid_tool_call: bool,
diag: Optional[Dict[str, Any]] = None,
) -> None:
"""Record a transient stream-drop and retry to ``agent.log``.
"""Structured WARNING to ``agent.log`` for a transient stream drop + retry.
Always logs a structured WARNING so users have a breadcrumb regardless
of UI verbosity. Subagents in particular benefit because their
retries no longer spam the parent's terminal — but the file log keeps
full detail (provider, error class, attempt, base_url, subagent_id).
When *diag* is provided (the per-attempt stream-diagnostic dict from
:func:`stream_diag_init`), the WARNING also captures upstream headers
(cf-ray, x-openrouter-provider, x-openrouter-id), HTTP status, bytes
streamed before the drop, and elapsed time on the dying attempt.
These are the breadcrumbs needed to answer "is one CF edge / one
downstream provider responsible, or is it random across runs?"
Always logged regardless of UI verbosity (subagent retries no longer spam the
parent's terminal but keep full detail here). With *diag*, also records upstream
headers, HTTP status, bytes/chunks streamed, elapsed and TTFB on the dying attempt —
enough to tell "one CF edge / downstream provider" from "random across runs".
"""
try:
try:
@@ -148,14 +116,11 @@ def log_stream_retry(
_summary = str(error)
if _summary and len(_summary) > 240:
_summary = _summary[:240] + "…"
# Inner-cause chain (httpx errors hide under openai.APIError).
try:
_chain = flatten_exception_chain(error)
except Exception:
_chain = type(error).__name__
# Per-attempt counters and upstream headers.
_now = time.time()
_bytes = 0
_chunks = 0
@@ -174,9 +139,7 @@ def log_stream_retry(
_ttfb = max(0.0, float(_first) - _started)
headers = diag.get("headers") or {}
if isinstance(headers, dict) and headers:
_headers_repr = " ".join(
f"{k}={v}" for k, v in headers.items()
)
_headers_repr = " ".join(f"{k}={v}" for k, v in headers.items())
if diag.get("http_status") is not None:
_http_status = str(diag.get("http_status"))
except Exception:
@@ -220,35 +183,16 @@ def emit_stream_drop(
mid_tool_call: bool,
diag: Optional[Dict[str, Any]] = None,
) -> None:
"""Emit a single user-visible line for a stream drop+retry.
"""One compact user-visible status line for a stream drop+retry, plus the full WARNING via log_stream_retry.
Both top-level agents and subagents announce drops in the UI — the
parent prefixes subagent lines with ``[subagent-N]`` via ``log_prefix``
so they're easy to attribute. All cases also write a structured
WARNING to ``agent.log`` via :func:`log_stream_retry` with the full
diagnostic detail (subagent_id, provider, base_url, error_type,
cf-ray, x-openrouter-provider, bytes/chunks, elapsed) for post-hoc
analysis.
The user-visible status line is intentionally compact: provider,
error class, attempt N/M, plus ``after Xs`` when the stream dropped
mid-flight. Full diagnostic detail goes to ``agent.log`` only —
``hermes logs --level WARNING | grep "Stream drop"`` to inspect.
Subagent lines get a ``[subagent-N]`` prefix from ``log_prefix``. ``after Xs``
distinguishes "couldn't connect" (0s) from "died mid-stream" (idle-kill / proxy timeout).
"""
kind = "drop mid tool-call" if mid_tool_call else "drop"
log_stream_retry(
agent,
kind=kind,
error=error,
attempt=attempt,
max_attempts=max_attempts,
mid_tool_call=mid_tool_call,
diag=diag,
agent, kind=kind, error=error, attempt=attempt, max_attempts=max_attempts, mid_tool_call=mid_tool_call, diag=diag
)
provider = agent.provider or "provider"
# Compose a brief "after Xs" suffix when we have timing data — helps
# the user distinguish "couldn't connect" (0s) from "died after 30s
# of streaming" (likely upstream idle-kill or proxy timeout).
_suffix = ""
if isinstance(diag, dict):
try:
@@ -262,10 +206,7 @@ def emit_stream_drop(
f"⚠️ {provider} stream {kind} ({type(error).__name__}){_suffix} "
f"— reconnecting, retry {attempt}/{max_attempts}"
)
agent._touch_activity(
f"stream retry {attempt}/{max_attempts} "
f"after {type(error).__name__}"
)
agent._touch_activity(f"stream retry {attempt}/{max_attempts} after {type(error).__name__}")
except Exception:
pass