diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 7bcfbaf0cb..fbb8a7fadc 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -1,27 +1,22 @@ """API error classification for smart failover and recovery. -Provides a structured taxonomy of API errors and a priority-ordered -classification pipeline that determines the correct recovery action -(retry, rotate credential, fallback to another provider, compress -context, or abort). - -Replaces scattered inline string-matching with a centralized classifier -that the main retry loop in run_agent.py consults for every API failure. +A priority-ordered pipeline maps an API exception to a ``ClassifiedError`` +whose recovery hints (retry, rotate credential, fallback, compress, abort) the +retry loop in run_agent.py consults instead of re-matching strings itself. """ from __future__ import annotations import enum +import json import logging from dataclasses import dataclass, field -from typing import Any, Dict, Optional +from typing import Any, Callable, Dict, Iterable, Iterator, Optional, Sequence, Tuple logger = logging.getLogger(__name__) -# Synthetic error code used when the OpenAI SDK rejects a provider's SSE -# ``data:`` field before Hermes receives a completion chunk. Keeping this -# distinct from generic JSON parse failures lets the classifier make narrow, -# provider-stream-specific recovery decisions without inventing an HTTP status. +# Synthetic code for the OpenAI SDK rejecting a provider's SSE ``data:`` field +# before any completion chunk arrives; distinct from generic JSON parse errors. PROVIDER_STREAM_NON_JSON_ERROR_CODE = "provider_stream_non_json_data" @@ -30,52 +25,37 @@ PROVIDER_STREAM_NON_JSON_ERROR_CODE = "provider_stream_non_json_data" class FailoverReason(enum.Enum): """Why an API call failed — determines recovery strategy.""" - # Authentication / authorization auth = "auth" # Transient auth (401/403) — refresh/rotate auth_permanent = "auth_permanent" # Auth failed after refresh — abort - # Billing / quota billing = "billing" # 402 or confirmed credit exhaustion — rotate immediately rate_limit = "rate_limit" # 429 or quota-based throttling — backoff then rotate - # Upstream model rate-limited (aggregator 429) — fallback to a different - # model, NOT credential rotation. The user's key is healthy. - upstream_rate_limit = "upstream_rate_limit" + upstream_rate_limit = "upstream_rate_limit" # Aggregator's upstream model 429 — fallback model, key is healthy - # Server-side overloaded = "overloaded" # 503/529 — provider overloaded, backoff server_error = "server_error" # 500/502 — internal server error, retry - # Transport timeout = "timeout" # Connection/read timeout — rebuild client + retry - # TLS certificate verification failure — deterministic for the host - # (TLS-inspecting proxy, missing/expired CA bundle, self-signed cert). - # Retrying reproduces the identical handshake failure, so fail fast - # with actionable guidance instead of burning retries. - ssl_cert_verification = "ssl_cert_verification" + ssl_cert_verification = "ssl_cert_verification" # Deterministic TLS chain failure — fail fast with guidance - # Context / payload context_overflow = "context_overflow" # Context too large — compress, not failover payload_too_large = "payload_too_large" # 413 — compress payload image_too_large = "image_too_large" # Native image part exceeds provider's per-image limit — shrink and retry - image_corrupt = "image_corrupt" # Provider says the image bytes are undecodable — shrinking won't help, strip and retry instead + image_corrupt = "image_corrupt" # Provider can't decode image bytes — strip and retry (shrinking won't help) - # Model / provider policy model_not_found = "model_not_found" # 404 or invalid model — fallback to different model - provider_policy_blocked = "provider_policy_blocked" # Aggregator (e.g. OpenRouter) blocked the only endpoint due to account data/privacy policy - content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — deterministic per-request, don't retry unchanged + provider_policy_blocked = "provider_policy_blocked" # Aggregator account data/privacy policy excluded the only endpoint + content_policy_blocked = "content_policy_blocked" # Provider safety filter rejected this prompt — don't retry unchanged - # Request format format_error = "format_error" # 400 bad request — abort or strip + retry invalid_encrypted_content = "invalid_encrypted_content" # Responses replay blob rejected — strip replay state and retry - multimodal_tool_content_unsupported = "multimodal_tool_content_unsupported" # Provider rejected list-type content in tool messages (e.g. Xiaomi MiMo) — downgrade to text and retry + multimodal_tool_content_unsupported = "multimodal_tool_content_unsupported" # Provider rejected list content in tool messages — downgrade to text - # Provider-specific thinking_signature = "thinking_signature" # Anthropic thinking block sig invalid long_context_tier = "long_context_tier" # Anthropic "extra usage" tier gate - oauth_long_context_beta_forbidden = "oauth_long_context_beta_forbidden" # Anthropic OAuth subscription rejects 1M context beta — disable beta and retry - llama_cpp_grammar_pattern = "llama_cpp_grammar_pattern" # llama.cpp json-schema-to-grammar rejects regex escapes in `pattern` / `format` — strip from tools and retry + oauth_long_context_beta_forbidden = "oauth_long_context_beta_forbidden" # Anthropic OAuth rejects 1M beta — disable beta and retry + llama_cpp_grammar_pattern = "llama_cpp_grammar_pattern" # llama.cpp grammar rejects regex `pattern`/`format` — strip from tools and retry - # Catch-all unknown = "unknown" # Unclassifiable — retry with backoff @@ -92,8 +72,7 @@ class ClassifiedError: message: str = "" error_context: Dict[str, Any] = field(default_factory=dict) - # Recovery action hints — the retry loop checks these instead of - # re-classifying the error itself. + # Recovery hints — the retry loop checks these instead of re-classifying. retryable: bool = True should_compress: bool = False should_rotate_credential: bool = False @@ -105,18 +84,13 @@ class ClassifiedError: @property def billing_unverified(self) -> bool: - """True when a ``billing`` verdict rests on an ambiguous body. - - Anthropic's "out of extra usage" 400 can also be a content-filter - rejection (#82154); surfaces must hedge rather than assert exhaustion. - """ + """True when a ``billing`` verdict rests on an ambiguous body (#82154).""" return bool(self.error_context.get("billing_unverified")) - # ── Provider-specific patterns ────────────────────────────────────────── -# Patterns that indicate billing exhaustion (not transient rate limit) +# Billing exhaustion (not transient rate limit). _BILLING_PATTERNS = [ "insufficient credits", "insufficient_quota", @@ -141,16 +115,10 @@ _BILLING_PATTERNS = [ "not available on the free tier", ] -# Billing-pattern matches that are NOT proof of billing exhaustion. Anthropic -# returns the identical "out of extra usage" body on a subscription OAuth -# token both when the overage bucket is genuinely depleted AND when its -# server-side content filter rejects part of the request (#82154) — the two -# are indistinguishable from the response. Classification stays ``billing`` -# (rotation + fallback remain the right recovery either way), but the -# ambiguity is carried in ``error_context`` so downstream surfaces hedge -# instead of asserting exhaustion as fact, and the credential pool applies a -# short cooldown instead of the one-hour billing bench (a content-filter -# rejection leaves the credential perfectly healthy). +# Billing matches that are NOT proof of exhaustion: Anthropic returns the same +# "out of extra usage" body for a content-filter rejection (#82154). Verdict +# stays ``billing`` but error_context marks it unverified so surfaces hedge +# and the credential pool uses a short cooldown instead of the 1h bench. _UNVERIFIED_BILLING_PATTERNS = ("out of extra usage",) @@ -160,14 +128,13 @@ def _billing_ambiguity_context(error_msg: str) -> Dict[str, Any]: return {"billing_unverified": True, "possible_content_filter": True} return {} -# xAI's explicit Grok credit-exhaustion code. Keep the HTTP 403 special case -# provider-scoped: other providers' generic billing codes historically remain -# auth failures when they arrive as 403. + +# xAI's explicit Grok credit-exhaustion code, returned as HTTP 403 rather than +# 402. The 403 special case stays provider-scoped: other providers' billing +# codes on a 403 remain auth failures. _XAI_SPENDING_LIMIT_ERROR_CODE = "personal-team-blocked:spending-limit" -# Structured provider codes that mean the account cannot serve paid traffic -# until credits/subscription capacity is restored. xAI returns its explicit -# Grok spending-limit signal as HTTP 403 rather than 402. +# Structured codes meaning the account cannot serve paid traffic. _BILLING_ERROR_CODES = frozenset({ "insufficient_quota", "billing_not_active", @@ -180,7 +147,7 @@ _BILLING_ERROR_CODES = frozenset({ _XAI_SPENDING_LIMIT_ERROR_CODE, }) -# Patterns that indicate rate limiting (transient, will resolve) +# Rate limiting (transient, will resolve). _RATE_LIMIT_PATTERNS = [ "rate limit", "rate_limit", @@ -197,26 +164,15 @@ _RATE_LIMIT_PATTERNS = [ "throttlingexception", "too many concurrent requests", "servicequotaexceededexception", - # Generic throttle prefix — Bedrock (and some proxies) surface throttling - # as "Throttling error: Too many tokens, please wait before trying - # again." Without this entry the message falls through to the - # context-overflow list (which contains "too many tokens") and the retry - # loop compresses a healthy session instead of backing off. Matched - # BEFORE _CONTEXT_OVERFLOW_PATTERNS in the message-only path, so the - # throttle wins. (port of anomalyco/opencode#37848's exclusion guard) + # Bedrock "Throttling error: Too many tokens..." also contains the overflow + # phrase "too many tokens"; rate limit is matched first so throttle wins. "throttling", ] -# Patterns that indicate provider-side overload, NOT a per-credential rate -# limit or billing problem. The credential is valid — the server is just -# busy — so the correct recovery is "back off and retry the same key", never -# "rotate the credential" (rotating exhausts the pool while the endpoint is -# still busy; a single-key user has nothing to rotate to). Some providers -# (notably Z.AI / Zhipu) reuse HTTP 429 for server-wide overload, so the 429 -# status path matches the body against this list before falling through to -# the rate_limit default. Phrases are kept narrow and overload-flavoured so a -# normal rate-limit message ("you have been rate-limited") doesn't hit this -# bucket. (#14038, #15297) +# Provider-side overload: the credential is valid, the server is busy, so back +# off and retry the same key — never rotate. Some providers (Z.AI/Zhipu) reuse +# HTTP 429 for this, so the 429 path checks these first. Kept narrow so a +# normal "you have been rate-limited" doesn't land here. (#14038, #15297) _OVERLOADED_PATTERNS = [ "overloaded", "temporarily overloaded", @@ -232,7 +188,7 @@ _OVERLOADED_PATTERNS = [ "over capacity", ] -# Usage-limit patterns that need disambiguation (could be billing OR rate_limit) +# Usage-limit patterns that need disambiguation (billing OR rate_limit). _USAGE_LIMIT_PATTERNS = [ "usage limit", "quota", @@ -240,7 +196,7 @@ _USAGE_LIMIT_PATTERNS = [ "key limit exceeded", ] -# Patterns confirming usage limit is transient (not billing) +# Signals that a usage limit is transient (periodic quota, not billing). _USAGE_LIMIT_TRANSIENT_SIGNALS = [ "try again", "retry", @@ -257,25 +213,18 @@ _USAGE_LIMIT_TRANSIENT_SIGNALS = [ "per second", ] -# Payload-too-large patterns detected from message text (no status_code attr). -# Proxies and some backends embed the HTTP status in the error message. +# Payload-too-large detected from message text (proxies embed the status). _PAYLOAD_TOO_LARGE_PATTERNS = [ "request entity too large", "payload too large", "error code: 413", - # Anthropic's structured 413 error type. Normally arrives with an HTTP - # 413 status (handled by the status path), but aggregators/proxies can - # re-wrap it into a plain message with no status attribute — route it to - # the same compression recovery. (port of anomalyco/opencode#37848) - "request_too_large", + "request_too_large", # Anthropic's 413 type, re-wrapped by proxies without a status "request exceeds the maximum size", ] -# Image-size patterns. Matched against 400 bodies (not 413) because most -# providers return a 400 with a specific image-too-big message before the -# whole request hits the 413 size limit. Anthropic's wording is the most -# important here (hard 5 MB per image, returned as -# "messages.N.content.K.image.source.base64: image exceeds 5 MB maximum"). +# Image-size rejections. Matched on 400 bodies (not 413): providers return a +# specific 400 before the whole request hits the size limit (Anthropic: hard +# 5 MB per image, "image exceeds 5 MB maximum"). _IMAGE_TOO_LARGE_PATTERNS = [ "image exceeds", # Anthropic: "image exceeds 5 MB maximum" "image too large", # generic @@ -284,42 +233,17 @@ _IMAGE_TOO_LARGE_PATTERNS = [ "image dimensions exceed", # Anthropic: "image dimensions exceed max allowed size: 8000 pixels" "dimensions exceed max allowed size", # Anthropic dimension-cap (wording variant) "max allowed size: 8000", # Anthropic dimension-cap (explicit pixel ceiling) - # Vendors that reject the same oversized image without using the word - # "image". MiniMax's Anthropic-compatible endpoint returns - # "media exceeds size limit: max 10485760 bytes (2013)" for a native - # image part above its 10 MB ceiling (#76039). Matched on the "media" - # fragment to mirror "image exceeds" above and catch reworded variants. - # A non-image media rejection (audio/video) that lands here is safe: the - # shrink pass finds no image parts, returns False, and the caller - # surfaces the original error unchanged. + # MiniMax Anthropic-compat: "media exceeds size limit: max 10485760 bytes" + # (#76039). A non-image media rejection landing here is harmless: the + # shrink pass finds no image parts and the original error surfaces. "media exceeds", "media too large", - # "request_too_large" on a request known to contain an image → image is - # the likely culprit; we still try the shrink path before giving up. ] -# Image-corruption patterns — distinct from _IMAGE_TOO_LARGE_PATTERNS above. -# These fire when the provider can decode the request but not the image -# bytes themselves (e.g. a re-serialized image part in replayed history that -# lost data along the way). Re-encoding/shrinking corrupt bytes does not fix -# corruption, so this list is routed to the strip-and-retry path -# (FailoverReason.image_corrupt), never to the shrink path. -# -# xAI wording: {"code":"invalid-argument","error":"...Invalid PNG image."} -# xAI has a second wording for the same failure class depending on where -# the truncation lands: "Invalid PNG image." for aligned truncation, -# "base64 string of provided image cannot be decoded" for unaligned -# truncation (confirmed by the issue reporter — same root cause, two wire -# messages). -# A third xAI wording covers the URL-image path — the provider downloads -# the image itself and rejects the fetched bytes: -# {"code":"invalid-argument","error":"code: 'Client specified an invalid -# argument', message: \"Downloaded response does not contain a valid JPG, -# PNG, WebP, or ICO image.\""} -# Matched as the full observed sentence on purpose — shorter fragments -# ("downloaded response does not contain a valid") also match non-image -# download failures and would misroute them into strip-and-retry. -# See: https://github.com/NousResearch/hermes-agent/issues/69078 +# Image bytes undecodable (e.g. re-serialized history lost data). Shrinking +# can't fix corruption, so these route to strip-and-retry, never shrink. +# xAI wordings (#69078); the last is the full sentence on purpose — shorter +# fragments also match non-image download failures. _IMAGE_CORRUPT_PATTERNS = [ "invalid png image", "invalid jpeg image", @@ -327,33 +251,22 @@ _IMAGE_CORRUPT_PATTERNS = [ "downloaded response does not contain a valid jpg, png, webp, or ico image", ] -# Providers that follow the OpenAI spec strictly require tool message -# ``content`` to be a string. Some (Anthropic native, Codex Responses, -# Gemini native, first-party OpenAI) extend this to accept a content-parts -# list (text + image_url) so screenshots from computer_use survive. Others -# (Xiaomi MiMo, some Alibaba endpoints, a long tail of OpenAI-compatible -# providers) reject the list with a 400 — the patterns below are the most -# common error shapes we see. Recovery: strip image parts from tool -# messages in-place, record the (provider, model) for the rest of the -# session so we don't waste another call learning the same lesson, retry. -# -# See: https://github.com/NousResearch/hermes-agent/issues/27344 +# Providers that reject list-type ``content`` in tool messages with a 400 +# (Xiaomi MiMo, some Alibaba endpoints, OpenAI-compat long tail). Recovery: +# strip image parts from tool messages, remember (provider, model), retry. (#27344) _MULTIMODAL_TOOL_CONTENT_PATTERNS = [ # Xiaomi MiMo: {"error":{"code":"400","message":"Param Incorrect","param":"text is not set"}} "text is not set", - # Generic "tool message must be string" shapes "tool message content must be a string", "tool content must be a string", "tool message must be a string", - # OpenAI-compat servers that reject list-type tool content with a - # schema-validation message + # OpenAI-compat schema-validation shapes "expected string, got list", "expected string, got array", # Alibaba/DashScope variant "tool_call.content must be string", ] -# Context overflow patterns _CONTEXT_OVERFLOW_PATTERNS = [ "context length", "context size", @@ -365,11 +278,9 @@ _CONTEXT_OVERFLOW_PATTERNS = [ "context window", "prompt is too long", "prompt exceeds max length", - # NOTE: bare "max_tokens" is load-bearing — the output-cap-retry path keys - # off it (e.g. "max_tokens: 65536 > context_window: 200000 ..."). Do NOT - # remove it. Provider empty-response advisories also contain "very low - # max_tokens", but those are intercepted by _EMPTY_PROVIDER_RESPONSE_PATTERNS - # BEFORE this list is consulted, so they never mis-route into compression. + # Bare "max_tokens" is load-bearing: the output-cap-retry path keys off it. + # Empty-response advisories mentioning it are intercepted earlier by + # _EMPTY_PROVIDER_RESPONSE_PATTERNS, so they never route into compression. "max_tokens", "maximum number of tokens", # vLLM / local inference server patterns @@ -394,13 +305,10 @@ _CONTEXT_OVERFLOW_PATTERNS = [ "max input token", "input token", "exceeds the maximum number of input tokens", - # Together/Fireworks-style: "Input length 131393 exceeds the maximum - # allowed input length of 131040 tokens." No other pattern in this list - # matches that wording. (port of anomalyco/opencode#37848) + # Together/Fireworks: "Input length N exceeds the maximum allowed input length of M tokens." "maximum allowed input length", ] -# Model not found patterns _MODEL_NOT_FOUND_PATTERNS = [ "is not a valid model", "invalid model", @@ -410,14 +318,8 @@ _MODEL_NOT_FOUND_PATTERNS = [ "no such model", "unknown model", "unsupported model", - # OpenRouter returns 404 with this message when none of the candidate - # endpoints for the selected model support tool/function calling. - # Classifying this as model_not_found triggers fallback to a different - # model or provider that does support tools. Without this entry the - # pattern falls through to ``unknown`` with ``retryable=True``, the - # retry loop burns all attempts on the same deterministic rejection, - # and the error surfaces as a confusing "model not found" message - # instead of automatically failing over. See PR #58446. + # OpenRouter 404 when no endpoint for the model supports tool calling; + # model_not_found triggers fallback instead of burning retries (#58446). "no endpoints found that support tool use", ] @@ -425,16 +327,10 @@ _MODEL_NOT_FOUND_PATTERNS = [ def _model_id_missing_known_prefix(model: str, provider: str) -> bool: """True when a bare model id is only known to the provider as ``vendor/id``. - Some providers answer a malformed model id with a naked 404 that names - nothing — NVIDIA NIM returns ``404 page not found`` for a bare - ``nemotron-3-ultra-550b-a55b``, indistinguishable from a bad endpoint - path. Consulting the curated catalogue tells the two apart: if the id - carries no ``/`` but the catalogue has exactly one entry ending in - ``/``, the prefix was dropped and the failure is deterministic. - - Never guesses — an id absent from the catalogue (a local NIM container, - a proxied model) returns False so genuine endpoint problems keep their - retryable ``unknown`` classification. + NVIDIA NIM answers a bare id with a naked ``404 page not found``; the + curated catalogue tells that apart from a bad endpoint. Never guesses: an + id absent from the catalogue returns False so real endpoint problems keep + their retryable ``unknown`` classification. """ name = (model or "").strip() if not name or "/" in name: @@ -447,27 +343,15 @@ def _model_id_missing_known_prefix(model: str, provider: str) -> bool: return False -# Malformed-message-array 400s. Deterministic request-shape rejections that -# describe the *transcript* being invalid, not a parameter. The canonical -# case: a stream dies mid-response and Hermes persists a content-less -# assistant stub; on the next turn the Anthropic message schema (and the -# litellm/Bedrock proxies in front of it) reject the whole request with -# "all messages must have non-empty content except for the optional final -# assistant message" / errorCode INVALID_REQUEST_BODY -# These are NOT context overflow — the input may be tiny — but a large -# session used to mis-route them into the compression loop via the generic -# "400 + large session" heuristic below, ending in "Cannot compress further" -# every retry (the input is unchanged, so compression cannot help). Match -# the message-shape signals explicitly and fail fast as a format_error so the -# loop stops looping. The empty-stub creation is the root cause (fixed in -# chat_completion_helpers); this pattern stops the misclassification symptom -# for transcripts that already contain a poisoned stub. -# Qwen/vLLM chat-template raise_exception("No user query found in messages") -# — shared between _INVALID_MESSAGE_BODY_PATTERNS (→ format_error) and the -# llama.cpp grammar exclusion guard below. Keeping a single constant prevents -# the two sites from silently drifting if the phrase is ever changed. +# Qwen/vLLM chat-template raise_exception("No user query found in messages"). +# Shared by _INVALID_MESSAGE_BODY_PATTERNS (→ format_error) and the llama.cpp +# grammar exclusion guard so the two sites cannot drift. _NO_USER_QUERY_SIGNAL = "no user query found" +# Malformed-message-array 400s: deterministic rejections of the *transcript* +# (e.g. a content-less assistant stub after a dead stream). NOT context +# overflow — the input may be tiny — so they must fail fast as format_error +# instead of thrashing the compression loop. _INVALID_MESSAGE_BODY_PATTERNS = [ "must have non-empty content", "messages must have non-empty", @@ -475,26 +359,14 @@ _INVALID_MESSAGE_BODY_PATTERNS = [ "text content blocks must be non-empty", "content field is required", "messages: at least one message is required", - # Qwen / vLLM chat templates raise this when the request has no surviving - # non-empty user turn (oversized session truncation, compression that - # dropped the only user message, or a resumed lineage that opens with - # assistant/tool). Deterministic — compression cannot invent a user - # query the template already rejected. Fail fast as format_error so we - # do not thrash the compression loop or mis-route into llama.cpp - # grammar recovery when local engines wrap the raise_exception as - # applyPromptTemplate / "Unable to generate parser for this template". + # Qwen/vLLM templates: no surviving non-empty user turn. Compression + # cannot invent one, and local engines may wrap this as a grammar error. _NO_USER_QUERY_SIGNAL, ] -# Request-validation patterns — the request is malformed and will fail -# identically on every retry. Some OpenAI-compatible gateways (notably -# codex.nekos.me) return these as 5xx instead of the standard 4xx, which -# makes the generic "5xx → retryable server_error" rule misfire: the retry -# loop hammers the same deterministic rejection 3+ times, then the -# transport-recovery path resets the counter and does it again, producing -# a request flood. When a 5xx body carries one of these unambiguous -# request-validation signals, classify as a non-retryable format_error so -# the loop fails fast and falls back instead of looping. +# Request-validation signals: malformed request, identical on every retry. +# Some gateways (codex.nekos.me) return these as 5xx, so the 5xx path also +# checks them to avoid a retry flood on a deterministic rejection. _REQUEST_VALIDATION_PATTERNS = [ "unknown parameter", "unsupported parameter", @@ -504,99 +376,49 @@ _REQUEST_VALIDATION_PATTERNS = [ "unsupported_parameter", ] -# Request parameters that Hermes sends on SOME routes only, paired with the -# providers/hosts where sending them is deliberate. -# -# When a host that is NOT in the allowed set rejects one of these fields, the -# client never put it in the body — the provider's own gateway injected it — -# so the 400 is a server-side flake rather than a deterministic request-shape -# error. See ``_is_server_injected_param_rejection`` and the branch in -# ``_classify_400``. -# -# ``prompt_cache_retention`` is only sent for api.meta.ai and bedrock-mantle -# hosts (agent/transports/codex.py::_default_prompt_cache_retention_for_request). -# The Codex OAuth backend rejects it spontaneously on requests that provably -# never carried it. +# Parameters Hermes sends on SOME routes only → hosts where sending them is +# deliberate. A rejection from any other host means the provider's own gateway +# injected the field, so the 400 is a server-side flake, not our request shape. +# ``prompt_cache_retention``: only sent for api.meta.ai / bedrock-mantle +# (agent/transports/codex.py); the Codex OAuth backend rejects it spontaneously. _SERVER_INJECTED_PARAM_SENDERS: Dict[str, tuple] = { "prompt_cache_retention": ("meta", "muse", "msl", "model-api", "bedrock", "mantle"), } +_PARAM_REJECTION_WORDS = ("not supported", "unsupported", "unknown", "unrecognized") + def _is_server_injected_param_rejection(error_msg: str, provider: str) -> bool: - """True when a 400 blames a parameter this route never sends. + """True when a 400 blames a one-route-only parameter this route never sends. - ``error_msg`` is the lowercased, concatenated message text; ``provider`` is - the lowercased provider slug. A match means the rejection cannot be - attributed to our own request shape, so the error is transient and retrying - the identical request is the correct recovery. - - Deliberately conservative: it fires only for known one-route-only - parameters AND only when the current provider is not one of the routes that - actually sends them, so a genuine client-side bad parameter (``max_tokens`` - on a GPT-5 model) still fails fast as a ``format_error``. + Conservative: fires only for known parameters AND only when ``provider`` + is not a route that sends them, so a genuine client-side bad parameter + (``max_tokens`` on GPT-5) still fails fast as ``format_error``. """ if not error_msg: return False provider_slug = (provider or "").strip().lower() for param, senders in _SERVER_INJECTED_PARAM_SENDERS.items(): - if param not in error_msg: + if param not in error_msg or not any(w in error_msg for w in _PARAM_REJECTION_WORDS): continue - # Require the message to actually be a rejection of that parameter, - # not an incidental mention. - if not ( - "not supported" in error_msg - or "unsupported" in error_msg - or "unknown" in error_msg - or "unrecognized" in error_msg - ): - continue - if any(sender in provider_slug for sender in senders): - # This route sends the field on purpose — a real request error. - return False - return True + return not any(sender in provider_slug for sender in senders) return False -# OpenRouter aggregator policy-block patterns. -# -# When a user's OpenRouter account privacy setting (or a per-request -# `provider.data_collection: deny` preference) excludes the only endpoint -# serving a model, OpenRouter returns 404 with a *specific* message that is -# distinct from "model not found": -# -# "No endpoints available matching your guardrail restrictions and -# data policy. Configure: https://openrouter.ai/settings/privacy" -# -# We classify this as `provider_policy_blocked` rather than -# `model_not_found` because: -# - The model *exists* — model_not_found is misleading in logs -# - Provider fallback won't help: the account-level setting applies to -# every call on the same OpenRouter account -# - The error body already contains the fix URL, so the user gets -# actionable guidance without us rewriting the message +# OpenRouter 404 when the account privacy setting (or per-request +# ``provider.data_collection: deny``) excludes the only endpoint for a model. +# Not model_not_found: the model exists, fallback can't help (account-level), +# and the body already carries the fix URL. _PROVIDER_POLICY_BLOCKED_PATTERNS = [ "no endpoints available matching your guardrail", "no endpoints available matching your data policy", "no endpoints found matching your data policy", ] -# Provider content-policy / safety-filter blocks. Distinct from -# ``provider_policy_blocked`` above (which is an OpenRouter *account*-level -# data/privacy guardrail) — these are *per-prompt* safety decisions made by -# the upstream model provider. They are deterministic for the unchanged -# request, so retrying the same prompt three times just reproduces the same -# block and burns paid attempts on a refusal. The recovery is to switch to a -# configured fallback model/provider immediately, or surface the block to -# the user with actionable guidance if no fallback exists. -# -# Patterns are intentionally narrow — each phrase is a verbatim string from -# a specific provider's safety pipeline, not a generic word like "policy" or -# "violation" that could collide with billing/auth/format errors: -# • OpenAI Codex cybersecurity refusal (gpt-5.5, the case from #18028) -# • OpenAI moderation refusal ("violates our usage policies", with -# "usage policies" disambiguating from billing's "exceeded ... policy") -# • Anthropic safety refusal ("prompt was flagged by ... safety system") -# • OpenAI Responses content filter +# Per-prompt provider safety-filter blocks (distinct from the account-level +# provider_policy_blocked). Deterministic for the unchanged request, so +# fallback immediately. Each phrase is verbatim from a specific provider — +# never a generic word like "policy" that could collide with billing/auth. _CONTENT_POLICY_BLOCKED_PATTERNS = [ # OpenAI Codex (#18028) — message may arrive without an HTTP status "flagged for possible cybersecurity risk", @@ -608,22 +430,11 @@ _CONTENT_POLICY_BLOCKED_PATTERNS = [ # Anthropic safety system "prompt was flagged by our safety", "responses cannot be generated due to safety", - # Generic content-filter wording seen on Azure / OpenAI Responses. - # ``content_filter`` (underscore) is the OpenAI-standard error/finish - # token surfaced verbatim by their SDKs when a request is blocked. - # ``responsibleaipolicyviolation`` is Azure OpenAI's error code. - # Deliberately NOT matching the space variant ("content filter") — it - # appears in benign config descriptions and tooltip text that providers - # echo back; the underscore form is provider-specific enough. + # OpenAI-standard token / Azure error code. Deliberately NOT the space + # variant "content filter", which appears in benign echoed config text. "content_filter", "responsibleaipolicyviolation", - # MiniMax output-layer safety filter. The error string is surfaced - # verbatim by MiniMax SDK / OpenAI-compatible endpoints, usually in the - # form "output new_sensitive (1027)" when the model's *output* (often a - # large tool-call argument block) trips the upstream safety filter and - # the SSE stream is truncated mid-flight. ``new_sensitive`` is the - # filter name and is narrow enough that billing / format / auth error - # strings will not collide. See #32421. + # MiniMax output-layer safety filter, "output new_sensitive (1027)" (#32421) "new_sensitive", ] @@ -641,20 +452,9 @@ _AUTH_PATTERNS = [ "access denied", ] -# Anthropic thinking block signature patterns -_THINKING_SIG_PATTERNS = [ - "signature", # Combined with "thinking" check -] - -# Message-string patterns that indicate a provider-side timeout even when -# the exception type is generic (e.g. RuntimeError from a local shim that -# wraps a subprocess timeout). Checked before the type-based transport -# heuristics so custom-provider "timed out" errors don't fall through to -# Provider empty-response advisories (OpenRouter / nano-gpt / similar). -# Checked before context-overflow matching because the advisory text often -# mentions "max_tokens" as a possible cause, which historically sat in -# _CONTEXT_OVERFLOW_PATTERNS and sent healthy sessions into a compression -# death spiral ending in "Cannot compress further". +# Provider empty-response advisories (OpenRouter / nano-gpt / similar). Checked +# before context-overflow matching because the text often mentions +# "max_tokens", which used to send healthy sessions into a compression spiral. _EMPTY_PROVIDER_RESPONSE_PATTERNS = [ "returned an empty response", "empty response despite retries", @@ -663,7 +463,8 @@ _EMPTY_PROVIDER_RESPONSE_PATTERNS = [ "empty response stream", ] -# the unknown bucket and get misreported as empty responses. +# Timeout wording from generic exception types (RuntimeError from a shim +# wrapping a subprocess timeout) that the type-based heuristics would miss. _TIMEOUT_MESSAGE_PATTERNS = [ "timed out", "turn timed out", @@ -673,22 +474,11 @@ _TIMEOUT_MESSAGE_PATTERNS = [ "upstream timed out", ] -# Connection-establishment / DNS failure message patterns. These surface -# when the exception TYPE is generic (RuntimeError/Exception from a local -# shim, MCP bridge, subprocess wrapper, or an SDK that re-raises without -# chaining) so the _TRANSPORT_ERROR_TYPES check never fires, and the error -# carries no HTTP status. Without message-level matching they fall through -# to FailoverReason.unknown, which misses the transport eager-fallback path -# in the retry loop (unknown retries the same dead endpoint for the full -# budget before fallback). Ported from anomalyco/opencode#40707, which hit -# the same bug shape: serialized midstream errors matched by type only. -# -# Deliberately EXCLUDES mid-stream disconnect strings ("connection reset by -# peer", "peer closed connection", "unexpected eof", "socket hang up") — -# those belong to _SERVER_DISCONNECT_PATTERNS, whose classification step -# runs later and routes large sessions to context-overflow compression. -# A connection that was never established cannot be a server-side overflow -# rejection, so these are safe to classify as plain retryable transport. +# Connect/DNS failures surfaced by generic exception types with no status, so +# _TRANSPORT_ERROR_TYPES never fires. Deliberately EXCLUDES mid-stream +# disconnect strings — those belong to _SERVER_DISCONNECT_PATTERNS, which may +# route large sessions to compression; a never-established connection cannot +# be an overflow rejection. _CONNECTION_MESSAGE_PATTERNS = [ # TCP connect failures "connection refused", @@ -710,7 +500,6 @@ _CONNECTION_MESSAGE_PATTERNS = [ "upstream connect error", ] -# Transport error type names _TRANSPORT_ERROR_TYPES = frozenset({ "ReadTimeout", "ConnectTimeout", "PoolTimeout", "ConnectError", "RemoteProtocolError", @@ -718,12 +507,8 @@ _TRANSPORT_ERROR_TYPES = frozenset({ "ConnectionAbortedError", "BrokenPipeError", "TimeoutError", "ReadError", "ServerDisconnectedError", - # SSL/TLS transport errors — transient mid-stream handshake/record - # failures that should retry rather than surface as a stalled session. - # ssl.SSLError subclasses OSError (caught by isinstance) but we list - # the type names here so provider-wrapped SSL errors (e.g. when the - # SDK re-raises without preserving the exception chain) still classify - # as transport rather than falling through to the unknown bucket. + # SSL type names listed so provider-wrapped SSL errors (chain lost) still + # classify as transport instead of unknown. "SSLError", "SSLZeroReturnError", "SSLWantReadError", "SSLWantWriteError", "SSLEOFError", "SSLSyscallError", # OpenAI SDK errors (not subclasses of Python builtins) @@ -731,12 +516,8 @@ _TRANSPORT_ERROR_TYPES = frozenset({ "APITimeoutError", }) -# Server disconnect patterns (no status code, but transport-level). -# These are the "ambiguous" patterns — a plain connection close could be -# transient transport hiccup OR server-side context overflow rejection -# (common when the API gateway disconnects instead of returning an HTTP -# error for oversized requests). A large session + one of these patterns -# triggers the context-overflow-with-compression recovery path. +# Ambiguous disconnects (no status): transient hiccup OR a gateway dropping an +# oversized request. A large session + one of these → context-overflow path. _SERVER_DISCONNECT_PATTERNS = [ "server disconnected", "peer closed connection", @@ -747,18 +528,9 @@ _SERVER_DISCONNECT_PATTERNS = [ "incomplete chunked read", ] -# SSL certificate verification failures — deterministic, NOT transient. -# -# A failed certificate chain (TLS-inspecting corporate proxy, missing -# custom CA in the trust store, expired certificate, self-signed cert) -# fails identically on every retry. Burning the retry budget before -# surfacing the error hides the actionable fix from the user for minutes. -# Inspired by Claude Code v2.1.199 (July 2026), which made SSL certificate -# errors fail immediately with a fix hint instead of retrying. -# -# Must be checked BEFORE _SSL_TRANSIENT_PATTERNS — "certificate verify -# failed" messages usually also contain "[SSL:" which would otherwise -# match the transient list and retry forever. +# SSL certificate verification failures are deterministic (proxy, missing CA, +# expired/self-signed cert) — fail fast. Checked BEFORE _SSL_TRANSIENT_PATTERNS +# because these messages usually also contain "[SSL:". _SSL_CERT_VERIFY_PATTERNS = [ "certificate verify failed", # Python ssl module canonical text "certificate_verify_failed", # OpenSSL error token @@ -770,22 +542,9 @@ _SSL_CERT_VERIFY_PATTERNS = [ "unable to verify the first certificate", # Node/undici phrasing (MCP bridges) ] -# SSL/TLS transient failure patterns — intentionally distinct from -# _SERVER_DISCONNECT_PATTERNS above. -# -# An SSL alert mid-stream is almost always a transport-layer hiccup -# (flaky network, mid-session TLS renegotiation failure, load balancer -# dropping the connection) — NOT a server-side context overflow signal. -# So we want the retry path but NOT the compression path; lumping these -# into _SERVER_DISCONNECT_PATTERNS would trigger unnecessary (and -# expensive) context compression on any large-session SSL hiccup. -# -# The OpenSSL library constructs error codes by prepending a format string -# to the uppercased alert reason; OpenSSL 3.x changed the separator -# (e.g. `SSLV3_ALERT_BAD_RECORD_MAC` → `SSL/TLS_ALERT_BAD_RECORD_MAC`), -# which silently stopped matching anything explicit. Matching on the -# stable substrings (`bad record mac`, `ssl alert`, `tls alert`, etc.) -# survives future OpenSSL format churn without code changes. +# Transient SSL alerts: retry but NOT compression (kept apart from +# _SERVER_DISCONNECT_PATTERNS). Matched on stable substrings because OpenSSL 3 +# changed token separators (SSLV3_ALERT_... → SSL/TLS_ALERT_...). _SSL_TRANSIENT_PATTERNS = [ # Space-separated (human-readable form, Python ssl module, most SDKs) "bad record mac", @@ -794,8 +553,7 @@ _SSL_TRANSIENT_PATTERNS = [ "ssl handshake failure", "tlsv1 alert", "sslv3 alert", - # Underscore-separated (OpenSSL error code tokens, e.g. - # `ERR_SSL_SSL/TLS_ALERT_BAD_RECORD_MAC`, `SSLV3_ALERT_BAD_RECORD_MAC`) + # Underscore-separated OpenSSL tokens "bad_record_mac", "ssl_alert", "tls_alert", @@ -805,8 +563,167 @@ _SSL_TRANSIENT_PATTERNS = [ ] +# ── Verdicts and rule tables ──────────────────────────────────────────── +# +# A verdict is ``(reason, hint_overrides)``; overrides not listed take the +# ClassifiedError defaults (retryable=True, everything else False/empty). +# Rule tables are ordered ``(patterns, reason, hints)`` triples matched +# first-hit; ``hints`` may be a callable of the error message. + +Verdict = Tuple[FailoverReason, Dict[str, Any]] +_ROTATE_FALLBACK = {"should_rotate_credential": True, "should_fallback": True} + +_V_BILLING: Verdict = (FailoverReason.billing, {"retryable": False, **_ROTATE_FALLBACK}) +_V_RATE_LIMIT: Verdict = (FailoverReason.rate_limit, dict(_ROTATE_FALLBACK)) +_V_OVERLOADED: Verdict = (FailoverReason.overloaded, {}) +_V_SERVER_ERROR: Verdict = (FailoverReason.server_error, {}) +_V_CONTEXT_OVERFLOW: Verdict = (FailoverReason.context_overflow, {"should_compress": True}) +_V_PAYLOAD_TOO_LARGE: Verdict = (FailoverReason.payload_too_large, {"should_compress": True}) +_V_MODEL_NOT_FOUND: Verdict = (FailoverReason.model_not_found, {"retryable": False, "should_fallback": True}) +_V_POLICY_BLOCKED: Verdict = (FailoverReason.provider_policy_blocked, {"retryable": False}) +_V_FORMAT_ERROR: Verdict = (FailoverReason.format_error, {"retryable": False, "should_fallback": True}) +_V_AUTH_ROTATE: Verdict = (FailoverReason.auth, {"retryable": False, **_ROTATE_FALLBACK}) +_V_AUTH_FALLBACK: Verdict = (FailoverReason.auth, {"retryable": False, "should_fallback": True}) +_V_TIMEOUT: Verdict = (FailoverReason.timeout, {}) +_V_IMAGE_TOO_LARGE: Verdict = (FailoverReason.image_too_large, {}) +_V_IMAGE_CORRUPT: Verdict = (FailoverReason.image_corrupt, {}) +_V_MULTIMODAL: Verdict = (FailoverReason.multimodal_tool_content_unsupported, {}) +_V_INVALID_ENCRYPTED: Verdict = (FailoverReason.invalid_encrypted_content, {}) + + +def _billing_hints(error_msg: str) -> Dict[str, Any]: + """Billing verdict carrying the #82154 ambiguity marker when applicable.""" + return {**_V_BILLING[1], "error_context": _billing_ambiguity_context(error_msg)} + + +def _rule(patterns: Sequence[str], verdict: Verdict, hints=None) -> tuple: + return (patterns, verdict[0], verdict[1] if hints is None else hints) + + +def _emit(result_fn: Callable[..., ClassifiedError], verdict: Verdict) -> ClassifiedError: + return result_fn(verdict[0], **verdict[1]) + + +def _first_match(error_msg: str, rules: Iterable[tuple], result_fn) -> Optional[ClassifiedError]: + """Return the verdict of the first rule whose pattern list hits ``error_msg``.""" + for patterns, reason, hints in rules: + if any(p in error_msg for p in patterns): + overrides: Dict[str, Any] = hints(error_msg) if callable(hints) else hints + return result_fn(reason, **overrides) + return None + + +# Image/tool-content 400s, ordered: multimodal recovery differs from image +# shrink; corrupt bytes need strip-and-retry, not shrink; image-shrink is a +# cheaper recovery than context compression for "exceeds" + "image" bodies. +_IMAGE_TOOL_RULES = ( + _rule(_MULTIMODAL_TOOL_CONTENT_PATTERNS, _V_MULTIMODAL), + _rule(_IMAGE_CORRUPT_PATTERNS, _V_IMAGE_CORRUPT), + _rule(_IMAGE_TOO_LARGE_PATTERNS, _V_IMAGE_TOO_LARGE), +) + +# Overflow signals arriving as 5xx (llama.cpp reports overflow as 500; busy / +# model-load OOM as 503). Empty-response advisories must not enter compression. +_OVERFLOW_AS_5XX_RULES = ( + _rule(_EMPTY_PROVIDER_RESPONSE_PATTERNS, _V_SERVER_ERROR), + _rule(_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW), +) + +# 404: Nous API surfaces credit depletion as a paid model vanishing from the +# Free Tier (billing, not missing model); OpenRouter policy block before +# model_not_found. +_404_RULES = ( + _rule(_BILLING_PATTERNS, _V_BILLING), + _rule(_PROVIDER_POLICY_BLOCKED_PATTERNS, _V_POLICY_BLOCKED), + _rule(_MODEL_NOT_FOUND_PATTERNS, _V_MODEL_NOT_FOUND), +) + +# 400 tail after the deterministic request-shape checks. Some providers return +# model-not-found / rate-limit / billing as 400 instead of 404/429/402. +_400_TAIL_RULES = _OVERFLOW_AS_5XX_RULES + ( + _rule(_PROVIDER_POLICY_BLOCKED_PATTERNS, _V_POLICY_BLOCKED), + _rule(_MODEL_NOT_FOUND_PATTERNS, _V_MODEL_NOT_FOUND), + _rule(_RATE_LIMIT_PATTERNS, _V_RATE_LIMIT), + _rule(_BILLING_PATTERNS, _V_BILLING, _billing_hints), +) + +# Status-less message path, head (before usage-limit disambiguation). +_MESSAGE_HEAD_RULES = (_rule(_PAYLOAD_TOO_LARGE_PATTERNS, _V_PAYLOAD_TOO_LARGE),) + _IMAGE_TOOL_RULES + +# Status-less message path, tail. Overload before rate_limit/billing so a +# message-only "overloaded" backs off instead of rotating; auth is +# non-retryable (same key always fails); policy block before model_not_found; +# timeout/connection wording last, classified as transport (never compression). +_MESSAGE_TAIL_RULES = ( + _rule(_OVERLOADED_PATTERNS, _V_OVERLOADED), + _rule(_BILLING_PATTERNS, _V_BILLING, _billing_hints), + _rule(_RATE_LIMIT_PATTERNS, _V_RATE_LIMIT), + _rule(_EMPTY_PROVIDER_RESPONSE_PATTERNS, _V_SERVER_ERROR), + _rule(_CONTEXT_OVERFLOW_PATTERNS, _V_CONTEXT_OVERFLOW), + _rule(_AUTH_PATTERNS, _V_AUTH_ROTATE), + _rule(_PROVIDER_POLICY_BLOCKED_PATTERNS, _V_POLICY_BLOCKED), + _rule(_MODEL_NOT_FOUND_PATTERNS, _V_MODEL_NOT_FOUND), + _rule(_TIMEOUT_MESSAGE_PATTERNS, _V_TIMEOUT), + _rule(_CONNECTION_MESSAGE_PATTERNS, _V_TIMEOUT), +) + +# Structured error code → verdict. The error-code rate_limit verdict rotates +# but does not set should_fallback (unlike the message/status paths). +_ERROR_CODE_VERDICTS: Dict[str, Verdict] = { + **dict.fromkeys(("resource_exhausted", "throttled", "rate_limit_exceeded"), + (FailoverReason.rate_limit, {"should_rotate_credential": True})), + **dict.fromkeys(_BILLING_ERROR_CODES, _V_BILLING), + **dict.fromkeys(("model_not_found", "model_not_available", "invalid_model"), _V_MODEL_NOT_FOUND), + **dict.fromkeys(("context_length_exceeded", "max_tokens_exceeded"), _V_CONTEXT_OVERFLOW), + "invalid_encrypted_content": _V_INVALID_ENCRYPTED, +} + +_5XX_VALIDATION_CODES = {"invalid_request_error", "unknown_parameter", "unsupported_parameter"} +_400_VALIDATION_CODES = {"unknown_parameter", "unsupported_parameter"} +_400_VALIDATION_PATTERNS = [p for p in _REQUEST_VALIDATION_PATTERNS if p != "invalid_request_error"] + + # ── Classification pipeline ───────────────────────────────────────────── +def _openrouter_wrapped_message(err_obj: dict) -> str: + """Lowercased inner message from OpenRouter's ``error.metadata.raw`` JSON wrapper.""" + metadata = err_obj.get("metadata", {}) + raw = metadata.get("raw") or "" if isinstance(metadata, dict) else "" + if not (isinstance(raw, str) and raw.strip()): + return "" + try: + inner = json.loads(raw) + except (json.JSONDecodeError, TypeError): + return "" + inner_err = inner.get("error", {}) if isinstance(inner, dict) else None + if isinstance(inner_err, dict): + return str(inner_err.get("message") or "").lower() + return "" + + +def _build_error_msg(error: Exception, body: Any) -> str: + """Lowercased str(error) + body message + OpenRouter-wrapped upstream message. + + str(error) alone may omit the body (OpenAI SDK's APIStatusError.__str__ + returns only the first arg), so body text is appended for pattern matching. + """ + raw_msg = str(error).lower() + body_msg = metadata_msg = "" + if isinstance(body, dict): + err_obj = body.get("error", {}) + if isinstance(err_obj, dict): + body_msg = str(err_obj.get("message") or "").lower() + metadata_msg = _openrouter_wrapped_message(err_obj) + if not body_msg: + body_msg = str(body.get("message") or "").lower() + parts = [raw_msg] + if body_msg and body_msg not in raw_msg: + parts.append(body_msg) + if metadata_msg and metadata_msg not in raw_msg and metadata_msg not in body_msg: + parts.append(metadata_msg) + return " ".join(parts) + + def classify_api_error( error: Exception, *, @@ -818,78 +735,19 @@ def classify_api_error( ) -> ClassifiedError: """Classify an API error into a structured recovery recommendation. - Priority-ordered pipeline: - 0. Plugin ``transform_api_error_classification`` hooks (first valid result wins) - 1. Special-case provider-specific patterns (thinking sigs, tier gates) - 2. HTTP status code + message-aware refinement - 3. Error code classification (from body) - 4. Message pattern matching (billing vs rate_limit vs context vs auth) - 5. SSL/TLS transient alert patterns → retry as timeout - 6. Server disconnect + large session → context overflow - 7. Transport error heuristics - 8. Fallback: unknown (retryable with backoff) - - Args: - error: The exception from the API call. - provider: Current provider name (e.g. "openrouter", "anthropic"). - model: Current model slug. - approx_tokens: Approximate token count of the current context. - context_length: Maximum context length for the current model. - - Returns: - ClassifiedError with reason and recovery action hints. + Priority order: plugin hooks → provider-specific special cases → HTTP + status → structured error code → message patterns → SSL → disconnect + + large session → transport types → unknown (retryable with backoff). """ status_code = _extract_status_code(error) error_type = type(error).__name__ - # Copilot/GitHub Models RateLimitError may not set .status_code; force 429 - # so downstream rate-limit handling (classifier reason, pool rotation, - # fallback gating) fires correctly instead of misclassifying as generic. + # Copilot/GitHub Models RateLimitError may not set .status_code; force 429. if status_code is None and error_type == "RateLimitError": status_code = 429 body = _extract_error_body(error) error_code = _extract_error_code(body) response_headers = _extract_response_headers(error) - - # Build a comprehensive error message string for pattern matching. - # str(error) alone may not include the body message (e.g. OpenAI SDK's - # APIStatusError.__str__ returns the first arg, not the body). Append - # the body message so patterns like "try again" in 402 disambiguation - # are detected even when only present in the structured body. - # - # Also extract metadata.raw — OpenRouter wraps upstream provider errors - # inside {"error": {"message": "Provider returned error", "metadata": - # {"raw": ""}}} and the real error message (e.g. - # "context length exceeded") is only in the inner JSON. - _raw_msg = str(error).lower() - _body_msg = "" - _metadata_msg = "" - if isinstance(body, dict): - _err_obj = body.get("error", {}) - if isinstance(_err_obj, dict): - _body_msg = str(_err_obj.get("message") or "").lower() - # Parse metadata.raw for wrapped provider errors - _metadata = _err_obj.get("metadata", {}) - if isinstance(_metadata, dict): - _raw_json = _metadata.get("raw") or "" - if isinstance(_raw_json, str) and _raw_json.strip(): - try: - import json - _inner = json.loads(_raw_json) - if isinstance(_inner, dict): - _inner_err = _inner.get("error", {}) - if isinstance(_inner_err, dict): - _metadata_msg = str(_inner_err.get("message") or "").lower() - except (json.JSONDecodeError, TypeError): - pass - if not _body_msg: - _body_msg = str(body.get("message") or "").lower() - # Combine all message sources for pattern matching - parts = [_raw_msg] - if _body_msg and _body_msg not in _raw_msg: - parts.append(_body_msg) - if _metadata_msg and _metadata_msg not in _raw_msg and _metadata_msg not in _body_msg: - parts.append(_metadata_msg) - error_msg = " ".join(parts) + error_msg = _build_error_msg(error, body) provider_lower = (provider or "").strip().lower() model_lower = (model or "").strip().lower() @@ -905,14 +763,9 @@ def classify_api_error( return ClassifiedError(**defaults) # ── 0. Plugin classifiers (first valid result wins) ───────────── - # - # Consulted BEFORE the built-in pipeline so a provider plugin can both - # add classifications the core patterns miss and correct ones they get - # wrong for its provider (see the ``transform_api_error_classification`` entry in - # hermes_cli.plugins.VALID_HOOKS for the callback contract). Callback - # exceptions are isolated inside invoke_hook and malformed returns are - # dropped by the helper, so a broken plugin can never break - # classification — the guard here only covers import/dispatch failure. + # Runs before the built-in pipeline so a provider plugin can add or correct + # classifications. invoke_hook isolates callback failures; this guard only + # covers import/dispatch failure. try: from hermes_cli.plugins import get_plugin_error_classification plugin_classification = get_plugin_error_classification( @@ -941,38 +794,15 @@ def classify_api_error( # ── 1. Provider-specific patterns (highest priority) ──────────── - # Provider content-policy / safety-filter block. The provider has made a - # deterministic refusal decision about THIS prompt — retrying unchanged - # just reproduces the same refusal and burns paid attempts. Must run - # before status-based classification so a 400 safety block isn't - # downgraded to a generic ``format_error`` and a status-less block - # (OpenAI Codex SDK can raise without one) isn't left in the retryable - # ``unknown`` bucket. See issue #18028. + # Deterministic per-prompt safety refusal. Before status classification so + # a 400 block isn't downgraded to format_error and a status-less block + # isn't left retryable (#18028). if any(p in error_msg for p in _CONTENT_POLICY_BLOCKED_PATTERNS): - return _result( - FailoverReason.content_policy_blocked, - retryable=False, - should_fallback=True, - ) + return _result(FailoverReason.content_policy_blocked, retryable=False, should_fallback=True) - # Anthropic thinking block recovery (400). Two distinct failure modes, - # same recovery (strip all reasoning_details and retry without thinking - # blocks — see the thinking_signature handler in conversation_loop.py): - # 1. Signature mismatch: a thinking block is signed against the full - # turn content; any upstream mutation (context compression, session - # truncation, message merging) invalidates the signature. - # Pattern: "signature" + "thinking". - # 2. Frozen-block mutation: Anthropic rejects any change to the - # thinking/redacted_thinking blocks in the *latest* assistant - # message — "`thinking` or `redacted_thinking` blocks in the latest - # assistant message cannot be modified. These blocks must remain as - # they were in the original response." This carries no "signature" - # token, so the original pattern missed it and the turn hard-aborted - # as a non-retryable client error instead of self-healing. - # Pattern: "thinking" + ("cannot be modified" | "must remain as they were"). - # Don't gate on provider — OpenRouter proxies Anthropic errors, so the - # provider may be "openrouter" even though the error is Anthropic-specific. - # The combined patterns are unique enough. + # Anthropic thinking-block 400s: signature mismatch after any transcript + # mutation, or "blocks in the latest assistant message cannot be modified". + # Not gated on provider — OpenRouter proxies Anthropic errors. if ( status_code == 400 and "thinking" in error_msg @@ -982,108 +812,39 @@ def classify_api_error( or "must remain as they were" in error_msg ) ): - return _result( - FailoverReason.thinking_signature, - retryable=True, - should_compress=False, - ) + return _result(FailoverReason.thinking_signature, retryable=True, should_compress=False) # Anthropic long-context tier gate (429 "extra usage" + "long context") - if ( - status_code == 429 - and "extra usage" in error_msg - and "long context" in error_msg - ): - return _result( - FailoverReason.long_context_tier, - retryable=True, - should_compress=True, - ) + if status_code == 429 and "extra usage" in error_msg and "long context" in error_msg: + return _result(FailoverReason.long_context_tier, retryable=True, should_compress=True) - # Anthropic OAuth subscription rejects the 1M-context beta header. - # Observed error body: "The long context beta is not yet available for - # this subscription." Returned as HTTP 400 from native Anthropic when - # the subscription doesn't include 1M context, even though the request - # carries ``anthropic-beta: context-1m-2025-08-07``. The recovery path - # in run_agent.py rebuilds the Anthropic client with the beta stripped - # and retries once. Pattern is narrow enough that it won't collide with - # the 429 tier-gate pattern above (different status, different phrase). - if ( - status_code == 400 - and "long context beta" in error_msg - and "not yet available" in error_msg - ): - return _result( - FailoverReason.oauth_long_context_beta_forbidden, - retryable=True, - should_compress=False, - ) + # Anthropic OAuth subscription rejects the 1M-context beta header (400 + # "The long context beta is not yet available for this subscription"); + # run_agent rebuilds the client without the beta and retries once. + if status_code == 400 and "long context beta" in error_msg and "not yet available" in error_msg: + return _result(FailoverReason.oauth_long_context_beta_forbidden, retryable=True, should_compress=False) - # llama.cpp's ``json-schema-to-grammar`` converter (used by its OAI - # server to build GBNF tool-call parsers) rejects regex escape classes - # like ``\d``/``\w``/``\s`` and most ``format`` values. MCP servers - # routinely emit ``"pattern": "\\d{4}-\\d{2}-\\d{2}"`` for date/phone/ - # email params. llama.cpp surfaces this as HTTP 400 with one of a few - # recognizable phrases; on match we strip ``pattern``/``format`` from - # ``self.tools`` in the retry loop and retry once. Cloud providers are - # unaffected — they accept these keywords and we never hit this branch. - # - # Exclude Qwen/vLLM template raise_exception("No user query found…") - # wrapped by some local engines as applyPromptTemplate / "Unable to - # generate parser for this template". That is a poisoned transcript - # shape (handled via _INVALID_MESSAGE_BODY_PATTERNS → format_error), - # not a tool-schema grammar rejection — matching it here strips - # pattern/format keywords and retries uselessly while the real fix - # is /new (or a successful compression that preserves a user turn). - if status_code == 400: - _llama_cpp_grammar_hit = ( - "error parsing grammar" in error_msg - or "json-schema-to-grammar" in error_msg - or ( - "unable to generate parser" in error_msg - and "template" in error_msg - ) - ) - else: - _llama_cpp_grammar_hit = False - if ( - _llama_cpp_grammar_hit - and _NO_USER_QUERY_SIGNAL not in error_msg - ): - return _result( - FailoverReason.llama_cpp_grammar_pattern, - retryable=True, - should_compress=False, - ) + # llama.cpp json-schema-to-grammar rejects regex escapes / ``format`` in + # tool schemas (400); the retry loop strips pattern/format and retries. + # Exclude the Qwen/vLLM "No user query found" template error that local + # engines wrap as "Unable to generate parser for this template" — that is + # a poisoned transcript (→ format_error), not a grammar problem. + llama_cpp_grammar_hit = status_code == 400 and ( + "error parsing grammar" in error_msg + or "json-schema-to-grammar" in error_msg + or ("unable to generate parser" in error_msg and "template" in error_msg) + ) + if llama_cpp_grammar_hit and _NO_USER_QUERY_SIGNAL not in error_msg: + return _result(FailoverReason.llama_cpp_grammar_pattern, retryable=True, should_compress=False) - # xAI Grok subscription entitlement errors. - # - # xAI returns "You have either run out of available resources or do not - # have an active Grok subscription" through two distinct code paths: - # - # • HTTP 403 — status_code is set; _classify_by_status (step 2) routes - # it to FailoverReason.auth correctly, and _is_entitlement_failure - # then prevents the credential-refresh loop. - # - # • SSE ``type=error`` frame — surfaced as _StreamErrorEvent with - # status_code=None. _classify_by_status is skipped entirely, and - # "grok subscription" / "out of available resources" appear in none - # of the message-pattern lists below. Without this guard the error - # falls through to FailoverReason.unknown (retryable=True), burning - # max_retries before the agent stops — and _is_entitlement_failure - # is never called because it only runs under FailoverReason.auth. - # - # Both X Premium+ and SuperGrok subscribers hit this path when their - # subscription tier does not cover the requested model or feature. + # xAI Grok subscription entitlement. As HTTP 403 the status path handles + # it; as an SSE ``type=error`` frame there is no status and the message + # matches no pattern list, so it would burn max_retries as ``unknown``. if ( "do not have an active grok subscription" in error_msg or ("out of available resources" in error_msg and "grok" in error_msg) ): - return _result( - FailoverReason.auth, - retryable=False, - should_fallback=True, - ) + return _result(FailoverReason.auth, retryable=False, should_fallback=True) # ── 2. HTTP status code classification ────────────────────────── @@ -1099,22 +860,15 @@ def classify_api_error( if classified is not None: return classified - # Local MoA streaming compatibility errors are adapter-shape bugs, not a - # provider outage. Falling back to another model would silently switch the - # user's selected MoA route to a single-model answer (#55933 follow-up). + # Local MoA streaming adapter-shape bugs are not a provider outage; falling + # back would silently replace the MoA route with a single model (#55933). if provider_lower == "moa" and ( "'types.SimpleNamespace' object is not iterable" in str(error) or "'types.SimpleNamespace' object has no attribute 'index'" in str(error) ): - return _result( - FailoverReason.format_error, - retryable=False, - should_fallback=False, - ) + return _result(FailoverReason.format_error, retryable=False, should_fallback=False) - # Local MoA config drift is deterministic: a persisted session can retain - # a preset name that was later renamed/deleted. Retrying the same lookup - # cannot recover and makes a clear config error look like an API outage. + # Persisted MoA preset name that was renamed/deleted — deterministic config error. from agent.errors import MoAPresetNotFoundError if isinstance(error, MoAPresetNotFoundError): @@ -1138,103 +892,48 @@ def classify_api_error( if classified is not None: return classified - # ── 5. SSL certificate verification failures → fail fast ──────── - # A broken certificate chain (TLS-inspecting proxy, missing custom CA, - # expired/self-signed cert) is deterministic for the host — every retry - # reproduces the identical handshake failure. Fail immediately with - # actionable guidance instead of burning the retry budget first. - # Checked BEFORE the transient-SSL patterns: cert-verify messages also - # contain "[ssl:" which would otherwise match the transient list. - # Inspired by Claude Code v2.1.199 (July 2026). + # ── 5. SSL: deterministic cert failure → fail fast; transient alert → retry + # Cert-verify first: those messages also contain "[ssl:". Transient alerts + # are classified before the disconnect check so a large session doesn't + # compress on a flaky TLS handshake. if any(p in error_msg for p in _SSL_CERT_VERIFY_PATTERNS): - return _result( - FailoverReason.ssl_cert_verification, - retryable=False, - should_fallback=False, - ) - - # ── 5b. SSL/TLS transient errors → retry as timeout (not compression) ── - # SSL alerts mid-stream are transport hiccups, not server-side context - # overflow signals. Classify before the disconnect check so a large - # session doesn't incorrectly trigger context compression when the real - # cause is a flaky TLS handshake. Also matches when the error is - # wrapped in a generic exception whose message string carries the SSL - # alert text but the type isn't ssl.SSLError (happens with some SDKs - # that re-raise without chaining). + return _result(FailoverReason.ssl_cert_verification, retryable=False, should_fallback=False) if any(p in error_msg for p in _SSL_TRANSIENT_PATTERNS): return _result(FailoverReason.timeout, retryable=True) # ── 6. Server disconnect + large session → context overflow ───── - # Must come BEFORE generic transport error catch — a disconnect on - # a large session is more likely context overflow than a transient - # transport hiccup. Without this ordering, RemoteProtocolError - # always maps to timeout regardless of session size. - - is_disconnect = any(p in error_msg for p in _SERVER_DISCONNECT_PATTERNS) - if is_disconnect and not status_code: - # Reasoning-model override: a transport disconnect on a reasoning - # model is much more likely the upstream proxy idle-killing a - # long thinking stream than a true context overflow — even on - # large sessions. The default disconnect+large-session routing - # below would otherwise send the user into the compression - # branch (should_compress=True) and silently delete - # conversation history on a phantom context-length error. - # Reasoning models have multi-minute thinking phases that - # routinely exceed the cloud gateway's idle window (NVIDIA - # NIM ~120s — first-party repro at NVIDIA/NemoClaw#4846; - # OpenAI worker / Anthropic stream-idle similar). The - # per-reasoning-model stale-timeout floor in - # agent/reasoning_timeouts.py raises the stale-detector - # threshold to tolerate long thinking, so a true - # transport-layer failure here is recoverable via the retry - # path — not via context compression. Reclassify as timeout. - # (Part 1 of Fixes #52310.) + # Before the generic transport catch: a disconnect on a large session is + # more likely an overflow rejection than a transport hiccup. + if any(p in error_msg for p in _SERVER_DISCONNECT_PATTERNS) and not status_code: + # Reasoning models: a disconnect is far more likely the gateway + # idle-killing a long thinking stream than overflow — never compress + # (and silently drop history) on a phantom overflow (#52310). from agent.reasoning_timeouts import get_reasoning_stale_timeout_floor if get_reasoning_stale_timeout_floor(model) is not None: return _result(FailoverReason.timeout, retryable=True) - # Absolute token/message-count thresholds are only a proxy for smaller - # context windows. Large-context sessions can have hundreds of - # messages while still being far below their actual token budget. + # Absolute thresholds only proxy for smaller context windows. is_large = approx_tokens > context_length * 0.6 or ( context_length <= 256000 and (approx_tokens > 120000 or num_messages > 200) ) if is_large: - return _result( - FailoverReason.context_overflow, - retryable=True, - should_compress=True, - ) + return _result(FailoverReason.context_overflow, retryable=True, should_compress=True) return _result(FailoverReason.timeout, retryable=True) - # ── 7b. Stale-call circuit breaker → failover immediately ────── - # _check_stale_giveup() in agent/chat_completion_helpers.py raises a - # RuntimeError when the provider has been unresponsive for N - # consecutive stale attempts (default 5). The error is NOT a transport - # timeout — the circuit breaker fires *before* any network call to avoid - # an indefinite stall. Without this classification the RuntimeError - # falls through to FailoverReason.unknown (retryable=True), which burns - # all max_retries against the same dead provider (each retry hitting the - # circuit breaker instantly with zero network overhead) before fallback - # is attempted. Classify as non-retryable + should_fallback so the - # retry loop activates the next fallback provider on the first hit. + # ── 7. Stale-call circuit breaker → failover immediately ──────── + # _check_stale_giveup() raises RuntimeError before any network call; as + # ``unknown`` it would burn every retry instantly against the dead provider. if ( error_type == "RuntimeError" and "consecutive stale attempts" in error_msg and "aborting this call" in error_msg ): - return _result( - FailoverReason.timeout, - retryable=False, - should_fallback=True, - ) + return _result(FailoverReason.timeout, retryable=False, should_fallback=True) # ── 8. Transport / timeout heuristics ─────────────────────────── - if error_type in _TRANSPORT_ERROR_TYPES or isinstance(error, (TimeoutError, ConnectionError, OSError)): return _result(FailoverReason.timeout, retryable=True) # ── 9. Fallback: unknown ──────────────────────────────────────── - return _result(FailoverReason.unknown, retryable=True) @@ -1257,125 +956,47 @@ def _classify_by_status( """Classify based on HTTP status code with message-aware refinement.""" if status_code == 401: - # Not retryable on its own — credential pool rotation and - # provider-specific refresh (Codex, Anthropic, Nous) run before - # the retryability check in run_agent.py. If those succeed, the - # loop `continue`s. If they fail, retryable=False ensures we - # hit the client-error abort path (which tries fallback first). - return result_fn( - FailoverReason.auth, - retryable=False, - should_rotate_credential=True, - should_fallback=True, - ) + # Not retryable on its own: credential rotation / provider refresh run + # before the retryability check; if they fail, the client-error abort + # path (fallback first) is correct. + return result_fn(FailoverReason.auth, retryable=False, **_ROTATE_FALLBACK) if status_code == 403: - # OpenRouter 403 "key limit exceeded" is actually billing. Other - # providers also use 403 for account-plan or credit exhaustion. + # OpenRouter 403 "key limit exceeded" and similar plan/credit exhaustion are billing. if ( - ( - provider == "xai-oauth" - and error_code.lower() == _XAI_SPENDING_LIMIT_ERROR_CODE - ) + (provider == "xai-oauth" and error_code.lower() == _XAI_SPENDING_LIMIT_ERROR_CODE) or "key limit exceeded" in error_msg or "spending limit" in error_msg or any(p in error_msg for p in _BILLING_PATTERNS) ): - return result_fn( - FailoverReason.billing, - retryable=False, - should_rotate_credential=True, - should_fallback=True, - ) - return result_fn( - FailoverReason.auth, - retryable=False, - should_fallback=True, - ) + return _emit(result_fn, _V_BILLING) + return _emit(result_fn, _V_AUTH_FALLBACK) if status_code == 402: return _classify_402(error_msg, result_fn) if status_code == 404: - # Nous API currently surfaces HA/NAS credit depletion as a paid model - # becoming unavailable on the Free Tier, returned as 404 rather than - # 402. Treat that as entitlement/billing exhaustion, not a missing - # model, so the retry loop can show credit/top-up guidance. - if any(p in error_msg for p in _BILLING_PATTERNS): - return result_fn( - FailoverReason.billing, - retryable=False, - should_rotate_credential=True, - should_fallback=True, - ) - # OpenRouter policy-block 404 — distinct from "model not found". - # The model exists; the user's account privacy setting excludes the - # only endpoint serving it. Falling back to another provider won't - # help (same account setting applies). The error body already - # contains the fix URL, so just surface it. - if any(p in error_msg for p in _PROVIDER_POLICY_BLOCKED_PATTERNS): - return result_fn( - FailoverReason.provider_policy_blocked, - retryable=False, - should_fallback=False, - ) - if any(p in error_msg for p in _MODEL_NOT_FOUND_PATTERNS): - return result_fn( - FailoverReason.model_not_found, - retryable=False, - should_fallback=True, - ) - # A bare id that the provider's catalogue only knows in prefixed form - # is a malformed model id, not a routing glitch — NVIDIA NIM answers - # one with a naked ``404 page not found`` that names nothing, so the - # generic branch below burns three retries and reports what looks - # like an outage (#78796). Deterministic: don't retry, and let the - # model_not_found surface carry the real cause. + classified = _first_match(error_msg, _404_RULES, result_fn) + if classified is not None: + return classified + # Bare id the catalogue only knows prefixed → malformed id (NVIDIA NIM + # "404 page not found"), deterministic (#78796). if _model_id_missing_known_prefix(model, provider): - return result_fn( - FailoverReason.model_not_found, - retryable=False, - should_fallback=True, - ) - # Generic 404 with no "model not found" signal — could be a wrong - # endpoint path (common with local llama.cpp / Ollama / vLLM when - # the URL is slightly misconfigured), a proxy routing glitch, or - # a transient backend issue. Classifying these as model_not_found - # silently falls back to a different provider and tells the model - # the model is missing, which is wrong and wastes a turn. Treat - # as unknown so the retry loop surfaces the real error instead. - return result_fn( - FailoverReason.unknown, - retryable=True, - ) + return _emit(result_fn, _V_MODEL_NOT_FOUND) + # Generic 404 (wrong endpoint path, proxy glitch): model_not_found would + # silently fall back and misreport; stay unknown so the real error surfaces. + return result_fn(FailoverReason.unknown, retryable=True) if status_code == 413: - return result_fn( - FailoverReason.payload_too_large, - retryable=True, - should_compress=True, - ) + return result_fn(FailoverReason.payload_too_large, retryable=True, should_compress=True) if status_code == 429: - # Already checked long_context_tier above. Some providers (notably - # Z.AI / Zhipu) reuse HTTP 429 for server-wide overload — same status - # code as a true per-credential rate limit, but the credential is - # valid and the correct recovery is "back off and retry the same key", - # NOT "rotate the credential" (which exhausts the pool while the - # endpoint is still busy, and does nothing for a single-key user). - # Disambiguate on the error body so an overload 429 takes the - # transient-overload path instead of burning the pool. (#14038) + # Z.AI/Zhipu reuse 429 for server-wide overload: back off on the same + # key instead of burning the pool (#14038). if any(p in error_msg for p in _OVERLOADED_PATTERNS): - return result_fn( - FailoverReason.overloaded, - retryable=True, - ) - # Distinguish an OpenRouter-aggregator upstream 429 (an upstream model - # like DeepSeek rate-limited OpenRouter's aggregate traffic) from an - # account-level 429 (the user's key is actually throttled). OpenRouter - # wraps upstream errors with the outer message "Provider returned - # error" — the user's key is healthy, so marking it exhausted / rotating - # is wrong and burns the key for ~24min. Fall back to a different model. + return result_fn(FailoverReason.overloaded, retryable=True) + # OpenRouter-wrapped upstream 429: the user's key is healthy, so + # fall back to another model rather than rotating/benching the key. if _is_openrouter_upstream_error(body, provider): upstream_provider = _extract_upstream_provider_name(body) ctx = {"upstream_provider": upstream_provider} if upstream_provider else {} @@ -1386,54 +1007,24 @@ def _classify_by_status( should_fallback=True, error_context=ctx, ) - # Account/subscription usage exhaustion is a quota wall, not a - # request-rate throttle. Anthropic returns this as 429, so the generic - # branch below used to retry it and Desktop rendered a provider error - # instead of the billing/quota recovery. Preserve periodic quotas when - # the response supplies an explicit reset/retry signal. - # - # The check covers the narrow #93419 core (Anthropic's - # ``usage_limit_reached``) plus the broader ``_USAGE_LIMIT_PATTERNS`` - # ("quota", "limit exceeded", "key limit exceeded") so other providers' - # hard quota walls also route to billing — but ONLY when the message is - # not itself an explicit rate-limit phrase. Without that guard, - # "Rate limit exceeded" ("limit exceeded" substring) would wrongly - # promote to non-retryable billing. (broadening + guard credit #39441) + # Quota walls returned as 429 (Anthropic ``usage_limit_reached``, other + # providers' "quota"/"limit exceeded", explicit billing phrases) are + # billing — but ONLY when the body is not itself an explicit rate-limit + # phrase ("Rate limit exceeded" contains "limit exceeded") and carries + # no reset/retry signal (#93419, #39441). has_usage_limit = ( error_code.lower() == "usage_limit_reached" or "usage_limit_reached" in error_msg or any(p in error_msg for p in _USAGE_LIMIT_PATTERNS) ) - # Explicit billing phrases in a 429 body are a hard wall regardless of - # usage-limit wording — a provider that wraps "insufficient credits" in - # a 429 (rather than 402) was previously retried as a rate limit and - # burned the pool. (credit #39441) has_billing = any(p in error_msg for p in _BILLING_PATTERNS) - has_explicit_rate_limit = any( - p in error_msg for p in _RATE_LIMIT_PATTERNS - ) - has_transient_signal = _has_usage_limit_transient_signal( - error_msg, - body, - response_headers, - ) if ( (has_billing or has_usage_limit) - and not has_explicit_rate_limit - and not has_transient_signal + and not any(p in error_msg for p in _RATE_LIMIT_PATTERNS) + and not _has_usage_limit_transient_signal(error_msg, body, response_headers) ): - return result_fn( - FailoverReason.billing, - retryable=False, - should_rotate_credential=True, - should_fallback=True, - ) - return result_fn( - FailoverReason.rate_limit, - retryable=True, - should_rotate_credential=True, - should_fallback=True, - ) + return _emit(result_fn, _V_BILLING) + return _emit(result_fn, _V_RATE_LIMIT) if status_code == 400: return _classify_400( @@ -1446,157 +1037,72 @@ def _classify_by_status( ) if status_code in {500, 502}: - # Some OpenAI-compatible gateways return request-validation errors - # with a 5xx status (codex.nekos.me returns 502 for unknown/ - # unsupported parameters). These are deterministic — every retry - # gets the identical rejection — so the generic "5xx → retryable - # server_error" rule turns one bad request into a retry flood. - # Detect the unambiguous request-validation signals (in either the - # message text or the structured error code) and fail fast. - # - # Exception: a parameter WE never sent on this route was injected by - # the provider/proxy itself, so the rejection is not deterministic and - # the generic retryable-5xx handling is correct. Mirrors the guard in - # _classify_400 — see _is_server_injected_param_rejection. + # Deterministic request-validation errors returned as 5xx + # (codex.nekos.me) must fail fast, not retry-flood — unless the + # rejected parameter was injected server-side (see _classify_400). if ( any(p in error_msg for p in _REQUEST_VALIDATION_PATTERNS) - or error_code.lower() in {"invalid_request_error", "unknown_parameter", - "unsupported_parameter"} + or error_code.lower() in _5XX_VALIDATION_CODES ) and not _is_server_injected_param_rejection(error_msg, provider): - return result_fn( - FailoverReason.format_error, - retryable=False, - should_fallback=True, - ) - # Some local inference servers (notably llama.cpp / llama-server) - # report context overflow with an HTTP 500 instead of the standard - # 400/413. The request-validation guard above already ran, so any - # remaining explicit context-overflow signal routes into the - # compression-and-retry path (mirroring _classify_400) instead of - # blind server_error retries that exhaust and drop the turn. - # Empty-response advisories that mention "max_tokens" must not enter - # that compression path. - if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS): - return result_fn( - FailoverReason.server_error, - retryable=True, - should_compress=False, - ) - if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS): - return result_fn( - FailoverReason.context_overflow, - retryable=True, - should_compress=True, - ) - return result_fn(FailoverReason.server_error, retryable=True) + return _emit(result_fn, _V_FORMAT_ERROR) + classified = _first_match(error_msg, _OVERFLOW_AS_5XX_RULES, result_fn) + return classified if classified is not None else result_fn(FailoverReason.server_error, retryable=True) if status_code in {503, 529}: - # Same overflow-as-5xx variant (server busy / model-load OOM, or a - # Cloudflare/Tailscale hop relabeling the status). Route explicit - # overflow bodies into compression; otherwise treat as transient - # overload and retry. - if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS): - return result_fn( - FailoverReason.server_error, - retryable=True, - should_compress=False, - ) - if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS): - return result_fn( - FailoverReason.context_overflow, - retryable=True, - should_compress=True, - ) - return result_fn(FailoverReason.overloaded, retryable=True) + classified = _first_match(error_msg, _OVERFLOW_AS_5XX_RULES, result_fn) + return classified if classified is not None else result_fn(FailoverReason.overloaded, retryable=True) - # 408 Request Timeout — a transient timing failure the server itself flags - # as safe to retry (RFC 9110 §15.5.9), not a malformed request. Commonly - # emitted by reverse proxies sitting in front of self-hosted backends - # (llama.cpp / Ollama / vLLM) when a long generation outruns the proxy's - # request-read window. Route to the dedicated ``timeout`` reason (rebuild - # client + retry) instead of falling through to the generic 4xx bucket - # below, which would abort the turn on a retry-safe error the same way it - # aborts a 400 Bad Request. + # 408 Request Timeout is retry-safe (RFC 9110 §15.5.9) — proxies in front + # of self-hosted backends emit it when generation outruns the read window. if status_code == 408: return result_fn(FailoverReason.timeout, retryable=True) - # Other 4xx — non-retryable if 400 <= status_code < 500: - return result_fn( - FailoverReason.format_error, - retryable=False, - should_fallback=True, - ) + return _emit(result_fn, _V_FORMAT_ERROR) - # Other 5xx — retryable if 500 <= status_code < 600: return result_fn(FailoverReason.server_error, retryable=True) return None -def _has_usage_limit_transient_signal( - error_msg: str, - body: dict, - response_headers, -) -> bool: +_RESET_FIELDS = ("resets_in_seconds", "resets_at", "reset_at", "retry_after") +_RESET_HEADERS = ("retry-after", "Retry-After", "x-ratelimit-reset", "X-RateLimit-Reset") + + +def _has_usage_limit_transient_signal(error_msg: str, body: dict, response_headers) -> bool: """Return whether a usage-limit response identifies a reset window.""" if any(pattern in error_msg for pattern in _USAGE_LIMIT_TRANSIENT_SIGNALS): return True - payloads = [body] if isinstance(body, dict) and isinstance(body.get("error"), dict): payloads.append(body["error"]) - reset_fields = ("resets_in_seconds", "resets_at", "reset_at", "retry_after") for payload in payloads: - if not isinstance(payload, dict): - continue - if any( - payload.get(field) is not None and payload.get(field) != "" - for field in reset_fields - ): + if isinstance(payload, dict) and any(payload.get(f) not in (None, "") for f in _RESET_FIELDS): return True - if response_headers and hasattr(response_headers, "get"): - for header in ( - "retry-after", - "Retry-After", - "x-ratelimit-reset", - "X-RateLimit-Reset", - ): - value = response_headers.get(header) - if value is not None and value != "": - return True + return any(response_headers.get(h) not in (None, "") for h in _RESET_HEADERS) return False def _classify_402(error_msg: str, result_fn) -> ClassifiedError: - """Disambiguate 402: billing exhaustion vs transient usage limit. + """Disambiguate 402: "usage limit, try again in 5 minutes" is a periodic quota, not billing.""" + if ( + any(p in error_msg for p in _USAGE_LIMIT_PATTERNS) + and any(p in error_msg for p in _USAGE_LIMIT_TRANSIENT_SIGNALS) + ): + return _emit(result_fn, _V_RATE_LIMIT) + return _emit(result_fn, _V_BILLING) - The key insight from OpenClaw: some 402s are transient rate limits - disguised as payment errors. "Usage limit, try again in 5 minutes" - is NOT a billing problem — it's a periodic quota that resets. - """ - # Check for transient usage-limit signals first - has_usage_limit = any(p in error_msg for p in _USAGE_LIMIT_PATTERNS) - has_transient_signal = any(p in error_msg for p in _USAGE_LIMIT_TRANSIENT_SIGNALS) - if has_usage_limit and has_transient_signal: - # Transient quota — treat as rate limit, not billing - return result_fn( - FailoverReason.rate_limit, - retryable=True, - should_rotate_credential=True, - should_fallback=True, - ) - - # Confirmed billing exhaustion - return result_fn( - FailoverReason.billing, - retryable=False, - should_rotate_credential=True, - should_fallback=True, - ) +def _body_message_candidates(body: dict) -> Iterator[Any]: + """Body message fields in priority order (OpenAI, flat, litellm/Bedrock proxy shapes).""" + err_obj = body.get("error", {}) + yield err_obj.get("message") if isinstance(err_obj, dict) else None + yield body.get("message") + yield body.get("errorMessage") + args = body.get("errorArgs") + yield args.get("reason") if isinstance(args, dict) else None def _classify_400( @@ -1612,116 +1118,41 @@ def _classify_400( result_fn, ) -> ClassifiedError: """Classify 400 Bad Request — context overflow, format error, or generic.""" + classified = _first_match(error_msg, _IMAGE_TOOL_RULES, result_fn) + if classified is not None: + return classified - # Multimodal tool content rejected from 400. Must be checked BEFORE - # image_too_large because the recovery is different (strip image parts - # from tool messages, mark the model as no-list-tool-content for the - # rest of the session) and BEFORE context_overflow because some of the - # patterns ("text is not set") are ambiguous in isolation but become - # specific when combined with a 400 on a request known to contain - # multimodal tool content. - if any(p in error_msg for p in _MULTIMODAL_TOOL_CONTENT_PATTERNS): - return result_fn( - FailoverReason.multimodal_tool_content_unsupported, - retryable=True, - ) - - # Image-corruption from 400 (xAI's undecodable-image check fires this way). - # Must be checked BEFORE image_too_large: both are image-shaped 400s, but - # corrupt bytes need strip-and-retry, not shrink-and-retry — shrinking - # can't repair a truncated/malformed PNG. - if any(p in error_msg for p in _IMAGE_CORRUPT_PATTERNS): - return result_fn( - FailoverReason.image_corrupt, - retryable=True, - ) - - # Image-too-large from 400 (Anthropic's 5 MB per-image check fires this way). - # Must be checked BEFORE context_overflow because messages can trip both - # patterns ("exceeds" + "image") and image-shrink is a cheaper recovery. - if any(p in error_msg for p in _IMAGE_TOO_LARGE_PATTERNS): - return result_fn( - FailoverReason.image_too_large, - retryable=True, - ) - - # Invalid encrypted reasoning replay blob (OpenAI Responses API). Must be - # checked BEFORE context_overflow because some surfaces emit messages that - # contain context-like phrasing ("encrypted content … could not be - # verified") which could otherwise trip the context_overflow heuristics. - # ``error_msg`` is lowercased upstream — match accordingly. + # Invalid encrypted reasoning replay blob (OpenAI Responses). Before + # context_overflow: "encrypted content … could not be verified" can trip + # the overflow heuristics. error_code_lower = (error_code or "").lower() if ( error_code_lower == "invalid_encrypted_content" or "invalid_encrypted_content" in error_msg - or ( - "encrypted content for item" in error_msg - and "could not be verified" in error_msg - ) + or ("encrypted content for item" in error_msg and "could not be verified" in error_msg) or "could not decrypt the provided encrypted_content" in error_msg ): - return result_fn( - FailoverReason.invalid_encrypted_content, - retryable=True, - should_fallback=False, - ) + return result_fn(FailoverReason.invalid_encrypted_content, retryable=True, should_fallback=False) - # Server-injected parameter rejection: a 400 blaming a request field the - # client never sent. MUST be checked BEFORE the request-validation branch - # below, which would otherwise class it as a deterministic format_error and - # abort the turn. - # - # Observed live on the Codex OAuth backend (chatgpt.com/backend-api/codex): - # it intermittently adds ``prompt_cache_retention`` to its own upstream - # call and then rejects it, so a byte-identical request succeeds on retry - # (measured ~20% failure over n=20 on a minimal 1-message request that - # provably carried no cache parameters). Retrying is the correct and only - # recovery; failing fast burnt an entire large-context request per attempt. + # A 400 blaming a field this route never sent (Codex OAuth backend injects + # and then rejects prompt_cache_retention ~20% of the time): transient, + # retry the identical request; never compress. Before the validation branch. if _is_server_injected_param_rejection(error_msg, provider): - return result_fn( - FailoverReason.server_error, - retryable=True, - # The request shape was fine — never route this into compression. - should_compress=False, - ) + return result_fn(FailoverReason.server_error, retryable=True, should_compress=False) - # Request-validation errors (unsupported / unknown parameter) MUST be - # checked BEFORE context_overflow. A GPT-5 model rejecting max_tokens - # returns: - # "Unsupported parameter: 'max_tokens' is not supported with this model. - # Use 'max_completion_tokens' instead." - # That string contains the literal substring "max_tokens", which historically - # sat in _CONTEXT_OVERFLOW_PATTERNS — so without this guard the 400 is - # misclassified as context_overflow, routed into the compression loop, - # re-sent with the same bad parameter, and ends in "Cannot compress - # further". These errors are deterministic (every retry gets the identical - # rejection), so classify as a non-retryable format_error and fall back. - # - # NOTE: we deliberately do NOT key off the generic ``invalid_request_error`` - # code here — OpenAI stamps that same code on genuine context-overflow 400s, - # so matching it would mis-route real overflows away from compression. The - # unambiguous signals are the explicit "unsupported/unknown parameter" - # message text and the specific parameter-level error codes. + # Unsupported/unknown parameter before context_overflow: GPT-5's + # "Unsupported parameter: 'max_tokens'…" contains the overflow pattern + # "max_tokens". Generic ``invalid_request_error`` is deliberately NOT used + # here — OpenAI stamps it on genuine overflow 400s too. if ( - any(p in error_msg for p in _REQUEST_VALIDATION_PATTERNS - if p != "invalid_request_error") - or error_code_lower in {"unknown_parameter", "unsupported_parameter"} + any(p in error_msg for p in _400_VALIDATION_PATTERNS) + or error_code_lower in _400_VALIDATION_CODES ): - return result_fn( - FailoverReason.format_error, - retryable=False, - should_fallback=True, - ) + return _emit(result_fn, _V_FORMAT_ERROR) - # Malformed message array (empty-content assistant stub, etc.). Must be - # checked BEFORE context_overflow: the input can be tiny, so the generic - # "400 + large session" heuristic would otherwise mis-route it into the - # compression loop and thrash until "Cannot compress further" on every - # retry (the request is unchanged, so compression cannot fix it). This is - # a deterministic request-shape rejection — fail fast as a non-retryable - # format_error and fall back. Checked against the message text AND the - # structured error code, since proxies (litellm/Bedrock) surface the - # signal in errorCode=INVALID_REQUEST_BODY. + # Malformed message array (empty-content assistant stub, etc.) before + # context_overflow: the input can be tiny and compression cannot fix it. + # Proxies (litellm/Bedrock) surface it as errorCode=INVALID_REQUEST_BODY. if ( any(p in error_msg for p in _INVALID_MESSAGE_BODY_PATTERNS) or error_code_lower == "invalid_request_body" @@ -1734,109 +1165,30 @@ def _classify_400( "approx_tokens=%s. error=%.200s", num_messages, approx_tokens, error_msg, ) - return result_fn( - FailoverReason.format_error, - retryable=False, - should_fallback=True, - ) + return _emit(result_fn, _V_FORMAT_ERROR) - # Empty-provider-response advisories must not enter compression. They - # often mention "max_tokens" as a possible cause and used to match the - # bare overflow pattern, then thrash compress until "Cannot compress - # further" on an otherwise healthy session (custom endpoints / nano-gpt). - if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS): - return result_fn( - FailoverReason.server_error, - retryable=True, - should_compress=False, - ) + classified = _first_match(error_msg, _400_TAIL_RULES, result_fn) + if classified is not None: + return classified - # Context overflow from 400 - if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS): - return result_fn( - FailoverReason.context_overflow, - retryable=True, - should_compress=True, - ) - - # Some providers return model-not-found as 400 instead of 404 (e.g. OpenRouter). - if any(p in error_msg for p in _PROVIDER_POLICY_BLOCKED_PATTERNS): - return result_fn( - FailoverReason.provider_policy_blocked, - retryable=False, - should_fallback=False, - ) - if any(p in error_msg for p in _MODEL_NOT_FOUND_PATTERNS): - return result_fn( - FailoverReason.model_not_found, - retryable=False, - should_fallback=True, - ) - - # Some providers return rate limit / billing errors as 400 instead of 429/402. - # Check these patterns before falling through to format_error. - if any(p in error_msg for p in _RATE_LIMIT_PATTERNS): - return result_fn( - FailoverReason.rate_limit, - retryable=True, - should_rotate_credential=True, - should_fallback=True, - ) - if any(p in error_msg for p in _BILLING_PATTERNS): - return result_fn( - FailoverReason.billing, - retryable=False, - should_rotate_credential=True, - should_fallback=True, - # "out of extra usage" on a 400 is ambiguous — it can also be a - # content-filter rejection (#82154). Mark the verdict unverified - # so downstream hedges and the pool skips the 1-hour bench. - error_context=_billing_ambiguity_context(error_msg), - ) - - # Generic 400 + large session → probable context overflow - # Anthropic sometimes returns a bare "Error" message when context is too large + # Generic 400 + large session → probable context overflow (Anthropic can + # return a bare "Error"). Proxy shapes are recognised so a long, descriptive + # rejection is not mistaken for a bare error. err_body_msg = "" if isinstance(body, dict): - err_obj = body.get("error", {}) - if isinstance(err_obj, dict): - err_body_msg = str(err_obj.get("message") or "").strip().lower() - # Responses API (and some providers) use flat body: {"message": "..."} - if not err_body_msg: - err_body_msg = str(body.get("message") or "").strip().lower() - # litellm / Bedrock proxies use a custom shape: {"errorMessage": "...", - # "errorCode": "...", "errorArgs": {"reason": "..."}}. Without these - # keys err_body_msg stays "" and a long, descriptive rejection is - # wrongly treated as a "generic" (bare) error below, which — on a - # large session — mis-routes into the compression loop. Recognize - # them so the is_generic heuristic sees the real message length. - if not err_body_msg: - err_body_msg = str(body.get("errorMessage") or "").strip().lower() - if not err_body_msg: - _args = body.get("errorArgs") - if isinstance(_args, dict): - err_body_msg = str(_args.get("reason") or "").strip().lower() + err_body_msg = next( + (m for m in (str(c or "").strip().lower() for c in _body_message_candidates(body)) if m), + "", + ) is_generic = len(err_body_msg) < 30 or err_body_msg in {"error", ""} - # Absolute token/message-count thresholds are only a proxy for smaller - # context windows. Large-context sessions can have many messages while - # still being far below their actual token budget. + # Absolute thresholds only proxy for smaller context windows. is_large = approx_tokens > context_length * 0.4 or ( context_length <= 256000 and (approx_tokens > 80000 or num_messages > 80) ) - if is_generic and is_large: - return result_fn( - FailoverReason.context_overflow, - retryable=True, - should_compress=True, - ) + return result_fn(FailoverReason.context_overflow, retryable=True, should_compress=True) - # Non-retryable format error - return result_fn( - FailoverReason.format_error, - retryable=False, - should_fallback=True, - ) + return _emit(result_fn, _V_FORMAT_ERROR) # ── Error code classification ─────────────────────────────────────────── @@ -1847,57 +1199,14 @@ def _classify_by_error_code( """Classify by structured error codes from the response body.""" code_lower = error_code.lower() - if ( - code_lower == PROVIDER_STREAM_NON_JSON_ERROR_CODE - and "request validation failed:" in error_msg - ): - # Some OpenAI-compatible endpoints encode deterministic request - # validation failures as plain-text ``event: error`` SSE data behind - # HTTP 200. Retrying the unchanged request cannot succeed, but a - # configured provider fallback still may. - return result_fn( - FailoverReason.format_error, - retryable=False, - should_fallback=True, - ) + # Deterministic request-validation failures encoded as plain-text + # ``event: error`` SSE data behind HTTP 200: retrying cannot succeed, a + # configured fallback still may. + if code_lower == PROVIDER_STREAM_NON_JSON_ERROR_CODE and "request validation failed:" in error_msg: + return _emit(result_fn, _V_FORMAT_ERROR) - if code_lower in {"resource_exhausted", "throttled", "rate_limit_exceeded"}: - return result_fn( - FailoverReason.rate_limit, - retryable=True, - should_rotate_credential=True, - ) - - if code_lower in _BILLING_ERROR_CODES: - return result_fn( - FailoverReason.billing, - retryable=False, - should_rotate_credential=True, - should_fallback=True, - ) - - if code_lower in {"model_not_found", "model_not_available", "invalid_model"}: - return result_fn( - FailoverReason.model_not_found, - retryable=False, - should_fallback=True, - ) - - if code_lower in {"context_length_exceeded", "max_tokens_exceeded"}: - return result_fn( - FailoverReason.context_overflow, - retryable=True, - should_compress=True, - ) - - if code_lower == "invalid_encrypted_content": - return result_fn( - FailoverReason.invalid_encrypted_content, - retryable=True, - should_fallback=False, - ) - - return None + verdict = _ERROR_CODE_VERDICTS.get(code_lower) + return _emit(result_fn, verdict) if verdict is not None else None # ── Message pattern classification ────────────────────────────────────── @@ -1911,187 +1220,50 @@ def _classify_by_message( result_fn, ) -> Optional[ClassifiedError]: """Classify based on error message patterns when no status code is available.""" + classified = _first_match(error_msg, _MESSAGE_HEAD_RULES, result_fn) + if classified is not None: + return classified - # Payload-too-large patterns (from message text when no status_code) - if any(p in error_msg for p in _PAYLOAD_TOO_LARGE_PATTERNS): - return result_fn( - FailoverReason.payload_too_large, - retryable=True, - should_compress=True, - ) + # Status-less usage limits need the same disambiguation as 402. + if any(p in error_msg for p in _USAGE_LIMIT_PATTERNS): + if any(p in error_msg for p in _USAGE_LIMIT_TRANSIENT_SIGNALS): + return _emit(result_fn, _V_RATE_LIMIT) + return _emit(result_fn, _V_BILLING) - # Multimodal tool content patterns (from message text when no status_code) - if any(p in error_msg for p in _MULTIMODAL_TOOL_CONTENT_PATTERNS): - return result_fn( - FailoverReason.multimodal_tool_content_unsupported, - retryable=True, - ) - - # Image-corruption patterns (from message text when no status_code) - if any(p in error_msg for p in _IMAGE_CORRUPT_PATTERNS): - return result_fn( - FailoverReason.image_corrupt, - retryable=True, - ) - - # Image-too-large patterns (from message text when no status_code) - if any(p in error_msg for p in _IMAGE_TOO_LARGE_PATTERNS): - return result_fn( - FailoverReason.image_too_large, - retryable=True, - ) - - # Usage-limit patterns need the same disambiguation as 402: some providers - # surface "usage limit" errors without an HTTP status code. A transient - # signal ("try again", "resets at", …) means it's a periodic quota, not - # billing exhaustion. - has_usage_limit = any(p in error_msg for p in _USAGE_LIMIT_PATTERNS) - if has_usage_limit: - has_transient_signal = any(p in error_msg for p in _USAGE_LIMIT_TRANSIENT_SIGNALS) - if has_transient_signal: - return result_fn( - FailoverReason.rate_limit, - retryable=True, - should_rotate_credential=True, - should_fallback=True, - ) - return result_fn( - FailoverReason.billing, - retryable=False, - should_rotate_credential=True, - should_fallback=True, - ) - - # Overloaded / server-busy patterns — must come BEFORE the rate_limit and - # billing checks so that a message-only "overloaded" (no 503/529 status, - # e.g. some Anthropic-compatible proxies) classifies as a transient - # overload (backoff + retry) instead of falling through to `unknown` or - # incorrectly triggering credential rotation. - if any(p in error_msg for p in _OVERLOADED_PATTERNS): - return result_fn( - FailoverReason.overloaded, - retryable=True, - ) - - # Billing patterns - if any(p in error_msg for p in _BILLING_PATTERNS): - return result_fn( - FailoverReason.billing, - retryable=False, - should_rotate_credential=True, - should_fallback=True, - # Status-less path: adapters can strip the HTTP status from the - # Anthropic "out of extra usage" 400, so the same ambiguity - # marking applies here (#82154). - error_context=_billing_ambiguity_context(error_msg), - ) - - # Rate limit patterns - if any(p in error_msg for p in _RATE_LIMIT_PATTERNS): - return result_fn( - FailoverReason.rate_limit, - retryable=True, - should_rotate_credential=True, - should_fallback=True, - ) - - # Empty-provider-response advisories (often mention "max_tokens") must - # retry without compression — see the matching 400-path guard above. - if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS): - return result_fn( - FailoverReason.server_error, - retryable=True, - should_compress=False, - ) - - # Context overflow patterns - if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS): - return result_fn( - FailoverReason.context_overflow, - retryable=True, - should_compress=True, - ) - - # Auth patterns - # Auth errors should NOT be retried directly — the credential is invalid and - # retrying with the same key will always fail. Set retryable=False so the - # caller triggers credential rotation (should_rotate_credential=True) or - # provider fallback rather than an immediate retry loop. - if any(p in error_msg for p in _AUTH_PATTERNS): - return result_fn( - FailoverReason.auth, - retryable=False, - should_rotate_credential=True, - should_fallback=True, - ) - - # Provider policy-block (aggregator-side guardrail) — check before - # model_not_found so we don't mis-label as a missing model. - if any(p in error_msg for p in _PROVIDER_POLICY_BLOCKED_PATTERNS): - return result_fn( - FailoverReason.provider_policy_blocked, - retryable=False, - should_fallback=False, - ) - - # Model not found patterns - if any(p in error_msg for p in _MODEL_NOT_FOUND_PATTERNS): - return result_fn( - FailoverReason.model_not_found, - retryable=False, - should_fallback=True, - ) - - # Timeout message patterns — generic exception types (e.g. RuntimeError) - # raised by local shims or custom providers that internally wrap a - # subprocess/HTTP timeout. Classified as transport timeout so the retry - # loop rebuilds the client instead of treating the turn as an empty - # model response. - if any(p in error_msg for p in _TIMEOUT_MESSAGE_PATTERNS): - return result_fn(FailoverReason.timeout, retryable=True) - - # Connection-establishment / DNS failure message patterns — same shim - # problem as the timeout patterns above: the wrapping exception type is - # generic, so _TRANSPORT_ERROR_TYPES never matches and the error would - # fall through to FailoverReason.unknown. Classified as timeout (the - # transport bucket) so the retry loop's eager transport fallback and - # client rebuild apply. Never routes to compression: a connection that - # was never established is not a context-overflow signal. - if any(p in error_msg for p in _CONNECTION_MESSAGE_PATTERNS): - return result_fn(FailoverReason.timeout, retryable=True) - - return None + return _first_match(error_msg, _MESSAGE_TAIL_RULES, result_fn) # ── Helpers ───────────────────────────────────────────────────────────── +def _cause_chain(error: Exception) -> Iterator[Any]: + """Yield the error and its __cause__/__context__ chain, at most 5 deep.""" + current = error + for _ in range(5): + yield current + cause = getattr(current, "__cause__", None) or getattr(current, "__context__", None) + if cause is None or cause is current: + return + current = cause + + def _extract_status_code(error: Exception) -> Optional[int]: """Walk the error and its cause chain to find an HTTP status code.""" - current = error - for _ in range(5): # Max depth to prevent infinite loops + for current in _cause_chain(error): code = getattr(current, "status_code", None) if isinstance(code, int): return code - # Some SDKs use .status instead of .status_code - code = getattr(current, "status", None) + code = getattr(current, "status", None) # some SDKs use .status if isinstance(code, int) and 100 <= code < 600: return code - # Walk cause chain - cause = getattr(current, "__cause__", None) or getattr(current, "__context__", None) - if cause is None or cause is current: - break - current = cause return None def _extract_error_body(error: Exception) -> dict: """Extract the structured error body from an SDK exception or its cause chain.""" - current = error - for _ in range(5): # Match _extract_status_code() traversal depth. + for current in _cause_chain(error): body = getattr(current, "body", None) if isinstance(body, dict): return body - # Some errors have .response.json() response = getattr(current, "response", None) if response is not None: try: @@ -2100,71 +1272,41 @@ def _extract_error_body(error: Exception) -> dict: return json_body except Exception: pass - cause = getattr(current, "__cause__", None) or getattr(current, "__context__", None) - if cause is None or cause is current: - break - current = cause return {} def _extract_response_headers(error: Exception): """Walk the error and its cause chain to find response headers.""" - current = error - for _ in range(5): - response = getattr(current, "response", None) - headers = getattr(response, "headers", None) + for current in _cause_chain(error): + headers = getattr(getattr(current, "response", None), "headers", None) if headers and hasattr(headers, "get"): return headers - cause = getattr(current, "__cause__", None) or getattr(current, "__context__", None) - if cause is None or cause is current: - break - current = cause return {} -def _extract_error_code(body: dict) -> str: - """Extract an error code string from the response body.""" - if not body: - return "" +def _code_from_payload(payload: Any, top_keys: Sequence[str], peek_message: bool) -> str: + """Code/type from ``payload.error`` or a top-level key; ``"400"`` is not a code. - def _code_from_payload(payload) -> str: - """Extract a code/type from a nested error payload dict (defensive).""" - if not isinstance(payload, dict): - return "" - payload_error = payload.get("error", {}) - if isinstance(payload_error, dict): - nested = payload_error.get("code") or payload_error.get("type") or "" - if isinstance(nested, str) and nested.strip() and nested.strip() != "400": - return nested.strip() - code = payload.get("code") or payload.get("error_code") or "" - if isinstance(code, (str, int)): - text = str(code).strip() - if text and text != "400": - return text + With ``peek_message``, a JSON string in ``error.message`` is parsed for a + nested code (Responses API surfaces ``invalid_encrypted_content`` this way). + """ + if not isinstance(payload, dict): return "" - - error_obj = body.get("error", {}) + error_obj = payload.get("error", {}) if isinstance(error_obj, dict): code = error_obj.get("code") or error_obj.get("type") or "" if isinstance(code, str) and code.strip() and code.strip() != "400": return code.strip() - - # Some providers wrap the real JSON error body as a string inside - # error.message — peek into it for a nested code (e.g. Responses API - # surfaces ``invalid_encrypted_content`` this way). message = error_obj.get("message") - if isinstance(message, str) and message.strip().startswith("{"): - import json + if peek_message and isinstance(message, str) and message.strip().startswith("{"): try: inner = json.loads(message) except (json.JSONDecodeError, TypeError): inner = None - nested_code = _code_from_payload(inner) + nested_code = _code_from_payload(inner, ("code", "error_code"), False) if nested_code: return nested_code - - # Top-level code - code = body.get("code") or body.get("error_code") or body.get("errorCode") or "" + code = next((payload.get(k) for k in top_keys if payload.get(k)), "") if isinstance(code, (str, int)): text = str(code).strip() if text and text != "400": @@ -2172,73 +1314,44 @@ def _extract_error_code(body: dict) -> str: return "" +def _extract_error_code(body: dict) -> str: + """Extract an error code string from the response body.""" + return _code_from_payload(body, ("code", "error_code", "errorCode"), True) if body else "" + + def _extract_message(error: Exception, body: dict) -> str: - """Extract the most informative error message.""" - # Try structured body first - if body: - error_obj = body.get("error", {}) - if isinstance(error_obj, dict): - msg = error_obj.get("message", "") - if isinstance(msg, str) and msg.strip(): - return msg.strip()[:500] - msg = body.get("message", "") + """Extract the most informative error message (structured body first).""" + for msg in _body_message_candidates(body or {}): if isinstance(msg, str) and msg.strip(): return msg.strip()[:500] - # litellm / Bedrock proxy shape: {"errorMessage": "...", - # "errorArgs": {"reason": "..."}}. - msg = body.get("errorMessage", "") - if isinstance(msg, str) and msg.strip(): - return msg.strip()[:500] - args = body.get("errorArgs") - if isinstance(args, dict): - reason = args.get("reason", "") - if isinstance(reason, str) and reason.strip(): - return reason.strip()[:500] - # Fallback to str(error) return str(error)[:500] def _is_openrouter_upstream_error(body: Any, provider: str) -> bool: - """Detect OpenRouter's aggregator-wrapped upstream provider errors. + """Detect OpenRouter's "Provider returned error" wrapper around an upstream failure. - OpenRouter returns errors from upstream model providers (DeepSeek, - Anthropic, etc.) wrapped with the outer message "Provider returned error" - and the real error nested in ``metadata.raw``. This signal means the - user's OpenRouter key is healthy — the upstream provider is the one that - failed — so credential rotation is the wrong recovery. + The user's OpenRouter key is healthy — the upstream provider failed — so + credential rotation is the wrong recovery. """ if not isinstance(body, dict): return False - provider_lower = (provider or "").strip().lower() err = body.get("error") if not isinstance(err, dict): return False - outer_msg = str(err.get("message") or "").strip().lower() - if outer_msg != "provider returned error": + if str(err.get("message") or "").strip().lower() != "provider returned error": return False - # Require either the explicit OpenRouter provider OR the metadata shape - # that only OpenRouter produces (metadata.raw / metadata.provider_name). - if provider_lower == "openrouter": + if (provider or "").strip().lower() == "openrouter": return True + # Otherwise require the metadata shape only OpenRouter produces. metadata = err.get("metadata") - if isinstance(metadata, dict) and ( - "raw" in metadata or "provider_name" in metadata - ): - return True - return False + return isinstance(metadata, dict) and ("raw" in metadata or "provider_name" in metadata) def _extract_upstream_provider_name(body: Any) -> Optional[str]: """Pull the upstream provider name out of OpenRouter's error metadata.""" - if not isinstance(body, dict): - return None - err = body.get("error") - if not isinstance(err, dict): - return None - metadata = err.get("metadata") - if not isinstance(metadata, dict): - return None - name = metadata.get("provider_name") + err = body.get("error") if isinstance(body, dict) else None + metadata = err.get("metadata") if isinstance(err, dict) else None + name = metadata.get("provider_name") if isinstance(metadata, dict) else None if isinstance(name, str) and name.strip(): return name.strip() return None