chore: anchor fresh-start history to upstream

This commit is contained in:
2026-09-16 15:06:54 +08:00
parent 64a9b43261
commit 012c9c6bed
3514 changed files with 48322 additions and 187346 deletions
+5 -56
View File
@@ -20,13 +20,6 @@ logger = logging.getLogger(__name__)
# before any completion chunk arrives; distinct from generic JSON parse errors.
PROVIDER_STREAM_NON_JSON_ERROR_CODE = "provider_stream_non_json_data"
# Same rejection with an EMPTY payload: the frame carried no ``data`` at all (``data:`` /
# ``event: ping`` / ``id:`` with no content). Per the SSE spec those are legal keepalives /
# no-ops, not malformed payloads — a degraded gateway answers every streaming request with
# them, so the session switches to non-streaming instead of re-streaming into the same
# window. See ``chat_completion_helpers._maybe_disable_streaming``.
PROVIDER_STREAM_EMPTY_FRAME_ERROR_CODE = "provider_stream_empty_frame"
# ── Error taxonomy ──────────────────────────────────────────────────────
@@ -92,21 +85,13 @@ class ClassifiedError:
# Billing exhaustion (not transient rate limit). "out of extra usage" is the
# Anthropic OAuth Pro/Max overage bucket depleted (HTTP 400).
# The Nous gateway's own words for "the free tier will not serve this" — a billing wall for a
# named account, the tier refusing for an anonymous one (see ``_WELCOME_403_NAMED_PATTERNS``).
_FREE_TIER_REFUSAL_PATTERNS = ("model_not_supported_on_free_tier", "not available on the free tier")
_BILLING_PATTERNS = (
"insufficient credits", "insufficient_quota", "insufficient balance", "credit balance",
"credits exhausted", "credits have been exhausted", "requires available credits",
"account balance is too low", "no usable credits", "top up your credits", "payment required",
"billing hard limit", "exceeded your current quota", "account is deactivated", "plan does not include",
"out of extra usage", "out of funds", "run out of funds", "balance_depleted",
*_FREE_TIER_REFUSAL_PATTERNS,
# LiteLLM proxies word a hard cap as "hard billing limit" (structured twin:
# ``terminal_quota_exhausted`` in _BILLING_ERROR_CODES). "terminal billing
# limit" free text is NOT matched: substring rules can't negate the
# "non-terminal billing limit" wording, and the structured code covers it.
"hard billing limit",
"model_not_supported_on_free_tier", "not available on the free tier",
)
# Not proof of exhaustion: Anthropic returns the same "out of extra usage" body
@@ -122,12 +107,7 @@ _XAI_SPENDING_LIMIT_ERROR_CODE = "personal-team-blocked:spending-limit"
_BILLING_ERROR_CODES = frozenset({
"insufficient_quota", "billing_not_active", "payment_required", "insufficient_credits",
"no_usable_credits", "balance_depleted", "model_not_supported_on_free_tier",
"member_spend_cap_exceeded", "terminal_quota_exhausted", _XAI_SPENDING_LIMIT_ERROR_CODE,
# OpenAI (and OpenAI-compatible aggregators) spend/usage-limit family:
# a credit balance or an org/project spend or usage cap is exhausted —
# terminal for this credential until limits are raised.
"credit_balance_exhausted", "organization_spend_limit_exceeded",
"organization_usage_limit_exceeded", "project_spend_limit_exceeded",
"member_spend_cap_exceeded", _XAI_SPENDING_LIMIT_ERROR_CODE,
})
# Transient rate limiting. Bedrock "Throttling error: Too many tokens" also
@@ -193,16 +173,11 @@ _IMAGE_CORRUPT_PATTERNS = (
# 400s rejecting list-type ``content`` in tool messages (Xiaomi MiMo "text is
# not set", Alibaba, OpenAI-compat long tail). Recovery: strip image parts from
# tool messages, remember (provider, model), retry. (#27344)
# NVIDIA NIM's Rust gateway never names the field: its serde rejection says the
# body "did not match any variant of untagged enum
# ChatCompletionRequestToolMessageContent", which is the same list-type tool
# content that every other wording here describes (#111231).
_MULTIMODAL_TOOL_CONTENT_PATTERNS = (
"text is not set", "tool message content must be a string", "tool content must be a string",
"tool message must be a string", "expected string, got list", "expected string, got array",
# Console Go / pydantic-v2 relays behind opencode-go (422, param ``messages.N.tool.content.str``, #104731).
"tool_call.content must be string", "tool.content.str", "input should be a valid string",
"chatcompletionrequesttoolmessagecontent",
)
# Local-inference memory/resource-ceiling rejections (oMLX/MLX memory guard,
@@ -519,7 +494,6 @@ class _Ctx:
approx_tokens: int
context_length: int
num_messages: int
base_url: str = "" # the route the call went to; "" when the caller did not say
def __post_init__(self) -> None:
self.error_type = type(self.error).__name__
@@ -556,13 +530,6 @@ def _plugin_verdict(c: _Ctx) -> Optional[Verdict]:
return verdict
# A welcome-host 403 that spells one of these out is a safety block or a billing wall, not the
# tier refusing. The free-tier refusal phrases are left OUT: on the free route they mean exactly
# "the tier refused", and an anonymous session has no credits to check.
_WELCOME_403_NAMED_PATTERNS = _CONTENT_POLICY_BLOCKED_PATTERNS + tuple(
p for p in _BILLING_PATTERNS if p not in _FREE_TIER_REFUSAL_PATTERNS)
def _nous_welcome_tier(c: _Ctx) -> Optional[Verdict]:
"""The Nous inference gateway's welcome-tier (free tier) refusals, read from the structured body.
@@ -586,10 +553,7 @@ def _nous_welcome_tier(c: _Ctx) -> Optional[Verdict]:
if refusal["retry_after"] > 0:
ctx["reset_at"] = time.time() + refusal["retry_after"]
return _v(_R.rate_limit, should_fallback=True, error_context=ctx)
# The route-keyed dark-tier 403 applies only to a 403 that says nothing else: a safety refusal
# or a billing wall on the welcome host keeps its own classification (and its own recovery).
plain_403 = c.provider == "nous" and not any(p in c.msg for p in _WELCOME_403_NAMED_PATTERNS)
kind = welcome_route_refusal(status, c.msg, c.base_url if plain_403 else None)
kind = welcome_route_refusal(status, c.msg)
if kind is None:
return None
ctx = {"welcome_route": kind}
@@ -725,12 +689,8 @@ _STAGES: Sequence[Callable[[_Ctx], Optional[Verdict]]] = (
def classify_api_error(
error: Exception, *, provider: str = "", model: str = "",
approx_tokens: int = 0, context_length: int = 200000, num_messages: int = 0,
base_url: str = "",
) -> ClassifiedError:
"""Classify an API error into a structured recovery recommendation (see ``_STAGES``).
``base_url`` (optional) is the route the call went to; the Nous welcome tier keys its
dark-tier 403 on it because that refusal carries no distinguishing message."""
"""Classify an API error into a structured recovery recommendation (see ``_STAGES``)."""
status_code = _extract_status_code(error)
# Copilot/GitHub Models RateLimitError may not set .status_code; force 429.
if status_code is None and type(error).__name__ == "RateLimitError":
@@ -738,7 +698,7 @@ def classify_api_error(
body = _extract_error_body(error)
c = _Ctx(
error, status_code, body, _build_error_msg(error, body), provider, model,
approx_tokens, context_length, num_messages, str(base_url or ""),
approx_tokens, context_length, num_messages,
)
verdict = next((v for v in (stage(c) for stage in _STAGES) if v is not None), _V_UNKNOWN)
base = {"status_code": status_code, "provider": provider, "model": model, "message": _extract_message(error, body)}
@@ -765,11 +725,6 @@ def _status_404(c: _Ctx) -> Verdict:
def _status_429(c: _Ctx) -> Verdict:
# A structured billing code is decisive: LiteLLM stamps
# ``terminal_quota_exhausted`` (a hard cap, not throttling) on 429s, and
# this handler always returns, so _by_error_code never sees the code.
if c.code in _BILLING_ERROR_CODES:
return _V_BILLING
# Z.AI/Zhipu reuse 429 for server-wide overload: back off on the same
# key instead of burning the pool (#14038).
if any(p in c.msg for p in _OVERLOADED_PATTERNS):
@@ -820,12 +775,6 @@ def _classify_400(c: _Ctx) -> Verdict:
if code == "invalid_encrypted_content" or "invalid_encrypted_content" in msg or (
"encrypted content for item" in msg and "could not be verified" in msg
) or "could not decrypt the provided encrypted_content" in msg or (
# Custom Responses endpoints wrap a replay rejection in a generic bad_request (#95834).
"encrypted content could not be decrypted or parsed" in msg
) or (
# OpenCode Zen wraps this OpenAI replay rejection in ``invalid_request_error`` (#111309).
"encrypted_content" in msg and "was not issued to this caller" in msg
) or (
# Azure Foundry (gpt-6-astra) rejects replayed reasoning from several prior responses this way (#105369).
"conflicting authenticated continuation identities" in msg
):