fix(desktop): distinguish provider quota exhaustion

This commit is contained in:
fangliquanflq
2026-08-24 04:46:20 +08:00
committed by Teknium
parent 6b3a7af73d
commit c2090ba6b4
12 changed files with 129 additions and 7 deletions
+20
View File
@@ -1338,6 +1338,26 @@ def _classify_by_status(
should_fallback=True,
error_context=ctx,
)
# Account/subscription usage exhaustion is a quota wall, not a
# request-rate throttle. Anthropic returns this as 429, so the generic
# branch below used to retry it and Desktop rendered a provider error
# instead of the billing/quota recovery. Preserve periodic quotas when
# the response supplies an explicit reset/retry signal.
has_usage_limit = (
error_code.lower() == "usage_limit_reached"
or "usage limit" in error_msg
or "usage_limit_reached" in error_msg
)
has_transient_signal = any(
p in error_msg for p in _USAGE_LIMIT_TRANSIENT_SIGNALS
)
if has_usage_limit and not has_transient_signal:
return result_fn(
FailoverReason.billing,
retryable=False,
should_rotate_credential=True,
should_fallback=True,
)
return result_fn(
FailoverReason.rate_limit,
retryable=True,
@@ -14,7 +14,7 @@ import { GlyphSpinner } from '@/components/ui/glyph-spinner'
import { useI18n } from '@/i18n'
import { displayPath, pathLeaf } from '@/lib/display-path'
import { Activity, AlertCircle, Clock, Command, FolderOpen, Globe, Hash, Loader2, Terminal } from '@/lib/icons'
import type { RuntimeReadinessResult } from '@/lib/runtime-readiness'
import { runtimeReadinessDisplay, type RuntimeReadinessResult } from '@/lib/runtime-readiness'
import { contextBarLabel, LiveDuration, usageContextLabel } from '@/lib/statusbar'
import { useStoreSelector } from '@/lib/use-session-slice'
import { cn } from '@/lib/utils'
@@ -286,13 +286,15 @@ export function useStatusbarItems({
const gatewayConnecting = gatewayState === 'connecting'
const inferenceReady = gatewayOpen && inferenceStatus?.ready === true
const gatewayDegraded = gatewayOpen || gatewayConnecting
const readinessDisplay = runtimeReadinessDisplay(inferenceStatus)
const gatewayDetail = gatewayOpen
? inferenceStatus?.ready
? copy.gatewayReady
: inferenceStatus
? copy.gatewayNeedsSetup
: copy.gatewayChecking
? {
checking: copy.gatewayChecking,
needs_setup: copy.gatewayNeedsSetup,
ready: copy.gatewayReady,
unavailable: copy.gatewayUnavailable
}[readinessDisplay]
: gatewayConnecting
? copy.gatewayConnecting
: copy.gatewayOffline
+1
View File
@@ -2255,6 +2255,7 @@ export const ar = defineLocale({
gateway: 'البوابة',
gatewayReady: 'البوابة جاهزة',
gatewayNeedsSetup: 'البوابة تحتاج إعدادا',
gatewayUnavailable: 'الاستدلال غير متاح',
gatewayChecking: 'جار فحص البوابة',
gatewayConnecting: 'جار اتصال البوابة',
gatewayOffline: 'البوابة غير متصلة',
+1
View File
@@ -2855,6 +2855,7 @@ export const en: Translations = {
gateway: 'Gateway',
gatewayReady: 'ready',
gatewayNeedsSetup: 'needs setup',
gatewayUnavailable: 'inference unavailable',
gatewayChecking: 'checking',
gatewayConnecting: 'connecting',
gatewayOffline: 'offline',
+1
View File
@@ -2523,6 +2523,7 @@ export const ja = defineLocale({
gateway: 'ゲートウェイ',
gatewayReady: '準備完了',
gatewayNeedsSetup: '設定が必要',
gatewayUnavailable: '推論を利用できません',
gatewayChecking: '確認中',
gatewayConnecting: '接続中',
gatewayOffline: 'オフライン',
+1
View File
@@ -2421,6 +2421,7 @@ export interface Translations {
gateway: string
gatewayReady: string
gatewayNeedsSetup: string
gatewayUnavailable: string
gatewayChecking: string
gatewayConnecting: string
gatewayOffline: string
+1
View File
@@ -2435,6 +2435,7 @@ export const zhHant = defineLocale({
gateway: '閘道',
gatewayReady: '就緒',
gatewayNeedsSetup: '需要設定',
gatewayUnavailable: '推論不可用',
gatewayChecking: '檢查中',
gatewayConnecting: '連線中',
gatewayOffline: '離線',
+1
View File
@@ -3020,6 +3020,7 @@ export const zh: Translations = {
gateway: '网关',
gatewayReady: '就绪',
gatewayNeedsSetup: '需要设置',
gatewayUnavailable: '推理不可用',
gatewayChecking: '检查中',
gatewayConnecting: '连接中',
gatewayOffline: '离线',
+30 -1
View File
@@ -1,6 +1,11 @@
import { describe, expect, it } from 'vitest'
import { evaluateRuntimeReadiness, fetchRuntimeReadinessSignals, interpretRuntimeReadiness } from './runtime-readiness'
import {
evaluateRuntimeReadiness,
fetchRuntimeReadinessSignals,
interpretRuntimeReadiness,
runtimeReadinessDisplay
} from './runtime-readiness'
describe('interpretRuntimeReadiness', () => {
it('prefers runtime_check when both signals exist', () => {
@@ -109,3 +114,27 @@ describe('evaluateRuntimeReadiness', () => {
expect(result.ready).toBe(true)
})
})
describe('runtimeReadinessDisplay', () => {
it('does not call configured credentials setup when runtime resolution fails', () => {
expect(
runtimeReadinessDisplay({
checksDisagree: true,
ready: false,
reason: 'Anthropic cannot serve the selected model.',
source: 'runtime_check'
})
).toBe('unavailable')
})
it('keeps needs-setup for an authoritative unconfigured result', () => {
expect(
runtimeReadinessDisplay({
checksDisagree: false,
ready: false,
reason: 'No provider configured.',
source: 'setup_status'
})
).toBe('needs_setup')
})
})
+17
View File
@@ -27,6 +27,8 @@ export interface RuntimeReadinessResult {
source: 'fallback' | 'runtime_check' | 'setup_status'
}
export type RuntimeReadinessDisplay = 'checking' | 'needs_setup' | 'ready' | 'unavailable'
export type RuntimeReadinessRequester = <T = unknown>(method: string, params?: Record<string, unknown>) => Promise<T>
const DEFAULT_NOT_READY_REASON = 'Add a provider credential before sending your first message.'
@@ -142,6 +144,21 @@ export function interpretRuntimeReadiness(
}
}
export function runtimeReadinessDisplay(status: RuntimeReadinessResult | null): RuntimeReadinessDisplay {
if (status === null) {
return 'checking'
}
if (status.ready) {
return 'ready'
}
// Credentials exist but runtime resolution failed. Calling that "needs
// setup" sends users back through onboarding for provider/quota failures
// that setup cannot repair; the reason tooltip carries the specific cause.
return status.checksDisagree ? 'unavailable' : 'needs_setup'
}
export async function evaluateRuntimeReadiness(
requestGateway: RuntimeReadinessRequester,
options: RuntimeReadinessOptions = {}
+29
View File
@@ -273,6 +273,35 @@ class TestClassifyApiError:
assert result.reason == FailoverReason.rate_limit
assert result.should_fallback is True
def test_anthropic_429_usage_limit_without_reset_is_billing(self):
e = MockAPIError(
"usage limit reached",
status_code=429,
body={
"error": {
"type": "usage_limit_reached",
"message": "Your account has reached its usage limit.",
}
},
)
result = classify_api_error(e, provider="anthropic", model="claude-opus-5")
assert result.reason == FailoverReason.billing
assert result.retryable is False
assert result.should_fallback is True
def test_anthropic_429_usage_limit_with_reset_stays_rate_limit(self):
e = MockAPIError(
"usage limit reached; resets at 2026-08-24T10:00:00Z",
status_code=429,
)
result = classify_api_error(e, provider="anthropic", model="claude-opus-5")
assert result.reason == FailoverReason.rate_limit
assert result.retryable is True
def test_alibaba_rate_increased_too_quickly(self):
"""Alibaba/DashScope returns a unique throttling message.
+19
View File
@@ -179,6 +179,25 @@ def test_exception_with_status_code_routes_through_classifier():
assert surface["code"] in ("rate_limit", "upstream_rate_limit")
def test_anthropic_usage_limit_routes_to_billing_recovery():
class FakeAPIError(Exception):
status_code = 429
surface = build_error_surface_from_exception(
FakeAPIError("usage limit reached"),
provider="anthropic",
model="claude-opus-5",
)
assert surface == {
"layer": LAYER_BILLING,
"code": "billing",
"retryable": False,
"provider": "anthropic",
"model": "claude-opus-5",
}
def test_exception_auth_status_routes_to_auth_layer():
class FakeAuthError(Exception):
status_code = 401