fix(desktop): distinguish provider quota exhaustion
This commit is contained in:
@@ -1338,6 +1338,26 @@ def _classify_by_status(
|
||||
should_fallback=True,
|
||||
error_context=ctx,
|
||||
)
|
||||
# Account/subscription usage exhaustion is a quota wall, not a
|
||||
# request-rate throttle. Anthropic returns this as 429, so the generic
|
||||
# branch below used to retry it and Desktop rendered a provider error
|
||||
# instead of the billing/quota recovery. Preserve periodic quotas when
|
||||
# the response supplies an explicit reset/retry signal.
|
||||
has_usage_limit = (
|
||||
error_code.lower() == "usage_limit_reached"
|
||||
or "usage limit" in error_msg
|
||||
or "usage_limit_reached" in error_msg
|
||||
)
|
||||
has_transient_signal = any(
|
||||
p in error_msg for p in _USAGE_LIMIT_TRANSIENT_SIGNALS
|
||||
)
|
||||
if has_usage_limit and not has_transient_signal:
|
||||
return result_fn(
|
||||
FailoverReason.billing,
|
||||
retryable=False,
|
||||
should_rotate_credential=True,
|
||||
should_fallback=True,
|
||||
)
|
||||
return result_fn(
|
||||
FailoverReason.rate_limit,
|
||||
retryable=True,
|
||||
|
||||
@@ -14,7 +14,7 @@ import { GlyphSpinner } from '@/components/ui/glyph-spinner'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { displayPath, pathLeaf } from '@/lib/display-path'
|
||||
import { Activity, AlertCircle, Clock, Command, FolderOpen, Globe, Hash, Loader2, Terminal } from '@/lib/icons'
|
||||
import type { RuntimeReadinessResult } from '@/lib/runtime-readiness'
|
||||
import { runtimeReadinessDisplay, type RuntimeReadinessResult } from '@/lib/runtime-readiness'
|
||||
import { contextBarLabel, LiveDuration, usageContextLabel } from '@/lib/statusbar'
|
||||
import { useStoreSelector } from '@/lib/use-session-slice'
|
||||
import { cn } from '@/lib/utils'
|
||||
@@ -286,13 +286,15 @@ export function useStatusbarItems({
|
||||
const gatewayConnecting = gatewayState === 'connecting'
|
||||
const inferenceReady = gatewayOpen && inferenceStatus?.ready === true
|
||||
const gatewayDegraded = gatewayOpen || gatewayConnecting
|
||||
const readinessDisplay = runtimeReadinessDisplay(inferenceStatus)
|
||||
|
||||
const gatewayDetail = gatewayOpen
|
||||
? inferenceStatus?.ready
|
||||
? copy.gatewayReady
|
||||
: inferenceStatus
|
||||
? copy.gatewayNeedsSetup
|
||||
: copy.gatewayChecking
|
||||
? {
|
||||
checking: copy.gatewayChecking,
|
||||
needs_setup: copy.gatewayNeedsSetup,
|
||||
ready: copy.gatewayReady,
|
||||
unavailable: copy.gatewayUnavailable
|
||||
}[readinessDisplay]
|
||||
: gatewayConnecting
|
||||
? copy.gatewayConnecting
|
||||
: copy.gatewayOffline
|
||||
|
||||
@@ -2255,6 +2255,7 @@ export const ar = defineLocale({
|
||||
gateway: 'البوابة',
|
||||
gatewayReady: 'البوابة جاهزة',
|
||||
gatewayNeedsSetup: 'البوابة تحتاج إعدادا',
|
||||
gatewayUnavailable: 'الاستدلال غير متاح',
|
||||
gatewayChecking: 'جار فحص البوابة',
|
||||
gatewayConnecting: 'جار اتصال البوابة',
|
||||
gatewayOffline: 'البوابة غير متصلة',
|
||||
|
||||
@@ -2855,6 +2855,7 @@ export const en: Translations = {
|
||||
gateway: 'Gateway',
|
||||
gatewayReady: 'ready',
|
||||
gatewayNeedsSetup: 'needs setup',
|
||||
gatewayUnavailable: 'inference unavailable',
|
||||
gatewayChecking: 'checking',
|
||||
gatewayConnecting: 'connecting',
|
||||
gatewayOffline: 'offline',
|
||||
|
||||
@@ -2523,6 +2523,7 @@ export const ja = defineLocale({
|
||||
gateway: 'ゲートウェイ',
|
||||
gatewayReady: '準備完了',
|
||||
gatewayNeedsSetup: '設定が必要',
|
||||
gatewayUnavailable: '推論を利用できません',
|
||||
gatewayChecking: '確認中',
|
||||
gatewayConnecting: '接続中',
|
||||
gatewayOffline: 'オフライン',
|
||||
|
||||
@@ -2421,6 +2421,7 @@ export interface Translations {
|
||||
gateway: string
|
||||
gatewayReady: string
|
||||
gatewayNeedsSetup: string
|
||||
gatewayUnavailable: string
|
||||
gatewayChecking: string
|
||||
gatewayConnecting: string
|
||||
gatewayOffline: string
|
||||
|
||||
@@ -2435,6 +2435,7 @@ export const zhHant = defineLocale({
|
||||
gateway: '閘道',
|
||||
gatewayReady: '就緒',
|
||||
gatewayNeedsSetup: '需要設定',
|
||||
gatewayUnavailable: '推論不可用',
|
||||
gatewayChecking: '檢查中',
|
||||
gatewayConnecting: '連線中',
|
||||
gatewayOffline: '離線',
|
||||
|
||||
@@ -3020,6 +3020,7 @@ export const zh: Translations = {
|
||||
gateway: '网关',
|
||||
gatewayReady: '就绪',
|
||||
gatewayNeedsSetup: '需要设置',
|
||||
gatewayUnavailable: '推理不可用',
|
||||
gatewayChecking: '检查中',
|
||||
gatewayConnecting: '连接中',
|
||||
gatewayOffline: '离线',
|
||||
|
||||
@@ -1,6 +1,11 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
|
||||
import { evaluateRuntimeReadiness, fetchRuntimeReadinessSignals, interpretRuntimeReadiness } from './runtime-readiness'
|
||||
import {
|
||||
evaluateRuntimeReadiness,
|
||||
fetchRuntimeReadinessSignals,
|
||||
interpretRuntimeReadiness,
|
||||
runtimeReadinessDisplay
|
||||
} from './runtime-readiness'
|
||||
|
||||
describe('interpretRuntimeReadiness', () => {
|
||||
it('prefers runtime_check when both signals exist', () => {
|
||||
@@ -109,3 +114,27 @@ describe('evaluateRuntimeReadiness', () => {
|
||||
expect(result.ready).toBe(true)
|
||||
})
|
||||
})
|
||||
|
||||
describe('runtimeReadinessDisplay', () => {
|
||||
it('does not call configured credentials setup when runtime resolution fails', () => {
|
||||
expect(
|
||||
runtimeReadinessDisplay({
|
||||
checksDisagree: true,
|
||||
ready: false,
|
||||
reason: 'Anthropic cannot serve the selected model.',
|
||||
source: 'runtime_check'
|
||||
})
|
||||
).toBe('unavailable')
|
||||
})
|
||||
|
||||
it('keeps needs-setup for an authoritative unconfigured result', () => {
|
||||
expect(
|
||||
runtimeReadinessDisplay({
|
||||
checksDisagree: false,
|
||||
ready: false,
|
||||
reason: 'No provider configured.',
|
||||
source: 'setup_status'
|
||||
})
|
||||
).toBe('needs_setup')
|
||||
})
|
||||
})
|
||||
|
||||
@@ -27,6 +27,8 @@ export interface RuntimeReadinessResult {
|
||||
source: 'fallback' | 'runtime_check' | 'setup_status'
|
||||
}
|
||||
|
||||
export type RuntimeReadinessDisplay = 'checking' | 'needs_setup' | 'ready' | 'unavailable'
|
||||
|
||||
export type RuntimeReadinessRequester = <T = unknown>(method: string, params?: Record<string, unknown>) => Promise<T>
|
||||
|
||||
const DEFAULT_NOT_READY_REASON = 'Add a provider credential before sending your first message.'
|
||||
@@ -142,6 +144,21 @@ export function interpretRuntimeReadiness(
|
||||
}
|
||||
}
|
||||
|
||||
export function runtimeReadinessDisplay(status: RuntimeReadinessResult | null): RuntimeReadinessDisplay {
|
||||
if (status === null) {
|
||||
return 'checking'
|
||||
}
|
||||
|
||||
if (status.ready) {
|
||||
return 'ready'
|
||||
}
|
||||
|
||||
// Credentials exist but runtime resolution failed. Calling that "needs
|
||||
// setup" sends users back through onboarding for provider/quota failures
|
||||
// that setup cannot repair; the reason tooltip carries the specific cause.
|
||||
return status.checksDisagree ? 'unavailable' : 'needs_setup'
|
||||
}
|
||||
|
||||
export async function evaluateRuntimeReadiness(
|
||||
requestGateway: RuntimeReadinessRequester,
|
||||
options: RuntimeReadinessOptions = {}
|
||||
|
||||
@@ -273,6 +273,35 @@ class TestClassifyApiError:
|
||||
assert result.reason == FailoverReason.rate_limit
|
||||
assert result.should_fallback is True
|
||||
|
||||
def test_anthropic_429_usage_limit_without_reset_is_billing(self):
|
||||
e = MockAPIError(
|
||||
"usage limit reached",
|
||||
status_code=429,
|
||||
body={
|
||||
"error": {
|
||||
"type": "usage_limit_reached",
|
||||
"message": "Your account has reached its usage limit.",
|
||||
}
|
||||
},
|
||||
)
|
||||
|
||||
result = classify_api_error(e, provider="anthropic", model="claude-opus-5")
|
||||
|
||||
assert result.reason == FailoverReason.billing
|
||||
assert result.retryable is False
|
||||
assert result.should_fallback is True
|
||||
|
||||
def test_anthropic_429_usage_limit_with_reset_stays_rate_limit(self):
|
||||
e = MockAPIError(
|
||||
"usage limit reached; resets at 2026-08-24T10:00:00Z",
|
||||
status_code=429,
|
||||
)
|
||||
|
||||
result = classify_api_error(e, provider="anthropic", model="claude-opus-5")
|
||||
|
||||
assert result.reason == FailoverReason.rate_limit
|
||||
assert result.retryable is True
|
||||
|
||||
def test_alibaba_rate_increased_too_quickly(self):
|
||||
"""Alibaba/DashScope returns a unique throttling message.
|
||||
|
||||
|
||||
@@ -179,6 +179,25 @@ def test_exception_with_status_code_routes_through_classifier():
|
||||
assert surface["code"] in ("rate_limit", "upstream_rate_limit")
|
||||
|
||||
|
||||
def test_anthropic_usage_limit_routes_to_billing_recovery():
|
||||
class FakeAPIError(Exception):
|
||||
status_code = 429
|
||||
|
||||
surface = build_error_surface_from_exception(
|
||||
FakeAPIError("usage limit reached"),
|
||||
provider="anthropic",
|
||||
model="claude-opus-5",
|
||||
)
|
||||
|
||||
assert surface == {
|
||||
"layer": LAYER_BILLING,
|
||||
"code": "billing",
|
||||
"retryable": False,
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-5",
|
||||
}
|
||||
|
||||
|
||||
def test_exception_auth_status_routes_to_auth_layer():
|
||||
class FakeAuthError(Exception):
|
||||
status_code = 401
|
||||
|
||||
Reference in New Issue
Block a user