feat(pricing): refresh Fireworks snapshot to 2026-07, cover full serverless catalog + cached picker pricing
- Refresh _OFFICIAL_DOCS_PRICING fireworks entries against current
docs.fireworks.ai/serverless/pricing: qwen3p6-plus is gone (replaced
by qwen3p7-plus); add glm-5p2/5p1, kimi-k2p7-code, deepseek-v4-flash,
minimax-m3/m2p7, gpt-oss-120b/20b, and the routers/*-fast tiers with
their distinct higher rates.
- Picker pricing via get_pricing_for_provider('fireworks'): pure dict
transform over the shared models.dev in-memory/disk cache (1h TTL) +
_pricing_cache memoization — no new network call on the picker path.
- Wire pricing display into the generic api-key-provider setup flow so
Fireworks model pickers show $/M columns like OpenRouter/Nous do.
- Invariant tests: plugin fallback_models all priced, fast tiers price
higher than standard, every row carries cache_read < input.
This commit is contained in:
@@ -1654,6 +1654,8 @@ def get_pricing_for_provider(provider: str, *, force_refresh: bool = False) -> d
|
||||
return _fetch_novita_pricing(force_refresh=force_refresh)
|
||||
if normalized == "deepinfra":
|
||||
return _fetch_deepinfra_pricing(force_refresh=force_refresh)
|
||||
if normalized == "fireworks":
|
||||
return _fireworks_pricing_from_models_dev(force_refresh=force_refresh)
|
||||
if normalized == "nous":
|
||||
api_key, base_url = _resolve_nous_pricing_credentials()
|
||||
if base_url:
|
||||
@@ -1670,6 +1672,55 @@ def get_pricing_for_provider(provider: str, *, force_refresh: bool = False) -> d
|
||||
return {}
|
||||
|
||||
|
||||
def _fireworks_pricing_from_models_dev(
|
||||
*,
|
||||
force_refresh: bool = False,
|
||||
) -> dict[str, dict[str, str]]:
|
||||
"""Derive Fireworks picker pricing from the models.dev registry cache.
|
||||
|
||||
No dedicated network fetch: ``fetch_models_dev()`` already maintains an
|
||||
in-memory + disk cache (1h TTL) that every picker surface shares, so this
|
||||
is a pure dict transform on the picker path — no added latency and no
|
||||
per-render network call. Results are additionally memoized in
|
||||
``_pricing_cache`` so repeated menu renders within a process are free.
|
||||
|
||||
models.dev publishes Fireworks costs in USD per 1M tokens; the shared
|
||||
pricing formatter expects per-token strings, so divide by 1M.
|
||||
"""
|
||||
cache_key = "models.dev/fireworks"
|
||||
if not force_refresh and cache_key in _pricing_cache:
|
||||
return _pricing_cache[cache_key]
|
||||
|
||||
result: dict[str, dict[str, str]] = {}
|
||||
try:
|
||||
from agent.models_dev import _get_provider_models
|
||||
|
||||
models = _get_provider_models("fireworks") or {}
|
||||
for mid, entry in models.items():
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
cost = entry.get("cost")
|
||||
if not isinstance(cost, dict):
|
||||
continue
|
||||
inp = cost.get("input")
|
||||
out = cost.get("output")
|
||||
if inp is None and out is None:
|
||||
continue
|
||||
row: dict[str, str] = {
|
||||
"prompt": str(float(inp or 0) / 1_000_000),
|
||||
"completion": str(float(out or 0) / 1_000_000),
|
||||
}
|
||||
cache_read = cost.get("cache_read")
|
||||
if cache_read:
|
||||
row["input_cache_read"] = str(float(cache_read) / 1_000_000)
|
||||
result[str(mid)] = row
|
||||
except Exception:
|
||||
result = {}
|
||||
|
||||
_pricing_cache[cache_key] = result
|
||||
return result
|
||||
|
||||
|
||||
def _fetch_novita_pricing(
|
||||
timeout: float = 8.0,
|
||||
*,
|
||||
|
||||
Reference in New Issue
Block a user