feat(pricing): refresh Fireworks snapshot to 2026-07, cover full serverless catalog + cached picker pricing

- Refresh _OFFICIAL_DOCS_PRICING fireworks entries against current
  docs.fireworks.ai/serverless/pricing: qwen3p6-plus is gone (replaced
  by qwen3p7-plus); add glm-5p2/5p1, kimi-k2p7-code, deepseek-v4-flash,
  minimax-m3/m2p7, gpt-oss-120b/20b, and the routers/*-fast tiers with
  their distinct higher rates.
- Picker pricing via get_pricing_for_provider('fireworks'): pure dict
  transform over the shared models.dev in-memory/disk cache (1h TTL) +
  _pricing_cache memoization — no new network call on the picker path.
- Wire pricing display into the generic api-key-provider setup flow so
  Fireworks model pickers show $/M columns like OpenRouter/Nous do.
- Invariant tests: plugin fallback_models all priced, fast tiers price
  higher than standard, every row carries cache_read < input.
This commit is contained in:
Teknium
2026-07-16 00:58:48 -07:00
parent 365620ab28
commit fe5c0cb6c3
4 changed files with 274 additions and 13 deletions
+51
View File
@@ -1654,6 +1654,8 @@ def get_pricing_for_provider(provider: str, *, force_refresh: bool = False) -> d
return _fetch_novita_pricing(force_refresh=force_refresh)
if normalized == "deepinfra":
return _fetch_deepinfra_pricing(force_refresh=force_refresh)
if normalized == "fireworks":
return _fireworks_pricing_from_models_dev(force_refresh=force_refresh)
if normalized == "nous":
api_key, base_url = _resolve_nous_pricing_credentials()
if base_url:
@@ -1670,6 +1672,55 @@ def get_pricing_for_provider(provider: str, *, force_refresh: bool = False) -> d
return {}
def _fireworks_pricing_from_models_dev(
*,
force_refresh: bool = False,
) -> dict[str, dict[str, str]]:
"""Derive Fireworks picker pricing from the models.dev registry cache.
No dedicated network fetch: ``fetch_models_dev()`` already maintains an
in-memory + disk cache (1h TTL) that every picker surface shares, so this
is a pure dict transform on the picker path — no added latency and no
per-render network call. Results are additionally memoized in
``_pricing_cache`` so repeated menu renders within a process are free.
models.dev publishes Fireworks costs in USD per 1M tokens; the shared
pricing formatter expects per-token strings, so divide by 1M.
"""
cache_key = "models.dev/fireworks"
if not force_refresh and cache_key in _pricing_cache:
return _pricing_cache[cache_key]
result: dict[str, dict[str, str]] = {}
try:
from agent.models_dev import _get_provider_models
models = _get_provider_models("fireworks") or {}
for mid, entry in models.items():
if not isinstance(entry, dict):
continue
cost = entry.get("cost")
if not isinstance(cost, dict):
continue
inp = cost.get("input")
out = cost.get("output")
if inp is None and out is None:
continue
row: dict[str, str] = {
"prompt": str(float(inp or 0) / 1_000_000),
"completion": str(float(out or 0) / 1_000_000),
}
cache_read = cost.get("cache_read")
if cache_read:
row["input_cache_read"] = str(float(cache_read) / 1_000_000)
result[str(mid)] = row
except Exception:
result = {}
_pricing_cache[cache_key] = result
return result
def _fetch_novita_pricing(
timeout: float = 8.0,
*,