"""A user-defined endpoint prices itself from its own upstream (sub2api's key-pricing API) — there is no canonical catalog to whitelist it into. """ import hermes_cli.inventory as inv import hermes_cli.models_pricing as mp def _stub_endpoint(monkeypatch, payload, *, calls=None, key="sk-test", reset=True): """Serve *payload* (or raise when it is an Exception) for any upstream GET.""" if reset: monkeypatch.setattr(mp, "_pricing_cache", {}) monkeypatch.setattr(mp, "_pricing_cache_retry_after", {}) monkeypatch.setattr(mp, "_custom_endpoint_api_key", lambda _slug: key) monkeypatch.setattr(mp, "get_pricing_for_provider", lambda *_a, **_kw: {}) def fake_get_json(url, headers, timeout, *_a, **_kw): if calls is not None: calls.append((url, headers.get("Authorization"))) if isinstance(payload, Exception): raise payload return payload monkeypatch.setattr(mp, "_get_json", fake_get_json) def _row(): return { "slug": "freemodel2api", "name": "FreeModel2API", "is_user_defined": True, "api_url": "http://124.232.163.75:3020/v1", "models": ["deepseek-flash", "glm-5.3", "no-price-model"], } def test_custom_endpoint_prices_reach_the_row_and_scale_by_rate_multiplier(monkeypatch): """Per-token upstream USD → the picker's $/Mtok columns, times the key's rate multiplier.""" calls = [] _stub_endpoint( monkeypatch, { "models": [ {"name": "deepseek-flash", "pricing": { "billing_mode": "token", "input_price": 1.5e-07, "output_price": 6e-07, "cache_read_price": 3e-09}}, # 2x group multiplier: every rate doubles. {"name": "glm-5.3", "pricing": { "billing_mode": "token", "input_price": 1e-06, "output_price": 2e-06}}, {"name": "no-price-model", "pricing": {"billing_mode": "unavailable", "intervals": []}}, ], "rate_multiplier": {"effective_rate_multiplier": 2}, }, calls=calls, ) rows = [_row()] inv._apply_pricing(rows) pricing = rows[0]["pricing"] assert pricing["deepseek-flash"] == {"input": "$0.30", "output": "$1.20", "cache": "$0.006", "free": False} assert pricing["glm-5.3"] == {"input": "$2.00", "output": "$4.00", "cache": None, "free": False} # "unavailable" carries no per-token rate — the model is simply absent, not zero-priced. assert "no-price-model" not in pricing assert calls == [("http://124.232.163.75:3020/v1/sub2api/pricing", "Bearer sk-test")] def test_non_sub2api_endpoint_goes_unpriced_without_breaking_the_row(monkeypatch): """An endpoint that is not a sub2api instance 404s / answers non-JSON; the row survives.""" _stub_endpoint(monkeypatch, ValueError("not JSON")) rows = [_row()] inv._apply_pricing(rows) assert "pricing" not in rows[0] assert rows[0]["models"] == ["deepseek-flash", "glm-5.3", "no-price-model"] def test_custom_endpoint_pricing_is_cached_only_on_the_picker_path(monkeypatch): """Picker opens read the cache; only the prewarm worker may start endpoint I/O.""" _stub_endpoint(monkeypatch, RuntimeError("network fetch started")) rows = [_row()] inv._apply_pricing(rows, cached_only=True) assert "pricing" not in rows[0] def test_warm_cache_answers_the_picker_path_without_io(monkeypatch): """The prewarm's result is what a later cached-only open renders.""" _stub_endpoint(monkeypatch, { "models": [{"name": "deepseek-flash", "pricing": { "billing_mode": "token", "input_price": 1.5e-07, "output_price": 6e-07}}], }) url = _row()["api_url"] assert mp.get_custom_endpoint_pricing("freemodel2api", url) # fills the cache _stub_endpoint(monkeypatch, RuntimeError("network fetch started"), reset=False) assert mp.get_custom_endpoint_pricing("freemodel2api", url, cached_only=True) rows = [_row()] inv._apply_pricing(rows, cached_only=True) assert rows[0]["pricing"]["deepseek-flash"]["input"] == "$0.15" def test_endpoint_probe_returns_formatted_prices_and_the_rate_multiplier(monkeypatch): """面板「测试」那一次:价格格式化成 $/Mtok(和选择器一致),倍率另给一个数 —— 价格里已经乘过倍率,但用户要看的就是这个数。""" _stub_endpoint(monkeypatch, { "models": [{"name": "deepseek-flash", "pricing": { "billing_mode": "token", "input_price": 1.5e-07, "output_price": 6e-07, "cache_read_price": 3e-09}}], "rate_multiplier": {"effective_rate_multiplier": 2}, }) monkeypatch.setattr(mp, "_RATE_MULTIPLIERS", {}) probe = mp.sub2api_endpoint_probe_pricing("http://h:3020/v1", "sk-test") assert probe["pricing"]["deepseek-flash"] == { "input": "$0.30", "output": "$1.20", "cache": "$0.006", "free": False} assert probe["rate_multiplier"] == 2.0 def test_endpoint_probe_on_a_non_sub2api_endpoint_is_empty_but_not_fatal(monkeypatch): _stub_endpoint(monkeypatch, ValueError("not JSON")) monkeypatch.setattr(mp, "_RATE_MULTIPLIERS", {}) probe = mp.sub2api_endpoint_probe_pricing("http://h:3020/v1", "sk-test") # 倍率 0 = 不知道,不是「倍率是 0」 assert probe == {"pricing": {}, "rate_multiplier": 0.0} def test_provider_form_prices_from_its_own_env_credentials(monkeypatch): """Same endpoint, provider form: FreeModel2API is a provider plugin, so its key and base URL come from its own env slots — no ``providers:`` entry, no custom endpoint row.""" from hermes_cli import auth calls = [] monkeypatch.setattr(mp, "_pricing_cache", {}) monkeypatch.setattr(mp, "_pricing_cache_retry_after", {}) monkeypatch.setattr( auth, "resolve_api_key_provider_credentials", lambda _pid: {"api_key": "sk-x", "base_url": "http://h:3020/v1"}) def fake_get_json(url, headers, timeout, *_a, **_kw): calls.append((url, headers.get("Authorization"))) return {"models": [{"name": "deepseek-flash", "pricing": { "billing_mode": "token", "input_price": 1.5e-07, "output_price": 6e-07}}]} monkeypatch.setattr(mp, "_get_json", fake_get_json) # Per-token USD, the shape _apply_pricing formats into the picker's $/Mtok columns. assert mp.get_pricing_for_provider("freemodel2api")["deepseek-flash"] == { "prompt": "1.5e-07", "completion": "6e-07"} assert calls == [("http://h:3020/v1/sub2api/pricing", "Bearer sk-x")] # The fetcher's remembered cache key must be the one cached_only reads back, else a picker open # (which never starts I/O) would show prices only on the second open. def boom(*_a, **_kw): raise RuntimeError("network fetch started") monkeypatch.setattr(mp, "_get_json", boom) assert mp.get_pricing_for_provider("freemodel2api", cached_only=True)["deepseek-flash"]["prompt"] == "1.5e-07"