diff --git a/agent/error_classifier.py b/agent/error_classifier.py index d28ddb0602..dd330863e3 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -108,6 +108,8 @@ _BILLING_PATTERNS = [ "credit balance", "credits exhausted", "credits have been exhausted", + "requires available credits", + "account balance is too low", "no usable credits", "top up your credits", "payment required", diff --git a/hermes_cli/main.py b/hermes_cli/main.py index ffb7ac93f8..abf48e320f 100644 --- a/hermes_cli/main.py +++ b/hermes_cli/main.py @@ -1078,6 +1078,67 @@ def _has_any_provider_configured() -> bool: return False +def _confirm_startup_expensive_model_override(args) -> None: + """Guard startup -m/--provider overrides before the first API call.""" + explicit_model = (getattr(args, "model", None) or "").strip() + explicit_provider = (getattr(args, "provider", None) or "").strip() + if not explicit_model and not explicit_provider: + return + + try: + from hermes_cli.config import load_config + from hermes_cli.model_cost_guard import expensive_model_warning + except Exception as exc: + logger.warning("startup model cost guard unavailable: %s", exc) + return + + try: + model_cfg = (load_config().get("model") or {}) + except Exception as exc: + logger.warning("startup model cost guard could not load config: %s", exc) + model_cfg = {} + if not isinstance(model_cfg, dict): + model_cfg = {} + + model = explicit_model or (model_cfg.get("default") or "").strip() + if not model: + return + provider = (explicit_provider or model_cfg.get("provider") or "").strip() + try: + warning = expensive_model_warning( + model, + provider=provider, + base_url=(model_cfg.get("base_url") or ""), + api_key=(model_cfg.get("api_key") or ""), + ) + except Exception as exc: + logger.warning("startup model cost guard failed for %s/%s: %s", provider, model, exc) + return + if warning is None: + return + + # Cost and provider-routing confirmation is intentionally independent of + # --yolo / --accept-hooks: those flags approve local command/tool risk, not + # paid aggregator spend or a surprising provider route. + message = warning.message + if not sys.stdin.isatty(): + sys.stderr.write(message + "\n") + sys.stderr.write( + "Refusing this startup model override in non-interactive mode. " + "Run interactively and confirm if you intend to use it.\n" + ) + raise SystemExit(1) + + sys.stderr.write(message + "\n") + try: + reply = input("Use this model for this invocation? [y/N] ").strip().lower() + except (EOFError, KeyboardInterrupt): + reply = "" + if reply not in {"y", "yes"}: + sys.stderr.write("Model override cancelled.\n") + raise SystemExit(1) + + def _session_browse_picker(sessions: list) -> Optional[str]: """Interactive curses-based session browser with live search filtering. @@ -2743,6 +2804,7 @@ def cmd_chat(args): os.environ["HERMES_SESSION_SOURCE"] = args.source _pin_kanban_board_env() + _confirm_startup_expensive_model_override(args) if use_tui: _launch_tui( @@ -11227,6 +11289,7 @@ def _try_termux_fast_cli_launch() -> bool: if getattr(args, "oneshot", None): _prepare_agent_startup(args) + _confirm_startup_expensive_model_override(args) _run_and_exit_oneshot( args.oneshot, model=getattr(args, "model", None), @@ -12921,6 +12984,7 @@ def main(): # Handle top-level --oneshot / -z: single-shot mode, stdout = final # response only, nothing else. Bypasses cli.py entirely. if getattr(args, "oneshot", None): + _confirm_startup_expensive_model_override(args) _run_and_exit_oneshot( args.oneshot, model=getattr(args, "model", None), diff --git a/hermes_cli/model_cost_guard.py b/hermes_cli/model_cost_guard.py index c3a327cde0..1ed8835b5b 100644 --- a/hermes_cli/model_cost_guard.py +++ b/hermes_cli/model_cost_guard.py @@ -147,13 +147,15 @@ def expensive_model_warning( output_cost = entry.output_cost_per_million source = entry.source + is_known_gpt55_pro_confusion = model.lower() == GPT55_PRO_OPENROUTER_ID + over_input = ( input_cost is not None and input_cost > INPUT_COST_WARNING_THRESHOLD ) over_output = ( output_cost is not None and output_cost > OUTPUT_COST_WARNING_THRESHOLD ) - if not over_input and not over_output: + if not over_input and not over_output and not is_known_gpt55_pro_confusion: return None lines = [ @@ -169,7 +171,7 @@ def expensive_model_warning( ] if source: lines.append(f"Pricing source: {source}.") - if model.lower() == GPT55_PRO_OPENROUTER_ID: + if is_known_gpt55_pro_confusion: lines.append(GPT55_SUGGESTION) lines.append("Confirm only if you intend to use this model.") diff --git a/tests/agent/test_error_classifier.py b/tests/agent/test_error_classifier.py index d2e5a19421..f27507ed9d 100644 --- a/tests/agent/test_error_classifier.py +++ b/tests/agent/test_error_classifier.py @@ -231,6 +231,38 @@ class TestClassifyApiError: assert result.retryable is False assert result.should_fallback is True + def test_404_requires_available_credits_is_billing(self): + e = MockAPIError( + "Not Found", + status_code=404, + body={ + "status": 404, + "message": ( + "Model 'openai/gpt-5.5-pro' requires available credits. " + "Your account balance is too low to use paid models — " + "add credits at https://portal.nousresearch.com or pick a free model." + ), + }, + ) + result = classify_api_error(e, provider="nous", model="openai/gpt-5.5-pro") + assert result.reason == FailoverReason.billing + assert result.retryable is False + assert result.should_fallback is True + + def test_wrapped_402_uses_nested_body_message(self): + inner = MockAPIError( + "inner", + status_code=402, + body={"error": {"message": "Usage limit reached, try again in 5 minutes"}}, + ) + outer = Exception("outer") + outer.__cause__ = inner + + result = classify_api_error(outer) + + assert result.reason == FailoverReason.rate_limit + assert result.retryable is True + assert result.message == "Usage limit reached, try again in 5 minutes" # ── Rate limit ── diff --git a/tests/hermes_cli/test_cli_startup_model_cost_guard.py b/tests/hermes_cli/test_cli_startup_model_cost_guard.py new file mode 100644 index 0000000000..341ad3aa73 --- /dev/null +++ b/tests/hermes_cli/test_cli_startup_model_cost_guard.py @@ -0,0 +1,209 @@ +from argparse import Namespace +import sys +import types + +import pytest + + +class _NonInteractiveStdin: + def isatty(self): + return False + + +def _chat_args(**overrides): + base = { + "continue_last": None, + "model": None, + "provider": None, + "resume": None, + "no_restore_cwd": False, + "toolsets": None, + "skills": None, + "tui": False, + "tui_dev": False, + "cli": True, + "verbose": None, + "quiet": True, + "query": "hello", + "image": None, + "worktree": False, + "checkpoints": False, + "pass_session_id": False, + "max_turns": None, + "ignore_rules": False, + "ignore_user_config": False, + "safe_mode": False, + "compact": False, + "source": None, + "yolo": False, + "accept_hooks": False, + } + base.update(overrides) + return Namespace(**base) + + +@pytest.fixture +def main_mod(monkeypatch): + import hermes_cli.main as mod + + monkeypatch.setattr(mod, "_has_any_provider_configured", lambda: True) + monkeypatch.setattr(mod, "_sync_bundled_skills_for_startup", lambda: None) + monkeypatch.setattr(mod, "_termux_should_prefetch_update_check", lambda: False) + monkeypatch.setattr(mod, "_pin_kanban_board_env", lambda: None) + monkeypatch.setattr(mod, "_resolve_session_by_name_or_id", lambda val: val) + monkeypatch.setattr(mod, "_oneshot_cleanup_done", False) + return mod + + +@pytest.fixture +def fake_cli(monkeypatch): + captured = {} + + def fake_cli_main(**kwargs): + captured.update(kwargs) + + monkeypatch.setitem(sys.modules, "cli", types.SimpleNamespace(main=fake_cli_main)) + return captured + + +@pytest.fixture +def codex_config(monkeypatch): + monkeypatch.setattr( + "hermes_cli.config.load_config", + lambda: { + "model": { + "provider": "openai-codex", + "default": "gpt-5.5", + "base_url": "https://chatgpt.com/backend-api/codex", + } + }, + ) + + +def test_cmd_chat_rejects_noninteractive_gpt55_pro_startup_override( + main_mod, fake_cli, codex_config, monkeypatch, capsys +): + monkeypatch.setattr(sys, "stdin", _NonInteractiveStdin()) + + with pytest.raises(SystemExit) as excinfo: + main_mod.cmd_chat(_chat_args(model="openai/gpt-5.5-pro")) + + assert excinfo.value.code == 1 + assert not fake_cli + err = capsys.readouterr().err + assert "EXPENSIVE MODEL WARNING" in err + assert "did you mean to select openai/gpt-5.5?" in err + assert "non-interactive" in err + + +def test_cmd_chat_rejects_noninteractive_gpt55_pro_even_with_yolo( + main_mod, fake_cli, codex_config, monkeypatch, capsys +): + monkeypatch.setattr(sys, "stdin", _NonInteractiveStdin()) + + with pytest.raises(SystemExit) as excinfo: + main_mod.cmd_chat(_chat_args(model="openai/gpt-5.5-pro", yolo=True)) + + assert excinfo.value.code == 1 + assert not fake_cli + assert "EXPENSIVE MODEL WARNING" in capsys.readouterr().err + + +def test_cmd_chat_allows_interactive_gpt55_pro_when_confirmed( + main_mod, fake_cli, codex_config, monkeypatch +): + monkeypatch.setattr(sys.stdin, "isatty", lambda: True) + monkeypatch.setattr("builtins.input", lambda _prompt: "yes") + + main_mod.cmd_chat(_chat_args(model="openai/gpt-5.5-pro")) + + assert fake_cli["model"] == "openai/gpt-5.5-pro" + + +def test_cmd_chat_cancels_interactive_gpt55_pro_when_not_confirmed( + main_mod, fake_cli, codex_config, monkeypatch, capsys +): + monkeypatch.setattr(sys.stdin, "isatty", lambda: True) + monkeypatch.setattr("builtins.input", lambda _prompt: "n") + + with pytest.raises(SystemExit) as excinfo: + main_mod.cmd_chat(_chat_args(model="openai/gpt-5.5-pro")) + + assert excinfo.value.code == 1 + assert not fake_cli + assert "Model override cancelled" in capsys.readouterr().err + + +def test_cmd_chat_cancels_interactive_gpt55_pro_on_eof( + main_mod, fake_cli, codex_config, monkeypatch, capsys +): + monkeypatch.setattr(sys.stdin, "isatty", lambda: True) + + def raise_eof(_prompt): + raise EOFError + + monkeypatch.setattr("builtins.input", raise_eof) + + with pytest.raises(SystemExit) as excinfo: + main_mod.cmd_chat(_chat_args(model="openai/gpt-5.5-pro")) + + assert excinfo.value.code == 1 + assert not fake_cli + assert "Model override cancelled" in capsys.readouterr().err + + +def test_cmd_chat_rejects_noninteractive_provider_only_override_when_default_is_expensive( + main_mod, fake_cli, monkeypatch, capsys +): + monkeypatch.setattr(sys, "stdin", _NonInteractiveStdin()) + monkeypatch.setattr( + "hermes_cli.config.load_config", + lambda: {"model": {"provider": "openai-codex", "default": "openai/gpt-5.5-pro"}}, + ) + + with pytest.raises(SystemExit) as excinfo: + main_mod.cmd_chat(_chat_args(model=None, provider="nous")) + + assert excinfo.value.code == 1 + assert not fake_cli + assert "EXPENSIVE MODEL WARNING" in capsys.readouterr().err + + +def test_cmd_chat_allows_noninteractive_safe_codex_startup_override( + main_mod, fake_cli, monkeypatch +): + monkeypatch.setattr(sys, "stdin", _NonInteractiveStdin()) + monkeypatch.setattr( + "hermes_cli.config.load_config", + lambda: {"model": {"provider": "openai-codex", "default": "gpt-5.5"}}, + ) + + main_mod.cmd_chat(_chat_args(model="gpt-5.5", provider="openai-codex")) + + assert fake_cli["model"] == "gpt-5.5" + assert fake_cli["provider"] == "openai-codex" + + +def test_top_level_oneshot_rejects_noninteractive_gpt55_pro_startup_override( + main_mod, codex_config, monkeypatch, capsys +): + monkeypatch.setattr(sys, "stdin", _NonInteractiveStdin()) + monkeypatch.setattr(sys, "argv", ["hermes", "-z", "hello", "-m", "openai/gpt-5.5-pro"]) + monkeypatch.setattr(main_mod, "_prepare_agent_startup", lambda _args: None) + + called = False + + def fake_run_and_exit_oneshot(*_args, **_kwargs): + nonlocal called + called = True + + monkeypatch.setattr(main_mod, "_run_and_exit_oneshot", fake_run_and_exit_oneshot) + + with pytest.raises(SystemExit) as excinfo: + main_mod.main() + + assert excinfo.value.code == 1 + assert called is False + err = capsys.readouterr().err + assert "EXPENSIVE MODEL WARNING" in err + assert "non-interactive" in err diff --git a/tests/hermes_cli/test_model_cost_guard.py b/tests/hermes_cli/test_model_cost_guard.py index acf20347b2..c346f2ef7f 100644 --- a/tests/hermes_cli/test_model_cost_guard.py +++ b/tests/hermes_cli/test_model_cost_guard.py @@ -122,6 +122,18 @@ def test_openai_gpt55_pro_adds_suggestion(monkeypatch): assert "did you mean to select openai/gpt-5.5?" in warning.message +def test_openai_gpt55_pro_warns_even_without_pricing(monkeypatch): + monkeypatch.setattr("agent.models_dev.get_model_info", lambda *_args, **_kwargs: None) + monkeypatch.setattr("agent.usage_pricing.get_pricing_entry", lambda *_args, **_kwargs: None) + + warning = expensive_model_warning("openai/gpt-5.5-pro", provider="openai-codex") + + assert warning is not None + assert warning.input_cost_per_million is None + assert warning.output_cost_per_million is None + assert "did you mean to select openai/gpt-5.5?" in warning.message + + def test_openai_gpt55_pro_warns_for_nous_portal_pricing(monkeypatch): monkeypatch.setattr("agent.models_dev.get_model_info", lambda *_args, **_kwargs: None) monkeypatch.setattr(