From c64a1fbd6caceefca811507cd4abed12ae6d2eca Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:57:37 -0700 Subject: [PATCH] refactor(hermes_cli): repack literal tables, one-line trivial branches, share _drop_authorization --- hermes_cli/models_catalog_static.py | 171 ++++++++++++---------------- hermes_cli/models_local.py | 25 +--- hermes_cli/models_pricing.py | 13 +-- hermes_cli/models_validate.py | 95 +++++----------- 4 files changed, 115 insertions(+), 189 deletions(-) diff --git a/hermes_cli/models_catalog_static.py b/hermes_cli/models_catalog_static.py index c1b3f118e5..9e001d7709 100644 --- a/hermes_cli/models_catalog_static.py +++ b/hermes_cli/models_catalog_static.py @@ -26,18 +26,16 @@ OPENROUTER_MODELS: list[tuple[str, str]] = [ for mid in ( "anthropic/claude-fable-5.1", "anthropic/claude-fable-5", "anthropic/claude-opus-5", "anthropic/claude-opus-5-fast", "anthropic/claude-opus-4.8", "anthropic/claude-opus-4.8-fast", - "anthropic/claude-sonnet-5", "anthropic/claude-haiku-4.5", "openai/gpt-5.6-sol", - "openai/gpt-5.6-sol-pro", "openai/gpt-5.6-terra", "openai/gpt-5.6-terra-pro", "openai/gpt-5.6-luna", - "openai/gpt-5.6-luna-pro", "openai/gpt-5.5", "openai/gpt-5.5-pro", "openai/gpt-5.4-mini", - "google/gemini-3.1-pro-preview", "google/gemini-3.8-flash", "google/gemini-3.7-flash", - "x-ai/grok-4.6", "deepseek/deepseek-v4-pro", "deepseek/deepseek-v4-pro-0813", - "deepseek/deepseek-v4-flash", "deepseek/deepseek-v4-flash-0731", "qwen/qwen3.8-max", - "qwen/qwen3.8-flash", "moonshotai/kimi-k3", "minimax/minimax-m3", "z-ai/glm-5.3", + "anthropic/claude-sonnet-5", "anthropic/claude-haiku-4.5", "openai/gpt-5.6-sol", "openai/gpt-5.6-sol-pro", + "openai/gpt-5.6-terra", "openai/gpt-5.6-terra-pro", "openai/gpt-5.6-luna", "openai/gpt-5.6-luna-pro", + "openai/gpt-5.5", "openai/gpt-5.5-pro", "openai/gpt-5.4-mini", "google/gemini-3.1-pro-preview", + "google/gemini-3.8-flash", "google/gemini-3.7-flash", "x-ai/grok-4.6", "deepseek/deepseek-v4-pro", + "deepseek/deepseek-v4-pro-0813", "deepseek/deepseek-v4-flash", "deepseek/deepseek-v4-flash-0731", + "qwen/qwen3.8-max", "qwen/qwen3.8-flash", "moonshotai/kimi-k3", "minimax/minimax-m3", "z-ai/glm-5.3", "z-ai/glm-5.3-flash", "z-ai/glm-5.2", "xiaomi/mimo-v2.5-pro", "tencent/hy4-preview", "tencent/hy3", - "stepfun/step-3.7-flash", "nvidia/nemotron-3-super-120b-a12b", "meta/muse-spark-1.2", - "sakana/fugu-ultra", "openrouter/pareto-code", "thinkingmachines/inkling:free", - "thinkingmachines/inkling-small:free", "minimax/minimax-m3:free", "z-ai/glm-5.2:free", - "poolside/laguna-s-2.1:free", "poolside/laguna-xs-2.1:free", + "stepfun/step-3.7-flash", "nvidia/nemotron-3-super-120b-a12b", "meta/muse-spark-1.2", "sakana/fugu-ultra", + "openrouter/pareto-code", "thinkingmachines/inkling:free", "thinkingmachines/inkling-small:free", + "minimax/minimax-m3:free", "z-ai/glm-5.2:free", "poolside/laguna-s-2.1:free", "poolside/laguna-xs-2.1:free", "nvidia/nemotron-3-super-120b-a12b:free", "nvidia/nemotron-3-ultra-550b-a55b:free", "nvidia/nemotron-3.5-lightning:free", ) @@ -45,8 +43,7 @@ OPENROUTER_MODELS: list[tuple[str, str]] = [ # OpenRouter entries the Nous Portal does not carry (routing/fast variants, free tier). _OPENROUTER_ONLY = { - "anthropic/claude-opus-5-fast", "anthropic/claude-opus-4.8-fast", "meta/muse-spark-1.2", - "openrouter/pareto-code", + "anthropic/claude-opus-5-fast", "anthropic/claude-opus-4.8-fast", "meta/muse-spark-1.2", "openrouter/pareto-code", } @@ -124,8 +121,7 @@ def _xai_curated_models() -> list[str]: # Native OpenAI Chat Completions (api.openai.com); also the head of the Copilot list. _OPENAI_CHAT_MODELS = [ - "gpt-5.4", "gpt-5.4-mini", "gpt-5-mini", "gpt-5.3-codex", "gpt-5.2-codex", "gpt-4.1", "gpt-4o", - "gpt-4o-mini", + "gpt-5.4", "gpt-5.4-mini", "gpt-5-mini", "gpt-5.3-codex", "gpt-5.2-codex", "gpt-4.1", "gpt-4o", "gpt-4o-mini", ] _MINIMAX_MODELS = ["MiniMax-M3", "MiniMax-M2.7", "MiniMax-M2.5", "MiniMax-M2.1", "MiniMax-M2"] _TENCENT_MODELS = ["hy4-preview", "hy3", "hy3-preview"] @@ -143,9 +139,8 @@ _ALIBABA_CODING_PLAN_MODELS = [ ] # Verified against a live Token Plan subscription (key tier ``sk-sp-...``). _ALIBABA_TOKEN_PLAN_MODELS = [ - "qwen3.8-max-preview", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash", - "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v3.2", "kimi-k2.7-code", "kimi-k2.6", - "kimi-k2.5", "glm-5.2", "glm-5.1", "glm-5", + "qwen3.8-max-preview", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash", "deepseek-v4-pro", + "deepseek-v4-flash", "deepseek-v3.2", "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "glm-5.2", "glm-5.1", "glm-5", ] _XAI_MODELS = _xai_curated_models() @@ -164,13 +159,11 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "xai-oauth": list(_XAI_MODELS), "copilot-acp": ["copilot-acp"], "copilot": _OPENAI_CHAT_MODELS + [ - "claude-sonnet-4.6", "claude-sonnet-5", "claude-sonnet-4", "claude-sonnet-4.5", - "claude-haiku-4.5", "gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3-flash-preview", - "gemini-2.5-pro", + "claude-sonnet-4.6", "claude-sonnet-5", "claude-sonnet-4", "claude-sonnet-4.5", "claude-haiku-4.5", + "gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3-flash-preview", "gemini-2.5-pro", ], "gemini": [ - "gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3.6-flash", - "gemini-3.1-flash-lite-preview", + "gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3.6-flash", "gemini-3.1-flash-lite-preview", ], "zai": [ "glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5", "glm-5v-turbo", "glm-5-turbo", @@ -184,9 +177,8 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "z-ai/glm-5.3", "z-ai/glm-5.2", "moonshotai/kimi-k2.6", "minimaxai/minimax-m3", ], "kimi-coding": [ - "kimi-k3", "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "kimi-for-coding", - "kimi-for-coding-highspeed", "kimi-k2-thinking", "kimi-k2-thinking-turbo", - "kimi-k2-turbo-preview", "kimi-k2-0905-preview", + "kimi-k3", "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "kimi-for-coding", "kimi-for-coding-highspeed", + "kimi-k2-thinking", "kimi-k2-thinking-turbo", "kimi-k2-turbo-preview", "kimi-k2-0905-preview", ], "kimi-coding-cn": [ "kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed", "kimi-k2.6", "kimi-k2.5", @@ -194,8 +186,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { ], "stepfun": ["step-3.5-flash", "step-3.5-flash-2603"], "moonshot": [ - "kimi-k3", "kimi-k2.6", "kimi-k2.5", "kimi-k2-thinking", "kimi-k2-turbo-preview", - "kimi-k2-0905-preview", + "kimi-k3", "kimi-k2.6", "kimi-k2.5", "kimi-k2-thinking", "kimi-k2-turbo-preview", "kimi-k2-0905-preview", ], "minimax": list(_MINIMAX_MODELS), "minimax-oauth": ["MiniMax-M3", "MiniMax-M2.7", "MiniMax-M2.7-highspeed"], @@ -219,19 +210,18 @@ _PROVIDER_MODELS: dict[str, list[str]] = { # _LIVE_FIRST_PICKER_PROVIDERS, so this is a discovery floor: live entries lead in the picker # and stale curated names never pollute the top. "x-preview-f-free" = "Ox Alpha" stealth model. "opencode-zen": [ - "x-preview-f-free", "kimi-k3", "kimi-k2.5", "kimi-k2.6", "gpt-5.6-sol", "gpt-5.6-terra", - "gpt-5.6-luna", "gpt-5.5", "gpt-5.5-pro", "gpt-5.4-pro", "gpt-5.4", "gpt-5.4-mini", - "gpt-5.4-nano", "gpt-5.3-codex", "gpt-5.3-codex-spark", "gpt-5.2", "gpt-5.2-codex", "gpt-5.1", - "gpt-5.1-codex", "gpt-5.1-codex-max", "gpt-5.1-codex-mini", "gpt-5", "gpt-5-codex", - "gpt-5-nano", "claude-fable-5", "claude-opus-5", "claude-sonnet-5", "claude-opus-4-8", - "claude-opus-4-7", "claude-opus-4-6", "claude-opus-4-5", "claude-sonnet-4-6", - "claude-sonnet-4-5", "claude-sonnet-4", "claude-haiku-4-5", "gemini-3.7-flash", - "gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro", - "gemini-3-flash", "grok-4.6", "grok-4.5", "grok-build-0.1", "muse-spark-1.2", "minimax-m3", - "minimax-m2.7", "minimax-m2.5", "glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5", - "kimi-k2.7-code", "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-free", - "qwen3.6-plus", "qwen3.5-plus", "big-pickle", "mimo-v2.5-free", "hy3-free", "laguna-s-2.1-free", - "nemotron-3-ultra-free", "nemotron-3.5-lightning-free", "muse-spark-1.2-contributor-free", + "x-preview-f-free", "kimi-k3", "kimi-k2.5", "kimi-k2.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna", + "gpt-5.5", "gpt-5.5-pro", "gpt-5.4-pro", "gpt-5.4", "gpt-5.4-mini", "gpt-5.4-nano", "gpt-5.3-codex", + "gpt-5.3-codex-spark", "gpt-5.2", "gpt-5.2-codex", "gpt-5.1", "gpt-5.1-codex", "gpt-5.1-codex-max", + "gpt-5.1-codex-mini", "gpt-5", "gpt-5-codex", "gpt-5-nano", "claude-fable-5", "claude-opus-5", + "claude-sonnet-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", "claude-opus-4-5", + "claude-sonnet-4-6", "claude-sonnet-4-5", "claude-sonnet-4", "claude-haiku-4-5", "gemini-3.7-flash", + "gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro", "gemini-3-flash", + "grok-4.6", "grok-4.5", "grok-build-0.1", "muse-spark-1.2", "minimax-m3", "minimax-m2.7", "minimax-m2.5", + "glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5", "kimi-k2.7-code", "deepseek-v4-pro", + "deepseek-v4-flash", "deepseek-v4-flash-free", "qwen3.6-plus", "qwen3.5-plus", "big-pickle", "mimo-v2.5-free", + "hy3-free", "laguna-s-2.1-free", "nemotron-3-ultra-free", "nemotron-3.5-lightning-free", + "muse-spark-1.2-contributor-free", ], # OpenCode keyless free tier — OFFLINE FLOOR only. provider_model_ids("opencode-free") # revalidates live against GET /zen/v1/models and filters to the anonymous tier, so this list @@ -269,11 +259,10 @@ _PROVIDER_MODELS: dict[str, list[str]] = { # Static fallback when live discovery (ListFoundationModels + ListInferenceProfiles) is # unavailable. Inference-profile IDs (us.*) because most models require them. "bedrock": [ - "us.anthropic.claude-sonnet-5", "us.anthropic.claude-sonnet-4-6", - "us.anthropic.claude-opus-4-6-v1", "us.anthropic.claude-haiku-4-5-20251001-v1:0", - "us.anthropic.claude-sonnet-4-5-20250929-v1:0", "openai.gpt-5.5", "openai.gpt-5.6-sol", - "openai.gpt-5.6-terra", "openai.gpt-5.6-luna", "us.amazon.nova-pro-v1:0", - "us.amazon.nova-lite-v1:0", "us.amazon.nova-micro-v1:0", "deepseek.v3.2", + "us.anthropic.claude-sonnet-5", "us.anthropic.claude-sonnet-4-6", "us.anthropic.claude-opus-4-6-v1", + "us.anthropic.claude-haiku-4-5-20251001-v1:0", "us.anthropic.claude-sonnet-4-5-20250929-v1:0", + "openai.gpt-5.5", "openai.gpt-5.6-sol", "openai.gpt-5.6-terra", "openai.gpt-5.6-luna", + "us.amazon.nova-pro-v1:0", "us.amazon.nova-lite-v1:0", "us.amazon.nova-micro-v1:0", "deepseek.v3.2", "us.meta.llama4-maverick-17b-instruct-v1:0", "us.meta.llama4-scout-17b-instruct-v1:0", ], # Azure Foundry models depend on the user's endpoint configuration. @@ -329,8 +318,7 @@ CANONICAL_PROVIDERS: list[ProviderEntry] = [ProviderEntry(*row) for row in ( ("huggingface", "Hugging Face", "Hugging Face Inference Providers"), ("gemini", "Google AI Studio", "Google AI Studio (Native Gemini API)"), ("vertex", "Google Vertex AI", "Google Vertex AI (Gemini via GCP; OAuth2 service account or ADC, GCP billing/quotas)"), - ("deepseek", "DeepSeek", "DeepSeek (V3, R1, coder, direct API)"), - ("xai", "xAI", "xAI Grok (Direct API)"), + ("deepseek", "DeepSeek", "DeepSeek (V3, R1, coder, direct API)"), ("xai", "xAI", "xAI Grok (Direct API)"), ("zai", "Z.AI / GLM", "Z.AI / GLM (Zhipu direct API)"), ("kimi-coding", "Kimi / Kimi Coding Plan", "Kimi Coding Plan (api.kimi.com & Moonshot API)"), ("kimi-coding-cn", "Kimi / Moonshot (China)", "Kimi / Moonshot China (Domestic direct API)"), @@ -446,34 +434,31 @@ _PROVIDER_ALIASES = dict(( ("glm", "zai"), ("z-ai", "zai"), ("z.ai", "zai"), ("zhipu", "zai"), ("github", "copilot"), ("github-copilot", "copilot"), ("github-models", "copilot"), ("github-model", "copilot"), ("github-copilot-acp", "copilot-acp"), ("copilot-acp-agent", "copilot-acp"), ("google", "gemini"), - ("google-gemini", "gemini"), ("google-ai-studio", "gemini"), ("google-vertex", "vertex"), - ("vertex-ai", "vertex"), ("gcp-vertex", "vertex"), ("vertexai", "vertex"), ("kimi", "kimi-coding"), - ("moonshot", "kimi-coding"), ("kimi-cn", "kimi-coding-cn"), ("moonshot-cn", "kimi-coding-cn"), - ("step", "stepfun"), ("stepfun-coding-plan", "stepfun"), ("arcee-ai", "arcee"), - ("arceeai", "arcee"), ("gmi-cloud", "gmi"), ("gmicloud", "gmi"), ("fireworks-ai", "fireworks"), - ("fw", "fireworks"), ("actual-computer", "actual"), ("actualcomputer", "actual"), ("aci", "actual"), - ("nebius", "nebius-token-factory"), ("nebius-tokenfactory", "nebius-token-factory"), - ("nebius-tf", "nebius-token-factory"), ("token-factory", "nebius-token-factory"), - ("tokenfactory", "nebius-token-factory"), ("minimax-china", "minimax-cn"), - ("minimax_cn", "minimax-cn"), ("minimax-portal", "minimax-oauth"), + ("google-gemini", "gemini"), ("google-ai-studio", "gemini"), ("google-vertex", "vertex"), ("vertex-ai", "vertex"), + ("gcp-vertex", "vertex"), ("vertexai", "vertex"), ("kimi", "kimi-coding"), ("moonshot", "kimi-coding"), + ("kimi-cn", "kimi-coding-cn"), ("moonshot-cn", "kimi-coding-cn"), ("step", "stepfun"), + ("stepfun-coding-plan", "stepfun"), ("arcee-ai", "arcee"), ("arceeai", "arcee"), ("gmi-cloud", "gmi"), + ("gmicloud", "gmi"), ("fireworks-ai", "fireworks"), ("fw", "fireworks"), ("actual-computer", "actual"), + ("actualcomputer", "actual"), ("aci", "actual"), ("nebius", "nebius-token-factory"), + ("nebius-tokenfactory", "nebius-token-factory"), ("nebius-tf", "nebius-token-factory"), + ("token-factory", "nebius-token-factory"), ("tokenfactory", "nebius-token-factory"), + ("minimax-china", "minimax-cn"), ("minimax_cn", "minimax-cn"), ("minimax-portal", "minimax-oauth"), ("minimax-global", "minimax-oauth"), ("minimax_oauth", "minimax-oauth"), ("claude", "anthropic"), - ("claude-code", "anthropic"), ("deep-seek", "deepseek"), ("opencode", "opencode-zen"), - ("zen", "opencode-zen"), ("go", "opencode-go"), ("opencode-go-sub", "opencode-go"), - ("free", "opencode-free"), ("opencode_free", "opencode-free"), ("aigateway", "ai-gateway"), - ("vercel", "ai-gateway"), ("vercel-ai-gateway", "ai-gateway"), ("kilo", "kilocode"), - ("kilo-code", "kilocode"), ("kilo-gateway", "kilocode"), ("dashscope", "alibaba"), - ("aliyun", "alibaba"), ("qwen", "alibaba"), ("alibaba-cloud", "alibaba"), - ("qwen-portal", "qwen-oauth"), ("hf", "huggingface"), ("hugging-face", "huggingface"), - ("huggingface-hub", "huggingface"), ("novita-ai", "novita"), ("novitaai", "novita"), - ("mimo", "xiaomi"), ("xiaomi-mimo", "xiaomi"), ("tencent", "tencent-tokenhub"), - ("tokenhub", "tencent-tokenhub"), ("tencent-cloud", "tencent-tokenhub"), - ("tencentmaas", "tencent-tokenhub"), ("tokenplan", "tencent-tokenplan"), - ("tencent-lkeap", "tencent-tokenplan"), ("aws", "bedrock"), ("aws-bedrock", "bedrock"), - ("amazon-bedrock", "bedrock"), ("amazon", "bedrock"), ("grok", "xai"), ("grok-oauth", "xai-oauth"), - ("xai-oauth", "xai-oauth"), ("x-ai-oauth", "xai-oauth"), ("xai-grok-oauth", "xai-oauth"), - ("x-ai", "xai"), ("x.ai", "xai"), ("nim", "nvidia"), ("nvidia-nim", "nvidia"), - ("build-nvidia", "nvidia"), ("nemotron", "nvidia"), ("lmstudio", "lmstudio"), - ("lm-studio", "lmstudio"), ("lm_studio", "lmstudio"), + ("claude-code", "anthropic"), ("deep-seek", "deepseek"), ("opencode", "opencode-zen"), ("zen", "opencode-zen"), + ("go", "opencode-go"), ("opencode-go-sub", "opencode-go"), ("free", "opencode-free"), + ("opencode_free", "opencode-free"), ("aigateway", "ai-gateway"), ("vercel", "ai-gateway"), + ("vercel-ai-gateway", "ai-gateway"), ("kilo", "kilocode"), ("kilo-code", "kilocode"), + ("kilo-gateway", "kilocode"), ("dashscope", "alibaba"), ("aliyun", "alibaba"), ("qwen", "alibaba"), + ("alibaba-cloud", "alibaba"), ("qwen-portal", "qwen-oauth"), ("hf", "huggingface"), + ("hugging-face", "huggingface"), ("huggingface-hub", "huggingface"), ("novita-ai", "novita"), + ("novitaai", "novita"), ("mimo", "xiaomi"), ("xiaomi-mimo", "xiaomi"), ("tencent", "tencent-tokenhub"), + ("tokenhub", "tencent-tokenhub"), ("tencent-cloud", "tencent-tokenhub"), ("tencentmaas", "tencent-tokenhub"), + ("tokenplan", "tencent-tokenplan"), ("tencent-lkeap", "tencent-tokenplan"), ("aws", "bedrock"), + ("aws-bedrock", "bedrock"), ("amazon-bedrock", "bedrock"), ("amazon", "bedrock"), ("grok", "xai"), + ("grok-oauth", "xai-oauth"), ("xai-oauth", "xai-oauth"), ("x-ai-oauth", "xai-oauth"), + ("xai-grok-oauth", "xai-oauth"), ("x-ai", "xai"), ("x.ai", "xai"), ("nim", "nvidia"), ("nvidia-nim", "nvidia"), + ("build-nvidia", "nvidia"), ("nemotron", "nvidia"), ("lmstudio", "lmstudio"), ("lm-studio", "lmstudio"), + ("lm_studio", "lmstudio"), ("ollama", "custom"), # bare "ollama" = local; use "ollama-cloud" for cloud ("ollama_cloud", "ollama-cloud"), )) @@ -545,9 +530,8 @@ _OPENAI_FAST_MODE_PREFIXES: tuple[str, ...] = ("gpt-", "o1", "o3", "o4") # /models are the subscription-tier source of truth), and providers with dedicated live-endpoint # branches (copilot, anthropic, ai-gateway, ollama-cloud, custom, stepfun, openai-codex). _MODELS_DEV_PREFERRED: frozenset[str] = frozenset({ - "opencode-go", "opencode-zen", "deepseek", "kilocode", "fireworks", "mistral", "togetherai", - "cohere", "perplexity", "groq", "nvidia", "huggingface", "zai", "gemini", "google", "xai", - "xai-oauth", + "opencode-go", "opencode-zen", "deepseek", "kilocode", "fireworks", "mistral", "togetherai", "cohere", + "perplexity", "groq", "nvidia", "huggingface", "zai", "gemini", "google", "xai", "xai-oauth", }) @@ -561,24 +545,19 @@ _KEYLESS_STABLE_CACHE_PROVIDERS = frozenset({"opencode-free"}) # Claude IDs use hyphens (Anthropic native) but Copilot's API only accepts dot-notation, so a # copilot + hyphenated default would otherwise hit HTTP 400 "model_not_supported". _COPILOT_MODEL_ALIASES = dict(( - ("openai/gpt-5", "gpt-5-mini"), ("openai/gpt-5-chat", "gpt-5-mini"), - ("openai/gpt-5-mini", "gpt-5-mini"), ("openai/gpt-5-nano", "gpt-5-mini"), - ("openai/gpt-4.1", "gpt-4.1"), ("openai/gpt-4.1-mini", "gpt-4.1"), - ("openai/gpt-4.1-nano", "gpt-4.1"), ("openai/gpt-4o", "gpt-4o"), - ("openai/gpt-4o-mini", "gpt-4o-mini"), ("openai/o1", "gpt-5.2"), ("openai/o1-mini", "gpt-5-mini"), - ("openai/o1-preview", "gpt-5.2"), ("openai/o3", "gpt-5.3-codex"), ("openai/o3-mini", "gpt-5-mini"), - ("openai/o4-mini", "gpt-5-mini"), ("anthropic/claude-opus-4.6", "claude-opus-4.6"), - ("anthropic/claude-sonnet-5", "claude-sonnet-5"), - ("anthropic/claude-sonnet-4.6", "claude-sonnet-4.6"), - ("anthropic/claude-sonnet-4", "claude-sonnet-4"), - ("anthropic/claude-sonnet-4.5", "claude-sonnet-4.5"), - ("anthropic/claude-haiku-4.5", "claude-haiku-4.5"), ("claude-sonnet-5", "claude-sonnet-5"), - ("claude-opus-4-6", "claude-opus-4.6"), ("claude-sonnet-4-6", "claude-sonnet-4.6"), - ("claude-sonnet-4-0", "claude-sonnet-4"), ("claude-sonnet-4-5", "claude-sonnet-4.5"), - ("claude-haiku-4-5", "claude-haiku-4.5"), ("anthropic/claude-opus-4-6", "claude-opus-4.6"), - ("anthropic/claude-sonnet-4-6", "claude-sonnet-4.6"), - ("anthropic/claude-sonnet-4-0", "claude-sonnet-4"), - ("anthropic/claude-sonnet-4-5", "claude-sonnet-4.5"), + ("openai/gpt-5", "gpt-5-mini"), ("openai/gpt-5-chat", "gpt-5-mini"), ("openai/gpt-5-mini", "gpt-5-mini"), + ("openai/gpt-5-nano", "gpt-5-mini"), ("openai/gpt-4.1", "gpt-4.1"), ("openai/gpt-4.1-mini", "gpt-4.1"), + ("openai/gpt-4.1-nano", "gpt-4.1"), ("openai/gpt-4o", "gpt-4o"), ("openai/gpt-4o-mini", "gpt-4o-mini"), + ("openai/o1", "gpt-5.2"), ("openai/o1-mini", "gpt-5-mini"), ("openai/o1-preview", "gpt-5.2"), + ("openai/o3", "gpt-5.3-codex"), ("openai/o3-mini", "gpt-5-mini"), ("openai/o4-mini", "gpt-5-mini"), + ("anthropic/claude-opus-4.6", "claude-opus-4.6"), ("anthropic/claude-sonnet-5", "claude-sonnet-5"), + ("anthropic/claude-sonnet-4.6", "claude-sonnet-4.6"), ("anthropic/claude-sonnet-4", "claude-sonnet-4"), + ("anthropic/claude-sonnet-4.5", "claude-sonnet-4.5"), ("anthropic/claude-haiku-4.5", "claude-haiku-4.5"), + ("claude-sonnet-5", "claude-sonnet-5"), ("claude-opus-4-6", "claude-opus-4.6"), + ("claude-sonnet-4-6", "claude-sonnet-4.6"), ("claude-sonnet-4-0", "claude-sonnet-4"), + ("claude-sonnet-4-5", "claude-sonnet-4.5"), ("claude-haiku-4-5", "claude-haiku-4.5"), + ("anthropic/claude-opus-4-6", "claude-opus-4.6"), ("anthropic/claude-sonnet-4-6", "claude-sonnet-4.6"), + ("anthropic/claude-sonnet-4-0", "claude-sonnet-4"), ("anthropic/claude-sonnet-4-5", "claude-sonnet-4.5"), ("anthropic/claude-haiku-4-5", "claude-haiku-4.5"), )) diff --git a/hermes_cli/models_local.py b/hermes_cli/models_local.py index 9de0a34f9f..858df73b18 100644 --- a/hermes_cli/models_local.py +++ b/hermes_cli/models_local.py @@ -112,11 +112,8 @@ def _get_ollama_base_url() -> str: return model_base except (OSError, RuntimeError, TypeError, ValueError): pass - env_host = os.getenv("OLLAMA_HOST", "").strip() - if env_host: - return _ollama_host_from_env(env_host) - return "http://localhost:11434" + return _ollama_host_from_env(env_host) if env_host else "http://localhost:11434" def _api_key_from_provider_config(entry: dict, *env_keys: str) -> str: @@ -239,7 +236,6 @@ def probe_ollama_local_models( if time.monotonic() - failed_at < _OLLAMA_LOCAL_PROBE_FAILURE_TTL: return None _OLLAMA_LOCAL_PROBE_FAILURE_CACHE.pop(failure_key, None) - try: request_headers = {"User-Agent": _HERMES_USER_AGENT, **(headers or {})} req = urllib.request.Request(root.rstrip("/") + "/api/tags", headers=request_headers) @@ -354,10 +350,7 @@ def _ollama_local_catalog(force_refresh: bool) -> list[str]: base_url = _get_ollama_base_url() headers = _get_ollama_native_headers(base_url) if should_use_ollama_native_catalog("ollama", base_url, headers=headers): - if headers: - native_models = fetch_ollama_local_models(base_url, headers=headers) - else: - native_models = fetch_ollama_local_models(base_url) + native_models = fetch_ollama_local_models(base_url, headers=headers) if headers else fetch_ollama_local_models(base_url) native_key = _ollama_probe_cache_key(_root_for_ollama_native_api(base_url), headers or None) if native_models or _OLLAMA_LOCAL_PROBE_REACHABLE.get(native_key) is True: return native_models or [] @@ -375,13 +368,10 @@ def _lmstudio_server_root(base_url: Optional[str]) -> Optional[str]: def _lmstudio_request_headers(api_key: Optional[str] = None) -> dict: - """Build HTTP headers for LM Studio native API requests.""" + """HTTP headers for LM Studio native API requests.""" from hermes_cli.models import _HERMES_USER_AGENT - headers = {"User-Agent": _HERMES_USER_AGENT} token = str(api_key or "").strip() - if token: - headers["Authorization"] = f"Bearer {token}" - return headers + return {"User-Agent": _HERMES_USER_AGENT, **({"Authorization": f"Bearer {token}"} if token else {})} def _lmstudio_fetch_raw_models( @@ -609,9 +599,8 @@ def ollama_model_supports_thinking( return None token = str(api_key or "").strip() - headers = {"Authorization": f"Bearer {token}"} if token else {} try: - with httpx.Client(timeout=timeout, headers=headers) as client: + with httpx.Client(timeout=timeout, headers={"Authorization": f"Bearer {token}"} if token else {}) as client: resp = client.post(f"{server_url}/api/show", json={"name": bare_model}) if resp.status_code != 200: return None @@ -646,9 +635,7 @@ def _load_ollama_cloud_cache(*, ignore_ttl: bool = False) -> Optional[dict]: try: data = _read_json_cache(_ollama_cloud_cache_path()) - if data is None: - return None - models = data.get("models") + models = data.get("models") if data is not None else None if not (isinstance(models, list) and models): return None if not ignore_ttl and (time.time() - data.get("cached_at", 0)) > _OLLAMA_CLOUD_CACHE_TTL: diff --git a/hermes_cli/models_pricing.py b/hermes_cli/models_pricing.py index 6f095e2a40..2bebefb5db 100644 --- a/hermes_cli/models_pricing.py +++ b/hermes_cli/models_pricing.py @@ -234,18 +234,14 @@ def fetch_models_with_pricing( result: dict[str, dict[str, Any]] = {} for item in payload.get("data", []): - mid = item.get("id") - pricing = item.get("pricing") + mid, pricing = item.get("id"), item.get("pricing") if mid and isinstance(pricing, dict): entry = _pricing_entry(pricing) # Sale chrome is Nous Portal-only; never copy pricing.original for other catalogs. original = pricing.get("original") if include_sale_original else None if isinstance(original, dict): - orig_entry = { - key: str(original[key]) - for key in ("prompt", "completion", "input_cache_read", "input_cache_write") - if original.get(key) not in (None, "") - } + orig_entry = {key: str(original[key]) for key in ("prompt", "completion", "input_cache_read", "input_cache_write") + if original.get(key) not in (None, "")} if orig_entry.get("prompt") or orig_entry.get("completion"): entry["original"] = orig_entry result[mid] = entry @@ -270,8 +266,7 @@ def fetch_ai_gateway_pricing(timeout: float = 8.0, *, force_refresh: bool = Fals result: dict[str, dict[str, str]] = {} for item in _catalog_items(payload): - mid = item.get("id") - pricing = item.get("pricing") + mid, pricing = item.get("id"), item.get("pricing") if mid and isinstance(pricing, dict): result[mid] = _pricing_entry(pricing, "input", "output") return _cache_catalog(cache_key, result) diff --git a/hermes_cli/models_validate.py b/hermes_cli/models_validate.py index 2bb8b0dca8..18192737a5 100644 --- a/hermes_cli/models_validate.py +++ b/hermes_cli/models_validate.py @@ -160,11 +160,8 @@ def _parse_openrouter_preset(req: _Request) -> Optional[dict[str, Any]]: else: preset_base, preset_slug = req.requested.split(marker, 1) if re.fullmatch(r"[A-Za-z0-9._~-]+", preset_slug) is None: - return _reject( - "OpenRouter preset slugs must be non-empty URL-safe " - "identifiers using only letters, digits, '.', '_', " - "'~', or '-'." - ) + return _reject("OpenRouter preset slugs must be non-empty URL-safe identifiers using only " + "letters, digits, '.', '_', '~', or '-'.") req.preset_suffix = f"{marker}{preset_slug}" if not preset_base: return _soft_accept(None) @@ -185,27 +182,19 @@ def _validate_lmstudio(req: _Request) -> dict[str, Any]: if models is None: return _reject(f"Could not reach LM Studio's `/api/v1/models` to validate `{req.requested}`.") if not models: - return _reject( - f"LM Studio is reachable but no chat-capable models are loaded. " - f"Load `{req.requested}` in LM Studio (Developer tab → Load Model) and try again." - ) + return _reject("LM Studio is reachable but no chat-capable models are loaded. " + f"Load `{req.requested}` in LM Studio (Developer tab → Load Model) and try again.") if req.lookup in set(models): return _accept() return _reject(f"Model `{req.requested}` was not found in LM Studio's model listing.") -def _drop_authorization(headers: dict[str, str]) -> None: - for key in tuple(headers): - if key.lower() == "authorization": - del headers[key] - - def _ollama_probe_headers(req: _Request) -> dict[str, str]: """Headers for the Ollama native probe. Configured ``providers.ollama.extra_headers`` apply only when the probed endpoint is the configured one (never leak them to a different host). Caller headers win; a caller ``api_key`` becomes the Authorization header unless the caller sent one.""" from hermes_cli import models as _m - from hermes_cli.models_local import _configured_ollama_base_url + from hermes_cli.models_local import _configured_ollama_base_url, _drop_authorization configured_base = _configured_ollama_base_url() configured_allowed = not configured_base or _m._same_ollama_native_root(req.base_url or "", configured_base) @@ -243,8 +232,7 @@ def _validate_ollama_native(req: _Request) -> Optional[dict[str, Any]]: f"Note: could not reach this Ollama endpoint's `/api/tags` model listing to validate `{req.requested}`. " "Hermes will save the model name, but local Ollama model discovery could not verify it." ) - match = _match_in_catalog(req.lookup, models, auto_correct=False, - suggest_label="Similar local Ollama models") + match = _match_in_catalog(req.lookup, models, auto_correct=False, suggest_label="Similar local Ollama models") if match.exact: return _accept() empty_hint = " No models are currently listed by `/api/tags`." if not models else "" @@ -274,10 +262,8 @@ def _validate_custom(req: _Request) -> dict[str, Any]: f"{match.suggestion_text}" ) if probe.get("used_fallback"): - message += ( - f"\n Endpoint verification succeeded after trying `{probe.get('resolved_base_url')}`. " - f"Consider saving that as your base URL." - ) + message += (f"\n Endpoint verification succeeded after trying `{probe.get('resolved_base_url')}`. " + "Consider saving that as your base URL.") return _soft_accept(message) message = ( @@ -285,10 +271,8 @@ def _validate_custom(req: _Request) -> dict[str, Any]: f"Hermes will still save `{req.requested}`, but the endpoint should expose `/models` for verification." ) if anthropic_style: - message += ( - "\n Many Anthropic-compatible proxies do not implement the Models API " - "(GET /v1/models). The model name has been accepted without verification." - ) + message += ("\n Many Anthropic-compatible proxies do not implement the Models API (GET /v1/models). " + "The model name has been accepted without verification.") if probe.get("suggested_base_url"): message += f"\n If this server expects `/v1`, try base URL: `{probe.get('suggested_base_url')}`" # Anthropic-style proxies routinely lack /v1/models, so only they are accepted unverified. @@ -328,11 +312,9 @@ def _validate_static_catalog(req: _Request) -> Optional[dict[str, Any]]: return _accept() base_guess = req.lookup[: -len(CODEX_CONTEXT_VARIANT_SUFFIX)] return _reject( - f"`{req.requested}` is not a valid large-context variant — " - f"`{base_guess}` enforces the standard 272K window on " - f"Codex, so no `-900k` option exists for it. Pick the " - f"base model, or a verified variant from the `/model` " - f"picker (e.g. `gpt-5.6-sol-900k`)." + f"`{req.requested}` is not a valid large-context variant — `{base_guess}` enforces the " + "standard 272K window on Codex, so no `-900k` option exists for it. Pick the base model, " + "or a verified variant from the `/model` picker (e.g. `gpt-5.6-sol-900k`)." ) if not catalog: return None @@ -349,12 +331,9 @@ def _validate_static_catalog(req: _Request) -> Optional[dict[str, Any]]: lower = req.lookup.strip().lower() if prefixes and not any(lower.startswith(p) for p in prefixes): return _reject( - f"`{req.requested}` doesn't look like a {label} model " - f"and isn't in its listing, so it was not accepted. If it " - f"belongs to another configured provider, switch with " - f"`--provider ` (or select it from the `/model` " - f"picker)." - f"{match.suggestion_text}" + f"`{req.requested}` doesn't look like a {label} model and isn't in its listing, so it was not " + "accepted. If it belongs to another configured provider, switch with `--provider ` " + f"(or select it from the `/model` picker).{match.suggestion_text}" ) return _soft_accept( f"Note: `{req.requested}` was not found in the {label} model listing. " @@ -405,10 +384,9 @@ def _validate_anthropic_messages(req: _Request) -> dict[str, Any]: models = _m.fetch_api_models(req.api_key, req.base_url, api_mode=req.api_mode) verdict = _match_in_catalog(req.lookup, models).verdict(req) if models is not None else None return verdict or _soft_accept( - f"Note: could not verify `{req.requested}` against this endpoint's " - f"model listing. Many Anthropic-compatible proxies do not " - f"implement GET /v1/models. The model name has been accepted " - f"without verification." + f"Note: could not verify `{req.requested}` against this endpoint's model listing. Many " + "Anthropic-compatible proxies do not implement GET /v1/models. The model name has been accepted " + "without verification." ) @@ -437,12 +415,9 @@ def _validate_live_listing(req: _Request) -> Optional[dict[str, Any]]: if api_models is None: return None if req.normalized == "gemini": - # Gemini's OpenAI-compat listing prefixes ids with "models/"; curated list and user - # input use the bare id, so strip before comparing. - api_models = [ - m[len("models/"):] if isinstance(m, str) and m.startswith("models/") else m - for m in api_models - ] + # Gemini's OpenAI-compat listing prefixes ids with "models/"; curated list and user input + # use the bare id, so strip before comparing. + api_models = [m[len("models/"):] if isinstance(m, str) and m.startswith("models/") else m for m in api_models] match = _match_in_catalog(req.lookup, api_models) if match.exact: return _accept() @@ -471,22 +446,15 @@ def _validate_live_listing(req: _Request) -> Optional[dict[str, Any]]: if not listing_authoritative and _m._model_in_provider_catalog( (variant_base or req.lookup).lower(), _m._provider_keys(req.normalized) ): - return _accept_with_note( - f"Note: `{req.requested}` was not found in the live /v1/models listing " - f"but exists in the curated catalog — accepted." - ) + return _accept_with_note(f"Note: `{req.requested}` was not found in the live /v1/models listing " + "but exists in the curated catalog — accepted.") # Nous: the Portal's recommended-models feed can list a model before the curated list or the # docs-hosted manifest catches up; `hermes chat` already accepts those at model-list build # time, so mirror that source of truth for per-message /model validation. if req.normalized == "nous" and req.lookup.lower() in _nous_portal_recommended_names(): - return _accept_with_note( - f"Note: `{req.requested}` was not found in the live /v1/models " - f"listing but is a current Nous Portal recommendation — accepted." - ) - return _reject( - f"Model `{req.requested}` was not found in this provider's model listing." - f"{match.suggestion_text}" - ) + return _accept_with_note(f"Note: `{req.requested}` was not found in the live /v1/models listing " + "but is a current Nous Portal recommendation — accepted.") + return _reject(f"Model `{req.requested}` was not found in this provider's model listing.{match.suggestion_text}") def _validate_bedrock(req: _Request) -> Optional[dict[str, Any]]: @@ -497,8 +465,7 @@ def _validate_bedrock(req: _Request) -> Optional[dict[str, Any]]: region = resolve_bedrock_runtime_region() discovered_ids = {m["id"] for m in discover_bedrock_models(region)} - match = _match_in_catalog(req.requested, list(discovered_ids), auto_correct=False, - suggest_cutoff=0.4) + match = _match_in_catalog(req.requested, list(discovered_ids), auto_correct=False, suggest_cutoff=0.4) if match.exact: return _accept() # Still accept (custom inference profiles / cross-account access), but warn. @@ -520,10 +487,8 @@ def _validate_catalog_fallback(req: _Request) -> dict[str, Any]: label = _m._PROVIDER_LABELS.get(req.normalized, req.normalized) catalog = _static_catalog(req.normalized) if not catalog: - return _soft_accept( - f"Note: could not reach the {label} API to validate `{req.requested}`. " - f"If the service isn't down, this model may not be valid." - ) + return _soft_accept(f"Note: could not reach the {label} API to validate `{req.requested}`. " + "If the service isn't down, this model may not be valid.") match = _match_in_catalog(req.lookup, catalog, case_insensitive=True) if match.exact: return _accept()