refactor(hermes_cli): repack literal tables, one-line trivial branches, share _drop_authorization

This commit is contained in:
Teknium
2026-09-02 21:57:37 -07:00
parent b30c4f1667
commit c64a1fbd6c
4 changed files with 115 additions and 189 deletions
+75 -96
View File
@@ -26,18 +26,16 @@ OPENROUTER_MODELS: list[tuple[str, str]] = [
for mid in (
"anthropic/claude-fable-5.1", "anthropic/claude-fable-5", "anthropic/claude-opus-5",
"anthropic/claude-opus-5-fast", "anthropic/claude-opus-4.8", "anthropic/claude-opus-4.8-fast",
"anthropic/claude-sonnet-5", "anthropic/claude-haiku-4.5", "openai/gpt-5.6-sol",
"openai/gpt-5.6-sol-pro", "openai/gpt-5.6-terra", "openai/gpt-5.6-terra-pro", "openai/gpt-5.6-luna",
"openai/gpt-5.6-luna-pro", "openai/gpt-5.5", "openai/gpt-5.5-pro", "openai/gpt-5.4-mini",
"google/gemini-3.1-pro-preview", "google/gemini-3.8-flash", "google/gemini-3.7-flash",
"x-ai/grok-4.6", "deepseek/deepseek-v4-pro", "deepseek/deepseek-v4-pro-0813",
"deepseek/deepseek-v4-flash", "deepseek/deepseek-v4-flash-0731", "qwen/qwen3.8-max",
"qwen/qwen3.8-flash", "moonshotai/kimi-k3", "minimax/minimax-m3", "z-ai/glm-5.3",
"anthropic/claude-sonnet-5", "anthropic/claude-haiku-4.5", "openai/gpt-5.6-sol", "openai/gpt-5.6-sol-pro",
"openai/gpt-5.6-terra", "openai/gpt-5.6-terra-pro", "openai/gpt-5.6-luna", "openai/gpt-5.6-luna-pro",
"openai/gpt-5.5", "openai/gpt-5.5-pro", "openai/gpt-5.4-mini", "google/gemini-3.1-pro-preview",
"google/gemini-3.8-flash", "google/gemini-3.7-flash", "x-ai/grok-4.6", "deepseek/deepseek-v4-pro",
"deepseek/deepseek-v4-pro-0813", "deepseek/deepseek-v4-flash", "deepseek/deepseek-v4-flash-0731",
"qwen/qwen3.8-max", "qwen/qwen3.8-flash", "moonshotai/kimi-k3", "minimax/minimax-m3", "z-ai/glm-5.3",
"z-ai/glm-5.3-flash", "z-ai/glm-5.2", "xiaomi/mimo-v2.5-pro", "tencent/hy4-preview", "tencent/hy3",
"stepfun/step-3.7-flash", "nvidia/nemotron-3-super-120b-a12b", "meta/muse-spark-1.2",
"sakana/fugu-ultra", "openrouter/pareto-code", "thinkingmachines/inkling:free",
"thinkingmachines/inkling-small:free", "minimax/minimax-m3:free", "z-ai/glm-5.2:free",
"poolside/laguna-s-2.1:free", "poolside/laguna-xs-2.1:free",
"stepfun/step-3.7-flash", "nvidia/nemotron-3-super-120b-a12b", "meta/muse-spark-1.2", "sakana/fugu-ultra",
"openrouter/pareto-code", "thinkingmachines/inkling:free", "thinkingmachines/inkling-small:free",
"minimax/minimax-m3:free", "z-ai/glm-5.2:free", "poolside/laguna-s-2.1:free", "poolside/laguna-xs-2.1:free",
"nvidia/nemotron-3-super-120b-a12b:free", "nvidia/nemotron-3-ultra-550b-a55b:free",
"nvidia/nemotron-3.5-lightning:free",
)
@@ -45,8 +43,7 @@ OPENROUTER_MODELS: list[tuple[str, str]] = [
# OpenRouter entries the Nous Portal does not carry (routing/fast variants, free tier).
_OPENROUTER_ONLY = {
"anthropic/claude-opus-5-fast", "anthropic/claude-opus-4.8-fast", "meta/muse-spark-1.2",
"openrouter/pareto-code",
"anthropic/claude-opus-5-fast", "anthropic/claude-opus-4.8-fast", "meta/muse-spark-1.2", "openrouter/pareto-code",
}
@@ -124,8 +121,7 @@ def _xai_curated_models() -> list[str]:
# Native OpenAI Chat Completions (api.openai.com); also the head of the Copilot list.
_OPENAI_CHAT_MODELS = [
"gpt-5.4", "gpt-5.4-mini", "gpt-5-mini", "gpt-5.3-codex", "gpt-5.2-codex", "gpt-4.1", "gpt-4o",
"gpt-4o-mini",
"gpt-5.4", "gpt-5.4-mini", "gpt-5-mini", "gpt-5.3-codex", "gpt-5.2-codex", "gpt-4.1", "gpt-4o", "gpt-4o-mini",
]
_MINIMAX_MODELS = ["MiniMax-M3", "MiniMax-M2.7", "MiniMax-M2.5", "MiniMax-M2.1", "MiniMax-M2"]
_TENCENT_MODELS = ["hy4-preview", "hy3", "hy3-preview"]
@@ -143,9 +139,8 @@ _ALIBABA_CODING_PLAN_MODELS = [
]
# Verified against a live Token Plan subscription (key tier ``sk-sp-...``).
_ALIBABA_TOKEN_PLAN_MODELS = [
"qwen3.8-max-preview", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash",
"deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v3.2", "kimi-k2.7-code", "kimi-k2.6",
"kimi-k2.5", "glm-5.2", "glm-5.1", "glm-5",
"qwen3.8-max-preview", "qwen3.7-max", "qwen3.7-plus", "qwen3.6-plus", "qwen3.6-flash", "deepseek-v4-pro",
"deepseek-v4-flash", "deepseek-v3.2", "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "glm-5.2", "glm-5.1", "glm-5",
]
_XAI_MODELS = _xai_curated_models()
@@ -164,13 +159,11 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
"xai-oauth": list(_XAI_MODELS),
"copilot-acp": ["copilot-acp"],
"copilot": _OPENAI_CHAT_MODELS + [
"claude-sonnet-4.6", "claude-sonnet-5", "claude-sonnet-4", "claude-sonnet-4.5",
"claude-haiku-4.5", "gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3-flash-preview",
"gemini-2.5-pro",
"claude-sonnet-4.6", "claude-sonnet-5", "claude-sonnet-4", "claude-sonnet-4.5", "claude-haiku-4.5",
"gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3-flash-preview", "gemini-2.5-pro",
],
"gemini": [
"gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3.6-flash",
"gemini-3.1-flash-lite-preview",
"gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3.6-flash", "gemini-3.1-flash-lite-preview",
],
"zai": [
"glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5", "glm-5v-turbo", "glm-5-turbo",
@@ -184,9 +177,8 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
"z-ai/glm-5.3", "z-ai/glm-5.2", "moonshotai/kimi-k2.6", "minimaxai/minimax-m3",
],
"kimi-coding": [
"kimi-k3", "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "kimi-for-coding",
"kimi-for-coding-highspeed", "kimi-k2-thinking", "kimi-k2-thinking-turbo",
"kimi-k2-turbo-preview", "kimi-k2-0905-preview",
"kimi-k3", "kimi-k2.7-code", "kimi-k2.6", "kimi-k2.5", "kimi-for-coding", "kimi-for-coding-highspeed",
"kimi-k2-thinking", "kimi-k2-thinking-turbo", "kimi-k2-turbo-preview", "kimi-k2-0905-preview",
],
"kimi-coding-cn": [
"kimi-k3", "kimi-k2.7-code", "kimi-k2.7-code-highspeed", "kimi-k2.6", "kimi-k2.5",
@@ -194,8 +186,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
],
"stepfun": ["step-3.5-flash", "step-3.5-flash-2603"],
"moonshot": [
"kimi-k3", "kimi-k2.6", "kimi-k2.5", "kimi-k2-thinking", "kimi-k2-turbo-preview",
"kimi-k2-0905-preview",
"kimi-k3", "kimi-k2.6", "kimi-k2.5", "kimi-k2-thinking", "kimi-k2-turbo-preview", "kimi-k2-0905-preview",
],
"minimax": list(_MINIMAX_MODELS),
"minimax-oauth": ["MiniMax-M3", "MiniMax-M2.7", "MiniMax-M2.7-highspeed"],
@@ -219,19 +210,18 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
# _LIVE_FIRST_PICKER_PROVIDERS, so this is a discovery floor: live entries lead in the picker
# and stale curated names never pollute the top. "x-preview-f-free" = "Ox Alpha" stealth model.
"opencode-zen": [
"x-preview-f-free", "kimi-k3", "kimi-k2.5", "kimi-k2.6", "gpt-5.6-sol", "gpt-5.6-terra",
"gpt-5.6-luna", "gpt-5.5", "gpt-5.5-pro", "gpt-5.4-pro", "gpt-5.4", "gpt-5.4-mini",
"gpt-5.4-nano", "gpt-5.3-codex", "gpt-5.3-codex-spark", "gpt-5.2", "gpt-5.2-codex", "gpt-5.1",
"gpt-5.1-codex", "gpt-5.1-codex-max", "gpt-5.1-codex-mini", "gpt-5", "gpt-5-codex",
"gpt-5-nano", "claude-fable-5", "claude-opus-5", "claude-sonnet-5", "claude-opus-4-8",
"claude-opus-4-7", "claude-opus-4-6", "claude-opus-4-5", "claude-sonnet-4-6",
"claude-sonnet-4-5", "claude-sonnet-4", "claude-haiku-4-5", "gemini-3.7-flash",
"gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro",
"gemini-3-flash", "grok-4.6", "grok-4.5", "grok-build-0.1", "muse-spark-1.2", "minimax-m3",
"minimax-m2.7", "minimax-m2.5", "glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5",
"kimi-k2.7-code", "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-free",
"qwen3.6-plus", "qwen3.5-plus", "big-pickle", "mimo-v2.5-free", "hy3-free", "laguna-s-2.1-free",
"nemotron-3-ultra-free", "nemotron-3.5-lightning-free", "muse-spark-1.2-contributor-free",
"x-preview-f-free", "kimi-k3", "kimi-k2.5", "kimi-k2.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna",
"gpt-5.5", "gpt-5.5-pro", "gpt-5.4-pro", "gpt-5.4", "gpt-5.4-mini", "gpt-5.4-nano", "gpt-5.3-codex",
"gpt-5.3-codex-spark", "gpt-5.2", "gpt-5.2-codex", "gpt-5.1", "gpt-5.1-codex", "gpt-5.1-codex-max",
"gpt-5.1-codex-mini", "gpt-5", "gpt-5-codex", "gpt-5-nano", "claude-fable-5", "claude-opus-5",
"claude-sonnet-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", "claude-opus-4-5",
"claude-sonnet-4-6", "claude-sonnet-4-5", "claude-sonnet-4", "claude-haiku-4-5", "gemini-3.7-flash",
"gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro", "gemini-3-flash",
"grok-4.6", "grok-4.5", "grok-build-0.1", "muse-spark-1.2", "minimax-m3", "minimax-m2.7", "minimax-m2.5",
"glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5", "kimi-k2.7-code", "deepseek-v4-pro",
"deepseek-v4-flash", "deepseek-v4-flash-free", "qwen3.6-plus", "qwen3.5-plus", "big-pickle", "mimo-v2.5-free",
"hy3-free", "laguna-s-2.1-free", "nemotron-3-ultra-free", "nemotron-3.5-lightning-free",
"muse-spark-1.2-contributor-free",
],
# OpenCode keyless free tier — OFFLINE FLOOR only. provider_model_ids("opencode-free")
# revalidates live against GET /zen/v1/models and filters to the anonymous tier, so this list
@@ -269,11 +259,10 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
# Static fallback when live discovery (ListFoundationModels + ListInferenceProfiles) is
# unavailable. Inference-profile IDs (us.*) because most models require them.
"bedrock": [
"us.anthropic.claude-sonnet-5", "us.anthropic.claude-sonnet-4-6",
"us.anthropic.claude-opus-4-6-v1", "us.anthropic.claude-haiku-4-5-20251001-v1:0",
"us.anthropic.claude-sonnet-4-5-20250929-v1:0", "openai.gpt-5.5", "openai.gpt-5.6-sol",
"openai.gpt-5.6-terra", "openai.gpt-5.6-luna", "us.amazon.nova-pro-v1:0",
"us.amazon.nova-lite-v1:0", "us.amazon.nova-micro-v1:0", "deepseek.v3.2",
"us.anthropic.claude-sonnet-5", "us.anthropic.claude-sonnet-4-6", "us.anthropic.claude-opus-4-6-v1",
"us.anthropic.claude-haiku-4-5-20251001-v1:0", "us.anthropic.claude-sonnet-4-5-20250929-v1:0",
"openai.gpt-5.5", "openai.gpt-5.6-sol", "openai.gpt-5.6-terra", "openai.gpt-5.6-luna",
"us.amazon.nova-pro-v1:0", "us.amazon.nova-lite-v1:0", "us.amazon.nova-micro-v1:0", "deepseek.v3.2",
"us.meta.llama4-maverick-17b-instruct-v1:0", "us.meta.llama4-scout-17b-instruct-v1:0",
],
# Azure Foundry models depend on the user's endpoint configuration.
@@ -329,8 +318,7 @@ CANONICAL_PROVIDERS: list[ProviderEntry] = [ProviderEntry(*row) for row in (
("huggingface", "Hugging Face", "Hugging Face Inference Providers"),
("gemini", "Google AI Studio", "Google AI Studio (Native Gemini API)"),
("vertex", "Google Vertex AI", "Google Vertex AI (Gemini via GCP; OAuth2 service account or ADC, GCP billing/quotas)"),
("deepseek", "DeepSeek", "DeepSeek (V3, R1, coder, direct API)"),
("xai", "xAI", "xAI Grok (Direct API)"),
("deepseek", "DeepSeek", "DeepSeek (V3, R1, coder, direct API)"), ("xai", "xAI", "xAI Grok (Direct API)"),
("zai", "Z.AI / GLM", "Z.AI / GLM (Zhipu direct API)"),
("kimi-coding", "Kimi / Kimi Coding Plan", "Kimi Coding Plan (api.kimi.com & Moonshot API)"),
("kimi-coding-cn", "Kimi / Moonshot (China)", "Kimi / Moonshot China (Domestic direct API)"),
@@ -446,34 +434,31 @@ _PROVIDER_ALIASES = dict((
("glm", "zai"), ("z-ai", "zai"), ("z.ai", "zai"), ("zhipu", "zai"), ("github", "copilot"),
("github-copilot", "copilot"), ("github-models", "copilot"), ("github-model", "copilot"),
("github-copilot-acp", "copilot-acp"), ("copilot-acp-agent", "copilot-acp"), ("google", "gemini"),
("google-gemini", "gemini"), ("google-ai-studio", "gemini"), ("google-vertex", "vertex"),
("vertex-ai", "vertex"), ("gcp-vertex", "vertex"), ("vertexai", "vertex"), ("kimi", "kimi-coding"),
("moonshot", "kimi-coding"), ("kimi-cn", "kimi-coding-cn"), ("moonshot-cn", "kimi-coding-cn"),
("step", "stepfun"), ("stepfun-coding-plan", "stepfun"), ("arcee-ai", "arcee"),
("arceeai", "arcee"), ("gmi-cloud", "gmi"), ("gmicloud", "gmi"), ("fireworks-ai", "fireworks"),
("fw", "fireworks"), ("actual-computer", "actual"), ("actualcomputer", "actual"), ("aci", "actual"),
("nebius", "nebius-token-factory"), ("nebius-tokenfactory", "nebius-token-factory"),
("nebius-tf", "nebius-token-factory"), ("token-factory", "nebius-token-factory"),
("tokenfactory", "nebius-token-factory"), ("minimax-china", "minimax-cn"),
("minimax_cn", "minimax-cn"), ("minimax-portal", "minimax-oauth"),
("google-gemini", "gemini"), ("google-ai-studio", "gemini"), ("google-vertex", "vertex"), ("vertex-ai", "vertex"),
("gcp-vertex", "vertex"), ("vertexai", "vertex"), ("kimi", "kimi-coding"), ("moonshot", "kimi-coding"),
("kimi-cn", "kimi-coding-cn"), ("moonshot-cn", "kimi-coding-cn"), ("step", "stepfun"),
("stepfun-coding-plan", "stepfun"), ("arcee-ai", "arcee"), ("arceeai", "arcee"), ("gmi-cloud", "gmi"),
("gmicloud", "gmi"), ("fireworks-ai", "fireworks"), ("fw", "fireworks"), ("actual-computer", "actual"),
("actualcomputer", "actual"), ("aci", "actual"), ("nebius", "nebius-token-factory"),
("nebius-tokenfactory", "nebius-token-factory"), ("nebius-tf", "nebius-token-factory"),
("token-factory", "nebius-token-factory"), ("tokenfactory", "nebius-token-factory"),
("minimax-china", "minimax-cn"), ("minimax_cn", "minimax-cn"), ("minimax-portal", "minimax-oauth"),
("minimax-global", "minimax-oauth"), ("minimax_oauth", "minimax-oauth"), ("claude", "anthropic"),
("claude-code", "anthropic"), ("deep-seek", "deepseek"), ("opencode", "opencode-zen"),
("zen", "opencode-zen"), ("go", "opencode-go"), ("opencode-go-sub", "opencode-go"),
("free", "opencode-free"), ("opencode_free", "opencode-free"), ("aigateway", "ai-gateway"),
("vercel", "ai-gateway"), ("vercel-ai-gateway", "ai-gateway"), ("kilo", "kilocode"),
("kilo-code", "kilocode"), ("kilo-gateway", "kilocode"), ("dashscope", "alibaba"),
("aliyun", "alibaba"), ("qwen", "alibaba"), ("alibaba-cloud", "alibaba"),
("qwen-portal", "qwen-oauth"), ("hf", "huggingface"), ("hugging-face", "huggingface"),
("huggingface-hub", "huggingface"), ("novita-ai", "novita"), ("novitaai", "novita"),
("mimo", "xiaomi"), ("xiaomi-mimo", "xiaomi"), ("tencent", "tencent-tokenhub"),
("tokenhub", "tencent-tokenhub"), ("tencent-cloud", "tencent-tokenhub"),
("tencentmaas", "tencent-tokenhub"), ("tokenplan", "tencent-tokenplan"),
("tencent-lkeap", "tencent-tokenplan"), ("aws", "bedrock"), ("aws-bedrock", "bedrock"),
("amazon-bedrock", "bedrock"), ("amazon", "bedrock"), ("grok", "xai"), ("grok-oauth", "xai-oauth"),
("xai-oauth", "xai-oauth"), ("x-ai-oauth", "xai-oauth"), ("xai-grok-oauth", "xai-oauth"),
("x-ai", "xai"), ("x.ai", "xai"), ("nim", "nvidia"), ("nvidia-nim", "nvidia"),
("build-nvidia", "nvidia"), ("nemotron", "nvidia"), ("lmstudio", "lmstudio"),
("lm-studio", "lmstudio"), ("lm_studio", "lmstudio"),
("claude-code", "anthropic"), ("deep-seek", "deepseek"), ("opencode", "opencode-zen"), ("zen", "opencode-zen"),
("go", "opencode-go"), ("opencode-go-sub", "opencode-go"), ("free", "opencode-free"),
("opencode_free", "opencode-free"), ("aigateway", "ai-gateway"), ("vercel", "ai-gateway"),
("vercel-ai-gateway", "ai-gateway"), ("kilo", "kilocode"), ("kilo-code", "kilocode"),
("kilo-gateway", "kilocode"), ("dashscope", "alibaba"), ("aliyun", "alibaba"), ("qwen", "alibaba"),
("alibaba-cloud", "alibaba"), ("qwen-portal", "qwen-oauth"), ("hf", "huggingface"),
("hugging-face", "huggingface"), ("huggingface-hub", "huggingface"), ("novita-ai", "novita"),
("novitaai", "novita"), ("mimo", "xiaomi"), ("xiaomi-mimo", "xiaomi"), ("tencent", "tencent-tokenhub"),
("tokenhub", "tencent-tokenhub"), ("tencent-cloud", "tencent-tokenhub"), ("tencentmaas", "tencent-tokenhub"),
("tokenplan", "tencent-tokenplan"), ("tencent-lkeap", "tencent-tokenplan"), ("aws", "bedrock"),
("aws-bedrock", "bedrock"), ("amazon-bedrock", "bedrock"), ("amazon", "bedrock"), ("grok", "xai"),
("grok-oauth", "xai-oauth"), ("xai-oauth", "xai-oauth"), ("x-ai-oauth", "xai-oauth"),
("xai-grok-oauth", "xai-oauth"), ("x-ai", "xai"), ("x.ai", "xai"), ("nim", "nvidia"), ("nvidia-nim", "nvidia"),
("build-nvidia", "nvidia"), ("nemotron", "nvidia"), ("lmstudio", "lmstudio"), ("lm-studio", "lmstudio"),
("lm_studio", "lmstudio"),
("ollama", "custom"), # bare "ollama" = local; use "ollama-cloud" for cloud
("ollama_cloud", "ollama-cloud"),
))
@@ -545,9 +530,8 @@ _OPENAI_FAST_MODE_PREFIXES: tuple[str, ...] = ("gpt-", "o1", "o3", "o4")
# /models are the subscription-tier source of truth), and providers with dedicated live-endpoint
# branches (copilot, anthropic, ai-gateway, ollama-cloud, custom, stepfun, openai-codex).
_MODELS_DEV_PREFERRED: frozenset[str] = frozenset({
"opencode-go", "opencode-zen", "deepseek", "kilocode", "fireworks", "mistral", "togetherai",
"cohere", "perplexity", "groq", "nvidia", "huggingface", "zai", "gemini", "google", "xai",
"xai-oauth",
"opencode-go", "opencode-zen", "deepseek", "kilocode", "fireworks", "mistral", "togetherai", "cohere",
"perplexity", "groq", "nvidia", "huggingface", "zai", "gemini", "google", "xai", "xai-oauth",
})
@@ -561,24 +545,19 @@ _KEYLESS_STABLE_CACHE_PROVIDERS = frozenset({"opencode-free"})
# Claude IDs use hyphens (Anthropic native) but Copilot's API only accepts dot-notation, so a
# copilot + hyphenated default would otherwise hit HTTP 400 "model_not_supported".
_COPILOT_MODEL_ALIASES = dict((
("openai/gpt-5", "gpt-5-mini"), ("openai/gpt-5-chat", "gpt-5-mini"),
("openai/gpt-5-mini", "gpt-5-mini"), ("openai/gpt-5-nano", "gpt-5-mini"),
("openai/gpt-4.1", "gpt-4.1"), ("openai/gpt-4.1-mini", "gpt-4.1"),
("openai/gpt-4.1-nano", "gpt-4.1"), ("openai/gpt-4o", "gpt-4o"),
("openai/gpt-4o-mini", "gpt-4o-mini"), ("openai/o1", "gpt-5.2"), ("openai/o1-mini", "gpt-5-mini"),
("openai/o1-preview", "gpt-5.2"), ("openai/o3", "gpt-5.3-codex"), ("openai/o3-mini", "gpt-5-mini"),
("openai/o4-mini", "gpt-5-mini"), ("anthropic/claude-opus-4.6", "claude-opus-4.6"),
("anthropic/claude-sonnet-5", "claude-sonnet-5"),
("anthropic/claude-sonnet-4.6", "claude-sonnet-4.6"),
("anthropic/claude-sonnet-4", "claude-sonnet-4"),
("anthropic/claude-sonnet-4.5", "claude-sonnet-4.5"),
("anthropic/claude-haiku-4.5", "claude-haiku-4.5"), ("claude-sonnet-5", "claude-sonnet-5"),
("claude-opus-4-6", "claude-opus-4.6"), ("claude-sonnet-4-6", "claude-sonnet-4.6"),
("claude-sonnet-4-0", "claude-sonnet-4"), ("claude-sonnet-4-5", "claude-sonnet-4.5"),
("claude-haiku-4-5", "claude-haiku-4.5"), ("anthropic/claude-opus-4-6", "claude-opus-4.6"),
("anthropic/claude-sonnet-4-6", "claude-sonnet-4.6"),
("anthropic/claude-sonnet-4-0", "claude-sonnet-4"),
("anthropic/claude-sonnet-4-5", "claude-sonnet-4.5"),
("openai/gpt-5", "gpt-5-mini"), ("openai/gpt-5-chat", "gpt-5-mini"), ("openai/gpt-5-mini", "gpt-5-mini"),
("openai/gpt-5-nano", "gpt-5-mini"), ("openai/gpt-4.1", "gpt-4.1"), ("openai/gpt-4.1-mini", "gpt-4.1"),
("openai/gpt-4.1-nano", "gpt-4.1"), ("openai/gpt-4o", "gpt-4o"), ("openai/gpt-4o-mini", "gpt-4o-mini"),
("openai/o1", "gpt-5.2"), ("openai/o1-mini", "gpt-5-mini"), ("openai/o1-preview", "gpt-5.2"),
("openai/o3", "gpt-5.3-codex"), ("openai/o3-mini", "gpt-5-mini"), ("openai/o4-mini", "gpt-5-mini"),
("anthropic/claude-opus-4.6", "claude-opus-4.6"), ("anthropic/claude-sonnet-5", "claude-sonnet-5"),
("anthropic/claude-sonnet-4.6", "claude-sonnet-4.6"), ("anthropic/claude-sonnet-4", "claude-sonnet-4"),
("anthropic/claude-sonnet-4.5", "claude-sonnet-4.5"), ("anthropic/claude-haiku-4.5", "claude-haiku-4.5"),
("claude-sonnet-5", "claude-sonnet-5"), ("claude-opus-4-6", "claude-opus-4.6"),
("claude-sonnet-4-6", "claude-sonnet-4.6"), ("claude-sonnet-4-0", "claude-sonnet-4"),
("claude-sonnet-4-5", "claude-sonnet-4.5"), ("claude-haiku-4-5", "claude-haiku-4.5"),
("anthropic/claude-opus-4-6", "claude-opus-4.6"), ("anthropic/claude-sonnet-4-6", "claude-sonnet-4.6"),
("anthropic/claude-sonnet-4-0", "claude-sonnet-4"), ("anthropic/claude-sonnet-4-5", "claude-sonnet-4.5"),
("anthropic/claude-haiku-4-5", "claude-haiku-4.5"),
))
+6 -19
View File
@@ -112,11 +112,8 @@ def _get_ollama_base_url() -> str:
return model_base
except (OSError, RuntimeError, TypeError, ValueError):
pass
env_host = os.getenv("OLLAMA_HOST", "").strip()
if env_host:
return _ollama_host_from_env(env_host)
return "http://localhost:11434"
return _ollama_host_from_env(env_host) if env_host else "http://localhost:11434"
def _api_key_from_provider_config(entry: dict, *env_keys: str) -> str:
@@ -239,7 +236,6 @@ def probe_ollama_local_models(
if time.monotonic() - failed_at < _OLLAMA_LOCAL_PROBE_FAILURE_TTL:
return None
_OLLAMA_LOCAL_PROBE_FAILURE_CACHE.pop(failure_key, None)
try:
request_headers = {"User-Agent": _HERMES_USER_AGENT, **(headers or {})}
req = urllib.request.Request(root.rstrip("/") + "/api/tags", headers=request_headers)
@@ -354,10 +350,7 @@ def _ollama_local_catalog(force_refresh: bool) -> list[str]:
base_url = _get_ollama_base_url()
headers = _get_ollama_native_headers(base_url)
if should_use_ollama_native_catalog("ollama", base_url, headers=headers):
if headers:
native_models = fetch_ollama_local_models(base_url, headers=headers)
else:
native_models = fetch_ollama_local_models(base_url)
native_models = fetch_ollama_local_models(base_url, headers=headers) if headers else fetch_ollama_local_models(base_url)
native_key = _ollama_probe_cache_key(_root_for_ollama_native_api(base_url), headers or None)
if native_models or _OLLAMA_LOCAL_PROBE_REACHABLE.get(native_key) is True:
return native_models or []
@@ -375,13 +368,10 @@ def _lmstudio_server_root(base_url: Optional[str]) -> Optional[str]:
def _lmstudio_request_headers(api_key: Optional[str] = None) -> dict:
"""Build HTTP headers for LM Studio native API requests."""
"""HTTP headers for LM Studio native API requests."""
from hermes_cli.models import _HERMES_USER_AGENT
headers = {"User-Agent": _HERMES_USER_AGENT}
token = str(api_key or "").strip()
if token:
headers["Authorization"] = f"Bearer {token}"
return headers
return {"User-Agent": _HERMES_USER_AGENT, **({"Authorization": f"Bearer {token}"} if token else {})}
def _lmstudio_fetch_raw_models(
@@ -609,9 +599,8 @@ def ollama_model_supports_thinking(
return None
token = str(api_key or "").strip()
headers = {"Authorization": f"Bearer {token}"} if token else {}
try:
with httpx.Client(timeout=timeout, headers=headers) as client:
with httpx.Client(timeout=timeout, headers={"Authorization": f"Bearer {token}"} if token else {}) as client:
resp = client.post(f"{server_url}/api/show", json={"name": bare_model})
if resp.status_code != 200:
return None
@@ -646,9 +635,7 @@ def _load_ollama_cloud_cache(*, ignore_ttl: bool = False) -> Optional[dict]:
try:
data = _read_json_cache(_ollama_cloud_cache_path())
if data is None:
return None
models = data.get("models")
models = data.get("models") if data is not None else None
if not (isinstance(models, list) and models):
return None
if not ignore_ttl and (time.time() - data.get("cached_at", 0)) > _OLLAMA_CLOUD_CACHE_TTL:
+4 -9
View File
@@ -234,18 +234,14 @@ def fetch_models_with_pricing(
result: dict[str, dict[str, Any]] = {}
for item in payload.get("data", []):
mid = item.get("id")
pricing = item.get("pricing")
mid, pricing = item.get("id"), item.get("pricing")
if mid and isinstance(pricing, dict):
entry = _pricing_entry(pricing)
# Sale chrome is Nous Portal-only; never copy pricing.original for other catalogs.
original = pricing.get("original") if include_sale_original else None
if isinstance(original, dict):
orig_entry = {
key: str(original[key])
for key in ("prompt", "completion", "input_cache_read", "input_cache_write")
if original.get(key) not in (None, "")
}
orig_entry = {key: str(original[key]) for key in ("prompt", "completion", "input_cache_read", "input_cache_write")
if original.get(key) not in (None, "")}
if orig_entry.get("prompt") or orig_entry.get("completion"):
entry["original"] = orig_entry
result[mid] = entry
@@ -270,8 +266,7 @@ def fetch_ai_gateway_pricing(timeout: float = 8.0, *, force_refresh: bool = Fals
result: dict[str, dict[str, str]] = {}
for item in _catalog_items(payload):
mid = item.get("id")
pricing = item.get("pricing")
mid, pricing = item.get("id"), item.get("pricing")
if mid and isinstance(pricing, dict):
result[mid] = _pricing_entry(pricing, "input", "output")
return _cache_catalog(cache_key, result)
+30 -65
View File
@@ -160,11 +160,8 @@ def _parse_openrouter_preset(req: _Request) -> Optional[dict[str, Any]]:
else:
preset_base, preset_slug = req.requested.split(marker, 1)
if re.fullmatch(r"[A-Za-z0-9._~-]+", preset_slug) is None:
return _reject(
"OpenRouter preset slugs must be non-empty URL-safe "
"identifiers using only letters, digits, '.', '_', "
"'~', or '-'."
)
return _reject("OpenRouter preset slugs must be non-empty URL-safe identifiers using only "
"letters, digits, '.', '_', '~', or '-'.")
req.preset_suffix = f"{marker}{preset_slug}"
if not preset_base:
return _soft_accept(None)
@@ -185,27 +182,19 @@ def _validate_lmstudio(req: _Request) -> dict[str, Any]:
if models is None:
return _reject(f"Could not reach LM Studio's `/api/v1/models` to validate `{req.requested}`.")
if not models:
return _reject(
f"LM Studio is reachable but no chat-capable models are loaded. "
f"Load `{req.requested}` in LM Studio (Developer tab → Load Model) and try again."
)
return _reject("LM Studio is reachable but no chat-capable models are loaded. "
f"Load `{req.requested}` in LM Studio (Developer tab → Load Model) and try again.")
if req.lookup in set(models):
return _accept()
return _reject(f"Model `{req.requested}` was not found in LM Studio's model listing.")
def _drop_authorization(headers: dict[str, str]) -> None:
for key in tuple(headers):
if key.lower() == "authorization":
del headers[key]
def _ollama_probe_headers(req: _Request) -> dict[str, str]:
"""Headers for the Ollama native probe. Configured ``providers.ollama.extra_headers`` apply only
when the probed endpoint is the configured one (never leak them to a different host). Caller
headers win; a caller ``api_key`` becomes the Authorization header unless the caller sent one."""
from hermes_cli import models as _m
from hermes_cli.models_local import _configured_ollama_base_url
from hermes_cli.models_local import _configured_ollama_base_url, _drop_authorization
configured_base = _configured_ollama_base_url()
configured_allowed = not configured_base or _m._same_ollama_native_root(req.base_url or "", configured_base)
@@ -243,8 +232,7 @@ def _validate_ollama_native(req: _Request) -> Optional[dict[str, Any]]:
f"Note: could not reach this Ollama endpoint's `/api/tags` model listing to validate `{req.requested}`. "
"Hermes will save the model name, but local Ollama model discovery could not verify it."
)
match = _match_in_catalog(req.lookup, models, auto_correct=False,
suggest_label="Similar local Ollama models")
match = _match_in_catalog(req.lookup, models, auto_correct=False, suggest_label="Similar local Ollama models")
if match.exact:
return _accept()
empty_hint = " No models are currently listed by `/api/tags`." if not models else ""
@@ -274,10 +262,8 @@ def _validate_custom(req: _Request) -> dict[str, Any]:
f"{match.suggestion_text}"
)
if probe.get("used_fallback"):
message += (
f"\n Endpoint verification succeeded after trying `{probe.get('resolved_base_url')}`. "
f"Consider saving that as your base URL."
)
message += (f"\n Endpoint verification succeeded after trying `{probe.get('resolved_base_url')}`. "
"Consider saving that as your base URL.")
return _soft_accept(message)
message = (
@@ -285,10 +271,8 @@ def _validate_custom(req: _Request) -> dict[str, Any]:
f"Hermes will still save `{req.requested}`, but the endpoint should expose `/models` for verification."
)
if anthropic_style:
message += (
"\n Many Anthropic-compatible proxies do not implement the Models API "
"(GET /v1/models). The model name has been accepted without verification."
)
message += ("\n Many Anthropic-compatible proxies do not implement the Models API (GET /v1/models). "
"The model name has been accepted without verification.")
if probe.get("suggested_base_url"):
message += f"\n If this server expects `/v1`, try base URL: `{probe.get('suggested_base_url')}`"
# Anthropic-style proxies routinely lack /v1/models, so only they are accepted unverified.
@@ -328,11 +312,9 @@ def _validate_static_catalog(req: _Request) -> Optional[dict[str, Any]]:
return _accept()
base_guess = req.lookup[: -len(CODEX_CONTEXT_VARIANT_SUFFIX)]
return _reject(
f"`{req.requested}` is not a valid large-context variant — "
f"`{base_guess}` enforces the standard 272K window on "
f"Codex, so no `-900k` option exists for it. Pick the "
f"base model, or a verified variant from the `/model` "
f"picker (e.g. `gpt-5.6-sol-900k`)."
f"`{req.requested}` is not a valid large-context variant — `{base_guess}` enforces the "
"standard 272K window on Codex, so no `-900k` option exists for it. Pick the base model, "
"or a verified variant from the `/model` picker (e.g. `gpt-5.6-sol-900k`)."
)
if not catalog:
return None
@@ -349,12 +331,9 @@ def _validate_static_catalog(req: _Request) -> Optional[dict[str, Any]]:
lower = req.lookup.strip().lower()
if prefixes and not any(lower.startswith(p) for p in prefixes):
return _reject(
f"`{req.requested}` doesn't look like a {label} model "
f"and isn't in its listing, so it was not accepted. If it "
f"belongs to another configured provider, switch with "
f"`--provider <slug>` (or select it from the `/model` "
f"picker)."
f"{match.suggestion_text}"
f"`{req.requested}` doesn't look like a {label} model and isn't in its listing, so it was not "
"accepted. If it belongs to another configured provider, switch with `--provider <slug>` "
f"(or select it from the `/model` picker).{match.suggestion_text}"
)
return _soft_accept(
f"Note: `{req.requested}` was not found in the {label} model listing. "
@@ -405,10 +384,9 @@ def _validate_anthropic_messages(req: _Request) -> dict[str, Any]:
models = _m.fetch_api_models(req.api_key, req.base_url, api_mode=req.api_mode)
verdict = _match_in_catalog(req.lookup, models).verdict(req) if models is not None else None
return verdict or _soft_accept(
f"Note: could not verify `{req.requested}` against this endpoint's "
f"model listing. Many Anthropic-compatible proxies do not "
f"implement GET /v1/models. The model name has been accepted "
f"without verification."
f"Note: could not verify `{req.requested}` against this endpoint's model listing. Many "
"Anthropic-compatible proxies do not implement GET /v1/models. The model name has been accepted "
"without verification."
)
@@ -437,12 +415,9 @@ def _validate_live_listing(req: _Request) -> Optional[dict[str, Any]]:
if api_models is None:
return None
if req.normalized == "gemini":
# Gemini's OpenAI-compat listing prefixes ids with "models/"; curated list and user
# input use the bare id, so strip before comparing.
api_models = [
m[len("models/"):] if isinstance(m, str) and m.startswith("models/") else m
for m in api_models
]
# Gemini's OpenAI-compat listing prefixes ids with "models/"; curated list and user input
# use the bare id, so strip before comparing.
api_models = [m[len("models/"):] if isinstance(m, str) and m.startswith("models/") else m for m in api_models]
match = _match_in_catalog(req.lookup, api_models)
if match.exact:
return _accept()
@@ -471,22 +446,15 @@ def _validate_live_listing(req: _Request) -> Optional[dict[str, Any]]:
if not listing_authoritative and _m._model_in_provider_catalog(
(variant_base or req.lookup).lower(), _m._provider_keys(req.normalized)
):
return _accept_with_note(
f"Note: `{req.requested}` was not found in the live /v1/models listing "
f"but exists in the curated catalog — accepted."
)
return _accept_with_note(f"Note: `{req.requested}` was not found in the live /v1/models listing "
"but exists in the curated catalog — accepted.")
# Nous: the Portal's recommended-models feed can list a model before the curated list or the
# docs-hosted manifest catches up; `hermes chat` already accepts those at model-list build
# time, so mirror that source of truth for per-message /model validation.
if req.normalized == "nous" and req.lookup.lower() in _nous_portal_recommended_names():
return _accept_with_note(
f"Note: `{req.requested}` was not found in the live /v1/models "
f"listing but is a current Nous Portal recommendation — accepted."
)
return _reject(
f"Model `{req.requested}` was not found in this provider's model listing."
f"{match.suggestion_text}"
)
return _accept_with_note(f"Note: `{req.requested}` was not found in the live /v1/models listing "
"but is a current Nous Portal recommendation — accepted.")
return _reject(f"Model `{req.requested}` was not found in this provider's model listing.{match.suggestion_text}")
def _validate_bedrock(req: _Request) -> Optional[dict[str, Any]]:
@@ -497,8 +465,7 @@ def _validate_bedrock(req: _Request) -> Optional[dict[str, Any]]:
region = resolve_bedrock_runtime_region()
discovered_ids = {m["id"] for m in discover_bedrock_models(region)}
match = _match_in_catalog(req.requested, list(discovered_ids), auto_correct=False,
suggest_cutoff=0.4)
match = _match_in_catalog(req.requested, list(discovered_ids), auto_correct=False, suggest_cutoff=0.4)
if match.exact:
return _accept()
# Still accept (custom inference profiles / cross-account access), but warn.
@@ -520,10 +487,8 @@ def _validate_catalog_fallback(req: _Request) -> dict[str, Any]:
label = _m._PROVIDER_LABELS.get(req.normalized, req.normalized)
catalog = _static_catalog(req.normalized)
if not catalog:
return _soft_accept(
f"Note: could not reach the {label} API to validate `{req.requested}`. "
f"If the service isn't down, this model may not be valid."
)
return _soft_accept(f"Note: could not reach the {label} API to validate `{req.requested}`. "
"If the service isn't down, this model may not be valid.")
match = _match_in_catalog(req.lookup, catalog, case_insensitive=True)
if match.exact:
return _accept()