feat(models): DeepSeek V4.1 Flash on the Nous Portal and OpenRouter pickers

Add deepseek/deepseek-v4.1-flash to OPENROUTER_MODELS (Nous list derives from it),
regenerate the docs manifest, and give the slug its own 1M context entry and 600s
reasoning-stale floor — the longest-key-first scan otherwise lands the new slug on
the 128K `deepseek` catch-all and no floor. Live probed on both routes: echoed
model matches, usage.cost billed.
This commit is contained in:
Teknium
2026-09-10 09:11:35 -07:00
parent ee35a4624f
commit 073c57872a
4 changed files with 11 additions and 4 deletions
+1 -1
View File
@@ -343,7 +343,7 @@ DEFAULT_CONTEXT_LENGTHS = {
# (version-less canonical id, 2026-09 Flash refresh) needs a discrete entry or the
# longest-key-first scan falls through to the 128K ``deepseek`` catch-all below.
# https://api-docs.deepseek.com/zh-cn/quick_start/pricing
"deepseek-v4-pro": 1_000_000, "deepseek-v4-flash": 1_000_000, "deepseek-chat": 1_000_000,
"deepseek-v4-pro": 1_000_000, "deepseek-v4.1-flash": 1_000_000, "deepseek-v4-flash": 1_000_000, "deepseek-chat": 1_000_000,
"deepseek-reasoner": 1_000_000, "deepseek-flash": 1_000_000, "deepseek": 128000,
# Meta; Muse Spark family (1.1/1.2/1.3, -contributor(-free), meta/ prefixed) is 1M per OpenRouter,
# models.dev and api.commandcode.ai /models — keep the "muse-spark" prefix (bare "muse" would match
+1 -1
View File
@@ -22,7 +22,7 @@ _REASONING_STALE_TIMEOUT_FLOORS: dict[int, tuple[str, ...]] = {
# DeepSeek R1 / V4 (reasoning_content streamed before final content).
# ``deepseek-flash`` is the version-less canonical Flash id (2026-09 Flash refresh);
# ``deepseek-v4-flash`` still aliases onto it server-side.
"deepseek-r1", "deepseek-reasoner", "deepseek-flash", "deepseek-v4-flash", "deepseek-v4-pro",
"deepseek-r1", "deepseek-reasoner", "deepseek-flash", "deepseek-v4-flash", "deepseek-v4.1-flash", "deepseek-v4-pro",
# OpenAI o-series: each variant enumerated so bare ``o1`` cannot over-match ``olmo-1``.
"o1", "o1-mini", "o1-pro", "o1-preview", "o3", "o3-pro",
# Mythos-class named models (claude-fable-5): 1M ctx + 128K output, a heavier thinking
+1 -1
View File
@@ -35,7 +35,7 @@ OPENROUTER_MODELS: list[tuple[str, str]] = [
"openai/gpt-5.6-terra", "openai/gpt-5.6-terra-pro", "openai/gpt-5.6-luna", "openai/gpt-5.6-luna-pro",
"openai/gpt-5.5", "openai/gpt-5.5-pro", "openai/gpt-5.4-mini", "google/gemini-3.1-pro-preview",
"google/gemini-3.8-flash", "google/gemini-3.7-flash", "x-ai/grok-4.6", "deepseek/deepseek-v4-pro",
"deepseek/deepseek-v4-pro-0813", "deepseek/deepseek-v4-flash-0731",
"deepseek/deepseek-v4-pro-0813", "deepseek/deepseek-v4.1-flash", "deepseek/deepseek-v4-flash-0731",
"qwen/qwen3.8-max-0902", "qwen/qwen3.8-flash", "moonshotai/kimi-k3", "minimax/minimax-m3", "z-ai/glm-5.3",
"z-ai/glm-5.3-flash", "z-ai/glm-5.2", "xiaomi/mimo-v2.5-pro", "tencent/hy4-preview", "tencent/hy3",
"stepfun/step-3.7-flash", "nvidia/nemotron-3-super-120b-a12b", "meta/muse-spark-1.2",
+8 -1
View File
@@ -1,6 +1,6 @@
{
"version": 1,
"updated_at": "2026-09-05T09:33:43Z",
"updated_at": "2026-09-10T16:09:08Z",
"metadata": {
"source": "hermes-agent repo",
"docs": "https://hermes-agent.nousresearch.com/docs/reference/model-catalog"
@@ -128,6 +128,10 @@
"id": "deepseek/deepseek-v4-pro-0813",
"description": "dated snapshot of v4-pro"
},
{
"id": "deepseek/deepseek-v4.1-flash",
"description": ""
},
{
"id": "deepseek/deepseek-v4-flash-0731",
"description": "dated snapshot of v4-flash"
@@ -330,6 +334,9 @@
{
"id": "deepseek/deepseek-v4-pro-0813"
},
{
"id": "deepseek/deepseek-v4.1-flash"
},
{
"id": "deepseek/deepseek-v4-flash-0731"
},