diff --git a/agent/agent_init.py b/agent/agent_init.py index 94ab40b876..5e9549e95c 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -1831,6 +1831,9 @@ def init_agent( except Exception: agent.show_commentary = True + # Window (seconds) for the bounded /fast auto|cold modes (agent.fast_mode). + agent.fast_auto_seconds = (_agent_cfg.get("agent") or {}).get("fast_auto_seconds", 60) + # LM Studio can either be explicitly preloaded through LM Studio's # management API (the historical Hermes behavior) or left to LM Studio's # just-in-time / Auto-Evict chat-completions path. Keep the default diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 5f7467e291..e352f40ca9 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -34,6 +34,7 @@ from agent.error_classifier import ( PROVIDER_STREAM_NON_JSON_ERROR_CODE, ) from agent.errors import EmptyStreamError +from agent.fast_mode import effective_request_overrides from agent.turn_context import substitute_api_content from agent.gemini_native_adapter import is_native_gemini_base_url from agent.model_metadata import is_local_endpoint @@ -1985,6 +1986,10 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non _wire_reasoning_config = _reasoning_config_for_wire(agent) if tools_for_api is None: tools_for_api = agent.tools + # The one place request_overrides are consumed: static /fast values are + # already pinned in agent.request_overrides; auto/cold windows layer the + # fast override here, per request, only while the window is open. + _request_overrides = effective_request_overrides(agent) if agent.api_mode == "anthropic_messages": _transport = agent._get_transport() @@ -2004,7 +2009,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non preserve_dots=agent._anthropic_preserve_dots(), context_length=ctx_len, base_url=getattr(agent, "_anthropic_base_url", None), - fast_mode=(agent.request_overrides or {}).get("speed") == "fast", + fast_mode=_request_overrides.get("speed") == "fast", drop_context_1m_beta=bool(getattr(agent, "_oauth_1m_beta_disabled", False)), ) # Nous Portal reads ``tags`` and ``session_id`` as top-level body fields @@ -2097,7 +2102,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non base_url=agent.base_url, max_tokens=agent.max_tokens, timeout=agent._resolved_api_call_timeout(), - request_overrides=agent.request_overrides, + request_overrides=_request_overrides, provider=getattr(agent, "provider", None), is_github_responses=is_github_responses, is_codex_backend=is_codex_backend, @@ -2249,7 +2254,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non ephemeral_max_output_tokens=_ephemeral_out, max_tokens_param_fn=agent._max_tokens_param, reasoning_config=_wire_reasoning_config, - request_overrides=agent.request_overrides, + request_overrides=_request_overrides, session_id=getattr(agent, "session_id", None), cache_scope_id=_cache_scope_id, provider_profile=_profile, @@ -2282,7 +2287,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non ephemeral_max_output_tokens=_ephemeral_out, max_tokens_param_fn=agent._max_tokens_param, reasoning_config=_wire_reasoning_config, - request_overrides=agent.request_overrides, + request_overrides=_request_overrides, session_id=getattr(agent, "session_id", None), cache_scope_id=_cache_scope_id, model_lower=(agent.model or "").lower(), diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 4c2d5d4949..d404207119 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -41,6 +41,7 @@ from agent.conversation_compression import ( from agent.context_engine import automatic_compaction_status_message from agent.display import KawaiiSpinner from agent.error_classifier import FailoverReason, classify_api_error +from agent.fast_mode import begin_turn as begin_fast_mode_turn from agent.message_metadata import append_message from agent.turn_context import ( PreflightCompressionTimedOut, @@ -2042,6 +2043,7 @@ def run_conversation( agent._last_compaction_in_place = False agent._last_compression_attempt_recorded = False agent._last_compression_attempt_in_place = None + begin_fast_mode_turn(agent, conversation_history) # Adopt any ~/.hermes/.env credential/base-url edits made since the last # turn — a Settings save updates .env but not this worker's client, which diff --git a/agent/fast_mode.py b/agent/fast_mode.py new file mode 100644 index 0000000000..b8121f6295 --- /dev/null +++ b/agent/fast_mode.py @@ -0,0 +1,63 @@ +"""Bounded fast-mode windows (``/fast auto`` and ``/fast cold``). + +``agent.service_tier`` is ``None`` (normal), ``"priority"`` (static fast), +``"auto"`` or ``"cold"``. The static value is pinned into +``agent.request_overrides`` at agent build time; the two bounded modes +instead open a wall-clock window at each user-turn boundary and layer the +provider's fast override onto the request kwargs only while it is open: + +- ``auto`` — every user turn opens a window of ``agent.fast_auto_seconds``. +- ``cold`` — only the first turn of a session (no prior history) opens it. + +Only per-request params (``service_tier`` / ``speed``) vary between requests; +the system prompt, tools, and messages are untouched, so the prompt cache is +preserved across the window boundary. +""" + +from __future__ import annotations + +import time +from typing import Any + +BOUNDED_MODES = frozenset({"auto", "cold"}) +DEFAULT_WINDOW_SECONDS = 60 + + +def begin_turn(agent: Any, conversation_history: Any) -> None: + """Open (or refuse) the fast window at a user-turn boundary.""" + mode = getattr(agent, "service_tier", None) + agent._fast_until = 0.0 + if mode not in BOUNDED_MODES: + return + if mode == "cold" and any( + isinstance(m, dict) and m.get("role") in ("user", "assistant", "tool") + for m in (conversation_history or ()) + ): + return + try: + window = float(getattr(agent, "fast_auto_seconds", DEFAULT_WINDOW_SECONDS)) + except (TypeError, ValueError): + window = DEFAULT_WINDOW_SECONDS + agent._fast_until = time.monotonic() + max(window, 0.0) + + +def effective_request_overrides(agent: Any) -> dict[str, Any]: + """``agent.request_overrides`` plus the fast override while the window is open.""" + overrides = dict(getattr(agent, "request_overrides", None) or {}) + if getattr(agent, "service_tier", None) not in BOUNDED_MODES: + return overrides + if time.monotonic() >= getattr(agent, "_fast_until", 0.0): + return overrides + from hermes_cli.models import resolve_fast_mode_overrides + + base_url = getattr(agent, "base_url", None) + if getattr(agent, "api_mode", None) == "anthropic_messages": + base_url = getattr(agent, "_anthropic_base_url", None) or base_url + fast = resolve_fast_mode_overrides( + getattr(agent, "model", None), + provider=getattr(agent, "provider", None), + base_url=base_url, + ) + if fast: + overrides.update(fast) + return overrides diff --git a/cli-config.yaml.example b/cli-config.yaml.example index 848482adfc..a0954258d9 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -1181,7 +1181,17 @@ agent: # "claude-opus-4.6": "high" # bare model name also works # "deepseek/deepseek-v4-pro": "xhigh" # dots and dashes are interchangeable reasoning_overrides: {} - + + # Fast mode (OpenAI Priority Processing / xAI Grok 4.6 / Anthropic Fast Mode + # on Opus 4.8+). Premium pricing; only sent to first-party endpoints. + # "" / "normal" - off (default) + # "fast" - every request + # "auto" - only the first fast_auto_seconds of every turn + # "cold" - that window on the first turn of a session only + # Also: /fast normal|fast|auto|cold [--global] + service_tier: "" + fast_auto_seconds: 60 + # Custom personalities (use with /personality command). # Built-ins (helpful, concise, technical, creative, teacher, kawaii, catgirl, # pirate, shakespeare, surfer, noir, uwu, philosopher, hype) are always diff --git a/cli.py b/cli.py index 5918f0564a..5797c81318 100644 --- a/cli.py +++ b/cli.py @@ -402,12 +402,14 @@ def _parse_reasoning_config(effort) -> dict | None: def _parse_service_tier_config(raw: str) -> str | None: - """Parse a persisted service-tier preference into a Responses API value.""" + """Parse a persisted fast-mode preference: None, "priority", "auto", or "cold".""" value = str(raw or "").strip().lower() if not value or value in {"normal", "default", "standard", "off", "none"}: return None if value in {"fast", "priority", "on"}: return "priority" + if value in {"auto", "cold"}: + return value logger.warning("Unknown service_tier '%s', ignoring", raw) return None diff --git a/gateway/run.py b/gateway/run.py index 3f2b91d0c7..05c8a59bb8 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -8903,12 +8903,18 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # configured extra_body (chat_template_kwargs, etc.) never reached the # model on the gateway path -- only /fast service-tier overrides did. service_tier = getattr(self, "_service_tier", None) - if not service_tier: + if service_tier != "priority": + # None (normal) or auto/cold — the bounded window is applied per + # request by agent.fast_mode, not pinned into request_overrides. route["request_overrides"] = base_request_overrides return route try: - overrides = resolve_fast_mode_overrides(route["model"]) + overrides = resolve_fast_mode_overrides( + route["model"], + provider=runtime["provider"], + base_url=runtime["base_url"], + ) except Exception: overrides = None # Fast-mode overrides (service_tier / speed) are top-level keys and do @@ -10367,6 +10373,8 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew return None if value in {"fast", "priority", "on"}: return "priority" + if value in {"auto", "cold"}: + return value logger.warning("Unknown service_tier '%s', ignoring", raw) return None diff --git a/gateway/slash_commands.py b/gateway/slash_commands.py index 5493e849dc..e8d84e6dcb 100644 --- a/gateway/slash_commands.py +++ b/gateway/slash_commands.py @@ -4122,6 +4122,9 @@ class GatewaySlashCommandsMixin: tier = None saved_value = "normal" label = t("gateway.fast.label_normal") + elif value in {"auto", "cold"}: + tier = saved_value = value + label = value.upper() else: return t("gateway.fast.unknown_arg", arg=value) self._service_tier = tier @@ -4144,7 +4147,8 @@ class GatewaySlashCommandsMixin: if not args or args == "status": is_fast = self._service_tier == "priority" - status = t("gateway.fast.status_fast") if is_fast else t("gateway.fast.status_normal") + mode = "fast" if is_fast else (self._service_tier or "normal") + status = {"fast": t("gateway.fast.status_fast"), "normal": t("gateway.fast.status_normal")}.get(mode, mode) async def _on_fast_choice(_chat_id: str, value: str) -> str: return _apply_fast_selection(value, persist=persist_global) @@ -4162,7 +4166,17 @@ class GatewaySlashCommandsMixin: { "value": "normal", "label": t("gateway.fast.choice_normal"), - "is_current": not is_fast, + "is_current": mode == "normal", + }, + { + "value": "auto", + "label": t("gateway.fast.choice_auto"), + "is_current": mode == "auto", + }, + { + "value": "cold", + "label": t("gateway.fast.choice_cold"), + "is_current": mode == "cold", }, ], on_choice_selected=_on_fast_choice, diff --git a/hermes_cli/cli_agent_setup_mixin.py b/hermes_cli/cli_agent_setup_mixin.py index ae5ce2a3e5..46eae97b0d 100644 --- a/hermes_cli/cli_agent_setup_mixin.py +++ b/hermes_cli/cli_agent_setup_mixin.py @@ -353,12 +353,18 @@ class CLIAgentSetupMixin: } service_tier = getattr(self, "service_tier", None) - if not service_tier: + if service_tier != "priority": + # None (normal) or auto/cold — the bounded window is applied per + # request by agent.fast_mode, not pinned into request_overrides. route["request_overrides"] = None return route try: - overrides = resolve_fast_mode_overrides(route["model"]) + overrides = resolve_fast_mode_overrides( + route["model"], + provider=runtime["provider"], + base_url=runtime["base_url"], + ) except Exception: overrides = None route["request_overrides"] = overrides diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index 7fe5737319..b78ff6157c 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -3989,9 +3989,9 @@ class CLICommandsMixin: parts = cmd.strip().split(maxsplit=1) if len(parts) < 2 or parts[1].strip().lower() == "status": - status = "fast" if self.service_tier == "priority" else "normal" + status = {"priority": "fast", None: "normal"}.get(self.service_tier, self.service_tier) _cprint(f" {_ACCENT}{feature_name}: {status}{_RST}") - _cprint(f" {_DIM}Usage: /fast [normal|fast|status] [--global]{_RST}") + _cprint(f" {_DIM}Usage: /fast [normal|fast|auto|cold|status] [--global]{_RST}") return arg_tokens = parts[1].strip().lower().split() @@ -4009,9 +4009,13 @@ class CLICommandsMixin: self.service_tier = None saved_value = "normal" label = "NORMAL" + elif arg in {"auto", "cold"}: + self.service_tier = arg + saved_value = arg + label = arg.upper() else: _cprint(f" {_DIM}(._.) Unknown argument: {arg}{_RST}") - _cprint(f" {_DIM}Usage: /fast [normal|fast|status] [--global]{_RST}") + _cprint(f" {_DIM}Usage: /fast [normal|fast|auto|cold|status] [--global]{_RST}") return self.agent = None # Force agent re-init with new service-tier config diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py index 23af46886d..f2c2ea7faf 100644 --- a/hermes_cli/commands.py +++ b/hermes_cli/commands.py @@ -297,9 +297,9 @@ COMMAND_REGISTRY: list[CommandDef] = [ args_hint="[level|show|hide|full|clamp] [--global]", subcommands=("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "show", "hide", "on", "off", "full", "clamp", "--global"), desktop="advanced"), - CommandDef("fast", "Toggle fast mode — OpenAI Priority Processing / Anthropic Fast Mode (Normal/Fast)", "Configuration", - args_hint="[normal|fast|status] [--global]", - subcommands=("normal", "fast", "status", "on", "off", "--global"), + CommandDef("fast", "Fast mode — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)", "Configuration", + args_hint="[normal|fast|auto|cold|status] [--global]", + subcommands=("normal", "fast", "auto", "cold", "status", "on", "off", "--global"), desktop="advanced"), CommandDef("skin", "Show or change the display skin/theme", "Configuration", cli_only=True, args_hint="[name]", argument_mode="options"), diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index a43ac5b09d..7bce1e561a 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -151,7 +151,10 @@ DEFAULT_CONFIG = { # leaves the budget untouched. "cost_threshold_usd": 0.25, }, + # Fast mode: "" / "normal" (off), "fast" (always), "auto" (first + # fast_auto_seconds of every turn), "cold" (first turn of a session only). "service_tier": "", + "fast_auto_seconds": 60, # Tool-use enforcement: injects system prompt guidance that tells the # model to actually call tools instead of describing intended actions. # Values: "auto" (default — applies to gpt/codex models), true/false diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 2c5bc9652d..7968a3f4d9 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -3993,7 +3993,36 @@ def _is_anthropic_fast_model(model_id: Optional[str]) -> bool: return any(v in base for v in ("opus-4-8", "opus-4.8", "opus-5")) -def resolve_fast_mode_overrides(model_id: Optional[str]) -> dict[str, Any] | None: +def _fast_mode_route_supported( + model_id: Optional[str], provider: Optional[str], base_url: Optional[str] +) -> bool: + """Only the first-party endpoint that bills for fast mode may receive its params. + + OpenRouter, Nous, Copilot, Azure, Bedrock, and custom base_urls either + strip ``service_tier``/``speed`` (charging nothing) or 400 on them. + """ + from urllib.parse import urlparse + + from agent.model_metadata import is_grok_46_family + + if _is_anthropic_fast_model(model_id): + allowed = {"anthropic": "api.anthropic.com"} + elif is_grok_46_family(str(model_id or "")): + allowed = {"xai": "api.x.ai"} + else: + allowed = {"openai": "api.openai.com", "openai-codex": "chatgpt.com"} + if provider and normalize_provider(provider) not in allowed: + return False + host = (urlparse(str(base_url or "")).hostname or "").lower() + return not host or host in allowed.values() + + +def resolve_fast_mode_overrides( + model_id: Optional[str], + *, + provider: Optional[str] = None, + base_url: Optional[str] = None, +) -> dict[str, Any] | None: """Return request_overrides for fast/priority mode, or None if unsupported. Returns provider-appropriate overrides: @@ -4001,12 +4030,21 @@ def resolve_fast_mode_overrides(model_id: Optional[str]) -> dict[str, Any] | Non - Anthropic models: ``{"speed": "fast"}`` (Anthropic Fast Mode beta) - Grok 4.6: ``{"service_tier": "priority"}`` (xAI Priority Processing) + When ``provider``/``base_url`` are given the result is also gated on the + route (see ``_fast_mode_route_supported``) so proxies never see the + params. This is the single fast-mode gate for static ``/fast fast`` and + the bounded ``auto``/``cold`` windows in ``agent.fast_mode``. + The overrides are injected into the API request kwargs by - ``_build_api_kwargs`` in run_agent.py — each API path handles its own - keys (service_tier for OpenAI/Codex, speed for Anthropic Messages). + ``build_api_kwargs`` — each API path handles its own keys + (service_tier for OpenAI/Codex, speed for Anthropic Messages). """ if not model_supports_fast_mode(model_id): return None + if (provider or base_url) and not _fast_mode_route_supported( + model_id, provider, base_url + ): + return None if _is_anthropic_fast_model(model_id): return {"speed": "fast"} return {"service_tier": "priority"} diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index a2bce1d852..9433f6f16d 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -1429,8 +1429,8 @@ _SCHEMA_OVERRIDES: Dict[str, Dict[str, Any]] = { }, "agent.service_tier": { "type": "select", - "description": "API service tier (OpenAI/Anthropic)", - "options": ["", "auto", "default", "flex"], + "description": "Fast mode: fast = always, auto = first N seconds of each turn, cold = first turn only", + "options": ["", "normal", "fast", "auto", "cold"], }, "delegation.reasoning_effort": { "type": "select", diff --git a/locales/af.yaml b/locales/af.yaml index 21806156ab..363563e9a4 100644 --- a/locales/af.yaml +++ b/locales/af.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nHuidige modus: `{mode}`\\n\\nKies \\'n opsie:" choice_fast: "fast — Priority Processing aan" choice_normal: "normal — standaardverwerking" + choice_auto: "auto — vinnig vir die eerste sekondes van elke beurt" + choice_cold: "cold — vinnig slegs vir die eerste beurt van 'n sessie" footer: status: "📎 Looptyd-voetstuk: **{state}**\nVelde: `{fields}`\nPlatform: `{platform}`" diff --git a/locales/ar.yaml b/locales/ar.yaml index 1f68628a3b..abdc371ead 100644 --- a/locales/ar.yaml +++ b/locales/ar.yaml @@ -160,6 +160,8 @@ gateway: picker_title: "⚡ **المعالجة ذات الأولوية**\n\nالوضع الحالي: `{mode}`\n\nاختر خيارًا:" choice_fast: "fast — المعالجة ذات الأولوية مُفعّلة" choice_normal: "normal — المعالجة القياسية" + choice_auto: "auto — سريع في الثواني الأولى من كل دور" + choice_cold: "cold — سريع في الدور الأول من الجلسة فقط" footer: status: "📎 تذييل التشغيل: **{state}**\nالحقول: `{fields}`\nالمنصّة: `{platform}`" diff --git a/locales/de.yaml b/locales/de.yaml index bc00bfe32f..d6e1528088 100644 --- a/locales/de.yaml +++ b/locales/de.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nAktueller Modus: `{mode}`\\n\\nOption wählen:" choice_fast: "fast — Priority Processing an" choice_normal: "normal — Standardverarbeitung" + choice_auto: "auto — schnell in den ersten Sekunden jedes Zugs" + choice_cold: "cold — schnell nur im ersten Zug einer Sitzung" footer: status: "📎 Laufzeit-Fußzeile: **{state}**\nFelder: `{fields}`\nPlattform: `{platform}`" diff --git a/locales/en.yaml b/locales/en.yaml index 2adac023f2..9b06ae1e96 100644 --- a/locales/en.yaml +++ b/locales/en.yaml @@ -141,8 +141,8 @@ gateway: fast: not_supported: "⚡ /fast is only available for OpenAI models that support Priority Processing." - status: "⚡ Priority Processing\n\nCurrent mode: `{mode}`\n\n_Usage:_ `/fast `" - unknown_arg: "⚠️ Unknown argument: `{arg}`\n\n**Valid options:** normal, fast, status" + status: "⚡ Priority Processing\n\nCurrent mode: `{mode}`\n\n_Usage:_ `/fast `" + unknown_arg: "⚠️ Unknown argument: `{arg}`\n\n**Valid options:** normal, fast, auto, cold, status" saved: "⚡ ✓ Priority Processing: **{label}** (saved to config)\n_(takes effect on next message)_" session_only: "⚡ ✓ Priority Processing: **{label}** (this session only)" label_fast: "FAST" @@ -152,6 +152,8 @@ gateway: picker_title: "⚡ **Priority Processing**\n\nCurrent mode: `{mode}`\n\nPick an option:" choice_fast: "fast — Priority Processing on" choice_normal: "normal — standard processing" + choice_auto: "auto — fast for the first seconds of every turn" + choice_cold: "cold — fast for the first turn of a session only" footer: status: "📎 Runtime footer: **{state}**\nFields: `{fields}`\nPlatform: `{platform}`" diff --git a/locales/es.yaml b/locales/es.yaml index 6b06a52afb..06cd2e9e23 100644 --- a/locales/es.yaml +++ b/locales/es.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nModo actual: `{mode}`\\n\\nElige una opción:" choice_fast: "fast — Priority Processing activado" choice_normal: "normal — procesamiento estándar" + choice_auto: "auto — rápido en los primeros segundos de cada turno" + choice_cold: "cold — rápido solo en el primer turno de una sesión" footer: status: "📎 Pie de ejecución: **{state}**\nCampos: `{fields}`\nPlataforma: `{platform}`" diff --git a/locales/fr.yaml b/locales/fr.yaml index 4ce9760969..4f1faa6cbf 100644 --- a/locales/fr.yaml +++ b/locales/fr.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nMode actuel : `{mode}`\\n\\nChoisissez une option :" choice_fast: "fast — Priority Processing activé" choice_normal: "normal — traitement standard" + choice_auto: "auto — rapide pendant les premières secondes de chaque tour" + choice_cold: "cold — rapide uniquement au premier tour d'une session" footer: status: "📎 Pied de page d'exécution : **{state}**\nChamps : `{fields}`\nPlateforme : `{platform}`" diff --git a/locales/ga.yaml b/locales/ga.yaml index 92ef5363ea..843dc5a1c2 100644 --- a/locales/ga.yaml +++ b/locales/ga.yaml @@ -141,6 +141,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nMód reatha: `{mode}`\\n\\nRoghnaigh rogha:" choice_fast: "fast — Priority Processing ar siúl" choice_normal: "normal — gnáthphróiseáil" + choice_auto: "auto — tapa do na chéad soicindí de gach seal" + choice_cold: "cold — tapa don chéad seal de sheisiún amháin" footer: status: "📎 Buntásc rite: **{state}**\nRéimsí: `{fields}`\nArdán: `{platform}`" diff --git a/locales/hu.yaml b/locales/hu.yaml index b8feb1b994..d134d4372f 100644 --- a/locales/hu.yaml +++ b/locales/hu.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nJelenlegi mód: `{mode}`\\n\\nVálassz egy opciót:" choice_fast: "fast — Priority Processing bekapcsolva" choice_normal: "normal — normál feldolgozás" + choice_auto: "auto — gyors minden kör első másodperceiben" + choice_cold: "cold — gyors csak a munkamenet első körében" footer: status: "📎 Futási idejű lábléc: **{state}**\nMezők: `{fields}`\nPlatform: `{platform}`" diff --git a/locales/it.yaml b/locales/it.yaml index 758be5d8a7..a3480541bc 100644 --- a/locales/it.yaml +++ b/locales/it.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nModalità attuale: `{mode}`\\n\\nScegli un\\'opzione:" choice_fast: "fast — Priority Processing attivo" choice_normal: "normal — elaborazione standard" + choice_auto: "auto — veloce nei primi secondi di ogni turno" + choice_cold: "cold — veloce solo nel primo turno di una sessione" footer: status: "📎 Footer di runtime: **{state}**\nCampi: `{fields}`\nPiattaforma: `{platform}`" diff --git a/locales/ja.yaml b/locales/ja.yaml index 28b41682aa..b691daf93d 100644 --- a/locales/ja.yaml +++ b/locales/ja.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\n現在のモード: `{mode}`\\n\\nオプションを選択:" choice_fast: "fast — Priority Processing オン" choice_normal: "normal — 標準処理" + choice_auto: "auto — 各ターンの最初の数秒間だけ高速" + choice_cold: "cold — セッションの最初のターンのみ高速" footer: status: "📎 ランタイムフッター: **{state}**\nフィールド: `{fields}`\nプラットフォーム: `{platform}`" diff --git a/locales/ko.yaml b/locales/ko.yaml index ecf58bbc68..f7f4f25a06 100644 --- a/locales/ko.yaml +++ b/locales/ko.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\n현재 모드: `{mode}`\\n\\n옵션을 선택하세요:" choice_fast: "fast — Priority Processing 켜기" choice_normal: "normal — 표준 처리" + choice_auto: "auto — 매 턴의 처음 몇 초 동안 빠름" + choice_cold: "cold — 세션의 첫 턴에만 빠름" footer: status: "📎 런타임 푸터: **{state}**\n필드: `{fields}`\n플랫폼: `{platform}`" diff --git a/locales/pt.yaml b/locales/pt.yaml index 1ac1fd4b00..f18100340e 100644 --- a/locales/pt.yaml +++ b/locales/pt.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nModo atual: `{mode}`\\n\\nEscolha uma opção:" choice_fast: "fast — Priority Processing ativado" choice_normal: "normal — processamento padrão" + choice_auto: "auto — rápido nos primeiros segundos de cada turno" + choice_cold: "cold — rápido apenas no primeiro turno de uma sessão" footer: status: "📎 Rodapé de execução: **{state}**\nCampos: `{fields}`\nPlataforma: `{platform}`" diff --git a/locales/ru.yaml b/locales/ru.yaml index 51c892e02e..7fd2c285f2 100644 --- a/locales/ru.yaml +++ b/locales/ru.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nТекущий режим: `{mode}`\\n\\nВыберите вариант:" choice_fast: "fast — Priority Processing включён" choice_normal: "normal — стандартная обработка" + choice_auto: "auto — быстро в первые секунды каждого хода" + choice_cold: "cold — быстро только на первом ходе сессии" footer: status: "📎 Нижний колонтитул среды выполнения: **{state}**\nПоля: `{fields}`\nПлатформа: `{platform}`" diff --git a/locales/tr.yaml b/locales/tr.yaml index a88b1d0586..7af791fb74 100644 --- a/locales/tr.yaml +++ b/locales/tr.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nMevcut mod: `{mode}`\\n\\nBir seçenek seçin:" choice_fast: "fast — Priority Processing açık" choice_normal: "normal — standart işleme" + choice_auto: "auto — her turun ilk saniyelerinde hızlı" + choice_cold: "cold — yalnızca oturumun ilk turunda hızlı" footer: status: "📎 Çalışma zamanı altbilgisi: **{state}**\nAlanlar: `{fields}`\nPlatform: `{platform}`" diff --git a/locales/uk.yaml b/locales/uk.yaml index 730a052cd5..5e63f0bdad 100644 --- a/locales/uk.yaml +++ b/locales/uk.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\nПоточний режим: `{mode}`\\n\\nОберіть варіант:" choice_fast: "fast — Priority Processing увімкнено" choice_normal: "normal — стандартна обробка" + choice_auto: "auto — швидко в перші секунди кожного ходу" + choice_cold: "cold — швидко лише на першому ході сесії" footer: status: "📎 Нижній колонтитул середовища: **{state}**\nПоля: `{fields}`\nПлатформа: `{platform}`" diff --git a/locales/zh-hant.yaml b/locales/zh-hant.yaml index 9468fbba1c..b9091c2aa2 100644 --- a/locales/zh-hant.yaml +++ b/locales/zh-hant.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **Priority Processing**\\n\\n目前模式:`{mode}`\\n\\n請選擇:" choice_fast: "fast — 開啟 Priority Processing" choice_normal: "normal — 標準處理" + choice_auto: "auto — 每輪的前幾秒快速" + choice_cold: "cold — 僅會話的第一輪快速" footer: status: "📎 執行階段頁尾:**{state}**\n欄位:`{fields}`\n平台:`{platform}`" diff --git a/locales/zh.yaml b/locales/zh.yaml index f659de9a24..dde25d0e82 100644 --- a/locales/zh.yaml +++ b/locales/zh.yaml @@ -137,6 +137,8 @@ gateway: picker_title: "⚡ **优先处理**\\n\\n当前模式:`{mode}`\\n\\n请选择:" choice_fast: "fast — 开启优先处理" choice_normal: "normal — 标准处理" + choice_auto: "auto — 每轮的前几秒快速" + choice_cold: "cold — 仅会话的第一轮快速" footer: status: "📎 运行时页脚:**{state}**\n字段:`{fields}`\n平台:`{platform}`" diff --git a/tests/agent/test_fast_mode_auto.py b/tests/agent/test_fast_mode_auto.py new file mode 100644 index 0000000000..a9af81a8df --- /dev/null +++ b/tests/agent/test_fast_mode_auto.py @@ -0,0 +1,143 @@ +"""Bounded /fast auto|cold windows and the shared route-aware gate.""" + +from types import SimpleNamespace + +from agent import fast_mode + + +def _agent(**kw): + base = dict( + service_tier="auto", + model="gpt-5.4", + provider="openai", + base_url="https://api.openai.com/v1", + api_mode="chat_completions", + request_overrides={"extra_body": {"keep": 1}}, + fast_auto_seconds=60, + ) + base.update(kw) + return SimpleNamespace(**base) + + +def test_bounded_fast_window_policy(monkeypatch): + clock = [1000.0] + monkeypatch.setattr(fast_mode.time, "monotonic", lambda: clock[0]) + + # auto: window open -> fast override layered over existing overrides + agent = _agent() + fast_mode.begin_turn(agent, conversation_history=[]) + assert fast_mode.effective_request_overrides(agent) == { + "extra_body": {"keep": 1}, + "service_tier": "priority", + } + assert agent.request_overrides == {"extra_body": {"keep": 1}} # never mutated + + # window expired -> override absent + clock[0] += 61 + assert fast_mode.effective_request_overrides(agent) == {"extra_body": {"keep": 1}} + + # auto re-opens on the next turn + fast_mode.begin_turn(agent, conversation_history=[{"role": "user", "content": "x"}]) + assert "service_tier" in fast_mode.effective_request_overrides(agent) + + # cold: prior history -> no window at all + cold = _agent(service_tier="cold") + fast_mode.begin_turn(cold, conversation_history=[{"role": "user", "content": "x"}]) + assert "service_tier" not in fast_mode.effective_request_overrides(cold) + fast_mode.begin_turn(cold, conversation_history=None) + assert fast_mode.effective_request_overrides(cold)["service_tier"] == "priority" + + # Anthropic route uses the speed param + anth = _agent( + service_tier="auto", + model="claude-opus-5", + provider="anthropic", + base_url="https://api.anthropic.com", + api_mode="anthropic_messages", + ) + fast_mode.begin_turn(anth, conversation_history=[]) + assert fast_mode.effective_request_overrides(anth)["speed"] == "fast" + + # unsupported routes never get fast params, in auto or static mode + from hermes_cli.models import resolve_fast_mode_overrides + + for provider, base_url in ( + ("openrouter", "https://openrouter.ai/api/v1"), + ("nous", "https://inference-api.nousresearch.com/v1"), + ("copilot", "https://api.githubcopilot.com"), + ("azure", "https://foo.openai.azure.com"), + ("custom", "http://10.0.0.1:8000/v1"), + ("openai", "https://proxy.example.com/v1"), + ): + proxied = _agent(provider=provider, base_url=base_url) + fast_mode.begin_turn(proxied, conversation_history=[]) + assert "service_tier" not in fast_mode.effective_request_overrides(proxied), provider + assert resolve_fast_mode_overrides("gpt-5.4", provider=provider, base_url=base_url) is None + assert resolve_fast_mode_overrides( + "claude-opus-5", provider="bedrock", base_url="https://bedrock-runtime.us-east-1.amazonaws.com" + ) is None + # first-party routes (and the legacy model-only call) still resolve + assert resolve_fast_mode_overrides("gpt-5.4", provider="openai-codex", base_url="https://chatgpt.com/backend-api/codex") + assert resolve_fast_mode_overrides("grok-4.6", provider="xai", base_url="https://api.x.ai/v1") + assert resolve_fast_mode_overrides("gpt-5.4") == {"service_tier": "priority"} + + # normal / static modes are untouched by the window logic + static = _agent(service_tier="priority", request_overrides={"service_tier": "priority"}) + fast_mode.begin_turn(static, conversation_history=[]) + assert fast_mode.effective_request_overrides(static) == {"service_tier": "priority"} + off = _agent(service_tier=None) + fast_mode.begin_turn(off, conversation_history=[]) + assert fast_mode.effective_request_overrides(off) == {"extra_body": {"keep": 1}} + + +def test_fast_auto_and_cold_parse_and_slash_command(monkeypatch): + import hermes_cli.config as config_mod + + if not hasattr(config_mod, "save_env_value_secure"): + config_mod.save_env_value_secure = lambda key, value: {"success": True} + import cli as cli_mod + from gateway.run import GatewayRunner + from hermes_cli.commands import COMMAND_REGISTRY + from hermes_cli.config import DEFAULT_CONFIG + + # config parsing: CLI, gateway, TUI all accept auto/cold; default stays off + for raw, expected in (("auto", "auto"), ("COLD", "cold"), ("fast", "priority"), ("", None), ("bogus", None)): + assert cli_mod._parse_service_tier_config(raw) == expected + monkeypatch.setattr( + "gateway.run._load_gateway_runtime_config", lambda: {"agent": {"service_tier": raw}} + ) + assert GatewayRunner._load_service_tier() == expected + assert DEFAULT_CONFIG["agent"]["service_tier"] == "" + assert DEFAULT_CONFIG["agent"]["fast_auto_seconds"] == 60 + + # /fast auto — session-scoped, agent rebuilt, status reports the mode + fast_cmd = next(c for c in COMMAND_REGISTRY if c.name == "fast") + assert {"auto", "cold"} <= set(fast_cmd.subcommands) + printed = [] + monkeypatch.setattr(cli_mod, "_cprint", lambda *a, **k: printed.append(" ".join(map(str, a)))) + monkeypatch.setattr(cli_mod, "save_config_value", lambda *a, **k: (_ for _ in ()).throw(AssertionError("no config write"))) + stub = SimpleNamespace( + service_tier=None, model="gpt-5.4", agent=object(), _fast_command_available=lambda: True + ) + cli_mod.HermesCLI._handle_fast_command(stub, "/fast auto") + assert stub.service_tier == "auto" + assert stub.agent is None + cli_mod.HermesCLI._handle_fast_command(stub, "/fast status") + assert any("auto" in line for line in printed) + cli_mod.HermesCLI._handle_fast_command(stub, "/fast cold") + assert stub.service_tier == "cold" + + # auto/cold do NOT pin a static override into the turn route + route_stub = SimpleNamespace( + model="gpt-5.4", api_key="k", base_url="https://api.openai.com/v1", provider="openai", + api_mode="chat_completions", acp_command=None, acp_args=[], _credential_pool=None, + service_tier="auto", + ) + assert cli_mod.HermesCLI._resolve_turn_agent_config(route_stub, "hi")["request_overrides"] is None + route_stub.service_tier = "priority" + assert cli_mod.HermesCLI._resolve_turn_agent_config(route_stub, "hi")["request_overrides"] == { + "service_tier": "priority" + } + route_stub.base_url = "https://openrouter.ai/api/v1" + route_stub.provider = "openrouter" + assert cli_mod.HermesCLI._resolve_turn_agent_config(route_stub, "hi")["request_overrides"] is None diff --git a/tests/cli/test_fast_command.py b/tests/cli/test_fast_command.py index 23170bc300..203dfe329d 100644 --- a/tests/cli/test_fast_command.py +++ b/tests/cli/test_fast_command.py @@ -159,8 +159,8 @@ class TestFastModeRouting(unittest.TestCase): stub = SimpleNamespace( model="gpt-5.4", api_key="primary-key", - base_url="https://openrouter.ai/api/v1", - provider="openrouter", + base_url="https://api.openai.com/v1", + provider="openai", api_mode="chat_completions", acp_command=None, acp_args=[], @@ -171,11 +171,16 @@ class TestFastModeRouting(unittest.TestCase): route = cli_mod.HermesCLI._resolve_turn_agent_config(stub, "hi") # Provider should NOT have changed - assert route["runtime"]["provider"] == "openrouter" + assert route["runtime"]["provider"] == "openai" assert route["runtime"]["api_mode"] == "chat_completions" # But request_overrides should be set assert route["request_overrides"] == {"service_tier": "priority"} + # Proxied routes (OpenRouter etc.) strip/400 on the param — never sent. + stub.base_url = "https://openrouter.ai/api/v1" + stub.provider = "openrouter" + assert cli_mod.HermesCLI._resolve_turn_agent_config(stub, "hi")["request_overrides"] is None + def test_turn_route_keeps_primary_runtime_when_model_has_no_fast_backend(self): cli_mod = _import_cli() stub = SimpleNamespace( diff --git a/tests/gateway/test_choice_picker.py b/tests/gateway/test_choice_picker.py index a2c9a52961..c8e6712ec0 100644 --- a/tests/gateway/test_choice_picker.py +++ b/tests/gateway/test_choice_picker.py @@ -126,7 +126,7 @@ class TestFastChoicePicker: assert result is None values = [c["value"] for c in adapter.calls[0]["choices"]] - assert values == ["fast", "normal"] + assert values == ["fast", "normal", "auto", "cold"] @pytest.mark.asyncio async def test_fast_picker_selection_is_session_scoped(self, tmp_path, monkeypatch): diff --git a/tests/gateway/test_fast_command.py b/tests/gateway/test_fast_command.py index c714b76e84..b8792ecce4 100644 --- a/tests/gateway/test_fast_command.py +++ b/tests/gateway/test_fast_command.py @@ -109,8 +109,8 @@ def test_turn_route_injects_priority_processing_without_changing_runtime(): runner._service_tier = "priority" runtime_kwargs = { "api_key": "***", - "base_url": "https://openrouter.ai/api/v1", - "provider": "openrouter", + "base_url": "https://api.openai.com/v1", + "provider": "openai", "api_mode": "chat_completions", "command": None, "args": [], @@ -119,10 +119,15 @@ def test_turn_route_injects_priority_processing_without_changing_runtime(): route = gateway_run.GatewayRunner._resolve_turn_agent_config(runner, "hi", "gpt-5.4", runtime_kwargs) - assert route["runtime"]["provider"] == "openrouter" + assert route["runtime"]["provider"] == "openai" assert route["runtime"]["api_mode"] == "chat_completions" assert route["request_overrides"] == {"service_tier": "priority"} + # Proxied routes never receive the param (OpenRouter strips it / others 400). + runtime_kwargs.update(base_url="https://openrouter.ai/api/v1", provider="openrouter") + route = gateway_run.GatewayRunner._resolve_turn_agent_config(runner, "hi", "gpt-5.4", runtime_kwargs) + assert route["request_overrides"] == {} + @pytest.mark.asyncio async def test_handle_fast_command_global_flag_persists_config(monkeypatch, tmp_path): diff --git a/tests/gateway/test_turn_request_overrides.py b/tests/gateway/test_turn_request_overrides.py index c985125176..bd5da602d8 100644 --- a/tests/gateway/test_turn_request_overrides.py +++ b/tests/gateway/test_turn_request_overrides.py @@ -53,7 +53,7 @@ def test_provider_request_overrides_merged_under_fast_mode(monkeypatch): """/fast active: provider extra_body AND the service-tier marker both survive.""" monkeypatch.setattr( "hermes_cli.models.resolve_fast_mode_overrides", - lambda model_id: {"service_tier": "priority"}, + lambda model_id, **_route: {"service_tier": "priority"}, ) runner = _runner(service_tier="priority") rk = _runtime_kwargs(request_overrides=PROVIDER_OVERRIDES) diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index fcb059da82..724efd7d45 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -8504,7 +8504,7 @@ def test_config_set_fast_updates_live_agent_session_scoped(monkeypatch): monkeypatch.setattr(server, "_emit", lambda *args: emits.append(args)) monkeypatch.setattr( "hermes_cli.models.resolve_fast_mode_overrides", - lambda _model_id: {"service_tier": "priority"}, + lambda _model_id, **_route: {"service_tier": "priority"}, ) try: @@ -8583,7 +8583,7 @@ def test_config_set_fast_rejects_unsupported_model(monkeypatch): ) monkeypatch.setattr( "hermes_cli.models.resolve_fast_mode_overrides", - lambda _model_id: None, + lambda _model_id, **_route: None, ) try: diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 9a519659eb..b50ee9d265 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -6113,6 +6113,8 @@ def _load_service_tier() -> str | None: return None if raw in {"fast", "priority", "on"}: return "priority" + if raw in {"auto", "cold"}: + return raw return None @@ -14644,19 +14646,20 @@ def _(rid, params: dict) -> dict: raw = str(value or "").strip().lower() agent = session.get("agent") if session else None if agent is not None: - current_fast = getattr(agent, "service_tier", None) == "priority" + current_tier = getattr(agent, "service_tier", None) elif session is not None and session.get("create_service_tier_override") is not None: # Pre-build session with a pinned tier (desktop draft pick or an # earlier session-scoped toggle) — report/toggle from the pin, not # the global default. - current_fast = session["create_service_tier_override"] == "priority" + current_tier = session["create_service_tier_override"] or None else: - current_fast = _load_service_tier() == "priority" + current_tier = _load_service_tier() + current_fast = current_tier == "priority" if raw in {"status"}: return _ok( rid, - {"key": key, "value": "fast" if current_fast else "normal"}, + {"key": key, "value": {"priority": "fast", None: "normal"}.get(current_tier, current_tier)}, ) if raw in {"", "toggle"}: @@ -14665,6 +14668,8 @@ def _(rid, params: dict) -> dict: nv = "fast" elif raw in {"normal", "off"}: nv = "normal" + elif raw in {"auto", "cold"}: + nv = raw else: return _err(rid, 4002, f"unknown fast mode: {value}") @@ -14690,7 +14695,11 @@ def _(rid, params: dict) -> dict: 4002, "fast mode is not available without a selected model", ) - overrides = resolve_fast_mode_overrides(target_model) + overrides = resolve_fast_mode_overrides( + target_model, + provider=getattr(agent, "provider", None), + base_url=getattr(agent, "base_url", None), + ) if overrides is None: return _err( rid, @@ -14707,13 +14716,11 @@ def _(rid, params: dict) -> dict: # build ("switch one session, switches everywhere"). Pin the # create override so lazily-built sessions and rebuilds (/new, # deferred resume) keep the choice; "" pins normal explicitly. - session["create_service_tier_override"] = ( - "priority" if nv == "fast" else "" - ) + session["create_service_tier_override"] = {"fast": "priority", "normal": ""}.get(nv, nv) else: _write_config_key("agent.service_tier", nv) if agent is not None: - agent.service_tier = "priority" if nv == "fast" else None + agent.service_tier = {"fast": "priority", "normal": None}.get(nv, nv) current_overrides = dict(getattr(agent, "request_overrides", {}) or {}) current_overrides.pop("service_tier", None) current_overrides.pop("speed", None) @@ -16897,6 +16904,8 @@ def _mirror_slash_side_effects(sid: str, session: dict, command: str) -> str: agent.service_tier = "priority" elif mode in {"normal", "off"}: agent.service_tier = None + elif mode in {"auto", "cold"}: + agent.service_tier = mode _emit("session.info", sid, _session_info(agent, session)) elif name == "reload-mcp" and agent and hasattr(agent, "reload_mcp_tools"): agent.reload_mcp_tools() diff --git a/website/docs/reference/slash-commands.md b/website/docs/reference/slash-commands.md index 41dd223b7f..f7a281d44f 100644 --- a/website/docs/reference/slash-commands.md +++ b/website/docs/reference/slash-commands.md @@ -81,7 +81,7 @@ Type `/` in the CLI to open the autocomplete menu. Built-in commands are case-in | `/personality` | Set a predefined personality. `/personality none` (or `default` / `neutral`) clears the overlay and returns to base behavior. | | `/verbose` | Cycle tool progress display: off → new → all → verbose. Can be [enabled for messaging](#notes) via config. | | `/focus [on\|off\|status]` | Toggle **focus view** — a display-only reduced-output mode showing just your prompt and the final response. Composes with `/verbose`: turning it on snaps tool progress to `off` and remembers your previous mode, and `/focus off` restores it. Each turn ends with a dim recovery line (`⋯ 7 tool lines hidden · /focus off to show`) and a persistent `◉ focus` badge sits in the status bar so you always know you're in the reduced view. Nothing is sent differently to the model — detail is hidden, never discarded. | -| `/fast [normal\|fast\|status]` | Toggle fast mode — OpenAI Priority Processing / Anthropic Fast Mode. Options: `normal`, `fast`, `status`. | +| `/fast [normal\|fast\|auto\|cold\|status]` | Fast mode — OpenAI Priority Processing / Anthropic Fast Mode. `fast` = every request; `auto` = only requests in the first `agent.fast_auto_seconds` (default 60s) of each turn; `cold` = that same window on the first turn of a session only. Default `normal` (off). See [Fast mode](../user-guide/configuration.md#fast-mode). | | `/reasoning [level\|show\|hide\|full\|clamp] [--global]` | Manage reasoning effort and display. Levels include `none` / `minimal` / `low` / `medium` / `high` / `xhigh` / `max` / `ultra`. `show` / `hide` (or `on` / `off`) toggle reasoning display; `full` and `clamp` adjust how reasoning is shown. `--global` persists effort to config. | | `/skin` | Show or change the display skin/theme | | `/export [profile] [-o out.tar.gz]` | **CLI only.** Pack a profile into a shareable `.tar.gz` — skills, memory, persona, crons, plugins, settings, and (from the desktop) themes and layout. Credentials (`auth.json`, `.env`) are stripped. Defaults to the active profile and `.tar.gz` in the current directory. Same archive as `hermes profile export`; for a versioned, updatable share use a [profile distribution](../user-guide/profile-distributions.md) instead. | @@ -246,7 +246,7 @@ The messaging gateway supports the following built-in commands inside Telegram, | `/model [provider:model]` | Show or change the model. Supports provider switches (`/model zai:glm-5`), custom endpoints (`/model custom:model`), named custom providers (`/model custom:local:qwen`), auto-detect (`/model custom`), and user-defined aliases (`/model fav`, `/model grok` — see [Custom model aliases](#custom-model-aliases)). Use `--global` to persist the change to config.yaml. **Note:** `/model` can only switch between already-configured providers. To add a new provider or set up API keys, use `hermes model` from your terminal (outside the chat session). **Cost note:** a mid-session model switch resets the prompt cache (the cache key includes the model), so the next message re-reads the whole conversation at full input price. | | `/codex-runtime [auto\|codex_app_server\|on\|off]` | Toggle the optional [Codex app-server runtime](../user-guide/features/codex-app-server-runtime). Persists to `model.openai_runtime` in config.yaml and evicts the cached agent so the next message picks up the new runtime. Effective on next session. | | `/personality [name]` | Set a personality overlay for the session. `/personality none` (or `default` / `neutral`) clears it. | -| `/fast [normal\|fast\|status]` | Toggle fast mode — OpenAI Priority Processing / Anthropic Fast Mode. | +| `/fast [normal\|fast\|auto\|cold\|status]` | Fast mode — OpenAI Priority Processing / Anthropic Fast Mode. `auto`/`cold` open a bounded fast window per turn / per session. | | `/retry` | Retry the last message. | | `/undo` | Remove the last exchange. | | `/sethome` (alias: `/set-home`) | Mark the current chat as the platform home channel for deliveries. | diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 95c0a1c030..2e2778760d 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -1701,6 +1701,27 @@ There is no `hermes config set` support for `reasoning_overrides` keys — edit The override applies automatically everywhere: CLI startup, messaging gateway, Desktop/TUI, cron jobs, `/model` mid-session switches, and fallback model activation. +## Fast Mode + +Fast mode asks the provider for faster output at a premium price: OpenAI [Priority Processing](https://openai.com/api-priority-processing/) (`service_tier: priority`), xAI Priority Processing on Grok 4.6, and Anthropic [Fast Mode](https://platform.claude.com/docs/en/build-with-claude/fast-mode) (`speed: fast`, Opus 4.8 / Opus 5 only). It is **off by default**. + +```yaml +agent: + service_tier: "" # "" / normal | fast | auto | cold + fast_auto_seconds: 60 # window for auto / cold +``` + +| Mode | When fast params are sent | Use it for | +|------|---------------------------|------------| +| `normal` (default, `""`) | Never | Cheapest; standard latency | +| `fast` | Every request | Long interactive sessions where you always want speed | +| `auto` | Requests in the first `fast_auto_seconds` of **every** turn | Snappy first reply; long tool loops fall back to standard pricing | +| `cold` | Same window, but only on the **first turn** of a session (no prior history) | Fast onboarding reply, standard pricing afterwards | + +`/fast normal|fast|auto|cold` switches the mode for the session; add `--global` to persist to `config.yaml`. `/fast` alone shows the current mode. + +**Cost note:** both providers bill fast requests at a multiplier on standard rates (Anthropic: $10 / $50 per MTok in/out on Opus 4.8 and Opus 5), stacking with prompt-cache pricing. `auto`/`cold` bound that premium to the window only. Fast params are only sent to the first-party endpoint that supports them (`api.openai.com` / Codex subscription, `api.anthropic.com`, `api.x.ai`); OpenRouter, Nous Portal, Copilot, Azure, Bedrock, and custom `base_url` routes never receive them in any mode. Only the per-request parameter changes between requests — the system prompt, tools, and messages stay byte-identical, so the prompt cache survives the window boundary. + ## Tool-Use Enforcement Some models occasionally describe intended actions as text instead of making tool calls ("I would run the tests..." instead of actually calling the terminal). Tool-use enforcement injects system prompt guidance that steers the model back to actually calling tools.