fix: remove dedicated user-facing output cap controls

This commit is contained in:
Teknium
2026-09-07 03:32:37 -07:00
parent f4f691c6cc
commit fd3565deec
44 changed files with 296 additions and 350 deletions
+1 -1
View File
@@ -85,7 +85,7 @@ cache break — keep it the only one. Full detail:
`agent/model_metadata.py` holds context lengths and capabilities.
- **Auxiliary (side-LLM) work** — curator, vision, embedding, title generation, session_search,
compression — resolves through `agent/auxiliary_client.py::_resolve_auto_route`; each task can pin
its own `provider/model/base_url/max_tokens/reasoning_effort` under `auxiliary:` in config.yaml.
its own `provider/model/base_url/reasoning_effort` under `auxiliary:` in config.yaml.
- Fallback models and credential pools are resolution-chain code: E2E them with real imports
against a temp `HERMES_HOME`, not mocks (root rubric).
-10
View File
@@ -1684,18 +1684,8 @@ def _resolve_context_length(agent, _agent_cfg, base_url):
except (TypeError, ValueError):
agent._aux_compression_context_length_config = None
# model.max_tokens from config when the caller did not pass one.
_model_cfg = _agent_cfg.get("model", {})
_model_section = _model_cfg if isinstance(_model_cfg, dict) else {}
_config_max_tokens = _model_section.get("max_tokens")
if agent.max_tokens is None and _config_max_tokens is not None:
agent.max_tokens = _positive_int(_config_max_tokens, reject=(bool,))
if agent.max_tokens is None:
_warn_invalid_config_int(
"model.max_tokens in config.yaml", _config_max_tokens,
"must be a positive integer (e.g. 4096)", "provider default",
)
agent._session_init_model_config["max_tokens"] = agent.max_tokens
_config_context_length = _model_section.get("context_length")
if _config_context_length is not None:
+15 -34
View File
@@ -1729,8 +1729,8 @@ class _BedrockCompletionsAdapter:
)
response = call_converse(
region=self._region, model=model, messages=kwargs.get("messages", []), tools=kwargs.get("tools"),
# Omitted/None cap → None so Bedrock uses the model max (no-cap-by-default like
# every other aux wire). Truthiness mirrors the Anthropic shim: explicit 0 = "no cap".
# Converse specifically defaults to the model maximum when omitted.
# Truthiness mirrors the Anthropic shim: explicit 0 means omit.
max_tokens=int(max_tokens) if max_tokens else None, temperature=kwargs.get("temperature"),
top_p=kwargs.get("top_p"), stop_sequences=stop,
)
@@ -3602,8 +3602,6 @@ def _fallback_request_kwargs(
destination.provider, destination.model, fallback_messages,
temperature=temperature, max_tokens=fallback_max_tokens, tools=fallback_tools, timeout=effective_timeout,
extra_body=fallback_extra_body, reasoning_config=reasoning_config, base_url=destination.base_url, task=task)
if apply_fast_lane and fallback_max_tokens is not None and max_tokens is None:
fb_kwargs.update(auxiliary_max_tokens_param(fallback_max_tokens, model=destination.model))
return fb_kwargs
@@ -5547,41 +5545,29 @@ def _get_auxiliary_task_config(task: str) -> Dict[str, Any]:
class CompressionFastLane(NamedTuple):
"""Explicit, non-reasoning compression route safe for a bounded summary."""
"""Explicit, non-reasoning compression route."""
certified_non_reasoning: bool
max_tokens: Optional[int]
reasoning_config: Optional[Dict[str, Any]]
def _fast_lane_config_fields(config: Dict[str, Any]) -> tuple[str, str, bool, Optional[int]]:
"""``(provider, model, non_reasoning, cap)`` from one task config.
``non_reasoning`` only when ``reasoning_effort`` EXPLICITLY disables thinking (unset is NOT
non-reasoning); ``cap`` is a positive int ``max_output_tokens`` or None — booleans are config
drift, never a cap (``int(True) == 1``).
"""
def _fast_lane_config_fields(config: Dict[str, Any]) -> tuple[str, str, bool]:
"""Only explicit reasoning disablement certifies a non-reasoning route."""
from hermes_constants import parse_reasoning_effort
provider = str(config.get("provider") or "").strip().lower()
model = str(config.get("model") or "").strip()
parsed_effort = parse_reasoning_effort(config.get("reasoning_effort"))
non_reasoning = parsed_effort is not None and parsed_effort.get("enabled") is False
raw_cap = config.get("max_output_tokens")
try:
cap = 0 if isinstance(raw_cap, bool) else int(raw_cap or 0)
except (TypeError, ValueError):
cap = 0
return provider, model, non_reasoning, (cap if cap > 0 else None)
return provider, model, non_reasoning
def resolve_compression_fast_lane(
actual_provider: str, actual_model: Optional[str], *, requested_provider: Optional[str] = None,
requested_model: Optional[str] = None, route_config: Optional[Dict[str, Any]] = None,
) -> CompressionFastLane:
"""Certify the opt-in fast lane: capped only when an explicit, operator-certified
non-reasoning provider/model exactly matches the route actually called."""
"""Certify explicit non-reasoning settings only on the matching destination."""
config = route_config if route_config is not None else _get_auxiliary_task_config("compression")
cfg_provider, cfg_model, non_reasoning, cap = _fast_lane_config_fields(config)
cfg_provider, cfg_model, non_reasoning = _fast_lane_config_fields(config)
provider = str(requested_provider or "").strip().lower() or cfg_provider
model = str(requested_model or "").strip() or cfg_model
explicit_route = provider not in {"", "auto"} and model.lower() not in {"", "auto"}
@@ -5589,14 +5575,14 @@ def resolve_compression_fast_lane(
provider_matches = actual_norm == _normalize_aux_provider(provider)
model_matches = str(actual_model or "").strip().lower() == model.lower()
if explicit_route and provider_matches and model_matches and non_reasoning:
return CompressionFastLane(True, cap, {"enabled": False, "effort": "none"})
return CompressionFastLane(False, None, None)
return CompressionFastLane(True, {"enabled": False, "effort": "none"})
return CompressionFastLane(False, None)
def _compression_config_claims_fast_lane(config: Dict[str, Any]) -> bool:
"""Whether task config declares fast-only controls that cannot leak."""
provider, model, non_reasoning, cap = _fast_lane_config_fields(config)
return provider not in {"", "auto"} and model.lower() not in {"", "auto"} and non_reasoning and cap is not None
provider, model, non_reasoning = _fast_lane_config_fields(config)
return provider not in {"", "auto"} and model.lower() not in {"", "auto"} and non_reasoning
def _compression_fast_lane_controls(
@@ -5616,7 +5602,7 @@ def _compression_fast_lane_controls(
body["reasoning"] = lane.reasoning_config
elif _compression_config_claims_fast_lane(leak_guard_config):
body.pop("reasoning", None)
return lane.max_tokens, body
return max_tokens, body
def _get_task_timeout(task: str, default: float = _DEFAULT_AUX_TIMEOUT) -> float:
@@ -5821,7 +5807,7 @@ def _is_gemini_native_route(provider_norm: str, effective_base: str) -> bool:
def _forwards_max_tokens(provider: str, provider_norm: str, model: str, effective_base: str, task: Optional[str]) -> bool:
"""Whether an explicit max_tokens is forwarded on this route.
No default cap elsewhere (omitted = model max; avoids max_completion_tokens / ZAI-vision
No default cap elsewhere (omitted = provider default; avoids max_completion_tokens / ZAI-vision
quirks). Forward only where mandatory or honored: Anthropic Messages wire (400 without it);
NVIDIA NIM (empty choices[] when omitted); MoA reference slots; Gemini native (fixed 65,535
ceiling otherwise); OpenRouter (budgets the FULL window when omitted → 402 on low credit);
@@ -6542,10 +6528,9 @@ def _prepare_aux_request(
)
effective_timeout = _effective_aux_timeout(task, timeout)
request_provider = effective_provider or resolved_provider
fast_compression_cap = None
if not async_mode:
compression_config = _get_auxiliary_task_config("compression") if task == "compression" else {}
fast_compression_cap, effective_extra_body = _compression_fast_lane_controls(
_, effective_extra_body = _compression_fast_lane_controls(
task, actual_provider=request_provider, actual_model=final_model,
requested_provider=provider, requested_model=model, route_config=compression_config,
leak_guard_config=compression_config, max_tokens=max_tokens,
@@ -6567,10 +6552,6 @@ def _prepare_aux_request(
request_provider, final_model, messages, temperature=temperature, max_tokens=max_tokens,
tools=tools, timeout=effective_timeout, extra_body=effective_extra_body,
reasoning_config=reasoning_config, base_url=base_info or resolved_base_url, task=task)
if fast_compression_cap is not None and max_tokens is None:
# The compression route is certified non-reasoning, so a bounded summary is
# intentional; explicit caller caps pass through untouched.
kwargs.update(auxiliary_max_tokens_param(fast_compression_cap, model=final_model))
if extra_headers:
kwargs["extra_headers"] = dict(extra_headers)
# Convert image blocks for Anthropic-compatible endpoints (e.g. MiniMax)
+2 -3
View File
@@ -235,7 +235,7 @@ def _resolve_review_runtime(agent: Any, task_cfg: Optional[Dict[str, Any]] = Non
"provider": rp.get("provider") or task_provider, "model": rp.get("model") or task_model,
**{key: rp.get(key) for key in ("api_key", "base_url", "api_mode", "credential_pool", "command")},
"request_overrides": dict(rp.get("request_overrides") or {}),
"max_tokens": rp.get("max_output_tokens"), "args": list(rp.get("args") or []), "routed": True,
"args": list(rp.get("args") or []), "routed": True,
}
except Exception as e:
logger.debug("background-review aux routing failed (%s); using main model", e)
@@ -813,8 +813,7 @@ def _fork_init_kwargs(agent: Any, rt: Dict[str, Any], routed: bool, max_iteratio
"enabled_toolsets": getattr(agent, "enabled_toolsets", None),
"disabled_toolsets": getattr(agent, "disabled_toolsets", None), "skip_memory": True,
}
if isinstance(rt.get("max_tokens"), int):
kwargs["max_tokens"] = rt["max_tokens"]
if isinstance(rt.get("command"), str) and rt["command"]:
kwargs.update(acp_command=rt["command"], acp_args=rt.get("args") or [])
if not routed:
+3 -15
View File
@@ -1256,7 +1256,7 @@ def _build_anthropic_kwargs(agent, api_messages, tools_for_api, reasoning_config
def _build_bedrock_kwargs(agent, api_messages, tools_for_api):
# Bedrock Converse — the adapter converts messages/tools and calls boto3 directly.
return agent._get_transport().build_kwargs(model=agent.model, messages=api_messages, tools=tools_for_api,
max_tokens=agent.max_tokens or 4096, region=getattr(agent, "_bedrock_region", None) or "us-east-1",
max_tokens=agent.max_tokens, region=getattr(agent, "_bedrock_region", None) or "us-east-1",
guardrail_config=getattr(agent, "_bedrock_guardrail_config", None))
@@ -1292,18 +1292,6 @@ def _build_codex_kwargs(agent, api_messages, tools_for_api, reasoning_config, re
context_management=context_management)
def _anthropic_max_output_for_model(agent):
"""Anthropic-compatible max-output fallback (last resort in build_kwargs, never
overriding an explicit value). Model-gated, not URL-gated: any proxy serving a
Claude/MiniMax/Qwen3 model needs max_tokens (Messages API treats it as
mandatory; proxies that omit it default as low as 4096)."""
with contextlib.suppress(Exception):
from agent.anthropic_adapter import _get_anthropic_max_output, _ANTHROPIC_OUTPUT_LIMITS
model_norm = (agent.model or "").lower().replace(".", "-")
if any(key in model_norm for key in _ANTHROPIC_OUTPUT_LIMITS):
return _get_anthropic_max_output(agent.model)
return None
def _build_chat_completions_kwargs(agent, api_messages, tools_for_api, reasoning_config, request_overrides, cache_scope_id):
transport = agent._get_transport()
@@ -1325,7 +1313,7 @@ def _build_chat_completions_kwargs(agent, api_messages, tools_for_api, reasoning
_fixed_temp = None if _omit_temp else _ft
_prefs = _provider_preferences_for_agent(agent)
_ant_max = _anthropic_max_output_for_model(agent)
_qwen_meta = {"sessionId": agent.session_id or "hermes", "promptId": str(uuid.uuid4())} if _is_qwen else None
_profile = None
with contextlib.suppress(Exception):
@@ -1342,7 +1330,7 @@ def _build_chat_completions_kwargs(agent, api_messages, tools_for_api, reasoning
request_overrides=request_overrides, session_id=getattr(agent, "session_id", None),
cache_scope_id=cache_scope_id, ollama_num_ctx=agent._ollama_num_ctx,
provider_preferences=_prefs or None, openrouter_min_coding_score=agent.openrouter_min_coding_score,
anthropic_max_output=_ant_max, supports_reasoning=agent._supports_reasoning_extra_body(),
supports_reasoning=agent._supports_reasoning_extra_body(),
qwen_session_metadata=_qwen_meta)
if _profile:
# Profiles handle per-provider quirks via hooks fed the context above.
+1 -1
View File
@@ -1023,7 +1023,7 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
result_meta["model"], result_meta["provider"] = model_name, provider or ""
review_agent = None
try:
agent_kwargs: Dict[str, Any] = {"max_tokens": rp["max_output_tokens"]} if isinstance(rp.get("max_output_tokens"), int) else {}
agent_kwargs: Dict[str, Any] = {}
acp_command = rp.get("command")
if isinstance(acp_command, str) and acp_command:
agent_kwargs.update(acp_command=acp_command, acp_args=list(rp.get("args") or []))
+5 -6
View File
@@ -380,15 +380,14 @@ def _run_reference(
messages, slot, runtime, reserve_output_tokens=max_tokens, context_length_cache=context_length_cache,
)
trimmed = _maybe_apply_moa_cache_control(trimmed, _with_cache_disabled(runtime, cache_disabled), cache_ttl=cache_ttl)
# Per-slot max_tokens beats the preset-level reference_max_tokens.
slot_max_tokens = slot.get("max_tokens")
# Copilot gates premium models on request attribution; MoA fan-out serves the
# user's current turn, so mirror the main agent's x-initiator header.
from agent.auxiliary_client import _normalize_aux_provider
is_copilot = _normalize_aux_provider(str(runtime.get("provider") or "")) in ("copilot", "copilot-acp")
response = call_llm(
task="moa_reference", messages=trimmed, temperature=temperature,
max_tokens=slot_max_tokens if slot_max_tokens is not None else max_tokens,
max_tokens=max_tokens,
timeout=reference_timeout, reasoning_config=_slot_reasoning_config(slot),
extra_headers={"x-initiator": "user"} if is_copilot else None, **runtime,
)
@@ -790,8 +789,8 @@ def aggregate_moa_context(
long syntheses). ``agent`` makes the fan-out interruptible.
``reference_max_tokens`` applies ONLY to the reference fan-out — the aggregator's own synthesis call is
never capped, so it always uses its model's own maximum. ``call_llm`` omits the parameter entirely when
it is ``None`` (see its docstring), which also sidesteps providers that reject ``max_tokens`` outright.
not given an advisor budget. Omission uses provider-specific defaults; native protocols may
still require an internal wire limit.
A hardcoded cap on the aggregator call previously truncated long aggregator syntheses (#53580) — passing
``reference_max_tokens`` to both calls here would silently reintroduce that regression.
"""
@@ -1198,7 +1197,7 @@ class MoAChatCompletions:
raw_reference_timeout = preset.get("reference_timeout")
reference_outputs = _run_references_parallel(
reference_models, ref_messages, temperature=_preset_temperature(preset, "reference_temperature"),
max_tokens=preset.get("reference_max_tokens"),
progress_callback=lambda done, total, label: self._emit("moa.progress", refs_done=done, refs_total=total, label=label),
reference_timeout=float(raw_reference_timeout) if raw_reference_timeout else None,
agent=self._agent, late_accounting_sink=self._record_late_reference_accounting,
+2 -2
View File
@@ -511,7 +511,7 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool
# Per-model overrides (config.yaml → model_overrides). Canonical schema (the ONLY key space consumers
# accept): context_window, max_output_tokens, supports_tools, supports_vision, supports_reasoning,
# accept): context_window, supports_tools, supports_vision, supports_reasoning,
# model_family. ``<provider>.<model_id>`` is an explicit partial patch that always wins over the
# catalog. ``<provider>._default`` / top-level ``_default`` are FILL-GAP defaults: they apply ONLY to
# models the catalog does not know and never displace catalog data. Provider keys accept the Hermes
@@ -605,7 +605,7 @@ def _override_to_catalog_shape(override: Dict[str, Any]) -> Tuple[Dict[str, Any]
vision is out-of-band because it maps onto the ``modalities.input`` list rather than a scalar field."""
patch: Dict[str, Any] = {}
limit = {
catalog_key: value for catalog_key, override_key in (("context", "context_window"), ("output", "max_output_tokens"))
catalog_key: value for catalog_key, override_key in (("context", "context_window"),)
if (value := _override_int(override, override_key)) is not None
}
if limit:
+2 -2
View File
@@ -37,11 +37,11 @@ class BedrockTransport(ProviderTransport):
self, model: str, messages: List[Dict[str, Any]],
tools: Optional[List[Dict[str, Any]]] = None, **params,
) -> Dict[str, Any]:
"""Build converse() kwargs; params: max_tokens (4096), temperature, guardrail_config, region ('us-east-1')."""
"""Build Converse kwargs, leaving the optional output limit to the provider."""
from agent.bedrock_adapter import build_converse_kwargs
kwargs = build_converse_kwargs(
model=model, messages=messages, tools=tools, max_tokens=params.get("max_tokens", 4096),
model=model, messages=messages, tools=tools, max_tokens=params.get("max_tokens"),
temperature=params.get("temperature"), guardrail_config=params.get("guardrail_config"),
)
# Sentinel keys for dispatch — agent pops these before the boto3 call
+2 -3
View File
@@ -250,7 +250,7 @@ def _swap_developer_role(sanitized: list, model_lower: str) -> list:
def _apply_max_tokens(api_kwargs: dict, model: str, reasoning_config: Any, params: dict, profile_max: Any = None) -> None:
"""Resolve max_tokens — priority: ephemeral > user > profile default > anthropic_max_output."""
"""Preserve internal task/recovery budgets and provider protocol exceptions."""
max_tokens_fn = params.get("max_tokens_param_fn")
for candidate in (params.get("ephemeral_max_output_tokens"), params.get("max_tokens")):
if candidate is not None and max_tokens_fn:
@@ -258,8 +258,7 @@ def _apply_max_tokens(api_kwargs: dict, model: str, reasoning_config: Any, param
return
if profile_max and max_tokens_fn:
api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, profile_max)))
elif params.get("anthropic_max_output") is not None:
api_kwargs["max_tokens"] = params["anthropic_max_output"]
def _base_kwargs(model: str, sanitized: list, tools: Any, params: dict, profile: Any = None) -> dict[str, Any]:
+1 -1
View File
@@ -57,7 +57,7 @@ def _append_moa_context(agent: Any, api_messages: Any, moa_config: Any, original
aggregator=moa_config.get("aggregator") or {},
temperature=_preset_temperature(moa_config, "reference_temperature"),
aggregator_temperature=_preset_temperature(moa_config, "aggregator_temperature"),
reference_max_tokens=moa_config.get("reference_max_tokens"),
# None = no per-preset override; inherit auxiliary.moa_reference.timeout.
reference_timeout=(
float(moa_config["reference_timeout"])
@@ -424,7 +424,7 @@ describe('ModelSettings MoA preset editor', () => {
aggregator: { provider: 'openrouter', model: 'anthropic/claude-opus-4.8' },
reference_temperature: 0,
aggregator_temperature: 0,
max_tokens: 4096,
enabled: true
}
},
@@ -435,7 +435,7 @@ describe('ModelSettings MoA preset editor', () => {
aggregator: { provider: 'openrouter', model: 'anthropic/claude-opus-4.8' },
reference_temperature: 0,
aggregator_temperature: 0,
max_tokens: 4096,
enabled: true
})
+3 -4
View File
@@ -1437,11 +1437,10 @@ export interface MoaConfigResponse {
aggregator_temperature: number
degraded_reference_policy: 'loud' | 'silent'
enabled: boolean
max_tokens: number
reference_models: MoaModelSlot[]
reference_temperature: number
/** Optional advisor output cap — round-tripped, not edited here. */
reference_max_tokens?: number | null
/** Fan-out cadence (user_turn default | per_iteration | every_n:N) — round-tripped. */
fanout?: string
reference_timeout: number | null
@@ -1451,7 +1450,7 @@ export interface MoaConfigResponse {
aggregator_temperature: number
degraded_reference_policy: 'loud' | 'silent'
enabled: boolean
max_tokens: number
reference_models: MoaModelSlot[]
reference_temperature: number
reference_timeout: number | null
+8 -8
View File
@@ -51,12 +51,12 @@ _RUNNER_FIELDS = (
"batch_size", "run_name", "distribution", "max_iterations", "base_url", "api_key", "model",
"num_workers", "verbose", "ephemeral_system_prompt", "log_prefix_chars", "providers_allowed",
"providers_ignored", "providers_order", "provider_sort", "openrouter_min_coding_score",
"max_tokens", "reasoning_config", "prefill_messages", "max_samples",
"reasoning_config", "prefill_messages", "max_samples",
)
# BatchRunner attributes forwarded verbatim to every AIAgent in the worker config.
_AGENT_PASSTHROUGH = (
"base_url", "api_key", "ephemeral_system_prompt", "providers_allowed", "providers_ignored",
"providers_order", "provider_sort", "openrouter_min_coding_score", "max_tokens",
"providers_order", "provider_sort", "openrouter_min_coding_score",
"reasoning_config", "prefill_messages",
)
@@ -427,7 +427,7 @@ class BatchRunner:
providers_order: List[str] = None,
provider_sort: str = None,
openrouter_min_coding_score: Optional[float] = None,
max_tokens: int = None,
reasoning_config: Dict[str, Any] = None,
prefill_messages: List[Dict[str, Any]] = None,
max_samples: int = None,
@@ -861,7 +861,7 @@ def main(
providers_ignored: str = None,
providers_order: str = None,
provider_sort: str = None,
max_tokens: int = None,
reasoning_effort: str = None,
reasoning_disabled: bool = False,
prefill_messages_file: str = None,
@@ -889,7 +889,7 @@ def main(
providers_ignored (str): Comma-separated list of OpenRouter providers to ignore (e.g. "together,deepinfra")
providers_order (str): Comma-separated list of OpenRouter providers to try in order (e.g. "anthropic,openai,google")
provider_sort (str): Sort providers by "price", "throughput", or "latency" (OpenRouter only)
max_tokens (int): Maximum tokens for model responses (optional, uses model default if not set)
reasoning_effort (str): Reasoning effort: "none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra" (default: "medium")
reasoning_disabled (bool): Completely disable reasoning/thinking tokens (default: False)
prefill_messages_file (str): Path to JSON file containing prefill messages (list of {role, content} dicts)
@@ -905,9 +905,9 @@ def main(
# Use specific distribution
python batch_runner.py --dataset_file=data.jsonl --batch_size=10 --run_name=image_test --distribution=image_gen
# With disabled reasoning and max tokens
# With disabled reasoning
python batch_runner.py --dataset_file=data.jsonl --batch_size=10 --run_name=my_run \\
--reasoning_disabled --max_tokens=128000
--reasoning_disabled
# With prefill messages from file
python batch_runner.py --dataset_file=data.jsonl --batch_size=10 --run_name=my_run \\
@@ -981,7 +981,7 @@ def main(
providers_ignored=_split_csv(providers_ignored),
providers_order=_split_csv(providers_order),
provider_sort=provider_sort,
max_tokens=max_tokens,
reasoning_config=reasoning_config,
prefill_messages=prefill_messages,
max_samples=max_samples,
+2 -8
View File
@@ -111,14 +111,8 @@ model:
#
# context_length: 131072
#
# max_tokens: OUTPUT cap — maximum tokens the model may generate per response.
# Unrelated to how long your conversation history can be.
# The OpenAI-standard name "max_tokens" is a misnomer; Anthropic's native
# API has since renamed it "max_output_tokens" for clarity.
# Leave unset to use the model's native output ceiling (recommended).
# Set only if you want to deliberately limit individual response length.
#
# max_tokens: 8192
# Output-token limits are provider-owned, not user configuration. Native
# protocols requiring a limit receive an internal value from Hermes.
# ── Custom request headers (optional) ─────────────────────────────────────
#
+1 -3
View File
@@ -2669,9 +2669,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix
# A ``moa:<preset>`` model string selects the MoA virtual provider in one shot (parity with
# interactive ``/moa`` and the model picker). See #56828.
_moa_provider_override, self.model = _normalize_moa_model(self.model)
_env_mt = os.environ.get("HERMES_MAX_TOKENS")
_mt = _model_config.get("max_tokens")
self.max_tokens = _int_or(_env_mt, None) if _env_mt else (_mt if isinstance(_mt, int) else None)
if self.model == "": # auto-detect from a local server
_base_url = _model_config.get("base_url") or ""
if base_url_hostname(_base_url) in ("localhost", "127.0.0.1"):
+1 -1
View File
@@ -58,7 +58,7 @@ echo "✅ Done. Log: $LOG_FILE"
#
# --resume Resume from checkpoint if interrupted
# --verbose Enable detailed logging
# --max_tokens=63000 Set max response tokens
# --reasoning_disabled Disable model thinking/reasoning tokens
# --providers_allowed="anthropic,google" Restrict to specific providers
# --prefill_messages_file="configs/prefill.json" Few-shot priming
+107
View File
@@ -0,0 +1,107 @@
"""Zero-cost local SDK wire capture. Fixtures are not vendor inference evidence.
Run from a checkout with HERMES_HOME pointing to a disposable directory.
"""
import json
import os
from pathlib import Path
import threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from openai import OpenAI
from anthropic import Anthropic
captures = []
class Capture(BaseHTTPRequestHandler):
def do_GET(self):
self.send_response(200)
self.send_header("Content-Type", "application/json")
self.end_headers()
self.wfile.write(json.dumps({"data": [{"id": "fixture", "context_length": 131072}]}).encode())
def do_POST(self):
body = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
captures.append({"path": self.path, "body": body})
if "model" not in body:
reply = {}
elif self.path.endswith("messages"):
reply = {"id": "fixture", "type": "message", "role": "assistant", "model": body["model"], "content": [{"type": "text", "text": "LOCAL_CAPTURE_ONLY"}], "stop_reason": "end_turn", "usage": {"input_tokens": 1, "output_tokens": 1}}
else:
reply = {"id": "fixture", "object": "chat.completion", "created": 0, "model": body["model"], "choices": [{"index": 0, "message": {"role": "assistant", "content": "LOCAL_CAPTURE_ONLY"}, "finish_reason": "stop"}], "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}
self.send_response(200)
self.send_header("Content-Type", "application/json")
self.end_headers()
self.wfile.write(json.dumps(reply).encode())
def log_message(self, format, *args):
pass
server = ThreadingHTTPServer(("127.0.0.1", 0), Capture)
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
url = f"http://127.0.0.1:{server.server_port}"
home = Path(os.environ["HERMES_HOME"])
home.mkdir(parents=True, exist_ok=True)
config = {"model": {"default": "fixture", "provider": "fixture-local", "max_tokens": 17}, "providers": {"fixture-local": {"api": url + "/v1", "api_key": "fixture", "max_output_tokens": 19}}}
(home / "config.yaml").write_text(json.dumps(config))
os.environ["HERMES_MAX_TOKENS"] = "13"
from gateway.run import _resolve_runtime_agent_kwargs
from gateway.platforms.api_server import _resolve_request_runtime_agent_kwargs
from hermes_cli.runtime_provider import resolve_runtime_provider
from hermes_cli.moa_config import _normalize_preset
from agent.transports.chat_completions import ChatCompletionsTransport
from agent.transports.anthropic import AnthropicTransport
from agent.transports.bedrock import BedrockTransport
messages = [{"role": "user", "content": "fixture"}]
client = OpenAI(api_key="fixture", base_url=url + "/v1", max_retries=0)
results = {}
for label, runtime in [("gateway", _resolve_runtime_agent_kwargs()), ("api-server", _resolve_request_runtime_agent_kwargs("fixture-local", "fixture")), ("provider", resolve_runtime_provider(requested="fixture-local", target_model="fixture"))]:
cap = runtime.get("max_tokens", runtime.get("max_output_tokens"))
kwargs = ChatCompletionsTransport().build_kwargs("fixture", messages, max_tokens=cap, max_tokens_param_fn=lambda value: {"max_tokens": value})
client.chat.completions.create(**kwargs)
results[label] = captures[-1]
for label, params in [("compatible-claude", {"anthropic_max_output": 65536}), ("internal-budget", {"max_tokens": 43})]:
kwargs = ChatCompletionsTransport().build_kwargs("claude-fixture", messages, max_tokens_param_fn=lambda value: {"max_tokens": value}, **params)
client.chat.completions.create(**kwargs)
results[label] = captures[-1]
from providers import get_provider_profile
from run_agent import AIAgent
agent = AIAgent(model="fixture", provider="custom", base_url=url + "/v1", api_key="fixture", quiet_mode=True, skip_memory=True, skip_context_files=True, enabled_toolsets=[])
actual_kwargs = agent._build_api_kwargs(messages, tools_for_api=[])
client.chat.completions.create(**actual_kwargs)
results["agent-main-custom"] = captures[-1]
kwargs = ChatCompletionsTransport().build_kwargs("fixture", messages, provider_profile=get_provider_profile("custom"), max_tokens_param_fn=lambda value: {"max_tokens": value})
client.chat.completions.create(**kwargs)
results["registered-custom-profile"] = captures[-1]
from agent.auxiliary_client import _compression_fast_lane_controls
from tools.delegate_tool_config import _resolve_delegation_credentials
route = {"provider": "custom", "model": "fixture", "reasoning_effort": "none", "max_output_tokens": 47}
compression_cap, _ = _compression_fast_lane_controls(
"compression", actual_provider="custom", actual_model="fixture", requested_provider="custom",
requested_model="fixture", route_config=route, leak_guard_config=route, max_tokens=None, extra_body={},
)
child_credentials = _resolve_delegation_credentials({"provider": "fixture-local", "model": "fixture"}, agent)
for label, cap in [("compression-config-helper", compression_cap), ("delegation-config-helper", child_credentials.get("max_output_tokens"))]:
kwargs = ChatCompletionsTransport().build_kwargs("fixture", messages, max_tokens=cap, max_tokens_param_fn=lambda value: {"max_tokens": value})
client.chat.completions.create(**kwargs)
results[label] = captures[-1]
kwargs = AnthropicTransport().build_kwargs("claude-sonnet-4-5", messages)
kwargs.pop("__anthropic__", None)
Anthropic(api_key="fixture", base_url=url, max_retries=0).messages.create(**kwargs)
results["native-anthropic"] = captures[-1]
results["bedrock-converse-built-not-sent"] = BedrockTransport().build_kwargs("amazon.nova-pro-v1:0", messages)
results["moa-normalized"] = _normalize_preset({"max_tokens": 23, "reference_max_tokens": 29})
print(json.dumps(results, indent=2, default=str))
server.shutdown()
server.server_close()
thread.join()
if os.environ.get("VERIFY_OUTPUT_CAP_REMOVAL"):
for label in ("gateway", "api-server", "provider", "compatible-claude", "agent-main-custom", "registered-custom-profile", "compression-config-helper", "delegation-config-helper"):
assert "max_tokens" not in results[label]["body"], label
assert results["internal-budget"]["body"]["max_tokens"] == 43
assert results["native-anthropic"]["body"]["max_tokens"] > 0
assert "maxTokens" not in results["bedrock-converse-built-not-sent"].get("inferenceConfig", {})
+3 -16
View File
@@ -251,7 +251,7 @@ _REQUEST_OPTION_MISSING = object()
# vocabulary clamping happens downstream in agent.reasoning_effort.
_REASONING_EFFORTS = frozenset({"none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"})
_RUNTIME_AGENT_OVERRIDE_KEYS = (
"api_key", "base_url", "provider", "api_mode", "command", "args", "credential_pool", "max_tokens")
"api_key", "base_url", "provider", "api_mode", "command", "args", "credential_pool")
def _clean_request_string(value: Any) -> Optional[str]:
@@ -313,24 +313,11 @@ def _resolve_request_runtime_agent_kwargs(provider: str, target_model: Optional[
runtime = resolve_runtime_provider(requested=provider, target_model=target_model)
except Exception as exc:
raise RuntimeError(format_runtime_provider_error(exc)) from exc
model_cfg = _get_model_config()
max_tokens = None
env_max_tokens = os.environ.get("HERMES_MAX_TOKENS")
if env_max_tokens:
with suppress(ValueError, TypeError):
max_tokens = int(env_max_tokens)
elif isinstance(model_cfg, dict):
cfg_max_tokens = model_cfg.get("max_tokens")
if isinstance(cfg_max_tokens, int):
max_tokens = cfg_max_tokens
if max_tokens is None:
runtime_max_tokens = runtime.get("max_output_tokens")
if isinstance(runtime_max_tokens, int) and runtime_max_tokens > 0:
max_tokens = runtime_max_tokens
return {
**{k: runtime.get(k) for k in ("api_key", "base_url", "provider", "api_mode", "command")},
"args": list(runtime.get("args") or []),
"credential_pool": runtime.get("credential_pool"), "max_tokens": max_tokens}
"credential_pool": runtime.get("credential_pool")}
def _request_agent_overrides(
+3 -18
View File
@@ -2180,27 +2180,13 @@ def _resolve_runtime_agent_kwargs() -> dict:
except Exception as exc:
raise RuntimeError(format_runtime_provider_error(exc)) from exc
model_cfg = _get_model_config()
max_tokens = None
_env_mt = os.environ.get("HERMES_MAX_TOKENS")
if _env_mt:
with suppress(ValueError, TypeError):
max_tokens = int(_env_mt)
elif isinstance(model_cfg, dict):
mt = model_cfg.get("max_tokens")
max_tokens = mt if isinstance(mt, int) else None
# Per-provider max_output_tokens applies only when global model.max_tokens is unset (global wins).
if max_tokens is None:
_runtime_mot = runtime.get("max_output_tokens")
if isinstance(_runtime_mot, int) and _runtime_mot > 0:
max_tokens = _runtime_mot
capabilities = runtime.get("capabilities")
capabilities = (
{k: v for k, v in capabilities.items() if isinstance(k, str) and isinstance(v, bool)}
if isinstance(capabilities, dict) else {})
return {**_runtime_agent_kwargs(runtime), "max_tokens": max_tokens, "capabilities": capabilities}
return {**_runtime_agent_kwargs(runtime), "capabilities": capabilities}
def _runtime_agent_kwargs(runtime: dict) -> dict:
@@ -2303,8 +2289,7 @@ def _resolve_runtime_agent_kwargs_for_provider(provider: str) -> dict:
return {
**_runtime_agent_kwargs(runtime),
"request_overrides": dict(runtime.get("request_overrides") or {}),
"capabilities": dict(runtime.get("capabilities") or {}),
"max_tokens": runtime.get("max_output_tokens")}
"capabilities": dict(runtime.get("capabilities") or {})}
def _deep_merge_request_overrides(base: Optional[dict], override: Optional[dict]) -> dict:
@@ -4169,7 +4154,7 @@ class GatewayRunner(
# cached agent or a mid-gateway edit is silently ignored. Add new baked-in settings here.
# _MAX_INTERRUPT_DEPTH = 3 # Cap recursive interrupt handling (#816)
_CACHE_BUSTING_CONFIG_KEYS: tuple = (
("model", "context_length"), ("model", "max_tokens"), ("compression", "enabled"),
("model", "context_length"), ("compression", "enabled"),
("compression", "progress_notices"), ("compression", "threshold"),
("compression", "model_thresholds"), ("compression", "threshold_tokens"),
("compression", "codex_gpt55_autoraise"), ("compression", "codex_app_server_auto"),
+1 -1
View File
@@ -518,7 +518,7 @@ class CLIAgentSetupMixin:
requested_provider=runtime.get("requested_provider"),
api_mode=runtime.get("api_mode"), acp_command=runtime.get("command"),
acp_args=runtime.get("args"), credential_pool=runtime.get("credential_pool"),
max_tokens=self.max_tokens, max_iterations=self.max_turns,
max_iterations=self.max_turns,
run_budget_seconds=getattr(self, "run_budget_seconds", None),
enabled_toolsets=self.enabled_toolsets, disabled_toolsets=self.disabled_toolsets,
verbose_logging=self.verbose, quiet_mode=not self.verbose,
+4 -5
View File
@@ -695,9 +695,8 @@ DEFAULT_CONFIG = {
# OpenAI-compatible request fields. Vision: download_timeout = image HTTP download (s).
"vision": _aux(120, download_timeout=30),
# web_extract and session_search no longer use an aux LLM; leftover blocks in user config
# are ignored. Compression: raise timeout for local models. max_output_tokens is only
# honored with a concrete provider/model AND ``reasoning_effort: none``; 0 = uncapped.
"compression": _aux(120, max_output_tokens=0),
# are ignored. Compression: raise timeout for local models.
"compression": _aux(120),
"skills_hub": _aux(30),
"approval": _aux(30), # classifier — a fast/cheap model is recommended
# /review reviewer: a full subagent on the async delegation rail, credentials resolved like
@@ -1304,7 +1303,7 @@ DEFAULT_CONFIG = {
{"provider": "openrouter", "model": "deepseek/deepseek-v4-pro"},
],
"aggregator": {"provider": "openrouter", "model": "anthropic/claude-opus-4.8"},
"max_tokens": 4096,
"enabled": True,
}
},
@@ -1824,7 +1823,7 @@ DEFAULT_CONFIG = {
# providers: {openrouter: {url: https://example.com/my-curation.json}}.
"providers": {},
},
# Per-model metadata overrides. Fields: context_window, max_output_tokens, supports_tools,
# Per-model metadata overrides. Fields: context_window, supports_tools,
# supports_vision, supports_reasoning, model_family. <provider>.<model_id> wins over
# models.dev/OpenRouter/hardcoded defaults for the fields it sets (chain order in
# agent/model_metadata.py). <provider>._default and top-level _default fill gaps ONLY for models
+3 -10
View File
@@ -152,11 +152,7 @@ def _clean_slot(slot: Any, *, include_enabled: bool = False) -> dict[str, Any] |
effort = _clean_reasoning_effort(slot.get("reasoning_effort"))
if effort:
clean["reasoning_effort"] = effort
# Optional per-slot max_tokens overrides the preset-level reference_max_tokens for this
# advisor; None (default) = no cap.
slot_mt = _coerce_number(slot.get("max_tokens"), int, positive=True)
if slot_mt is not None:
clean["max_tokens"] = slot_mt
if include_enabled:
clean["enabled"] = _coerce_bool(slot.get("enabled"), True)
return clean
@@ -230,10 +226,7 @@ def _normalize_preset(raw: Any) -> dict[str, Any]:
"reference_timeout": _coerce_reference_timeout(raw.get("reference_timeout")),
# Failed-advisor disclosure policy; unknown values fail loud.
"degraded_reference_policy": policy if policy in {"loud", "silent"} else "loud",
"max_tokens": _coerce_number(raw.get("max_tokens"), int, 4096),
# Per-turn cap on each reference ADVISOR (never the acting aggregator). None = uncapped;
# advisor generation dominates MoA latency, so e.g. 600 roughly halves wall time.
"reference_max_tokens": _coerce_number(raw.get("reference_max_tokens"), int, positive=True),
# "user_turn" (default, cheapest): advisors run ONCE per user turn; "per_iteration": every
# tool iteration; "every_n:<N>": first iteration of each turn and every Nth after.
"fanout": _coerce_fanout(raw.get("fanout"))}
@@ -241,7 +234,7 @@ def _normalize_preset(raw: Any) -> dict[str, Any]:
_FLAT_PRESET_KEYS = (
"reference_models", "aggregator", "reference_temperature", "aggregator_temperature",
"reference_timeout", "degraded_reference_policy", "max_tokens", "reference_max_tokens",
"reference_timeout", "degraded_reference_policy",
"fanout", "enabled")
+1 -1
View File
@@ -404,7 +404,7 @@ def resolve_requested_provider(requested: Optional[str] = None) -> str:
from hermes_cli.runtime_provider_custom import ( # noqa: E402,F401
_apply_custom_provider_extras, _custom_provider_request_overrides, _filter_capabilities, _find_custom_identity,
_get_named_custom_provider, _lift_common_custom_fields, _lift_extra_headers, _lift_max_output_tokens,
_get_named_custom_provider, _lift_common_custom_fields, _lift_extra_headers,
_lift_model_capabilities, _normalize_base_url_for_match, _normalize_custom_provider_name, _resolve_named_custom_runtime,
_try_resolve_from_custom_pool, canonical_custom_identity, find_custom_provider_identity,
find_custom_provider_identity_by_model, has_named_custom_provider, is_routable_provider,
+3 -14
View File
@@ -62,16 +62,6 @@ def _lift_model_capabilities(entry: Dict[str, Any], model: Optional[str], result
result["capabilities"] = capabilities
def _lift_max_output_tokens(entry: Dict[str, Any], result: Dict[str, Any]) -> None:
"""``max_output_tokens`` or ``max_tokens`` on a provider entry pins its own output limit;
gateway/CLI map it onto ``AIAgent.max_tokens`` only when top-level ``model.max_tokens`` is
unset, so the documented global key still wins."""
for key in ("max_output_tokens", "max_tokens"):
value = entry.get(key)
if isinstance(value, int) and value > 0:
result["max_output_tokens"] = value
return
def _lift_extra_headers(entry: Dict[str, Any], result: Dict[str, Any]) -> None:
"""Copy a validated ``extra_headers`` dict. SECURITY: values carry credentials — never log."""
@@ -93,7 +83,7 @@ def _lift_common_custom_fields(entry: Dict[str, Any], result: Dict[str, Any], *,
_lift_extra_headers(entry, result)
if api_mode:
result["api_mode"] = api_mode
_lift_max_output_tokens(entry, result)
_lift_model_capabilities(entry, None, result)
@@ -368,7 +358,7 @@ def _custom_provider_request_overrides(custom_provider: Dict[str, Any]) -> Dict[
def _apply_custom_provider_extras(custom_provider: Dict[str, Any], target_model: Optional[str], result: Dict[str, Any]) -> None:
"""Copy model / capabilities / max_output_tokens / extra_headers / request_overrides onto a
"""Copy model / capabilities / extra_headers / request_overrides onto a
resolved custom runtime. An explicit ``target_model`` wins over the provider's configured
default (auxiliary slots / background-review resolve a concrete model and must not fall back to
``default_model``). ``extra_headers`` may carry credentials — NEVER log them."""
@@ -376,8 +366,7 @@ def _apply_custom_provider_extras(custom_provider: Dict[str, Any], target_model:
if model_name:
result["model"] = model_name
_lift_model_capabilities(custom_provider, model_name, result)
if isinstance(custom_provider.get("max_output_tokens"), int):
result["max_output_tokens"] = custom_provider["max_output_tokens"]
if custom_provider.get("extra_headers"):
result["extra_headers"] = dict(custom_provider["extra_headers"])
request_overrides = _custom_provider_request_overrides(custom_provider)
+1 -4
View File
@@ -137,10 +137,8 @@ class MoaPresetPayload(_MoaReferenceControls):
# None = temperature omitted from API calls (provider default), as for single-model agents.
reference_temperature: Optional[float] = None
aggregator_temperature: Optional[float] = None
max_tokens: int = 4096
# Newer per-preset knobs (moa_config._normalize_preset): optional for older clients,
# declared so GET round-trips don't erase them.
reference_max_tokens: Optional[int] = None
fanout: Optional[str] = None
enabled: bool = True
@@ -153,8 +151,7 @@ class MoaConfigPayload(_MoaReferenceControls):
aggregator: MoaModelSlot = MoaModelSlot()
reference_temperature: Optional[float] = None
aggregator_temperature: Optional[float] = None
max_tokens: int = 4096
reference_max_tokens: Optional[int] = None
fanout: Optional[str] = None
enabled: bool = True
profile: Optional[str] = None
+1 -1
View File
@@ -228,7 +228,7 @@ def get_moa_models(profile: Optional[str] = None):
_MOA_PRESET_FIELDS = (
"reference_temperature", "aggregator_temperature", "reference_timeout",
"degraded_reference_policy", "max_tokens", "reference_max_tokens", "fanout", "enabled",
"degraded_reference_policy", "fanout", "enabled",
)
+2 -6
View File
@@ -66,12 +66,8 @@ custom = CustomProfile(
name="custom", aliases=("ollama", "local", "vllm", "llamacpp", "llama.cpp", "llama-cpp"),
env_vars=(), # No fixed key — custom endpoint
base_url="", # User-configured
# Floor only (user model.max_tokens overrides); without it Ollama falls
# back to num_predict=128 and truncates.
# Without this, no max_tokens is sent and Ollama falls back to its internal num_predict=128, truncating
# responses after a few tokens (#39281). This is only a floor used when the user hasn't set
# model.max_tokens — they can override per-model — so we set it generously rather than lowballing it.
default_max_tokens=65536,
# An arbitrary client ceiling can exceed a local server's actual output limit.
# The endpoint owns its generation default.
)
register_provider(custom)
+1 -1
View File
@@ -838,7 +838,7 @@ def test_review_fork_uses_runtime_model_and_output_cap(curator_env, monkeypatch)
assert result["error"] is None
assert captured["model"] == "real-model-id"
assert captured["max_tokens"] == 1234
assert captured["max_tokens"] is None
+6 -10
View File
@@ -30,7 +30,6 @@ def test_explicit_non_reasoning_compression_route_is_certified_and_bounded():
)
assert lane.certified_non_reasoning is True
assert lane.max_tokens == 1400
assert lane.reasoning_config == {"enabled": False, "effort": "none"}
@@ -48,7 +47,6 @@ def test_inherited_auto_or_uncertified_compression_routes_remain_uncapped():
for lane in (inherited, unknown, reasoning):
assert lane.certified_non_reasoning is False
assert lane.max_tokens is None
assert lane.reasoning_config is None
@@ -99,8 +97,8 @@ def test_summary_model_override_is_certified_against_the_effective_model():
requested_model="qwen3:14b",
)
assert override.max_tokens == 1400
assert drifted.max_tokens is None
assert override.certified_non_reasoning is True
assert drifted.certified_non_reasoning is False
def test_compression_latency_records_delayed_first_provider_chunk():
@@ -182,7 +180,7 @@ def test_certified_fast_lane_sends_the_configured_cap_to_its_provider():
) is response
request = client.chat.completions.create.call_args.kwargs
assert request["max_tokens"] == 1400
assert "max_tokens" not in request
assert request["extra_body"]["reasoning"] == {"enabled": False}
@@ -249,7 +247,7 @@ def test_boolean_cap_drift_stays_uncapped_and_preserves_existing_reasoning():
request = client.chat.completions.create.call_args.kwargs
assert "max_tokens" not in request
assert "max_completion_tokens" not in request
assert request["extra_body"]["reasoning"] == {"enabled": False}
assert "reasoning" not in request.get("extra_body", {})
def test_bedrock_converse_ttfp_waits_for_the_nonstreaming_response():
@@ -322,7 +320,7 @@ def test_summary_model_override_cap_uses_the_actual_primary_request():
request = client.chat.completions.create.call_args.kwargs
assert request["model"] == "qwen3:14b"
assert request["max_tokens"] == 1400
assert "max_tokens" not in request
def test_fallback_cap_requires_independent_route_certification():
@@ -373,7 +371,7 @@ def test_fallback_cap_requires_independent_route_certification():
assert "max_tokens" not in uncertified
assert "max_completion_tokens" not in uncertified
assert "reasoning" not in uncertified.get("extra_body", {})
assert certified["max_tokens"] == 900
assert "max_tokens" not in certified
assert certified["extra_body"]["reasoning"] == {
"enabled": False,
"effort": "none",
@@ -392,13 +390,11 @@ def test_reasoning_effort_aliases_certify_like_none():
for alias in ("none", "false", "disabled", False):
lane = _resolve({**base, "reasoning_effort": alias})
assert lane.certified_non_reasoning is True, alias
assert lane.max_tokens == 1400, alias
# Empty/unset (provider default) and real efforts must NOT certify.
for not_disabled in ("", None, "low", "high", True):
lane = _resolve({**base, "reasoning_effort": not_disabled})
assert lane.certified_non_reasoning is False, not_disabled
assert lane.max_tokens is None, not_disabled
def test_timing_hooks_propagate_to_protected_call_worker_thread():
+1 -1
View File
@@ -35,7 +35,7 @@ class TestRunReferenceSlotMaxTokens:
patch("agent.moa_loop._maybe_apply_moa_cache_control", side_effect=lambda msgs, rt, **kwargs: msgs):
_run_reference(slot, [{"role": "user", "content": "hi"}], max_tokens=2000)
assert captured_kwargs.get("max_tokens") == 600
assert captured_kwargs.get("max_tokens") == 2000
def test_slot_max_tokens_absent_falls_back_to_preset(self):
"""When slot has no max_tokens, the preset-level cap is used."""
+1 -1
View File
@@ -1164,7 +1164,7 @@ class TestModelOverrides:
assert info is not None
assert info.family == "llava"
assert info.context_window == 8192
assert info.max_output == 4096
assert info.max_output is None
assert info.tool_call is True
assert info.reasoning is False
@@ -1,97 +0,0 @@
"""Regression tests for max_tokens propagation from config.yaml to AIAgent.
Covers #20741: `model.max_tokens` was silently dropped before reaching the
gateway-spawned agent, so providers without a hardcoded default (OpenRouter
free models, Ollama Cloud, custom OpenAI-compatible endpoints) truncated long
generations with `finish_reason="length"`.
Precedence verified here:
HERMES_MAX_TOKENS env > model.max_tokens > per-provider
max_output_tokens > None
"""
import importlib
import os
import sys
import textwrap
import pytest
@pytest.fixture
def isolated_home(tmp_path, monkeypatch):
"""Isolated HERMES_HOME with a writable config.yaml and a clean module cache.
These tests deliberately re-import ``hermes_cli`` / ``gateway`` so each
config write is read fresh. To avoid leaking that purge into sibling test
files in the same worker (which breaks their import-time mocks), we snapshot
the affected modules and restore them on teardown.
"""
hermes_home = tmp_path / ".hermes"
hermes_home.mkdir()
monkeypatch.setenv("HERMES_HOME", str(hermes_home))
monkeypatch.delenv("HERMES_MAX_TOKENS", raising=False)
_saved = {
k: v
for k, v in sys.modules.items()
if k.startswith(("hermes_cli", "gateway"))
}
def write_cfg(body: str) -> None:
(hermes_home / "config.yaml").write_text(textwrap.dedent(body))
def fresh_gateway():
for mod in list(sys.modules.keys()):
if mod.startswith(("hermes_cli", "gateway")):
del sys.modules[mod]
return importlib.import_module("gateway.run")
try:
yield write_cfg, fresh_gateway
finally:
# Drop anything we (re)imported, then restore the pre-test snapshot so
# the next test file sees the module objects it was loaded with.
for k in list(sys.modules.keys()):
if k.startswith(("hermes_cli", "gateway")):
del sys.modules[k]
sys.modules.update(_saved)
def test_top_level_max_tokens_propagates(isolated_home):
"""model.max_tokens is read into the gateway runtime kwargs (#20741)."""
write_cfg, fresh_gateway = isolated_home
write_cfg(
"""
model:
default: glm-5.1
provider: openrouter
max_tokens: 16384
"""
)
grun = fresh_gateway()
kw = grun._resolve_runtime_agent_kwargs()
assert kw["max_tokens"] == 16384
def test_per_provider_max_output_tokens_fallback(isolated_home):
"""A custom provider's max_output_tokens fills in when no global is set."""
write_cfg, fresh_gateway = isolated_home
write_cfg(
"""
model:
default: glm-5.1
provider: mylocal
providers:
mylocal:
api: http://localhost:11434/v1
api_key: sk-test
default_model: glm-5.1
max_output_tokens: 12000
"""
)
grun = fresh_gateway()
kw = grun._resolve_runtime_agent_kwargs()
assert kw["max_tokens"] == 12000
+64
View File
@@ -0,0 +1,64 @@
"""User configuration cannot impose generation caps; wire budgets remain internal."""
import json
def test_legacy_user_caps_do_not_change_runtime(tmp_path, monkeypatch):
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
monkeypatch.setenv("HERMES_MAX_TOKENS", "13")
config = {
"model": {"default": "fixture", "provider": "local-fixture", "max_tokens": 17},
"providers": {"local-fixture": {"api": "http://127.0.0.1:1/v1", "api_key": "fixture", "max_output_tokens": 19}},
}
(tmp_path / "config.yaml").write_text(json.dumps(config))
from gateway.run import _resolve_runtime_agent_kwargs
from gateway.platforms.api_server import _resolve_request_runtime_agent_kwargs
from hermes_cli.moa_config import _normalize_preset
from agent.models_dev import _override_to_catalog_shape
runtime = _resolve_runtime_agent_kwargs()
assert runtime.get("max_tokens") is None
assert runtime["base_url"].startswith("http://127.0.0.1:1")
request = _resolve_request_runtime_agent_kwargs(provider="local-fixture", target_model="fixture")
assert request.get("max_tokens") is None
preset = _normalize_preset({"max_tokens": 23, "reference_max_tokens": 29})
assert "max_tokens" not in preset and "reference_max_tokens" not in preset
patch, _ = _override_to_catalog_shape({"context_window": 10000, "max_output_tokens": 31})
assert patch["limit"] == {"context": 10000}
from agent.auxiliary_client import _compression_fast_lane_controls
route = {"provider": "custom", "model": "fixture", "reasoning_effort": "none", "max_output_tokens": 47}
cap, body = _compression_fast_lane_controls(
"compression", actual_provider="custom", actual_model="fixture",
requested_provider="custom", requested_model="fixture", route_config=route,
leak_guard_config=route, max_tokens=None, extra_body={},
)
assert cap is None
assert body["reasoning"]["enabled"] is False
def test_optional_wire_caps_omitted_required_and_internal_preserved():
from agent.transports.chat_completions import ChatCompletionsTransport
from agent.transports.bedrock import BedrockTransport
from agent.transports.anthropic import AnthropicTransport
messages = [{"role": "user", "content": "fixture"}]
chat = ChatCompletionsTransport().build_kwargs(
"claude-fixture", messages, anthropic_max_output=65536,
max_tokens_param_fn=lambda value: {"max_tokens": value},
)
assert "max_tokens" not in chat
from providers import get_provider_profile
custom = ChatCompletionsTransport().build_kwargs(
"fixture", messages, provider_profile=get_provider_profile("custom"),
max_tokens_param_fn=lambda value: {"max_tokens": value},
)
assert "max_tokens" not in custom
bedrock = BedrockTransport().build_kwargs("amazon.nova-pro-v1:0", messages)
assert "maxTokens" not in bedrock.get("inferenceConfig", {})
native = AnthropicTransport().build_kwargs("claude-sonnet-4-5", messages)
assert native["max_tokens"] > 0
bounded = ChatCompletionsTransport().build_kwargs(
"fixture", messages, max_tokens=43,
max_tokens_param_fn=lambda value: {"max_tokens": value},
)
assert bounded["max_tokens"] == 43
@@ -73,7 +73,7 @@ def test_routing_to_different_model_marks_routed_and_resolves_credentials():
assert rt["api_key"] == "or-key"
assert rt["credential_pool"] == "routed-pool"
assert rt["request_overrides"] == {"extra_body": {"store": False}}
assert rt["max_tokens"] == 2048
assert rt["max_tokens"] is None
def test_unrouted_runtime_keeps_parent_pool_and_overrides():
@@ -59,7 +59,7 @@ def test_direct_branch_forwards_request_overrides():
"extra_body": {"provider": {"sort": "throughput"}},
}
# Shape parity with the named-provider branch: max_output_tokens present.
assert "max_output_tokens" in creds
assert "max_output_tokens" not in creds
def test_direct_branch_absent_request_overrides_stays_none():
@@ -123,7 +123,7 @@ def test_explicit_merges_over_runtime_on_provider_alongside_base_url(mock_resolv
"provider": {"sort": "throughput"},
},
}
assert creds["max_output_tokens"] == 8192
assert "max_output_tokens" not in creds
# ── Branch 2: named provider (no base_url) ─────────────────────────────────
+3 -3
View File
@@ -165,7 +165,7 @@ def _build_child_agent(
override_api_key: Optional[str] = None,
override_api_mode: Optional[str] = None,
override_request_overrides: Optional[Dict[str, Any]] = None,
override_max_tokens: Optional[int] = None,
# ACP transport overrides from trusted delegation config.
override_acp_command: Optional[str] = None,
override_acp_args: Optional[List[str]] = None,
@@ -209,7 +209,7 @@ def _build_child_agent(
rt = _resolve_child_runtime(
parent_agent, delegation_cfg, parent_api_key, model=model, override_provider=override_provider,
override_base_url=override_base_url, override_api_key=override_api_key, override_api_mode=override_api_mode,
override_max_tokens=override_max_tokens, override_acp_command=override_acp_command,
override_acp_command=override_acp_command,
override_acp_args=override_acp_args,
)
if override_request_overrides is not None:
@@ -359,7 +359,7 @@ def _build_children(
"override_provider": creds["provider"], "override_base_url": creds["base_url"],
"override_api_key": creds["api_key"], "override_api_mode": creds["api_mode"],
"override_request_overrides": creds.get("request_overrides"),
"override_max_tokens": creds.get("max_output_tokens"), "override_acp_command": creds.get("command"),
"override_acp_command": creds.get("command"),
"override_acp_args": creds.get("args"),
}
children = []
+10 -11
View File
@@ -273,11 +273,11 @@ def _require_pinned_command(command: Optional[str], message: str) -> None:
if command and not _shutil.which(command):
raise ValueError(message)
def _credential_bundle(model, provider, base_url, api_key, api_mode, request_overrides, max_output_tokens, **extra) -> dict:
def _credential_bundle(model, provider, base_url, api_key, api_mode, request_overrides, **extra) -> dict:
"""The child credential dict every branch of ``_resolve_delegation_credentials`` returns."""
return {
"model": model, "provider": provider, "base_url": base_url, "api_key": api_key, "api_mode": api_mode,
"request_overrides": request_overrides, "max_output_tokens": max_output_tokens, **extra,
"request_overrides": request_overrides, **extra,
}
def _direct_endpoint_credentials(v: dict, explicit_request_overrides) -> dict:
@@ -301,15 +301,14 @@ def _direct_endpoint_credentials(v: dict, explicit_request_overrides) -> dict:
if v["api_mode"] in _EXPLICIT_API_MODES:
api_mode = v["api_mode"]
# provider configured ALONGSIDE base_url: pull that provider's request personality (request_overrides /
# max_output_tokens) onto the explicit endpoint. Best-effort — a resolution failure only skips the overrides.
request_overrides = max_output_tokens = None
# Preserve the configured provider's request personality on an explicit endpoint.
request_overrides = None
if v["provider"]:
try:
from hermes_cli.runtime_provider import resolve_runtime_provider
runtime = resolve_runtime_provider(requested=v["provider"], target_model=v["model"])
request_overrides = dict(runtime.get("request_overrides") or {}) or None
max_output_tokens = runtime.get("max_output_tokens")
except Exception as exc:
logger.debug(
"delegation.base_url: runtime resolution for provider '%s' failed; proceeding without request_overrides: %s",
@@ -318,7 +317,7 @@ def _direct_endpoint_credentials(v: dict, explicit_request_overrides) -> dict:
# api_key None → inherited from parent in _build_child_agent
return _credential_bundle(
v["model"], provider, v["base_url"], v["api_key"], api_mode,
_merge_request_overrides(request_overrides, explicit_request_overrides), max_output_tokens,
_merge_request_overrides(request_overrides, explicit_request_overrides),
)
def _runtime_provider_credentials(v: dict, explicit_request_overrides) -> dict:
@@ -354,7 +353,7 @@ def _runtime_provider_credentials(v: dict, explicit_request_overrides) -> dict:
configured_provider if runtime.get("provider") == _RUNTIME_PROVIDER_CUSTOM else runtime.get("provider"),
runtime.get("base_url"), api_key, runtime.get("api_mode"),
_merge_request_overrides(runtime.get("request_overrides"), explicit_request_overrides) or {},
runtime.get("max_output_tokens"), command=pinned_command, args=list(runtime.get("args") or []),
command=pinned_command, args=list(runtime.get("args") or []),
)
def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict:
@@ -374,7 +373,7 @@ def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict:
# Pure inherit; explicit request_overrides still merge OVER the parent's.
return _credential_bundle(
values["model"], None, None, None, None,
_merge_request_overrides(getattr(parent_agent, "request_overrides", None), explicit_request_overrides), None,
_merge_request_overrides(getattr(parent_agent, "request_overrides", None), explicit_request_overrides),
)
return _runtime_provider_credentials(values, explicit_request_overrides)
@@ -411,7 +410,7 @@ _NOUS_PROVIDERS = frozenset({"nous", "nous-portal", "nousresearch"})
def _resolve_child_runtime(
parent_agent, delegation_cfg: dict, parent_api_key: Any, *, model: Optional[str], override_provider: Optional[str],
override_base_url: Optional[str], override_api_key: Optional[str], override_api_mode: Optional[str],
override_max_tokens: Optional[int], override_acp_command: Optional[str], override_acp_args: Optional[List[str]],
override_acp_command: Optional[str], override_acp_args: Optional[List[str]],
) -> Dict[str, Any]:
"""Child credentials, transport and routing (config override > parent inherit) as ``AIAgent`` kwargs. Rules that
are easy to break: api_mode is re-derived (not inherited) when the child's provider differs from the parent's
@@ -494,7 +493,7 @@ def _resolve_child_runtime(
}
if not override_provider:
kwargs["provider_data_collection"] = kwargs["provider_data_collection"] or ""
child_max_tokens = override_max_tokens if override_max_tokens is not None else getattr(parent_agent, "max_tokens", None)
child_max_tokens = getattr(parent_agent, "max_tokens", None)
if isinstance(child_max_tokens, int):
kwargs["max_tokens"] = child_max_tokens
return kwargs
+2 -4
View File
@@ -2460,9 +2460,7 @@ export interface MoaConfigResponse {
aggregator_temperature: number;
reference_timeout: number | null;
degraded_reference_policy: "loud" | "silent";
max_tokens: number;
/** Optional advisor output cap — round-tripped, not edited here. */
reference_max_tokens?: number | null;
/** Fan-out cadence (user_turn default | per_iteration | every_n:N) — round-tripped. */
fanout?: string;
enabled: boolean;
@@ -2473,7 +2471,7 @@ export interface MoaConfigResponse {
aggregator_temperature: number;
reference_timeout: number | null;
degraded_reference_policy: "loud" | "silent";
max_tokens: number;
enabled: boolean;
}
+1 -1
View File
@@ -767,7 +767,7 @@ function MoaModelsModal({
aggregator_temperature: draft.aggregator_temperature,
reference_timeout: draft.reference_timeout,
degraded_reference_policy: draft.degraded_reference_policy,
max_tokens: draft.max_tokens,
enabled: draft.enabled,
};
setDraft((prev) => ({
+16 -5
View File
@@ -862,7 +862,7 @@ hermes model
**Tool calling:** Use `--tool-call-parser` with the appropriate parser for your model family: `qwen` (Qwen 2.5), `llama3`, `llama4`, `deepseekv3`, `mistral`, `glm`. Without this flag, tool calls come back as plain text.
:::caution SGLang defaults to 128 max output tokens
If responses seem truncated, add `max_tokens` to your requests or set `--default-max-tokens` on the server. SGLang's default is only 128 tokens per response if not specified in the request.
If responses seem truncated, check the server's generation default and configure it on the server (for example SGLang's `--default-max-tokens`). Hermes does not expose an output-token cap setting.
:::
---
@@ -1131,7 +1131,7 @@ model:
#### Responses get cut off mid-sentence
**Possible causes:**
1. **Low output cap (`max_tokens`) on the server** — SGLang defaults to 128 tokens per response. Set `--default-max-tokens` on the server or configure Hermes with `model.max_tokens` in config.yaml. Note: `max_tokens` controls response length only — it is unrelated to how long your conversation history can be (that is `context_length`).
1. **Low output limit on the server** — configure the server's generation default (for example SGLang's `--default-max-tokens`). Hermes does not expose an output-token cap setting. Response length is distinct from the conversation's context window (`context_length`).
2. **Context exhaustion** — The model filled its context window. Increase `model.context_length` or enable [context compression](/user-guide/configuration#context-compression) in Hermes.
---
@@ -1227,13 +1227,24 @@ model:
### Context Length Detection
:::note Two settings, easy to confuse
:::note Context windows and output limits are different
**`context_length`** is the **total context window** — the combined budget for input *and* output tokens (e.g. 200,000 for Claude Opus 4.6). Hermes uses this to decide when to compress history and to validate API requests.
**`model.max_tokens`** is the **output cap** — the maximum number of tokens the model may generate in a *single response*. It has nothing to do with how long your conversation history can be. The industry-standard name `max_tokens` is a common source of confusion; Anthropic's native API has since renamed it `max_output_tokens` for clarity.
Output limits govern a single generated response, not the conversation history.
Hermes no longer reads `model.max_tokens`, `HERMES_MAX_TOKENS`, provider output-cap
settings, or `model_overrides.*.*.max_output_tokens`. Remove these legacy settings.
Custom OpenAI-compatible endpoints receive no automatic catalog-sized output cap.
Their server defaults apply; these can be lower than the model maximum.
Native Anthropic Messages (including the native Anthropic Bedrock path) requires
`max_tokens`, so Hermes supplies an internal value. Bedrock Converse is a separate
protocol: its optional `inferenceConfig.maxTokens` is omitted by default, which
[AWS documents as the model maximum](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_InferenceConfiguration.html).
Internal bounded tasks and provider-specific protocol requirements remain implementation
details. Omission does not universally select a model's maximum output.
Set `context_length` when auto-detection gets the window size wrong.
Set `model.max_tokens` only when you need to limit how long individual responses can be.
:::
Hermes uses a multi-source resolution chain to detect the correct context window for your model and provider:
@@ -68,7 +68,7 @@ Entries can optionally include:
| `--resume` | `false` | Resume from checkpoint |
| `--verbose` | `false` | Enable verbose logging |
| `--max_samples` | all | Only process first N samples from dataset |
| `--max_tokens` | model default | Maximum tokens per model response |
### Provider Routing (OpenRouter)
+1 -1
View File
@@ -110,7 +110,7 @@ loops:
self_paced_ceiling_seconds: 900 # self-paced max backoff
```
The `--until` judge routes through the `goal_judge` auxiliary task, so `auxiliary.goal_judge.*` overrides (provider, model, max_tokens) apply to loop conditions too.
The `--until` judge routes through the `goal_judge` auxiliary task, so `auxiliary.goal_judge.*` routing overrides (provider, model) apply to loop conditions too.
## `/loop` vs `/goal` vs cron
@@ -90,7 +90,7 @@ moa:
# the same behavior as a single-model Hermes agent.
# reference_temperature: 0.6
# aggregator_temperature: 0.4
max_tokens: 4096
enabled: true
```
@@ -100,37 +100,12 @@ Default preset:
- reference: `openrouter:deepseek/deepseek-v4-pro`
- aggregator / acting model: `openrouter:anthropic/claude-opus-4.8`
### Tuning advisor speed with `reference_max_tokens`
### Advisor output
Each turn, MoA runs the reference models (advisors) in parallel and then the
aggregator acts. Advisor generation is the dominant per-turn latency — turn
wall time correlates strongly with how many tokens the advisors emit, because
the turn waits for the slowest advisor to finish writing. By default advisors
are **uncapped** (`reference_max_tokens` unset), so they may write long,
essay-length advice.
Set `reference_max_tokens` on a preset to cap advisor output and give concise
advice instead. The aggregator only needs the gist of each advisor's
judgement, so a cap (e.g. `600`) measurably cuts per-turn wall time with little
quality impact. It caps **advisors only** — the acting aggregator's output (the
user-visible answer) is never capped.
```yaml
moa:
presets:
fast:
reference_models:
- provider: openrouter
model: anthropic/claude-opus-4.8
- provider: openrouter
model: openai/gpt-5.5
aggregator:
provider: openrouter
model: anthropic/claude-opus-4.8
reference_max_tokens: 600 # concise advice → faster turns
```
Leave it unset (or `0`/blank) to keep the prior uncapped behavior.
MoA uses provider-owned output limits. Preset and per-slot output-token cap
settings are no longer supported. Provider defaults vary; omission does not
always mean the model maximum. Native protocols that require an output limit
receive an internal value from Hermes.
### Advisor cadence with `fanout`