From fd3565deecb7ddcc1faf500c1cbd8ae9f76d7483 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Mon, 7 Sep 2026 03:32:37 -0700 Subject: [PATCH] fix: remove dedicated user-facing output cap controls --- agent/AGENTS.md | 2 +- agent/agent_init.py | 10 -- agent/auxiliary_client.py | 49 +++----- agent/background_review.py | 5 +- agent/chat_completion_helpers.py | 18 +-- agent/curator.py | 2 +- agent/moa_loop.py | 11 +- agent/models_dev.py | 4 +- agent/transports/bedrock.py | 4 +- agent/transports/chat_completions.py | 5 +- agent/turn_request_assembly.py | 2 +- .../src/app/settings/model-settings.test.tsx | 4 +- apps/desktop/src/types/hermes.ts | 7 +- batch_runner.py | 16 +-- cli-config.yaml.example | 10 +- cli.py | 4 +- datagen-config-examples/run_browser_tasks.sh | 2 +- evals/output_caps_local_capture.py | 107 ++++++++++++++++++ gateway/platforms/api_server.py | 19 +--- gateway/run.py | 21 +--- hermes_cli/cli_agent_setup_mixin.py | 2 +- hermes_cli/config_defaults.py | 9 +- hermes_cli/moa_config.py | 13 +-- hermes_cli/runtime_provider.py | 2 +- hermes_cli/runtime_provider_custom.py | 17 +-- hermes_cli/web_models.py | 5 +- hermes_cli/web_routers/models.py | 2 +- plugins/model-providers/custom/__init__.py | 8 +- tests/agent/test_curator.py | 2 +- tests/agent/test_fast_compression_lane.py | 16 +-- tests/agent/test_moa_slot_max_tokens.py | 2 +- tests/agent/test_models_dev.py | 2 +- tests/gateway/test_max_tokens_propagation.py | 97 ---------------- tests/gateway/test_output_caps_removed.py | 64 +++++++++++ .../test_background_review_cost_controls.py | 2 +- .../tools/test_delegate_request_overrides.py | 4 +- tools/delegate_tool.py | 6 +- tools/delegate_tool_config.py | 21 ++-- web/src/lib/api.ts | 6 +- web/src/pages/ModelsPage.tsx | 2 +- website/docs/integrations/providers.md | 21 +++- .../user-guide/features/batch-processing.md | 2 +- website/docs/user-guide/features/loops.md | 2 +- .../user-guide/features/mixture-of-agents.md | 37 +----- 44 files changed, 296 insertions(+), 350 deletions(-) create mode 100644 evals/output_caps_local_capture.py delete mode 100644 tests/gateway/test_max_tokens_propagation.py create mode 100644 tests/gateway/test_output_caps_removed.py diff --git a/agent/AGENTS.md b/agent/AGENTS.md index 7fb81c5a66..e1f162fa77 100644 --- a/agent/AGENTS.md +++ b/agent/AGENTS.md @@ -85,7 +85,7 @@ cache break — keep it the only one. Full detail: `agent/model_metadata.py` holds context lengths and capabilities. - **Auxiliary (side-LLM) work** — curator, vision, embedding, title generation, session_search, compression — resolves through `agent/auxiliary_client.py::_resolve_auto_route`; each task can pin - its own `provider/model/base_url/max_tokens/reasoning_effort` under `auxiliary:` in config.yaml. + its own `provider/model/base_url/reasoning_effort` under `auxiliary:` in config.yaml. - Fallback models and credential pools are resolution-chain code: E2E them with real imports against a temp `HERMES_HOME`, not mocks (root rubric). diff --git a/agent/agent_init.py b/agent/agent_init.py index 562dfcc328..63e1dbc919 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -1684,18 +1684,8 @@ def _resolve_context_length(agent, _agent_cfg, base_url): except (TypeError, ValueError): agent._aux_compression_context_length_config = None - # model.max_tokens from config when the caller did not pass one. _model_cfg = _agent_cfg.get("model", {}) _model_section = _model_cfg if isinstance(_model_cfg, dict) else {} - _config_max_tokens = _model_section.get("max_tokens") - if agent.max_tokens is None and _config_max_tokens is not None: - agent.max_tokens = _positive_int(_config_max_tokens, reject=(bool,)) - if agent.max_tokens is None: - _warn_invalid_config_int( - "model.max_tokens in config.yaml", _config_max_tokens, - "must be a positive integer (e.g. 4096)", "provider default", - ) - agent._session_init_model_config["max_tokens"] = agent.max_tokens _config_context_length = _model_section.get("context_length") if _config_context_length is not None: diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 04487ccb07..34a12bf2cc 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1729,8 +1729,8 @@ class _BedrockCompletionsAdapter: ) response = call_converse( region=self._region, model=model, messages=kwargs.get("messages", []), tools=kwargs.get("tools"), - # Omitted/None cap → None so Bedrock uses the model max (no-cap-by-default like - # every other aux wire). Truthiness mirrors the Anthropic shim: explicit 0 = "no cap". + # Converse specifically defaults to the model maximum when omitted. + # Truthiness mirrors the Anthropic shim: explicit 0 means omit. max_tokens=int(max_tokens) if max_tokens else None, temperature=kwargs.get("temperature"), top_p=kwargs.get("top_p"), stop_sequences=stop, ) @@ -3602,8 +3602,6 @@ def _fallback_request_kwargs( destination.provider, destination.model, fallback_messages, temperature=temperature, max_tokens=fallback_max_tokens, tools=fallback_tools, timeout=effective_timeout, extra_body=fallback_extra_body, reasoning_config=reasoning_config, base_url=destination.base_url, task=task) - if apply_fast_lane and fallback_max_tokens is not None and max_tokens is None: - fb_kwargs.update(auxiliary_max_tokens_param(fallback_max_tokens, model=destination.model)) return fb_kwargs @@ -5547,41 +5545,29 @@ def _get_auxiliary_task_config(task: str) -> Dict[str, Any]: class CompressionFastLane(NamedTuple): - """Explicit, non-reasoning compression route safe for a bounded summary.""" + """Explicit, non-reasoning compression route.""" certified_non_reasoning: bool - max_tokens: Optional[int] reasoning_config: Optional[Dict[str, Any]] -def _fast_lane_config_fields(config: Dict[str, Any]) -> tuple[str, str, bool, Optional[int]]: - """``(provider, model, non_reasoning, cap)`` from one task config. - - ``non_reasoning`` only when ``reasoning_effort`` EXPLICITLY disables thinking (unset is NOT - non-reasoning); ``cap`` is a positive int ``max_output_tokens`` or None — booleans are config - drift, never a cap (``int(True) == 1``). - """ +def _fast_lane_config_fields(config: Dict[str, Any]) -> tuple[str, str, bool]: + """Only explicit reasoning disablement certifies a non-reasoning route.""" from hermes_constants import parse_reasoning_effort provider = str(config.get("provider") or "").strip().lower() model = str(config.get("model") or "").strip() parsed_effort = parse_reasoning_effort(config.get("reasoning_effort")) non_reasoning = parsed_effort is not None and parsed_effort.get("enabled") is False - raw_cap = config.get("max_output_tokens") - try: - cap = 0 if isinstance(raw_cap, bool) else int(raw_cap or 0) - except (TypeError, ValueError): - cap = 0 - return provider, model, non_reasoning, (cap if cap > 0 else None) + return provider, model, non_reasoning def resolve_compression_fast_lane( actual_provider: str, actual_model: Optional[str], *, requested_provider: Optional[str] = None, requested_model: Optional[str] = None, route_config: Optional[Dict[str, Any]] = None, ) -> CompressionFastLane: - """Certify the opt-in fast lane: capped only when an explicit, operator-certified - non-reasoning provider/model exactly matches the route actually called.""" + """Certify explicit non-reasoning settings only on the matching destination.""" config = route_config if route_config is not None else _get_auxiliary_task_config("compression") - cfg_provider, cfg_model, non_reasoning, cap = _fast_lane_config_fields(config) + cfg_provider, cfg_model, non_reasoning = _fast_lane_config_fields(config) provider = str(requested_provider or "").strip().lower() or cfg_provider model = str(requested_model or "").strip() or cfg_model explicit_route = provider not in {"", "auto"} and model.lower() not in {"", "auto"} @@ -5589,14 +5575,14 @@ def resolve_compression_fast_lane( provider_matches = actual_norm == _normalize_aux_provider(provider) model_matches = str(actual_model or "").strip().lower() == model.lower() if explicit_route and provider_matches and model_matches and non_reasoning: - return CompressionFastLane(True, cap, {"enabled": False, "effort": "none"}) - return CompressionFastLane(False, None, None) + return CompressionFastLane(True, {"enabled": False, "effort": "none"}) + return CompressionFastLane(False, None) def _compression_config_claims_fast_lane(config: Dict[str, Any]) -> bool: """Whether task config declares fast-only controls that cannot leak.""" - provider, model, non_reasoning, cap = _fast_lane_config_fields(config) - return provider not in {"", "auto"} and model.lower() not in {"", "auto"} and non_reasoning and cap is not None + provider, model, non_reasoning = _fast_lane_config_fields(config) + return provider not in {"", "auto"} and model.lower() not in {"", "auto"} and non_reasoning def _compression_fast_lane_controls( @@ -5616,7 +5602,7 @@ def _compression_fast_lane_controls( body["reasoning"] = lane.reasoning_config elif _compression_config_claims_fast_lane(leak_guard_config): body.pop("reasoning", None) - return lane.max_tokens, body + return max_tokens, body def _get_task_timeout(task: str, default: float = _DEFAULT_AUX_TIMEOUT) -> float: @@ -5821,7 +5807,7 @@ def _is_gemini_native_route(provider_norm: str, effective_base: str) -> bool: def _forwards_max_tokens(provider: str, provider_norm: str, model: str, effective_base: str, task: Optional[str]) -> bool: """Whether an explicit max_tokens is forwarded on this route. - No default cap elsewhere (omitted = model max; avoids max_completion_tokens / ZAI-vision + No default cap elsewhere (omitted = provider default; avoids max_completion_tokens / ZAI-vision quirks). Forward only where mandatory or honored: Anthropic Messages wire (400 without it); NVIDIA NIM (empty choices[] when omitted); MoA reference slots; Gemini native (fixed 65,535 ceiling otherwise); OpenRouter (budgets the FULL window when omitted → 402 on low credit); @@ -6542,10 +6528,9 @@ def _prepare_aux_request( ) effective_timeout = _effective_aux_timeout(task, timeout) request_provider = effective_provider or resolved_provider - fast_compression_cap = None if not async_mode: compression_config = _get_auxiliary_task_config("compression") if task == "compression" else {} - fast_compression_cap, effective_extra_body = _compression_fast_lane_controls( + _, effective_extra_body = _compression_fast_lane_controls( task, actual_provider=request_provider, actual_model=final_model, requested_provider=provider, requested_model=model, route_config=compression_config, leak_guard_config=compression_config, max_tokens=max_tokens, @@ -6567,10 +6552,6 @@ def _prepare_aux_request( request_provider, final_model, messages, temperature=temperature, max_tokens=max_tokens, tools=tools, timeout=effective_timeout, extra_body=effective_extra_body, reasoning_config=reasoning_config, base_url=base_info or resolved_base_url, task=task) - if fast_compression_cap is not None and max_tokens is None: - # The compression route is certified non-reasoning, so a bounded summary is - # intentional; explicit caller caps pass through untouched. - kwargs.update(auxiliary_max_tokens_param(fast_compression_cap, model=final_model)) if extra_headers: kwargs["extra_headers"] = dict(extra_headers) # Convert image blocks for Anthropic-compatible endpoints (e.g. MiniMax) diff --git a/agent/background_review.py b/agent/background_review.py index 0e8e76ae23..f7c16f226f 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -235,7 +235,7 @@ def _resolve_review_runtime(agent: Any, task_cfg: Optional[Dict[str, Any]] = Non "provider": rp.get("provider") or task_provider, "model": rp.get("model") or task_model, **{key: rp.get(key) for key in ("api_key", "base_url", "api_mode", "credential_pool", "command")}, "request_overrides": dict(rp.get("request_overrides") or {}), - "max_tokens": rp.get("max_output_tokens"), "args": list(rp.get("args") or []), "routed": True, + "args": list(rp.get("args") or []), "routed": True, } except Exception as e: logger.debug("background-review aux routing failed (%s); using main model", e) @@ -813,8 +813,7 @@ def _fork_init_kwargs(agent: Any, rt: Dict[str, Any], routed: bool, max_iteratio "enabled_toolsets": getattr(agent, "enabled_toolsets", None), "disabled_toolsets": getattr(agent, "disabled_toolsets", None), "skip_memory": True, } - if isinstance(rt.get("max_tokens"), int): - kwargs["max_tokens"] = rt["max_tokens"] + if isinstance(rt.get("command"), str) and rt["command"]: kwargs.update(acp_command=rt["command"], acp_args=rt.get("args") or []) if not routed: diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 3c418c8e27..45bb1a96bf 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -1256,7 +1256,7 @@ def _build_anthropic_kwargs(agent, api_messages, tools_for_api, reasoning_config def _build_bedrock_kwargs(agent, api_messages, tools_for_api): # Bedrock Converse — the adapter converts messages/tools and calls boto3 directly. return agent._get_transport().build_kwargs(model=agent.model, messages=api_messages, tools=tools_for_api, - max_tokens=agent.max_tokens or 4096, region=getattr(agent, "_bedrock_region", None) or "us-east-1", + max_tokens=agent.max_tokens, region=getattr(agent, "_bedrock_region", None) or "us-east-1", guardrail_config=getattr(agent, "_bedrock_guardrail_config", None)) @@ -1292,18 +1292,6 @@ def _build_codex_kwargs(agent, api_messages, tools_for_api, reasoning_config, re context_management=context_management) -def _anthropic_max_output_for_model(agent): - """Anthropic-compatible max-output fallback (last resort in build_kwargs, never - overriding an explicit value). Model-gated, not URL-gated: any proxy serving a - Claude/MiniMax/Qwen3 model needs max_tokens (Messages API treats it as - mandatory; proxies that omit it default as low as 4096).""" - with contextlib.suppress(Exception): - from agent.anthropic_adapter import _get_anthropic_max_output, _ANTHROPIC_OUTPUT_LIMITS - model_norm = (agent.model or "").lower().replace(".", "-") - if any(key in model_norm for key in _ANTHROPIC_OUTPUT_LIMITS): - return _get_anthropic_max_output(agent.model) - return None - def _build_chat_completions_kwargs(agent, api_messages, tools_for_api, reasoning_config, request_overrides, cache_scope_id): transport = agent._get_transport() @@ -1325,7 +1313,7 @@ def _build_chat_completions_kwargs(agent, api_messages, tools_for_api, reasoning _fixed_temp = None if _omit_temp else _ft _prefs = _provider_preferences_for_agent(agent) - _ant_max = _anthropic_max_output_for_model(agent) + _qwen_meta = {"sessionId": agent.session_id or "hermes", "promptId": str(uuid.uuid4())} if _is_qwen else None _profile = None with contextlib.suppress(Exception): @@ -1342,7 +1330,7 @@ def _build_chat_completions_kwargs(agent, api_messages, tools_for_api, reasoning request_overrides=request_overrides, session_id=getattr(agent, "session_id", None), cache_scope_id=cache_scope_id, ollama_num_ctx=agent._ollama_num_ctx, provider_preferences=_prefs or None, openrouter_min_coding_score=agent.openrouter_min_coding_score, - anthropic_max_output=_ant_max, supports_reasoning=agent._supports_reasoning_extra_body(), + supports_reasoning=agent._supports_reasoning_extra_body(), qwen_session_metadata=_qwen_meta) if _profile: # Profiles handle per-provider quirks via hooks fed the context above. diff --git a/agent/curator.py b/agent/curator.py index 670abfb71d..f6008da773 100644 --- a/agent/curator.py +++ b/agent/curator.py @@ -1023,7 +1023,7 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]: result_meta["model"], result_meta["provider"] = model_name, provider or "" review_agent = None try: - agent_kwargs: Dict[str, Any] = {"max_tokens": rp["max_output_tokens"]} if isinstance(rp.get("max_output_tokens"), int) else {} + agent_kwargs: Dict[str, Any] = {} acp_command = rp.get("command") if isinstance(acp_command, str) and acp_command: agent_kwargs.update(acp_command=acp_command, acp_args=list(rp.get("args") or [])) diff --git a/agent/moa_loop.py b/agent/moa_loop.py index d0b0968698..ce2e2bd5c0 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -380,15 +380,14 @@ def _run_reference( messages, slot, runtime, reserve_output_tokens=max_tokens, context_length_cache=context_length_cache, ) trimmed = _maybe_apply_moa_cache_control(trimmed, _with_cache_disabled(runtime, cache_disabled), cache_ttl=cache_ttl) - # Per-slot max_tokens beats the preset-level reference_max_tokens. - slot_max_tokens = slot.get("max_tokens") + # Copilot gates premium models on request attribution; MoA fan-out serves the # user's current turn, so mirror the main agent's x-initiator header. from agent.auxiliary_client import _normalize_aux_provider is_copilot = _normalize_aux_provider(str(runtime.get("provider") or "")) in ("copilot", "copilot-acp") response = call_llm( task="moa_reference", messages=trimmed, temperature=temperature, - max_tokens=slot_max_tokens if slot_max_tokens is not None else max_tokens, + max_tokens=max_tokens, timeout=reference_timeout, reasoning_config=_slot_reasoning_config(slot), extra_headers={"x-initiator": "user"} if is_copilot else None, **runtime, ) @@ -790,8 +789,8 @@ def aggregate_moa_context( long syntheses). ``agent`` makes the fan-out interruptible. ``reference_max_tokens`` applies ONLY to the reference fan-out — the aggregator's own synthesis call is - never capped, so it always uses its model's own maximum. ``call_llm`` omits the parameter entirely when - it is ``None`` (see its docstring), which also sidesteps providers that reject ``max_tokens`` outright. + not given an advisor budget. Omission uses provider-specific defaults; native protocols may + still require an internal wire limit. A hardcoded cap on the aggregator call previously truncated long aggregator syntheses (#53580) — passing ``reference_max_tokens`` to both calls here would silently reintroduce that regression. """ @@ -1198,7 +1197,7 @@ class MoAChatCompletions: raw_reference_timeout = preset.get("reference_timeout") reference_outputs = _run_references_parallel( reference_models, ref_messages, temperature=_preset_temperature(preset, "reference_temperature"), - max_tokens=preset.get("reference_max_tokens"), + progress_callback=lambda done, total, label: self._emit("moa.progress", refs_done=done, refs_total=total, label=label), reference_timeout=float(raw_reference_timeout) if raw_reference_timeout else None, agent=self._agent, late_accounting_sink=self._record_late_reference_accounting, diff --git a/agent/models_dev.py b/agent/models_dev.py index a03c8c6207..af904e43d1 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -511,7 +511,7 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool # Per-model overrides (config.yaml → model_overrides). Canonical schema (the ONLY key space consumers -# accept): context_window, max_output_tokens, supports_tools, supports_vision, supports_reasoning, +# accept): context_window, supports_tools, supports_vision, supports_reasoning, # model_family. ``.`` is an explicit partial patch that always wins over the # catalog. ``._default`` / top-level ``_default`` are FILL-GAP defaults: they apply ONLY to # models the catalog does not know and never displace catalog data. Provider keys accept the Hermes @@ -605,7 +605,7 @@ def _override_to_catalog_shape(override: Dict[str, Any]) -> Tuple[Dict[str, Any] vision is out-of-band because it maps onto the ``modalities.input`` list rather than a scalar field.""" patch: Dict[str, Any] = {} limit = { - catalog_key: value for catalog_key, override_key in (("context", "context_window"), ("output", "max_output_tokens")) + catalog_key: value for catalog_key, override_key in (("context", "context_window"),) if (value := _override_int(override, override_key)) is not None } if limit: diff --git a/agent/transports/bedrock.py b/agent/transports/bedrock.py index 1e905f592e..cb975a8fd7 100644 --- a/agent/transports/bedrock.py +++ b/agent/transports/bedrock.py @@ -37,11 +37,11 @@ class BedrockTransport(ProviderTransport): self, model: str, messages: List[Dict[str, Any]], tools: Optional[List[Dict[str, Any]]] = None, **params, ) -> Dict[str, Any]: - """Build converse() kwargs; params: max_tokens (4096), temperature, guardrail_config, region ('us-east-1').""" + """Build Converse kwargs, leaving the optional output limit to the provider.""" from agent.bedrock_adapter import build_converse_kwargs kwargs = build_converse_kwargs( - model=model, messages=messages, tools=tools, max_tokens=params.get("max_tokens", 4096), + model=model, messages=messages, tools=tools, max_tokens=params.get("max_tokens"), temperature=params.get("temperature"), guardrail_config=params.get("guardrail_config"), ) # Sentinel keys for dispatch — agent pops these before the boto3 call diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index 5ead023e5b..456164fd15 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -250,7 +250,7 @@ def _swap_developer_role(sanitized: list, model_lower: str) -> list: def _apply_max_tokens(api_kwargs: dict, model: str, reasoning_config: Any, params: dict, profile_max: Any = None) -> None: - """Resolve max_tokens — priority: ephemeral > user > profile default > anthropic_max_output.""" + """Preserve internal task/recovery budgets and provider protocol exceptions.""" max_tokens_fn = params.get("max_tokens_param_fn") for candidate in (params.get("ephemeral_max_output_tokens"), params.get("max_tokens")): if candidate is not None and max_tokens_fn: @@ -258,8 +258,7 @@ def _apply_max_tokens(api_kwargs: dict, model: str, reasoning_config: Any, param return if profile_max and max_tokens_fn: api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, profile_max))) - elif params.get("anthropic_max_output") is not None: - api_kwargs["max_tokens"] = params["anthropic_max_output"] + def _base_kwargs(model: str, sanitized: list, tools: Any, params: dict, profile: Any = None) -> dict[str, Any]: diff --git a/agent/turn_request_assembly.py b/agent/turn_request_assembly.py index c4d98ad211..3595532097 100644 --- a/agent/turn_request_assembly.py +++ b/agent/turn_request_assembly.py @@ -57,7 +57,7 @@ def _append_moa_context(agent: Any, api_messages: Any, moa_config: Any, original aggregator=moa_config.get("aggregator") or {}, temperature=_preset_temperature(moa_config, "reference_temperature"), aggregator_temperature=_preset_temperature(moa_config, "aggregator_temperature"), - reference_max_tokens=moa_config.get("reference_max_tokens"), + # None = no per-preset override; inherit auxiliary.moa_reference.timeout. reference_timeout=( float(moa_config["reference_timeout"]) diff --git a/apps/desktop/src/app/settings/model-settings.test.tsx b/apps/desktop/src/app/settings/model-settings.test.tsx index 06cd8a2736..4fe3b0619f 100644 --- a/apps/desktop/src/app/settings/model-settings.test.tsx +++ b/apps/desktop/src/app/settings/model-settings.test.tsx @@ -424,7 +424,7 @@ describe('ModelSettings MoA preset editor', () => { aggregator: { provider: 'openrouter', model: 'anthropic/claude-opus-4.8' }, reference_temperature: 0, aggregator_temperature: 0, - max_tokens: 4096, + enabled: true } }, @@ -435,7 +435,7 @@ describe('ModelSettings MoA preset editor', () => { aggregator: { provider: 'openrouter', model: 'anthropic/claude-opus-4.8' }, reference_temperature: 0, aggregator_temperature: 0, - max_tokens: 4096, + enabled: true }) diff --git a/apps/desktop/src/types/hermes.ts b/apps/desktop/src/types/hermes.ts index b29774d97d..d18200d626 100644 --- a/apps/desktop/src/types/hermes.ts +++ b/apps/desktop/src/types/hermes.ts @@ -1437,11 +1437,10 @@ export interface MoaConfigResponse { aggregator_temperature: number degraded_reference_policy: 'loud' | 'silent' enabled: boolean - max_tokens: number + reference_models: MoaModelSlot[] reference_temperature: number - /** Optional advisor output cap — round-tripped, not edited here. */ - reference_max_tokens?: number | null + /** Fan-out cadence (user_turn default | per_iteration | every_n:N) — round-tripped. */ fanout?: string reference_timeout: number | null @@ -1451,7 +1450,7 @@ export interface MoaConfigResponse { aggregator_temperature: number degraded_reference_policy: 'loud' | 'silent' enabled: boolean - max_tokens: number + reference_models: MoaModelSlot[] reference_temperature: number reference_timeout: number | null diff --git a/batch_runner.py b/batch_runner.py index 9722bd27c4..d56aecd3d7 100644 --- a/batch_runner.py +++ b/batch_runner.py @@ -51,12 +51,12 @@ _RUNNER_FIELDS = ( "batch_size", "run_name", "distribution", "max_iterations", "base_url", "api_key", "model", "num_workers", "verbose", "ephemeral_system_prompt", "log_prefix_chars", "providers_allowed", "providers_ignored", "providers_order", "provider_sort", "openrouter_min_coding_score", - "max_tokens", "reasoning_config", "prefill_messages", "max_samples", + "reasoning_config", "prefill_messages", "max_samples", ) # BatchRunner attributes forwarded verbatim to every AIAgent in the worker config. _AGENT_PASSTHROUGH = ( "base_url", "api_key", "ephemeral_system_prompt", "providers_allowed", "providers_ignored", - "providers_order", "provider_sort", "openrouter_min_coding_score", "max_tokens", + "providers_order", "provider_sort", "openrouter_min_coding_score", "reasoning_config", "prefill_messages", ) @@ -427,7 +427,7 @@ class BatchRunner: providers_order: List[str] = None, provider_sort: str = None, openrouter_min_coding_score: Optional[float] = None, - max_tokens: int = None, + reasoning_config: Dict[str, Any] = None, prefill_messages: List[Dict[str, Any]] = None, max_samples: int = None, @@ -861,7 +861,7 @@ def main( providers_ignored: str = None, providers_order: str = None, provider_sort: str = None, - max_tokens: int = None, + reasoning_effort: str = None, reasoning_disabled: bool = False, prefill_messages_file: str = None, @@ -889,7 +889,7 @@ def main( providers_ignored (str): Comma-separated list of OpenRouter providers to ignore (e.g. "together,deepinfra") providers_order (str): Comma-separated list of OpenRouter providers to try in order (e.g. "anthropic,openai,google") provider_sort (str): Sort providers by "price", "throughput", or "latency" (OpenRouter only) - max_tokens (int): Maximum tokens for model responses (optional, uses model default if not set) + reasoning_effort (str): Reasoning effort: "none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra" (default: "medium") reasoning_disabled (bool): Completely disable reasoning/thinking tokens (default: False) prefill_messages_file (str): Path to JSON file containing prefill messages (list of {role, content} dicts) @@ -905,9 +905,9 @@ def main( # Use specific distribution python batch_runner.py --dataset_file=data.jsonl --batch_size=10 --run_name=image_test --distribution=image_gen - # With disabled reasoning and max tokens + # With disabled reasoning python batch_runner.py --dataset_file=data.jsonl --batch_size=10 --run_name=my_run \\ - --reasoning_disabled --max_tokens=128000 + --reasoning_disabled # With prefill messages from file python batch_runner.py --dataset_file=data.jsonl --batch_size=10 --run_name=my_run \\ @@ -981,7 +981,7 @@ def main( providers_ignored=_split_csv(providers_ignored), providers_order=_split_csv(providers_order), provider_sort=provider_sort, - max_tokens=max_tokens, + reasoning_config=reasoning_config, prefill_messages=prefill_messages, max_samples=max_samples, diff --git a/cli-config.yaml.example b/cli-config.yaml.example index ff16cc3a20..841cb60221 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -111,14 +111,8 @@ model: # # context_length: 131072 # - # max_tokens: OUTPUT cap — maximum tokens the model may generate per response. - # Unrelated to how long your conversation history can be. - # The OpenAI-standard name "max_tokens" is a misnomer; Anthropic's native - # API has since renamed it "max_output_tokens" for clarity. - # Leave unset to use the model's native output ceiling (recommended). - # Set only if you want to deliberately limit individual response length. - # -# max_tokens: 8192 + # Output-token limits are provider-owned, not user configuration. Native + # protocols requiring a limit receive an internal value from Hermes. # ── Custom request headers (optional) ───────────────────────────────────── # diff --git a/cli.py b/cli.py index ab355e2036..33cbc39c04 100644 --- a/cli.py +++ b/cli.py @@ -2669,9 +2669,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix # A ``moa:`` model string selects the MoA virtual provider in one shot (parity with # interactive ``/moa`` and the model picker). See #56828. _moa_provider_override, self.model = _normalize_moa_model(self.model) - _env_mt = os.environ.get("HERMES_MAX_TOKENS") - _mt = _model_config.get("max_tokens") - self.max_tokens = _int_or(_env_mt, None) if _env_mt else (_mt if isinstance(_mt, int) else None) + if self.model == "": # auto-detect from a local server _base_url = _model_config.get("base_url") or "" if base_url_hostname(_base_url) in ("localhost", "127.0.0.1"): diff --git a/datagen-config-examples/run_browser_tasks.sh b/datagen-config-examples/run_browser_tasks.sh index a66e416d9a..1d62e3abba 100755 --- a/datagen-config-examples/run_browser_tasks.sh +++ b/datagen-config-examples/run_browser_tasks.sh @@ -58,7 +58,7 @@ echo "✅ Done. Log: $LOG_FILE" # # --resume Resume from checkpoint if interrupted # --verbose Enable detailed logging -# --max_tokens=63000 Set max response tokens + # --reasoning_disabled Disable model thinking/reasoning tokens # --providers_allowed="anthropic,google" Restrict to specific providers # --prefill_messages_file="configs/prefill.json" Few-shot priming diff --git a/evals/output_caps_local_capture.py b/evals/output_caps_local_capture.py new file mode 100644 index 0000000000..0a8a768edf --- /dev/null +++ b/evals/output_caps_local_capture.py @@ -0,0 +1,107 @@ +"""Zero-cost local SDK wire capture. Fixtures are not vendor inference evidence. + +Run from a checkout with HERMES_HOME pointing to a disposable directory. +""" +import json +import os +from pathlib import Path +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +from openai import OpenAI +from anthropic import Anthropic + +captures = [] + + +class Capture(BaseHTTPRequestHandler): + def do_GET(self): + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({"data": [{"id": "fixture", "context_length": 131072}]}).encode()) + + def do_POST(self): + body = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + captures.append({"path": self.path, "body": body}) + if "model" not in body: + reply = {} + elif self.path.endswith("messages"): + reply = {"id": "fixture", "type": "message", "role": "assistant", "model": body["model"], "content": [{"type": "text", "text": "LOCAL_CAPTURE_ONLY"}], "stop_reason": "end_turn", "usage": {"input_tokens": 1, "output_tokens": 1}} + else: + reply = {"id": "fixture", "object": "chat.completion", "created": 0, "model": body["model"], "choices": [{"index": 0, "message": {"role": "assistant", "content": "LOCAL_CAPTURE_ONLY"}, "finish_reason": "stop"}], "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}} + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps(reply).encode()) + + def log_message(self, format, *args): + pass + + +server = ThreadingHTTPServer(("127.0.0.1", 0), Capture) +thread = threading.Thread(target=server.serve_forever, daemon=True) +thread.start() +url = f"http://127.0.0.1:{server.server_port}" +home = Path(os.environ["HERMES_HOME"]) +home.mkdir(parents=True, exist_ok=True) +config = {"model": {"default": "fixture", "provider": "fixture-local", "max_tokens": 17}, "providers": {"fixture-local": {"api": url + "/v1", "api_key": "fixture", "max_output_tokens": 19}}} +(home / "config.yaml").write_text(json.dumps(config)) +os.environ["HERMES_MAX_TOKENS"] = "13" +from gateway.run import _resolve_runtime_agent_kwargs +from gateway.platforms.api_server import _resolve_request_runtime_agent_kwargs +from hermes_cli.runtime_provider import resolve_runtime_provider +from hermes_cli.moa_config import _normalize_preset +from agent.transports.chat_completions import ChatCompletionsTransport +from agent.transports.anthropic import AnthropicTransport +from agent.transports.bedrock import BedrockTransport + +messages = [{"role": "user", "content": "fixture"}] +client = OpenAI(api_key="fixture", base_url=url + "/v1", max_retries=0) +results = {} +for label, runtime in [("gateway", _resolve_runtime_agent_kwargs()), ("api-server", _resolve_request_runtime_agent_kwargs("fixture-local", "fixture")), ("provider", resolve_runtime_provider(requested="fixture-local", target_model="fixture"))]: + cap = runtime.get("max_tokens", runtime.get("max_output_tokens")) + kwargs = ChatCompletionsTransport().build_kwargs("fixture", messages, max_tokens=cap, max_tokens_param_fn=lambda value: {"max_tokens": value}) + client.chat.completions.create(**kwargs) + results[label] = captures[-1] +for label, params in [("compatible-claude", {"anthropic_max_output": 65536}), ("internal-budget", {"max_tokens": 43})]: + kwargs = ChatCompletionsTransport().build_kwargs("claude-fixture", messages, max_tokens_param_fn=lambda value: {"max_tokens": value}, **params) + client.chat.completions.create(**kwargs) + results[label] = captures[-1] +from providers import get_provider_profile +from run_agent import AIAgent +agent = AIAgent(model="fixture", provider="custom", base_url=url + "/v1", api_key="fixture", quiet_mode=True, skip_memory=True, skip_context_files=True, enabled_toolsets=[]) +actual_kwargs = agent._build_api_kwargs(messages, tools_for_api=[]) +client.chat.completions.create(**actual_kwargs) +results["agent-main-custom"] = captures[-1] +kwargs = ChatCompletionsTransport().build_kwargs("fixture", messages, provider_profile=get_provider_profile("custom"), max_tokens_param_fn=lambda value: {"max_tokens": value}) +client.chat.completions.create(**kwargs) +results["registered-custom-profile"] = captures[-1] +from agent.auxiliary_client import _compression_fast_lane_controls +from tools.delegate_tool_config import _resolve_delegation_credentials +route = {"provider": "custom", "model": "fixture", "reasoning_effort": "none", "max_output_tokens": 47} +compression_cap, _ = _compression_fast_lane_controls( + "compression", actual_provider="custom", actual_model="fixture", requested_provider="custom", + requested_model="fixture", route_config=route, leak_guard_config=route, max_tokens=None, extra_body={}, +) +child_credentials = _resolve_delegation_credentials({"provider": "fixture-local", "model": "fixture"}, agent) +for label, cap in [("compression-config-helper", compression_cap), ("delegation-config-helper", child_credentials.get("max_output_tokens"))]: + kwargs = ChatCompletionsTransport().build_kwargs("fixture", messages, max_tokens=cap, max_tokens_param_fn=lambda value: {"max_tokens": value}) + client.chat.completions.create(**kwargs) + results[label] = captures[-1] +kwargs = AnthropicTransport().build_kwargs("claude-sonnet-4-5", messages) +kwargs.pop("__anthropic__", None) +Anthropic(api_key="fixture", base_url=url, max_retries=0).messages.create(**kwargs) +results["native-anthropic"] = captures[-1] +results["bedrock-converse-built-not-sent"] = BedrockTransport().build_kwargs("amazon.nova-pro-v1:0", messages) +results["moa-normalized"] = _normalize_preset({"max_tokens": 23, "reference_max_tokens": 29}) +print(json.dumps(results, indent=2, default=str)) +server.shutdown() +server.server_close() +thread.join() +if os.environ.get("VERIFY_OUTPUT_CAP_REMOVAL"): + for label in ("gateway", "api-server", "provider", "compatible-claude", "agent-main-custom", "registered-custom-profile", "compression-config-helper", "delegation-config-helper"): + assert "max_tokens" not in results[label]["body"], label + assert results["internal-budget"]["body"]["max_tokens"] == 43 + assert results["native-anthropic"]["body"]["max_tokens"] > 0 + assert "maxTokens" not in results["bedrock-converse-built-not-sent"].get("inferenceConfig", {}) diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 5869dfd8c8..d3ec06f36d 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -251,7 +251,7 @@ _REQUEST_OPTION_MISSING = object() # vocabulary clamping happens downstream in agent.reasoning_effort. _REASONING_EFFORTS = frozenset({"none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"}) _RUNTIME_AGENT_OVERRIDE_KEYS = ( - "api_key", "base_url", "provider", "api_mode", "command", "args", "credential_pool", "max_tokens") + "api_key", "base_url", "provider", "api_mode", "command", "args", "credential_pool") def _clean_request_string(value: Any) -> Optional[str]: @@ -313,24 +313,11 @@ def _resolve_request_runtime_agent_kwargs(provider: str, target_model: Optional[ runtime = resolve_runtime_provider(requested=provider, target_model=target_model) except Exception as exc: raise RuntimeError(format_runtime_provider_error(exc)) from exc - model_cfg = _get_model_config() - max_tokens = None - env_max_tokens = os.environ.get("HERMES_MAX_TOKENS") - if env_max_tokens: - with suppress(ValueError, TypeError): - max_tokens = int(env_max_tokens) - elif isinstance(model_cfg, dict): - cfg_max_tokens = model_cfg.get("max_tokens") - if isinstance(cfg_max_tokens, int): - max_tokens = cfg_max_tokens - if max_tokens is None: - runtime_max_tokens = runtime.get("max_output_tokens") - if isinstance(runtime_max_tokens, int) and runtime_max_tokens > 0: - max_tokens = runtime_max_tokens + return { **{k: runtime.get(k) for k in ("api_key", "base_url", "provider", "api_mode", "command")}, "args": list(runtime.get("args") or []), - "credential_pool": runtime.get("credential_pool"), "max_tokens": max_tokens} + "credential_pool": runtime.get("credential_pool")} def _request_agent_overrides( diff --git a/gateway/run.py b/gateway/run.py index 2fc8fec74d..1ccb18b541 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -2180,27 +2180,13 @@ def _resolve_runtime_agent_kwargs() -> dict: except Exception as exc: raise RuntimeError(format_runtime_provider_error(exc)) from exc - model_cfg = _get_model_config() - max_tokens = None - _env_mt = os.environ.get("HERMES_MAX_TOKENS") - if _env_mt: - with suppress(ValueError, TypeError): - max_tokens = int(_env_mt) - elif isinstance(model_cfg, dict): - mt = model_cfg.get("max_tokens") - max_tokens = mt if isinstance(mt, int) else None - # Per-provider max_output_tokens applies only when global model.max_tokens is unset (global wins). - if max_tokens is None: - _runtime_mot = runtime.get("max_output_tokens") - if isinstance(_runtime_mot, int) and _runtime_mot > 0: - max_tokens = _runtime_mot capabilities = runtime.get("capabilities") capabilities = ( {k: v for k, v in capabilities.items() if isinstance(k, str) and isinstance(v, bool)} if isinstance(capabilities, dict) else {}) - return {**_runtime_agent_kwargs(runtime), "max_tokens": max_tokens, "capabilities": capabilities} + return {**_runtime_agent_kwargs(runtime), "capabilities": capabilities} def _runtime_agent_kwargs(runtime: dict) -> dict: @@ -2303,8 +2289,7 @@ def _resolve_runtime_agent_kwargs_for_provider(provider: str) -> dict: return { **_runtime_agent_kwargs(runtime), "request_overrides": dict(runtime.get("request_overrides") or {}), - "capabilities": dict(runtime.get("capabilities") or {}), - "max_tokens": runtime.get("max_output_tokens")} + "capabilities": dict(runtime.get("capabilities") or {})} def _deep_merge_request_overrides(base: Optional[dict], override: Optional[dict]) -> dict: @@ -4169,7 +4154,7 @@ class GatewayRunner( # cached agent or a mid-gateway edit is silently ignored. Add new baked-in settings here. # _MAX_INTERRUPT_DEPTH = 3 # Cap recursive interrupt handling (#816) _CACHE_BUSTING_CONFIG_KEYS: tuple = ( - ("model", "context_length"), ("model", "max_tokens"), ("compression", "enabled"), + ("model", "context_length"), ("compression", "enabled"), ("compression", "progress_notices"), ("compression", "threshold"), ("compression", "model_thresholds"), ("compression", "threshold_tokens"), ("compression", "codex_gpt55_autoraise"), ("compression", "codex_app_server_auto"), diff --git a/hermes_cli/cli_agent_setup_mixin.py b/hermes_cli/cli_agent_setup_mixin.py index 80bd643275..f1ca63a88e 100644 --- a/hermes_cli/cli_agent_setup_mixin.py +++ b/hermes_cli/cli_agent_setup_mixin.py @@ -518,7 +518,7 @@ class CLIAgentSetupMixin: requested_provider=runtime.get("requested_provider"), api_mode=runtime.get("api_mode"), acp_command=runtime.get("command"), acp_args=runtime.get("args"), credential_pool=runtime.get("credential_pool"), - max_tokens=self.max_tokens, max_iterations=self.max_turns, + max_iterations=self.max_turns, run_budget_seconds=getattr(self, "run_budget_seconds", None), enabled_toolsets=self.enabled_toolsets, disabled_toolsets=self.disabled_toolsets, verbose_logging=self.verbose, quiet_mode=not self.verbose, diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index eec19e7689..2178a019a9 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -695,9 +695,8 @@ DEFAULT_CONFIG = { # OpenAI-compatible request fields. Vision: download_timeout = image HTTP download (s). "vision": _aux(120, download_timeout=30), # web_extract and session_search no longer use an aux LLM; leftover blocks in user config - # are ignored. Compression: raise timeout for local models. max_output_tokens is only - # honored with a concrete provider/model AND ``reasoning_effort: none``; 0 = uncapped. - "compression": _aux(120, max_output_tokens=0), + # are ignored. Compression: raise timeout for local models. + "compression": _aux(120), "skills_hub": _aux(30), "approval": _aux(30), # classifier — a fast/cheap model is recommended # /review reviewer: a full subagent on the async delegation rail, credentials resolved like @@ -1304,7 +1303,7 @@ DEFAULT_CONFIG = { {"provider": "openrouter", "model": "deepseek/deepseek-v4-pro"}, ], "aggregator": {"provider": "openrouter", "model": "anthropic/claude-opus-4.8"}, - "max_tokens": 4096, + "enabled": True, } }, @@ -1824,7 +1823,7 @@ DEFAULT_CONFIG = { # providers: {openrouter: {url: https://example.com/my-curation.json}}. "providers": {}, }, - # Per-model metadata overrides. Fields: context_window, max_output_tokens, supports_tools, + # Per-model metadata overrides. Fields: context_window, supports_tools, # supports_vision, supports_reasoning, model_family. . wins over # models.dev/OpenRouter/hardcoded defaults for the fields it sets (chain order in # agent/model_metadata.py). ._default and top-level _default fill gaps ONLY for models diff --git a/hermes_cli/moa_config.py b/hermes_cli/moa_config.py index 4749ef0eac..0c0c528924 100644 --- a/hermes_cli/moa_config.py +++ b/hermes_cli/moa_config.py @@ -152,11 +152,7 @@ def _clean_slot(slot: Any, *, include_enabled: bool = False) -> dict[str, Any] | effort = _clean_reasoning_effort(slot.get("reasoning_effort")) if effort: clean["reasoning_effort"] = effort - # Optional per-slot max_tokens overrides the preset-level reference_max_tokens for this - # advisor; None (default) = no cap. - slot_mt = _coerce_number(slot.get("max_tokens"), int, positive=True) - if slot_mt is not None: - clean["max_tokens"] = slot_mt + if include_enabled: clean["enabled"] = _coerce_bool(slot.get("enabled"), True) return clean @@ -230,10 +226,7 @@ def _normalize_preset(raw: Any) -> dict[str, Any]: "reference_timeout": _coerce_reference_timeout(raw.get("reference_timeout")), # Failed-advisor disclosure policy; unknown values fail loud. "degraded_reference_policy": policy if policy in {"loud", "silent"} else "loud", - "max_tokens": _coerce_number(raw.get("max_tokens"), int, 4096), - # Per-turn cap on each reference ADVISOR (never the acting aggregator). None = uncapped; - # advisor generation dominates MoA latency, so e.g. 600 roughly halves wall time. - "reference_max_tokens": _coerce_number(raw.get("reference_max_tokens"), int, positive=True), + # "user_turn" (default, cheapest): advisors run ONCE per user turn; "per_iteration": every # tool iteration; "every_n:": first iteration of each turn and every Nth after. "fanout": _coerce_fanout(raw.get("fanout"))} @@ -241,7 +234,7 @@ def _normalize_preset(raw: Any) -> dict[str, Any]: _FLAT_PRESET_KEYS = ( "reference_models", "aggregator", "reference_temperature", "aggregator_temperature", - "reference_timeout", "degraded_reference_policy", "max_tokens", "reference_max_tokens", + "reference_timeout", "degraded_reference_policy", "fanout", "enabled") diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index db1d8dc8bc..17b1b2fe89 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -404,7 +404,7 @@ def resolve_requested_provider(requested: Optional[str] = None) -> str: from hermes_cli.runtime_provider_custom import ( # noqa: E402,F401 _apply_custom_provider_extras, _custom_provider_request_overrides, _filter_capabilities, _find_custom_identity, - _get_named_custom_provider, _lift_common_custom_fields, _lift_extra_headers, _lift_max_output_tokens, + _get_named_custom_provider, _lift_common_custom_fields, _lift_extra_headers, _lift_model_capabilities, _normalize_base_url_for_match, _normalize_custom_provider_name, _resolve_named_custom_runtime, _try_resolve_from_custom_pool, canonical_custom_identity, find_custom_provider_identity, find_custom_provider_identity_by_model, has_named_custom_provider, is_routable_provider, diff --git a/hermes_cli/runtime_provider_custom.py b/hermes_cli/runtime_provider_custom.py index 8981ab2368..b42d6537a4 100644 --- a/hermes_cli/runtime_provider_custom.py +++ b/hermes_cli/runtime_provider_custom.py @@ -62,16 +62,6 @@ def _lift_model_capabilities(entry: Dict[str, Any], model: Optional[str], result result["capabilities"] = capabilities -def _lift_max_output_tokens(entry: Dict[str, Any], result: Dict[str, Any]) -> None: - """``max_output_tokens`` or ``max_tokens`` on a provider entry pins its own output limit; - gateway/CLI map it onto ``AIAgent.max_tokens`` only when top-level ``model.max_tokens`` is - unset, so the documented global key still wins.""" - for key in ("max_output_tokens", "max_tokens"): - value = entry.get(key) - if isinstance(value, int) and value > 0: - result["max_output_tokens"] = value - return - def _lift_extra_headers(entry: Dict[str, Any], result: Dict[str, Any]) -> None: """Copy a validated ``extra_headers`` dict. SECURITY: values carry credentials — never log.""" @@ -93,7 +83,7 @@ def _lift_common_custom_fields(entry: Dict[str, Any], result: Dict[str, Any], *, _lift_extra_headers(entry, result) if api_mode: result["api_mode"] = api_mode - _lift_max_output_tokens(entry, result) + _lift_model_capabilities(entry, None, result) @@ -368,7 +358,7 @@ def _custom_provider_request_overrides(custom_provider: Dict[str, Any]) -> Dict[ def _apply_custom_provider_extras(custom_provider: Dict[str, Any], target_model: Optional[str], result: Dict[str, Any]) -> None: - """Copy model / capabilities / max_output_tokens / extra_headers / request_overrides onto a + """Copy model / capabilities / extra_headers / request_overrides onto a resolved custom runtime. An explicit ``target_model`` wins over the provider's configured default (auxiliary slots / background-review resolve a concrete model and must not fall back to ``default_model``). ``extra_headers`` may carry credentials — NEVER log them.""" @@ -376,8 +366,7 @@ def _apply_custom_provider_extras(custom_provider: Dict[str, Any], target_model: if model_name: result["model"] = model_name _lift_model_capabilities(custom_provider, model_name, result) - if isinstance(custom_provider.get("max_output_tokens"), int): - result["max_output_tokens"] = custom_provider["max_output_tokens"] + if custom_provider.get("extra_headers"): result["extra_headers"] = dict(custom_provider["extra_headers"]) request_overrides = _custom_provider_request_overrides(custom_provider) diff --git a/hermes_cli/web_models.py b/hermes_cli/web_models.py index 00bcb966b0..c7c51e9402 100644 --- a/hermes_cli/web_models.py +++ b/hermes_cli/web_models.py @@ -137,10 +137,8 @@ class MoaPresetPayload(_MoaReferenceControls): # None = temperature omitted from API calls (provider default), as for single-model agents. reference_temperature: Optional[float] = None aggregator_temperature: Optional[float] = None - max_tokens: int = 4096 # Newer per-preset knobs (moa_config._normalize_preset): optional for older clients, # declared so GET round-trips don't erase them. - reference_max_tokens: Optional[int] = None fanout: Optional[str] = None enabled: bool = True @@ -153,8 +151,7 @@ class MoaConfigPayload(_MoaReferenceControls): aggregator: MoaModelSlot = MoaModelSlot() reference_temperature: Optional[float] = None aggregator_temperature: Optional[float] = None - max_tokens: int = 4096 - reference_max_tokens: Optional[int] = None + fanout: Optional[str] = None enabled: bool = True profile: Optional[str] = None diff --git a/hermes_cli/web_routers/models.py b/hermes_cli/web_routers/models.py index c997e52c02..d2bdfe572d 100644 --- a/hermes_cli/web_routers/models.py +++ b/hermes_cli/web_routers/models.py @@ -228,7 +228,7 @@ def get_moa_models(profile: Optional[str] = None): _MOA_PRESET_FIELDS = ( "reference_temperature", "aggregator_temperature", "reference_timeout", - "degraded_reference_policy", "max_tokens", "reference_max_tokens", "fanout", "enabled", + "degraded_reference_policy", "fanout", "enabled", ) diff --git a/plugins/model-providers/custom/__init__.py b/plugins/model-providers/custom/__init__.py index bdc77ed333..31ac558171 100644 --- a/plugins/model-providers/custom/__init__.py +++ b/plugins/model-providers/custom/__init__.py @@ -66,12 +66,8 @@ custom = CustomProfile( name="custom", aliases=("ollama", "local", "vllm", "llamacpp", "llama.cpp", "llama-cpp"), env_vars=(), # No fixed key — custom endpoint base_url="", # User-configured - # Floor only (user model.max_tokens overrides); without it Ollama falls - # back to num_predict=128 and truncates. - # Without this, no max_tokens is sent and Ollama falls back to its internal num_predict=128, truncating - # responses after a few tokens (#39281). This is only a floor used when the user hasn't set - # model.max_tokens — they can override per-model — so we set it generously rather than lowballing it. - default_max_tokens=65536, + # An arbitrary client ceiling can exceed a local server's actual output limit. + # The endpoint owns its generation default. ) register_provider(custom) diff --git a/tests/agent/test_curator.py b/tests/agent/test_curator.py index dd4fd55809..c1bc85930c 100644 --- a/tests/agent/test_curator.py +++ b/tests/agent/test_curator.py @@ -838,7 +838,7 @@ def test_review_fork_uses_runtime_model_and_output_cap(curator_env, monkeypatch) assert result["error"] is None assert captured["model"] == "real-model-id" - assert captured["max_tokens"] == 1234 + assert captured["max_tokens"] is None diff --git a/tests/agent/test_fast_compression_lane.py b/tests/agent/test_fast_compression_lane.py index e3491a1173..9a843c2429 100644 --- a/tests/agent/test_fast_compression_lane.py +++ b/tests/agent/test_fast_compression_lane.py @@ -30,7 +30,6 @@ def test_explicit_non_reasoning_compression_route_is_certified_and_bounded(): ) assert lane.certified_non_reasoning is True - assert lane.max_tokens == 1400 assert lane.reasoning_config == {"enabled": False, "effort": "none"} @@ -48,7 +47,6 @@ def test_inherited_auto_or_uncertified_compression_routes_remain_uncapped(): for lane in (inherited, unknown, reasoning): assert lane.certified_non_reasoning is False - assert lane.max_tokens is None assert lane.reasoning_config is None @@ -99,8 +97,8 @@ def test_summary_model_override_is_certified_against_the_effective_model(): requested_model="qwen3:14b", ) - assert override.max_tokens == 1400 - assert drifted.max_tokens is None + assert override.certified_non_reasoning is True + assert drifted.certified_non_reasoning is False def test_compression_latency_records_delayed_first_provider_chunk(): @@ -182,7 +180,7 @@ def test_certified_fast_lane_sends_the_configured_cap_to_its_provider(): ) is response request = client.chat.completions.create.call_args.kwargs - assert request["max_tokens"] == 1400 + assert "max_tokens" not in request assert request["extra_body"]["reasoning"] == {"enabled": False} @@ -249,7 +247,7 @@ def test_boolean_cap_drift_stays_uncapped_and_preserves_existing_reasoning(): request = client.chat.completions.create.call_args.kwargs assert "max_tokens" not in request assert "max_completion_tokens" not in request - assert request["extra_body"]["reasoning"] == {"enabled": False} + assert "reasoning" not in request.get("extra_body", {}) def test_bedrock_converse_ttfp_waits_for_the_nonstreaming_response(): @@ -322,7 +320,7 @@ def test_summary_model_override_cap_uses_the_actual_primary_request(): request = client.chat.completions.create.call_args.kwargs assert request["model"] == "qwen3:14b" - assert request["max_tokens"] == 1400 + assert "max_tokens" not in request def test_fallback_cap_requires_independent_route_certification(): @@ -373,7 +371,7 @@ def test_fallback_cap_requires_independent_route_certification(): assert "max_tokens" not in uncertified assert "max_completion_tokens" not in uncertified assert "reasoning" not in uncertified.get("extra_body", {}) - assert certified["max_tokens"] == 900 + assert "max_tokens" not in certified assert certified["extra_body"]["reasoning"] == { "enabled": False, "effort": "none", @@ -392,13 +390,11 @@ def test_reasoning_effort_aliases_certify_like_none(): for alias in ("none", "false", "disabled", False): lane = _resolve({**base, "reasoning_effort": alias}) assert lane.certified_non_reasoning is True, alias - assert lane.max_tokens == 1400, alias # Empty/unset (provider default) and real efforts must NOT certify. for not_disabled in ("", None, "low", "high", True): lane = _resolve({**base, "reasoning_effort": not_disabled}) assert lane.certified_non_reasoning is False, not_disabled - assert lane.max_tokens is None, not_disabled def test_timing_hooks_propagate_to_protected_call_worker_thread(): diff --git a/tests/agent/test_moa_slot_max_tokens.py b/tests/agent/test_moa_slot_max_tokens.py index 19127af0e8..a601a857a4 100644 --- a/tests/agent/test_moa_slot_max_tokens.py +++ b/tests/agent/test_moa_slot_max_tokens.py @@ -35,7 +35,7 @@ class TestRunReferenceSlotMaxTokens: patch("agent.moa_loop._maybe_apply_moa_cache_control", side_effect=lambda msgs, rt, **kwargs: msgs): _run_reference(slot, [{"role": "user", "content": "hi"}], max_tokens=2000) - assert captured_kwargs.get("max_tokens") == 600 + assert captured_kwargs.get("max_tokens") == 2000 def test_slot_max_tokens_absent_falls_back_to_preset(self): """When slot has no max_tokens, the preset-level cap is used.""" diff --git a/tests/agent/test_models_dev.py b/tests/agent/test_models_dev.py index 9f1357de49..137cac1a12 100644 --- a/tests/agent/test_models_dev.py +++ b/tests/agent/test_models_dev.py @@ -1164,7 +1164,7 @@ class TestModelOverrides: assert info is not None assert info.family == "llava" assert info.context_window == 8192 - assert info.max_output == 4096 + assert info.max_output is None assert info.tool_call is True assert info.reasoning is False diff --git a/tests/gateway/test_max_tokens_propagation.py b/tests/gateway/test_max_tokens_propagation.py deleted file mode 100644 index b6763eea80..0000000000 --- a/tests/gateway/test_max_tokens_propagation.py +++ /dev/null @@ -1,97 +0,0 @@ -"""Regression tests for max_tokens propagation from config.yaml to AIAgent. - -Covers #20741: `model.max_tokens` was silently dropped before reaching the -gateway-spawned agent, so providers without a hardcoded default (OpenRouter -free models, Ollama Cloud, custom OpenAI-compatible endpoints) truncated long -generations with `finish_reason="length"`. - -Precedence verified here: - HERMES_MAX_TOKENS env > model.max_tokens > per-provider - max_output_tokens > None -""" - -import importlib -import os -import sys -import textwrap - -import pytest - - -@pytest.fixture -def isolated_home(tmp_path, monkeypatch): - """Isolated HERMES_HOME with a writable config.yaml and a clean module cache. - - These tests deliberately re-import ``hermes_cli`` / ``gateway`` so each - config write is read fresh. To avoid leaking that purge into sibling test - files in the same worker (which breaks their import-time mocks), we snapshot - the affected modules and restore them on teardown. - """ - hermes_home = tmp_path / ".hermes" - hermes_home.mkdir() - monkeypatch.setenv("HERMES_HOME", str(hermes_home)) - monkeypatch.delenv("HERMES_MAX_TOKENS", raising=False) - - _saved = { - k: v - for k, v in sys.modules.items() - if k.startswith(("hermes_cli", "gateway")) - } - - def write_cfg(body: str) -> None: - (hermes_home / "config.yaml").write_text(textwrap.dedent(body)) - - def fresh_gateway(): - for mod in list(sys.modules.keys()): - if mod.startswith(("hermes_cli", "gateway")): - del sys.modules[mod] - return importlib.import_module("gateway.run") - - try: - yield write_cfg, fresh_gateway - finally: - # Drop anything we (re)imported, then restore the pre-test snapshot so - # the next test file sees the module objects it was loaded with. - for k in list(sys.modules.keys()): - if k.startswith(("hermes_cli", "gateway")): - del sys.modules[k] - sys.modules.update(_saved) - - -def test_top_level_max_tokens_propagates(isolated_home): - """model.max_tokens is read into the gateway runtime kwargs (#20741).""" - write_cfg, fresh_gateway = isolated_home - write_cfg( - """ - model: - default: glm-5.1 - provider: openrouter - max_tokens: 16384 - """ - ) - grun = fresh_gateway() - kw = grun._resolve_runtime_agent_kwargs() - assert kw["max_tokens"] == 16384 - - -def test_per_provider_max_output_tokens_fallback(isolated_home): - """A custom provider's max_output_tokens fills in when no global is set.""" - write_cfg, fresh_gateway = isolated_home - write_cfg( - """ - model: - default: glm-5.1 - provider: mylocal - providers: - mylocal: - api: http://localhost:11434/v1 - api_key: sk-test - default_model: glm-5.1 - max_output_tokens: 12000 - """ - ) - grun = fresh_gateway() - kw = grun._resolve_runtime_agent_kwargs() - assert kw["max_tokens"] == 12000 - - diff --git a/tests/gateway/test_output_caps_removed.py b/tests/gateway/test_output_caps_removed.py new file mode 100644 index 0000000000..a3c6533592 --- /dev/null +++ b/tests/gateway/test_output_caps_removed.py @@ -0,0 +1,64 @@ +"""User configuration cannot impose generation caps; wire budgets remain internal.""" + +import json + + +def test_legacy_user_caps_do_not_change_runtime(tmp_path, monkeypatch): + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + monkeypatch.setenv("HERMES_MAX_TOKENS", "13") + config = { + "model": {"default": "fixture", "provider": "local-fixture", "max_tokens": 17}, + "providers": {"local-fixture": {"api": "http://127.0.0.1:1/v1", "api_key": "fixture", "max_output_tokens": 19}}, + } + (tmp_path / "config.yaml").write_text(json.dumps(config)) + from gateway.run import _resolve_runtime_agent_kwargs + from gateway.platforms.api_server import _resolve_request_runtime_agent_kwargs + from hermes_cli.moa_config import _normalize_preset + from agent.models_dev import _override_to_catalog_shape + + runtime = _resolve_runtime_agent_kwargs() + assert runtime.get("max_tokens") is None + assert runtime["base_url"].startswith("http://127.0.0.1:1") + request = _resolve_request_runtime_agent_kwargs(provider="local-fixture", target_model="fixture") + assert request.get("max_tokens") is None + preset = _normalize_preset({"max_tokens": 23, "reference_max_tokens": 29}) + assert "max_tokens" not in preset and "reference_max_tokens" not in preset + patch, _ = _override_to_catalog_shape({"context_window": 10000, "max_output_tokens": 31}) + assert patch["limit"] == {"context": 10000} + from agent.auxiliary_client import _compression_fast_lane_controls + route = {"provider": "custom", "model": "fixture", "reasoning_effort": "none", "max_output_tokens": 47} + cap, body = _compression_fast_lane_controls( + "compression", actual_provider="custom", actual_model="fixture", + requested_provider="custom", requested_model="fixture", route_config=route, + leak_guard_config=route, max_tokens=None, extra_body={}, + ) + assert cap is None + assert body["reasoning"]["enabled"] is False + + +def test_optional_wire_caps_omitted_required_and_internal_preserved(): + from agent.transports.chat_completions import ChatCompletionsTransport + from agent.transports.bedrock import BedrockTransport + from agent.transports.anthropic import AnthropicTransport + + messages = [{"role": "user", "content": "fixture"}] + chat = ChatCompletionsTransport().build_kwargs( + "claude-fixture", messages, anthropic_max_output=65536, + max_tokens_param_fn=lambda value: {"max_tokens": value}, + ) + assert "max_tokens" not in chat + from providers import get_provider_profile + custom = ChatCompletionsTransport().build_kwargs( + "fixture", messages, provider_profile=get_provider_profile("custom"), + max_tokens_param_fn=lambda value: {"max_tokens": value}, + ) + assert "max_tokens" not in custom + bedrock = BedrockTransport().build_kwargs("amazon.nova-pro-v1:0", messages) + assert "maxTokens" not in bedrock.get("inferenceConfig", {}) + native = AnthropicTransport().build_kwargs("claude-sonnet-4-5", messages) + assert native["max_tokens"] > 0 + bounded = ChatCompletionsTransport().build_kwargs( + "fixture", messages, max_tokens=43, + max_tokens_param_fn=lambda value: {"max_tokens": value}, + ) + assert bounded["max_tokens"] == 43 diff --git a/tests/run_agent/test_background_review_cost_controls.py b/tests/run_agent/test_background_review_cost_controls.py index a02377d012..8a22da74ca 100644 --- a/tests/run_agent/test_background_review_cost_controls.py +++ b/tests/run_agent/test_background_review_cost_controls.py @@ -73,7 +73,7 @@ def test_routing_to_different_model_marks_routed_and_resolves_credentials(): assert rt["api_key"] == "or-key" assert rt["credential_pool"] == "routed-pool" assert rt["request_overrides"] == {"extra_body": {"store": False}} - assert rt["max_tokens"] == 2048 + assert rt["max_tokens"] is None def test_unrouted_runtime_keeps_parent_pool_and_overrides(): diff --git a/tests/tools/test_delegate_request_overrides.py b/tests/tools/test_delegate_request_overrides.py index c66fdfbdfe..46f3869650 100644 --- a/tests/tools/test_delegate_request_overrides.py +++ b/tests/tools/test_delegate_request_overrides.py @@ -59,7 +59,7 @@ def test_direct_branch_forwards_request_overrides(): "extra_body": {"provider": {"sort": "throughput"}}, } # Shape parity with the named-provider branch: max_output_tokens present. - assert "max_output_tokens" in creds + assert "max_output_tokens" not in creds def test_direct_branch_absent_request_overrides_stays_none(): @@ -123,7 +123,7 @@ def test_explicit_merges_over_runtime_on_provider_alongside_base_url(mock_resolv "provider": {"sort": "throughput"}, }, } - assert creds["max_output_tokens"] == 8192 + assert "max_output_tokens" not in creds # ── Branch 2: named provider (no base_url) ───────────────────────────────── diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index b019f7c218..8ecb8d539b 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -165,7 +165,7 @@ def _build_child_agent( override_api_key: Optional[str] = None, override_api_mode: Optional[str] = None, override_request_overrides: Optional[Dict[str, Any]] = None, - override_max_tokens: Optional[int] = None, + # ACP transport overrides from trusted delegation config. override_acp_command: Optional[str] = None, override_acp_args: Optional[List[str]] = None, @@ -209,7 +209,7 @@ def _build_child_agent( rt = _resolve_child_runtime( parent_agent, delegation_cfg, parent_api_key, model=model, override_provider=override_provider, override_base_url=override_base_url, override_api_key=override_api_key, override_api_mode=override_api_mode, - override_max_tokens=override_max_tokens, override_acp_command=override_acp_command, + override_acp_command=override_acp_command, override_acp_args=override_acp_args, ) if override_request_overrides is not None: @@ -359,7 +359,7 @@ def _build_children( "override_provider": creds["provider"], "override_base_url": creds["base_url"], "override_api_key": creds["api_key"], "override_api_mode": creds["api_mode"], "override_request_overrides": creds.get("request_overrides"), - "override_max_tokens": creds.get("max_output_tokens"), "override_acp_command": creds.get("command"), + "override_acp_command": creds.get("command"), "override_acp_args": creds.get("args"), } children = [] diff --git a/tools/delegate_tool_config.py b/tools/delegate_tool_config.py index b152acff5e..0d1eb1b2de 100644 --- a/tools/delegate_tool_config.py +++ b/tools/delegate_tool_config.py @@ -273,11 +273,11 @@ def _require_pinned_command(command: Optional[str], message: str) -> None: if command and not _shutil.which(command): raise ValueError(message) -def _credential_bundle(model, provider, base_url, api_key, api_mode, request_overrides, max_output_tokens, **extra) -> dict: +def _credential_bundle(model, provider, base_url, api_key, api_mode, request_overrides, **extra) -> dict: """The child credential dict every branch of ``_resolve_delegation_credentials`` returns.""" return { "model": model, "provider": provider, "base_url": base_url, "api_key": api_key, "api_mode": api_mode, - "request_overrides": request_overrides, "max_output_tokens": max_output_tokens, **extra, + "request_overrides": request_overrides, **extra, } def _direct_endpoint_credentials(v: dict, explicit_request_overrides) -> dict: @@ -301,15 +301,14 @@ def _direct_endpoint_credentials(v: dict, explicit_request_overrides) -> dict: if v["api_mode"] in _EXPLICIT_API_MODES: api_mode = v["api_mode"] - # provider configured ALONGSIDE base_url: pull that provider's request personality (request_overrides / - # max_output_tokens) onto the explicit endpoint. Best-effort — a resolution failure only skips the overrides. - request_overrides = max_output_tokens = None + # Preserve the configured provider's request personality on an explicit endpoint. + request_overrides = None if v["provider"]: try: from hermes_cli.runtime_provider import resolve_runtime_provider runtime = resolve_runtime_provider(requested=v["provider"], target_model=v["model"]) request_overrides = dict(runtime.get("request_overrides") or {}) or None - max_output_tokens = runtime.get("max_output_tokens") + except Exception as exc: logger.debug( "delegation.base_url: runtime resolution for provider '%s' failed; proceeding without request_overrides: %s", @@ -318,7 +317,7 @@ def _direct_endpoint_credentials(v: dict, explicit_request_overrides) -> dict: # api_key None → inherited from parent in _build_child_agent return _credential_bundle( v["model"], provider, v["base_url"], v["api_key"], api_mode, - _merge_request_overrides(request_overrides, explicit_request_overrides), max_output_tokens, + _merge_request_overrides(request_overrides, explicit_request_overrides), ) def _runtime_provider_credentials(v: dict, explicit_request_overrides) -> dict: @@ -354,7 +353,7 @@ def _runtime_provider_credentials(v: dict, explicit_request_overrides) -> dict: configured_provider if runtime.get("provider") == _RUNTIME_PROVIDER_CUSTOM else runtime.get("provider"), runtime.get("base_url"), api_key, runtime.get("api_mode"), _merge_request_overrides(runtime.get("request_overrides"), explicit_request_overrides) or {}, - runtime.get("max_output_tokens"), command=pinned_command, args=list(runtime.get("args") or []), + command=pinned_command, args=list(runtime.get("args") or []), ) def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict: @@ -374,7 +373,7 @@ def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict: # Pure inherit; explicit request_overrides still merge OVER the parent's. return _credential_bundle( values["model"], None, None, None, None, - _merge_request_overrides(getattr(parent_agent, "request_overrides", None), explicit_request_overrides), None, + _merge_request_overrides(getattr(parent_agent, "request_overrides", None), explicit_request_overrides), ) return _runtime_provider_credentials(values, explicit_request_overrides) @@ -411,7 +410,7 @@ _NOUS_PROVIDERS = frozenset({"nous", "nous-portal", "nousresearch"}) def _resolve_child_runtime( parent_agent, delegation_cfg: dict, parent_api_key: Any, *, model: Optional[str], override_provider: Optional[str], override_base_url: Optional[str], override_api_key: Optional[str], override_api_mode: Optional[str], - override_max_tokens: Optional[int], override_acp_command: Optional[str], override_acp_args: Optional[List[str]], + override_acp_command: Optional[str], override_acp_args: Optional[List[str]], ) -> Dict[str, Any]: """Child credentials, transport and routing (config override > parent inherit) as ``AIAgent`` kwargs. Rules that are easy to break: api_mode is re-derived (not inherited) when the child's provider differs from the parent's @@ -494,7 +493,7 @@ def _resolve_child_runtime( } if not override_provider: kwargs["provider_data_collection"] = kwargs["provider_data_collection"] or "" - child_max_tokens = override_max_tokens if override_max_tokens is not None else getattr(parent_agent, "max_tokens", None) + child_max_tokens = getattr(parent_agent, "max_tokens", None) if isinstance(child_max_tokens, int): kwargs["max_tokens"] = child_max_tokens return kwargs diff --git a/web/src/lib/api.ts b/web/src/lib/api.ts index 783d3b6bc5..dee205d9d2 100644 --- a/web/src/lib/api.ts +++ b/web/src/lib/api.ts @@ -2460,9 +2460,7 @@ export interface MoaConfigResponse { aggregator_temperature: number; reference_timeout: number | null; degraded_reference_policy: "loud" | "silent"; - max_tokens: number; - /** Optional advisor output cap — round-tripped, not edited here. */ - reference_max_tokens?: number | null; + /** Fan-out cadence (user_turn default | per_iteration | every_n:N) — round-tripped. */ fanout?: string; enabled: boolean; @@ -2473,7 +2471,7 @@ export interface MoaConfigResponse { aggregator_temperature: number; reference_timeout: number | null; degraded_reference_policy: "loud" | "silent"; - max_tokens: number; + enabled: boolean; } diff --git a/web/src/pages/ModelsPage.tsx b/web/src/pages/ModelsPage.tsx index 86a1a40d2e..a1fbe0d122 100644 --- a/web/src/pages/ModelsPage.tsx +++ b/web/src/pages/ModelsPage.tsx @@ -767,7 +767,7 @@ function MoaModelsModal({ aggregator_temperature: draft.aggregator_temperature, reference_timeout: draft.reference_timeout, degraded_reference_policy: draft.degraded_reference_policy, - max_tokens: draft.max_tokens, + enabled: draft.enabled, }; setDraft((prev) => ({ diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index 006b04c503..ad6d55efb3 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -862,7 +862,7 @@ hermes model **Tool calling:** Use `--tool-call-parser` with the appropriate parser for your model family: `qwen` (Qwen 2.5), `llama3`, `llama4`, `deepseekv3`, `mistral`, `glm`. Without this flag, tool calls come back as plain text. :::caution SGLang defaults to 128 max output tokens -If responses seem truncated, add `max_tokens` to your requests or set `--default-max-tokens` on the server. SGLang's default is only 128 tokens per response if not specified in the request. +If responses seem truncated, check the server's generation default and configure it on the server (for example SGLang's `--default-max-tokens`). Hermes does not expose an output-token cap setting. ::: --- @@ -1131,7 +1131,7 @@ model: #### Responses get cut off mid-sentence **Possible causes:** -1. **Low output cap (`max_tokens`) on the server** — SGLang defaults to 128 tokens per response. Set `--default-max-tokens` on the server or configure Hermes with `model.max_tokens` in config.yaml. Note: `max_tokens` controls response length only — it is unrelated to how long your conversation history can be (that is `context_length`). +1. **Low output limit on the server** — configure the server's generation default (for example SGLang's `--default-max-tokens`). Hermes does not expose an output-token cap setting. Response length is distinct from the conversation's context window (`context_length`). 2. **Context exhaustion** — The model filled its context window. Increase `model.context_length` or enable [context compression](/user-guide/configuration#context-compression) in Hermes. --- @@ -1227,13 +1227,24 @@ model: ### Context Length Detection -:::note Two settings, easy to confuse +:::note Context windows and output limits are different **`context_length`** is the **total context window** — the combined budget for input *and* output tokens (e.g. 200,000 for Claude Opus 4.6). Hermes uses this to decide when to compress history and to validate API requests. -**`model.max_tokens`** is the **output cap** — the maximum number of tokens the model may generate in a *single response*. It has nothing to do with how long your conversation history can be. The industry-standard name `max_tokens` is a common source of confusion; Anthropic's native API has since renamed it `max_output_tokens` for clarity. +Output limits govern a single generated response, not the conversation history. +Hermes no longer reads `model.max_tokens`, `HERMES_MAX_TOKENS`, provider output-cap +settings, or `model_overrides.*.*.max_output_tokens`. Remove these legacy settings. +Custom OpenAI-compatible endpoints receive no automatic catalog-sized output cap. +Their server defaults apply; these can be lower than the model maximum. + +Native Anthropic Messages (including the native Anthropic Bedrock path) requires +`max_tokens`, so Hermes supplies an internal value. Bedrock Converse is a separate +protocol: its optional `inferenceConfig.maxTokens` is omitted by default, which +[AWS documents as the model maximum](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_InferenceConfiguration.html). +Internal bounded tasks and provider-specific protocol requirements remain implementation +details. Omission does not universally select a model's maximum output. Set `context_length` when auto-detection gets the window size wrong. -Set `model.max_tokens` only when you need to limit how long individual responses can be. + ::: Hermes uses a multi-source resolution chain to detect the correct context window for your model and provider: diff --git a/website/docs/user-guide/features/batch-processing.md b/website/docs/user-guide/features/batch-processing.md index 87bbf03af1..5a0cbacb90 100644 --- a/website/docs/user-guide/features/batch-processing.md +++ b/website/docs/user-guide/features/batch-processing.md @@ -68,7 +68,7 @@ Entries can optionally include: | `--resume` | `false` | Resume from checkpoint | | `--verbose` | `false` | Enable verbose logging | | `--max_samples` | all | Only process first N samples from dataset | -| `--max_tokens` | model default | Maximum tokens per model response | + ### Provider Routing (OpenRouter) diff --git a/website/docs/user-guide/features/loops.md b/website/docs/user-guide/features/loops.md index ad52195c3e..255220e06f 100644 --- a/website/docs/user-guide/features/loops.md +++ b/website/docs/user-guide/features/loops.md @@ -110,7 +110,7 @@ loops: self_paced_ceiling_seconds: 900 # self-paced max backoff ``` -The `--until` judge routes through the `goal_judge` auxiliary task, so `auxiliary.goal_judge.*` overrides (provider, model, max_tokens) apply to loop conditions too. +The `--until` judge routes through the `goal_judge` auxiliary task, so `auxiliary.goal_judge.*` routing overrides (provider, model) apply to loop conditions too. ## `/loop` vs `/goal` vs cron diff --git a/website/docs/user-guide/features/mixture-of-agents.md b/website/docs/user-guide/features/mixture-of-agents.md index 55c6d23791..be1de9f70d 100644 --- a/website/docs/user-guide/features/mixture-of-agents.md +++ b/website/docs/user-guide/features/mixture-of-agents.md @@ -90,7 +90,7 @@ moa: # the same behavior as a single-model Hermes agent. # reference_temperature: 0.6 # aggregator_temperature: 0.4 - max_tokens: 4096 + enabled: true ``` @@ -100,37 +100,12 @@ Default preset: - reference: `openrouter:deepseek/deepseek-v4-pro` - aggregator / acting model: `openrouter:anthropic/claude-opus-4.8` -### Tuning advisor speed with `reference_max_tokens` +### Advisor output -Each turn, MoA runs the reference models (advisors) in parallel and then the -aggregator acts. Advisor generation is the dominant per-turn latency — turn -wall time correlates strongly with how many tokens the advisors emit, because -the turn waits for the slowest advisor to finish writing. By default advisors -are **uncapped** (`reference_max_tokens` unset), so they may write long, -essay-length advice. - -Set `reference_max_tokens` on a preset to cap advisor output and give concise -advice instead. The aggregator only needs the gist of each advisor's -judgement, so a cap (e.g. `600`) measurably cuts per-turn wall time with little -quality impact. It caps **advisors only** — the acting aggregator's output (the -user-visible answer) is never capped. - -```yaml -moa: - presets: - fast: - reference_models: - - provider: openrouter - model: anthropic/claude-opus-4.8 - - provider: openrouter - model: openai/gpt-5.5 - aggregator: - provider: openrouter - model: anthropic/claude-opus-4.8 - reference_max_tokens: 600 # concise advice → faster turns -``` - -Leave it unset (or `0`/blank) to keep the prior uncapped behavior. +MoA uses provider-owned output limits. Preset and per-slot output-token cap +settings are no longer supported. Provider defaults vary; omission does not +always mean the model maximum. Native protocols that require an output limit +receive an internal value from Hermes. ### Advisor cadence with `fanout`