fix: remove dedicated user-facing output cap controls
This commit is contained in:
+1
-1
@@ -85,7 +85,7 @@ cache break — keep it the only one. Full detail:
|
||||
`agent/model_metadata.py` holds context lengths and capabilities.
|
||||
- **Auxiliary (side-LLM) work** — curator, vision, embedding, title generation, session_search,
|
||||
compression — resolves through `agent/auxiliary_client.py::_resolve_auto_route`; each task can pin
|
||||
its own `provider/model/base_url/max_tokens/reasoning_effort` under `auxiliary:` in config.yaml.
|
||||
its own `provider/model/base_url/reasoning_effort` under `auxiliary:` in config.yaml.
|
||||
- Fallback models and credential pools are resolution-chain code: E2E them with real imports
|
||||
against a temp `HERMES_HOME`, not mocks (root rubric).
|
||||
|
||||
|
||||
@@ -1684,18 +1684,8 @@ def _resolve_context_length(agent, _agent_cfg, base_url):
|
||||
except (TypeError, ValueError):
|
||||
agent._aux_compression_context_length_config = None
|
||||
|
||||
# model.max_tokens from config when the caller did not pass one.
|
||||
_model_cfg = _agent_cfg.get("model", {})
|
||||
_model_section = _model_cfg if isinstance(_model_cfg, dict) else {}
|
||||
_config_max_tokens = _model_section.get("max_tokens")
|
||||
if agent.max_tokens is None and _config_max_tokens is not None:
|
||||
agent.max_tokens = _positive_int(_config_max_tokens, reject=(bool,))
|
||||
if agent.max_tokens is None:
|
||||
_warn_invalid_config_int(
|
||||
"model.max_tokens in config.yaml", _config_max_tokens,
|
||||
"must be a positive integer (e.g. 4096)", "provider default",
|
||||
)
|
||||
agent._session_init_model_config["max_tokens"] = agent.max_tokens
|
||||
|
||||
_config_context_length = _model_section.get("context_length")
|
||||
if _config_context_length is not None:
|
||||
|
||||
+15
-34
@@ -1729,8 +1729,8 @@ class _BedrockCompletionsAdapter:
|
||||
)
|
||||
response = call_converse(
|
||||
region=self._region, model=model, messages=kwargs.get("messages", []), tools=kwargs.get("tools"),
|
||||
# Omitted/None cap → None so Bedrock uses the model max (no-cap-by-default like
|
||||
# every other aux wire). Truthiness mirrors the Anthropic shim: explicit 0 = "no cap".
|
||||
# Converse specifically defaults to the model maximum when omitted.
|
||||
# Truthiness mirrors the Anthropic shim: explicit 0 means omit.
|
||||
max_tokens=int(max_tokens) if max_tokens else None, temperature=kwargs.get("temperature"),
|
||||
top_p=kwargs.get("top_p"), stop_sequences=stop,
|
||||
)
|
||||
@@ -3602,8 +3602,6 @@ def _fallback_request_kwargs(
|
||||
destination.provider, destination.model, fallback_messages,
|
||||
temperature=temperature, max_tokens=fallback_max_tokens, tools=fallback_tools, timeout=effective_timeout,
|
||||
extra_body=fallback_extra_body, reasoning_config=reasoning_config, base_url=destination.base_url, task=task)
|
||||
if apply_fast_lane and fallback_max_tokens is not None and max_tokens is None:
|
||||
fb_kwargs.update(auxiliary_max_tokens_param(fallback_max_tokens, model=destination.model))
|
||||
return fb_kwargs
|
||||
|
||||
|
||||
@@ -5547,41 +5545,29 @@ def _get_auxiliary_task_config(task: str) -> Dict[str, Any]:
|
||||
|
||||
|
||||
class CompressionFastLane(NamedTuple):
|
||||
"""Explicit, non-reasoning compression route safe for a bounded summary."""
|
||||
"""Explicit, non-reasoning compression route."""
|
||||
|
||||
certified_non_reasoning: bool
|
||||
max_tokens: Optional[int]
|
||||
reasoning_config: Optional[Dict[str, Any]]
|
||||
|
||||
|
||||
def _fast_lane_config_fields(config: Dict[str, Any]) -> tuple[str, str, bool, Optional[int]]:
|
||||
"""``(provider, model, non_reasoning, cap)`` from one task config.
|
||||
|
||||
``non_reasoning`` only when ``reasoning_effort`` EXPLICITLY disables thinking (unset is NOT
|
||||
non-reasoning); ``cap`` is a positive int ``max_output_tokens`` or None — booleans are config
|
||||
drift, never a cap (``int(True) == 1``).
|
||||
"""
|
||||
def _fast_lane_config_fields(config: Dict[str, Any]) -> tuple[str, str, bool]:
|
||||
"""Only explicit reasoning disablement certifies a non-reasoning route."""
|
||||
from hermes_constants import parse_reasoning_effort
|
||||
provider = str(config.get("provider") or "").strip().lower()
|
||||
model = str(config.get("model") or "").strip()
|
||||
parsed_effort = parse_reasoning_effort(config.get("reasoning_effort"))
|
||||
non_reasoning = parsed_effort is not None and parsed_effort.get("enabled") is False
|
||||
raw_cap = config.get("max_output_tokens")
|
||||
try:
|
||||
cap = 0 if isinstance(raw_cap, bool) else int(raw_cap or 0)
|
||||
except (TypeError, ValueError):
|
||||
cap = 0
|
||||
return provider, model, non_reasoning, (cap if cap > 0 else None)
|
||||
return provider, model, non_reasoning
|
||||
|
||||
|
||||
def resolve_compression_fast_lane(
|
||||
actual_provider: str, actual_model: Optional[str], *, requested_provider: Optional[str] = None,
|
||||
requested_model: Optional[str] = None, route_config: Optional[Dict[str, Any]] = None,
|
||||
) -> CompressionFastLane:
|
||||
"""Certify the opt-in fast lane: capped only when an explicit, operator-certified
|
||||
non-reasoning provider/model exactly matches the route actually called."""
|
||||
"""Certify explicit non-reasoning settings only on the matching destination."""
|
||||
config = route_config if route_config is not None else _get_auxiliary_task_config("compression")
|
||||
cfg_provider, cfg_model, non_reasoning, cap = _fast_lane_config_fields(config)
|
||||
cfg_provider, cfg_model, non_reasoning = _fast_lane_config_fields(config)
|
||||
provider = str(requested_provider or "").strip().lower() or cfg_provider
|
||||
model = str(requested_model or "").strip() or cfg_model
|
||||
explicit_route = provider not in {"", "auto"} and model.lower() not in {"", "auto"}
|
||||
@@ -5589,14 +5575,14 @@ def resolve_compression_fast_lane(
|
||||
provider_matches = actual_norm == _normalize_aux_provider(provider)
|
||||
model_matches = str(actual_model or "").strip().lower() == model.lower()
|
||||
if explicit_route and provider_matches and model_matches and non_reasoning:
|
||||
return CompressionFastLane(True, cap, {"enabled": False, "effort": "none"})
|
||||
return CompressionFastLane(False, None, None)
|
||||
return CompressionFastLane(True, {"enabled": False, "effort": "none"})
|
||||
return CompressionFastLane(False, None)
|
||||
|
||||
|
||||
def _compression_config_claims_fast_lane(config: Dict[str, Any]) -> bool:
|
||||
"""Whether task config declares fast-only controls that cannot leak."""
|
||||
provider, model, non_reasoning, cap = _fast_lane_config_fields(config)
|
||||
return provider not in {"", "auto"} and model.lower() not in {"", "auto"} and non_reasoning and cap is not None
|
||||
provider, model, non_reasoning = _fast_lane_config_fields(config)
|
||||
return provider not in {"", "auto"} and model.lower() not in {"", "auto"} and non_reasoning
|
||||
|
||||
|
||||
def _compression_fast_lane_controls(
|
||||
@@ -5616,7 +5602,7 @@ def _compression_fast_lane_controls(
|
||||
body["reasoning"] = lane.reasoning_config
|
||||
elif _compression_config_claims_fast_lane(leak_guard_config):
|
||||
body.pop("reasoning", None)
|
||||
return lane.max_tokens, body
|
||||
return max_tokens, body
|
||||
|
||||
|
||||
def _get_task_timeout(task: str, default: float = _DEFAULT_AUX_TIMEOUT) -> float:
|
||||
@@ -5821,7 +5807,7 @@ def _is_gemini_native_route(provider_norm: str, effective_base: str) -> bool:
|
||||
def _forwards_max_tokens(provider: str, provider_norm: str, model: str, effective_base: str, task: Optional[str]) -> bool:
|
||||
"""Whether an explicit max_tokens is forwarded on this route.
|
||||
|
||||
No default cap elsewhere (omitted = model max; avoids max_completion_tokens / ZAI-vision
|
||||
No default cap elsewhere (omitted = provider default; avoids max_completion_tokens / ZAI-vision
|
||||
quirks). Forward only where mandatory or honored: Anthropic Messages wire (400 without it);
|
||||
NVIDIA NIM (empty choices[] when omitted); MoA reference slots; Gemini native (fixed 65,535
|
||||
ceiling otherwise); OpenRouter (budgets the FULL window when omitted → 402 on low credit);
|
||||
@@ -6542,10 +6528,9 @@ def _prepare_aux_request(
|
||||
)
|
||||
effective_timeout = _effective_aux_timeout(task, timeout)
|
||||
request_provider = effective_provider or resolved_provider
|
||||
fast_compression_cap = None
|
||||
if not async_mode:
|
||||
compression_config = _get_auxiliary_task_config("compression") if task == "compression" else {}
|
||||
fast_compression_cap, effective_extra_body = _compression_fast_lane_controls(
|
||||
_, effective_extra_body = _compression_fast_lane_controls(
|
||||
task, actual_provider=request_provider, actual_model=final_model,
|
||||
requested_provider=provider, requested_model=model, route_config=compression_config,
|
||||
leak_guard_config=compression_config, max_tokens=max_tokens,
|
||||
@@ -6567,10 +6552,6 @@ def _prepare_aux_request(
|
||||
request_provider, final_model, messages, temperature=temperature, max_tokens=max_tokens,
|
||||
tools=tools, timeout=effective_timeout, extra_body=effective_extra_body,
|
||||
reasoning_config=reasoning_config, base_url=base_info or resolved_base_url, task=task)
|
||||
if fast_compression_cap is not None and max_tokens is None:
|
||||
# The compression route is certified non-reasoning, so a bounded summary is
|
||||
# intentional; explicit caller caps pass through untouched.
|
||||
kwargs.update(auxiliary_max_tokens_param(fast_compression_cap, model=final_model))
|
||||
if extra_headers:
|
||||
kwargs["extra_headers"] = dict(extra_headers)
|
||||
# Convert image blocks for Anthropic-compatible endpoints (e.g. MiniMax)
|
||||
|
||||
@@ -235,7 +235,7 @@ def _resolve_review_runtime(agent: Any, task_cfg: Optional[Dict[str, Any]] = Non
|
||||
"provider": rp.get("provider") or task_provider, "model": rp.get("model") or task_model,
|
||||
**{key: rp.get(key) for key in ("api_key", "base_url", "api_mode", "credential_pool", "command")},
|
||||
"request_overrides": dict(rp.get("request_overrides") or {}),
|
||||
"max_tokens": rp.get("max_output_tokens"), "args": list(rp.get("args") or []), "routed": True,
|
||||
"args": list(rp.get("args") or []), "routed": True,
|
||||
}
|
||||
except Exception as e:
|
||||
logger.debug("background-review aux routing failed (%s); using main model", e)
|
||||
@@ -813,8 +813,7 @@ def _fork_init_kwargs(agent: Any, rt: Dict[str, Any], routed: bool, max_iteratio
|
||||
"enabled_toolsets": getattr(agent, "enabled_toolsets", None),
|
||||
"disabled_toolsets": getattr(agent, "disabled_toolsets", None), "skip_memory": True,
|
||||
}
|
||||
if isinstance(rt.get("max_tokens"), int):
|
||||
kwargs["max_tokens"] = rt["max_tokens"]
|
||||
|
||||
if isinstance(rt.get("command"), str) and rt["command"]:
|
||||
kwargs.update(acp_command=rt["command"], acp_args=rt.get("args") or [])
|
||||
if not routed:
|
||||
|
||||
@@ -1256,7 +1256,7 @@ def _build_anthropic_kwargs(agent, api_messages, tools_for_api, reasoning_config
|
||||
def _build_bedrock_kwargs(agent, api_messages, tools_for_api):
|
||||
# Bedrock Converse — the adapter converts messages/tools and calls boto3 directly.
|
||||
return agent._get_transport().build_kwargs(model=agent.model, messages=api_messages, tools=tools_for_api,
|
||||
max_tokens=agent.max_tokens or 4096, region=getattr(agent, "_bedrock_region", None) or "us-east-1",
|
||||
max_tokens=agent.max_tokens, region=getattr(agent, "_bedrock_region", None) or "us-east-1",
|
||||
guardrail_config=getattr(agent, "_bedrock_guardrail_config", None))
|
||||
|
||||
|
||||
@@ -1292,18 +1292,6 @@ def _build_codex_kwargs(agent, api_messages, tools_for_api, reasoning_config, re
|
||||
context_management=context_management)
|
||||
|
||||
|
||||
def _anthropic_max_output_for_model(agent):
|
||||
"""Anthropic-compatible max-output fallback (last resort in build_kwargs, never
|
||||
overriding an explicit value). Model-gated, not URL-gated: any proxy serving a
|
||||
Claude/MiniMax/Qwen3 model needs max_tokens (Messages API treats it as
|
||||
mandatory; proxies that omit it default as low as 4096)."""
|
||||
with contextlib.suppress(Exception):
|
||||
from agent.anthropic_adapter import _get_anthropic_max_output, _ANTHROPIC_OUTPUT_LIMITS
|
||||
model_norm = (agent.model or "").lower().replace(".", "-")
|
||||
if any(key in model_norm for key in _ANTHROPIC_OUTPUT_LIMITS):
|
||||
return _get_anthropic_max_output(agent.model)
|
||||
return None
|
||||
|
||||
|
||||
def _build_chat_completions_kwargs(agent, api_messages, tools_for_api, reasoning_config, request_overrides, cache_scope_id):
|
||||
transport = agent._get_transport()
|
||||
@@ -1325,7 +1313,7 @@ def _build_chat_completions_kwargs(agent, api_messages, tools_for_api, reasoning
|
||||
_fixed_temp = None if _omit_temp else _ft
|
||||
|
||||
_prefs = _provider_preferences_for_agent(agent)
|
||||
_ant_max = _anthropic_max_output_for_model(agent)
|
||||
|
||||
_qwen_meta = {"sessionId": agent.session_id or "hermes", "promptId": str(uuid.uuid4())} if _is_qwen else None
|
||||
_profile = None
|
||||
with contextlib.suppress(Exception):
|
||||
@@ -1342,7 +1330,7 @@ def _build_chat_completions_kwargs(agent, api_messages, tools_for_api, reasoning
|
||||
request_overrides=request_overrides, session_id=getattr(agent, "session_id", None),
|
||||
cache_scope_id=cache_scope_id, ollama_num_ctx=agent._ollama_num_ctx,
|
||||
provider_preferences=_prefs or None, openrouter_min_coding_score=agent.openrouter_min_coding_score,
|
||||
anthropic_max_output=_ant_max, supports_reasoning=agent._supports_reasoning_extra_body(),
|
||||
supports_reasoning=agent._supports_reasoning_extra_body(),
|
||||
qwen_session_metadata=_qwen_meta)
|
||||
if _profile:
|
||||
# Profiles handle per-provider quirks via hooks fed the context above.
|
||||
|
||||
+1
-1
@@ -1023,7 +1023,7 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
|
||||
result_meta["model"], result_meta["provider"] = model_name, provider or ""
|
||||
review_agent = None
|
||||
try:
|
||||
agent_kwargs: Dict[str, Any] = {"max_tokens": rp["max_output_tokens"]} if isinstance(rp.get("max_output_tokens"), int) else {}
|
||||
agent_kwargs: Dict[str, Any] = {}
|
||||
acp_command = rp.get("command")
|
||||
if isinstance(acp_command, str) and acp_command:
|
||||
agent_kwargs.update(acp_command=acp_command, acp_args=list(rp.get("args") or []))
|
||||
|
||||
+5
-6
@@ -380,15 +380,14 @@ def _run_reference(
|
||||
messages, slot, runtime, reserve_output_tokens=max_tokens, context_length_cache=context_length_cache,
|
||||
)
|
||||
trimmed = _maybe_apply_moa_cache_control(trimmed, _with_cache_disabled(runtime, cache_disabled), cache_ttl=cache_ttl)
|
||||
# Per-slot max_tokens beats the preset-level reference_max_tokens.
|
||||
slot_max_tokens = slot.get("max_tokens")
|
||||
|
||||
# Copilot gates premium models on request attribution; MoA fan-out serves the
|
||||
# user's current turn, so mirror the main agent's x-initiator header.
|
||||
from agent.auxiliary_client import _normalize_aux_provider
|
||||
is_copilot = _normalize_aux_provider(str(runtime.get("provider") or "")) in ("copilot", "copilot-acp")
|
||||
response = call_llm(
|
||||
task="moa_reference", messages=trimmed, temperature=temperature,
|
||||
max_tokens=slot_max_tokens if slot_max_tokens is not None else max_tokens,
|
||||
max_tokens=max_tokens,
|
||||
timeout=reference_timeout, reasoning_config=_slot_reasoning_config(slot),
|
||||
extra_headers={"x-initiator": "user"} if is_copilot else None, **runtime,
|
||||
)
|
||||
@@ -790,8 +789,8 @@ def aggregate_moa_context(
|
||||
long syntheses). ``agent`` makes the fan-out interruptible.
|
||||
|
||||
``reference_max_tokens`` applies ONLY to the reference fan-out — the aggregator's own synthesis call is
|
||||
never capped, so it always uses its model's own maximum. ``call_llm`` omits the parameter entirely when
|
||||
it is ``None`` (see its docstring), which also sidesteps providers that reject ``max_tokens`` outright.
|
||||
not given an advisor budget. Omission uses provider-specific defaults; native protocols may
|
||||
still require an internal wire limit.
|
||||
A hardcoded cap on the aggregator call previously truncated long aggregator syntheses (#53580) — passing
|
||||
``reference_max_tokens`` to both calls here would silently reintroduce that regression.
|
||||
"""
|
||||
@@ -1198,7 +1197,7 @@ class MoAChatCompletions:
|
||||
raw_reference_timeout = preset.get("reference_timeout")
|
||||
reference_outputs = _run_references_parallel(
|
||||
reference_models, ref_messages, temperature=_preset_temperature(preset, "reference_temperature"),
|
||||
max_tokens=preset.get("reference_max_tokens"),
|
||||
|
||||
progress_callback=lambda done, total, label: self._emit("moa.progress", refs_done=done, refs_total=total, label=label),
|
||||
reference_timeout=float(raw_reference_timeout) if raw_reference_timeout else None,
|
||||
agent=self._agent, late_accounting_sink=self._record_late_reference_accounting,
|
||||
|
||||
+2
-2
@@ -511,7 +511,7 @@ def lookup_models_dev_context(provider: str, model: str, *, allow_network: bool
|
||||
|
||||
|
||||
# Per-model overrides (config.yaml → model_overrides). Canonical schema (the ONLY key space consumers
|
||||
# accept): context_window, max_output_tokens, supports_tools, supports_vision, supports_reasoning,
|
||||
# accept): context_window, supports_tools, supports_vision, supports_reasoning,
|
||||
# model_family. ``<provider>.<model_id>`` is an explicit partial patch that always wins over the
|
||||
# catalog. ``<provider>._default`` / top-level ``_default`` are FILL-GAP defaults: they apply ONLY to
|
||||
# models the catalog does not know and never displace catalog data. Provider keys accept the Hermes
|
||||
@@ -605,7 +605,7 @@ def _override_to_catalog_shape(override: Dict[str, Any]) -> Tuple[Dict[str, Any]
|
||||
vision is out-of-band because it maps onto the ``modalities.input`` list rather than a scalar field."""
|
||||
patch: Dict[str, Any] = {}
|
||||
limit = {
|
||||
catalog_key: value for catalog_key, override_key in (("context", "context_window"), ("output", "max_output_tokens"))
|
||||
catalog_key: value for catalog_key, override_key in (("context", "context_window"),)
|
||||
if (value := _override_int(override, override_key)) is not None
|
||||
}
|
||||
if limit:
|
||||
|
||||
@@ -37,11 +37,11 @@ class BedrockTransport(ProviderTransport):
|
||||
self, model: str, messages: List[Dict[str, Any]],
|
||||
tools: Optional[List[Dict[str, Any]]] = None, **params,
|
||||
) -> Dict[str, Any]:
|
||||
"""Build converse() kwargs; params: max_tokens (4096), temperature, guardrail_config, region ('us-east-1')."""
|
||||
"""Build Converse kwargs, leaving the optional output limit to the provider."""
|
||||
from agent.bedrock_adapter import build_converse_kwargs
|
||||
|
||||
kwargs = build_converse_kwargs(
|
||||
model=model, messages=messages, tools=tools, max_tokens=params.get("max_tokens", 4096),
|
||||
model=model, messages=messages, tools=tools, max_tokens=params.get("max_tokens"),
|
||||
temperature=params.get("temperature"), guardrail_config=params.get("guardrail_config"),
|
||||
)
|
||||
# Sentinel keys for dispatch — agent pops these before the boto3 call
|
||||
|
||||
@@ -250,7 +250,7 @@ def _swap_developer_role(sanitized: list, model_lower: str) -> list:
|
||||
|
||||
|
||||
def _apply_max_tokens(api_kwargs: dict, model: str, reasoning_config: Any, params: dict, profile_max: Any = None) -> None:
|
||||
"""Resolve max_tokens — priority: ephemeral > user > profile default > anthropic_max_output."""
|
||||
"""Preserve internal task/recovery budgets and provider protocol exceptions."""
|
||||
max_tokens_fn = params.get("max_tokens_param_fn")
|
||||
for candidate in (params.get("ephemeral_max_output_tokens"), params.get("max_tokens")):
|
||||
if candidate is not None and max_tokens_fn:
|
||||
@@ -258,8 +258,7 @@ def _apply_max_tokens(api_kwargs: dict, model: str, reasoning_config: Any, param
|
||||
return
|
||||
if profile_max and max_tokens_fn:
|
||||
api_kwargs.update(max_tokens_fn(_raise_gemini_thinking_max_tokens(model, reasoning_config, profile_max)))
|
||||
elif params.get("anthropic_max_output") is not None:
|
||||
api_kwargs["max_tokens"] = params["anthropic_max_output"]
|
||||
|
||||
|
||||
|
||||
def _base_kwargs(model: str, sanitized: list, tools: Any, params: dict, profile: Any = None) -> dict[str, Any]:
|
||||
|
||||
@@ -57,7 +57,7 @@ def _append_moa_context(agent: Any, api_messages: Any, moa_config: Any, original
|
||||
aggregator=moa_config.get("aggregator") or {},
|
||||
temperature=_preset_temperature(moa_config, "reference_temperature"),
|
||||
aggregator_temperature=_preset_temperature(moa_config, "aggregator_temperature"),
|
||||
reference_max_tokens=moa_config.get("reference_max_tokens"),
|
||||
|
||||
# None = no per-preset override; inherit auxiliary.moa_reference.timeout.
|
||||
reference_timeout=(
|
||||
float(moa_config["reference_timeout"])
|
||||
|
||||
@@ -424,7 +424,7 @@ describe('ModelSettings MoA preset editor', () => {
|
||||
aggregator: { provider: 'openrouter', model: 'anthropic/claude-opus-4.8' },
|
||||
reference_temperature: 0,
|
||||
aggregator_temperature: 0,
|
||||
max_tokens: 4096,
|
||||
|
||||
enabled: true
|
||||
}
|
||||
},
|
||||
@@ -435,7 +435,7 @@ describe('ModelSettings MoA preset editor', () => {
|
||||
aggregator: { provider: 'openrouter', model: 'anthropic/claude-opus-4.8' },
|
||||
reference_temperature: 0,
|
||||
aggregator_temperature: 0,
|
||||
max_tokens: 4096,
|
||||
|
||||
enabled: true
|
||||
})
|
||||
|
||||
|
||||
@@ -1437,11 +1437,10 @@ export interface MoaConfigResponse {
|
||||
aggregator_temperature: number
|
||||
degraded_reference_policy: 'loud' | 'silent'
|
||||
enabled: boolean
|
||||
max_tokens: number
|
||||
|
||||
reference_models: MoaModelSlot[]
|
||||
reference_temperature: number
|
||||
/** Optional advisor output cap — round-tripped, not edited here. */
|
||||
reference_max_tokens?: number | null
|
||||
|
||||
/** Fan-out cadence (user_turn default | per_iteration | every_n:N) — round-tripped. */
|
||||
fanout?: string
|
||||
reference_timeout: number | null
|
||||
@@ -1451,7 +1450,7 @@ export interface MoaConfigResponse {
|
||||
aggregator_temperature: number
|
||||
degraded_reference_policy: 'loud' | 'silent'
|
||||
enabled: boolean
|
||||
max_tokens: number
|
||||
|
||||
reference_models: MoaModelSlot[]
|
||||
reference_temperature: number
|
||||
reference_timeout: number | null
|
||||
|
||||
+8
-8
@@ -51,12 +51,12 @@ _RUNNER_FIELDS = (
|
||||
"batch_size", "run_name", "distribution", "max_iterations", "base_url", "api_key", "model",
|
||||
"num_workers", "verbose", "ephemeral_system_prompt", "log_prefix_chars", "providers_allowed",
|
||||
"providers_ignored", "providers_order", "provider_sort", "openrouter_min_coding_score",
|
||||
"max_tokens", "reasoning_config", "prefill_messages", "max_samples",
|
||||
"reasoning_config", "prefill_messages", "max_samples",
|
||||
)
|
||||
# BatchRunner attributes forwarded verbatim to every AIAgent in the worker config.
|
||||
_AGENT_PASSTHROUGH = (
|
||||
"base_url", "api_key", "ephemeral_system_prompt", "providers_allowed", "providers_ignored",
|
||||
"providers_order", "provider_sort", "openrouter_min_coding_score", "max_tokens",
|
||||
"providers_order", "provider_sort", "openrouter_min_coding_score",
|
||||
"reasoning_config", "prefill_messages",
|
||||
)
|
||||
|
||||
@@ -427,7 +427,7 @@ class BatchRunner:
|
||||
providers_order: List[str] = None,
|
||||
provider_sort: str = None,
|
||||
openrouter_min_coding_score: Optional[float] = None,
|
||||
max_tokens: int = None,
|
||||
|
||||
reasoning_config: Dict[str, Any] = None,
|
||||
prefill_messages: List[Dict[str, Any]] = None,
|
||||
max_samples: int = None,
|
||||
@@ -861,7 +861,7 @@ def main(
|
||||
providers_ignored: str = None,
|
||||
providers_order: str = None,
|
||||
provider_sort: str = None,
|
||||
max_tokens: int = None,
|
||||
|
||||
reasoning_effort: str = None,
|
||||
reasoning_disabled: bool = False,
|
||||
prefill_messages_file: str = None,
|
||||
@@ -889,7 +889,7 @@ def main(
|
||||
providers_ignored (str): Comma-separated list of OpenRouter providers to ignore (e.g. "together,deepinfra")
|
||||
providers_order (str): Comma-separated list of OpenRouter providers to try in order (e.g. "anthropic,openai,google")
|
||||
provider_sort (str): Sort providers by "price", "throughput", or "latency" (OpenRouter only)
|
||||
max_tokens (int): Maximum tokens for model responses (optional, uses model default if not set)
|
||||
|
||||
reasoning_effort (str): Reasoning effort: "none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra" (default: "medium")
|
||||
reasoning_disabled (bool): Completely disable reasoning/thinking tokens (default: False)
|
||||
prefill_messages_file (str): Path to JSON file containing prefill messages (list of {role, content} dicts)
|
||||
@@ -905,9 +905,9 @@ def main(
|
||||
# Use specific distribution
|
||||
python batch_runner.py --dataset_file=data.jsonl --batch_size=10 --run_name=image_test --distribution=image_gen
|
||||
|
||||
# With disabled reasoning and max tokens
|
||||
# With disabled reasoning
|
||||
python batch_runner.py --dataset_file=data.jsonl --batch_size=10 --run_name=my_run \\
|
||||
--reasoning_disabled --max_tokens=128000
|
||||
--reasoning_disabled
|
||||
|
||||
# With prefill messages from file
|
||||
python batch_runner.py --dataset_file=data.jsonl --batch_size=10 --run_name=my_run \\
|
||||
@@ -981,7 +981,7 @@ def main(
|
||||
providers_ignored=_split_csv(providers_ignored),
|
||||
providers_order=_split_csv(providers_order),
|
||||
provider_sort=provider_sort,
|
||||
max_tokens=max_tokens,
|
||||
|
||||
reasoning_config=reasoning_config,
|
||||
prefill_messages=prefill_messages,
|
||||
max_samples=max_samples,
|
||||
|
||||
@@ -111,14 +111,8 @@ model:
|
||||
#
|
||||
# context_length: 131072
|
||||
#
|
||||
# max_tokens: OUTPUT cap — maximum tokens the model may generate per response.
|
||||
# Unrelated to how long your conversation history can be.
|
||||
# The OpenAI-standard name "max_tokens" is a misnomer; Anthropic's native
|
||||
# API has since renamed it "max_output_tokens" for clarity.
|
||||
# Leave unset to use the model's native output ceiling (recommended).
|
||||
# Set only if you want to deliberately limit individual response length.
|
||||
#
|
||||
# max_tokens: 8192
|
||||
# Output-token limits are provider-owned, not user configuration. Native
|
||||
# protocols requiring a limit receive an internal value from Hermes.
|
||||
|
||||
# ── Custom request headers (optional) ─────────────────────────────────────
|
||||
#
|
||||
|
||||
@@ -2669,9 +2669,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix
|
||||
# A ``moa:<preset>`` model string selects the MoA virtual provider in one shot (parity with
|
||||
# interactive ``/moa`` and the model picker). See #56828.
|
||||
_moa_provider_override, self.model = _normalize_moa_model(self.model)
|
||||
_env_mt = os.environ.get("HERMES_MAX_TOKENS")
|
||||
_mt = _model_config.get("max_tokens")
|
||||
self.max_tokens = _int_or(_env_mt, None) if _env_mt else (_mt if isinstance(_mt, int) else None)
|
||||
|
||||
if self.model == "": # auto-detect from a local server
|
||||
_base_url = _model_config.get("base_url") or ""
|
||||
if base_url_hostname(_base_url) in ("localhost", "127.0.0.1"):
|
||||
|
||||
@@ -58,7 +58,7 @@ echo "✅ Done. Log: $LOG_FILE"
|
||||
#
|
||||
# --resume Resume from checkpoint if interrupted
|
||||
# --verbose Enable detailed logging
|
||||
# --max_tokens=63000 Set max response tokens
|
||||
|
||||
# --reasoning_disabled Disable model thinking/reasoning tokens
|
||||
# --providers_allowed="anthropic,google" Restrict to specific providers
|
||||
# --prefill_messages_file="configs/prefill.json" Few-shot priming
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
"""Zero-cost local SDK wire capture. Fixtures are not vendor inference evidence.
|
||||
|
||||
Run from a checkout with HERMES_HOME pointing to a disposable directory.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import threading
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
|
||||
from openai import OpenAI
|
||||
from anthropic import Anthropic
|
||||
|
||||
captures = []
|
||||
|
||||
|
||||
class Capture(BaseHTTPRequestHandler):
|
||||
def do_GET(self):
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.end_headers()
|
||||
self.wfile.write(json.dumps({"data": [{"id": "fixture", "context_length": 131072}]}).encode())
|
||||
|
||||
def do_POST(self):
|
||||
body = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
|
||||
captures.append({"path": self.path, "body": body})
|
||||
if "model" not in body:
|
||||
reply = {}
|
||||
elif self.path.endswith("messages"):
|
||||
reply = {"id": "fixture", "type": "message", "role": "assistant", "model": body["model"], "content": [{"type": "text", "text": "LOCAL_CAPTURE_ONLY"}], "stop_reason": "end_turn", "usage": {"input_tokens": 1, "output_tokens": 1}}
|
||||
else:
|
||||
reply = {"id": "fixture", "object": "chat.completion", "created": 0, "model": body["model"], "choices": [{"index": 0, "message": {"role": "assistant", "content": "LOCAL_CAPTURE_ONLY"}, "finish_reason": "stop"}], "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.end_headers()
|
||||
self.wfile.write(json.dumps(reply).encode())
|
||||
|
||||
def log_message(self, format, *args):
|
||||
pass
|
||||
|
||||
|
||||
server = ThreadingHTTPServer(("127.0.0.1", 0), Capture)
|
||||
thread = threading.Thread(target=server.serve_forever, daemon=True)
|
||||
thread.start()
|
||||
url = f"http://127.0.0.1:{server.server_port}"
|
||||
home = Path(os.environ["HERMES_HOME"])
|
||||
home.mkdir(parents=True, exist_ok=True)
|
||||
config = {"model": {"default": "fixture", "provider": "fixture-local", "max_tokens": 17}, "providers": {"fixture-local": {"api": url + "/v1", "api_key": "fixture", "max_output_tokens": 19}}}
|
||||
(home / "config.yaml").write_text(json.dumps(config))
|
||||
os.environ["HERMES_MAX_TOKENS"] = "13"
|
||||
from gateway.run import _resolve_runtime_agent_kwargs
|
||||
from gateway.platforms.api_server import _resolve_request_runtime_agent_kwargs
|
||||
from hermes_cli.runtime_provider import resolve_runtime_provider
|
||||
from hermes_cli.moa_config import _normalize_preset
|
||||
from agent.transports.chat_completions import ChatCompletionsTransport
|
||||
from agent.transports.anthropic import AnthropicTransport
|
||||
from agent.transports.bedrock import BedrockTransport
|
||||
|
||||
messages = [{"role": "user", "content": "fixture"}]
|
||||
client = OpenAI(api_key="fixture", base_url=url + "/v1", max_retries=0)
|
||||
results = {}
|
||||
for label, runtime in [("gateway", _resolve_runtime_agent_kwargs()), ("api-server", _resolve_request_runtime_agent_kwargs("fixture-local", "fixture")), ("provider", resolve_runtime_provider(requested="fixture-local", target_model="fixture"))]:
|
||||
cap = runtime.get("max_tokens", runtime.get("max_output_tokens"))
|
||||
kwargs = ChatCompletionsTransport().build_kwargs("fixture", messages, max_tokens=cap, max_tokens_param_fn=lambda value: {"max_tokens": value})
|
||||
client.chat.completions.create(**kwargs)
|
||||
results[label] = captures[-1]
|
||||
for label, params in [("compatible-claude", {"anthropic_max_output": 65536}), ("internal-budget", {"max_tokens": 43})]:
|
||||
kwargs = ChatCompletionsTransport().build_kwargs("claude-fixture", messages, max_tokens_param_fn=lambda value: {"max_tokens": value}, **params)
|
||||
client.chat.completions.create(**kwargs)
|
||||
results[label] = captures[-1]
|
||||
from providers import get_provider_profile
|
||||
from run_agent import AIAgent
|
||||
agent = AIAgent(model="fixture", provider="custom", base_url=url + "/v1", api_key="fixture", quiet_mode=True, skip_memory=True, skip_context_files=True, enabled_toolsets=[])
|
||||
actual_kwargs = agent._build_api_kwargs(messages, tools_for_api=[])
|
||||
client.chat.completions.create(**actual_kwargs)
|
||||
results["agent-main-custom"] = captures[-1]
|
||||
kwargs = ChatCompletionsTransport().build_kwargs("fixture", messages, provider_profile=get_provider_profile("custom"), max_tokens_param_fn=lambda value: {"max_tokens": value})
|
||||
client.chat.completions.create(**kwargs)
|
||||
results["registered-custom-profile"] = captures[-1]
|
||||
from agent.auxiliary_client import _compression_fast_lane_controls
|
||||
from tools.delegate_tool_config import _resolve_delegation_credentials
|
||||
route = {"provider": "custom", "model": "fixture", "reasoning_effort": "none", "max_output_tokens": 47}
|
||||
compression_cap, _ = _compression_fast_lane_controls(
|
||||
"compression", actual_provider="custom", actual_model="fixture", requested_provider="custom",
|
||||
requested_model="fixture", route_config=route, leak_guard_config=route, max_tokens=None, extra_body={},
|
||||
)
|
||||
child_credentials = _resolve_delegation_credentials({"provider": "fixture-local", "model": "fixture"}, agent)
|
||||
for label, cap in [("compression-config-helper", compression_cap), ("delegation-config-helper", child_credentials.get("max_output_tokens"))]:
|
||||
kwargs = ChatCompletionsTransport().build_kwargs("fixture", messages, max_tokens=cap, max_tokens_param_fn=lambda value: {"max_tokens": value})
|
||||
client.chat.completions.create(**kwargs)
|
||||
results[label] = captures[-1]
|
||||
kwargs = AnthropicTransport().build_kwargs("claude-sonnet-4-5", messages)
|
||||
kwargs.pop("__anthropic__", None)
|
||||
Anthropic(api_key="fixture", base_url=url, max_retries=0).messages.create(**kwargs)
|
||||
results["native-anthropic"] = captures[-1]
|
||||
results["bedrock-converse-built-not-sent"] = BedrockTransport().build_kwargs("amazon.nova-pro-v1:0", messages)
|
||||
results["moa-normalized"] = _normalize_preset({"max_tokens": 23, "reference_max_tokens": 29})
|
||||
print(json.dumps(results, indent=2, default=str))
|
||||
server.shutdown()
|
||||
server.server_close()
|
||||
thread.join()
|
||||
if os.environ.get("VERIFY_OUTPUT_CAP_REMOVAL"):
|
||||
for label in ("gateway", "api-server", "provider", "compatible-claude", "agent-main-custom", "registered-custom-profile", "compression-config-helper", "delegation-config-helper"):
|
||||
assert "max_tokens" not in results[label]["body"], label
|
||||
assert results["internal-budget"]["body"]["max_tokens"] == 43
|
||||
assert results["native-anthropic"]["body"]["max_tokens"] > 0
|
||||
assert "maxTokens" not in results["bedrock-converse-built-not-sent"].get("inferenceConfig", {})
|
||||
@@ -251,7 +251,7 @@ _REQUEST_OPTION_MISSING = object()
|
||||
# vocabulary clamping happens downstream in agent.reasoning_effort.
|
||||
_REASONING_EFFORTS = frozenset({"none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"})
|
||||
_RUNTIME_AGENT_OVERRIDE_KEYS = (
|
||||
"api_key", "base_url", "provider", "api_mode", "command", "args", "credential_pool", "max_tokens")
|
||||
"api_key", "base_url", "provider", "api_mode", "command", "args", "credential_pool")
|
||||
|
||||
|
||||
def _clean_request_string(value: Any) -> Optional[str]:
|
||||
@@ -313,24 +313,11 @@ def _resolve_request_runtime_agent_kwargs(provider: str, target_model: Optional[
|
||||
runtime = resolve_runtime_provider(requested=provider, target_model=target_model)
|
||||
except Exception as exc:
|
||||
raise RuntimeError(format_runtime_provider_error(exc)) from exc
|
||||
model_cfg = _get_model_config()
|
||||
max_tokens = None
|
||||
env_max_tokens = os.environ.get("HERMES_MAX_TOKENS")
|
||||
if env_max_tokens:
|
||||
with suppress(ValueError, TypeError):
|
||||
max_tokens = int(env_max_tokens)
|
||||
elif isinstance(model_cfg, dict):
|
||||
cfg_max_tokens = model_cfg.get("max_tokens")
|
||||
if isinstance(cfg_max_tokens, int):
|
||||
max_tokens = cfg_max_tokens
|
||||
if max_tokens is None:
|
||||
runtime_max_tokens = runtime.get("max_output_tokens")
|
||||
if isinstance(runtime_max_tokens, int) and runtime_max_tokens > 0:
|
||||
max_tokens = runtime_max_tokens
|
||||
|
||||
return {
|
||||
**{k: runtime.get(k) for k in ("api_key", "base_url", "provider", "api_mode", "command")},
|
||||
"args": list(runtime.get("args") or []),
|
||||
"credential_pool": runtime.get("credential_pool"), "max_tokens": max_tokens}
|
||||
"credential_pool": runtime.get("credential_pool")}
|
||||
|
||||
|
||||
def _request_agent_overrides(
|
||||
|
||||
+3
-18
@@ -2180,27 +2180,13 @@ def _resolve_runtime_agent_kwargs() -> dict:
|
||||
except Exception as exc:
|
||||
raise RuntimeError(format_runtime_provider_error(exc)) from exc
|
||||
|
||||
model_cfg = _get_model_config()
|
||||
max_tokens = None
|
||||
_env_mt = os.environ.get("HERMES_MAX_TOKENS")
|
||||
if _env_mt:
|
||||
with suppress(ValueError, TypeError):
|
||||
max_tokens = int(_env_mt)
|
||||
elif isinstance(model_cfg, dict):
|
||||
mt = model_cfg.get("max_tokens")
|
||||
max_tokens = mt if isinstance(mt, int) else None
|
||||
# Per-provider max_output_tokens applies only when global model.max_tokens is unset (global wins).
|
||||
if max_tokens is None:
|
||||
_runtime_mot = runtime.get("max_output_tokens")
|
||||
if isinstance(_runtime_mot, int) and _runtime_mot > 0:
|
||||
max_tokens = _runtime_mot
|
||||
|
||||
capabilities = runtime.get("capabilities")
|
||||
capabilities = (
|
||||
{k: v for k, v in capabilities.items() if isinstance(k, str) and isinstance(v, bool)}
|
||||
if isinstance(capabilities, dict) else {})
|
||||
|
||||
return {**_runtime_agent_kwargs(runtime), "max_tokens": max_tokens, "capabilities": capabilities}
|
||||
return {**_runtime_agent_kwargs(runtime), "capabilities": capabilities}
|
||||
|
||||
|
||||
def _runtime_agent_kwargs(runtime: dict) -> dict:
|
||||
@@ -2303,8 +2289,7 @@ def _resolve_runtime_agent_kwargs_for_provider(provider: str) -> dict:
|
||||
return {
|
||||
**_runtime_agent_kwargs(runtime),
|
||||
"request_overrides": dict(runtime.get("request_overrides") or {}),
|
||||
"capabilities": dict(runtime.get("capabilities") or {}),
|
||||
"max_tokens": runtime.get("max_output_tokens")}
|
||||
"capabilities": dict(runtime.get("capabilities") or {})}
|
||||
|
||||
|
||||
def _deep_merge_request_overrides(base: Optional[dict], override: Optional[dict]) -> dict:
|
||||
@@ -4169,7 +4154,7 @@ class GatewayRunner(
|
||||
# cached agent or a mid-gateway edit is silently ignored. Add new baked-in settings here.
|
||||
# _MAX_INTERRUPT_DEPTH = 3 # Cap recursive interrupt handling (#816)
|
||||
_CACHE_BUSTING_CONFIG_KEYS: tuple = (
|
||||
("model", "context_length"), ("model", "max_tokens"), ("compression", "enabled"),
|
||||
("model", "context_length"), ("compression", "enabled"),
|
||||
("compression", "progress_notices"), ("compression", "threshold"),
|
||||
("compression", "model_thresholds"), ("compression", "threshold_tokens"),
|
||||
("compression", "codex_gpt55_autoraise"), ("compression", "codex_app_server_auto"),
|
||||
|
||||
@@ -518,7 +518,7 @@ class CLIAgentSetupMixin:
|
||||
requested_provider=runtime.get("requested_provider"),
|
||||
api_mode=runtime.get("api_mode"), acp_command=runtime.get("command"),
|
||||
acp_args=runtime.get("args"), credential_pool=runtime.get("credential_pool"),
|
||||
max_tokens=self.max_tokens, max_iterations=self.max_turns,
|
||||
max_iterations=self.max_turns,
|
||||
run_budget_seconds=getattr(self, "run_budget_seconds", None),
|
||||
enabled_toolsets=self.enabled_toolsets, disabled_toolsets=self.disabled_toolsets,
|
||||
verbose_logging=self.verbose, quiet_mode=not self.verbose,
|
||||
|
||||
@@ -695,9 +695,8 @@ DEFAULT_CONFIG = {
|
||||
# OpenAI-compatible request fields. Vision: download_timeout = image HTTP download (s).
|
||||
"vision": _aux(120, download_timeout=30),
|
||||
# web_extract and session_search no longer use an aux LLM; leftover blocks in user config
|
||||
# are ignored. Compression: raise timeout for local models. max_output_tokens is only
|
||||
# honored with a concrete provider/model AND ``reasoning_effort: none``; 0 = uncapped.
|
||||
"compression": _aux(120, max_output_tokens=0),
|
||||
# are ignored. Compression: raise timeout for local models.
|
||||
"compression": _aux(120),
|
||||
"skills_hub": _aux(30),
|
||||
"approval": _aux(30), # classifier — a fast/cheap model is recommended
|
||||
# /review reviewer: a full subagent on the async delegation rail, credentials resolved like
|
||||
@@ -1304,7 +1303,7 @@ DEFAULT_CONFIG = {
|
||||
{"provider": "openrouter", "model": "deepseek/deepseek-v4-pro"},
|
||||
],
|
||||
"aggregator": {"provider": "openrouter", "model": "anthropic/claude-opus-4.8"},
|
||||
"max_tokens": 4096,
|
||||
|
||||
"enabled": True,
|
||||
}
|
||||
},
|
||||
@@ -1824,7 +1823,7 @@ DEFAULT_CONFIG = {
|
||||
# providers: {openrouter: {url: https://example.com/my-curation.json}}.
|
||||
"providers": {},
|
||||
},
|
||||
# Per-model metadata overrides. Fields: context_window, max_output_tokens, supports_tools,
|
||||
# Per-model metadata overrides. Fields: context_window, supports_tools,
|
||||
# supports_vision, supports_reasoning, model_family. <provider>.<model_id> wins over
|
||||
# models.dev/OpenRouter/hardcoded defaults for the fields it sets (chain order in
|
||||
# agent/model_metadata.py). <provider>._default and top-level _default fill gaps ONLY for models
|
||||
|
||||
@@ -152,11 +152,7 @@ def _clean_slot(slot: Any, *, include_enabled: bool = False) -> dict[str, Any] |
|
||||
effort = _clean_reasoning_effort(slot.get("reasoning_effort"))
|
||||
if effort:
|
||||
clean["reasoning_effort"] = effort
|
||||
# Optional per-slot max_tokens overrides the preset-level reference_max_tokens for this
|
||||
# advisor; None (default) = no cap.
|
||||
slot_mt = _coerce_number(slot.get("max_tokens"), int, positive=True)
|
||||
if slot_mt is not None:
|
||||
clean["max_tokens"] = slot_mt
|
||||
|
||||
if include_enabled:
|
||||
clean["enabled"] = _coerce_bool(slot.get("enabled"), True)
|
||||
return clean
|
||||
@@ -230,10 +226,7 @@ def _normalize_preset(raw: Any) -> dict[str, Any]:
|
||||
"reference_timeout": _coerce_reference_timeout(raw.get("reference_timeout")),
|
||||
# Failed-advisor disclosure policy; unknown values fail loud.
|
||||
"degraded_reference_policy": policy if policy in {"loud", "silent"} else "loud",
|
||||
"max_tokens": _coerce_number(raw.get("max_tokens"), int, 4096),
|
||||
# Per-turn cap on each reference ADVISOR (never the acting aggregator). None = uncapped;
|
||||
# advisor generation dominates MoA latency, so e.g. 600 roughly halves wall time.
|
||||
"reference_max_tokens": _coerce_number(raw.get("reference_max_tokens"), int, positive=True),
|
||||
|
||||
# "user_turn" (default, cheapest): advisors run ONCE per user turn; "per_iteration": every
|
||||
# tool iteration; "every_n:<N>": first iteration of each turn and every Nth after.
|
||||
"fanout": _coerce_fanout(raw.get("fanout"))}
|
||||
@@ -241,7 +234,7 @@ def _normalize_preset(raw: Any) -> dict[str, Any]:
|
||||
|
||||
_FLAT_PRESET_KEYS = (
|
||||
"reference_models", "aggregator", "reference_temperature", "aggregator_temperature",
|
||||
"reference_timeout", "degraded_reference_policy", "max_tokens", "reference_max_tokens",
|
||||
"reference_timeout", "degraded_reference_policy",
|
||||
"fanout", "enabled")
|
||||
|
||||
|
||||
|
||||
@@ -404,7 +404,7 @@ def resolve_requested_provider(requested: Optional[str] = None) -> str:
|
||||
|
||||
from hermes_cli.runtime_provider_custom import ( # noqa: E402,F401
|
||||
_apply_custom_provider_extras, _custom_provider_request_overrides, _filter_capabilities, _find_custom_identity,
|
||||
_get_named_custom_provider, _lift_common_custom_fields, _lift_extra_headers, _lift_max_output_tokens,
|
||||
_get_named_custom_provider, _lift_common_custom_fields, _lift_extra_headers,
|
||||
_lift_model_capabilities, _normalize_base_url_for_match, _normalize_custom_provider_name, _resolve_named_custom_runtime,
|
||||
_try_resolve_from_custom_pool, canonical_custom_identity, find_custom_provider_identity,
|
||||
find_custom_provider_identity_by_model, has_named_custom_provider, is_routable_provider,
|
||||
|
||||
@@ -62,16 +62,6 @@ def _lift_model_capabilities(entry: Dict[str, Any], model: Optional[str], result
|
||||
result["capabilities"] = capabilities
|
||||
|
||||
|
||||
def _lift_max_output_tokens(entry: Dict[str, Any], result: Dict[str, Any]) -> None:
|
||||
"""``max_output_tokens`` or ``max_tokens`` on a provider entry pins its own output limit;
|
||||
gateway/CLI map it onto ``AIAgent.max_tokens`` only when top-level ``model.max_tokens`` is
|
||||
unset, so the documented global key still wins."""
|
||||
for key in ("max_output_tokens", "max_tokens"):
|
||||
value = entry.get(key)
|
||||
if isinstance(value, int) and value > 0:
|
||||
result["max_output_tokens"] = value
|
||||
return
|
||||
|
||||
|
||||
def _lift_extra_headers(entry: Dict[str, Any], result: Dict[str, Any]) -> None:
|
||||
"""Copy a validated ``extra_headers`` dict. SECURITY: values carry credentials — never log."""
|
||||
@@ -93,7 +83,7 @@ def _lift_common_custom_fields(entry: Dict[str, Any], result: Dict[str, Any], *,
|
||||
_lift_extra_headers(entry, result)
|
||||
if api_mode:
|
||||
result["api_mode"] = api_mode
|
||||
_lift_max_output_tokens(entry, result)
|
||||
|
||||
_lift_model_capabilities(entry, None, result)
|
||||
|
||||
|
||||
@@ -368,7 +358,7 @@ def _custom_provider_request_overrides(custom_provider: Dict[str, Any]) -> Dict[
|
||||
|
||||
|
||||
def _apply_custom_provider_extras(custom_provider: Dict[str, Any], target_model: Optional[str], result: Dict[str, Any]) -> None:
|
||||
"""Copy model / capabilities / max_output_tokens / extra_headers / request_overrides onto a
|
||||
"""Copy model / capabilities / extra_headers / request_overrides onto a
|
||||
resolved custom runtime. An explicit ``target_model`` wins over the provider's configured
|
||||
default (auxiliary slots / background-review resolve a concrete model and must not fall back to
|
||||
``default_model``). ``extra_headers`` may carry credentials — NEVER log them."""
|
||||
@@ -376,8 +366,7 @@ def _apply_custom_provider_extras(custom_provider: Dict[str, Any], target_model:
|
||||
if model_name:
|
||||
result["model"] = model_name
|
||||
_lift_model_capabilities(custom_provider, model_name, result)
|
||||
if isinstance(custom_provider.get("max_output_tokens"), int):
|
||||
result["max_output_tokens"] = custom_provider["max_output_tokens"]
|
||||
|
||||
if custom_provider.get("extra_headers"):
|
||||
result["extra_headers"] = dict(custom_provider["extra_headers"])
|
||||
request_overrides = _custom_provider_request_overrides(custom_provider)
|
||||
|
||||
@@ -137,10 +137,8 @@ class MoaPresetPayload(_MoaReferenceControls):
|
||||
# None = temperature omitted from API calls (provider default), as for single-model agents.
|
||||
reference_temperature: Optional[float] = None
|
||||
aggregator_temperature: Optional[float] = None
|
||||
max_tokens: int = 4096
|
||||
# Newer per-preset knobs (moa_config._normalize_preset): optional for older clients,
|
||||
# declared so GET round-trips don't erase them.
|
||||
reference_max_tokens: Optional[int] = None
|
||||
fanout: Optional[str] = None
|
||||
enabled: bool = True
|
||||
|
||||
@@ -153,8 +151,7 @@ class MoaConfigPayload(_MoaReferenceControls):
|
||||
aggregator: MoaModelSlot = MoaModelSlot()
|
||||
reference_temperature: Optional[float] = None
|
||||
aggregator_temperature: Optional[float] = None
|
||||
max_tokens: int = 4096
|
||||
reference_max_tokens: Optional[int] = None
|
||||
|
||||
fanout: Optional[str] = None
|
||||
enabled: bool = True
|
||||
profile: Optional[str] = None
|
||||
|
||||
@@ -228,7 +228,7 @@ def get_moa_models(profile: Optional[str] = None):
|
||||
|
||||
_MOA_PRESET_FIELDS = (
|
||||
"reference_temperature", "aggregator_temperature", "reference_timeout",
|
||||
"degraded_reference_policy", "max_tokens", "reference_max_tokens", "fanout", "enabled",
|
||||
"degraded_reference_policy", "fanout", "enabled",
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -66,12 +66,8 @@ custom = CustomProfile(
|
||||
name="custom", aliases=("ollama", "local", "vllm", "llamacpp", "llama.cpp", "llama-cpp"),
|
||||
env_vars=(), # No fixed key — custom endpoint
|
||||
base_url="", # User-configured
|
||||
# Floor only (user model.max_tokens overrides); without it Ollama falls
|
||||
# back to num_predict=128 and truncates.
|
||||
# Without this, no max_tokens is sent and Ollama falls back to its internal num_predict=128, truncating
|
||||
# responses after a few tokens (#39281). This is only a floor used when the user hasn't set
|
||||
# model.max_tokens — they can override per-model — so we set it generously rather than lowballing it.
|
||||
default_max_tokens=65536,
|
||||
# An arbitrary client ceiling can exceed a local server's actual output limit.
|
||||
# The endpoint owns its generation default.
|
||||
)
|
||||
|
||||
register_provider(custom)
|
||||
|
||||
@@ -838,7 +838,7 @@ def test_review_fork_uses_runtime_model_and_output_cap(curator_env, monkeypatch)
|
||||
|
||||
assert result["error"] is None
|
||||
assert captured["model"] == "real-model-id"
|
||||
assert captured["max_tokens"] == 1234
|
||||
assert captured["max_tokens"] is None
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -30,7 +30,6 @@ def test_explicit_non_reasoning_compression_route_is_certified_and_bounded():
|
||||
)
|
||||
|
||||
assert lane.certified_non_reasoning is True
|
||||
assert lane.max_tokens == 1400
|
||||
assert lane.reasoning_config == {"enabled": False, "effort": "none"}
|
||||
|
||||
|
||||
@@ -48,7 +47,6 @@ def test_inherited_auto_or_uncertified_compression_routes_remain_uncapped():
|
||||
|
||||
for lane in (inherited, unknown, reasoning):
|
||||
assert lane.certified_non_reasoning is False
|
||||
assert lane.max_tokens is None
|
||||
assert lane.reasoning_config is None
|
||||
|
||||
|
||||
@@ -99,8 +97,8 @@ def test_summary_model_override_is_certified_against_the_effective_model():
|
||||
requested_model="qwen3:14b",
|
||||
)
|
||||
|
||||
assert override.max_tokens == 1400
|
||||
assert drifted.max_tokens is None
|
||||
assert override.certified_non_reasoning is True
|
||||
assert drifted.certified_non_reasoning is False
|
||||
|
||||
|
||||
def test_compression_latency_records_delayed_first_provider_chunk():
|
||||
@@ -182,7 +180,7 @@ def test_certified_fast_lane_sends_the_configured_cap_to_its_provider():
|
||||
) is response
|
||||
|
||||
request = client.chat.completions.create.call_args.kwargs
|
||||
assert request["max_tokens"] == 1400
|
||||
assert "max_tokens" not in request
|
||||
assert request["extra_body"]["reasoning"] == {"enabled": False}
|
||||
|
||||
|
||||
@@ -249,7 +247,7 @@ def test_boolean_cap_drift_stays_uncapped_and_preserves_existing_reasoning():
|
||||
request = client.chat.completions.create.call_args.kwargs
|
||||
assert "max_tokens" not in request
|
||||
assert "max_completion_tokens" not in request
|
||||
assert request["extra_body"]["reasoning"] == {"enabled": False}
|
||||
assert "reasoning" not in request.get("extra_body", {})
|
||||
|
||||
|
||||
def test_bedrock_converse_ttfp_waits_for_the_nonstreaming_response():
|
||||
@@ -322,7 +320,7 @@ def test_summary_model_override_cap_uses_the_actual_primary_request():
|
||||
|
||||
request = client.chat.completions.create.call_args.kwargs
|
||||
assert request["model"] == "qwen3:14b"
|
||||
assert request["max_tokens"] == 1400
|
||||
assert "max_tokens" not in request
|
||||
|
||||
|
||||
def test_fallback_cap_requires_independent_route_certification():
|
||||
@@ -373,7 +371,7 @@ def test_fallback_cap_requires_independent_route_certification():
|
||||
assert "max_tokens" not in uncertified
|
||||
assert "max_completion_tokens" not in uncertified
|
||||
assert "reasoning" not in uncertified.get("extra_body", {})
|
||||
assert certified["max_tokens"] == 900
|
||||
assert "max_tokens" not in certified
|
||||
assert certified["extra_body"]["reasoning"] == {
|
||||
"enabled": False,
|
||||
"effort": "none",
|
||||
@@ -392,13 +390,11 @@ def test_reasoning_effort_aliases_certify_like_none():
|
||||
for alias in ("none", "false", "disabled", False):
|
||||
lane = _resolve({**base, "reasoning_effort": alias})
|
||||
assert lane.certified_non_reasoning is True, alias
|
||||
assert lane.max_tokens == 1400, alias
|
||||
|
||||
# Empty/unset (provider default) and real efforts must NOT certify.
|
||||
for not_disabled in ("", None, "low", "high", True):
|
||||
lane = _resolve({**base, "reasoning_effort": not_disabled})
|
||||
assert lane.certified_non_reasoning is False, not_disabled
|
||||
assert lane.max_tokens is None, not_disabled
|
||||
|
||||
|
||||
def test_timing_hooks_propagate_to_protected_call_worker_thread():
|
||||
|
||||
@@ -35,7 +35,7 @@ class TestRunReferenceSlotMaxTokens:
|
||||
patch("agent.moa_loop._maybe_apply_moa_cache_control", side_effect=lambda msgs, rt, **kwargs: msgs):
|
||||
_run_reference(slot, [{"role": "user", "content": "hi"}], max_tokens=2000)
|
||||
|
||||
assert captured_kwargs.get("max_tokens") == 600
|
||||
assert captured_kwargs.get("max_tokens") == 2000
|
||||
|
||||
def test_slot_max_tokens_absent_falls_back_to_preset(self):
|
||||
"""When slot has no max_tokens, the preset-level cap is used."""
|
||||
|
||||
@@ -1164,7 +1164,7 @@ class TestModelOverrides:
|
||||
assert info is not None
|
||||
assert info.family == "llava"
|
||||
assert info.context_window == 8192
|
||||
assert info.max_output == 4096
|
||||
assert info.max_output is None
|
||||
assert info.tool_call is True
|
||||
assert info.reasoning is False
|
||||
|
||||
|
||||
@@ -1,97 +0,0 @@
|
||||
"""Regression tests for max_tokens propagation from config.yaml to AIAgent.
|
||||
|
||||
Covers #20741: `model.max_tokens` was silently dropped before reaching the
|
||||
gateway-spawned agent, so providers without a hardcoded default (OpenRouter
|
||||
free models, Ollama Cloud, custom OpenAI-compatible endpoints) truncated long
|
||||
generations with `finish_reason="length"`.
|
||||
|
||||
Precedence verified here:
|
||||
HERMES_MAX_TOKENS env > model.max_tokens > per-provider
|
||||
max_output_tokens > None
|
||||
"""
|
||||
|
||||
import importlib
|
||||
import os
|
||||
import sys
|
||||
import textwrap
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def isolated_home(tmp_path, monkeypatch):
|
||||
"""Isolated HERMES_HOME with a writable config.yaml and a clean module cache.
|
||||
|
||||
These tests deliberately re-import ``hermes_cli`` / ``gateway`` so each
|
||||
config write is read fresh. To avoid leaking that purge into sibling test
|
||||
files in the same worker (which breaks their import-time mocks), we snapshot
|
||||
the affected modules and restore them on teardown.
|
||||
"""
|
||||
hermes_home = tmp_path / ".hermes"
|
||||
hermes_home.mkdir()
|
||||
monkeypatch.setenv("HERMES_HOME", str(hermes_home))
|
||||
monkeypatch.delenv("HERMES_MAX_TOKENS", raising=False)
|
||||
|
||||
_saved = {
|
||||
k: v
|
||||
for k, v in sys.modules.items()
|
||||
if k.startswith(("hermes_cli", "gateway"))
|
||||
}
|
||||
|
||||
def write_cfg(body: str) -> None:
|
||||
(hermes_home / "config.yaml").write_text(textwrap.dedent(body))
|
||||
|
||||
def fresh_gateway():
|
||||
for mod in list(sys.modules.keys()):
|
||||
if mod.startswith(("hermes_cli", "gateway")):
|
||||
del sys.modules[mod]
|
||||
return importlib.import_module("gateway.run")
|
||||
|
||||
try:
|
||||
yield write_cfg, fresh_gateway
|
||||
finally:
|
||||
# Drop anything we (re)imported, then restore the pre-test snapshot so
|
||||
# the next test file sees the module objects it was loaded with.
|
||||
for k in list(sys.modules.keys()):
|
||||
if k.startswith(("hermes_cli", "gateway")):
|
||||
del sys.modules[k]
|
||||
sys.modules.update(_saved)
|
||||
|
||||
|
||||
def test_top_level_max_tokens_propagates(isolated_home):
|
||||
"""model.max_tokens is read into the gateway runtime kwargs (#20741)."""
|
||||
write_cfg, fresh_gateway = isolated_home
|
||||
write_cfg(
|
||||
"""
|
||||
model:
|
||||
default: glm-5.1
|
||||
provider: openrouter
|
||||
max_tokens: 16384
|
||||
"""
|
||||
)
|
||||
grun = fresh_gateway()
|
||||
kw = grun._resolve_runtime_agent_kwargs()
|
||||
assert kw["max_tokens"] == 16384
|
||||
|
||||
|
||||
def test_per_provider_max_output_tokens_fallback(isolated_home):
|
||||
"""A custom provider's max_output_tokens fills in when no global is set."""
|
||||
write_cfg, fresh_gateway = isolated_home
|
||||
write_cfg(
|
||||
"""
|
||||
model:
|
||||
default: glm-5.1
|
||||
provider: mylocal
|
||||
providers:
|
||||
mylocal:
|
||||
api: http://localhost:11434/v1
|
||||
api_key: sk-test
|
||||
default_model: glm-5.1
|
||||
max_output_tokens: 12000
|
||||
"""
|
||||
)
|
||||
grun = fresh_gateway()
|
||||
kw = grun._resolve_runtime_agent_kwargs()
|
||||
assert kw["max_tokens"] == 12000
|
||||
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
"""User configuration cannot impose generation caps; wire budgets remain internal."""
|
||||
|
||||
import json
|
||||
|
||||
|
||||
def test_legacy_user_caps_do_not_change_runtime(tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
monkeypatch.setenv("HERMES_MAX_TOKENS", "13")
|
||||
config = {
|
||||
"model": {"default": "fixture", "provider": "local-fixture", "max_tokens": 17},
|
||||
"providers": {"local-fixture": {"api": "http://127.0.0.1:1/v1", "api_key": "fixture", "max_output_tokens": 19}},
|
||||
}
|
||||
(tmp_path / "config.yaml").write_text(json.dumps(config))
|
||||
from gateway.run import _resolve_runtime_agent_kwargs
|
||||
from gateway.platforms.api_server import _resolve_request_runtime_agent_kwargs
|
||||
from hermes_cli.moa_config import _normalize_preset
|
||||
from agent.models_dev import _override_to_catalog_shape
|
||||
|
||||
runtime = _resolve_runtime_agent_kwargs()
|
||||
assert runtime.get("max_tokens") is None
|
||||
assert runtime["base_url"].startswith("http://127.0.0.1:1")
|
||||
request = _resolve_request_runtime_agent_kwargs(provider="local-fixture", target_model="fixture")
|
||||
assert request.get("max_tokens") is None
|
||||
preset = _normalize_preset({"max_tokens": 23, "reference_max_tokens": 29})
|
||||
assert "max_tokens" not in preset and "reference_max_tokens" not in preset
|
||||
patch, _ = _override_to_catalog_shape({"context_window": 10000, "max_output_tokens": 31})
|
||||
assert patch["limit"] == {"context": 10000}
|
||||
from agent.auxiliary_client import _compression_fast_lane_controls
|
||||
route = {"provider": "custom", "model": "fixture", "reasoning_effort": "none", "max_output_tokens": 47}
|
||||
cap, body = _compression_fast_lane_controls(
|
||||
"compression", actual_provider="custom", actual_model="fixture",
|
||||
requested_provider="custom", requested_model="fixture", route_config=route,
|
||||
leak_guard_config=route, max_tokens=None, extra_body={},
|
||||
)
|
||||
assert cap is None
|
||||
assert body["reasoning"]["enabled"] is False
|
||||
|
||||
|
||||
def test_optional_wire_caps_omitted_required_and_internal_preserved():
|
||||
from agent.transports.chat_completions import ChatCompletionsTransport
|
||||
from agent.transports.bedrock import BedrockTransport
|
||||
from agent.transports.anthropic import AnthropicTransport
|
||||
|
||||
messages = [{"role": "user", "content": "fixture"}]
|
||||
chat = ChatCompletionsTransport().build_kwargs(
|
||||
"claude-fixture", messages, anthropic_max_output=65536,
|
||||
max_tokens_param_fn=lambda value: {"max_tokens": value},
|
||||
)
|
||||
assert "max_tokens" not in chat
|
||||
from providers import get_provider_profile
|
||||
custom = ChatCompletionsTransport().build_kwargs(
|
||||
"fixture", messages, provider_profile=get_provider_profile("custom"),
|
||||
max_tokens_param_fn=lambda value: {"max_tokens": value},
|
||||
)
|
||||
assert "max_tokens" not in custom
|
||||
bedrock = BedrockTransport().build_kwargs("amazon.nova-pro-v1:0", messages)
|
||||
assert "maxTokens" not in bedrock.get("inferenceConfig", {})
|
||||
native = AnthropicTransport().build_kwargs("claude-sonnet-4-5", messages)
|
||||
assert native["max_tokens"] > 0
|
||||
bounded = ChatCompletionsTransport().build_kwargs(
|
||||
"fixture", messages, max_tokens=43,
|
||||
max_tokens_param_fn=lambda value: {"max_tokens": value},
|
||||
)
|
||||
assert bounded["max_tokens"] == 43
|
||||
@@ -73,7 +73,7 @@ def test_routing_to_different_model_marks_routed_and_resolves_credentials():
|
||||
assert rt["api_key"] == "or-key"
|
||||
assert rt["credential_pool"] == "routed-pool"
|
||||
assert rt["request_overrides"] == {"extra_body": {"store": False}}
|
||||
assert rt["max_tokens"] == 2048
|
||||
assert rt["max_tokens"] is None
|
||||
|
||||
|
||||
def test_unrouted_runtime_keeps_parent_pool_and_overrides():
|
||||
|
||||
@@ -59,7 +59,7 @@ def test_direct_branch_forwards_request_overrides():
|
||||
"extra_body": {"provider": {"sort": "throughput"}},
|
||||
}
|
||||
# Shape parity with the named-provider branch: max_output_tokens present.
|
||||
assert "max_output_tokens" in creds
|
||||
assert "max_output_tokens" not in creds
|
||||
|
||||
|
||||
def test_direct_branch_absent_request_overrides_stays_none():
|
||||
@@ -123,7 +123,7 @@ def test_explicit_merges_over_runtime_on_provider_alongside_base_url(mock_resolv
|
||||
"provider": {"sort": "throughput"},
|
||||
},
|
||||
}
|
||||
assert creds["max_output_tokens"] == 8192
|
||||
assert "max_output_tokens" not in creds
|
||||
|
||||
|
||||
# ── Branch 2: named provider (no base_url) ─────────────────────────────────
|
||||
|
||||
@@ -165,7 +165,7 @@ def _build_child_agent(
|
||||
override_api_key: Optional[str] = None,
|
||||
override_api_mode: Optional[str] = None,
|
||||
override_request_overrides: Optional[Dict[str, Any]] = None,
|
||||
override_max_tokens: Optional[int] = None,
|
||||
|
||||
# ACP transport overrides from trusted delegation config.
|
||||
override_acp_command: Optional[str] = None,
|
||||
override_acp_args: Optional[List[str]] = None,
|
||||
@@ -209,7 +209,7 @@ def _build_child_agent(
|
||||
rt = _resolve_child_runtime(
|
||||
parent_agent, delegation_cfg, parent_api_key, model=model, override_provider=override_provider,
|
||||
override_base_url=override_base_url, override_api_key=override_api_key, override_api_mode=override_api_mode,
|
||||
override_max_tokens=override_max_tokens, override_acp_command=override_acp_command,
|
||||
override_acp_command=override_acp_command,
|
||||
override_acp_args=override_acp_args,
|
||||
)
|
||||
if override_request_overrides is not None:
|
||||
@@ -359,7 +359,7 @@ def _build_children(
|
||||
"override_provider": creds["provider"], "override_base_url": creds["base_url"],
|
||||
"override_api_key": creds["api_key"], "override_api_mode": creds["api_mode"],
|
||||
"override_request_overrides": creds.get("request_overrides"),
|
||||
"override_max_tokens": creds.get("max_output_tokens"), "override_acp_command": creds.get("command"),
|
||||
"override_acp_command": creds.get("command"),
|
||||
"override_acp_args": creds.get("args"),
|
||||
}
|
||||
children = []
|
||||
|
||||
@@ -273,11 +273,11 @@ def _require_pinned_command(command: Optional[str], message: str) -> None:
|
||||
if command and not _shutil.which(command):
|
||||
raise ValueError(message)
|
||||
|
||||
def _credential_bundle(model, provider, base_url, api_key, api_mode, request_overrides, max_output_tokens, **extra) -> dict:
|
||||
def _credential_bundle(model, provider, base_url, api_key, api_mode, request_overrides, **extra) -> dict:
|
||||
"""The child credential dict every branch of ``_resolve_delegation_credentials`` returns."""
|
||||
return {
|
||||
"model": model, "provider": provider, "base_url": base_url, "api_key": api_key, "api_mode": api_mode,
|
||||
"request_overrides": request_overrides, "max_output_tokens": max_output_tokens, **extra,
|
||||
"request_overrides": request_overrides, **extra,
|
||||
}
|
||||
|
||||
def _direct_endpoint_credentials(v: dict, explicit_request_overrides) -> dict:
|
||||
@@ -301,15 +301,14 @@ def _direct_endpoint_credentials(v: dict, explicit_request_overrides) -> dict:
|
||||
if v["api_mode"] in _EXPLICIT_API_MODES:
|
||||
api_mode = v["api_mode"]
|
||||
|
||||
# provider configured ALONGSIDE base_url: pull that provider's request personality (request_overrides /
|
||||
# max_output_tokens) onto the explicit endpoint. Best-effort — a resolution failure only skips the overrides.
|
||||
request_overrides = max_output_tokens = None
|
||||
# Preserve the configured provider's request personality on an explicit endpoint.
|
||||
request_overrides = None
|
||||
if v["provider"]:
|
||||
try:
|
||||
from hermes_cli.runtime_provider import resolve_runtime_provider
|
||||
runtime = resolve_runtime_provider(requested=v["provider"], target_model=v["model"])
|
||||
request_overrides = dict(runtime.get("request_overrides") or {}) or None
|
||||
max_output_tokens = runtime.get("max_output_tokens")
|
||||
|
||||
except Exception as exc:
|
||||
logger.debug(
|
||||
"delegation.base_url: runtime resolution for provider '%s' failed; proceeding without request_overrides: %s",
|
||||
@@ -318,7 +317,7 @@ def _direct_endpoint_credentials(v: dict, explicit_request_overrides) -> dict:
|
||||
# api_key None → inherited from parent in _build_child_agent
|
||||
return _credential_bundle(
|
||||
v["model"], provider, v["base_url"], v["api_key"], api_mode,
|
||||
_merge_request_overrides(request_overrides, explicit_request_overrides), max_output_tokens,
|
||||
_merge_request_overrides(request_overrides, explicit_request_overrides),
|
||||
)
|
||||
|
||||
def _runtime_provider_credentials(v: dict, explicit_request_overrides) -> dict:
|
||||
@@ -354,7 +353,7 @@ def _runtime_provider_credentials(v: dict, explicit_request_overrides) -> dict:
|
||||
configured_provider if runtime.get("provider") == _RUNTIME_PROVIDER_CUSTOM else runtime.get("provider"),
|
||||
runtime.get("base_url"), api_key, runtime.get("api_mode"),
|
||||
_merge_request_overrides(runtime.get("request_overrides"), explicit_request_overrides) or {},
|
||||
runtime.get("max_output_tokens"), command=pinned_command, args=list(runtime.get("args") or []),
|
||||
command=pinned_command, args=list(runtime.get("args") or []),
|
||||
)
|
||||
|
||||
def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict:
|
||||
@@ -374,7 +373,7 @@ def _resolve_delegation_credentials(cfg: dict, parent_agent) -> dict:
|
||||
# Pure inherit; explicit request_overrides still merge OVER the parent's.
|
||||
return _credential_bundle(
|
||||
values["model"], None, None, None, None,
|
||||
_merge_request_overrides(getattr(parent_agent, "request_overrides", None), explicit_request_overrides), None,
|
||||
_merge_request_overrides(getattr(parent_agent, "request_overrides", None), explicit_request_overrides),
|
||||
)
|
||||
return _runtime_provider_credentials(values, explicit_request_overrides)
|
||||
|
||||
@@ -411,7 +410,7 @@ _NOUS_PROVIDERS = frozenset({"nous", "nous-portal", "nousresearch"})
|
||||
def _resolve_child_runtime(
|
||||
parent_agent, delegation_cfg: dict, parent_api_key: Any, *, model: Optional[str], override_provider: Optional[str],
|
||||
override_base_url: Optional[str], override_api_key: Optional[str], override_api_mode: Optional[str],
|
||||
override_max_tokens: Optional[int], override_acp_command: Optional[str], override_acp_args: Optional[List[str]],
|
||||
override_acp_command: Optional[str], override_acp_args: Optional[List[str]],
|
||||
) -> Dict[str, Any]:
|
||||
"""Child credentials, transport and routing (config override > parent inherit) as ``AIAgent`` kwargs. Rules that
|
||||
are easy to break: api_mode is re-derived (not inherited) when the child's provider differs from the parent's
|
||||
@@ -494,7 +493,7 @@ def _resolve_child_runtime(
|
||||
}
|
||||
if not override_provider:
|
||||
kwargs["provider_data_collection"] = kwargs["provider_data_collection"] or ""
|
||||
child_max_tokens = override_max_tokens if override_max_tokens is not None else getattr(parent_agent, "max_tokens", None)
|
||||
child_max_tokens = getattr(parent_agent, "max_tokens", None)
|
||||
if isinstance(child_max_tokens, int):
|
||||
kwargs["max_tokens"] = child_max_tokens
|
||||
return kwargs
|
||||
|
||||
+2
-4
@@ -2460,9 +2460,7 @@ export interface MoaConfigResponse {
|
||||
aggregator_temperature: number;
|
||||
reference_timeout: number | null;
|
||||
degraded_reference_policy: "loud" | "silent";
|
||||
max_tokens: number;
|
||||
/** Optional advisor output cap — round-tripped, not edited here. */
|
||||
reference_max_tokens?: number | null;
|
||||
|
||||
/** Fan-out cadence (user_turn default | per_iteration | every_n:N) — round-tripped. */
|
||||
fanout?: string;
|
||||
enabled: boolean;
|
||||
@@ -2473,7 +2471,7 @@ export interface MoaConfigResponse {
|
||||
aggregator_temperature: number;
|
||||
reference_timeout: number | null;
|
||||
degraded_reference_policy: "loud" | "silent";
|
||||
max_tokens: number;
|
||||
|
||||
enabled: boolean;
|
||||
}
|
||||
|
||||
|
||||
@@ -767,7 +767,7 @@ function MoaModelsModal({
|
||||
aggregator_temperature: draft.aggregator_temperature,
|
||||
reference_timeout: draft.reference_timeout,
|
||||
degraded_reference_policy: draft.degraded_reference_policy,
|
||||
max_tokens: draft.max_tokens,
|
||||
|
||||
enabled: draft.enabled,
|
||||
};
|
||||
setDraft((prev) => ({
|
||||
|
||||
@@ -862,7 +862,7 @@ hermes model
|
||||
**Tool calling:** Use `--tool-call-parser` with the appropriate parser for your model family: `qwen` (Qwen 2.5), `llama3`, `llama4`, `deepseekv3`, `mistral`, `glm`. Without this flag, tool calls come back as plain text.
|
||||
|
||||
:::caution SGLang defaults to 128 max output tokens
|
||||
If responses seem truncated, add `max_tokens` to your requests or set `--default-max-tokens` on the server. SGLang's default is only 128 tokens per response if not specified in the request.
|
||||
If responses seem truncated, check the server's generation default and configure it on the server (for example SGLang's `--default-max-tokens`). Hermes does not expose an output-token cap setting.
|
||||
:::
|
||||
|
||||
---
|
||||
@@ -1131,7 +1131,7 @@ model:
|
||||
#### Responses get cut off mid-sentence
|
||||
|
||||
**Possible causes:**
|
||||
1. **Low output cap (`max_tokens`) on the server** — SGLang defaults to 128 tokens per response. Set `--default-max-tokens` on the server or configure Hermes with `model.max_tokens` in config.yaml. Note: `max_tokens` controls response length only — it is unrelated to how long your conversation history can be (that is `context_length`).
|
||||
1. **Low output limit on the server** — configure the server's generation default (for example SGLang's `--default-max-tokens`). Hermes does not expose an output-token cap setting. Response length is distinct from the conversation's context window (`context_length`).
|
||||
2. **Context exhaustion** — The model filled its context window. Increase `model.context_length` or enable [context compression](/user-guide/configuration#context-compression) in Hermes.
|
||||
|
||||
---
|
||||
@@ -1227,13 +1227,24 @@ model:
|
||||
|
||||
### Context Length Detection
|
||||
|
||||
:::note Two settings, easy to confuse
|
||||
:::note Context windows and output limits are different
|
||||
**`context_length`** is the **total context window** — the combined budget for input *and* output tokens (e.g. 200,000 for Claude Opus 4.6). Hermes uses this to decide when to compress history and to validate API requests.
|
||||
|
||||
**`model.max_tokens`** is the **output cap** — the maximum number of tokens the model may generate in a *single response*. It has nothing to do with how long your conversation history can be. The industry-standard name `max_tokens` is a common source of confusion; Anthropic's native API has since renamed it `max_output_tokens` for clarity.
|
||||
Output limits govern a single generated response, not the conversation history.
|
||||
Hermes no longer reads `model.max_tokens`, `HERMES_MAX_TOKENS`, provider output-cap
|
||||
settings, or `model_overrides.*.*.max_output_tokens`. Remove these legacy settings.
|
||||
Custom OpenAI-compatible endpoints receive no automatic catalog-sized output cap.
|
||||
Their server defaults apply; these can be lower than the model maximum.
|
||||
|
||||
Native Anthropic Messages (including the native Anthropic Bedrock path) requires
|
||||
`max_tokens`, so Hermes supplies an internal value. Bedrock Converse is a separate
|
||||
protocol: its optional `inferenceConfig.maxTokens` is omitted by default, which
|
||||
[AWS documents as the model maximum](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_InferenceConfiguration.html).
|
||||
Internal bounded tasks and provider-specific protocol requirements remain implementation
|
||||
details. Omission does not universally select a model's maximum output.
|
||||
|
||||
Set `context_length` when auto-detection gets the window size wrong.
|
||||
Set `model.max_tokens` only when you need to limit how long individual responses can be.
|
||||
|
||||
:::
|
||||
|
||||
Hermes uses a multi-source resolution chain to detect the correct context window for your model and provider:
|
||||
|
||||
@@ -68,7 +68,7 @@ Entries can optionally include:
|
||||
| `--resume` | `false` | Resume from checkpoint |
|
||||
| `--verbose` | `false` | Enable verbose logging |
|
||||
| `--max_samples` | all | Only process first N samples from dataset |
|
||||
| `--max_tokens` | model default | Maximum tokens per model response |
|
||||
|
||||
|
||||
### Provider Routing (OpenRouter)
|
||||
|
||||
|
||||
@@ -110,7 +110,7 @@ loops:
|
||||
self_paced_ceiling_seconds: 900 # self-paced max backoff
|
||||
```
|
||||
|
||||
The `--until` judge routes through the `goal_judge` auxiliary task, so `auxiliary.goal_judge.*` overrides (provider, model, max_tokens) apply to loop conditions too.
|
||||
The `--until` judge routes through the `goal_judge` auxiliary task, so `auxiliary.goal_judge.*` routing overrides (provider, model) apply to loop conditions too.
|
||||
|
||||
## `/loop` vs `/goal` vs cron
|
||||
|
||||
|
||||
@@ -90,7 +90,7 @@ moa:
|
||||
# the same behavior as a single-model Hermes agent.
|
||||
# reference_temperature: 0.6
|
||||
# aggregator_temperature: 0.4
|
||||
max_tokens: 4096
|
||||
|
||||
enabled: true
|
||||
```
|
||||
|
||||
@@ -100,37 +100,12 @@ Default preset:
|
||||
- reference: `openrouter:deepseek/deepseek-v4-pro`
|
||||
- aggregator / acting model: `openrouter:anthropic/claude-opus-4.8`
|
||||
|
||||
### Tuning advisor speed with `reference_max_tokens`
|
||||
### Advisor output
|
||||
|
||||
Each turn, MoA runs the reference models (advisors) in parallel and then the
|
||||
aggregator acts. Advisor generation is the dominant per-turn latency — turn
|
||||
wall time correlates strongly with how many tokens the advisors emit, because
|
||||
the turn waits for the slowest advisor to finish writing. By default advisors
|
||||
are **uncapped** (`reference_max_tokens` unset), so they may write long,
|
||||
essay-length advice.
|
||||
|
||||
Set `reference_max_tokens` on a preset to cap advisor output and give concise
|
||||
advice instead. The aggregator only needs the gist of each advisor's
|
||||
judgement, so a cap (e.g. `600`) measurably cuts per-turn wall time with little
|
||||
quality impact. It caps **advisors only** — the acting aggregator's output (the
|
||||
user-visible answer) is never capped.
|
||||
|
||||
```yaml
|
||||
moa:
|
||||
presets:
|
||||
fast:
|
||||
reference_models:
|
||||
- provider: openrouter
|
||||
model: anthropic/claude-opus-4.8
|
||||
- provider: openrouter
|
||||
model: openai/gpt-5.5
|
||||
aggregator:
|
||||
provider: openrouter
|
||||
model: anthropic/claude-opus-4.8
|
||||
reference_max_tokens: 600 # concise advice → faster turns
|
||||
```
|
||||
|
||||
Leave it unset (or `0`/blank) to keep the prior uncapped behavior.
|
||||
MoA uses provider-owned output limits. Preset and per-slot output-token cap
|
||||
settings are no longer supported. Provider defaults vary; omission does not
|
||||
always mean the model maximum. Native protocols that require an output limit
|
||||
receive an internal value from Hermes.
|
||||
|
||||
### Advisor cadence with `fanout`
|
||||
|
||||
|
||||
Reference in New Issue
Block a user