diff --git a/agent/background_review.py b/agent/background_review.py index f7c16f226f..28b9991c9e 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -813,7 +813,8 @@ def _fork_init_kwargs(agent: Any, rt: Dict[str, Any], routed: bool, max_iteratio "enabled_toolsets": getattr(agent, "enabled_toolsets", None), "disabled_toolsets": getattr(agent, "disabled_toolsets", None), "skip_memory": True, } - + if isinstance(rt.get("max_tokens"), int): + kwargs["max_tokens"] = rt["max_tokens"] if isinstance(rt.get("command"), str) and rt["command"]: kwargs.update(acp_command=rt["command"], acp_args=rt.get("args") or []) if not routed: diff --git a/agent/models_dev.py b/agent/models_dev.py index af904e43d1..5dd3e9707c 100644 --- a/agent/models_dev.py +++ b/agent/models_dev.py @@ -765,7 +765,7 @@ def get_model_info(provider_id: str, model_id: str, *, allow_network: bool = Fal only for unknown ones. ``allow_network`` defaults to False — cost guard and inventory are hot paths. ``model_overrides`` entries use the SAME canonical schema as every other consumer (``context_window``, - ``max_output_tokens``, ``supports_*``, ``model_family``) — they are translated into the catalog shape at + ``supports_*``, ``model_family``) — they are translated into the catalog shape at this boundary, and sub-dicts (``limit``, ``modalities``) are merged rather than clobbered. See #84482, #8731. """ diff --git a/agent/secret_scope.py b/agent/secret_scope.py index d7b85a6e09..7b84dcbdc4 100644 --- a/agent/secret_scope.py +++ b/agent/secret_scope.py @@ -65,7 +65,7 @@ def current_secret_scope() -> Optional[Mapping[str, str]]: _GLOBAL_ENV_EXACT = frozenset({ # Hermes runtime / deployment "HERMES_HOME", "HERMES_PROFILE", "HERMES_GATEWAY_LOCK_DIR", - "HERMES_MAX_ITERATIONS", "HERMES_MAX_TOKENS", "HERMES_API_TIMEOUT", + "HERMES_MAX_ITERATIONS", "HERMES_API_TIMEOUT", "HERMES_REDACT_SECRETS", "HERMES_NOUS_TIMEOUT_SECONDS", "_HERMES_GATEWAY", # OS / interpreter diff --git a/evals/output_caps_local_capture.py b/evals/output_caps_local_capture.py index 0a8a768edf..db7dd5fb58 100644 --- a/evals/output_caps_local_capture.py +++ b/evals/output_caps_local_capture.py @@ -31,9 +31,15 @@ class Capture(BaseHTTPRequestHandler): else: reply = {"id": "fixture", "object": "chat.completion", "created": 0, "model": body["model"], "choices": [{"index": 0, "message": {"role": "assistant", "content": "LOCAL_CAPTURE_ONLY"}, "finish_reason": "stop"}], "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}} self.send_response(200) - self.send_header("Content-Type", "application/json") - self.end_headers() - self.wfile.write(json.dumps(reply).encode()) + if body.get("stream"): + self.send_header("Content-Type", "text/event-stream") + self.end_headers() + chunk = {"id": "fixture", "object": "chat.completion.chunk", "created": 0, "model": body["model"], "choices": [{"index": 0, "delta": {"content": "LOCAL_CAPTURE_ONLY"}, "finish_reason": "stop"}], "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}} + self.wfile.write(("data: " + json.dumps(chunk) + "\n\ndata: [DONE]\n\n").encode()) + else: + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps(reply).encode()) def log_message(self, format, *args): pass @@ -95,6 +101,9 @@ Anthropic(api_key="fixture", base_url=url, max_retries=0).messages.create(**kwar results["native-anthropic"] = captures[-1] results["bedrock-converse-built-not-sent"] = BedrockTransport().build_kwargs("amazon.nova-pro-v1:0", messages) results["moa-normalized"] = _normalize_preset({"max_tokens": 23, "reference_max_tokens": 29}) +if os.environ.get("FULL_OUTPUT_CAP_SURFACES"): + from output_caps_surfaces import exercise_surfaces + results.update(exercise_surfaces(agent, captures, url, home, config)) print(json.dumps(results, indent=2, default=str)) server.shutdown() server.server_close() diff --git a/evals/output_caps_scope.md b/evals/output_caps_scope.md new file mode 100644 index 0000000000..1e6f39b15d --- /dev/null +++ b/evals/output_caps_scope.md @@ -0,0 +1,68 @@ +# Output-cap removal: scope and local evidence + +Run from a checkout (Python dependencies installed): + +```sh +HERMES_HOME=$(mktemp -d) FULL_OUTPUT_CAP_SURFACES=1 \ + VERIFY_OUTPUT_CAP_REMOVAL=1 PYTHONPATH=$PWD \ + .venv/bin/python evals/output_caps_local_capture.py +``` + +The HTTP fixture returns `LOCAL_CAPTURE_ONLY`, never vendor inference. The same +harness runs against the unchanged baseline with `VERIFY_OUTPUT_CAP_REMOVAL` +unset, using that checkout's working directory and `PYTHONPATH`. Capture bodies, +not the fake response length, are the oracle. The fixture exercises installed +OpenAI, Anthropic and boto3 SDKs; the latter is a local Converse capture, **not an +AWS call**. The PTY transcript is saved under the disposable `HERMES_HOME`. + +## Boundary + +Removed: `HERMES_MAX_TOKENS`, `model.max_tokens`, dedicated named/custom provider +`max_output_tokens` fields, model metadata output overrides, batch CLI cap, +MoA preset/reference/slot caps, and auxiliary compression user caps. Dedicated +caps are no longer lifted through gateway/API/CLI, child, review or curator +runtime resolution. Stale configuration is ignored; no global request-field +scrubber is introduced. + +Preserved: + +- **Protocol fields and internal task arguments:** native Anthropic Messages + requires `max_tokens`; native Anthropic on Bedrock is distinct from Converse. + Converse `inferenceConfig.maxTokens` is optional. Provider defaults need not + equal a model's maximum. +- **Provider implementation constraints:** existing NVIDIA, Qwen OAuth, Meta, + Kimi, Opencode and native Gemini profile/adapter budgets are unchanged. This + change does not establish fresh vendor acceptance or billing for those paths. +- **Task APIs:** `AIAgent(max_tokens=...)`, auxiliary `call_llm` budgets and + `llm.oneshot` remain callable by application tasks, not user model settings. + Desktop `requestOneShot` serves commit-message and other short generation + tasks; the project-idea action also uses the stateless RPC. There is no output + limit input in that UI. The optional Desktop maxTokens wrapper currently has + no explicit maxTokens-valued caller; its task RPC contract is retained. +- **MCP security policy:** `SamplingHandler.max_tokens_cap` bounds a particular + server's sampling requests, alongside rate/timeout/tool-round limits. Its + Nix, migration, example and documentation settings belong to that separate + trust boundary and remain supported. +- **Generic protocol passthrough:** `request_overrides`/`extra_body` remain + arbitrary request-shaping mechanisms. They are not dedicated output-cap + controls; stripping arbitrary native fields would break other protocols. +- Session `--max-tokens` selection filters, observed usage, context limits, + provider-error recovery and tool-output size budgets are unrelated. + +## Observed before / after + +| Local production surface | Baseline cap | Fixed cap | +| --- | --- | --- | +| Main agent conversation | 17 | omitted | +| Child build and completed lifecycle | 17 | omitted | +| Full MoA: two advisors + aggregator | 23 / 23 / omitted | all omitted | +| Compression `call_llm` | 47 | omitted | +| CLI one-shot through PTY | 13 | omitted | +| Converse through boto3 to loopback | 4096 | omitted | +| Native Anthropic SDK | 16384 | 16384 | +| Explicit internal task budget | 43 | 43 | + +The initial capture also covers real gateway/API/custom-provider resolvers and +the registered custom provider profile. It does not prove remote provider +behavior, billing, a native Desktop interaction, or curator/review scheduling. +Those surfaces must be validated separately rather than inferred from builders. diff --git a/evals/output_caps_surfaces.py b/evals/output_caps_surfaces.py new file mode 100644 index 0000000000..3906c928f5 --- /dev/null +++ b/evals/output_caps_surfaces.py @@ -0,0 +1,123 @@ +"""Full local execution controls; never sends a request to an inference vendor.""" +import json +import os +import pty +import select +import subprocess +import sys +import time +from pathlib import Path + + +def exercise_surfaces(agent, captures, url, home, config): + results = {} + start = len(captures) + response = agent.run_conversation("Return the fixture response.") + assert response["final_response"] == "LOCAL_CAPTURE_ONLY", response + results["full-agent"] = captures[start:] + + from tools.delegate_tool import _build_child_agent, _run_single_child + child = _build_child_agent(0, "Return fixture response", None, [], "fixture", 2, 1, agent) + start = len(captures) + response = _run_single_child(0, "Return fixture response", child, agent) + assert response.get("status") == "completed", response + results["full-child"] = captures[start:] + + slot = {"provider": "fixture-local", "model": "fixture", "max_tokens": 23} + config["moa"] = {"default_preset": "fixture", "presets": {"fixture": { + "reference_models": [slot, slot], "aggregator": slot, + "max_tokens": 29, "reference_max_tokens": 31, + }}} + config["auxiliary"] = {"compression": {"provider": "fixture-local", "model": "fixture", "reasoning_effort": "none", "max_output_tokens": 47}} + config["_setup_done"] = True + (home / "config.yaml").write_text(json.dumps(config)) + from agent.moa_loop import MoAClient + start = len(captures) + response = MoAClient("fixture").chat.completions.create(model="fixture", messages=[{"role": "user", "content": "Return fixture"}]) + assert response.choices[0].message.content == "LOCAL_CAPTURE_ONLY" + results["full-moa"] = captures[start:] + assert len(results["full-moa"]) == 3, results["full-moa"] + + from agent.auxiliary_client import call_llm + start = len(captures) + response = call_llm(task="compression", messages=[{"role": "user", "content": "Return fixture"}]) + assert response.choices[0].message.content == "LOCAL_CAPTURE_ONLY" + results["full-compression-call"] = captures[start:] + + from agent.curator import _run_llm_review + start = len(captures) + response = _run_llm_review("Return the fixture response without tools.") + assert response["final"] == "LOCAL_CAPTURE_ONLY", response + results["full-curator-fork"] = captures[start:] + + from agent.background_review import build_cache_parity_fork + start = len(captures) + review, _, _ = build_cache_parity_fork(agent, {}, max_iterations=2) + try: + response = review.run_conversation("Return fixture response without tools.") + assert response["final_response"] == "LOCAL_CAPTURE_ONLY", response + finally: + review.close() + results["full-review-fork"] = captures[start:] + + agent.max_tokens = 43 + start = len(captures) + review, _, _ = build_cache_parity_fork(agent, {}, max_iterations=2) + try: + response = review.run_conversation("Return fixture response without tools.") + assert response["final_response"] == "LOCAL_CAPTURE_ONLY", response + finally: + review.close() + agent.max_tokens = None + results["internal-review-budget"] = captures[start:] + assert results["internal-review-budget"][0]["body"]["max_tokens"] == 43 + + import boto3 + from agent.transports.bedrock import BedrockTransport + native = boto3.client("bedrock-runtime", region_name="us-east-1", endpoint_url=url, + aws_access_key_id="fixture", aws_secret_access_key="fixture") + kwargs = BedrockTransport().build_kwargs("amazon.nova-pro-v1:0", [{"role": "user", "content": "fixture"}]) + kwargs.pop("__bedrock_converse__", None) + kwargs.pop("__bedrock_region__", None) + start = len(captures) + native.converse(**kwargs) + results["bedrock-sdk-local-not-aws"] = captures[start:] + + master, slave = pty.openpty() + env = {k: v for k, v in os.environ.items() if not any(s in k for s in ("API_KEY", "TOKEN", "SECRET"))} + env.update(HOME=str(home), HERMES_HOME=str(home), HERMES_MAX_TOKENS="13", TERM="xterm", NO_COLOR="1") + start = len(captures) + proc = subprocess.Popen([sys.executable, "-m", "hermes_cli.main", "chat", "--provider", "fixture-local", "-m", "fixture", "-q", "Return fixture"], stdin=slave, stdout=slave, stderr=slave, cwd=os.getcwd(), env=env) + os.close(slave) + output = bytearray() + deadline = time.monotonic() + 60 + try: + while time.monotonic() < deadline: + if select.select([master], [], [], 0.1)[0]: + try: + chunk = os.read(master, 65536) + except OSError: + break + if not chunk: + break + output.extend(chunk) + elif proc.poll() is not None: + break + if proc.poll() is None: + proc.terminate() + proc.wait(timeout=10) + finally: + os.close(master) + (home / "cli-pty.txt").write_bytes(output) + assert proc.returncode == 0 and b"LOCAL_CAPTURE_ONLY" in output, output.decode(errors="replace") + results["cli-pty"] = captures[start:] + assert results["cli-pty"] + if os.environ.get("VERIFY_OUTPUT_CAP_REMOVAL"): + for label, requests in results.items(): + if label == "internal-review-budget": + continue + assert requests, label + for request in requests: + assert "max_tokens" not in request["body"] and "max_completion_tokens" not in request["body"], (label, request) + assert "maxTokens" not in request["body"].get("inferenceConfig", {}), (label, request) + return results diff --git a/tests/agent/test_curator.py b/tests/agent/test_curator.py index c1bc85930c..b898c14f9d 100644 --- a/tests/agent/test_curator.py +++ b/tests/agent/test_curator.py @@ -838,7 +838,7 @@ def test_review_fork_uses_runtime_model_and_output_cap(curator_env, monkeypatch) assert result["error"] is None assert captured["model"] == "real-model-id" - assert captured["max_tokens"] is None + assert captured.get("max_tokens") is None diff --git a/tests/agent/test_fast_compression_lane.py b/tests/agent/test_fast_compression_lane.py index 9a843c2429..919142dd5c 100644 --- a/tests/agent/test_fast_compression_lane.py +++ b/tests/agent/test_fast_compression_lane.py @@ -19,7 +19,7 @@ def _resolve(config, *, provider="ollama", model="qwen3:8b", requested_model=Non ) -def test_explicit_non_reasoning_compression_route_is_certified_and_bounded(): +def test_explicit_non_reasoning_compression_route_is_certified(): lane = _resolve( { "provider": "ollama", @@ -155,7 +155,7 @@ def test_compression_latency_records_delayed_first_provider_chunk(): assert timings["summary_generation_ms"] >= timings["time_to_first_progress_ms"] -def test_certified_fast_lane_sends_the_configured_cap_to_its_provider(): +def test_certified_fast_lane_ignores_legacy_cap_and_preserves_reasoning(): from agent.auxiliary_client import call_llm config = { @@ -217,7 +217,7 @@ def test_uncertified_effective_primary_route_does_not_receive_fast_cap(): assert "reasoning" not in request.get("extra_body", {}) -def test_boolean_cap_drift_stays_uncapped_and_preserves_existing_reasoning(): +def test_legacy_boolean_cap_does_not_bypass_route_certification(): from agent.auxiliary_client import call_llm config = { @@ -323,7 +323,7 @@ def test_summary_model_override_cap_uses_the_actual_primary_request(): assert "max_tokens" not in request -def test_fallback_cap_requires_independent_route_certification(): +def test_fallback_reasoning_requires_independent_route_certification(): from agent.auxiliary_client import _call_fallback_candidate_sync response = object() diff --git a/tests/agent/test_moa_slot_max_tokens.py b/tests/agent/test_moa_slot_max_tokens.py index a601a857a4..4aceb8e4b9 100644 --- a/tests/agent/test_moa_slot_max_tokens.py +++ b/tests/agent/test_moa_slot_max_tokens.py @@ -13,10 +13,10 @@ import pytest class TestRunReferenceSlotMaxTokens: - """_run_reference should prefer slot-level max_tokens over preset-level.""" + """Legacy slot settings cannot override internal advisor task budgets.""" - def test_slot_max_tokens_overrides_preset_level(self): - """When slot has max_tokens, it overrides the preset-level cap.""" + def test_legacy_slot_max_tokens_cannot_override_internal_budget(self): + """An explicit internal caller budget wins over a stale user setting.""" from agent.moa_loop import _run_reference captured_kwargs: dict = {} diff --git a/tests/agent/test_models_dev.py b/tests/agent/test_models_dev.py index 137cac1a12..be39c5092e 100644 --- a/tests/agent/test_models_dev.py +++ b/tests/agent/test_models_dev.py @@ -1164,7 +1164,7 @@ class TestModelOverrides: assert info is not None assert info.family == "llava" assert info.context_window == 8192 - assert info.max_output is None + assert info.max_output == 0 assert info.tool_call is True assert info.reasoning is False diff --git a/tests/run_agent/test_background_review_cost_controls.py b/tests/run_agent/test_background_review_cost_controls.py index 8a22da74ca..20921bf40c 100644 --- a/tests/run_agent/test_background_review_cost_controls.py +++ b/tests/run_agent/test_background_review_cost_controls.py @@ -73,7 +73,7 @@ def test_routing_to_different_model_marks_routed_and_resolves_credentials(): assert rt["api_key"] == "or-key" assert rt["credential_pool"] == "routed-pool" assert rt["request_overrides"] == {"extra_body": {"store": False}} - assert rt["max_tokens"] is None + assert rt.get("max_tokens") is None def test_unrouted_runtime_keeps_parent_pool_and_overrides(): diff --git a/tests/tools/test_delegate_request_overrides.py b/tests/tools/test_delegate_request_overrides.py index 46f3869650..ddc2b4a06c 100644 --- a/tests/tools/test_delegate_request_overrides.py +++ b/tests/tools/test_delegate_request_overrides.py @@ -58,7 +58,7 @@ def test_direct_branch_forwards_request_overrides(): assert creds["request_overrides"] == { "extra_body": {"provider": {"sort": "throughput"}}, } - # Shape parity with the named-provider branch: max_output_tokens present. + # Dedicated user cap controls are not part of child credentials. assert "max_output_tokens" not in creds @@ -93,7 +93,7 @@ def test_explicit_merges_over_runtime_on_provider_alongside_base_url(mock_resolv """Precedence on the provider-alongside-base_url path (#98237 interplay): explicit delegation.request_overrides merges OVER the named provider's runtime overrides — runtime extra_body keys survive unless redefined, - explicit top-level keys win, and max_output_tokens is preserved.""" + explicit top-level keys win, and legacy max_output_tokens is ignored.""" mock_resolve.return_value = { "provider": "custom", "base_url": "https://provider-default.example/v1", diff --git a/website/docs/user-guide/skills/optional/devops/devops-actual-setup.md b/website/docs/user-guide/skills/optional/devops/devops-actual-setup.md index 450eb93bc9..e4e9b6dd41 100644 --- a/website/docs/user-guide/skills/optional/devops/devops-actual-setup.md +++ b/website/docs/user-guide/skills/optional/devops/devops-actual-setup.md @@ -141,8 +141,8 @@ a human in a browser. `actual models load` takes the INSTALLED name from `actual models list`. 4. **Reasoning models returning empty content.** GLM/Qwen reasoning variants emit thinking in a separate `reasoning` field and can burn a small - `max_tokens` entirely on reasoning. Give generous max_tokens before - assuming failure. + output budget entirely on reasoning. Check the server-side output default + before assuming failure. 5. **Do not create a custom provider named `actual`.** Older setup guides (pre first-class support) wrote `providers.actual.*` config blocks. On current Hermes the built-in provider wins the name; stale custom blocks diff --git a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/integrations/providers.md b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/integrations/providers.md index e65738b72b..d999056279 100644 --- a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/integrations/providers.md +++ b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/integrations/providers.md @@ -720,7 +720,7 @@ hermes model **工具调用:** 使用 `--tool-call-parser` 并选择适合你模型系列的解析器:`qwen`(Qwen 2.5)、`llama3`、`llama4`、`deepseekv3`、`mistral`、`glm`。没有此标志,工具调用将以纯文本返回。 :::caution SGLang 默认最大输出 128 tokens -如果响应看起来被截断,在请求中添加 `max_tokens` 或在服务器上设置 `--default-max-tokens`。SGLang 的默认值是每次响应仅 128 tokens(如果请求中未指定)。 +如果响应看起来被截断,在服务器上设置 `--default-max-tokens`。SGLang 的默认值是每次响应仅 128 tokens(如果请求中未指定)。 ::: --- @@ -977,7 +977,7 @@ model: #### 响应在句子中间被截断 **可能原因:** -1. **服务器上的输出上限(`max_tokens`)过低** — SGLang 默认每次响应 128 tokens。在服务器上设置 `--default-max-tokens`,或在 config.yaml 中配置 `model.max_tokens`。注意:`max_tokens` 只控制响应长度——与对话历史可以有多长无关(那是 `context_length`)。 +1. **服务器上的输出上限(`max_tokens`)过低** — SGLang 默认每次响应 128 tokens。在服务器上设置 `--default-max-tokens`;Hermes 不再提供用户输出上限设置。注意:`max_tokens` 只控制响应长度——与对话历史可以有多长无关(那是 `context_length`)。 2. **上下文耗尽** — 模型填满了上下文窗口。增加 `model.context_length` 或在 Hermes 中启用[上下文压缩](/user-guide/configuration#context-compression)。 --- @@ -1075,10 +1075,10 @@ model: :::note 两个设置,容易混淆 **`context_length`** 是**总上下文窗口**——输入和输出 token 的合计预算(例如 Claude Opus 4.6 为 200,000)。Hermes 用它来决定何时压缩历史记录以及验证 API 请求。 -**`model.max_tokens`** 是**输出上限**——模型在*单次响应*中最多可生成的 token 数。与对话历史可以有多长无关。行业标准名称 `max_tokens` 是常见的混淆来源;Anthropic 的原生 API 已将其重命名为 `max_output_tokens` 以更清晰。 +**输出上限**限制单次响应,而非对话历史。Hermes 不再读取 `model.max_tokens`、`HERMES_MAX_TOKENS` 或提供商及模型的输出上限设置。兼容端点使用服务器默认值;该值不一定是模型最大值。原生 Anthropic Messages 仍要求 `max_tokens`,Hermes 会提供内部值。Bedrock Converse 的可选输出限制默认省略。 当自动检测获取的窗口大小不正确时,设置 `context_length`。 -仅当需要限制单次响应长度时,才设置 `model.max_tokens`。 +请删除旧的用户输出上限配置;内部任务预算与 MCP 采样安全预算保持不变。 ::: Hermes 使用多源解析链来检测模型和提供商的正确上下文窗口: diff --git a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/features/batch-processing.md b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/features/batch-processing.md index 0c62b94ba4..6068a52c4a 100644 --- a/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/features/batch-processing.md +++ b/website/i18n/zh-Hans/docusaurus-plugin-content-docs/current/user-guide/features/batch-processing.md @@ -68,7 +68,6 @@ python batch_runner.py --list_distributions | `--resume` | `false` | 从断点恢复 | | `--verbose` | `false` | 启用详细日志 | | `--max_samples` | 全部 | 仅处理数据集中前 N 条样本 | -| `--max_tokens` | 模型默认值 | 每次模型响应的最大 token 数 | ### 供应商路由(OpenRouter)