test: exercise output-cap removal across native and child surfaces
This commit is contained in:
@@ -813,7 +813,8 @@ def _fork_init_kwargs(agent: Any, rt: Dict[str, Any], routed: bool, max_iteratio
|
||||
"enabled_toolsets": getattr(agent, "enabled_toolsets", None),
|
||||
"disabled_toolsets": getattr(agent, "disabled_toolsets", None), "skip_memory": True,
|
||||
}
|
||||
|
||||
if isinstance(rt.get("max_tokens"), int):
|
||||
kwargs["max_tokens"] = rt["max_tokens"]
|
||||
if isinstance(rt.get("command"), str) and rt["command"]:
|
||||
kwargs.update(acp_command=rt["command"], acp_args=rt.get("args") or [])
|
||||
if not routed:
|
||||
|
||||
+1
-1
@@ -765,7 +765,7 @@ def get_model_info(provider_id: str, model_id: str, *, allow_network: bool = Fal
|
||||
only for unknown ones. ``allow_network`` defaults to False — cost guard and inventory are hot paths.
|
||||
|
||||
``model_overrides`` entries use the SAME canonical schema as every other consumer (``context_window``,
|
||||
``max_output_tokens``, ``supports_*``, ``model_family``) — they are translated into the catalog shape at
|
||||
``supports_*``, ``model_family``) — they are translated into the catalog shape at
|
||||
this boundary, and sub-dicts (``limit``, ``modalities``) are merged rather than clobbered. See #84482,
|
||||
#8731.
|
||||
"""
|
||||
|
||||
@@ -65,7 +65,7 @@ def current_secret_scope() -> Optional[Mapping[str, str]]:
|
||||
_GLOBAL_ENV_EXACT = frozenset({
|
||||
# Hermes runtime / deployment
|
||||
"HERMES_HOME", "HERMES_PROFILE", "HERMES_GATEWAY_LOCK_DIR",
|
||||
"HERMES_MAX_ITERATIONS", "HERMES_MAX_TOKENS", "HERMES_API_TIMEOUT",
|
||||
"HERMES_MAX_ITERATIONS", "HERMES_API_TIMEOUT",
|
||||
"HERMES_REDACT_SECRETS", "HERMES_NOUS_TIMEOUT_SECONDS",
|
||||
"_HERMES_GATEWAY",
|
||||
# OS / interpreter
|
||||
|
||||
@@ -31,9 +31,15 @@ class Capture(BaseHTTPRequestHandler):
|
||||
else:
|
||||
reply = {"id": "fixture", "object": "chat.completion", "created": 0, "model": body["model"], "choices": [{"index": 0, "message": {"role": "assistant", "content": "LOCAL_CAPTURE_ONLY"}, "finish_reason": "stop"}], "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}
|
||||
self.send_response(200)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.end_headers()
|
||||
self.wfile.write(json.dumps(reply).encode())
|
||||
if body.get("stream"):
|
||||
self.send_header("Content-Type", "text/event-stream")
|
||||
self.end_headers()
|
||||
chunk = {"id": "fixture", "object": "chat.completion.chunk", "created": 0, "model": body["model"], "choices": [{"index": 0, "delta": {"content": "LOCAL_CAPTURE_ONLY"}, "finish_reason": "stop"}], "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}
|
||||
self.wfile.write(("data: " + json.dumps(chunk) + "\n\ndata: [DONE]\n\n").encode())
|
||||
else:
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.end_headers()
|
||||
self.wfile.write(json.dumps(reply).encode())
|
||||
|
||||
def log_message(self, format, *args):
|
||||
pass
|
||||
@@ -95,6 +101,9 @@ Anthropic(api_key="fixture", base_url=url, max_retries=0).messages.create(**kwar
|
||||
results["native-anthropic"] = captures[-1]
|
||||
results["bedrock-converse-built-not-sent"] = BedrockTransport().build_kwargs("amazon.nova-pro-v1:0", messages)
|
||||
results["moa-normalized"] = _normalize_preset({"max_tokens": 23, "reference_max_tokens": 29})
|
||||
if os.environ.get("FULL_OUTPUT_CAP_SURFACES"):
|
||||
from output_caps_surfaces import exercise_surfaces
|
||||
results.update(exercise_surfaces(agent, captures, url, home, config))
|
||||
print(json.dumps(results, indent=2, default=str))
|
||||
server.shutdown()
|
||||
server.server_close()
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
# Output-cap removal: scope and local evidence
|
||||
|
||||
Run from a checkout (Python dependencies installed):
|
||||
|
||||
```sh
|
||||
HERMES_HOME=$(mktemp -d) FULL_OUTPUT_CAP_SURFACES=1 \
|
||||
VERIFY_OUTPUT_CAP_REMOVAL=1 PYTHONPATH=$PWD \
|
||||
.venv/bin/python evals/output_caps_local_capture.py
|
||||
```
|
||||
|
||||
The HTTP fixture returns `LOCAL_CAPTURE_ONLY`, never vendor inference. The same
|
||||
harness runs against the unchanged baseline with `VERIFY_OUTPUT_CAP_REMOVAL`
|
||||
unset, using that checkout's working directory and `PYTHONPATH`. Capture bodies,
|
||||
not the fake response length, are the oracle. The fixture exercises installed
|
||||
OpenAI, Anthropic and boto3 SDKs; the latter is a local Converse capture, **not an
|
||||
AWS call**. The PTY transcript is saved under the disposable `HERMES_HOME`.
|
||||
|
||||
## Boundary
|
||||
|
||||
Removed: `HERMES_MAX_TOKENS`, `model.max_tokens`, dedicated named/custom provider
|
||||
`max_output_tokens` fields, model metadata output overrides, batch CLI cap,
|
||||
MoA preset/reference/slot caps, and auxiliary compression user caps. Dedicated
|
||||
caps are no longer lifted through gateway/API/CLI, child, review or curator
|
||||
runtime resolution. Stale configuration is ignored; no global request-field
|
||||
scrubber is introduced.
|
||||
|
||||
Preserved:
|
||||
|
||||
- **Protocol fields and internal task arguments:** native Anthropic Messages
|
||||
requires `max_tokens`; native Anthropic on Bedrock is distinct from Converse.
|
||||
Converse `inferenceConfig.maxTokens` is optional. Provider defaults need not
|
||||
equal a model's maximum.
|
||||
- **Provider implementation constraints:** existing NVIDIA, Qwen OAuth, Meta,
|
||||
Kimi, Opencode and native Gemini profile/adapter budgets are unchanged. This
|
||||
change does not establish fresh vendor acceptance or billing for those paths.
|
||||
- **Task APIs:** `AIAgent(max_tokens=...)`, auxiliary `call_llm` budgets and
|
||||
`llm.oneshot` remain callable by application tasks, not user model settings.
|
||||
Desktop `requestOneShot` serves commit-message and other short generation
|
||||
tasks; the project-idea action also uses the stateless RPC. There is no output
|
||||
limit input in that UI. The optional Desktop maxTokens wrapper currently has
|
||||
no explicit maxTokens-valued caller; its task RPC contract is retained.
|
||||
- **MCP security policy:** `SamplingHandler.max_tokens_cap` bounds a particular
|
||||
server's sampling requests, alongside rate/timeout/tool-round limits. Its
|
||||
Nix, migration, example and documentation settings belong to that separate
|
||||
trust boundary and remain supported.
|
||||
- **Generic protocol passthrough:** `request_overrides`/`extra_body` remain
|
||||
arbitrary request-shaping mechanisms. They are not dedicated output-cap
|
||||
controls; stripping arbitrary native fields would break other protocols.
|
||||
- Session `--max-tokens` selection filters, observed usage, context limits,
|
||||
provider-error recovery and tool-output size budgets are unrelated.
|
||||
|
||||
## Observed before / after
|
||||
|
||||
| Local production surface | Baseline cap | Fixed cap |
|
||||
| --- | --- | --- |
|
||||
| Main agent conversation | 17 | omitted |
|
||||
| Child build and completed lifecycle | 17 | omitted |
|
||||
| Full MoA: two advisors + aggregator | 23 / 23 / omitted | all omitted |
|
||||
| Compression `call_llm` | 47 | omitted |
|
||||
| CLI one-shot through PTY | 13 | omitted |
|
||||
| Converse through boto3 to loopback | 4096 | omitted |
|
||||
| Native Anthropic SDK | 16384 | 16384 |
|
||||
| Explicit internal task budget | 43 | 43 |
|
||||
|
||||
The initial capture also covers real gateway/API/custom-provider resolvers and
|
||||
the registered custom provider profile. It does not prove remote provider
|
||||
behavior, billing, a native Desktop interaction, or curator/review scheduling.
|
||||
Those surfaces must be validated separately rather than inferred from builders.
|
||||
@@ -0,0 +1,123 @@
|
||||
"""Full local execution controls; never sends a request to an inference vendor."""
|
||||
import json
|
||||
import os
|
||||
import pty
|
||||
import select
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def exercise_surfaces(agent, captures, url, home, config):
|
||||
results = {}
|
||||
start = len(captures)
|
||||
response = agent.run_conversation("Return the fixture response.")
|
||||
assert response["final_response"] == "LOCAL_CAPTURE_ONLY", response
|
||||
results["full-agent"] = captures[start:]
|
||||
|
||||
from tools.delegate_tool import _build_child_agent, _run_single_child
|
||||
child = _build_child_agent(0, "Return fixture response", None, [], "fixture", 2, 1, agent)
|
||||
start = len(captures)
|
||||
response = _run_single_child(0, "Return fixture response", child, agent)
|
||||
assert response.get("status") == "completed", response
|
||||
results["full-child"] = captures[start:]
|
||||
|
||||
slot = {"provider": "fixture-local", "model": "fixture", "max_tokens": 23}
|
||||
config["moa"] = {"default_preset": "fixture", "presets": {"fixture": {
|
||||
"reference_models": [slot, slot], "aggregator": slot,
|
||||
"max_tokens": 29, "reference_max_tokens": 31,
|
||||
}}}
|
||||
config["auxiliary"] = {"compression": {"provider": "fixture-local", "model": "fixture", "reasoning_effort": "none", "max_output_tokens": 47}}
|
||||
config["_setup_done"] = True
|
||||
(home / "config.yaml").write_text(json.dumps(config))
|
||||
from agent.moa_loop import MoAClient
|
||||
start = len(captures)
|
||||
response = MoAClient("fixture").chat.completions.create(model="fixture", messages=[{"role": "user", "content": "Return fixture"}])
|
||||
assert response.choices[0].message.content == "LOCAL_CAPTURE_ONLY"
|
||||
results["full-moa"] = captures[start:]
|
||||
assert len(results["full-moa"]) == 3, results["full-moa"]
|
||||
|
||||
from agent.auxiliary_client import call_llm
|
||||
start = len(captures)
|
||||
response = call_llm(task="compression", messages=[{"role": "user", "content": "Return fixture"}])
|
||||
assert response.choices[0].message.content == "LOCAL_CAPTURE_ONLY"
|
||||
results["full-compression-call"] = captures[start:]
|
||||
|
||||
from agent.curator import _run_llm_review
|
||||
start = len(captures)
|
||||
response = _run_llm_review("Return the fixture response without tools.")
|
||||
assert response["final"] == "LOCAL_CAPTURE_ONLY", response
|
||||
results["full-curator-fork"] = captures[start:]
|
||||
|
||||
from agent.background_review import build_cache_parity_fork
|
||||
start = len(captures)
|
||||
review, _, _ = build_cache_parity_fork(agent, {}, max_iterations=2)
|
||||
try:
|
||||
response = review.run_conversation("Return fixture response without tools.")
|
||||
assert response["final_response"] == "LOCAL_CAPTURE_ONLY", response
|
||||
finally:
|
||||
review.close()
|
||||
results["full-review-fork"] = captures[start:]
|
||||
|
||||
agent.max_tokens = 43
|
||||
start = len(captures)
|
||||
review, _, _ = build_cache_parity_fork(agent, {}, max_iterations=2)
|
||||
try:
|
||||
response = review.run_conversation("Return fixture response without tools.")
|
||||
assert response["final_response"] == "LOCAL_CAPTURE_ONLY", response
|
||||
finally:
|
||||
review.close()
|
||||
agent.max_tokens = None
|
||||
results["internal-review-budget"] = captures[start:]
|
||||
assert results["internal-review-budget"][0]["body"]["max_tokens"] == 43
|
||||
|
||||
import boto3
|
||||
from agent.transports.bedrock import BedrockTransport
|
||||
native = boto3.client("bedrock-runtime", region_name="us-east-1", endpoint_url=url,
|
||||
aws_access_key_id="fixture", aws_secret_access_key="fixture")
|
||||
kwargs = BedrockTransport().build_kwargs("amazon.nova-pro-v1:0", [{"role": "user", "content": "fixture"}])
|
||||
kwargs.pop("__bedrock_converse__", None)
|
||||
kwargs.pop("__bedrock_region__", None)
|
||||
start = len(captures)
|
||||
native.converse(**kwargs)
|
||||
results["bedrock-sdk-local-not-aws"] = captures[start:]
|
||||
|
||||
master, slave = pty.openpty()
|
||||
env = {k: v for k, v in os.environ.items() if not any(s in k for s in ("API_KEY", "TOKEN", "SECRET"))}
|
||||
env.update(HOME=str(home), HERMES_HOME=str(home), HERMES_MAX_TOKENS="13", TERM="xterm", NO_COLOR="1")
|
||||
start = len(captures)
|
||||
proc = subprocess.Popen([sys.executable, "-m", "hermes_cli.main", "chat", "--provider", "fixture-local", "-m", "fixture", "-q", "Return fixture"], stdin=slave, stdout=slave, stderr=slave, cwd=os.getcwd(), env=env)
|
||||
os.close(slave)
|
||||
output = bytearray()
|
||||
deadline = time.monotonic() + 60
|
||||
try:
|
||||
while time.monotonic() < deadline:
|
||||
if select.select([master], [], [], 0.1)[0]:
|
||||
try:
|
||||
chunk = os.read(master, 65536)
|
||||
except OSError:
|
||||
break
|
||||
if not chunk:
|
||||
break
|
||||
output.extend(chunk)
|
||||
elif proc.poll() is not None:
|
||||
break
|
||||
if proc.poll() is None:
|
||||
proc.terminate()
|
||||
proc.wait(timeout=10)
|
||||
finally:
|
||||
os.close(master)
|
||||
(home / "cli-pty.txt").write_bytes(output)
|
||||
assert proc.returncode == 0 and b"LOCAL_CAPTURE_ONLY" in output, output.decode(errors="replace")
|
||||
results["cli-pty"] = captures[start:]
|
||||
assert results["cli-pty"]
|
||||
if os.environ.get("VERIFY_OUTPUT_CAP_REMOVAL"):
|
||||
for label, requests in results.items():
|
||||
if label == "internal-review-budget":
|
||||
continue
|
||||
assert requests, label
|
||||
for request in requests:
|
||||
assert "max_tokens" not in request["body"] and "max_completion_tokens" not in request["body"], (label, request)
|
||||
assert "maxTokens" not in request["body"].get("inferenceConfig", {}), (label, request)
|
||||
return results
|
||||
@@ -838,7 +838,7 @@ def test_review_fork_uses_runtime_model_and_output_cap(curator_env, monkeypatch)
|
||||
|
||||
assert result["error"] is None
|
||||
assert captured["model"] == "real-model-id"
|
||||
assert captured["max_tokens"] is None
|
||||
assert captured.get("max_tokens") is None
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -19,7 +19,7 @@ def _resolve(config, *, provider="ollama", model="qwen3:8b", requested_model=Non
|
||||
)
|
||||
|
||||
|
||||
def test_explicit_non_reasoning_compression_route_is_certified_and_bounded():
|
||||
def test_explicit_non_reasoning_compression_route_is_certified():
|
||||
lane = _resolve(
|
||||
{
|
||||
"provider": "ollama",
|
||||
@@ -155,7 +155,7 @@ def test_compression_latency_records_delayed_first_provider_chunk():
|
||||
assert timings["summary_generation_ms"] >= timings["time_to_first_progress_ms"]
|
||||
|
||||
|
||||
def test_certified_fast_lane_sends_the_configured_cap_to_its_provider():
|
||||
def test_certified_fast_lane_ignores_legacy_cap_and_preserves_reasoning():
|
||||
from agent.auxiliary_client import call_llm
|
||||
|
||||
config = {
|
||||
@@ -217,7 +217,7 @@ def test_uncertified_effective_primary_route_does_not_receive_fast_cap():
|
||||
assert "reasoning" not in request.get("extra_body", {})
|
||||
|
||||
|
||||
def test_boolean_cap_drift_stays_uncapped_and_preserves_existing_reasoning():
|
||||
def test_legacy_boolean_cap_does_not_bypass_route_certification():
|
||||
from agent.auxiliary_client import call_llm
|
||||
|
||||
config = {
|
||||
@@ -323,7 +323,7 @@ def test_summary_model_override_cap_uses_the_actual_primary_request():
|
||||
assert "max_tokens" not in request
|
||||
|
||||
|
||||
def test_fallback_cap_requires_independent_route_certification():
|
||||
def test_fallback_reasoning_requires_independent_route_certification():
|
||||
from agent.auxiliary_client import _call_fallback_candidate_sync
|
||||
|
||||
response = object()
|
||||
|
||||
@@ -13,10 +13,10 @@ import pytest
|
||||
|
||||
|
||||
class TestRunReferenceSlotMaxTokens:
|
||||
"""_run_reference should prefer slot-level max_tokens over preset-level."""
|
||||
"""Legacy slot settings cannot override internal advisor task budgets."""
|
||||
|
||||
def test_slot_max_tokens_overrides_preset_level(self):
|
||||
"""When slot has max_tokens, it overrides the preset-level cap."""
|
||||
def test_legacy_slot_max_tokens_cannot_override_internal_budget(self):
|
||||
"""An explicit internal caller budget wins over a stale user setting."""
|
||||
from agent.moa_loop import _run_reference
|
||||
|
||||
captured_kwargs: dict = {}
|
||||
|
||||
@@ -1164,7 +1164,7 @@ class TestModelOverrides:
|
||||
assert info is not None
|
||||
assert info.family == "llava"
|
||||
assert info.context_window == 8192
|
||||
assert info.max_output is None
|
||||
assert info.max_output == 0
|
||||
assert info.tool_call is True
|
||||
assert info.reasoning is False
|
||||
|
||||
|
||||
@@ -73,7 +73,7 @@ def test_routing_to_different_model_marks_routed_and_resolves_credentials():
|
||||
assert rt["api_key"] == "or-key"
|
||||
assert rt["credential_pool"] == "routed-pool"
|
||||
assert rt["request_overrides"] == {"extra_body": {"store": False}}
|
||||
assert rt["max_tokens"] is None
|
||||
assert rt.get("max_tokens") is None
|
||||
|
||||
|
||||
def test_unrouted_runtime_keeps_parent_pool_and_overrides():
|
||||
|
||||
@@ -58,7 +58,7 @@ def test_direct_branch_forwards_request_overrides():
|
||||
assert creds["request_overrides"] == {
|
||||
"extra_body": {"provider": {"sort": "throughput"}},
|
||||
}
|
||||
# Shape parity with the named-provider branch: max_output_tokens present.
|
||||
# Dedicated user cap controls are not part of child credentials.
|
||||
assert "max_output_tokens" not in creds
|
||||
|
||||
|
||||
@@ -93,7 +93,7 @@ def test_explicit_merges_over_runtime_on_provider_alongside_base_url(mock_resolv
|
||||
"""Precedence on the provider-alongside-base_url path (#98237 interplay):
|
||||
explicit delegation.request_overrides merges OVER the named provider's
|
||||
runtime overrides — runtime extra_body keys survive unless redefined,
|
||||
explicit top-level keys win, and max_output_tokens is preserved."""
|
||||
explicit top-level keys win, and legacy max_output_tokens is ignored."""
|
||||
mock_resolve.return_value = {
|
||||
"provider": "custom",
|
||||
"base_url": "https://provider-default.example/v1",
|
||||
|
||||
@@ -141,8 +141,8 @@ a human in a browser.
|
||||
`actual models load` takes the INSTALLED name from `actual models list`.
|
||||
4. **Reasoning models returning empty content.** GLM/Qwen reasoning variants
|
||||
emit thinking in a separate `reasoning` field and can burn a small
|
||||
`max_tokens` entirely on reasoning. Give generous max_tokens before
|
||||
assuming failure.
|
||||
output budget entirely on reasoning. Check the server-side output default
|
||||
before assuming failure.
|
||||
5. **Do not create a custom provider named `actual`.** Older setup guides
|
||||
(pre first-class support) wrote `providers.actual.*` config blocks. On
|
||||
current Hermes the built-in provider wins the name; stale custom blocks
|
||||
|
||||
+4
-4
@@ -720,7 +720,7 @@ hermes model
|
||||
**工具调用:** 使用 `--tool-call-parser` 并选择适合你模型系列的解析器:`qwen`(Qwen 2.5)、`llama3`、`llama4`、`deepseekv3`、`mistral`、`glm`。没有此标志,工具调用将以纯文本返回。
|
||||
|
||||
:::caution SGLang 默认最大输出 128 tokens
|
||||
如果响应看起来被截断,在请求中添加 `max_tokens` 或在服务器上设置 `--default-max-tokens`。SGLang 的默认值是每次响应仅 128 tokens(如果请求中未指定)。
|
||||
如果响应看起来被截断,在服务器上设置 `--default-max-tokens`。SGLang 的默认值是每次响应仅 128 tokens(如果请求中未指定)。
|
||||
:::
|
||||
|
||||
---
|
||||
@@ -977,7 +977,7 @@ model:
|
||||
#### 响应在句子中间被截断
|
||||
|
||||
**可能原因:**
|
||||
1. **服务器上的输出上限(`max_tokens`)过低** — SGLang 默认每次响应 128 tokens。在服务器上设置 `--default-max-tokens`,或在 config.yaml 中配置 `model.max_tokens`。注意:`max_tokens` 只控制响应长度——与对话历史可以有多长无关(那是 `context_length`)。
|
||||
1. **服务器上的输出上限(`max_tokens`)过低** — SGLang 默认每次响应 128 tokens。在服务器上设置 `--default-max-tokens`;Hermes 不再提供用户输出上限设置。注意:`max_tokens` 只控制响应长度——与对话历史可以有多长无关(那是 `context_length`)。
|
||||
2. **上下文耗尽** — 模型填满了上下文窗口。增加 `model.context_length` 或在 Hermes 中启用[上下文压缩](/user-guide/configuration#context-compression)。
|
||||
|
||||
---
|
||||
@@ -1075,10 +1075,10 @@ model:
|
||||
:::note 两个设置,容易混淆
|
||||
**`context_length`** 是**总上下文窗口**——输入和输出 token 的合计预算(例如 Claude Opus 4.6 为 200,000)。Hermes 用它来决定何时压缩历史记录以及验证 API 请求。
|
||||
|
||||
**`model.max_tokens`** 是**输出上限**——模型在*单次响应*中最多可生成的 token 数。与对话历史可以有多长无关。行业标准名称 `max_tokens` 是常见的混淆来源;Anthropic 的原生 API 已将其重命名为 `max_output_tokens` 以更清晰。
|
||||
**输出上限**限制单次响应,而非对话历史。Hermes 不再读取 `model.max_tokens`、`HERMES_MAX_TOKENS` 或提供商及模型的输出上限设置。兼容端点使用服务器默认值;该值不一定是模型最大值。原生 Anthropic Messages 仍要求 `max_tokens`,Hermes 会提供内部值。Bedrock Converse 的可选输出限制默认省略。
|
||||
|
||||
当自动检测获取的窗口大小不正确时,设置 `context_length`。
|
||||
仅当需要限制单次响应长度时,才设置 `model.max_tokens`。
|
||||
请删除旧的用户输出上限配置;内部任务预算与 MCP 采样安全预算保持不变。
|
||||
:::
|
||||
|
||||
Hermes 使用多源解析链来检测模型和提供商的正确上下文窗口:
|
||||
|
||||
-1
@@ -68,7 +68,6 @@ python batch_runner.py --list_distributions
|
||||
| `--resume` | `false` | 从断点恢复 |
|
||||
| `--verbose` | `false` | 启用详细日志 |
|
||||
| `--max_samples` | 全部 | 仅处理数据集中前 N 条样本 |
|
||||
| `--max_tokens` | 模型默认值 | 每次模型响应的最大 token 数 |
|
||||
|
||||
### 供应商路由(OpenRouter)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user