From 7d368c7d2cda1338ab6411e79e18a84002efb574 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Sun, 13 Sep 2026 20:10:55 -0700 Subject: [PATCH] docs: background_review.reasoning_effort applies on the routed path; same-model warns once --- agent/background_review.py | 6 +++--- website/docs/user-guide/configuration.md | 2 +- website/docs/user-guide/features/memory.md | 2 +- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/agent/background_review.py b/agent/background_review.py index 9627bc92bc..ee4557b59a 100644 --- a/agent/background_review.py +++ b/agent/background_review.py @@ -923,9 +923,9 @@ def build_cache_parity_fork( # OAuth-only providers, session-scoped creds and credential pools. _rt = _resolve_review_runtime(agent, task_cfg) _routed = bool(_rt.get("routed")) - # A configured effort is silently dropped on the same-model path (cache parity) — say so once, - # visible, instead of leaving the set-but-ignored key invisible (#104116). Routed forks are a - # separate, tracked issue (#94825) and are left alone. + # A configured effort is dropped on the same-model path (cache parity) — say so once, visible, + # instead of leaving the set-but-ignored key invisible (#104116). Routed forks honor it + # (_routed_reasoning_config). if not _routed and write_origin == "background_review": _warn_ignored_reasoning_effort(agent, task_cfg) review_agent = AIAgent(**_fork_init_kwargs(agent, _rt, _routed, max_iterations, task_cfg)) diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 52cfa28403..b73afaadd5 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -1372,7 +1372,7 @@ Auxiliary task blocks additionally accept a `reasoning_effort` knob: This is the per-task counterpart of the global `agent.reasoning_effort`: run compression at `low` or vision at `none` to cut side-task latency and cost when your main model is an expensive reasoning model, without touching your main chat behavior. It applies to auxiliary-client tasks such as `vision`, `compression`, `title_generation`, and `curator`, across all three auxiliary wire formats (chat completions, Codex Responses, Anthropic Messages). An explicit `extra_body.reasoning` on the same task wins over the shorthand. -**Background review is different:** a same-model review fork always inherits the parent's reasoning effort. `auxiliary.background_review.reasoning_effort` is ignored on that path, including when the parent provider/model is explicitly selected. This preserves byte-identical reasoning settings, system prompt, full conversation snapshot, and tool definitions for prompt-cache parity; there is no independent-effort switch for same-model reviews. See [background review reasoning](/user-guide/features/memory#same-model-review-reasoning). The separate routed-fork effort issue is tracked in [#94825](https://github.com/NousResearch/hermes-agent/issues/94825). +**Background review is different:** a same-model review fork always inherits the parent's reasoning effort. `auxiliary.background_review.reasoning_effort` is ignored on that path, including when the parent provider/model is explicitly selected. This preserves byte-identical reasoning settings, system prompt, full conversation snapshot, and tool definitions for prompt-cache parity; there is no independent-effort switch for same-model reviews. See [background review reasoning](/user-guide/features/memory#same-model-review-reasoning). When the review is routed to a different provider/model, `reasoning_effort` applies to that routed fork (unset = the routed provider's default). Hermes prints a one-time warning when the key is set but the review runs on the main model. **MoA also uses a different configuration:** reasoning depth for Mixture-of-Agents is configured **per slot** in the MoA preset (`moa.presets..reference_models[].reasoning_effort` / `aggregator.reasoning_effort`), not on the `moa_reference`/`moa_aggregator` auxiliary blocks — see [Mixture of Agents](/user-guide/features/mixture-of-agents). diff --git a/website/docs/user-guide/features/memory.md b/website/docs/user-guide/features/memory.md index bb916ac822..6e252e6a1a 100644 --- a/website/docs/user-guide/features/memory.md +++ b/website/docs/user-guide/features/memory.md @@ -351,7 +351,7 @@ A review using the same model as the parent **always inherits the parent's reaso Reasoning settings, the system prompt, the full conversation snapshot, and tool definitions stay byte-identical to the parent at fork birth so the review can reuse its prompt-cache prefix. Changing only the review's thinking level would break that parity. There is no independent-effort switch for same-model reviews. -To reduce review work without changing the main conversation's effort, adjust `memory.nudge_interval` / `skills.creation_nudge_interval`, disable automatic reviews as described below, or route reviews to a different model. A different-model route uses a digest and does not share the parent's warm prefix; its separate task-effort bug is tracked in [#94825](https://github.com/NousResearch/hermes-agent/issues/94825). These frequency and routing controls do not decouple same-model reasoning. +To reduce review work without changing the main conversation's effort, adjust `memory.nudge_interval` / `skills.creation_nudge_interval`, disable automatic reviews as described below, or route reviews to a different model. A different-model route uses a digest and does not share the parent's warm prefix; on that route `auxiliary.background_review.reasoning_effort` IS honored (unset = the routed provider's default). A one-time warning is printed when the key is set but the review stays on the main model. These frequency and routing controls do not decouple same-model reasoning. ### Disabling automatic reviews (`enabled`)