From 0b69a6ac021c454ef6496d943ccd61d241e51dda Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Sat, 1 Aug 2026 16:09:56 -0500 Subject: [PATCH] feat(kanban): let a task pin its own thinking depth MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A task could already pin a model and provider, but not how hard the worker thinks: reasoning effort came from the assigned profile's config and nothing per-task could reach it. Pairing a small model with high effort, or a big one with thinking off, meant editing the worker profile itself. Adds a tasks.reasoning_effort column (migrated, NULL = inherit the profile) with set_reasoning_effort(), a create_task kwarg, and a --reasoning spawn flag. Kept deliberately independent of model_override: a task may run the profile's own model at a different depth, and clearing a model override no longer resets the depth the operator chose. "none" is a value (thinking off), not a clear. --reasoning is new on the CLI too — the level was only reachable through the /reasoning slash command, so the dispatcher had no flag to pass. It overrides agent.reasoning_effort for one run and is never persisted. --- cli.py | 19 ++++++++ hermes_cli/_parser.py | 23 +++++++++ hermes_cli/kanban_db.py | 101 +++++++++++++++++++++++++++++++++++++++- hermes_cli/main.py | 1 + 4 files changed, 143 insertions(+), 1 deletion(-) diff --git a/cli.py b/cli.py index 6a81546911..0dfe7b34fc 100644 --- a/cli.py +++ b/cli.py @@ -4205,6 +4205,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): model: str = None, toolsets: List[str] = None, provider: str = None, + reasoning: str = None, api_key: str = None, base_url: str = None, max_turns: int = None, @@ -4222,6 +4223,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): model: Model to use (default: from env or claude-sonnet) toolsets: List of toolsets to enable (default: all) provider: Inference provider ("auto", "openrouter", "nous", "openai-codex", "zai", "kimi-coding", "minimax", "minimax-cn") + reasoning: Reasoning effort override for this run (none|minimal|low|medium|high|xhigh|max|ultra). Wins over config. api_key: API key (default: from environment) base_url: API base URL (default: OpenRouter) max_turns: Maximum tool-calling iterations shared with subagents (default: 500) @@ -4487,6 +4489,20 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin): # shared chokepoint in hermes_constants (Closes #21256). from hermes_constants import resolve_reasoning_config self.reasoning_config = resolve_reasoning_config(CLI_CONFIG, self.model) + # An explicit --reasoning wins over config for this run only (never + # persisted). Kanban's dispatcher uses it to pin a task's thinking + # depth without touching the worker profile's config.yaml. An + # unparseable level is ignored with a warning rather than silently + # swapping in the default — same contract as the config path. + if reasoning is not None and str(reasoning).strip(): + _cli_reasoning = _parse_reasoning_config(reasoning) + if _cli_reasoning is None: + logger.warning( + "Unknown --reasoning '%s', keeping the configured level", + reasoning, + ) + else: + self.reasoning_config = _cli_reasoning self.service_tier = _parse_service_tier_config( CLI_CONFIG["agent"].get("service_tier", "") ) @@ -17842,6 +17858,7 @@ def main( skills: str | list[str] | tuple[str, ...] = None, model: str = None, provider: str = None, + reasoning: str = None, api_key: str = None, base_url: str = None, max_turns: int = None, @@ -17870,6 +17887,7 @@ def main( skills: Comma-separated or repeated list of skills to preload for the session model: Model to use (default: anthropic/claude-opus-4-20250514) provider: Inference provider ("auto", "openrouter", "nous", "openai-codex", "zai", "kimi-coding", "minimax", "minimax-cn") + reasoning: Reasoning effort for this run (none|minimal|low|medium|high|xhigh|max|ultra). Overrides agent.reasoning_effort. api_key: API key for authentication base_url: Base URL for the API max_turns: Maximum tool-calling iterations (default: 60) @@ -17984,6 +18002,7 @@ def main( model=model, toolsets=toolsets_list, provider=provider, + reasoning=reasoning, api_key=api_key, base_url=base_url, max_turns=max_turns, diff --git a/hermes_cli/_parser.py b/hermes_cli/_parser.py index cb5be6b935..b5098f6c98 100644 --- a/hermes_cli/_parser.py +++ b/hermes_cli/_parser.py @@ -147,6 +147,18 @@ def build_top_level_parser(): "under model.provider — use `hermes setup` or edit the file to change it." ), ) + _inherited_flag( + parser, + "--reasoning", + default=None, + metavar="LEVEL", + help=( + "Reasoning effort for this invocation: none, minimal, low, medium, " + "high, xhigh, max, or ultra. Overrides agent.reasoning_effort in " + "config.yaml for this run only; the persistent level lives there " + "(or per-model under agent.reasoning_overrides)." + ), + ) parser.add_argument( "-t", "--toolsets", @@ -299,6 +311,17 @@ def build_top_level_parser(): default=argparse.SUPPRESS, help="Comma-separated toolsets to enable", ) + _inherited_flag( + chat_parser, + "--reasoning", + default=argparse.SUPPRESS, + metavar="LEVEL", + help=( + "Reasoning effort for this session: none, minimal, low, medium, " + "high, xhigh, max, or ultra. Overrides agent.reasoning_effort for " + "this run only (same levels as the /reasoning slash command)." + ), + ) _inherited_flag( chat_parser, "-s", diff --git a/hermes_cli/kanban_db.py b/hermes_cli/kanban_db.py index 72fa92cf20..b64d54e53a 100644 --- a/hermes_cli/kanban_db.py +++ b/hermes_cli/kanban_db.py @@ -133,6 +133,30 @@ VALID_BLOCK_KINDS = {"dependency", "needs_input", "capability", "transient"} # not dispatcher spawn/crash/timeout failures. BLOCK_RECURRENCE_LIMIT = 2 VALID_WORKSPACE_KINDS = {"scratch", "worktree", "dir"} + + +def normalize_reasoning_effort(effort: Optional[str]) -> Optional[str]: + """Normalize a per-task reasoning effort into a storable level. + + Accepts any level in ``hermes_constants.VALID_REASONING_EFFORTS`` plus + ``"none"`` (thinking disabled), case-insensitively. Empty / None means + "inherit the worker profile's own ``agent.reasoning_effort``" and stores + NULL. Anything else is rejected rather than silently dropped — a typo'd + level must not quietly hand the task back to the profile default. + """ + from hermes_constants import VALID_REASONING_EFFORTS + + value = str(effort or "").strip().lower() + if not value: + return None + if value == "none" or value in VALID_REASONING_EFFORTS: + return value + allowed = ", ".join(("none", *VALID_REASONING_EFFORTS)) + raise ValueError( + f"reasoning_effort must be one of {allowed}, got {effort!r}" + ) + + KNOWN_TOOLSET_NAMES = frozenset(name.casefold() for name in get_toolset_names()) _IS_WINDOWS = sys.platform == "win32" KANBAN_ATTACHMENT_MAX_BYTES = 25 * 1024 * 1024 @@ -929,6 +953,12 @@ class Task: # model (pre-existing behaviour). Solves the "model from provider A, # profile configured for provider B" mismatch class. provider_override: Optional[str] = None + # Per-task reasoning effort for the worker (one of + # ``hermes_constants.VALID_REASONING_EFFORTS``, or ``"none"`` for thinking + # off). When set, the dispatcher passes ``--reasoning `` so the + # worker runs at that depth regardless of the profile's + # ``agent.reasoning_effort``. NULL = the worker profile's own setting. + reasoning_effort: Optional[str] = None # Per-task override for the consecutive-failure circuit breaker. # The value is the failure count at which the breaker trips — e.g. # ``max_retries=1`` blocks on the first failure (zero retries), @@ -1032,6 +1062,11 @@ class Task: if "provider_override" in keys and row["provider_override"] else None ), + reasoning_effort=( + row["reasoning_effort"] + if "reasoning_effort" in keys and row["reasoning_effort"] + else None + ), max_retries=( row["max_retries"] if "max_retries" in keys else None ), @@ -1200,6 +1235,11 @@ CREATE TABLE IF NOT EXISTS tasks ( -- worker resolves the model against the right backend instead of the -- profile's configured provider. NULL = profile provider. provider_override TEXT, + -- Per-task reasoning effort for the worker (minimal|low|medium|high| + -- xhigh|max|ultra, or 'none' for thinking off). When set, the dispatcher + -- passes --reasoning so the worker runs at that depth regardless + -- of the profile's agent.reasoning_effort. NULL = profile setting. + reasoning_effort TEXT, -- Per-task override for the consecutive-failure circuit breaker. -- The value is the failure count at which the breaker trips — e.g. -- ``max_retries=1`` blocks on the first failure. NULL (the common @@ -2388,6 +2428,13 @@ def _migrate_add_optional_columns(conn: sqlite3.Connection) -> None: conn, "tasks", "provider_override", "provider_override TEXT" ) + if "reasoning_effort" not in cols: + # Per-task thinking depth for the worker. NULL = the worker profile's + # own agent.reasoning_effort, which is what existing rows were getting. + _add_column_if_missing( + conn, "tasks", "reasoning_effort", "reasoning_effort TEXT" + ) + if "goal_mode" not in cols: # Ralph-style goal loop toggle for the dispatched worker. 0 (the # default) = classic single-shot worker, preserving the behaviour @@ -2851,6 +2898,7 @@ def create_task( max_retries: Optional[int] = None, model_override: Optional[str] = None, provider_override: Optional[str] = None, + reasoning_effort: Optional[str] = None, goal_mode: bool = False, goal_max_turns: Optional[int] = None, initial_status: str = "running", @@ -2887,6 +2935,11 @@ def create_task( config — passed to the worker as ``-m [--provider ]``. ``provider_override`` requires ``model_override``. + ``reasoning_effort`` pins the worker's thinking depth for this task + (``minimal``…``ultra``, or ``none`` to disable thinking), passed as + ``--reasoning ``. It is independent of ``model_override``: a task + can run the profile's own model at a different depth. + ``project_source_task_id`` is an internal cross-profile fallback for a worker-created child. When the active profile cannot resolve ``project_id`` in its own projects.db, a matching canonical project-linked task in this @@ -2895,6 +2948,7 @@ def create_task( """ model_override = (model_override or "").strip() or None provider_override = (provider_override or "").strip() or None + reasoning_effort = normalize_reasoning_effort(reasoning_effort) if provider_override and not model_override: raise ValueError("provider_override requires a model_override") assignee = _canonical_assignee(assignee) @@ -3162,8 +3216,9 @@ def create_task( branch_name, project_id, tenant, idempotency_key, max_runtime_seconds, skills, max_retries, model_override, provider_override, + reasoning_effort, goal_mode, goal_max_turns, session_id - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) """, ( task_id, @@ -3185,6 +3240,7 @@ def create_task( int(max_retries) if max_retries is not None else None, model_override, provider_override, + reasoning_effort, 1 if goal_mode else 0, int(goal_max_turns) if goal_max_turns is not None else None, session_id, @@ -3427,6 +3483,44 @@ def set_model_override( return True +def set_reasoning_effort( + conn: sqlite3.Connection, + task_id: str, + effort: Optional[str], +) -> bool: + """Set (or clear) the per-task reasoning effort. + + ``effort=None`` (or empty) clears the override — the worker falls back to + its profile's own ``agent.reasoning_effort``. ``"none"`` is a real value, + not a clear: it pins thinking OFF for this task. + + Deliberately independent of :func:`set_model_override`: a task may run the + profile's own model at a different depth, and clearing a model override + must not silently reset the depth the operator chose. Like the model + override, it takes effect on the NEXT dispatch, so it is settable on a + running task. Returns True on success. + """ + effort = normalize_reasoning_effort(effort) + with write_txn(conn): + row = conn.execute( + "SELECT status FROM tasks WHERE id = ?", (task_id,) + ).fetchone() + if not row: + return False + if row["status"] == "archived": + raise RuntimeError( + f"cannot set reasoning effort on archived task {task_id}" + ) + conn.execute( + "UPDATE tasks SET reasoning_effort = ? WHERE id = ?", + (effort, task_id), + ) + _append_event( + conn, task_id, "reasoning_effort_set", {"reasoning_effort": effort} + ) + return True + + # --------------------------------------------------------------------------- # Links # --------------------------------------------------------------------------- @@ -9026,6 +9120,11 @@ def _default_spawn( # the classic mis-set that stalls a board). if task.provider_override: cmd.extend(["--provider", task.provider_override]) + # Per-task thinking depth. Independent of the model override — a task can + # run the profile's own model at a different depth — so this is its own + # branch, not a nested one. + if task.reasoning_effort: + cmd.extend(["--reasoning", task.reasoning_effort]) worker_toolsets = _resolve_worker_cli_toolsets(env.get("HERMES_HOME")) if worker_toolsets: cmd.extend(["--toolsets", ",".join(worker_toolsets)]) diff --git a/hermes_cli/main.py b/hermes_cli/main.py index a8fb2ffc47..cfc331d9b1 100644 --- a/hermes_cli/main.py +++ b/hermes_cli/main.py @@ -2693,6 +2693,7 @@ def cmd_chat(args): kwargs = { "model": args.model, "provider": getattr(args, "provider", None), + "reasoning": getattr(args, "reasoning", None), "toolsets": args.toolsets, "skills": getattr(args, "skills", None), "verbose": getattr(args, "verbose", None),