fix(agent): reasoning-off continuation reaches the wire on the legacy chat path; reset one-shot flag per turn

Follow-up to the #99622 salvage:
- agent/transports/chat_completions.py: the legacy (no provider profile)
  chat_completions path always re-emitted extra_body.reasoning with
  enabled=True, so both reasoning_effort: none and the one-shot
  length-continuation override went out as {enabled: true, effort: none}.
  Honor enabled=False / effort=none the way the profile path does.
- agent/conversation_loop.py: reset agent._ephemeral_reasoning_off at
  turn start so a flag armed by an interrupted/errored turn can never
  strip thinking from the next turn's first request.
- User-facing hints now name the real slash command (/reasoning); the
  /thinkon//thinkoff commands do not exist.
- tests: wire-level regression (continuation request carries
  reasoning.enabled=false) and a stale-flag turn-scope test.
This commit is contained in:
Teknium
2026-09-01 21:22:56 -07:00
parent fb76fb0526
commit 2a0605a807
4 changed files with 59 additions and 6 deletions
+9 -3
View File
@@ -2160,6 +2160,12 @@ def run_conversation(
failed = False
codex_ack_continuations = 0
length_continue_retries = 0
# One-shot "continue without thinking" override is turn-scoped: a
# thinking-only truncation arms it right before the continuation restart,
# and build_api_kwargs consumes it on that call. If the turn is
# interrupted/errors between arm and consume, it must not fire on the
# next turn's first request.
agent._ephemeral_reasoning_off = False
# Total outer-loop exceptions this turn (#92450) — see _MAX_OUTER_LOOP_ERRORS.
_outer_error_count = 0
truncated_tool_call_retries = 0
@@ -4163,7 +4169,7 @@ def run_conversation(
"The model used all its output tokens on reasoning "
"and had none left for the actual response.\n\n"
"To fix this:\n"
"→ Lower reasoning effort: `/thinkon low` or `/thinkon minimal`\n"
"→ Lower reasoning effort: `/reasoning low` or `/reasoning minimal`\n"
"→ Or switch to a larger/non-reasoning model with `/model`"
)
agent._cleanup_task_resources(effective_task_id)
@@ -4406,8 +4412,8 @@ def run_conversation(
"continuation attempt — its reasoning "
"consumed the entire budget each time.\n\n"
"To fix this:\n"
"→ Lower reasoning effort: `/thinkon low` "
"or `/thinkoff`\n"
"→ Lower reasoning effort: `/reasoning low` "
"or `/reasoning none`\n"
"→ Or raise max_tokens for this model"
)
# Unanswered continue nudges made every later turn re-truncate.
+10 -1
View File
@@ -767,9 +767,18 @@ class ChatCompletionsTransport(ProviderTransport):
extra_body["reasoning"] = gh_reasoning
else:
_effort = "medium"
_enabled = True
if reasoning_config and isinstance(reasoning_config, dict):
_effort = reasoning_config.get("effort", "medium") or "medium"
extra_body["reasoning"] = {"enabled": True, "effort": _effort}
# Honor an explicit "thinking off" (agent.reasoning_effort:
# none / the one-shot length-continuation override) the same
# way the provider-profile path does — never re-enable it.
if reasoning_config.get("enabled") is False or _effort == "none":
_enabled = False
if _enabled:
extra_body["reasoning"] = {"enabled": True, "effort": _effort}
else:
extra_body["reasoning"] = {"enabled": False, "effort": "none"}
if provider_name == "gemini":
raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config)
@@ -225,4 +225,42 @@ class TestThinkingOnlyTruncation:
assert result["completed"] is True
assert "visible part one." in (result["final_response"] or "")
assert "and the ending." in (result["final_response"] or "")
assert _no_empty_assistant_rows(result["messages"]) == []
assert _no_empty_assistant_rows(result["messages"]) == []
class TestReasoningOffReachesTheWire:
def test_continuation_request_carries_reasoning_off_on_the_wire(self, loop_agent):
"""The flag is only useful if the continuation REQUEST goes out with
thinking disabled — assert the OpenRouter extra_body, not the flag."""
loop_agent.reasoning_config = {"enabled": True, "effort": "high"}
loop_agent._supports_reasoning_extra_body = lambda: True
loop_agent.client.chat.completions.create.side_effect = [
_thinking_only_length_response(),
_full_response("Here is the full answer."),
]
result = _run(loop_agent, "write me a long report")
assert result["completed"] is True
calls = loop_agent.client.chat.completions.create.call_args_list
assert len(calls) == 2
first = (calls[0].kwargs.get("extra_body") or {}).get("reasoning")
second = (calls[1].kwargs.get("extra_body") or {}).get("reasoning")
assert first == {"enabled": True, "effort": "high"}, first
assert second is not None and second.get("enabled") is False, (
f"continuation must be sent with thinking off, got {second!r}"
)
def test_stale_flag_does_not_leak_into_next_turn(self, loop_agent):
"""A flag armed by a previous turn that never reached build_api_kwargs
(interrupt/error between arm and consume) must not silently strip
thinking from the next turn's first request."""
loop_agent.reasoning_config = {"enabled": True, "effort": "high"}
loop_agent._supports_reasoning_extra_body = lambda: True
loop_agent._ephemeral_reasoning_off = True # stale from a prior turn
loop_agent.client.chat.completions.create.side_effect = [
_full_response("fresh turn answer."),
]
result = _run(loop_agent, "hello")
assert result["completed"] is True
calls = loop_agent.client.chat.completions.create.call_args_list
first = (calls[0].kwargs.get("extra_body") or {}).get("reasoning")
assert first == {"enabled": True, "effort": "high"}, first
+1 -1
View File
@@ -4421,7 +4421,7 @@ class TestRunConversation:
# Should have a user-friendly response (not None)
assert result["final_response"] is not None
assert "Thinking Budget Exhausted" in result["final_response"]
assert "/thinkon" in result["final_response"]
assert "/reasoning" in result["final_response"]
def test_length_with_tool_calls_returns_partial_without_executing_tools(self, agent):