fix(agent): reasoning-off continuation reaches the wire on the legacy chat path; reset one-shot flag per turn
Follow-up to the #99622 salvage: - agent/transports/chat_completions.py: the legacy (no provider profile) chat_completions path always re-emitted extra_body.reasoning with enabled=True, so both reasoning_effort: none and the one-shot length-continuation override went out as {enabled: true, effort: none}. Honor enabled=False / effort=none the way the profile path does. - agent/conversation_loop.py: reset agent._ephemeral_reasoning_off at turn start so a flag armed by an interrupted/errored turn can never strip thinking from the next turn's first request. - User-facing hints now name the real slash command (/reasoning); the /thinkon//thinkoff commands do not exist. - tests: wire-level regression (continuation request carries reasoning.enabled=false) and a stale-flag turn-scope test.
This commit is contained in:
@@ -2160,6 +2160,12 @@ def run_conversation(
|
||||
failed = False
|
||||
codex_ack_continuations = 0
|
||||
length_continue_retries = 0
|
||||
# One-shot "continue without thinking" override is turn-scoped: a
|
||||
# thinking-only truncation arms it right before the continuation restart,
|
||||
# and build_api_kwargs consumes it on that call. If the turn is
|
||||
# interrupted/errors between arm and consume, it must not fire on the
|
||||
# next turn's first request.
|
||||
agent._ephemeral_reasoning_off = False
|
||||
# Total outer-loop exceptions this turn (#92450) — see _MAX_OUTER_LOOP_ERRORS.
|
||||
_outer_error_count = 0
|
||||
truncated_tool_call_retries = 0
|
||||
@@ -4163,7 +4169,7 @@ def run_conversation(
|
||||
"The model used all its output tokens on reasoning "
|
||||
"and had none left for the actual response.\n\n"
|
||||
"To fix this:\n"
|
||||
"→ Lower reasoning effort: `/thinkon low` or `/thinkon minimal`\n"
|
||||
"→ Lower reasoning effort: `/reasoning low` or `/reasoning minimal`\n"
|
||||
"→ Or switch to a larger/non-reasoning model with `/model`"
|
||||
)
|
||||
agent._cleanup_task_resources(effective_task_id)
|
||||
@@ -4406,8 +4412,8 @@ def run_conversation(
|
||||
"continuation attempt — its reasoning "
|
||||
"consumed the entire budget each time.\n\n"
|
||||
"To fix this:\n"
|
||||
"→ Lower reasoning effort: `/thinkon low` "
|
||||
"or `/thinkoff`\n"
|
||||
"→ Lower reasoning effort: `/reasoning low` "
|
||||
"or `/reasoning none`\n"
|
||||
"→ Or raise max_tokens for this model"
|
||||
)
|
||||
# Unanswered continue nudges made every later turn re-truncate.
|
||||
|
||||
@@ -767,9 +767,18 @@ class ChatCompletionsTransport(ProviderTransport):
|
||||
extra_body["reasoning"] = gh_reasoning
|
||||
else:
|
||||
_effort = "medium"
|
||||
_enabled = True
|
||||
if reasoning_config and isinstance(reasoning_config, dict):
|
||||
_effort = reasoning_config.get("effort", "medium") or "medium"
|
||||
extra_body["reasoning"] = {"enabled": True, "effort": _effort}
|
||||
# Honor an explicit "thinking off" (agent.reasoning_effort:
|
||||
# none / the one-shot length-continuation override) the same
|
||||
# way the provider-profile path does — never re-enable it.
|
||||
if reasoning_config.get("enabled") is False or _effort == "none":
|
||||
_enabled = False
|
||||
if _enabled:
|
||||
extra_body["reasoning"] = {"enabled": True, "effort": _effort}
|
||||
else:
|
||||
extra_body["reasoning"] = {"enabled": False, "effort": "none"}
|
||||
|
||||
if provider_name == "gemini":
|
||||
raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config)
|
||||
|
||||
@@ -225,4 +225,42 @@ class TestThinkingOnlyTruncation:
|
||||
assert result["completed"] is True
|
||||
assert "visible part one." in (result["final_response"] or "")
|
||||
assert "and the ending." in (result["final_response"] or "")
|
||||
assert _no_empty_assistant_rows(result["messages"]) == []
|
||||
assert _no_empty_assistant_rows(result["messages"]) == []
|
||||
|
||||
class TestReasoningOffReachesTheWire:
|
||||
def test_continuation_request_carries_reasoning_off_on_the_wire(self, loop_agent):
|
||||
"""The flag is only useful if the continuation REQUEST goes out with
|
||||
thinking disabled — assert the OpenRouter extra_body, not the flag."""
|
||||
loop_agent.reasoning_config = {"enabled": True, "effort": "high"}
|
||||
loop_agent._supports_reasoning_extra_body = lambda: True
|
||||
loop_agent.client.chat.completions.create.side_effect = [
|
||||
_thinking_only_length_response(),
|
||||
_full_response("Here is the full answer."),
|
||||
]
|
||||
result = _run(loop_agent, "write me a long report")
|
||||
assert result["completed"] is True
|
||||
|
||||
calls = loop_agent.client.chat.completions.create.call_args_list
|
||||
assert len(calls) == 2
|
||||
first = (calls[0].kwargs.get("extra_body") or {}).get("reasoning")
|
||||
second = (calls[1].kwargs.get("extra_body") or {}).get("reasoning")
|
||||
assert first == {"enabled": True, "effort": "high"}, first
|
||||
assert second is not None and second.get("enabled") is False, (
|
||||
f"continuation must be sent with thinking off, got {second!r}"
|
||||
)
|
||||
|
||||
def test_stale_flag_does_not_leak_into_next_turn(self, loop_agent):
|
||||
"""A flag armed by a previous turn that never reached build_api_kwargs
|
||||
(interrupt/error between arm and consume) must not silently strip
|
||||
thinking from the next turn's first request."""
|
||||
loop_agent.reasoning_config = {"enabled": True, "effort": "high"}
|
||||
loop_agent._supports_reasoning_extra_body = lambda: True
|
||||
loop_agent._ephemeral_reasoning_off = True # stale from a prior turn
|
||||
loop_agent.client.chat.completions.create.side_effect = [
|
||||
_full_response("fresh turn answer."),
|
||||
]
|
||||
result = _run(loop_agent, "hello")
|
||||
assert result["completed"] is True
|
||||
calls = loop_agent.client.chat.completions.create.call_args_list
|
||||
first = (calls[0].kwargs.get("extra_body") or {}).get("reasoning")
|
||||
assert first == {"enabled": True, "effort": "high"}, first
|
||||
|
||||
@@ -4421,7 +4421,7 @@ class TestRunConversation:
|
||||
# Should have a user-friendly response (not None)
|
||||
assert result["final_response"] is not None
|
||||
assert "Thinking Budget Exhausted" in result["final_response"]
|
||||
assert "/thinkon" in result["final_response"]
|
||||
assert "/reasoning" in result["final_response"]
|
||||
|
||||
|
||||
def test_length_with_tool_calls_returns_partial_without_executing_tools(self, agent):
|
||||
|
||||
Reference in New Issue
Block a user