diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index e413b83ff9..4c2d5d4949 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -2160,6 +2160,12 @@ def run_conversation( failed = False codex_ack_continuations = 0 length_continue_retries = 0 + # One-shot "continue without thinking" override is turn-scoped: a + # thinking-only truncation arms it right before the continuation restart, + # and build_api_kwargs consumes it on that call. If the turn is + # interrupted/errors between arm and consume, it must not fire on the + # next turn's first request. + agent._ephemeral_reasoning_off = False # Total outer-loop exceptions this turn (#92450) — see _MAX_OUTER_LOOP_ERRORS. _outer_error_count = 0 truncated_tool_call_retries = 0 @@ -4163,7 +4169,7 @@ def run_conversation( "The model used all its output tokens on reasoning " "and had none left for the actual response.\n\n" "To fix this:\n" - "→ Lower reasoning effort: `/thinkon low` or `/thinkon minimal`\n" + "→ Lower reasoning effort: `/reasoning low` or `/reasoning minimal`\n" "→ Or switch to a larger/non-reasoning model with `/model`" ) agent._cleanup_task_resources(effective_task_id) @@ -4406,8 +4412,8 @@ def run_conversation( "continuation attempt — its reasoning " "consumed the entire budget each time.\n\n" "To fix this:\n" - "→ Lower reasoning effort: `/thinkon low` " - "or `/thinkoff`\n" + "→ Lower reasoning effort: `/reasoning low` " + "or `/reasoning none`\n" "→ Or raise max_tokens for this model" ) # Unanswered continue nudges made every later turn re-truncate. diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index d2284b64d2..25b41be44d 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -767,9 +767,18 @@ class ChatCompletionsTransport(ProviderTransport): extra_body["reasoning"] = gh_reasoning else: _effort = "medium" + _enabled = True if reasoning_config and isinstance(reasoning_config, dict): _effort = reasoning_config.get("effort", "medium") or "medium" - extra_body["reasoning"] = {"enabled": True, "effort": _effort} + # Honor an explicit "thinking off" (agent.reasoning_effort: + # none / the one-shot length-continuation override) the same + # way the provider-profile path does — never re-enable it. + if reasoning_config.get("enabled") is False or _effort == "none": + _enabled = False + if _enabled: + extra_body["reasoning"] = {"enabled": True, "effort": _effort} + else: + extra_body["reasoning"] = {"enabled": False, "effort": "none"} if provider_name == "gemini": raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config) diff --git a/tests/run_agent/test_length_continuation_thinking_exhaustion.py b/tests/run_agent/test_length_continuation_thinking_exhaustion.py index eea5e8f7ff..5c8a0a25c4 100644 --- a/tests/run_agent/test_length_continuation_thinking_exhaustion.py +++ b/tests/run_agent/test_length_continuation_thinking_exhaustion.py @@ -225,4 +225,42 @@ class TestThinkingOnlyTruncation: assert result["completed"] is True assert "visible part one." in (result["final_response"] or "") assert "and the ending." in (result["final_response"] or "") - assert _no_empty_assistant_rows(result["messages"]) == [] \ No newline at end of file + assert _no_empty_assistant_rows(result["messages"]) == [] + +class TestReasoningOffReachesTheWire: + def test_continuation_request_carries_reasoning_off_on_the_wire(self, loop_agent): + """The flag is only useful if the continuation REQUEST goes out with + thinking disabled — assert the OpenRouter extra_body, not the flag.""" + loop_agent.reasoning_config = {"enabled": True, "effort": "high"} + loop_agent._supports_reasoning_extra_body = lambda: True + loop_agent.client.chat.completions.create.side_effect = [ + _thinking_only_length_response(), + _full_response("Here is the full answer."), + ] + result = _run(loop_agent, "write me a long report") + assert result["completed"] is True + + calls = loop_agent.client.chat.completions.create.call_args_list + assert len(calls) == 2 + first = (calls[0].kwargs.get("extra_body") or {}).get("reasoning") + second = (calls[1].kwargs.get("extra_body") or {}).get("reasoning") + assert first == {"enabled": True, "effort": "high"}, first + assert second is not None and second.get("enabled") is False, ( + f"continuation must be sent with thinking off, got {second!r}" + ) + + def test_stale_flag_does_not_leak_into_next_turn(self, loop_agent): + """A flag armed by a previous turn that never reached build_api_kwargs + (interrupt/error between arm and consume) must not silently strip + thinking from the next turn's first request.""" + loop_agent.reasoning_config = {"enabled": True, "effort": "high"} + loop_agent._supports_reasoning_extra_body = lambda: True + loop_agent._ephemeral_reasoning_off = True # stale from a prior turn + loop_agent.client.chat.completions.create.side_effect = [ + _full_response("fresh turn answer."), + ] + result = _run(loop_agent, "hello") + assert result["completed"] is True + calls = loop_agent.client.chat.completions.create.call_args_list + first = (calls[0].kwargs.get("extra_body") or {}).get("reasoning") + assert first == {"enabled": True, "effort": "high"}, first diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 069bd55d69..05af3b8d1d 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -4421,7 +4421,7 @@ class TestRunConversation: # Should have a user-friendly response (not None) assert result["final_response"] is not None assert "Thinking Budget Exhausted" in result["final_response"] - assert "/thinkon" in result["final_response"] + assert "/reasoning" in result["final_response"] def test_length_with_tool_calls_returns_partial_without_executing_tools(self, agent):