From fb76fb05265dda87baa6519bb78e6d65ca5009f6 Mon Sep 17 00:00:00 2001 From: AlexGabbia Date: Mon, 31 Aug 2026 18:54:03 +0200 Subject: [PATCH] fix(agent): thinking-only length truncations no longer wedge continuations GLM-5.3-flash on ollama-cloud with reasoning_effort=high can spend the ENTIRE output cap on reasoning delivered in a separate field and return finish_reason=length with no visible content (verified live: max_tokens=4096, completion_tokens=4096, content empty). The length-continuation path handled that shape badly: 1. the empty response was appended as an interim assistant fragment, poisoning the transcript until the pre-call sanitizer healed it (observed 3+ healings per turn on the reporting user's session); 2. every continuation re-ran with thinking ON, re-deriving the whole thinking budget against a growing context, so 4 attempts still produced nothing and the turn died with 'Response remains truncated after 4 continuation attempts'. Now: - interim assistant fragments with no visible content are never appended (whichever way they got empty); - a thinking-only truncation sets a one-shot reasoning-off override that build_api_kwargs consumes for the next request, so the continuation writes the answer instead of re-thinking it; - the ceiling exit clears a pending override and, when every fragment was empty, returns an actionable final_response instead of an invisible None. --- agent/chat_completion_helpers.py | 43 +++- agent/conversation_loop.py | 75 ++++-- ...length_continuation_thinking_exhaustion.py | 228 ++++++++++++++++++ 3 files changed, 328 insertions(+), 18 deletions(-) create mode 100644 tests/run_agent/test_length_continuation_thinking_exhaustion.py diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 501284461f..79d4b925c3 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -1932,8 +1932,43 @@ def interruptible_api_call(agent, api_kwargs: dict): +def _consume_ephemeral_reasoning_off(agent) -> bool: + """Consume the one-shot "answer without thinking" continuation flag. + + Set by the length-continuation path when a request returned reasoning + but NO visible content — the thinking phase consumed the entire output + cap (GLM-5.3 on ollama-cloud with reasoning_effort=high: reported live as + finish_reason="length", content="", completion_tokens == max_tokens). + + Continuation turns never replay the prior reasoning, so re-running with + thinking ON re-derives — and re-burns — the whole thinking budget from + scratch instead of writing the answer (observed: 4 futile continuations + then "Response remained truncated after 4 continuation attempts"). + When True is returned the caller must override the wire reasoning_config + with ``{"enabled": False, "effort": "none"}`` for exactly the next call. + """ + if getattr(agent, "_ephemeral_reasoning_off", False): + agent._ephemeral_reasoning_off = False + return True + return False + + +def _reasoning_config_for_wire(agent): + """``agent.reasoning_config`` with the one-shot reasoning-off override applied.""" + if _consume_ephemeral_reasoning_off(agent): + return { + **(agent.reasoning_config or {}), + "enabled": False, + "effort": "none", + } + return agent.reasoning_config + + def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = None) -> dict: """Build the keyword arguments dict for the active API mode.""" + # One-shot continuation override — consumed exactly once, on the FIRST + # request this call builds (only one api_mode branch runs per invocation). + _wire_reasoning_config = _reasoning_config_for_wire(agent) if tools_for_api is None: tools_for_api = agent.tools @@ -1950,7 +1985,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non messages=anthropic_messages, tools=tools_for_api, max_tokens=ephemeral_out if ephemeral_out is not None else agent.max_tokens, - reasoning_config=agent.reasoning_config, + reasoning_config=_wire_reasoning_config, is_oauth=agent._is_anthropic_oauth, preserve_dots=agent._anthropic_preserve_dots(), context_length=ctx_len, @@ -2042,7 +2077,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non model=agent.model, messages=_msgs_for_codex, tools=tools_for_api, - reasoning_config=agent.reasoning_config, + reasoning_config=_wire_reasoning_config, session_id=getattr(agent, "session_id", None), cache_scope_id=_cache_scope_id, base_url=agent.base_url, @@ -2199,7 +2234,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non max_tokens=agent.max_tokens, ephemeral_max_output_tokens=_ephemeral_out, max_tokens_param_fn=agent._max_tokens_param, - reasoning_config=agent.reasoning_config, + reasoning_config=_wire_reasoning_config, request_overrides=agent.request_overrides, session_id=getattr(agent, "session_id", None), cache_scope_id=_cache_scope_id, @@ -2232,7 +2267,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non max_tokens=agent.max_tokens, ephemeral_max_output_tokens=_ephemeral_out, max_tokens_param_fn=agent._max_tokens_param, - reasoning_config=agent.reasoning_config, + reasoning_config=_wire_reasoning_config, request_overrides=agent.request_overrides, session_id=getattr(agent, "session_id", None), cache_scope_id=_cache_scope_id, diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 9d2bff7385..e413b83ff9 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -4290,29 +4290,46 @@ def run_conversation( ) if assistant_message is not None and not _trunc_has_tool_calls: length_continue_retries += 1 - # An EMPTY partial-stream stub (stream dropped - # mid tool-call before any text was delivered) - # must not be appended as an interim assistant - # message: it would serialize as - # {"role": "assistant", "content": ""}, and + # An interim assistant message with NO visible + # content must not be appended — whichever way it + # got that way. An empty partial-stream stub + # (stream dropped before any text was delivered) + # and a response whose whole output budget went to + # reasoning delivered in a separate field (GLM-5.3 + # on ollama-cloud with reasoning_effort=high: + # finish_reason="length", content="", + # completion_tokens == max_tokens) both serialize + # as {"role": "assistant", "content": ""}, and # strict providers (Moonshot/Kimi via OpenRouter) # reject empty assistant content with HTTP 400 # ("message ... with role 'assistant' must not be # empty") on the very next replay — permanently - # poisoning the session history. There is no - # partial text to continue from anyway, so only - # the continuation user-message is appended. + # poisoning the session history until the pre-call + # sanitizer "heals" the hole (observed 3+ healings + # per turn). There is no partial text to continue + # from anyway, so only the continuation + # user-message is appended. + _interim_content = getattr(assistant_message, "content", None) _is_empty_partial_stub = ( getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID - and not getattr(assistant_message, "content", None) + and not _interim_content ) - if not _is_empty_partial_stub: + if not _interim_content and not _is_empty_partial_stub: + # Thinking-only truncation: the model spent the + # entire output cap on reasoning and produced no + # visible text. A continuation with thinking + # ON would re-think the whole context from + # scratch (continuations never replay prior + # reasoning) and re-burn the same budget, so + # the next call drops thinking for one request + # — the answer must be written, not re-derived. + agent._ephemeral_reasoning_off = True + if _interim_content: interim_msg = agent._build_assistant_message(assistant_message, finish_reason) # Marked so the ceiling exit can drop the fragment trail. interim_msg["_length_continuation_fragment"] = True append_message(messages, interim_msg) - if assistant_message.content: - truncated_response_parts.append(assistant_message.content) + truncated_response_parts.append(_interim_content) if length_continue_retries < 4: _is_partial_stream_stub = ( @@ -4356,13 +4373,43 @@ def run_conversation( break partial_response = agent._strip_think_blocks(_join_truncated_parts(truncated_response_parts)).strip() + # The pending one-shot reasoning-off override must + # not leak into the next turn when the 4th + # truncation goes straight to the ceiling exit + # without scheduling a continuation call to + # consume it. + agent._ephemeral_reasoning_off = False if partial_response: agent._vprint( f"{agent.log_prefix}⚠️ Response still truncated " - f"after 4 continuation attempts — keeping the " + f"after {length_continue_retries} continuation attempts — keeping the " f"partial response received so far.", force=True, ) + _ceiling_final = partial_response + else: + # Every fragment was empty — e.g. a thinking + # model that spent each attempt's whole cap on + # reasoning (GLM-5.3 on ollama-cloud). Return + # an actionable message instead of an invisible + # None result, which only surfaces as a bare + # error card. + agent._vprint( + f"{agent.log_prefix}⚠️ Response still truncated " + f"after {length_continue_retries} continuation attempts — no visible " + f"text was produced.", + force=True, + ) + _ceiling_final = ( + "⚠️ **No visible answer was produced.** The " + "model hit its output-token limit on every " + "continuation attempt — its reasoning " + "consumed the entire budget each time.\n\n" + "To fix this:\n" + "→ Lower reasoning effort: `/thinkon low` " + "or `/thinkoff`\n" + "→ Or raise max_tokens for this model" + ) # Unanswered continue nudges made every later turn re-truncate. _turn_start = ( current_turn_user_idx + 1 @@ -4390,7 +4437,7 @@ def run_conversation( agent._cleanup_task_resources(effective_task_id) agent._persist_session(messages, conversation_history) return { - "final_response": partial_response or None, + "final_response": _ceiling_final, "messages": messages, "api_calls": api_call_count, "completed": False, diff --git a/tests/run_agent/test_length_continuation_thinking_exhaustion.py b/tests/run_agent/test_length_continuation_thinking_exhaustion.py new file mode 100644 index 0000000000..eea5e8f7ff --- /dev/null +++ b/tests/run_agent/test_length_continuation_thinking_exhaustion.py @@ -0,0 +1,228 @@ +"""Regression tests for thinking-only length truncations. + +GLM-5.3-flash on ollama-cloud with reasoning_effort=high can burn the ENTIRE +output cap on reasoning delivered in a separate field and return +finish_reason="length" with NO visible content (verified live: max_tokens=4096 +→ completion_tokens=4096, reasoning ~18.5KB, content empty). + +The old continuation flow handled this badly: + 1. the empty response was appended as an interim assistant fragment, + poisoning the transcript until the pre-call sanitizer "healed" it + (observed 3+ healings per turn); + 2. every continuation re-ran with thinking ON, re-deriving — and re-burning + — the whole thinking budget against a growing context, so 4 attempts + still produced nothing and the turn died with + "Response remained truncated after 4 continuation attempts". + +The fix: skip empty interim fragments, and issue the continuation with a +one-shot reasoning-off override so the budget goes to writing the answer. +""" + +from __future__ import annotations + +from types import SimpleNamespace +from unittest.mock import MagicMock, patch + +import pytest + +from hermes_constants import FINISH_REASON_LENGTH + + +class _AgentStandIn: + """Minimal agent surface _reasoning_config_for_wire needs.""" + + def __init__(self, reasoning_config): + self.reasoning_config = reasoning_config + + +class TestReasoningOffOneShotOverride: + def test_flag_consumed_exactly_once(self): + from agent.chat_completion_helpers import _reasoning_config_for_wire + + agent = _AgentStandIn({"enabled": True, "effort": "high"}) + # Without the flag the reasoning config passes through untouched. + assert _reasoning_config_for_wire(agent) == { + "enabled": True, + "effort": "high", + } + + agent._ephemeral_reasoning_off = True + cfg = _reasoning_config_for_wire(agent) + assert cfg["enabled"] is False + assert cfg["effort"] == "none" + assert agent._ephemeral_reasoning_off is False, ( + "The one-shot override must be consumed by the first call." + ) + + # Subsequent calls keep the user's own reasoning config. + assert _reasoning_config_for_wire(agent) == { + "enabled": True, + "effort": "high", + } + + def test_flag_with_no_user_reasoning_config(self): + from agent.chat_completion_helpers import _reasoning_config_for_wire + + agent = _AgentStandIn(None) + agent._ephemeral_reasoning_off = True + cfg = _reasoning_config_for_wire(agent) + assert cfg == {"enabled": False, "effort": "none"} + + +@pytest.fixture() +def loop_agent(): + from run_agent import AIAgent + + with ( + patch("run_agent.get_tool_definitions", return_value=[]), + patch("run_agent.check_toolset_requirements", return_value={}), + patch("run_agent.OpenAI"), + ): + a = AIAgent( + api_key="test-key-1234567890", + base_url="https://openrouter.ai/api/v1", + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + ) + a.client = MagicMock() + a._cached_system_prompt = "You are helpful." + a._use_prompt_caching = False + a.compression_enabled = False + a.save_trajectories = False + return a + + +def _thinking_only_length_response(): + """finish_reason='length' with reasoning but zero visible content — the + live GLM-5.3-flash-on-ollama-cloud shape (normal response id, NOT the + partial-stream stub).""" + from tests.run_agent.test_run_agent import _mock_assistant_msg + + return SimpleNamespace( + id="chatcmpl-thinking-exhausted", + model="test/model", + choices=[SimpleNamespace( + index=0, + message=_mock_assistant_msg(content=""), + finish_reason=FINISH_REASON_LENGTH, + )], + usage=None, + ) + + +def _full_response(content): + from tests.run_agent.test_run_agent import _mock_response + + return _mock_response(content=content, finish_reason="stop") + + +def _truncated_text_response(content): + from tests.run_agent.test_run_agent import _mock_response + + return _mock_response(content=content, finish_reason=FINISH_REASON_LENGTH) + + +def _run(agent, message, history=None): + with ( + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + ): + return agent.run_conversation(message, conversation_history=history) + + +def _no_empty_assistant_rows(messages): + return [ + m for m in messages + if m.get("role") == "assistant" + and not (m.get("content") or "").strip() + and not m.get("tool_calls") + ] + + +class TestThinkingOnlyTruncation: + def test_retry_after_thinking_only_truncation_completes(self, loop_agent): + """One thinking-only truncation, then a normal answer: the retry must + drop thinking (one-shot), boost the output cap, and finish the turn.""" + loop_agent.client.chat.completions.create.side_effect = [ + _thinking_only_length_response(), + _full_response("Here is the full answer."), + ] + result = _run(loop_agent, "write me a long report") + + assert result["completed"] is True + assert "full answer" in (result["final_response"] or "") + assert _no_empty_assistant_rows(result["messages"]) == [], ( + "An empty (thinking-only) truncated response must never be " + "appended to the transcript." + ) + + calls = loop_agent.client.chat.completions.create.call_args_list + assert len(calls) == 2 + # Continuation retry boosts the output cap (2^1 × 4096 base floor). + assert calls[1].kwargs.get("max_tokens") == 8192, ( + "The continuation retry must request a larger output budget than " + "the request that truncated." + ) + assert loop_agent._ephemeral_reasoning_off is False, ( + "The one-shot reasoning-off override must be consumed by the " + "continuation call." + ) + + def test_thinking_only_truncation_sets_reasoning_off(self, loop_agent): + from tests.run_agent.test_run_agent import _mock_response + + loop_agent.client.chat.completions.create.side_effect = [ + _thinking_only_length_response(), + _mock_response( + content="done", finish_reason=FINISH_REASON_LENGTH + ), + _full_response("finally complete."), + ] + _run(loop_agent, "write me a long report") + + calls = loop_agent.client.chat.completions.create.call_args_list + assert len(calls) == 3 + # The thinking-only fragment set the flag; it was consumed by the + # next call, and the SECOND truncated fragment (which had visible + # text) does not set it again — so the third call sees thinking ON. + assert loop_agent._ephemeral_reasoning_off is False + + def test_full_ceiling_with_empty_fragments_still_settles(self, loop_agent): + """All four attempts thinking-only: the turn must exit through the + ceiling with an actionable final_response, no poisoned transcript, + and no leaked reasoning-off flag.""" + loop_agent.client.chat.completions.create.side_effect = [ + _thinking_only_length_response() for _ in range(4) + ] + result = _run(loop_agent, "write me a long report") + + assert result["completed"] is False + assert result["partial"] is True + assert "truncated after 4 continuation attempts" in (result.get("error") or "") + assert result["final_response"], ( + "An all-empty ceiling exit must still surface a user-facing " + "message instead of an invisible None." + ) + assert "reasoning" in (result["final_response"] or "").lower() + assert _no_empty_assistant_rows(result["messages"]) == [] + assert loop_agent._ephemeral_reasoning_off is False, ( + "The ceiling exit must clear the pending one-shot override so the " + "next turn does not silently lose thinking." + ) + + def test_mixed_fragments_keep_visible_text(self, loop_agent): + """A visible fragment followed by a thinking-only one: the visible + text must be stitched, the empty one skipped.""" + loop_agent.client.chat.completions.create.side_effect = [ + _truncated_text_response("visible part one. "), + _thinking_only_length_response(), + _full_response("and the ending."), + ] + result = _run(loop_agent, "write me a long report") + + assert result["completed"] is True + assert "visible part one." in (result["final_response"] or "") + assert "and the ending." in (result["final_response"] or "") + assert _no_empty_assistant_rows(result["messages"]) == [] \ No newline at end of file