fix(agent): thinking-only length truncations no longer wedge continuations

GLM-5.3-flash on ollama-cloud with reasoning_effort=high can spend the ENTIRE
output cap on reasoning delivered in a separate field and return
finish_reason=length with no visible content (verified live: max_tokens=4096,
completion_tokens=4096, content empty).

The length-continuation path handled that shape badly:
  1. the empty response was appended as an interim assistant fragment,
     poisoning the transcript until the pre-call sanitizer healed it
     (observed 3+ healings per turn on the reporting user's session);
  2. every continuation re-ran with thinking ON, re-deriving the whole
     thinking budget against a growing context, so 4 attempts still produced
     nothing and the turn died with 'Response remains truncated after 4
     continuation attempts'.

Now:
  - interim assistant fragments with no visible content are never appended
    (whichever way they got empty);
  - a thinking-only truncation sets a one-shot reasoning-off override that
    build_api_kwargs consumes for the next request, so the continuation
    writes the answer instead of re-thinking it;
  - the ceiling exit clears a pending override and, when every fragment was
    empty, returns an actionable final_response instead of an invisible None.
This commit is contained in:
AlexGabbia
2026-08-31 18:54:03 +02:00
committed by Teknium
parent df4b3733ba
commit fb76fb0526
3 changed files with 328 additions and 18 deletions
+39 -4
View File
@@ -1932,8 +1932,43 @@ def interruptible_api_call(agent, api_kwargs: dict):
def _consume_ephemeral_reasoning_off(agent) -> bool:
"""Consume the one-shot "answer without thinking" continuation flag.
Set by the length-continuation path when a request returned reasoning
but NO visible content — the thinking phase consumed the entire output
cap (GLM-5.3 on ollama-cloud with reasoning_effort=high: reported live as
finish_reason="length", content="", completion_tokens == max_tokens).
Continuation turns never replay the prior reasoning, so re-running with
thinking ON re-derives — and re-burns — the whole thinking budget from
scratch instead of writing the answer (observed: 4 futile continuations
then "Response remained truncated after 4 continuation attempts").
When True is returned the caller must override the wire reasoning_config
with ``{"enabled": False, "effort": "none"}`` for exactly the next call.
"""
if getattr(agent, "_ephemeral_reasoning_off", False):
agent._ephemeral_reasoning_off = False
return True
return False
def _reasoning_config_for_wire(agent):
"""``agent.reasoning_config`` with the one-shot reasoning-off override applied."""
if _consume_ephemeral_reasoning_off(agent):
return {
**(agent.reasoning_config or {}),
"enabled": False,
"effort": "none",
}
return agent.reasoning_config
def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = None) -> dict:
"""Build the keyword arguments dict for the active API mode."""
# One-shot continuation override — consumed exactly once, on the FIRST
# request this call builds (only one api_mode branch runs per invocation).
_wire_reasoning_config = _reasoning_config_for_wire(agent)
if tools_for_api is None:
tools_for_api = agent.tools
@@ -1950,7 +1985,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
messages=anthropic_messages,
tools=tools_for_api,
max_tokens=ephemeral_out if ephemeral_out is not None else agent.max_tokens,
reasoning_config=agent.reasoning_config,
reasoning_config=_wire_reasoning_config,
is_oauth=agent._is_anthropic_oauth,
preserve_dots=agent._anthropic_preserve_dots(),
context_length=ctx_len,
@@ -2042,7 +2077,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
model=agent.model,
messages=_msgs_for_codex,
tools=tools_for_api,
reasoning_config=agent.reasoning_config,
reasoning_config=_wire_reasoning_config,
session_id=getattr(agent, "session_id", None),
cache_scope_id=_cache_scope_id,
base_url=agent.base_url,
@@ -2199,7 +2234,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
max_tokens=agent.max_tokens,
ephemeral_max_output_tokens=_ephemeral_out,
max_tokens_param_fn=agent._max_tokens_param,
reasoning_config=agent.reasoning_config,
reasoning_config=_wire_reasoning_config,
request_overrides=agent.request_overrides,
session_id=getattr(agent, "session_id", None),
cache_scope_id=_cache_scope_id,
@@ -2232,7 +2267,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
max_tokens=agent.max_tokens,
ephemeral_max_output_tokens=_ephemeral_out,
max_tokens_param_fn=agent._max_tokens_param,
reasoning_config=agent.reasoning_config,
reasoning_config=_wire_reasoning_config,
request_overrides=agent.request_overrides,
session_id=getattr(agent, "session_id", None),
cache_scope_id=_cache_scope_id,
+61 -14
View File
@@ -4290,29 +4290,46 @@ def run_conversation(
)
if assistant_message is not None and not _trunc_has_tool_calls:
length_continue_retries += 1
# An EMPTY partial-stream stub (stream dropped
# mid tool-call before any text was delivered)
# must not be appended as an interim assistant
# message: it would serialize as
# {"role": "assistant", "content": ""}, and
# An interim assistant message with NO visible
# content must not be appended — whichever way it
# got that way. An empty partial-stream stub
# (stream dropped before any text was delivered)
# and a response whose whole output budget went to
# reasoning delivered in a separate field (GLM-5.3
# on ollama-cloud with reasoning_effort=high:
# finish_reason="length", content="",
# completion_tokens == max_tokens) both serialize
# as {"role": "assistant", "content": ""}, and
# strict providers (Moonshot/Kimi via OpenRouter)
# reject empty assistant content with HTTP 400
# ("message ... with role 'assistant' must not be
# empty") on the very next replay — permanently
# poisoning the session history. There is no
# partial text to continue from anyway, so only
# the continuation user-message is appended.
# poisoning the session history until the pre-call
# sanitizer "heals" the hole (observed 3+ healings
# per turn). There is no partial text to continue
# from anyway, so only the continuation
# user-message is appended.
_interim_content = getattr(assistant_message, "content", None)
_is_empty_partial_stub = (
getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID
and not getattr(assistant_message, "content", None)
and not _interim_content
)
if not _is_empty_partial_stub:
if not _interim_content and not _is_empty_partial_stub:
# Thinking-only truncation: the model spent the
# entire output cap on reasoning and produced no
# visible text. A continuation with thinking
# ON would re-think the whole context from
# scratch (continuations never replay prior
# reasoning) and re-burn the same budget, so
# the next call drops thinking for one request
# — the answer must be written, not re-derived.
agent._ephemeral_reasoning_off = True
if _interim_content:
interim_msg = agent._build_assistant_message(assistant_message, finish_reason)
# Marked so the ceiling exit can drop the fragment trail.
interim_msg["_length_continuation_fragment"] = True
append_message(messages, interim_msg)
if assistant_message.content:
truncated_response_parts.append(assistant_message.content)
truncated_response_parts.append(_interim_content)
if length_continue_retries < 4:
_is_partial_stream_stub = (
@@ -4356,13 +4373,43 @@ def run_conversation(
break
partial_response = agent._strip_think_blocks(_join_truncated_parts(truncated_response_parts)).strip()
# The pending one-shot reasoning-off override must
# not leak into the next turn when the 4th
# truncation goes straight to the ceiling exit
# without scheduling a continuation call to
# consume it.
agent._ephemeral_reasoning_off = False
if partial_response:
agent._vprint(
f"{agent.log_prefix}⚠️ Response still truncated "
f"after 4 continuation attempts — keeping the "
f"after {length_continue_retries} continuation attempts — keeping the "
f"partial response received so far.",
force=True,
)
_ceiling_final = partial_response
else:
# Every fragment was empty — e.g. a thinking
# model that spent each attempt's whole cap on
# reasoning (GLM-5.3 on ollama-cloud). Return
# an actionable message instead of an invisible
# None result, which only surfaces as a bare
# error card.
agent._vprint(
f"{agent.log_prefix}⚠️ Response still truncated "
f"after {length_continue_retries} continuation attempts — no visible "
f"text was produced.",
force=True,
)
_ceiling_final = (
"⚠️ **No visible answer was produced.** The "
"model hit its output-token limit on every "
"continuation attempt — its reasoning "
"consumed the entire budget each time.\n\n"
"To fix this:\n"
"→ Lower reasoning effort: `/thinkon low` "
"or `/thinkoff`\n"
"→ Or raise max_tokens for this model"
)
# Unanswered continue nudges made every later turn re-truncate.
_turn_start = (
current_turn_user_idx + 1
@@ -4390,7 +4437,7 @@ def run_conversation(
agent._cleanup_task_resources(effective_task_id)
agent._persist_session(messages, conversation_history)
return {
"final_response": partial_response or None,
"final_response": _ceiling_final,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
@@ -0,0 +1,228 @@
"""Regression tests for thinking-only length truncations.
GLM-5.3-flash on ollama-cloud with reasoning_effort=high can burn the ENTIRE
output cap on reasoning delivered in a separate field and return
finish_reason="length" with NO visible content (verified live: max_tokens=4096
→ completion_tokens=4096, reasoning ~18.5KB, content empty).
The old continuation flow handled this badly:
1. the empty response was appended as an interim assistant fragment,
poisoning the transcript until the pre-call sanitizer "healed" it
(observed 3+ healings per turn);
2. every continuation re-ran with thinking ON, re-deriving — and re-burning
— the whole thinking budget against a growing context, so 4 attempts
still produced nothing and the turn died with
"Response remained truncated after 4 continuation attempts".
The fix: skip empty interim fragments, and issue the continuation with a
one-shot reasoning-off override so the budget goes to writing the answer.
"""
from __future__ import annotations
from types import SimpleNamespace
from unittest.mock import MagicMock, patch
import pytest
from hermes_constants import FINISH_REASON_LENGTH
class _AgentStandIn:
"""Minimal agent surface _reasoning_config_for_wire needs."""
def __init__(self, reasoning_config):
self.reasoning_config = reasoning_config
class TestReasoningOffOneShotOverride:
def test_flag_consumed_exactly_once(self):
from agent.chat_completion_helpers import _reasoning_config_for_wire
agent = _AgentStandIn({"enabled": True, "effort": "high"})
# Without the flag the reasoning config passes through untouched.
assert _reasoning_config_for_wire(agent) == {
"enabled": True,
"effort": "high",
}
agent._ephemeral_reasoning_off = True
cfg = _reasoning_config_for_wire(agent)
assert cfg["enabled"] is False
assert cfg["effort"] == "none"
assert agent._ephemeral_reasoning_off is False, (
"The one-shot override must be consumed by the first call."
)
# Subsequent calls keep the user's own reasoning config.
assert _reasoning_config_for_wire(agent) == {
"enabled": True,
"effort": "high",
}
def test_flag_with_no_user_reasoning_config(self):
from agent.chat_completion_helpers import _reasoning_config_for_wire
agent = _AgentStandIn(None)
agent._ephemeral_reasoning_off = True
cfg = _reasoning_config_for_wire(agent)
assert cfg == {"enabled": False, "effort": "none"}
@pytest.fixture()
def loop_agent():
from run_agent import AIAgent
with (
patch("run_agent.get_tool_definitions", return_value=[]),
patch("run_agent.check_toolset_requirements", return_value={}),
patch("run_agent.OpenAI"),
):
a = AIAgent(
api_key="test-key-1234567890",
base_url="https://openrouter.ai/api/v1",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
a.client = MagicMock()
a._cached_system_prompt = "You are helpful."
a._use_prompt_caching = False
a.compression_enabled = False
a.save_trajectories = False
return a
def _thinking_only_length_response():
"""finish_reason='length' with reasoning but zero visible content — the
live GLM-5.3-flash-on-ollama-cloud shape (normal response id, NOT the
partial-stream stub)."""
from tests.run_agent.test_run_agent import _mock_assistant_msg
return SimpleNamespace(
id="chatcmpl-thinking-exhausted",
model="test/model",
choices=[SimpleNamespace(
index=0,
message=_mock_assistant_msg(content=""),
finish_reason=FINISH_REASON_LENGTH,
)],
usage=None,
)
def _full_response(content):
from tests.run_agent.test_run_agent import _mock_response
return _mock_response(content=content, finish_reason="stop")
def _truncated_text_response(content):
from tests.run_agent.test_run_agent import _mock_response
return _mock_response(content=content, finish_reason=FINISH_REASON_LENGTH)
def _run(agent, message, history=None):
with (
patch.object(agent, "_persist_session"),
patch.object(agent, "_save_trajectory"),
patch.object(agent, "_cleanup_task_resources"),
):
return agent.run_conversation(message, conversation_history=history)
def _no_empty_assistant_rows(messages):
return [
m for m in messages
if m.get("role") == "assistant"
and not (m.get("content") or "").strip()
and not m.get("tool_calls")
]
class TestThinkingOnlyTruncation:
def test_retry_after_thinking_only_truncation_completes(self, loop_agent):
"""One thinking-only truncation, then a normal answer: the retry must
drop thinking (one-shot), boost the output cap, and finish the turn."""
loop_agent.client.chat.completions.create.side_effect = [
_thinking_only_length_response(),
_full_response("Here is the full answer."),
]
result = _run(loop_agent, "write me a long report")
assert result["completed"] is True
assert "full answer" in (result["final_response"] or "")
assert _no_empty_assistant_rows(result["messages"]) == [], (
"An empty (thinking-only) truncated response must never be "
"appended to the transcript."
)
calls = loop_agent.client.chat.completions.create.call_args_list
assert len(calls) == 2
# Continuation retry boosts the output cap (2^1 × 4096 base floor).
assert calls[1].kwargs.get("max_tokens") == 8192, (
"The continuation retry must request a larger output budget than "
"the request that truncated."
)
assert loop_agent._ephemeral_reasoning_off is False, (
"The one-shot reasoning-off override must be consumed by the "
"continuation call."
)
def test_thinking_only_truncation_sets_reasoning_off(self, loop_agent):
from tests.run_agent.test_run_agent import _mock_response
loop_agent.client.chat.completions.create.side_effect = [
_thinking_only_length_response(),
_mock_response(
content="done", finish_reason=FINISH_REASON_LENGTH
),
_full_response("finally complete."),
]
_run(loop_agent, "write me a long report")
calls = loop_agent.client.chat.completions.create.call_args_list
assert len(calls) == 3
# The thinking-only fragment set the flag; it was consumed by the
# next call, and the SECOND truncated fragment (which had visible
# text) does not set it again — so the third call sees thinking ON.
assert loop_agent._ephemeral_reasoning_off is False
def test_full_ceiling_with_empty_fragments_still_settles(self, loop_agent):
"""All four attempts thinking-only: the turn must exit through the
ceiling with an actionable final_response, no poisoned transcript,
and no leaked reasoning-off flag."""
loop_agent.client.chat.completions.create.side_effect = [
_thinking_only_length_response() for _ in range(4)
]
result = _run(loop_agent, "write me a long report")
assert result["completed"] is False
assert result["partial"] is True
assert "truncated after 4 continuation attempts" in (result.get("error") or "")
assert result["final_response"], (
"An all-empty ceiling exit must still surface a user-facing "
"message instead of an invisible None."
)
assert "reasoning" in (result["final_response"] or "").lower()
assert _no_empty_assistant_rows(result["messages"]) == []
assert loop_agent._ephemeral_reasoning_off is False, (
"The ceiling exit must clear the pending one-shot override so the "
"next turn does not silently lose thinking."
)
def test_mixed_fragments_keep_visible_text(self, loop_agent):
"""A visible fragment followed by a thinking-only one: the visible
text must be stitched, the empty one skipped."""
loop_agent.client.chat.completions.create.side_effect = [
_truncated_text_response("visible part one. "),
_thinking_only_length_response(),
_full_response("and the ending."),
]
result = _run(loop_agent, "write me a long report")
assert result["completed"] is True
assert "visible part one." in (result["final_response"] or "")
assert "and the ending." in (result["final_response"] or "")
assert _no_empty_assistant_rows(result["messages"]) == []