fix(agent): thinking-only length truncations no longer wedge continuations
GLM-5.3-flash on ollama-cloud with reasoning_effort=high can spend the ENTIRE
output cap on reasoning delivered in a separate field and return
finish_reason=length with no visible content (verified live: max_tokens=4096,
completion_tokens=4096, content empty).
The length-continuation path handled that shape badly:
1. the empty response was appended as an interim assistant fragment,
poisoning the transcript until the pre-call sanitizer healed it
(observed 3+ healings per turn on the reporting user's session);
2. every continuation re-ran with thinking ON, re-deriving the whole
thinking budget against a growing context, so 4 attempts still produced
nothing and the turn died with 'Response remains truncated after 4
continuation attempts'.
Now:
- interim assistant fragments with no visible content are never appended
(whichever way they got empty);
- a thinking-only truncation sets a one-shot reasoning-off override that
build_api_kwargs consumes for the next request, so the continuation
writes the answer instead of re-thinking it;
- the ceiling exit clears a pending override and, when every fragment was
empty, returns an actionable final_response instead of an invisible None.
This commit is contained in:
@@ -1932,8 +1932,43 @@ def interruptible_api_call(agent, api_kwargs: dict):
|
||||
|
||||
|
||||
|
||||
def _consume_ephemeral_reasoning_off(agent) -> bool:
|
||||
"""Consume the one-shot "answer without thinking" continuation flag.
|
||||
|
||||
Set by the length-continuation path when a request returned reasoning
|
||||
but NO visible content — the thinking phase consumed the entire output
|
||||
cap (GLM-5.3 on ollama-cloud with reasoning_effort=high: reported live as
|
||||
finish_reason="length", content="", completion_tokens == max_tokens).
|
||||
|
||||
Continuation turns never replay the prior reasoning, so re-running with
|
||||
thinking ON re-derives — and re-burns — the whole thinking budget from
|
||||
scratch instead of writing the answer (observed: 4 futile continuations
|
||||
then "Response remained truncated after 4 continuation attempts").
|
||||
When True is returned the caller must override the wire reasoning_config
|
||||
with ``{"enabled": False, "effort": "none"}`` for exactly the next call.
|
||||
"""
|
||||
if getattr(agent, "_ephemeral_reasoning_off", False):
|
||||
agent._ephemeral_reasoning_off = False
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _reasoning_config_for_wire(agent):
|
||||
"""``agent.reasoning_config`` with the one-shot reasoning-off override applied."""
|
||||
if _consume_ephemeral_reasoning_off(agent):
|
||||
return {
|
||||
**(agent.reasoning_config or {}),
|
||||
"enabled": False,
|
||||
"effort": "none",
|
||||
}
|
||||
return agent.reasoning_config
|
||||
|
||||
|
||||
def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = None) -> dict:
|
||||
"""Build the keyword arguments dict for the active API mode."""
|
||||
# One-shot continuation override — consumed exactly once, on the FIRST
|
||||
# request this call builds (only one api_mode branch runs per invocation).
|
||||
_wire_reasoning_config = _reasoning_config_for_wire(agent)
|
||||
if tools_for_api is None:
|
||||
tools_for_api = agent.tools
|
||||
|
||||
@@ -1950,7 +1985,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
|
||||
messages=anthropic_messages,
|
||||
tools=tools_for_api,
|
||||
max_tokens=ephemeral_out if ephemeral_out is not None else agent.max_tokens,
|
||||
reasoning_config=agent.reasoning_config,
|
||||
reasoning_config=_wire_reasoning_config,
|
||||
is_oauth=agent._is_anthropic_oauth,
|
||||
preserve_dots=agent._anthropic_preserve_dots(),
|
||||
context_length=ctx_len,
|
||||
@@ -2042,7 +2077,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
|
||||
model=agent.model,
|
||||
messages=_msgs_for_codex,
|
||||
tools=tools_for_api,
|
||||
reasoning_config=agent.reasoning_config,
|
||||
reasoning_config=_wire_reasoning_config,
|
||||
session_id=getattr(agent, "session_id", None),
|
||||
cache_scope_id=_cache_scope_id,
|
||||
base_url=agent.base_url,
|
||||
@@ -2199,7 +2234,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
|
||||
max_tokens=agent.max_tokens,
|
||||
ephemeral_max_output_tokens=_ephemeral_out,
|
||||
max_tokens_param_fn=agent._max_tokens_param,
|
||||
reasoning_config=agent.reasoning_config,
|
||||
reasoning_config=_wire_reasoning_config,
|
||||
request_overrides=agent.request_overrides,
|
||||
session_id=getattr(agent, "session_id", None),
|
||||
cache_scope_id=_cache_scope_id,
|
||||
@@ -2232,7 +2267,7 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
|
||||
max_tokens=agent.max_tokens,
|
||||
ephemeral_max_output_tokens=_ephemeral_out,
|
||||
max_tokens_param_fn=agent._max_tokens_param,
|
||||
reasoning_config=agent.reasoning_config,
|
||||
reasoning_config=_wire_reasoning_config,
|
||||
request_overrides=agent.request_overrides,
|
||||
session_id=getattr(agent, "session_id", None),
|
||||
cache_scope_id=_cache_scope_id,
|
||||
|
||||
+61
-14
@@ -4290,29 +4290,46 @@ def run_conversation(
|
||||
)
|
||||
if assistant_message is not None and not _trunc_has_tool_calls:
|
||||
length_continue_retries += 1
|
||||
# An EMPTY partial-stream stub (stream dropped
|
||||
# mid tool-call before any text was delivered)
|
||||
# must not be appended as an interim assistant
|
||||
# message: it would serialize as
|
||||
# {"role": "assistant", "content": ""}, and
|
||||
# An interim assistant message with NO visible
|
||||
# content must not be appended — whichever way it
|
||||
# got that way. An empty partial-stream stub
|
||||
# (stream dropped before any text was delivered)
|
||||
# and a response whose whole output budget went to
|
||||
# reasoning delivered in a separate field (GLM-5.3
|
||||
# on ollama-cloud with reasoning_effort=high:
|
||||
# finish_reason="length", content="",
|
||||
# completion_tokens == max_tokens) both serialize
|
||||
# as {"role": "assistant", "content": ""}, and
|
||||
# strict providers (Moonshot/Kimi via OpenRouter)
|
||||
# reject empty assistant content with HTTP 400
|
||||
# ("message ... with role 'assistant' must not be
|
||||
# empty") on the very next replay — permanently
|
||||
# poisoning the session history. There is no
|
||||
# partial text to continue from anyway, so only
|
||||
# the continuation user-message is appended.
|
||||
# poisoning the session history until the pre-call
|
||||
# sanitizer "heals" the hole (observed 3+ healings
|
||||
# per turn). There is no partial text to continue
|
||||
# from anyway, so only the continuation
|
||||
# user-message is appended.
|
||||
_interim_content = getattr(assistant_message, "content", None)
|
||||
_is_empty_partial_stub = (
|
||||
getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID
|
||||
and not getattr(assistant_message, "content", None)
|
||||
and not _interim_content
|
||||
)
|
||||
if not _is_empty_partial_stub:
|
||||
if not _interim_content and not _is_empty_partial_stub:
|
||||
# Thinking-only truncation: the model spent the
|
||||
# entire output cap on reasoning and produced no
|
||||
# visible text. A continuation with thinking
|
||||
# ON would re-think the whole context from
|
||||
# scratch (continuations never replay prior
|
||||
# reasoning) and re-burn the same budget, so
|
||||
# the next call drops thinking for one request
|
||||
# — the answer must be written, not re-derived.
|
||||
agent._ephemeral_reasoning_off = True
|
||||
if _interim_content:
|
||||
interim_msg = agent._build_assistant_message(assistant_message, finish_reason)
|
||||
# Marked so the ceiling exit can drop the fragment trail.
|
||||
interim_msg["_length_continuation_fragment"] = True
|
||||
append_message(messages, interim_msg)
|
||||
if assistant_message.content:
|
||||
truncated_response_parts.append(assistant_message.content)
|
||||
truncated_response_parts.append(_interim_content)
|
||||
|
||||
if length_continue_retries < 4:
|
||||
_is_partial_stream_stub = (
|
||||
@@ -4356,13 +4373,43 @@ def run_conversation(
|
||||
break
|
||||
|
||||
partial_response = agent._strip_think_blocks(_join_truncated_parts(truncated_response_parts)).strip()
|
||||
# The pending one-shot reasoning-off override must
|
||||
# not leak into the next turn when the 4th
|
||||
# truncation goes straight to the ceiling exit
|
||||
# without scheduling a continuation call to
|
||||
# consume it.
|
||||
agent._ephemeral_reasoning_off = False
|
||||
if partial_response:
|
||||
agent._vprint(
|
||||
f"{agent.log_prefix}⚠️ Response still truncated "
|
||||
f"after 4 continuation attempts — keeping the "
|
||||
f"after {length_continue_retries} continuation attempts — keeping the "
|
||||
f"partial response received so far.",
|
||||
force=True,
|
||||
)
|
||||
_ceiling_final = partial_response
|
||||
else:
|
||||
# Every fragment was empty — e.g. a thinking
|
||||
# model that spent each attempt's whole cap on
|
||||
# reasoning (GLM-5.3 on ollama-cloud). Return
|
||||
# an actionable message instead of an invisible
|
||||
# None result, which only surfaces as a bare
|
||||
# error card.
|
||||
agent._vprint(
|
||||
f"{agent.log_prefix}⚠️ Response still truncated "
|
||||
f"after {length_continue_retries} continuation attempts — no visible "
|
||||
f"text was produced.",
|
||||
force=True,
|
||||
)
|
||||
_ceiling_final = (
|
||||
"⚠️ **No visible answer was produced.** The "
|
||||
"model hit its output-token limit on every "
|
||||
"continuation attempt — its reasoning "
|
||||
"consumed the entire budget each time.\n\n"
|
||||
"To fix this:\n"
|
||||
"→ Lower reasoning effort: `/thinkon low` "
|
||||
"or `/thinkoff`\n"
|
||||
"→ Or raise max_tokens for this model"
|
||||
)
|
||||
# Unanswered continue nudges made every later turn re-truncate.
|
||||
_turn_start = (
|
||||
current_turn_user_idx + 1
|
||||
@@ -4390,7 +4437,7 @@ def run_conversation(
|
||||
agent._cleanup_task_resources(effective_task_id)
|
||||
agent._persist_session(messages, conversation_history)
|
||||
return {
|
||||
"final_response": partial_response or None,
|
||||
"final_response": _ceiling_final,
|
||||
"messages": messages,
|
||||
"api_calls": api_call_count,
|
||||
"completed": False,
|
||||
|
||||
@@ -0,0 +1,228 @@
|
||||
"""Regression tests for thinking-only length truncations.
|
||||
|
||||
GLM-5.3-flash on ollama-cloud with reasoning_effort=high can burn the ENTIRE
|
||||
output cap on reasoning delivered in a separate field and return
|
||||
finish_reason="length" with NO visible content (verified live: max_tokens=4096
|
||||
→ completion_tokens=4096, reasoning ~18.5KB, content empty).
|
||||
|
||||
The old continuation flow handled this badly:
|
||||
1. the empty response was appended as an interim assistant fragment,
|
||||
poisoning the transcript until the pre-call sanitizer "healed" it
|
||||
(observed 3+ healings per turn);
|
||||
2. every continuation re-ran with thinking ON, re-deriving — and re-burning
|
||||
— the whole thinking budget against a growing context, so 4 attempts
|
||||
still produced nothing and the turn died with
|
||||
"Response remained truncated after 4 continuation attempts".
|
||||
|
||||
The fix: skip empty interim fragments, and issue the continuation with a
|
||||
one-shot reasoning-off override so the budget goes to writing the answer.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from hermes_constants import FINISH_REASON_LENGTH
|
||||
|
||||
|
||||
class _AgentStandIn:
|
||||
"""Minimal agent surface _reasoning_config_for_wire needs."""
|
||||
|
||||
def __init__(self, reasoning_config):
|
||||
self.reasoning_config = reasoning_config
|
||||
|
||||
|
||||
class TestReasoningOffOneShotOverride:
|
||||
def test_flag_consumed_exactly_once(self):
|
||||
from agent.chat_completion_helpers import _reasoning_config_for_wire
|
||||
|
||||
agent = _AgentStandIn({"enabled": True, "effort": "high"})
|
||||
# Without the flag the reasoning config passes through untouched.
|
||||
assert _reasoning_config_for_wire(agent) == {
|
||||
"enabled": True,
|
||||
"effort": "high",
|
||||
}
|
||||
|
||||
agent._ephemeral_reasoning_off = True
|
||||
cfg = _reasoning_config_for_wire(agent)
|
||||
assert cfg["enabled"] is False
|
||||
assert cfg["effort"] == "none"
|
||||
assert agent._ephemeral_reasoning_off is False, (
|
||||
"The one-shot override must be consumed by the first call."
|
||||
)
|
||||
|
||||
# Subsequent calls keep the user's own reasoning config.
|
||||
assert _reasoning_config_for_wire(agent) == {
|
||||
"enabled": True,
|
||||
"effort": "high",
|
||||
}
|
||||
|
||||
def test_flag_with_no_user_reasoning_config(self):
|
||||
from agent.chat_completion_helpers import _reasoning_config_for_wire
|
||||
|
||||
agent = _AgentStandIn(None)
|
||||
agent._ephemeral_reasoning_off = True
|
||||
cfg = _reasoning_config_for_wire(agent)
|
||||
assert cfg == {"enabled": False, "effort": "none"}
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def loop_agent():
|
||||
from run_agent import AIAgent
|
||||
|
||||
with (
|
||||
patch("run_agent.get_tool_definitions", return_value=[]),
|
||||
patch("run_agent.check_toolset_requirements", return_value={}),
|
||||
patch("run_agent.OpenAI"),
|
||||
):
|
||||
a = AIAgent(
|
||||
api_key="test-key-1234567890",
|
||||
base_url="https://openrouter.ai/api/v1",
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
)
|
||||
a.client = MagicMock()
|
||||
a._cached_system_prompt = "You are helpful."
|
||||
a._use_prompt_caching = False
|
||||
a.compression_enabled = False
|
||||
a.save_trajectories = False
|
||||
return a
|
||||
|
||||
|
||||
def _thinking_only_length_response():
|
||||
"""finish_reason='length' with reasoning but zero visible content — the
|
||||
live GLM-5.3-flash-on-ollama-cloud shape (normal response id, NOT the
|
||||
partial-stream stub)."""
|
||||
from tests.run_agent.test_run_agent import _mock_assistant_msg
|
||||
|
||||
return SimpleNamespace(
|
||||
id="chatcmpl-thinking-exhausted",
|
||||
model="test/model",
|
||||
choices=[SimpleNamespace(
|
||||
index=0,
|
||||
message=_mock_assistant_msg(content=""),
|
||||
finish_reason=FINISH_REASON_LENGTH,
|
||||
)],
|
||||
usage=None,
|
||||
)
|
||||
|
||||
|
||||
def _full_response(content):
|
||||
from tests.run_agent.test_run_agent import _mock_response
|
||||
|
||||
return _mock_response(content=content, finish_reason="stop")
|
||||
|
||||
|
||||
def _truncated_text_response(content):
|
||||
from tests.run_agent.test_run_agent import _mock_response
|
||||
|
||||
return _mock_response(content=content, finish_reason=FINISH_REASON_LENGTH)
|
||||
|
||||
|
||||
def _run(agent, message, history=None):
|
||||
with (
|
||||
patch.object(agent, "_persist_session"),
|
||||
patch.object(agent, "_save_trajectory"),
|
||||
patch.object(agent, "_cleanup_task_resources"),
|
||||
):
|
||||
return agent.run_conversation(message, conversation_history=history)
|
||||
|
||||
|
||||
def _no_empty_assistant_rows(messages):
|
||||
return [
|
||||
m for m in messages
|
||||
if m.get("role") == "assistant"
|
||||
and not (m.get("content") or "").strip()
|
||||
and not m.get("tool_calls")
|
||||
]
|
||||
|
||||
|
||||
class TestThinkingOnlyTruncation:
|
||||
def test_retry_after_thinking_only_truncation_completes(self, loop_agent):
|
||||
"""One thinking-only truncation, then a normal answer: the retry must
|
||||
drop thinking (one-shot), boost the output cap, and finish the turn."""
|
||||
loop_agent.client.chat.completions.create.side_effect = [
|
||||
_thinking_only_length_response(),
|
||||
_full_response("Here is the full answer."),
|
||||
]
|
||||
result = _run(loop_agent, "write me a long report")
|
||||
|
||||
assert result["completed"] is True
|
||||
assert "full answer" in (result["final_response"] or "")
|
||||
assert _no_empty_assistant_rows(result["messages"]) == [], (
|
||||
"An empty (thinking-only) truncated response must never be "
|
||||
"appended to the transcript."
|
||||
)
|
||||
|
||||
calls = loop_agent.client.chat.completions.create.call_args_list
|
||||
assert len(calls) == 2
|
||||
# Continuation retry boosts the output cap (2^1 × 4096 base floor).
|
||||
assert calls[1].kwargs.get("max_tokens") == 8192, (
|
||||
"The continuation retry must request a larger output budget than "
|
||||
"the request that truncated."
|
||||
)
|
||||
assert loop_agent._ephemeral_reasoning_off is False, (
|
||||
"The one-shot reasoning-off override must be consumed by the "
|
||||
"continuation call."
|
||||
)
|
||||
|
||||
def test_thinking_only_truncation_sets_reasoning_off(self, loop_agent):
|
||||
from tests.run_agent.test_run_agent import _mock_response
|
||||
|
||||
loop_agent.client.chat.completions.create.side_effect = [
|
||||
_thinking_only_length_response(),
|
||||
_mock_response(
|
||||
content="done", finish_reason=FINISH_REASON_LENGTH
|
||||
),
|
||||
_full_response("finally complete."),
|
||||
]
|
||||
_run(loop_agent, "write me a long report")
|
||||
|
||||
calls = loop_agent.client.chat.completions.create.call_args_list
|
||||
assert len(calls) == 3
|
||||
# The thinking-only fragment set the flag; it was consumed by the
|
||||
# next call, and the SECOND truncated fragment (which had visible
|
||||
# text) does not set it again — so the third call sees thinking ON.
|
||||
assert loop_agent._ephemeral_reasoning_off is False
|
||||
|
||||
def test_full_ceiling_with_empty_fragments_still_settles(self, loop_agent):
|
||||
"""All four attempts thinking-only: the turn must exit through the
|
||||
ceiling with an actionable final_response, no poisoned transcript,
|
||||
and no leaked reasoning-off flag."""
|
||||
loop_agent.client.chat.completions.create.side_effect = [
|
||||
_thinking_only_length_response() for _ in range(4)
|
||||
]
|
||||
result = _run(loop_agent, "write me a long report")
|
||||
|
||||
assert result["completed"] is False
|
||||
assert result["partial"] is True
|
||||
assert "truncated after 4 continuation attempts" in (result.get("error") or "")
|
||||
assert result["final_response"], (
|
||||
"An all-empty ceiling exit must still surface a user-facing "
|
||||
"message instead of an invisible None."
|
||||
)
|
||||
assert "reasoning" in (result["final_response"] or "").lower()
|
||||
assert _no_empty_assistant_rows(result["messages"]) == []
|
||||
assert loop_agent._ephemeral_reasoning_off is False, (
|
||||
"The ceiling exit must clear the pending one-shot override so the "
|
||||
"next turn does not silently lose thinking."
|
||||
)
|
||||
|
||||
def test_mixed_fragments_keep_visible_text(self, loop_agent):
|
||||
"""A visible fragment followed by a thinking-only one: the visible
|
||||
text must be stitched, the empty one skipped."""
|
||||
loop_agent.client.chat.completions.create.side_effect = [
|
||||
_truncated_text_response("visible part one. "),
|
||||
_thinking_only_length_response(),
|
||||
_full_response("and the ending."),
|
||||
]
|
||||
result = _run(loop_agent, "write me a long report")
|
||||
|
||||
assert result["completed"] is True
|
||||
assert "visible part one." in (result["final_response"] or "")
|
||||
assert "and the ending." in (result["final_response"] or "")
|
||||
assert _no_empty_assistant_rows(result["messages"]) == []
|
||||
Reference in New Issue
Block a user