fix(agent): honor model.streaming: false as a non-streaming escape hatch (#72901)
The conversation loop has forced stream=True for every turn — subagents included — since #3120 (always-prefer-streaming for liveness health checking). Self-hosted OpenAI-compatible backends with broken streaming tool-call paths (e.g. vLLM --tool-call-parser qwen3_xml + reasoning parser + MTP) can leak tool-call markup into plain text and return zero tool_calls, so delegated tasks silently no-op instead of executing. model.streaming was never a real config key, so users could not opt out. Seed agent._disable_streaming from model.streaming: false at init; the loop already routes that flag to the non-streaming path (the same path used when a provider rejects streaming at runtime). Default stays streaming-on, preserving #3120's behavior for everyone else. Orthogonal to display.streaming (token rendering). Tests: config->flag seeding (patched loader + real config.yaml E2E), legacy string model section, multi-agent config propagation.
This commit is contained in:
@@ -1851,6 +1851,34 @@ def init_agent(
|
||||
except Exception:
|
||||
agent.lmstudio_load_mode = "explicit"
|
||||
|
||||
# API-transport streaming (``model.streaming``, default true). The
|
||||
# conversation loop prefers ``stream=True`` for every turn — including
|
||||
# subagent turns — to get fine-grained liveness health-checking (#3120),
|
||||
# but self-hosted OpenAI-compatible backends with broken streaming
|
||||
# tool-call paths (e.g. vLLM ``--tool-call-parser qwen3_xml`` + a
|
||||
# reasoning parser can leak tool-call markup into plain text and return
|
||||
# zero ``tool_calls``, #72901) silently no-op instead of executing.
|
||||
# ``model.streaming: false`` seeds ``_disable_streaming`` so the session
|
||||
# uses the non-streaming path, which the loop already falls back to at
|
||||
# runtime when a provider rejects streaming. The setting is
|
||||
# session-scoped: it persists across mid-session model switches, mirroring
|
||||
# the runtime fallback's semantics. Orthogonal to ``display.streaming``
|
||||
# (token rendering) — display-only settings are untouched.
|
||||
agent._disable_streaming = False
|
||||
try:
|
||||
_model_section = _agent_cfg.get("model", {})
|
||||
if isinstance(_model_section, dict):
|
||||
_streaming = str(_model_section.get("streaming", "true")).strip().lower()
|
||||
if _streaming in {"false", "0", "no", "off"}:
|
||||
agent._disable_streaming = True
|
||||
elif _streaming not in {"true", "1", "yes", "on"}:
|
||||
logger.warning(
|
||||
"Invalid model.streaming=%r; expected a boolean. Using streaming (default).",
|
||||
_model_section.get("streaming"),
|
||||
)
|
||||
except Exception:
|
||||
agent._disable_streaming = False
|
||||
|
||||
try:
|
||||
agent._tool_guardrails = ToolCallGuardrailController(
|
||||
ToolCallGuardrailConfig.from_mapping(
|
||||
|
||||
@@ -82,6 +82,17 @@ model:
|
||||
# api_key: "your-key-here" # Uncomment to set here instead of .env
|
||||
base_url: "https://openrouter.ai/api/v1"
|
||||
|
||||
# Stream API responses from the provider (default: true). The agent core
|
||||
# prefers streaming for every turn — subagents included — for liveness
|
||||
# health-checking. Set false to force non-streaming requests for the whole
|
||||
# session (persists across mid-session model switches). Escape hatch for
|
||||
# self-hosted OpenAI-compatible servers whose streaming tool-call path is
|
||||
# broken (e.g. vLLM with --tool-call-parser qwen3_xml + a reasoning parser
|
||||
# can leak tool calls into plain text instead of returning tool_calls —
|
||||
# #72901). Orthogonal to display.streaming, which controls token rendering
|
||||
# only.
|
||||
# streaming: true
|
||||
|
||||
# Azure Foundry keyless auth example:
|
||||
# provider: "azure-foundry"
|
||||
# base_url: "https://<resource>.openai.azure.com/openai/v1"
|
||||
|
||||
@@ -0,0 +1,151 @@
|
||||
"""``model.streaming`` config seeds the session's streaming decision (#72901).
|
||||
|
||||
The conversation loop prefers ``stream=True`` for every turn — subagents
|
||||
included — for liveness health-checking (#3120). Self-hosted OpenAI-compatible
|
||||
backends with broken streaming tool-call paths (e.g. vLLM
|
||||
``--tool-call-parser qwen3_xml`` + reasoning parser) can leak tool-call markup
|
||||
into plain text and return zero ``tool_calls``, silently no-oping delegated
|
||||
tasks. ``model.streaming: false`` must seed ``_disable_streaming`` at agent
|
||||
init so the whole session (parent and subagents) uses the non-streaming path.
|
||||
"""
|
||||
import os
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from run_agent import AIAgent
|
||||
|
||||
_BASE = {
|
||||
"model": {
|
||||
"default": "test/model",
|
||||
"provider": "custom",
|
||||
"base_url": "http://127.0.0.1:9999/v1",
|
||||
"api_key": "x",
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
def _build_agent(config):
|
||||
with patch("hermes_cli.config.load_config_readonly", return_value=config):
|
||||
return AIAgent(
|
||||
api_key="x",
|
||||
base_url="http://127.0.0.1:9999/v1",
|
||||
model="test/model",
|
||||
provider="custom",
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
)
|
||||
|
||||
|
||||
@patch("run_agent.OpenAI")
|
||||
def test_streaming_false_seeds_disable_streaming(mock_openai):
|
||||
mock_openai.return_value = MagicMock()
|
||||
agent = _build_agent({"model": {**_BASE["model"], "streaming": False}})
|
||||
|
||||
assert agent._disable_streaming is True
|
||||
|
||||
|
||||
@patch("run_agent.OpenAI")
|
||||
def test_streaming_absent_keeps_streaming_enabled(mock_openai):
|
||||
mock_openai.return_value = MagicMock()
|
||||
agent = _build_agent(_BASE)
|
||||
|
||||
assert agent._disable_streaming is False
|
||||
|
||||
|
||||
@patch("run_agent.OpenAI")
|
||||
def test_streaming_true_keeps_streaming_enabled(mock_openai):
|
||||
mock_openai.return_value = MagicMock()
|
||||
agent = _build_agent({"model": {**_BASE["model"], "streaming": True}})
|
||||
|
||||
assert agent._disable_streaming is False
|
||||
|
||||
|
||||
@patch("run_agent.OpenAI")
|
||||
def test_streaming_string_false_seeds_disable_streaming(mock_openai):
|
||||
"""String falsy values ('false', '0') must also disable streaming —
|
||||
YAML users commonly quote booleans."""
|
||||
mock_openai.return_value = MagicMock()
|
||||
agent = _build_agent({"model": {**_BASE["model"], "streaming": "false"}})
|
||||
|
||||
assert agent._disable_streaming is True
|
||||
|
||||
|
||||
@patch("run_agent.OpenAI")
|
||||
def test_streaming_zero_seeds_disable_streaming(mock_openai):
|
||||
mock_openai.return_value = MagicMock()
|
||||
agent = _build_agent({"model": {**_BASE["model"], "streaming": 0}})
|
||||
|
||||
assert agent._disable_streaming is True
|
||||
|
||||
|
||||
@patch("run_agent.OpenAI")
|
||||
def test_streaming_invalid_value_keeps_streaming_enabled(mock_openai):
|
||||
"""Unrecognized values warn and keep the safe default (streaming on),
|
||||
rather than silently disabling or crashing init."""
|
||||
mock_openai.return_value = MagicMock()
|
||||
agent = _build_agent({"model": {**_BASE["model"], "streaming": "flase"}})
|
||||
|
||||
assert agent._disable_streaming is False
|
||||
|
||||
|
||||
@patch("run_agent.OpenAI")
|
||||
def test_missing_model_section_keeps_streaming_enabled(mock_openai):
|
||||
mock_openai.return_value = MagicMock()
|
||||
agent = _build_agent({})
|
||||
|
||||
assert agent._disable_streaming is False
|
||||
|
||||
|
||||
@patch("run_agent.OpenAI")
|
||||
def test_legacy_string_model_section_does_not_crash(mock_openai):
|
||||
"""The top-level ``model`` key is a legacy string; init must not crash."""
|
||||
mock_openai.return_value = MagicMock()
|
||||
agent = _build_agent({"model": "test/model"})
|
||||
|
||||
assert agent._disable_streaming is False
|
||||
|
||||
|
||||
@patch("run_agent.OpenAI")
|
||||
def test_streaming_false_applies_to_every_agent_built_from_config(mock_openai):
|
||||
"""Delegate children are constructed through the same init, so any agent
|
||||
(parent or subagent) built under this config gets the escape hatch —
|
||||
covering the reported failure surface."""
|
||||
mock_openai.return_value = MagicMock()
|
||||
cfg = {"model": {**_BASE["model"], "streaming": False}}
|
||||
|
||||
first = _build_agent(cfg)
|
||||
second = _build_agent(cfg)
|
||||
|
||||
assert first._disable_streaming is True
|
||||
assert second._disable_streaming is True
|
||||
|
||||
|
||||
@patch("run_agent.OpenAI")
|
||||
def test_streaming_false_read_from_real_config_file(mock_openai):
|
||||
"""End-to-end: a real config.yaml in HERMES_HOME (sandboxed per-test by
|
||||
conftest) with ``model.streaming: false`` must seed the flag through the
|
||||
actual config loader — not just the patched function."""
|
||||
mock_openai.return_value = MagicMock()
|
||||
home = Path(os.environ["HERMES_HOME"])
|
||||
(home / "config.yaml").write_text(
|
||||
"model:\n"
|
||||
" default: \"test/model\"\n"
|
||||
" provider: \"custom\"\n"
|
||||
" base_url: \"http://127.0.0.1:9999/v1\"\n"
|
||||
" api_key: \"x\"\n"
|
||||
" streaming: false\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
agent = AIAgent(
|
||||
api_key="x",
|
||||
base_url="http://127.0.0.1:9999/v1",
|
||||
model="test/model",
|
||||
provider="custom",
|
||||
quiet_mode=True,
|
||||
skip_context_files=True,
|
||||
skip_memory=True,
|
||||
)
|
||||
|
||||
assert agent._disable_streaming is True
|
||||
Reference in New Issue
Block a user