From c905c2b4b55beb63abc316aa7ca0b8a55456ae4b Mon Sep 17 00:00:00 2001 From: David Metcalfe <80915+DavidMetcalfe@users.noreply.github.com> Date: Thu, 6 Aug 2026 19:29:23 -0700 Subject: [PATCH] fix(agent): honor model.streaming: false as a non-streaming escape hatch (#72901) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The conversation loop has forced stream=True for every turn — subagents included — since #3120 (always-prefer-streaming for liveness health checking). Self-hosted OpenAI-compatible backends with broken streaming tool-call paths (e.g. vLLM --tool-call-parser qwen3_xml + reasoning parser + MTP) can leak tool-call markup into plain text and return zero tool_calls, so delegated tasks silently no-op instead of executing. model.streaming was never a real config key, so users could not opt out. Seed agent._disable_streaming from model.streaming: false at init; the loop already routes that flag to the non-streaming path (the same path used when a provider rejects streaming at runtime). Default stays streaming-on, preserving #3120's behavior for everyone else. Orthogonal to display.streaming (token rendering). Tests: config->flag seeding (patched loader + real config.yaml E2E), legacy string model section, multi-agent config propagation. --- agent/agent_init.py | 28 ++++ cli-config.yaml.example | 11 ++ .../run_agent/test_model_streaming_config.py | 151 ++++++++++++++++++ 3 files changed, 190 insertions(+) create mode 100644 tests/run_agent/test_model_streaming_config.py diff --git a/agent/agent_init.py b/agent/agent_init.py index 72c1513925..b4e2e9ae5d 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -1851,6 +1851,34 @@ def init_agent( except Exception: agent.lmstudio_load_mode = "explicit" + # API-transport streaming (``model.streaming``, default true). The + # conversation loop prefers ``stream=True`` for every turn — including + # subagent turns — to get fine-grained liveness health-checking (#3120), + # but self-hosted OpenAI-compatible backends with broken streaming + # tool-call paths (e.g. vLLM ``--tool-call-parser qwen3_xml`` + a + # reasoning parser can leak tool-call markup into plain text and return + # zero ``tool_calls``, #72901) silently no-op instead of executing. + # ``model.streaming: false`` seeds ``_disable_streaming`` so the session + # uses the non-streaming path, which the loop already falls back to at + # runtime when a provider rejects streaming. The setting is + # session-scoped: it persists across mid-session model switches, mirroring + # the runtime fallback's semantics. Orthogonal to ``display.streaming`` + # (token rendering) — display-only settings are untouched. + agent._disable_streaming = False + try: + _model_section = _agent_cfg.get("model", {}) + if isinstance(_model_section, dict): + _streaming = str(_model_section.get("streaming", "true")).strip().lower() + if _streaming in {"false", "0", "no", "off"}: + agent._disable_streaming = True + elif _streaming not in {"true", "1", "yes", "on"}: + logger.warning( + "Invalid model.streaming=%r; expected a boolean. Using streaming (default).", + _model_section.get("streaming"), + ) + except Exception: + agent._disable_streaming = False + try: agent._tool_guardrails = ToolCallGuardrailController( ToolCallGuardrailConfig.from_mapping( diff --git a/cli-config.yaml.example b/cli-config.yaml.example index 50f927c368..c45203cfcb 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -82,6 +82,17 @@ model: # api_key: "your-key-here" # Uncomment to set here instead of .env base_url: "https://openrouter.ai/api/v1" + # Stream API responses from the provider (default: true). The agent core + # prefers streaming for every turn — subagents included — for liveness + # health-checking. Set false to force non-streaming requests for the whole + # session (persists across mid-session model switches). Escape hatch for + # self-hosted OpenAI-compatible servers whose streaming tool-call path is + # broken (e.g. vLLM with --tool-call-parser qwen3_xml + a reasoning parser + # can leak tool calls into plain text instead of returning tool_calls — + # #72901). Orthogonal to display.streaming, which controls token rendering + # only. + # streaming: true + # Azure Foundry keyless auth example: # provider: "azure-foundry" # base_url: "https://.openai.azure.com/openai/v1" diff --git a/tests/run_agent/test_model_streaming_config.py b/tests/run_agent/test_model_streaming_config.py new file mode 100644 index 0000000000..1ed5073f12 --- /dev/null +++ b/tests/run_agent/test_model_streaming_config.py @@ -0,0 +1,151 @@ +"""``model.streaming`` config seeds the session's streaming decision (#72901). + +The conversation loop prefers ``stream=True`` for every turn — subagents +included — for liveness health-checking (#3120). Self-hosted OpenAI-compatible +backends with broken streaming tool-call paths (e.g. vLLM +``--tool-call-parser qwen3_xml`` + reasoning parser) can leak tool-call markup +into plain text and return zero ``tool_calls``, silently no-oping delegated +tasks. ``model.streaming: false`` must seed ``_disable_streaming`` at agent +init so the whole session (parent and subagents) uses the non-streaming path. +""" +import os +from pathlib import Path +from unittest.mock import MagicMock, patch + +from run_agent import AIAgent + +_BASE = { + "model": { + "default": "test/model", + "provider": "custom", + "base_url": "http://127.0.0.1:9999/v1", + "api_key": "x", + } +} + + +def _build_agent(config): + with patch("hermes_cli.config.load_config_readonly", return_value=config): + return AIAgent( + api_key="x", + base_url="http://127.0.0.1:9999/v1", + model="test/model", + provider="custom", + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + ) + + +@patch("run_agent.OpenAI") +def test_streaming_false_seeds_disable_streaming(mock_openai): + mock_openai.return_value = MagicMock() + agent = _build_agent({"model": {**_BASE["model"], "streaming": False}}) + + assert agent._disable_streaming is True + + +@patch("run_agent.OpenAI") +def test_streaming_absent_keeps_streaming_enabled(mock_openai): + mock_openai.return_value = MagicMock() + agent = _build_agent(_BASE) + + assert agent._disable_streaming is False + + +@patch("run_agent.OpenAI") +def test_streaming_true_keeps_streaming_enabled(mock_openai): + mock_openai.return_value = MagicMock() + agent = _build_agent({"model": {**_BASE["model"], "streaming": True}}) + + assert agent._disable_streaming is False + + +@patch("run_agent.OpenAI") +def test_streaming_string_false_seeds_disable_streaming(mock_openai): + """String falsy values ('false', '0') must also disable streaming — + YAML users commonly quote booleans.""" + mock_openai.return_value = MagicMock() + agent = _build_agent({"model": {**_BASE["model"], "streaming": "false"}}) + + assert agent._disable_streaming is True + + +@patch("run_agent.OpenAI") +def test_streaming_zero_seeds_disable_streaming(mock_openai): + mock_openai.return_value = MagicMock() + agent = _build_agent({"model": {**_BASE["model"], "streaming": 0}}) + + assert agent._disable_streaming is True + + +@patch("run_agent.OpenAI") +def test_streaming_invalid_value_keeps_streaming_enabled(mock_openai): + """Unrecognized values warn and keep the safe default (streaming on), + rather than silently disabling or crashing init.""" + mock_openai.return_value = MagicMock() + agent = _build_agent({"model": {**_BASE["model"], "streaming": "flase"}}) + + assert agent._disable_streaming is False + + +@patch("run_agent.OpenAI") +def test_missing_model_section_keeps_streaming_enabled(mock_openai): + mock_openai.return_value = MagicMock() + agent = _build_agent({}) + + assert agent._disable_streaming is False + + +@patch("run_agent.OpenAI") +def test_legacy_string_model_section_does_not_crash(mock_openai): + """The top-level ``model`` key is a legacy string; init must not crash.""" + mock_openai.return_value = MagicMock() + agent = _build_agent({"model": "test/model"}) + + assert agent._disable_streaming is False + + +@patch("run_agent.OpenAI") +def test_streaming_false_applies_to_every_agent_built_from_config(mock_openai): + """Delegate children are constructed through the same init, so any agent + (parent or subagent) built under this config gets the escape hatch — + covering the reported failure surface.""" + mock_openai.return_value = MagicMock() + cfg = {"model": {**_BASE["model"], "streaming": False}} + + first = _build_agent(cfg) + second = _build_agent(cfg) + + assert first._disable_streaming is True + assert second._disable_streaming is True + + +@patch("run_agent.OpenAI") +def test_streaming_false_read_from_real_config_file(mock_openai): + """End-to-end: a real config.yaml in HERMES_HOME (sandboxed per-test by + conftest) with ``model.streaming: false`` must seed the flag through the + actual config loader — not just the patched function.""" + mock_openai.return_value = MagicMock() + home = Path(os.environ["HERMES_HOME"]) + (home / "config.yaml").write_text( + "model:\n" + " default: \"test/model\"\n" + " provider: \"custom\"\n" + " base_url: \"http://127.0.0.1:9999/v1\"\n" + " api_key: \"x\"\n" + " streaming: false\n", + encoding="utf-8", + ) + + agent = AIAgent( + api_key="x", + base_url="http://127.0.0.1:9999/v1", + model="test/model", + provider="custom", + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + ) + + assert agent._disable_streaming is True