Files
hermes-agent/tests/run_agent/test_model_streaming_config.py
T
Teknium 2a95791992 simplify(compat): run_agent/model_tools/toolsets/acp/providers — drop 42 re-exports/aliases, repoint 15 callers + 99 test files
run_agent.py: delete the `# noqa: F401` re-export block (agent.process_bootstrap
OpenAI/_SafeWriter/_get_proxy_*, model_tools get_tool_definitions/
handle_function_call/check_toolset_requirements, FailoverReason,
_qwen_portal_headers/_routermint_headers, session_persistence names,
estimate_request_tokens_rough, ContextCompressor + friends, jittered_backoff,
prompt_builder names, message_sanitization names, tool_dispatch_helpers
names) — 41 names run_agent never used itself — and the `_STREAM_DIAG_HEADERS`
back-compat class alias (no in-tree reader). run_agent now imports only what
it uses (get_toolset_for_tool, is_local_endpoint, coalesce/uniquify tool-call
ids, cleanup_vm/get_active_env from terminal_tool_lifecycle).

agent/*: `_ra().X` late-binds that only reached a re-export now import the
defining module directly (agent_runtime_helpers -> process_bootstrap.OpenAI,
model_tools.handle_function_call, session_persistence._safe_session_filename_component;
agent_init -> model_tools.get_tool_definitions/check_toolset_requirements,
_lazy_headers("agent.client_lifecycle", ...) for qwen/routermint;
system_prompt -> agent.prompt_builder / model_tools directly, dropping its
own _ra() shim and the `_r` parameter threading). `_ra()` stays for
run_agent-resident names (logger, AIAgent, _hermes_home, _set_interrupt, ...).

toolsets.py: remove resolve_multiple_toolsets (shim-only, restored by
34abf954bd); tests/test_toolsets.py pins the same union behavior via
resolve_toolset over each name.

providers/__init__.py: drop the OMIT_TEMPERATURE re-export (no callers via the
package); ProviderProfile stays because __init__ uses it for annotations —
2 tests repointed to providers.base.

agent/iteration_budget.py: drop the "run_agent re-exports the class"
docstring pointer; 4 tests import IterationBudget from its home.

model_tools.py (arg_coercion names), agent/tool_executor.py, and
hermes_cli/cli_session_mixin.py repoints landed via a sibling commit on this
shared worktree.

Callers repointed: gateway/run.py, hermes_cli/cli_chat_turn_mixin.py,
hermes_cli/cli_tui_mixin.py, tui_gateway/session_workdir.py,
agent/transports/codex.py (one-line imports) + comment pointers in
tools/file_state.py, tools/schema_sanitizer.py, scripts/tool_search_livetest.py.
Tests: patch("run_agent.X") / monkeypatch.setattr(run_agent, "X") /
`from run_agent import X` -> defining module across 99 test files.
2026-09-03 13:28:22 -07:00

152 lines
5.1 KiB
Python

"""``model.streaming`` config seeds the session's streaming decision (#72901).
The conversation loop prefers ``stream=True`` for every turn — subagents
included — for liveness health-checking (#3120). Self-hosted OpenAI-compatible
backends with broken streaming tool-call paths (e.g. vLLM
``--tool-call-parser qwen3_xml`` + reasoning parser) can leak tool-call markup
into plain text and return zero ``tool_calls``, silently no-oping delegated
tasks. ``model.streaming: false`` must seed ``_disable_streaming`` at agent
init so the whole session (parent and subagents) uses the non-streaming path.
"""
import os
from pathlib import Path
from unittest.mock import MagicMock, patch
from run_agent import AIAgent
_BASE = {
"model": {
"default": "test/model",
"provider": "custom",
"base_url": "http://127.0.0.1:9999/v1",
"api_key": "x",
}
}
def _build_agent(config):
with patch("hermes_cli.config.load_config_readonly", return_value=config):
return AIAgent(
api_key="x",
base_url="http://127.0.0.1:9999/v1",
model="test/model",
provider="custom",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
@patch("agent.process_bootstrap.OpenAI")
def test_streaming_false_seeds_disable_streaming(mock_openai):
mock_openai.return_value = MagicMock()
agent = _build_agent({"model": {**_BASE["model"], "streaming": False}})
assert agent._disable_streaming is True
@patch("agent.process_bootstrap.OpenAI")
def test_streaming_absent_keeps_streaming_enabled(mock_openai):
mock_openai.return_value = MagicMock()
agent = _build_agent(_BASE)
assert agent._disable_streaming is False
@patch("agent.process_bootstrap.OpenAI")
def test_streaming_true_keeps_streaming_enabled(mock_openai):
mock_openai.return_value = MagicMock()
agent = _build_agent({"model": {**_BASE["model"], "streaming": True}})
assert agent._disable_streaming is False
@patch("agent.process_bootstrap.OpenAI")
def test_streaming_string_false_seeds_disable_streaming(mock_openai):
"""String falsy values ('false', '0') must also disable streaming —
YAML users commonly quote booleans."""
mock_openai.return_value = MagicMock()
agent = _build_agent({"model": {**_BASE["model"], "streaming": "false"}})
assert agent._disable_streaming is True
@patch("agent.process_bootstrap.OpenAI")
def test_streaming_zero_seeds_disable_streaming(mock_openai):
mock_openai.return_value = MagicMock()
agent = _build_agent({"model": {**_BASE["model"], "streaming": 0}})
assert agent._disable_streaming is True
@patch("agent.process_bootstrap.OpenAI")
def test_streaming_invalid_value_keeps_streaming_enabled(mock_openai):
"""Unrecognized values warn and keep the safe default (streaming on),
rather than silently disabling or crashing init."""
mock_openai.return_value = MagicMock()
agent = _build_agent({"model": {**_BASE["model"], "streaming": "flase"}})
assert agent._disable_streaming is False
@patch("agent.process_bootstrap.OpenAI")
def test_missing_model_section_keeps_streaming_enabled(mock_openai):
mock_openai.return_value = MagicMock()
agent = _build_agent({})
assert agent._disable_streaming is False
@patch("agent.process_bootstrap.OpenAI")
def test_legacy_string_model_section_does_not_crash(mock_openai):
"""The top-level ``model`` key is a legacy string; init must not crash."""
mock_openai.return_value = MagicMock()
agent = _build_agent({"model": "test/model"})
assert agent._disable_streaming is False
@patch("agent.process_bootstrap.OpenAI")
def test_streaming_false_applies_to_every_agent_built_from_config(mock_openai):
"""Delegate children are constructed through the same init, so any agent
(parent or subagent) built under this config gets the escape hatch —
covering the reported failure surface."""
mock_openai.return_value = MagicMock()
cfg = {"model": {**_BASE["model"], "streaming": False}}
first = _build_agent(cfg)
second = _build_agent(cfg)
assert first._disable_streaming is True
assert second._disable_streaming is True
@patch("agent.process_bootstrap.OpenAI")
def test_streaming_false_read_from_real_config_file(mock_openai):
"""End-to-end: a real config.yaml in HERMES_HOME (sandboxed per-test by
conftest) with ``model.streaming: false`` must seed the flag through the
actual config loader — not just the patched function."""
mock_openai.return_value = MagicMock()
home = Path(os.environ["HERMES_HOME"])
(home / "config.yaml").write_text(
"model:\n"
" default: \"test/model\"\n"
" provider: \"custom\"\n"
" base_url: \"http://127.0.0.1:9999/v1\"\n"
" api_key: \"x\"\n"
" streaming: false\n",
encoding="utf-8",
)
agent = AIAgent(
api_key="x",
base_url="http://127.0.0.1:9999/v1",
model="test/model",
provider="custom",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
assert agent._disable_streaming is True