Files
hermes-agent/tests/run_agent/test_malformed_tool_arguments.py
T
Teknium 2a95791992 simplify(compat): run_agent/model_tools/toolsets/acp/providers — drop 42 re-exports/aliases, repoint 15 callers + 99 test files
run_agent.py: delete the `# noqa: F401` re-export block (agent.process_bootstrap
OpenAI/_SafeWriter/_get_proxy_*, model_tools get_tool_definitions/
handle_function_call/check_toolset_requirements, FailoverReason,
_qwen_portal_headers/_routermint_headers, session_persistence names,
estimate_request_tokens_rough, ContextCompressor + friends, jittered_backoff,
prompt_builder names, message_sanitization names, tool_dispatch_helpers
names) — 41 names run_agent never used itself — and the `_STREAM_DIAG_HEADERS`
back-compat class alias (no in-tree reader). run_agent now imports only what
it uses (get_toolset_for_tool, is_local_endpoint, coalesce/uniquify tool-call
ids, cleanup_vm/get_active_env from terminal_tool_lifecycle).

agent/*: `_ra().X` late-binds that only reached a re-export now import the
defining module directly (agent_runtime_helpers -> process_bootstrap.OpenAI,
model_tools.handle_function_call, session_persistence._safe_session_filename_component;
agent_init -> model_tools.get_tool_definitions/check_toolset_requirements,
_lazy_headers("agent.client_lifecycle", ...) for qwen/routermint;
system_prompt -> agent.prompt_builder / model_tools directly, dropping its
own _ra() shim and the `_r` parameter threading). `_ra()` stays for
run_agent-resident names (logger, AIAgent, _hermes_home, _set_interrupt, ...).

toolsets.py: remove resolve_multiple_toolsets (shim-only, restored by
34abf954bd); tests/test_toolsets.py pins the same union behavior via
resolve_toolset over each name.

providers/__init__.py: drop the OMIT_TEMPERATURE re-export (no callers via the
package); ProviderProfile stays because __init__ uses it for annotations —
2 tests repointed to providers.base.

agent/iteration_budget.py: drop the "run_agent re-exports the class"
docstring pointer; 4 tests import IterationBudget from its home.

model_tools.py (arg_coercion names), agent/tool_executor.py, and
hermes_cli/cli_session_mixin.py repoints landed via a sibling commit on this
shared worktree.

Callers repointed: gateway/run.py, hermes_cli/cli_chat_turn_mixin.py,
hermes_cli/cli_tui_mixin.py, tui_gateway/session_workdir.py,
agent/transports/codex.py (one-line imports) + comment pointers in
tools/file_state.py, tools/schema_sanitizer.py, scripts/tool_search_livetest.py.
Tests: patch("run_agent.X") / monkeypatch.setattr(run_agent, "X") /
`from run_agent import X` -> defining module across 99 test files.
2026-09-03 13:28:22 -07:00

98 lines
3.2 KiB
Python

"""Malformed model tool arguments are rejected at the dispatch boundary."""
import json
from types import SimpleNamespace
from unittest.mock import MagicMock, patch
import pytest
from run_agent import AIAgent
def _make_agent() -> AIAgent:
tool_defs = [
{
"type": "function",
"function": {
"name": "web_search",
"description": "search",
"parameters": {"type": "object", "properties": {}},
},
}
]
with (
patch("model_tools.get_tool_definitions", return_value=tool_defs),
patch("model_tools.check_toolset_requirements", return_value={}),
patch("hermes_cli.config.load_config", return_value={}),
patch("agent.process_bootstrap.OpenAI"),
):
agent = AIAgent(
api_key="test-key-1234567890",
base_url="https://openrouter.ai/api/v1",
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
agent.client = MagicMock()
agent._flush_messages_to_session_db = MagicMock()
return agent
def _tool_call(call_id: str, arguments: str):
return SimpleNamespace(
id=call_id,
type="function",
function=SimpleNamespace(name="web_search", arguments=arguments),
)
@pytest.mark.parametrize("dispatch_mode", ["sequential", "concurrent"])
@pytest.mark.parametrize(
"bad_arguments",
[
pytest.param("not-json", id="malformed-json"),
pytest.param('"scalar"', id="scalar"),
pytest.param("[]", id="list"),
pytest.param("", id="empty"),
pytest.param('{"query": "cut off', id="truncated"),
],
)
def test_malformed_arguments_are_rejected_without_blocking_valid_sibling(
dispatch_mode: str,
bad_arguments: str,
):
agent = _make_agent()
assistant_message = SimpleNamespace(
content="",
tool_calls=[
_tool_call("call-bad", bad_arguments),
_tool_call("call-good", '{"query": "valid"}'),
],
)
messages = []
executed = []
def fake_dispatch(name, args, task_id, *positional, **kwargs):
call_id = kwargs.get("tool_call_id") or (positional[0] if positional else None)
executed.append((name, args, call_id))
return json.dumps({"ok": args["query"]})
with (
patch("model_tools.handle_function_call", side_effect=fake_dispatch),
patch.object(agent, "_invoke_tool", side_effect=fake_dispatch),
patch(
"agent.tool_executor.maybe_persist_tool_result",
side_effect=lambda **kwargs: kwargs["content"],
),
):
execute = getattr(agent, f"_execute_tool_calls_{dispatch_mode}")
execute(assistant_message, messages, "task-1")
assert executed == [("web_search", {"query": "valid"}, "call-good")]
assert [message["tool_call_id"] for message in messages] == ["call-bad", "call-good"]
assert len([message for message in messages if message["tool_call_id"] == "call-bad"]) == 1
assert '"error": "Invalid tool arguments"' in messages[0]["content"]
assert "JSON object" in messages[0]["content"]
assert json.loads(messages[1]["content"]) == {"ok": "valid"}