2a95791992
run_agent.py: delete the `# noqa: F401` re-export block (agent.process_bootstrap
OpenAI/_SafeWriter/_get_proxy_*, model_tools get_tool_definitions/
handle_function_call/check_toolset_requirements, FailoverReason,
_qwen_portal_headers/_routermint_headers, session_persistence names,
estimate_request_tokens_rough, ContextCompressor + friends, jittered_backoff,
prompt_builder names, message_sanitization names, tool_dispatch_helpers
names) — 41 names run_agent never used itself — and the `_STREAM_DIAG_HEADERS`
back-compat class alias (no in-tree reader). run_agent now imports only what
it uses (get_toolset_for_tool, is_local_endpoint, coalesce/uniquify tool-call
ids, cleanup_vm/get_active_env from terminal_tool_lifecycle).
agent/*: `_ra().X` late-binds that only reached a re-export now import the
defining module directly (agent_runtime_helpers -> process_bootstrap.OpenAI,
model_tools.handle_function_call, session_persistence._safe_session_filename_component;
agent_init -> model_tools.get_tool_definitions/check_toolset_requirements,
_lazy_headers("agent.client_lifecycle", ...) for qwen/routermint;
system_prompt -> agent.prompt_builder / model_tools directly, dropping its
own _ra() shim and the `_r` parameter threading). `_ra()` stays for
run_agent-resident names (logger, AIAgent, _hermes_home, _set_interrupt, ...).
toolsets.py: remove resolve_multiple_toolsets (shim-only, restored by
34abf954bd); tests/test_toolsets.py pins the same union behavior via
resolve_toolset over each name.
providers/__init__.py: drop the OMIT_TEMPERATURE re-export (no callers via the
package); ProviderProfile stays because __init__ uses it for annotations —
2 tests repointed to providers.base.
agent/iteration_budget.py: drop the "run_agent re-exports the class"
docstring pointer; 4 tests import IterationBudget from its home.
model_tools.py (arg_coercion names), agent/tool_executor.py, and
hermes_cli/cli_session_mixin.py repoints landed via a sibling commit on this
shared worktree.
Callers repointed: gateway/run.py, hermes_cli/cli_chat_turn_mixin.py,
hermes_cli/cli_tui_mixin.py, tui_gateway/session_workdir.py,
agent/transports/codex.py (one-line imports) + comment pointers in
tools/file_state.py, tools/schema_sanitizer.py, scripts/tool_search_livetest.py.
Tests: patch("run_agent.X") / monkeypatch.setattr(run_agent, "X") /
`from run_agent import X` -> defining module across 99 test files.
411 lines
15 KiB
Python
411 lines
15 KiB
Python
"""Tests for _check_compression_model_feasibility() — warns when the
|
|
auxiliary compression model's context is smaller than the main model's
|
|
compression threshold.
|
|
|
|
Two-phase design:
|
|
1. __init__ → runs the check, prints via _vprint (CLI), stores warning
|
|
2. run_conversation (first call) → replays stored warning through
|
|
status_callback (gateway platforms)
|
|
"""
|
|
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
import pytest
|
|
|
|
from run_agent import AIAgent
|
|
from agent.context_compressor import ContextCompressor
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _stable_aux_provider_config():
|
|
"""Keep feasibility tests independent from the developer's config.yaml."""
|
|
with patch(
|
|
"agent.auxiliary_client._resolve_task_provider_model",
|
|
return_value=("auto", None, None, None, None),
|
|
):
|
|
yield
|
|
|
|
|
|
def _make_agent(
|
|
*,
|
|
compression_enabled: bool = True,
|
|
threshold_percent: float = 0.50,
|
|
main_context: int = 200_000,
|
|
) -> AIAgent:
|
|
"""Build a minimal AIAgent with a compressor, skipping __init__."""
|
|
agent = AIAgent.__new__(AIAgent)
|
|
agent.model = "test-main-model"
|
|
agent.provider = "openrouter"
|
|
agent.base_url = "https://openrouter.ai/api/v1"
|
|
agent.api_key = "sk-test"
|
|
agent.api_mode = "chat_completions"
|
|
agent.quiet_mode = True
|
|
agent.log_prefix = ""
|
|
agent.compression_enabled = compression_enabled
|
|
agent._print_fn = None
|
|
agent.suppress_status_output = False
|
|
agent._stream_consumers = []
|
|
agent._executing_tools = False
|
|
agent._mute_post_response = False
|
|
agent.status_callback = None
|
|
agent.tool_progress_callback = None
|
|
agent._compression_warning = None
|
|
agent._aux_compression_context_length_config = None
|
|
agent._custom_providers = []
|
|
agent.tools = []
|
|
|
|
compressor = MagicMock(spec=ContextCompressor)
|
|
compressor.context_length = main_context
|
|
compressor.threshold_tokens = int(main_context * threshold_percent)
|
|
compressor.summary_target_ratio = 0.20
|
|
compressor.tail_token_budget = int(
|
|
compressor.threshold_tokens * compressor.summary_target_ratio
|
|
)
|
|
agent.context_compressor = compressor
|
|
|
|
return agent
|
|
|
|
|
|
# ── Core warning logic ──────────────────────────────────────────────
|
|
|
|
|
|
@patch("agent.model_metadata.get_model_context_length", return_value=80_000)
|
|
@patch("agent.auxiliary_client.get_text_auxiliary_client")
|
|
def test_auto_corrects_threshold_when_aux_context_below_threshold(mock_get_client, mock_ctx_len):
|
|
"""Auto-correction: aux >= 64K floor but < threshold → lower threshold
|
|
to aux_context so compression still works this session."""
|
|
agent = _make_agent(main_context=200_000, threshold_percent=0.50)
|
|
# threshold = 100,000 — aux has 80,000 (above 64K floor, below threshold)
|
|
mock_client = MagicMock()
|
|
mock_client.base_url = "https://openrouter.ai/api/v1"
|
|
mock_client.api_key = "sk-aux"
|
|
mock_get_client.return_value = (mock_client, "google/gemini-3-flash-preview")
|
|
|
|
messages = []
|
|
agent._emit_status = lambda msg: messages.append(msg)
|
|
|
|
agent._check_compression_model_feasibility()
|
|
|
|
assert len(messages) == 1
|
|
assert "Compression model" in messages[0]
|
|
assert "80,000" in messages[0] # aux context
|
|
assert "100,000" in messages[0] # old threshold
|
|
assert "Auto-lowered" in messages[0]
|
|
# Actionable persistence guidance included
|
|
assert "config.yaml" in messages[0]
|
|
assert "auxiliary:" in messages[0]
|
|
assert "compression:" in messages[0]
|
|
# 200K main is under the 512K small-context limit and 80K/200K = 40% sits
|
|
# below the 75% floor — a `threshold:` suggestion would be raised back to
|
|
# 75% and ignored (#67422), so the message must not offer one and must
|
|
# explain the recomputed trigger instead (0.75 * 200K = 150K).
|
|
assert "threshold:" not in messages[0]
|
|
assert "150,000" in messages[0]
|
|
# Warning stored for gateway replay
|
|
assert agent._compression_warning is not None
|
|
# Threshold on the live compressor was actually lowered to aux_context.
|
|
assert agent.context_compressor.threshold_tokens == 80_000
|
|
# Every threshold-derived budget must move with it. Keeping the original
|
|
# 20K tail here would protect 25% of the lowered threshold instead of the
|
|
# configured 20%, and larger real-world mismatches can make the tail's 1.5x
|
|
# soft ceiling wider than the entire compression trigger.
|
|
assert agent.context_compressor.tail_token_budget == 16_000
|
|
|
|
|
|
@patch("agent.model_metadata.get_model_context_length", return_value=32_768)
|
|
@patch("agent.auxiliary_client.get_text_auxiliary_client")
|
|
def test_rejects_aux_below_minimum_context(mock_get_client, mock_ctx_len):
|
|
"""Hard floor: aux context < MINIMUM_CONTEXT_LENGTH (64K) → session
|
|
refuses to start (ValueError), mirroring the main-model rejection."""
|
|
agent = _make_agent(main_context=200_000, threshold_percent=0.50)
|
|
mock_client = MagicMock()
|
|
mock_client.base_url = "https://openrouter.ai/api/v1"
|
|
mock_client.api_key = "sk-aux"
|
|
mock_get_client.return_value = (mock_client, "tiny-aux-model")
|
|
|
|
agent._emit_status = lambda msg: None
|
|
|
|
with pytest.raises(ValueError) as exc_info:
|
|
agent._check_compression_model_feasibility()
|
|
|
|
err = str(exc_info.value)
|
|
assert "tiny-aux-model" in err
|
|
assert "32,768" in err
|
|
assert "64,000" in err
|
|
assert "below the minimum" in err
|
|
|
|
|
|
|
|
|
|
def test_feasibility_check_passes_live_main_runtime():
|
|
"""Compression feasibility should probe using the live session runtime."""
|
|
agent = _make_agent(main_context=200_000, threshold_percent=0.50)
|
|
agent.model = "gpt-5.4"
|
|
agent.provider = "openai-codex"
|
|
agent.base_url = "https://chatgpt.com/backend-api/codex"
|
|
agent.api_key = "codex-token"
|
|
agent.api_mode = "codex_responses"
|
|
|
|
mock_client = MagicMock()
|
|
mock_client.base_url = "https://chatgpt.com/backend-api/codex"
|
|
mock_client.api_key = "codex-token"
|
|
|
|
with patch("agent.auxiliary_client.get_text_auxiliary_client", return_value=(mock_client, "gpt-5.4")) as mock_get_client, \
|
|
patch("agent.model_metadata.get_model_context_length", return_value=200_000):
|
|
agent._emit_status = lambda msg: None
|
|
agent._check_compression_model_feasibility()
|
|
|
|
mock_get_client.assert_called_once_with(
|
|
"compression",
|
|
main_runtime={
|
|
"model": "gpt-5.4",
|
|
"provider": "openai-codex",
|
|
"base_url": "https://chatgpt.com/backend-api/codex",
|
|
"api_key": "codex-token",
|
|
"api_mode": "codex_responses",
|
|
"auth_mode": "",
|
|
},
|
|
)
|
|
|
|
|
|
@patch("agent.model_metadata.get_model_context_length", return_value=1_000_000)
|
|
@patch("agent.auxiliary_client.get_text_auxiliary_client")
|
|
def test_feasibility_check_passes_config_context_length(mock_get_client, mock_ctx_len):
|
|
"""auxiliary.compression.context_length from config is forwarded to
|
|
get_model_context_length so custom endpoints that lack /models still
|
|
report the correct context window (fixes #8499)."""
|
|
agent = _make_agent(main_context=200_000, threshold_percent=0.85)
|
|
agent._aux_compression_context_length_config = 1_000_000
|
|
mock_client = MagicMock()
|
|
mock_client.base_url = "http://custom-endpoint:8080/v1"
|
|
mock_client.api_key = "sk-custom"
|
|
mock_get_client.return_value = (mock_client, "custom/big-model")
|
|
|
|
agent._emit_status = lambda msg: None
|
|
agent._check_compression_model_feasibility()
|
|
|
|
mock_ctx_len.assert_called_once_with(
|
|
"custom/big-model",
|
|
base_url="http://custom-endpoint:8080/v1",
|
|
api_key="sk-custom",
|
|
config_context_length=1_000_000,
|
|
provider="openrouter",
|
|
custom_providers=[],
|
|
)
|
|
|
|
|
|
|
|
|
|
def test_init_feasibility_check_uses_aux_context_override_from_config():
|
|
"""Lazy feasibility check should cache and forward auxiliary.compression.context_length.
|
|
|
|
NB: feasibility check is deferred from AIAgent.__init__ to the first
|
|
actual compression attempt (saves ~400ms cold startup on short sessions
|
|
that never trigger compression). The test drives the check explicitly
|
|
via ``agent._check_compression_model_feasibility()`` to assert the
|
|
config-override threading.
|
|
"""
|
|
|
|
class _StubCompressor:
|
|
def __init__(self, *args, **kwargs):
|
|
self.context_length = 200_000
|
|
self.threshold_tokens = 100_000
|
|
self.threshold_percent = 0.50
|
|
|
|
def get_tool_schemas(self):
|
|
return []
|
|
|
|
def on_session_start(self, *args, **kwargs):
|
|
return None
|
|
|
|
cfg = {
|
|
"auxiliary": {
|
|
"compression": {
|
|
"context_length": 1_000_000,
|
|
},
|
|
},
|
|
}
|
|
mock_client = MagicMock()
|
|
mock_client.base_url = "http://custom-endpoint:8080/v1"
|
|
mock_client.api_key = "sk-custom"
|
|
|
|
with (
|
|
patch("hermes_cli.config.load_config", return_value=cfg), patch("hermes_cli.config.load_config_readonly", return_value=cfg),
|
|
patch("model_tools.get_tool_definitions", return_value=[]),
|
|
patch("model_tools.check_toolset_requirements", return_value={}),
|
|
patch("agent.process_bootstrap.OpenAI"),
|
|
patch("agent.agent_init.ContextCompressor", new=_StubCompressor),
|
|
patch("agent.auxiliary_client.get_text_auxiliary_client", return_value=(mock_client, "custom/big-model")),
|
|
patch("agent.model_metadata.get_model_context_length", return_value=1_000_000) as mock_ctx_len,
|
|
):
|
|
agent = AIAgent(
|
|
api_key="test-key-1234567890",
|
|
base_url="https://openrouter.ai/api/v1",
|
|
quiet_mode=True,
|
|
skip_context_files=True,
|
|
skip_memory=True,
|
|
)
|
|
|
|
# Config override is captured eagerly in __init__ (still needed
|
|
# because the threshold-derivation logic at construction time
|
|
# consults it).
|
|
assert agent._aux_compression_context_length_config == 1_000_000
|
|
|
|
# The expensive feasibility probe is deferred. Drive it manually
|
|
# to validate the call shape still forwards the override correctly.
|
|
agent._check_compression_model_feasibility()
|
|
|
|
mock_ctx_len.assert_called_once_with(
|
|
"custom/big-model",
|
|
base_url="http://custom-endpoint:8080/v1",
|
|
api_key="sk-custom",
|
|
config_context_length=1_000_000,
|
|
provider="",
|
|
custom_providers=[],
|
|
)
|
|
|
|
|
|
@patch("agent.auxiliary_client.get_text_auxiliary_client")
|
|
def test_warns_when_no_auxiliary_provider(mock_get_client):
|
|
"""Warning emitted when no auxiliary provider is configured."""
|
|
agent = _make_agent()
|
|
mock_get_client.return_value = (None, None)
|
|
|
|
messages = []
|
|
agent._emit_status = lambda msg: messages.append(msg)
|
|
|
|
agent._check_compression_model_feasibility()
|
|
|
|
assert len(messages) == 1
|
|
assert "No auxiliary LLM provider" in messages[0]
|
|
assert agent._compression_warning is not None
|
|
|
|
|
|
def test_no_unavailable_warning_when_configured_fallback_chain_resolves():
|
|
"""Primary compression provider can be down if configured fallback works."""
|
|
agent = _make_agent(main_context=200_000, threshold_percent=0.50)
|
|
fallback_client = MagicMock()
|
|
fallback_client.base_url = "https://chatgpt.com/backend-api/codex"
|
|
fallback_client.api_key = "codex-oauth-token"
|
|
|
|
messages = []
|
|
agent._emit_status = lambda msg: messages.append(msg)
|
|
|
|
with patch(
|
|
"agent.auxiliary_client._resolve_task_provider_model",
|
|
return_value=("ollama-cloud", "deepseek-v4-flash:cloud", None, None, None),
|
|
), patch(
|
|
"agent.auxiliary_client.get_text_auxiliary_client",
|
|
return_value=(None, None),
|
|
), patch(
|
|
"agent.auxiliary_client._try_configured_fallback_for_unavailable_client",
|
|
return_value=(fallback_client, "gpt-5.4-mini", "fallback_chain[0](openai-codex)"),
|
|
) as mock_fallback, patch(
|
|
"agent.model_metadata.get_model_context_length",
|
|
return_value=200_000,
|
|
) as mock_ctx_len:
|
|
agent._check_compression_model_feasibility()
|
|
|
|
assert messages == []
|
|
assert agent._compression_warning is None
|
|
mock_fallback.assert_called_once_with("compression", "ollama-cloud")
|
|
mock_ctx_len.assert_called_once()
|
|
assert mock_ctx_len.call_args.args == ("gpt-5.4-mini",)
|
|
assert mock_ctx_len.call_args.kwargs["provider"] == "openai-codex"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ── Two-phase: __init__ + run_conversation replay ───────────────────
|
|
|
|
|
|
@patch("agent.model_metadata.get_model_context_length", return_value=80_000)
|
|
@patch("agent.auxiliary_client.get_text_auxiliary_client")
|
|
def test_warning_stored_for_gateway_replay(mock_get_client, mock_ctx_len):
|
|
"""__init__ stores the warning; _replay sends it through status_callback."""
|
|
agent = _make_agent(main_context=200_000, threshold_percent=0.50)
|
|
mock_client = MagicMock()
|
|
mock_client.base_url = "https://openrouter.ai/api/v1"
|
|
mock_client.api_key = "sk-aux"
|
|
mock_get_client.return_value = (mock_client, "google/gemini-3-flash-preview")
|
|
|
|
# Phase 1: __init__ — _emit_status prints (CLI) but callback is None
|
|
vprint_messages = []
|
|
agent._emit_status = lambda msg: vprint_messages.append(msg)
|
|
agent._check_compression_model_feasibility()
|
|
|
|
assert len(vprint_messages) == 1 # CLI got it
|
|
assert agent._compression_warning is not None # stored for replay
|
|
|
|
# Phase 2: gateway wires callback post-init, then run_conversation replays
|
|
callback_events = []
|
|
agent.status_callback = lambda ev, msg: callback_events.append((ev, msg))
|
|
agent._replay_compression_warning()
|
|
|
|
assert any(
|
|
ev == "lifecycle" and "Auto-lowered" in msg
|
|
for ev, msg in callback_events
|
|
)
|
|
|
|
|
|
@patch("agent.model_metadata.get_model_context_length", return_value=200_000)
|
|
@patch("agent.auxiliary_client.get_text_auxiliary_client")
|
|
def test_no_replay_when_no_warning(mock_get_client, mock_ctx_len):
|
|
"""_replay_compression_warning is a no-op when there's no stored warning."""
|
|
agent = _make_agent(main_context=200_000, threshold_percent=0.50)
|
|
mock_client = MagicMock()
|
|
mock_client.base_url = "https://openrouter.ai/api/v1"
|
|
mock_client.api_key = "sk-aux"
|
|
mock_get_client.return_value = (mock_client, "big-model")
|
|
|
|
agent._emit_status = lambda msg: None
|
|
agent._check_compression_model_feasibility()
|
|
|
|
assert agent._compression_warning is None
|
|
|
|
callback_events = []
|
|
agent.status_callback = lambda ev, msg: callback_events.append((ev, msg))
|
|
agent._replay_compression_warning()
|
|
|
|
assert len(callback_events) == 0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ── #67422: threshold suggestion must survive the small-context floor ────────
|
|
|
|
|
|
|
|
|
|
@patch("agent.model_metadata.get_model_context_length", return_value=300_000)
|
|
@patch("agent.auxiliary_client.get_text_auxiliary_client")
|
|
def test_threshold_suggestion_kept_for_large_context_main(mock_get_client, mock_ctx_len):
|
|
"""Main window >= 512K has no floor — any suggestion is honored, so the
|
|
`threshold:` option stays even below 75%."""
|
|
agent = _make_agent(main_context=1_000_000, threshold_percent=0.50)
|
|
# threshold = 500,000 — aux has 300,000
|
|
mock_client = MagicMock()
|
|
mock_client.base_url = "https://openrouter.ai/api/v1"
|
|
mock_client.api_key = "sk-aux"
|
|
mock_get_client.return_value = (mock_client, "google/gemini-3-flash-preview")
|
|
|
|
messages = []
|
|
agent._emit_status = lambda msg: messages.append(msg)
|
|
|
|
agent._check_compression_model_feasibility()
|
|
|
|
assert len(messages) == 1
|
|
assert "threshold: 0.30" in messages[0]
|
|
|
|
|
|
|
|
|