Files
EvoScientist/tests/test_message_budget.py
T
m4 421a664336 feat(runtime)!: complete legacy removal, local snapshot entries, and TTL cleanup
- Remove legacy provider profiles, admin-token auth, /model command,
  model picker widget, and config.yaml LLM fields (design doc section 10)
- Wire CLI/channels/cron and async sub-agents through the local snapshot
  entry; run creation rejects model config outside runtime_snapshot_id
- Add periodic run-snapshot TTL cleanup to the config service lifespan
- Isolate tests from the real config dir and activate the registry where
  run/model paths fail closed in bootstrap

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-07-21 18:10:23 +08:00

440 lines
16 KiB
Python

"""Tests for message-only context budgeting (design doc 6.5, 8.1, 8.3).
Every budget comes from the frozen ``ResolvedModelConfig.budget`` of the
run snapshot — the middleware's own ``snapshot_role`` selects which frozen
configuration sizes the budget (section 6.1 role mapping). Runs without an
explicit ``runtime_snapshot_id`` lazily create a local snapshot bound to
their thread (section 8.1); a bootstrap registry fails closed with
``MODEL_REGISTRY_NOT_READY``. There is no interim 32K/profile fallback.
"""
from __future__ import annotations
from contextlib import contextmanager
from types import SimpleNamespace
from unittest.mock import MagicMock, patch
import pytest
from langchain_core.messages import AIMessage, HumanMessage, ToolMessage
from langchain_ollama import ChatOllama
from EvoScientist.middleware.message_budget import (
_snapshot_message_budget,
count_message_text_tokens,
create_message_budget_middleware,
)
from EvoScientist.model_registry.errors import (
MODEL_CONFIG_OUTSIDE_SNAPSHOT,
MODEL_REGISTRY_NOT_READY,
SNAPSHOT_NOT_FOUND,
ModelRegistryError,
)
from EvoScientist.model_registry.runtime import SnapshotRuntime
from EvoScientist.model_registry.schemas import CredentialWrite, RegistryV4
from EvoScientist.model_registry.store import ModelRuntimeStore
from tests.registry_fixtures import (
ZHIPU_SECRET,
_verify_model,
make_active_store,
make_snapshot,
ollama_provider_payload,
zhipu_provider_payload,
)
@contextmanager
def _patched_config(configurable: dict | None):
"""Patch ``langgraph.config.get_config`` to expose ``configurable``."""
import langgraph.config as _lg_cfg
if configurable is None:
with patch.object(
_lg_cfg,
"get_config",
side_effect=RuntimeError("Called get_config outside of a runnable context"),
):
yield
else:
with patch.object(
_lg_cfg,
"get_config",
return_value={"configurable": configurable},
):
yield
@pytest.fixture
def store(tmp_path):
return make_active_store(tmp_path / "model-runtime")
@pytest.fixture
def runtime(store):
return SnapshotRuntime(store)
def _request(messages, tools=()):
return SimpleNamespace(messages=messages, tools=list(tools))
def _configurable_for(snapshot, **overrides):
configurable = {
"runtime_snapshot_id": snapshot.snapshot_id,
"thread_id": snapshot.thread_id,
}
configurable.update(overrides)
return configurable
def _make_split_limit_store(config_dir):
"""Active store whose auxiliary model has a smaller input limit.
primary (zhipu-glm/glm-5.2): context 1048576 → resolved_input_limit 1015808
auxiliary (local-ollama/qwen3): context 65536 → resolved_input_limit 32768
"""
store = ModelRuntimeStore(config_dir=config_dir)
payload = {
"version": 4,
"revision": 1,
"state": "bootstrap",
"defaults": {
"primary": {"provider_id": "zhipu-glm", "model_key": "glm-5.2"},
"auxiliary": {"provider_id": "local-ollama", "model_key": "qwen3"},
},
"providers": [
zhipu_provider_payload(),
ollama_provider_payload(context_window_tokens=65536),
],
}
registry = store.save_registry(
expected_revision=1,
registry=RegistryV4.model_validate(payload),
credential_writes=[
CredentialWrite(credential_id="zhipu-primary", secret_value=ZHIPU_SECRET)
],
)
assert registry.state == "active"
_verify_model(store, registry, "zhipu-glm", "glm-5.2")
_verify_model(store, registry, "local-ollama", "qwen3")
return store
def test_text_counter_excludes_attachment_payloads_and_counts_tool_results():
messages = [
HumanMessage(
content=[
{"type": "text", "text": "abcd"},
{"type": "image", "base64": "x" * 100_000},
{"type": "file", "data": "y" * 100_000},
]
),
ToolMessage(content="wxyz", tool_call_id="tool-1"),
]
assert count_message_text_tokens(messages) == 2
# =============================================================================
# Snapshot-driven budget recomputation (section 6.5)
# =============================================================================
class TestSnapshotMessageBudget:
"""The frozen fixture registry resolves to:
resolved_input_limit = 1048576 - 32768 = 1015808
reserves: system 4096, tools 8192, attachments 4096
"""
def test_base_mode_deducts_only_system_reserve(self, store):
snapshot = make_snapshot(store)
budget = _snapshot_message_budget(
snapshot, "primary", has_tools=False, has_attachments=False
)
message_budget = 1015808 - 4096
assert budget.input_limit == 1015808
assert budget.hard_tokens == int(message_budget * 0.90)
assert budget.soft_tokens == int(budget.hard_tokens * 0.70)
assert budget.keep_tokens == int(budget.hard_tokens * 0.35)
def test_tools_and_attachments_deduct_their_reserves(self, store):
snapshot = make_snapshot(store)
tools_only = _snapshot_message_budget(
snapshot, "primary", has_tools=True, has_attachments=False
)
full = _snapshot_message_budget(
snapshot, "primary", has_tools=True, has_attachments=True
)
assert tools_only.hard_tokens == int((1015808 - 4096 - 8192) * 0.90)
assert full.hard_tokens == int((1015808 - 4096 - 8192 - 4096) * 0.90)
def test_auxiliary_role_uses_auxiliary_frozen_limits(self, tmp_path):
"""T6 review fix: the budget role is threaded through, not hardcoded
to ``primary`` — a scheduler (auxiliary role) must size its budget
from the auxiliary model's frozen configuration."""
store = _make_split_limit_store(tmp_path / "model-runtime")
snapshot = make_snapshot(store)
primary_budget = _snapshot_message_budget(
snapshot, "primary", has_tools=False, has_attachments=False
)
auxiliary_budget = _snapshot_message_budget(
snapshot, "auxiliary", has_tools=False, has_attachments=False
)
assert primary_budget.input_limit == 1015808
assert auxiliary_budget.input_limit == 32768
assert auxiliary_budget.hard_tokens == int((32768 - 4096) * 0.90)
class TestSnapshotModeMiddleware:
def test_budget_uses_frozen_reserves_not_model_profile(self, store, runtime):
"""A misleading compile-time profile must not affect snapshot mode."""
model = MagicMock()
model.profile = {"max_input_tokens": 32_768}
middleware = create_message_budget_middleware(
model, MagicMock(), runtime=runtime
)
snapshot = make_snapshot(store)
with _patched_config(_configurable_for(snapshot)):
budget = middleware._budget_for_request(_request([HumanMessage("hi")]))
assert budget.input_limit == 1015808
assert budget.has_tools is True # conservative construction-time flag
assert budget.has_attachments is False
assert budget.hard_tokens == int((1015808 - 4096 - 8192) * 0.90)
def test_recomputed_per_call_with_current_attachments(self, store, runtime):
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), runtime=runtime
)
snapshot = make_snapshot(store)
plain = _request([HumanMessage(content="plain")])
attached = _request(
[HumanMessage(content=[{"type": "image", "url": "https://example.test/a"}])]
)
with _patched_config(_configurable_for(snapshot)):
first = middleware._budget_for_request(plain)
second = middleware._budget_for_request(attached)
third = middleware._budget_for_request(plain)
assert first.has_attachments is False
assert second.has_attachments is True
assert second.hard_tokens == int((1015808 - 4096 - 8192 - 4096) * 0.90)
# No caching across calls: the third call recomputes from scratch.
assert third == first
def test_has_tools_is_conservative_not_request_derived(self, store, runtime):
"""Tool mode comes from construction, not the request's bound tools."""
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), has_tools=True, runtime=runtime
)
snapshot = make_snapshot(store)
with _patched_config(_configurable_for(snapshot)):
budget = middleware._budget_for_request(_request([HumanMessage("hi")]))
# The request binds no tools at all, yet the tools reserve applies.
assert budget.has_tools is True
assert budget.hard_tokens == int((1015808 - 4096 - 8192) * 0.90)
def test_has_tools_false_when_agent_has_no_toolset(self, store, runtime):
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), has_tools=False, runtime=runtime
)
snapshot = make_snapshot(store)
with _patched_config(_configurable_for(snapshot)):
budget = middleware._budget_for_request(
_request([HumanMessage("hi")], tools=[{"name": "read_file"}])
)
assert budget.has_tools is False
assert budget.hard_tokens == int((1015808 - 4096) * 0.90)
def test_auxiliary_snapshot_role_sizes_budget_from_auxiliary(self, tmp_path):
"""T6 review fix: ``snapshot_role="auxiliary"`` (scheduler) sizes the
budget from the snapshot's auxiliary frozen configuration."""
store = _make_split_limit_store(tmp_path / "model-runtime")
runtime = SnapshotRuntime(store)
middleware = create_message_budget_middleware(
MagicMock(),
MagicMock(),
has_tools=False,
snapshot_role="auxiliary",
runtime=runtime,
)
snapshot = make_snapshot(store)
with _patched_config(_configurable_for(snapshot)):
budget = middleware._budget_for_request(_request([HumanMessage("hi")]))
assert budget.input_limit == 32768
assert budget.hard_tokens == int((32768 - 4096) * 0.90)
def test_unknown_snapshot_role_rejected(self, runtime):
with pytest.raises(ValueError, match="Unknown model role"):
create_message_budget_middleware(
MagicMock(), MagicMock(), snapshot_role="bogus", runtime=runtime
)
def test_snapshot_binding_is_verified(self, store, runtime):
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), runtime=runtime
)
snapshot = make_snapshot(store)
with (
_patched_config(_configurable_for(snapshot, thread_id="other")),
pytest.raises(ModelRegistryError) as excinfo,
):
middleware._budget_for_request(_request([HumanMessage("hi")]))
assert excinfo.value.code == SNAPSHOT_NOT_FOUND
def test_summarizer_model_uses_summary_role(self, store, runtime):
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), runtime=runtime
)
snapshot = make_snapshot(store)
with _patched_config(_configurable_for(snapshot)):
model = middleware.model
# summary → snapshot.auxiliary ?? snapshot.primary (section 6.1)
assert isinstance(model, ChatOllama)
assert model.model == "qwen3"
def test_summarizer_model_cached_per_snapshot(self, store, runtime):
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), runtime=runtime
)
snapshot = make_snapshot(store)
with _patched_config(_configurable_for(snapshot)):
assert middleware.model is middleware.model
# =============================================================================
# Local snapshot entry (section 8.1): no explicit snapshot ID
# =============================================================================
class TestLocalSnapshotEntry:
def test_missing_snapshot_lazily_creates_one_bound_to_the_thread(
self, store, runtime
):
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), has_tools=False, runtime=runtime
)
with _patched_config({"thread_id": "cron-thread-1"}):
budget = middleware._budget_for_request(_request([HumanMessage("hi")]))
assert budget.input_limit == 1015808
# The lazy snapshot is persisted, bound to the run's own thread.
row = store.find_active_run_snapshot(
deployment_id="local",
thread_id="cron-thread-1",
run_request_id="auto:cron-thread-1",
)
assert row is not None
def test_lazy_creation_is_shared_across_middleware_calls(self, store, runtime):
"""The same thread converges on one snapshot (idempotent create)."""
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), runtime=runtime
)
with _patched_config({"thread_id": "cron-thread-2"}):
first = middleware._snapshot()
middleware._budget_for_request(_request([HumanMessage("hi")]))
second = middleware._snapshot()
assert first.snapshot_id == second.snapshot_id
def test_missing_snapshot_and_thread_id_fails_closed(self, runtime):
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), runtime=runtime
)
with (
_patched_config({}),
pytest.raises(ModelRegistryError) as excinfo,
):
middleware._budget_for_request(_request([HumanMessage("hi")]))
assert excinfo.value.code == MODEL_CONFIG_OUTSIDE_SNAPSHOT
def test_bootstrap_registry_fails_with_not_ready(self, tmp_path):
"""No 32K fallback: bootstrap registry → MODEL_REGISTRY_NOT_READY."""
bootstrap_runtime = SnapshotRuntime(
ModelRuntimeStore(config_dir=tmp_path / "model-runtime")
)
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), runtime=bootstrap_runtime
)
with (
_patched_config({"thread_id": "cli-thread"}),
pytest.raises(ModelRegistryError) as excinfo,
):
middleware._budget_for_request(_request([HumanMessage("hi")]))
assert excinfo.value.code == MODEL_REGISTRY_NOT_READY
# =============================================================================
# Cutoff behavior (budget-driven summarization trigger)
# =============================================================================
def test_budget_middleware_uses_message_threshold_and_safe_tool_cutoff(tmp_path):
# A small context window keeps the trigger thresholds reachable in a test.
store = make_active_store(
tmp_path / "model-runtime-small",
context_window_tokens=32768,
max_output_tokens=8192,
min_effective_input_tokens=1024,
fixed_system_reserve_tokens=1024,
fixed_tools_reserve_tokens=2048,
fixed_attachments_reserve_tokens=1024,
)
runtime = SnapshotRuntime(store)
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), has_tools=False, runtime=runtime
)
snapshot = make_snapshot(store)
messages = [
HumanMessage(content="a" * 30_000),
AIMessage(
content="", tool_calls=[{"name": "read_file", "args": {}, "id": "tool-1"}]
),
ToolMessage(content="b" * 30_000, tool_call_id="tool-1"),
HumanMessage(content="c" * 30_000),
]
total = count_message_text_tokens(messages)
from EvoScientist.middleware.message_budget import _ACTIVE_BUDGET
with _patched_config(_configurable_for(snapshot)):
# Simulate the per-call budget activation that wrap_model_call sets.
token = _ACTIVE_BUDGET.set(middleware._budget_for_request(_request(messages)))
try:
assert middleware._should_summarize(messages, total) is True
cutoff = middleware._determine_cutoff_index(messages)
finally:
_ACTIVE_BUDGET.reset(token)
assert cutoff in {1, 3}
# A cutoff never leaves the tool response without its matching AI tool call.
if cutoff == 1:
assert isinstance(messages[cutoff], AIMessage)
def test_active_budget_outside_model_call_fails_loudly(runtime):
middleware = create_message_budget_middleware(
MagicMock(), MagicMock(), runtime=runtime
)
with pytest.raises(RuntimeError, match="outside a model call"):
middleware._active_budget()