From cb71d5f1b1e4bb4398f2f9aefd1782d5a0c85137 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Mon, 31 Aug 2026 11:51:11 -0700 Subject: [PATCH] fix(agent_init): clamp compressor window to Ollama num_ctx resolved after construction MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit model.ollama_num_ctx is resolved AFTER the context compressor is constructed, so a config that sets only ollama_num_ctx (without model.context_length) ran every request at the smaller served num_ctx while the compressor still targeted the probed GGUF window (e.g. 256K Gemma metadata). The compaction trigger then sat several times above the window the server actually serves and never fired — reproducing the original #57275 'blows past the limit' symptom on current main. Live repro (real imports, temp HERMES_HOME, config = {model: {ollama_num_ctx: 65536}}, probed window 262144): before: _ollama_num_ctx=65536, compressor.context_length=262144, threshold_tokens=196608 (300% of the served window) after: compressor.context_length=65536, threshold below the window The clamp is one-directional (a num_ctx larger than the resolved window never inflates the compressor) and reuses update_model() so every threshold-derived budget recalibrates. Overlaps #60103 (silent-clamp dead zone) — this is the init-order half. Reported by @Artemonim in #57275 (residual claim 3). --- agent/agent_init.py | 30 +++++++++++++++++++ tests/test_ollama_num_ctx.py | 57 ++++++++++++++++++++++++++++++++++++ 2 files changed, 87 insertions(+) diff --git a/agent/agent_init.py b/agent/agent_init.py index ea2046763e..29c0390580 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -3040,6 +3040,36 @@ def init_agent( "Ollama num_ctx: will request %d tokens (model max from /api/show)", agent._ollama_num_ctx, ) + # ── Recalibrate the compressor to the served window (#57275 claim 3) ── + # The compressor was constructed ABOVE this block from the probed model + # window (GGUF metadata can advertise 256K+), but every request below + # runs at num_ctx. A config that sets only model.ollama_num_ctx (without + # model.context_length) previously left the compressor targeting the + # probed window while the server truncated/rejected at num_ctx — the + # compaction trigger could sit several times ABOVE the real served + # window and never fire. Clamp the compressor's window to the effective + # num_ctx so threshold math operates on the context the server actually + # serves. (Overlaps #60103's silent-clamp dead zone; this is the + # init-order half.) + _cc_window = getattr(agent.context_compressor, "context_length", 0) or 0 + if ( + agent._ollama_num_ctx + and agent._ollama_num_ctx > 0 + and _cc_window + and agent._ollama_num_ctx < _cc_window + ): + _ra().logger.info( + "Compressor window clamped to Ollama num_ctx: %d -> %d", + _cc_window, agent._ollama_num_ctx, + ) + agent.context_compressor.update_model( + model=agent.model, + context_length=agent._ollama_num_ctx, + base_url=agent.base_url, + api_key=getattr(agent, "api_key", ""), + provider=agent.provider, + api_mode=agent.api_mode, + ) # Codex gpt-5.x autoraise notice: show at most once per profile/config # state. Without the persisted marker the notice re-fires on every agent diff --git a/tests/test_ollama_num_ctx.py b/tests/test_ollama_num_ctx.py index 3097fbbf4e..2ad57201c2 100644 --- a/tests/test_ollama_num_ctx.py +++ b/tests/test_ollama_num_ctx.py @@ -130,3 +130,60 @@ class TestQueryOllamaSupportsVision: with patch("agent.model_metadata.detect_local_server_type", return_value="vllm"): result = query_ollama_supports_vision("llava", "http://localhost:8000/v1") assert result is None + + +# ═══════════════════════════════════════════════════════════════════════ +# Level 3: init-order — compressor window must clamp to effective num_ctx +# (#57275 residual claim 3 / #60103 init-order half) +# ═══════════════════════════════════════════════════════════════════════ + + +class TestCompressorClampsToNumCtx: + """A config setting ONLY model.ollama_num_ctx (no model.context_length) + must not leave the compressor targeting the probed model window while + requests run at the smaller served num_ctx.""" + + def _build_agent(self, cfg, probed_ctx): + import agent.context_compressor as cc_mod + with ( + patch("run_agent.get_tool_definitions", return_value=[]), + patch("run_agent.check_toolset_requirements", return_value={}), + patch("run_agent.OpenAI"), + patch("hermes_cli.config.load_config", return_value=cfg), + patch("hermes_cli.config.load_config_readonly", return_value=cfg), + patch( + "agent.model_metadata.get_model_context_length", + return_value=probed_ctx, + ), + patch.object( + cc_mod, "get_model_context_length", return_value=probed_ctx, + ), + ): + from run_agent import AIAgent + return AIAgent( + model="gemma3:27b", + api_key="ollama", + base_url="http://localhost:11434/v1", + quiet_mode=True, + skip_context_files=True, + skip_memory=True, + ) + + def test_num_ctx_only_config_clamps_compressor_window(self): + agent = self._build_agent( + {"agent": {}, "model": {"ollama_num_ctx": 65536}}, probed_ctx=262144 + ) + assert agent._ollama_num_ctx == 65536 + # The compressor must target the served window, not the probed 256K — + # otherwise its trigger sits far above what the server accepts and + # compaction never fires (#57275 claim 3). + assert agent.context_compressor.context_length == 65536 + assert agent.context_compressor.threshold_tokens < 65536 + + def test_larger_num_ctx_does_not_inflate_compressor_window(self): + agent = self._build_agent( + {"agent": {}, "model": {"ollama_num_ctx": 131072}}, probed_ctx=65536 + ) + # num_ctx above the resolved window must not RAISE the compressor + # window: the clamp is one-directional. + assert agent.context_compressor.context_length == 65536