From 167448edca5919c4afacb95f10840f8e56f6d647 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 9 Sep 2026 04:56:27 -0700 Subject: [PATCH] perf(gateway): warm the Python toolchain probe in the boot warm-up MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The gateway's startup warm-up (_warm_turn_machinery_sync) primed the run_agent import graph, tool schemas and context files, but left the process-local Python toolchain probe to the first AIAgent. That agent starts the probe worker in its constructor and its first system-prompt build then blocks in get_environment_probe_line() on the python/pip subprocesses (~250 ms measured) before the first model request — with local models too, so provider metadata warm-up (#105999) does not cover it. Call the same resolver from the warm-up: it owns the agent.environment_probe opt-out check here, and the resolver itself owns remote-backend omission, the single worker, its cache and the bounded fail-open wait, so the first turn simply finds the line cached. No agent is constructed and no session prompt is frozen at startup; a disabled probe or a remote terminal backend leaves the host untouched. Slim redo of #106075 by @francip: same mechanism, folded into the existing sync warm-up instead of a new module + asyncio.gather + second suppress block; the two behavior tests are kept, the concurrency tests dropped. Co-authored-by: Franci Penov <49422+francip@users.noreply.github.com> --- gateway/run.py | 11 ++++- .../gateway/test_startup_environment_probe.py | 47 +++++++++++++++++++ 2 files changed, 57 insertions(+), 1 deletion(-) create mode 100644 tests/gateway/test_startup_environment_probe.py diff --git a/gateway/run.py b/gateway/run.py index c907ba4187..40215b808c 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -953,7 +953,7 @@ def _warm_turn_machinery_sync() -> int: """Synchronously initialize first-turn prerequisites (executor thread); returns the schema count. Covers the lazy init seen in skeleton turns: ``run_agent`` import graph, tool schemas (+ ``check_fn`` - TTL cache), context files.""" + TTL cache), context files, the local Python toolchain probe (#106064).""" import run_agent # noqa: F401 # heavy import graph, cached in sys.modules import model_tools @@ -964,6 +964,15 @@ def _warm_turn_machinery_sync() -> int: build_context_files_prompt() except Exception: logger.debug("context-file warm-up failed (non-fatal)", exc_info=True) + from hermes_cli.config import load_config_readonly + + agent_cfg = load_config_readonly().get("agent") + if not isinstance(agent_cfg, dict) or agent_cfg.get("environment_probe", True): + # The resolver owns remote-backend omission, the single worker, its cache and the bounded + # wait; calling it here is what the first prompt build would otherwise do on the hot path. + from tools.env_probe import get_environment_probe_line + + get_environment_probe_line() return len(tool_defs) diff --git a/tests/gateway/test_startup_environment_probe.py b/tests/gateway/test_startup_environment_probe.py new file mode 100644 index 0000000000..4c0b18b9e5 --- /dev/null +++ b/tests/gateway/test_startup_environment_probe.py @@ -0,0 +1,47 @@ +"""The gateway boot warm-up primes the local toolchain probe the first prompt build reads (#106064).""" + +import pytest + +import model_tools +from gateway import run as gateway_run +from tools import env_probe +from tools.terminal_scope import reset_terminal_scope, set_terminal_scope + + +@pytest.fixture(autouse=True) +def _probe_cache(monkeypatch): + monkeypatch.setattr(model_tools, "get_tool_definitions", lambda **_: []) + env_probe._reset_cache_for_tests() + yield + env_probe._reset_cache_for_tests() + + +def _warm(tmp_path, monkeypatch, agent_section: dict, backend: str) -> None: + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + (tmp_path / "config.yaml").write_text(f"agent: {agent_section!r}\n", encoding="utf-8") + token = set_terminal_scope({"TERMINAL_ENV": backend}) + try: + gateway_run._warm_turn_machinery_sync() + finally: + reset_terminal_scope(token) + + +def test_warmup_leaves_probe_cached_for_first_prompt(tmp_path, monkeypatch): + calls = [] + monkeypatch.setattr(env_probe, "_build_probe_line", lambda: calls.append(1) or "Python toolchain: fixture.") + + _warm(tmp_path, monkeypatch, {}, "local") + + assert env_probe._PROBE_DONE.is_set() + assert env_probe.get_environment_probe_line() == "Python toolchain: fixture." + assert calls == [1] # single worker; the first turn reuses the cache + + +@pytest.mark.parametrize("agent_section,backend", [({"environment_probe": False}, "local"), ({}, "ssh")]) +def test_warmup_skips_probe_when_disabled_or_remote(tmp_path, monkeypatch, agent_section, backend): + monkeypatch.setattr(env_probe, "_build_probe_line", lambda: pytest.fail("host inspected")) + + _warm(tmp_path, monkeypatch, agent_section, backend) + + assert not env_probe._PROBE_DONE.is_set() + assert env_probe._PROBE_THREAD is None