From 27a4023791acbee8355c89766d83baa9db28688f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 3 Sep 2026 23:55:56 -0700 Subject: [PATCH] docs+evals: AGENTS.md for the facade/siblings layout; codebase-navigability benchmark harness AGENTS.md: Project Structure tree reflects the decomposition (run_agent 1.5k not 12k, cli 4.6k not 11k, hermes_state facade + 21 siblings, web_routers/, evals/, test counts); new "Facade + siblings layout" section with the sibling families table and the rules that follow (find by topic, patch where production reads, compat pointers off limits, don't recreate god files); AIAgent/Agent Loop point at agent/turn_*.py and conversation_loop; CLI dispatch documents _SLASH_DISPATCH + the _handle__command convention and "Adding a Slash Command" no longer tells you to add an elif (there is no ladder to add to on either surface). evals/codebase_navigability/: what the codebase costs an agent, not the CPU. bench.py ~19k real "locate X" tasks from tests/ imports; tokens (tiktoken o200k) of the defining file vs the symbol, context-window fit, read windows, siblings, symbol CC lookup_sim.py paired grep+read simulation over 4k common symbols; tool calls + tokens returned static_metrics.py LOC split, size distributions, elif/nesting, radon CC/MI, import graph + SCC cycles runtime_bench.py fresh-interpreter import/CLI/hot-path/collection timings with tree-purity assertion tests/evals/test_codebase_navigability.py pins the resolver's facade/sibling behaviour. --- AGENTS.md | 93 ++++++-- evals/__init__.py | 0 evals/codebase_navigability/README.md | 64 +++++ evals/codebase_navigability/__init__.py | 0 evals/codebase_navigability/bench.py | 223 ++++++++++++++++++ evals/codebase_navigability/lookup_sim.py | 140 +++++++++++ evals/codebase_navigability/runtime_bench.py | 152 ++++++++++++ evals/codebase_navigability/static_metrics.py | 157 ++++++++++++ tests/evals/__init__.py | 0 tests/evals/test_codebase_navigability.py | 55 +++++ 10 files changed, 862 insertions(+), 22 deletions(-) create mode 100644 evals/__init__.py create mode 100644 evals/codebase_navigability/README.md create mode 100644 evals/codebase_navigability/__init__.py create mode 100644 evals/codebase_navigability/bench.py create mode 100644 evals/codebase_navigability/lookup_sim.py create mode 100644 evals/codebase_navigability/runtime_bench.py create mode 100644 evals/codebase_navigability/static_metrics.py create mode 100644 tests/evals/__init__.py create mode 100644 tests/evals/test_codebase_navigability.py diff --git a/AGENTS.md b/AGENTS.md index e140522e96..e8cb473c6d 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -268,19 +268,20 @@ entry points you'll actually edit. ``` hermes-agent/ -├── run_agent.py # AIAgent class — core conversation loop (~12k LOC) +├── run_agent.py # AIAgent class — public entry (~1.5k LOC); the turn loop lives in agent/turn_*.py ├── model_tools.py # Tool orchestration, discover_builtin_tools(), handle_function_call() ├── toolsets.py # Toolset definitions, _HERMES_CORE_TOOLS list -├── cli.py # HermesCLI class — interactive CLI orchestrator (~11k LOC) -├── hermes_state.py # SessionDB — SQLite session store (FTS5 search) +├── cli.py # HermesCLI class — interactive CLI orchestrator (~4.6k LOC + hermes_cli/cli_*_mixin.py) +├── hermes_state.py # SessionDB facade (~1.4k LOC); implementation in hermes_state_*.py (21 siblings) ├── hermes_constants.py # get_hermes_home(), display_hermes_home() — profile-aware paths ├── hermes_logging.py # setup_logging() — agent.log / errors.log / gateway.log (profile-aware) ├── batch_runner.py # Parallel batch processing -├── agent/ # Agent internals (provider adapters, memory, caching, compression, etc.) +├── agent/ # Agent internals: turn_*.py (loop phases), provider adapters, memory, compression, ... ├── hermes_cli/ # CLI subcommands, setup wizard, plugins loader, skin engine +│ └── web_routers/ # Dashboard FastAPI routers (one file per surface); web_server.py mounts them ├── tools/ # Tool implementations — auto-discovered via tools/registry.py │ └── environments/ # Terminal backends (local, docker, ssh, modal, daytona, singularity) -├── gateway/ # Messaging gateway — run.py + session.py + platforms/ +├── gateway/ # Messaging gateway — run.py facade + run_*.py phases + session*.py + platforms/ │ ├── platforms/ # Adapter per platform (telegram, discord, slack, whatsapp, │ │ # homeassistant, signal, matrix, mattermost, email, sms, │ │ # dingtalk, wecom, weixin, feishu, qqbot, bluebubbles, @@ -300,14 +301,49 @@ hermes-agent/ ├── skills/ # Built-in skills bundled with the repo ├── ui-tui/ # Ink (React) terminal UI — `hermes --tui` │ └── src/ # entry.tsx, app.tsx, gatewayClient.ts + app/components/hooks/lib -├── tui_gateway/ # Python JSON-RPC backend for the TUI +├── tui_gateway/ # Python JSON-RPC backend for the TUI — server.py + methods_*.py ├── acp_adapter/ # ACP server (VS Code / Zed / JetBrains integration) -├── cron/ # Scheduler — jobs.py, scheduler.py -├── scripts/ # run_tests.sh, release.py, auxiliary scripts +├── cron/ # Scheduler — jobs.py, scheduler.py (+ scheduler_*.py) +├── evals/ # Offline benchmarks (codebase_navigability/, compaction/, core_tool_deferral/, ...) +├── scripts/ # run_tests.sh, release.py, check_compat_pointers.py, auxiliary scripts ├── website/ # Docusaurus docs site -└── tests/ # Pytest suite (~17k tests across ~900 files as of May 2026) +└── tests/ # Pytest suite (~39k test functions across ~3.7k files as of Sep 2026) ``` +### Facade + siblings layout (Sep 2026 decomposition) + +Every former god file is now a **facade** plus a family of **siblings** named `_.py` +in the same directory. The facade keeps the public entry points and the names other packages +import; each sibling owns one topic and is where the code actually lives. Families with the most +siblings: + +| facade | siblings | topics | +|---|---|---| +| `hermes_state.py` | 21 `hermes_state_*.py` | schema, fts, search, compression, portability, gateway, errors, ... | +| `gateway/run.py` | 15 `gateway/run_*.py` | startup, adapters, inbound, turn, busy, goals, notifications, shutdown, ... | +| `tools/mcp_tool.py` | 15 `tools/mcp_tool_*.py` | config, discovery, transport, registration, content, errors, ... | +| `hermes_cli/kanban.py` | 14 `hermes_cli/kanban_*.py` | boards, db, db_connect, db_dispatch, db_notify, workspace, ... | +| `hermes_cli/web_server.py` | 13 `web_server_*.py` + 24 `web_routers/*.py` | one router file per dashboard surface | +| `hermes_cli/auth.py` | 12 `auth_*.py` | device_flow, codex, minimax, model_picker, commands, ... | +| `tools/browser_tool.py` | 11 `browser_tool_*.py` | cdp, cloud, install, lifecycle, session, real_profile, vision, ... | +| `cli.py` | 12 `hermes_cli/cli_*_mixin.py` | commands, stream, status_bar, billing, ... — mixed into `HermesCLI` | +| `run_agent.py` | `agent/turn_*.py`, `agent/agent_init.py`, `agent/conversation_loop.py` | turn phases: iteration_prep, api_call, api_error, overflow, truncation, recovery | + +Rules that follow from this layout: + +- **Find code by topic, not by facade.** `grep -rn "def name" /_*.py` lands on the + definition; reading the facade first is the expensive way (see `evals/codebase_navigability/`). +- **Siblings may import each other and late-import the facade** (inside functions) to avoid cycles. + A facade never imports a sibling at module level *and* gets imported by that sibling at module level. +- **Patch where production reads.** Many siblings deliberately do `from import name` inside + the function so tests can `monkeypatch.setattr(facade, "name", ...)`. Before writing a patch target, + check which binding the call site actually reads; a patch on the wrong module passes silently. +- **`scripts/check_compat_pointers.py` runs in CI.** Old import paths kept alive for external plugins + (`PLUGIN-COMPAT` blocks, `COMPAT_MANIFEST.md`, `compat_manifest.json`) are OFF LIMITS to in-tree code + and tests; they are removed on schedule by reverting one commit. Import from the defining module. +- **Don't recreate god files.** A sibling passing ~2,000 lines or a function passing ~300 is a signal + to split again along the same `_` convention. + **User config:** `~/.hermes/config.yaml` (settings), `~/.hermes/.env` (API keys only). **Logs:** `~/.hermes/logs/` — `agent.log` (INFO+), `errors.log` (WARNING+), `gateway.log` when running the gateway. Profile-aware via `get_hermes_home()`. @@ -350,7 +386,13 @@ run_agent.py, cli.py, batch_runner.py, environments/ --- -## AIAgent Class (run_agent.py) +## AIAgent Class (run_agent.py + agent/) + +`run_agent.py` is the public entry: the `AIAgent` class assembled from mixins +(`agent/turn_facade.py`, `agent/client_lifecycle.py`, `agent/stream_delivery.py`, +`agent/session_persistence.py`, `agent/compression_facade.py`, ...). Construction runs +`agent/agent_init.py::init_agent`; the turn itself is `agent/conversation_loop.py::run_conversation`, +which `AIAgent.run_conversation` forwards to after taking the session turn lease. The real `AIAgent.__init__` takes ~60 parameters (credentials, routing, callbacks, session context, budget, credential pool, etc.). The signature below is the @@ -388,8 +430,11 @@ class AIAgent: ### Agent Loop -The core loop is inside `run_conversation()` — entirely synchronous, with -interrupt checks, budget tracking, and a one-turn grace call: +The core loop is `agent/conversation_loop.py::run_conversation()` — entirely synchronous, with +interrupt checks, budget tracking, and a one-turn grace call. Each phase of an iteration is its own +module under `agent/turn_*.py` (`turn_iteration_prep`, `turn_request_assembly`, `turn_api_call`, +`turn_api_error`, `turn_overflow`, `turn_truncation`, `turn_recovery`, `turn_usage`, ...), so a change +to, say, overflow handling touches one ~600-line file. Schematically: ```python while (api_call_count < self.max_iterations and self.iteration_budget.remaining > 0) \ @@ -410,13 +455,17 @@ Reasoning content is stored in `assistant_msg["reasoning"]`. --- -## CLI Architecture (cli.py) +## CLI Architecture (cli.py + hermes_cli/cli_*_mixin.py) +- `cli.py` holds `HermesCLI` (REPL loop, config, slash dispatch); behaviour is split into mixins in + `hermes_cli/cli_commands_mixin.py`, `cli_stream_mixin.py`, `cli_status_bar_mixin.py`, `cli_billing_mixin.py`, ... - **Rich** for banner/panels, **prompt_toolkit** for input with autocomplete - **KawaiiSpinner** (`agent/display.py`) — animated faces during API calls, `┊` activity feed for tool results - `load_cli_config()` in cli.py merges hardcoded defaults + user config YAML - **Skin engine** (`hermes_cli/skin_engine.py`) — data-driven CLI theming; initialized from `display.skin` config key at startup; skins customize banner colors, spinner faces/verbs/wings, tool prefix, response box, branding text -- `process_command()` is a method on `HermesCLI` — dispatches on canonical command name resolved via `resolve_command()` from the central registry +- `process_command()` on `HermesCLI` resolves the canonical name via `resolve_command()` from the central + registry, then dispatches through the `_SLASH_DISPATCH` table (`canonical -> (method name, pass_arg)`), + falling back to a `_handle__command` method by naming convention. There is no `elif` ladder. - Skill slash commands: `agent/skill_commands.py` scans `~/.hermes/skills/`, injects as **user message** (not system prompt) to preserve prompt caching ### Slash Command Registry (`hermes_cli/commands.py`) @@ -438,16 +487,16 @@ All slash commands are defined in a central `COMMAND_REGISTRY` list of `CommandD CommandDef("mycommand", "Description of what it does", "Session", aliases=("mc",), args_hint="[arg]"), ``` -2. Add handler in `HermesCLI.process_command()` in `cli.py`: +2. Add the handler on the CLI. Name it `_handle_mycommand_command(self, cmd_original)` on the + relevant `hermes_cli/cli_*_mixin.py` and it is picked up by convention; or add an explicit entry to + `HermesCLI._SLASH_DISPATCH` in `cli.py` when the method name or arg-passing differs: ```python -elif canonical == "mycommand": - self._handle_mycommand(cmd_original) -``` -3. If the command is available in the gateway, add a handler in `gateway/run.py`: -```python -if canonical == "mycommand": - return await self._handle_mycommand(event) +"mycommand": ("_handle_mycommand", True), # (method name, pass cmd_original?) ``` +3. If the command is available in the gateway, add `_handle_mycommand_command(self, event)` on the + matching `gateway/slash_commands_*.py` mixin and list `"mycommand"` in `_IDLE_COMMANDS` (or + `_PLAIN_COMMANDS` if it must also work mid-run) in `gateway/run_busy.py`. Handlers are looked up + by name through `_command_handler_table`; there is no `if canonical == ...` chain to extend. 4. For persistent settings, use `save_config_value()` in `cli.py` **CommandDef fields:** diff --git a/evals/__init__.py b/evals/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evals/codebase_navigability/README.md b/evals/codebase_navigability/README.md new file mode 100644 index 0000000000..4e43d7bfbc --- /dev/null +++ b/evals/codebase_navigability/README.md @@ -0,0 +1,64 @@ +# Codebase navigability eval + +Measures what a codebase costs an **LLM agent** to work in, as opposed to what it costs the CPU. +Built for the Sep 2026 decomposition (PR #102117) and kept so future refactors are held to the same +numbers. Everything here is offline and deterministic; no model calls. + +## The three questions it answers + +1. **How much code has to be read to see one definition?** (`bench.py`) + Workload = every `from import ` in `tests/`, resolved through re-exports + to the module that *defines* the name. That is ~19k real "locate X" tasks nobody hand-picked. + Per task: tokens of the defining file, tokens of the symbol itself, overhead = file − symbol, + whether the file fits a 32k / 128k context, how many 2,000-line `read_file` windows it spans, + how many unrelated top-level siblings sit in the same file, and the symbol's cyclomatic complexity. + Tokens are real (`tiktoken` `o200k_base`); falls back to bytes/4 if tiktoken is missing and says so. + +2. **What does a careful agent actually pay to look a symbol up?** (`lookup_sim.py`) + Simulates the policy a good model follows with our tools: `grep -n` for the definition (1 call), + `read_file` a 200-line window around the hit (1 call), page forward in 2,000-line windows only while + the definition is still running. Charges tool calls and returned tokens. Same random sample of symbols + that exist on BOTH trees, so it is a paired comparison. + +3. **Shape and runtime.** (`static_metrics.py`, `runtime_bench.py`) + LOC split (code/comment/docstring), file and function size distributions, elif-chain lengths, + nesting depth, radon CC/MI, import graph (edges, fan-in/out, Tarjan SCC cycles); fresh-interpreter + import time / module count / RSS for the entry points, CLI end-to-end, in-process hot paths, pytest + collection, bytecode footprint. `runtime_bench.py` pins `PYTHONPATH` to the tree and asserts no + module resolved from another checkout (an editable install will silently cross-contaminate otherwise). + +## Usage + +```bash +# deps: tiktoken + radon (bench venv or the project venv) +uv pip install tiktoken radon + +# 1 + 2: pass two checkouts (git worktree add is the easy way to get the baseline) +git worktree add /tmp/base origin/main +python evals/codebase_navigability/bench.py /tmp/base base --out out/ +python evals/codebase_navigability/bench.py . head --out out/ +python evals/codebase_navigability/lookup_sim.py /tmp/base . --sample 4000 --out out/ + +# 3 +NAV_OUT=out/ python evals/codebase_navigability/static_metrics.py /tmp/base base +NAV_OUT=out/ python evals/codebase_navigability/static_metrics.py . head +NAV_OUT=out/ python evals/codebase_navigability/runtime_bench.py /tmp/base base 9 +NAV_OUT=out/ python evals/codebase_navigability/runtime_bench.py . head 9 +``` + +`bench.py` and `static_metrics.py` take ~2 min each on a 1M-line tree; `lookup_sim.py` ~10 min for +4,000 symbols (it tokenizes every window it "reads"); `runtime_bench.py` ~4 min per tree at 9 reps. + +## Reading the results honestly + +- `bench.py` measures the **naive** cost (read the whole defining file). It is the number that drops + ~3× when god files are split, and the one that decides whether a file fits a context window at all. +- `lookup_sim.py` measures the **skilled** cost. It barely moves with a file split, because grep + a + 200-line window already dodges file size. What moves it is (a) the definition itself getting shorter + and (b) token *density* of the surrounding code. Stripping comments makes each line denser, so a fixed + 200-line window costs more tokens after a comment-stripping refactor even though the code is smaller. + Both effects are real and pull in opposite directions; report both numbers, not the flattering one. +- Import time goes **up** with a split in Python (per-module overhead), and the import graph's largest + cycle typically grows (intra-file coupling becomes inter-module edges). Neither is hidden by these tools. + +Results for PR #102117 are in the PR description; raw JSON for that run lives in the PR thread. diff --git a/evals/codebase_navigability/__init__.py b/evals/codebase_navigability/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evals/codebase_navigability/bench.py b/evals/codebase_navigability/bench.py new file mode 100644 index 0000000000..307d41872f --- /dev/null +++ b/evals/codebase_navigability/bench.py @@ -0,0 +1,223 @@ +#!/usr/bin/env python3 +"""Agent-navigability benchmark: what does it cost an LLM agent to find and read a symbol? + +Workload (not hand-picked): every ``from import `` in tests/ on the tree under +test, resolved to the module that DEFINES the name on that tree. Each is one "locate and read X" task. + +Per task, three costs are measured in real tokenizer tokens (tiktoken o200k_base, the GPT-4o/o-series +tokenizer; other tokenizers differ by a roughly constant factor): + + file_tokens tokens to read the whole file that defines X (the naive "read_file" cost) + symbol_tokens tokens of X's own definition (irreducible: you must read this) + overhead file_tokens - symbol_tokens (what the file's SIZE costs you beyond the answer) + fits_128k / fits_32k can the defining file be loaded whole into that context? + windows_2k how many 2,000-line read_file windows the file spans (how many reads to scan it) + +Also: for each defining file, the number of OTHER top-level symbols it contains (how much unrelated +code sits next to the answer), and the depth/complexity of the symbol itself. + +Usage: python evals/codebase_navigability/bench.py