docs+evals: AGENTS.md for the facade/siblings layout; codebase-navigability benchmark harness
AGENTS.md: Project Structure tree reflects the decomposition (run_agent 1.5k not 12k, cli 4.6k not 11k,
hermes_state facade + 21 siblings, web_routers/, evals/, test counts); new "Facade + siblings layout"
section with the sibling families table and the rules that follow (find by topic, patch where production
reads, compat pointers off limits, don't recreate god files); AIAgent/Agent Loop point at agent/turn_*.py
and conversation_loop; CLI dispatch documents _SLASH_DISPATCH + the _handle_<name>_command convention and
"Adding a Slash Command" no longer tells you to add an elif (there is no ladder to add to on either surface).
evals/codebase_navigability/: what the codebase costs an agent, not the CPU.
bench.py ~19k real "locate X" tasks from tests/ imports; tokens (tiktoken o200k) of the defining
file vs the symbol, context-window fit, read windows, siblings, symbol CC
lookup_sim.py paired grep+read simulation over 4k common symbols; tool calls + tokens returned
static_metrics.py LOC split, size distributions, elif/nesting, radon CC/MI, import graph + SCC cycles
runtime_bench.py fresh-interpreter import/CLI/hot-path/collection timings with tree-purity assertion
tests/evals/test_codebase_navigability.py pins the resolver's facade/sibling behaviour.
This commit is contained in:
@@ -268,19 +268,20 @@ entry points you'll actually edit.
|
||||
|
||||
```
|
||||
hermes-agent/
|
||||
├── run_agent.py # AIAgent class — core conversation loop (~12k LOC)
|
||||
├── run_agent.py # AIAgent class — public entry (~1.5k LOC); the turn loop lives in agent/turn_*.py
|
||||
├── model_tools.py # Tool orchestration, discover_builtin_tools(), handle_function_call()
|
||||
├── toolsets.py # Toolset definitions, _HERMES_CORE_TOOLS list
|
||||
├── cli.py # HermesCLI class — interactive CLI orchestrator (~11k LOC)
|
||||
├── hermes_state.py # SessionDB — SQLite session store (FTS5 search)
|
||||
├── cli.py # HermesCLI class — interactive CLI orchestrator (~4.6k LOC + hermes_cli/cli_*_mixin.py)
|
||||
├── hermes_state.py # SessionDB facade (~1.4k LOC); implementation in hermes_state_*.py (21 siblings)
|
||||
├── hermes_constants.py # get_hermes_home(), display_hermes_home() — profile-aware paths
|
||||
├── hermes_logging.py # setup_logging() — agent.log / errors.log / gateway.log (profile-aware)
|
||||
├── batch_runner.py # Parallel batch processing
|
||||
├── agent/ # Agent internals (provider adapters, memory, caching, compression, etc.)
|
||||
├── agent/ # Agent internals: turn_*.py (loop phases), provider adapters, memory, compression, ...
|
||||
├── hermes_cli/ # CLI subcommands, setup wizard, plugins loader, skin engine
|
||||
│ └── web_routers/ # Dashboard FastAPI routers (one file per surface); web_server.py mounts them
|
||||
├── tools/ # Tool implementations — auto-discovered via tools/registry.py
|
||||
│ └── environments/ # Terminal backends (local, docker, ssh, modal, daytona, singularity)
|
||||
├── gateway/ # Messaging gateway — run.py + session.py + platforms/
|
||||
├── gateway/ # Messaging gateway — run.py facade + run_*.py phases + session*.py + platforms/
|
||||
│ ├── platforms/ # Adapter per platform (telegram, discord, slack, whatsapp,
|
||||
│ │ # homeassistant, signal, matrix, mattermost, email, sms,
|
||||
│ │ # dingtalk, wecom, weixin, feishu, qqbot, bluebubbles,
|
||||
@@ -300,14 +301,49 @@ hermes-agent/
|
||||
├── skills/ # Built-in skills bundled with the repo
|
||||
├── ui-tui/ # Ink (React) terminal UI — `hermes --tui`
|
||||
│ └── src/ # entry.tsx, app.tsx, gatewayClient.ts + app/components/hooks/lib
|
||||
├── tui_gateway/ # Python JSON-RPC backend for the TUI
|
||||
├── tui_gateway/ # Python JSON-RPC backend for the TUI — server.py + methods_*.py
|
||||
├── acp_adapter/ # ACP server (VS Code / Zed / JetBrains integration)
|
||||
├── cron/ # Scheduler — jobs.py, scheduler.py
|
||||
├── scripts/ # run_tests.sh, release.py, auxiliary scripts
|
||||
├── cron/ # Scheduler — jobs.py, scheduler.py (+ scheduler_*.py)
|
||||
├── evals/ # Offline benchmarks (codebase_navigability/, compaction/, core_tool_deferral/, ...)
|
||||
├── scripts/ # run_tests.sh, release.py, check_compat_pointers.py, auxiliary scripts
|
||||
├── website/ # Docusaurus docs site
|
||||
└── tests/ # Pytest suite (~17k tests across ~900 files as of May 2026)
|
||||
└── tests/ # Pytest suite (~39k test functions across ~3.7k files as of Sep 2026)
|
||||
```
|
||||
|
||||
### Facade + siblings layout (Sep 2026 decomposition)
|
||||
|
||||
Every former god file is now a **facade** plus a family of **siblings** named `<stem>_<topic>.py`
|
||||
in the same directory. The facade keeps the public entry points and the names other packages
|
||||
import; each sibling owns one topic and is where the code actually lives. Families with the most
|
||||
siblings:
|
||||
|
||||
| facade | siblings | topics |
|
||||
|---|---|---|
|
||||
| `hermes_state.py` | 21 `hermes_state_*.py` | schema, fts, search, compression, portability, gateway, errors, ... |
|
||||
| `gateway/run.py` | 15 `gateway/run_*.py` | startup, adapters, inbound, turn, busy, goals, notifications, shutdown, ... |
|
||||
| `tools/mcp_tool.py` | 15 `tools/mcp_tool_*.py` | config, discovery, transport, registration, content, errors, ... |
|
||||
| `hermes_cli/kanban.py` | 14 `hermes_cli/kanban_*.py` | boards, db, db_connect, db_dispatch, db_notify, workspace, ... |
|
||||
| `hermes_cli/web_server.py` | 13 `web_server_*.py` + 24 `web_routers/*.py` | one router file per dashboard surface |
|
||||
| `hermes_cli/auth.py` | 12 `auth_*.py` | device_flow, codex, minimax, model_picker, commands, ... |
|
||||
| `tools/browser_tool.py` | 11 `browser_tool_*.py` | cdp, cloud, install, lifecycle, session, real_profile, vision, ... |
|
||||
| `cli.py` | 12 `hermes_cli/cli_*_mixin.py` | commands, stream, status_bar, billing, ... — mixed into `HermesCLI` |
|
||||
| `run_agent.py` | `agent/turn_*.py`, `agent/agent_init.py`, `agent/conversation_loop.py` | turn phases: iteration_prep, api_call, api_error, overflow, truncation, recovery |
|
||||
|
||||
Rules that follow from this layout:
|
||||
|
||||
- **Find code by topic, not by facade.** `grep -rn "def name" <dir>/<stem>_*.py` lands on the
|
||||
definition; reading the facade first is the expensive way (see `evals/codebase_navigability/`).
|
||||
- **Siblings may import each other and late-import the facade** (inside functions) to avoid cycles.
|
||||
A facade never imports a sibling at module level *and* gets imported by that sibling at module level.
|
||||
- **Patch where production reads.** Many siblings deliberately do `from <facade> import name` inside
|
||||
the function so tests can `monkeypatch.setattr(facade, "name", ...)`. Before writing a patch target,
|
||||
check which binding the call site actually reads; a patch on the wrong module passes silently.
|
||||
- **`scripts/check_compat_pointers.py` runs in CI.** Old import paths kept alive for external plugins
|
||||
(`PLUGIN-COMPAT` blocks, `COMPAT_MANIFEST.md`, `compat_manifest.json`) are OFF LIMITS to in-tree code
|
||||
and tests; they are removed on schedule by reverting one commit. Import from the defining module.
|
||||
- **Don't recreate god files.** A sibling passing ~2,000 lines or a function passing ~300 is a signal
|
||||
to split again along the same `<stem>_<topic>` convention.
|
||||
|
||||
**User config:** `~/.hermes/config.yaml` (settings), `~/.hermes/.env` (API keys only).
|
||||
**Logs:** `~/.hermes/logs/` — `agent.log` (INFO+), `errors.log` (WARNING+),
|
||||
`gateway.log` when running the gateway. Profile-aware via `get_hermes_home()`.
|
||||
@@ -350,7 +386,13 @@ run_agent.py, cli.py, batch_runner.py, environments/
|
||||
|
||||
---
|
||||
|
||||
## AIAgent Class (run_agent.py)
|
||||
## AIAgent Class (run_agent.py + agent/)
|
||||
|
||||
`run_agent.py` is the public entry: the `AIAgent` class assembled from mixins
|
||||
(`agent/turn_facade.py`, `agent/client_lifecycle.py`, `agent/stream_delivery.py`,
|
||||
`agent/session_persistence.py`, `agent/compression_facade.py`, ...). Construction runs
|
||||
`agent/agent_init.py::init_agent`; the turn itself is `agent/conversation_loop.py::run_conversation`,
|
||||
which `AIAgent.run_conversation` forwards to after taking the session turn lease.
|
||||
|
||||
The real `AIAgent.__init__` takes ~60 parameters (credentials, routing, callbacks,
|
||||
session context, budget, credential pool, etc.). The signature below is the
|
||||
@@ -388,8 +430,11 @@ class AIAgent:
|
||||
|
||||
### Agent Loop
|
||||
|
||||
The core loop is inside `run_conversation()` — entirely synchronous, with
|
||||
interrupt checks, budget tracking, and a one-turn grace call:
|
||||
The core loop is `agent/conversation_loop.py::run_conversation()` — entirely synchronous, with
|
||||
interrupt checks, budget tracking, and a one-turn grace call. Each phase of an iteration is its own
|
||||
module under `agent/turn_*.py` (`turn_iteration_prep`, `turn_request_assembly`, `turn_api_call`,
|
||||
`turn_api_error`, `turn_overflow`, `turn_truncation`, `turn_recovery`, `turn_usage`, ...), so a change
|
||||
to, say, overflow handling touches one ~600-line file. Schematically:
|
||||
|
||||
```python
|
||||
while (api_call_count < self.max_iterations and self.iteration_budget.remaining > 0) \
|
||||
@@ -410,13 +455,17 @@ Reasoning content is stored in `assistant_msg["reasoning"]`.
|
||||
|
||||
---
|
||||
|
||||
## CLI Architecture (cli.py)
|
||||
## CLI Architecture (cli.py + hermes_cli/cli_*_mixin.py)
|
||||
|
||||
- `cli.py` holds `HermesCLI` (REPL loop, config, slash dispatch); behaviour is split into mixins in
|
||||
`hermes_cli/cli_commands_mixin.py`, `cli_stream_mixin.py`, `cli_status_bar_mixin.py`, `cli_billing_mixin.py`, ...
|
||||
- **Rich** for banner/panels, **prompt_toolkit** for input with autocomplete
|
||||
- **KawaiiSpinner** (`agent/display.py`) — animated faces during API calls, `┊` activity feed for tool results
|
||||
- `load_cli_config()` in cli.py merges hardcoded defaults + user config YAML
|
||||
- **Skin engine** (`hermes_cli/skin_engine.py`) — data-driven CLI theming; initialized from `display.skin` config key at startup; skins customize banner colors, spinner faces/verbs/wings, tool prefix, response box, branding text
|
||||
- `process_command()` is a method on `HermesCLI` — dispatches on canonical command name resolved via `resolve_command()` from the central registry
|
||||
- `process_command()` on `HermesCLI` resolves the canonical name via `resolve_command()` from the central
|
||||
registry, then dispatches through the `_SLASH_DISPATCH` table (`canonical -> (method name, pass_arg)`),
|
||||
falling back to a `_handle_<name>_command` method by naming convention. There is no `elif` ladder.
|
||||
- Skill slash commands: `agent/skill_commands.py` scans `~/.hermes/skills/`, injects as **user message** (not system prompt) to preserve prompt caching
|
||||
|
||||
### Slash Command Registry (`hermes_cli/commands.py`)
|
||||
@@ -438,16 +487,16 @@ All slash commands are defined in a central `COMMAND_REGISTRY` list of `CommandD
|
||||
CommandDef("mycommand", "Description of what it does", "Session",
|
||||
aliases=("mc",), args_hint="[arg]"),
|
||||
```
|
||||
2. Add handler in `HermesCLI.process_command()` in `cli.py`:
|
||||
2. Add the handler on the CLI. Name it `_handle_mycommand_command(self, cmd_original)` on the
|
||||
relevant `hermes_cli/cli_*_mixin.py` and it is picked up by convention; or add an explicit entry to
|
||||
`HermesCLI._SLASH_DISPATCH` in `cli.py` when the method name or arg-passing differs:
|
||||
```python
|
||||
elif canonical == "mycommand":
|
||||
self._handle_mycommand(cmd_original)
|
||||
```
|
||||
3. If the command is available in the gateway, add a handler in `gateway/run.py`:
|
||||
```python
|
||||
if canonical == "mycommand":
|
||||
return await self._handle_mycommand(event)
|
||||
"mycommand": ("_handle_mycommand", True), # (method name, pass cmd_original?)
|
||||
```
|
||||
3. If the command is available in the gateway, add `_handle_mycommand_command(self, event)` on the
|
||||
matching `gateway/slash_commands_*.py` mixin and list `"mycommand"` in `_IDLE_COMMANDS` (or
|
||||
`_PLAIN_COMMANDS` if it must also work mid-run) in `gateway/run_busy.py`. Handlers are looked up
|
||||
by name through `_command_handler_table`; there is no `if canonical == ...` chain to extend.
|
||||
4. For persistent settings, use `save_config_value()` in `cli.py`
|
||||
|
||||
**CommandDef fields:**
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
# Codebase navigability eval
|
||||
|
||||
Measures what a codebase costs an **LLM agent** to work in, as opposed to what it costs the CPU.
|
||||
Built for the Sep 2026 decomposition (PR #102117) and kept so future refactors are held to the same
|
||||
numbers. Everything here is offline and deterministic; no model calls.
|
||||
|
||||
## The three questions it answers
|
||||
|
||||
1. **How much code has to be read to see one definition?** (`bench.py`)
|
||||
Workload = every `from <first-party module> import <Name>` in `tests/`, resolved through re-exports
|
||||
to the module that *defines* the name. That is ~19k real "locate X" tasks nobody hand-picked.
|
||||
Per task: tokens of the defining file, tokens of the symbol itself, overhead = file − symbol,
|
||||
whether the file fits a 32k / 128k context, how many 2,000-line `read_file` windows it spans,
|
||||
how many unrelated top-level siblings sit in the same file, and the symbol's cyclomatic complexity.
|
||||
Tokens are real (`tiktoken` `o200k_base`); falls back to bytes/4 if tiktoken is missing and says so.
|
||||
|
||||
2. **What does a careful agent actually pay to look a symbol up?** (`lookup_sim.py`)
|
||||
Simulates the policy a good model follows with our tools: `grep -n` for the definition (1 call),
|
||||
`read_file` a 200-line window around the hit (1 call), page forward in 2,000-line windows only while
|
||||
the definition is still running. Charges tool calls and returned tokens. Same random sample of symbols
|
||||
that exist on BOTH trees, so it is a paired comparison.
|
||||
|
||||
3. **Shape and runtime.** (`static_metrics.py`, `runtime_bench.py`)
|
||||
LOC split (code/comment/docstring), file and function size distributions, elif-chain lengths,
|
||||
nesting depth, radon CC/MI, import graph (edges, fan-in/out, Tarjan SCC cycles); fresh-interpreter
|
||||
import time / module count / RSS for the entry points, CLI end-to-end, in-process hot paths, pytest
|
||||
collection, bytecode footprint. `runtime_bench.py` pins `PYTHONPATH` to the tree and asserts no
|
||||
module resolved from another checkout (an editable install will silently cross-contaminate otherwise).
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
# deps: tiktoken + radon (bench venv or the project venv)
|
||||
uv pip install tiktoken radon
|
||||
|
||||
# 1 + 2: pass two checkouts (git worktree add is the easy way to get the baseline)
|
||||
git worktree add /tmp/base origin/main
|
||||
python evals/codebase_navigability/bench.py /tmp/base base --out out/
|
||||
python evals/codebase_navigability/bench.py . head --out out/
|
||||
python evals/codebase_navigability/lookup_sim.py /tmp/base . --sample 4000 --out out/
|
||||
|
||||
# 3
|
||||
NAV_OUT=out/ python evals/codebase_navigability/static_metrics.py /tmp/base base
|
||||
NAV_OUT=out/ python evals/codebase_navigability/static_metrics.py . head
|
||||
NAV_OUT=out/ python evals/codebase_navigability/runtime_bench.py /tmp/base base 9
|
||||
NAV_OUT=out/ python evals/codebase_navigability/runtime_bench.py . head 9
|
||||
```
|
||||
|
||||
`bench.py` and `static_metrics.py` take ~2 min each on a 1M-line tree; `lookup_sim.py` ~10 min for
|
||||
4,000 symbols (it tokenizes every window it "reads"); `runtime_bench.py` ~4 min per tree at 9 reps.
|
||||
|
||||
## Reading the results honestly
|
||||
|
||||
- `bench.py` measures the **naive** cost (read the whole defining file). It is the number that drops
|
||||
~3× when god files are split, and the one that decides whether a file fits a context window at all.
|
||||
- `lookup_sim.py` measures the **skilled** cost. It barely moves with a file split, because grep + a
|
||||
200-line window already dodges file size. What moves it is (a) the definition itself getting shorter
|
||||
and (b) token *density* of the surrounding code. Stripping comments makes each line denser, so a fixed
|
||||
200-line window costs more tokens after a comment-stripping refactor even though the code is smaller.
|
||||
Both effects are real and pull in opposite directions; report both numbers, not the flattering one.
|
||||
- Import time goes **up** with a split in Python (per-module overhead), and the import graph's largest
|
||||
cycle typically grows (intra-file coupling becomes inter-module edges). Neither is hidden by these tools.
|
||||
|
||||
Results for PR #102117 are in the PR description; raw JSON for that run lives in the PR thread.
|
||||
@@ -0,0 +1,223 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Agent-navigability benchmark: what does it cost an LLM agent to find and read a symbol?
|
||||
|
||||
Workload (not hand-picked): every ``from <first-party module> import <Name>`` in tests/ on the tree under
|
||||
test, resolved to the module that DEFINES the name on that tree. Each is one "locate and read X" task.
|
||||
|
||||
Per task, three costs are measured in real tokenizer tokens (tiktoken o200k_base, the GPT-4o/o-series
|
||||
tokenizer; other tokenizers differ by a roughly constant factor):
|
||||
|
||||
file_tokens tokens to read the whole file that defines X (the naive "read_file" cost)
|
||||
symbol_tokens tokens of X's own definition (irreducible: you must read this)
|
||||
overhead file_tokens - symbol_tokens (what the file's SIZE costs you beyond the answer)
|
||||
fits_128k / fits_32k can the defining file be loaded whole into that context?
|
||||
windows_2k how many 2,000-line read_file windows the file spans (how many reads to scan it)
|
||||
|
||||
Also: for each defining file, the number of OTHER top-level symbols it contains (how much unrelated
|
||||
code sits next to the answer), and the depth/complexity of the symbol itself.
|
||||
|
||||
Usage: python evals/codebase_navigability/bench.py <tree> <label> [--out DIR]
|
||||
Compare two labels with: python evals/codebase_navigability/compare.py base.json head.json
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import ast
|
||||
import json
|
||||
import os
|
||||
import statistics
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
SKIP_TOP = {".git", "node_modules", "apps", "website", "build", ".venv", "venv", "MagicMock", "__pycache__",
|
||||
".worktrees", "dist", "evals", "skills", "optional-skills", "docs", "tests"}
|
||||
FIRST_PARTY_HINT = None # resolved from the tree: any top-level .py or package dir
|
||||
|
||||
|
||||
def _tokenizer():
|
||||
try:
|
||||
import tiktoken
|
||||
enc = tiktoken.get_encoding("o200k_base")
|
||||
return lambda s: len(enc.encode(s, disallowed_special=()))
|
||||
except Exception:
|
||||
sys.stderr.write("tiktoken unavailable; falling back to bytes/4 (label will say so)\n")
|
||||
return lambda s: len(s.encode("utf-8")) // 4
|
||||
|
||||
|
||||
def source_modules(tree: Path) -> dict[str, Path]:
|
||||
mods = {}
|
||||
for dp, dns, fns in os.walk(tree):
|
||||
rel = os.path.relpath(dp, tree)
|
||||
top = rel.split(os.sep)[0]
|
||||
if rel != "." and top in SKIP_TOP:
|
||||
dns[:] = []
|
||||
continue
|
||||
for f in fns:
|
||||
if f.endswith(".py"):
|
||||
p = Path(dp) / f
|
||||
m = os.path.relpath(p, tree)[:-3].replace(os.sep, ".")
|
||||
if m.endswith(".__init__"):
|
||||
m = m[:-9]
|
||||
mods[m] = p
|
||||
return mods
|
||||
|
||||
|
||||
def top_level_defs(src: str) -> dict[str, tuple[int, int, ast.AST]]:
|
||||
out = {}
|
||||
try:
|
||||
tree = ast.parse(src)
|
||||
except SyntaxError:
|
||||
return out
|
||||
for n in tree.body:
|
||||
names = []
|
||||
if isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
||||
names = [n.name]
|
||||
elif isinstance(n, ast.Assign):
|
||||
names = [t.id for t in n.targets if isinstance(t, ast.Name)]
|
||||
elif isinstance(n, ast.AnnAssign) and isinstance(n.target, ast.Name):
|
||||
names = [n.target.id]
|
||||
start = n.lineno
|
||||
if getattr(n, "decorator_list", None):
|
||||
start = min(d.lineno for d in n.decorator_list)
|
||||
for nm in names:
|
||||
out[nm] = (start, n.end_lineno, n)
|
||||
return out
|
||||
|
||||
|
||||
def test_imports(tree: Path, mods: dict[str, Path]) -> list[tuple[str, str, str]]:
|
||||
"""(test_file, module, name) for every first-party from-import in tests/."""
|
||||
tasks = []
|
||||
for dp, _, fns in os.walk(tree / "tests"):
|
||||
for f in fns:
|
||||
if not f.endswith(".py"):
|
||||
continue
|
||||
p = Path(dp) / f
|
||||
try:
|
||||
t = ast.parse(p.read_text(encoding="utf-8", errors="replace"))
|
||||
except SyntaxError:
|
||||
continue
|
||||
for n in ast.walk(t):
|
||||
if isinstance(n, ast.ImportFrom) and n.module and n.level == 0 and n.module in mods:
|
||||
for a in n.names:
|
||||
if a.name != "*":
|
||||
tasks.append((os.path.relpath(p, tree), n.module, a.name))
|
||||
return tasks
|
||||
|
||||
|
||||
def resolve_definer(mods, name, start_mod, cache, depth=0):
|
||||
"""Follow re-exports (from X import name / lazy tables) to the module whose top level DEFINES name."""
|
||||
if depth > 6 or start_mod not in mods:
|
||||
return None
|
||||
if start_mod not in cache:
|
||||
src = mods[start_mod].read_text(encoding="utf-8", errors="replace")
|
||||
cache[start_mod] = (src, top_level_defs(src), _reexports(src))
|
||||
src, defs, rex = cache[start_mod]
|
||||
if name in defs and not _is_alias(defs[name][2]):
|
||||
return start_mod # defined here (def/class/real assignment)
|
||||
if name in rex: # re-export: follow to the origin
|
||||
return resolve_definer(mods, rex[name][1], rex[name][0], cache, depth + 1)
|
||||
return start_mod if name in defs else None
|
||||
|
||||
|
||||
def _is_alias(node) -> bool:
|
||||
return isinstance(node, (ast.Assign, ast.AnnAssign)) and isinstance(getattr(node, "value", None), ast.Name)
|
||||
|
||||
|
||||
def _reexports(src: str) -> dict[str, tuple[str, str]]:
|
||||
"""name -> (module, original) for top-level `from m import orig as name`, plus PLUGIN-COMPAT lazy tables."""
|
||||
out = {}
|
||||
try:
|
||||
t = ast.parse(src)
|
||||
except SyntaxError:
|
||||
return out
|
||||
for n in t.body:
|
||||
if isinstance(n, ast.ImportFrom) and n.module and n.level == 0:
|
||||
for a in n.names:
|
||||
out[a.asname or a.name] = (n.module, a.name)
|
||||
elif isinstance(n, ast.Assign) and any(isinstance(x, ast.Name) and x.id == "_PLUGIN_COMPAT_LAZY" for x in n.targets) and isinstance(n.value, ast.Dict):
|
||||
for k, v in zip(n.value.keys, n.value.values):
|
||||
try:
|
||||
out[ast.literal_eval(k)] = tuple(ast.literal_eval(v))
|
||||
except Exception:
|
||||
pass
|
||||
return out
|
||||
|
||||
|
||||
def cyclomatic(node) -> int:
|
||||
c = 1
|
||||
for n in ast.walk(node):
|
||||
if isinstance(n, (ast.If, ast.For, ast.While, ast.ExceptHandler, ast.With, ast.Assert, ast.IfExp, ast.comprehension, ast.AsyncFor, ast.AsyncWith)):
|
||||
c += 1
|
||||
elif isinstance(n, ast.BoolOp):
|
||||
c += len(n.values) - 1
|
||||
elif isinstance(n, ast.Match):
|
||||
c += len(n.cases)
|
||||
return c
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("tree")
|
||||
ap.add_argument("label")
|
||||
ap.add_argument("--out", default=".")
|
||||
args = ap.parse_args()
|
||||
tree = Path(args.tree).resolve()
|
||||
tok = _tokenizer()
|
||||
mods = source_modules(tree)
|
||||
tasks = test_imports(tree, mods)
|
||||
cache: dict = {}
|
||||
file_tok_cache: dict[str, int] = {}
|
||||
rows = []
|
||||
unresolved = 0
|
||||
for test_file, mod, name in tasks:
|
||||
definer = resolve_definer(mods, name, mod, cache)
|
||||
if definer is None:
|
||||
unresolved += 1
|
||||
continue
|
||||
src, defs, _ = cache[definer]
|
||||
if name not in defs:
|
||||
unresolved += 1
|
||||
continue
|
||||
s, e, node = defs[name]
|
||||
if definer not in file_tok_cache:
|
||||
file_tok_cache[definer] = tok(src)
|
||||
lines = src.split("\n")
|
||||
seg = "\n".join(lines[s - 1:e])
|
||||
rows.append({
|
||||
"test": test_file, "imported_from": mod, "name": name, "definer": definer,
|
||||
"file_lines": len(lines), "file_tokens": file_tok_cache[definer],
|
||||
"symbol_lines": e - s + 1, "symbol_tokens": tok(seg),
|
||||
"siblings": len(defs) - 1, "cc": cyclomatic(node) if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)) else 0,
|
||||
"reexported": definer != mod,
|
||||
})
|
||||
# aggregate
|
||||
ft = [r["file_tokens"] for r in rows]
|
||||
ov = [r["file_tokens"] - r["symbol_tokens"] for r in rows]
|
||||
st = [r["symbol_tokens"] for r in rows]
|
||||
q = lambda a, p: (sorted(a)[min(len(a) - 1, int(len(a) * p))] if a else 0)
|
||||
summary = {
|
||||
"label": args.label, "tree": str(tree), "tokenizer": "o200k_base" if "tiktoken" in sys.modules else "bytes/4",
|
||||
"tasks": len(rows), "unresolved": unresolved, "distinct_symbols": len({(r["definer"], r["name"]) for r in rows}),
|
||||
"distinct_defining_files": len({r["definer"] for r in rows}),
|
||||
"file_tokens_mean": round(statistics.mean(ft)), "file_tokens_p50": q(ft, .5), "file_tokens_p90": q(ft, .9), "file_tokens_p99": q(ft, .99), "file_tokens_max": max(ft),
|
||||
"symbol_tokens_mean": round(statistics.mean(st)), "symbol_tokens_p50": q(st, .5),
|
||||
"overhead_tokens_mean": round(statistics.mean(ov)), "overhead_tokens_p50": q(ov, .5), "overhead_tokens_p90": q(ov, .9),
|
||||
"overhead_ratio_mean": round(statistics.mean(r["file_tokens"] / max(1, r["symbol_tokens"]) for r in rows), 1),
|
||||
"total_file_tokens_if_read_whole": sum(ft), "total_symbol_tokens": sum(st),
|
||||
"tasks_file_over_128k": sum(1 for x in ft if x > 128_000), "tasks_file_over_32k": sum(1 for x in ft if x > 32_000), "tasks_file_over_8k": sum(1 for x in ft if x > 8_000),
|
||||
"tasks_file_over_2000_lines": sum(1 for r in rows if r["file_lines"] > 2000),
|
||||
"read_windows_2k_mean": round(statistics.mean(-(-r["file_lines"] // 2000) for r in rows), 2),
|
||||
"siblings_mean": round(statistics.mean(r["siblings"] for r in rows), 1), "siblings_p90": q([r["siblings"] for r in rows], .9),
|
||||
"symbol_cc_mean": round(statistics.mean(r["cc"] for r in rows if r["cc"]), 2), "symbol_cc_p90": q([r["cc"] for r in rows if r["cc"]], .9),
|
||||
"symbol_lines_p50": q([r["symbol_lines"] for r in rows], .5), "symbol_lines_p90": q([r["symbol_lines"] for r in rows], .9),
|
||||
"reexport_hops": sum(1 for r in rows if r["reexported"]),
|
||||
}
|
||||
out = Path(args.out)
|
||||
out.mkdir(parents=True, exist_ok=True)
|
||||
(out / f"{args.label}.navigability.json").write_text(json.dumps({"summary": summary, "rows": rows}, indent=1))
|
||||
print(json.dumps(summary, indent=1))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,140 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Simulated agent lookups: how many tool calls / tokens to answer "show me the definition of X"?
|
||||
|
||||
Deterministic agent policy (the one a careful model actually follows, given read_file's 2,000-line window):
|
||||
1. grep -n "def X\\b|class X\\b|^X\\s*=" across the tree -> 1 call, returns (file, line) hits
|
||||
2. read_file(file, offset=hit_line-20, limit=W) for W in (200,) -> 1 call; if the symbol's end is past the
|
||||
window, read the next 2,000-line window until it is -> +1 call each
|
||||
3. done when the whole definition is in context
|
||||
|
||||
Costs charged per task:
|
||||
tool_calls = 1 grep + N reads
|
||||
tokens = tokens of grep output (all hits, one line each) + tokens of every read window returned
|
||||
Both trees get the same symbol set: the intersection of symbols that exist (by name) on both, so a symbol moving
|
||||
to a smaller file counts as a WIN and a symbol that was deleted on one side is excluded, not scored.
|
||||
|
||||
Usage: python evals/codebase_navigability/lookup_sim.py <base_tree> <head_tree> [--sample N] [--seed S]
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import ast
|
||||
import json
|
||||
import os
|
||||
import random
|
||||
import statistics
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
SKIP_TOP = {".git", "node_modules", "apps", "website", "build", ".venv", "venv", "MagicMock", "__pycache__", ".worktrees", "dist", "evals", "skills", "optional-skills", "docs", "tests"}
|
||||
WINDOW = 2000 # read_file's default max lines per call
|
||||
FIRST_READ = 200 # a careful agent's first targeted read around the grep hit (override: --first-read)
|
||||
MARGIN = 5 # lines of context above the grep hit
|
||||
|
||||
|
||||
def tokenizer():
|
||||
try:
|
||||
import tiktoken
|
||||
enc = tiktoken.get_encoding("o200k_base")
|
||||
return lambda s: len(enc.encode(s, disallowed_special=()))
|
||||
except Exception:
|
||||
return lambda s: len(s.encode()) // 4
|
||||
|
||||
|
||||
def index(tree: Path):
|
||||
"""name -> list of (relpath, start_line, end_line, lines_of_file)"""
|
||||
idx: dict[str, list] = {}
|
||||
files: dict[str, list[str]] = {}
|
||||
for dp, dns, fns in os.walk(tree):
|
||||
rel = os.path.relpath(dp, tree)
|
||||
if rel != "." and rel.split(os.sep)[0] in SKIP_TOP:
|
||||
dns[:] = []
|
||||
continue
|
||||
for f in fns:
|
||||
if not f.endswith(".py"):
|
||||
continue
|
||||
p = Path(dp) / f
|
||||
src = p.read_text(encoding="utf-8", errors="replace")
|
||||
lines = src.split("\n")
|
||||
r = os.path.relpath(p, tree)
|
||||
files[r] = lines
|
||||
try:
|
||||
t = ast.parse(src)
|
||||
except SyntaxError:
|
||||
continue
|
||||
for n in ast.walk(t):
|
||||
if isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
||||
s = min([d.lineno for d in n.decorator_list] + [n.lineno])
|
||||
idx.setdefault(n.name, []).append((r, s, n.end_lineno, len(lines)))
|
||||
return idx, files
|
||||
|
||||
|
||||
def grep_hits(name, idx):
|
||||
"""Simulate `grep -rn` for the symbol: every def/class of that name, one line each."""
|
||||
return idx.get(name, [])
|
||||
|
||||
|
||||
def simulate(name, idx, files, tok):
|
||||
hits = grep_hits(name, idx)
|
||||
if not hits:
|
||||
return None
|
||||
grep_out = "\n".join(f"{r}:{s}: def {name}(...)" for r, s, _, _ in hits)
|
||||
tokens = tok(grep_out)
|
||||
calls = 1
|
||||
# The agent opens the FIRST hit (deterministic; ambiguity costs are the same policy on both trees)
|
||||
r, s, e, n = hits[0]
|
||||
lines = files[r]
|
||||
start = max(1, s - MARGIN)
|
||||
# The agent doesn't know where the definition ends until it reads it: it asks for a window
|
||||
# sized to the typical definition (FIRST_READ) and pages forward only if the def keeps going.
|
||||
end = min(n, start + FIRST_READ - 1)
|
||||
tokens += tok("\n".join(lines[start - 1:end]))
|
||||
calls += 1
|
||||
while end < e: # definition continues past the window: page forward
|
||||
start = end + 1
|
||||
end = min(n, start + WINDOW - 1)
|
||||
tokens += tok("\n".join(lines[start - 1:end]))
|
||||
calls += 1
|
||||
exact = tok("\n".join(lines[s - 1:e]))
|
||||
return {"name": name, "file": r, "file_lines": n, "def_lines": e - s + 1, "hits": len(hits), "calls": calls, "tokens": tokens, "exact_tokens": exact}
|
||||
|
||||
|
||||
def main():
|
||||
global FIRST_READ
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("base"); ap.add_argument("head")
|
||||
ap.add_argument("--sample", type=int, default=3000); ap.add_argument("--seed", type=int, default=7)
|
||||
ap.add_argument("--first-read", type=int, default=FIRST_READ, help="lines in the first targeted read")
|
||||
ap.add_argument("--out", default=".")
|
||||
a = ap.parse_args()
|
||||
FIRST_READ = a.first_read
|
||||
tok = tokenizer()
|
||||
bi, bf = index(Path(a.base)); hi, hf = index(Path(a.head))
|
||||
common = sorted(set(bi) & set(hi))
|
||||
# weight the sample toward symbols that are actually looked up: public names, exclude dunders/tests
|
||||
common = [n for n in common if not n.startswith("__") and not n.startswith("test_")]
|
||||
random.Random(a.seed).shuffle(common)
|
||||
sample = common[: a.sample]
|
||||
B = [simulate(n, bi, bf, tok) for n in sample]
|
||||
H = [simulate(n, hi, hf, tok) for n in sample]
|
||||
pairs = [(b, h) for b, h in zip(B, H) if b and h]
|
||||
def agg(rows):
|
||||
c = [r["calls"] for r in rows]; t = [r["tokens"] for r in rows]
|
||||
return {"tasks": len(rows), "calls_mean": round(statistics.mean(c), 3), "calls_total": sum(c), "multi_window_tasks": sum(1 for x in c if x > 2),
|
||||
"tokens_mean": round(statistics.mean(t)), "tokens_p50": int(statistics.median(t)), "tokens_p90": sorted(t)[int(len(t) * .9)], "tokens_total": sum(t),
|
||||
"file_lines_p50": int(statistics.median(r["file_lines"] for r in rows)), "def_lines_p50": int(statistics.median(r["def_lines"] for r in rows)),
|
||||
"ambiguous_hits_mean": round(statistics.mean(r["hits"] for r in rows), 2),
|
||||
"exact_def_tokens_mean": round(statistics.mean(r["exact_tokens"] for r in rows)), "exact_def_tokens_total": sum(r["exact_tokens"] for r in rows)}
|
||||
res = {"common_symbols": len(common), "sampled": len(pairs), "seed": a.seed, "window": WINDOW, "first_read": FIRST_READ,
|
||||
"base": agg([b for b, _ in pairs]), "head": agg([h for _, h in pairs]),
|
||||
"head_cheaper": sum(1 for b, h in pairs if h["tokens"] < b["tokens"]), "head_costlier": sum(1 for b, h in pairs if h["tokens"] > b["tokens"]), "equal": sum(1 for b, h in pairs if h["tokens"] == b["tokens"]),
|
||||
"worst_regressions": sorted(({"name": h["name"], "base_tok": b["tokens"], "head_tok": h["tokens"], "head_file": h["file"]} for b, h in pairs), key=lambda x: x["base_tok"] - x["head_tok"])[:8],
|
||||
"best_wins": sorted(({"name": h["name"], "base_tok": b["tokens"], "head_tok": h["tokens"], "base_file": b["file"], "head_file": h["file"]} for b, h in pairs), key=lambda x: x["head_tok"] - x["base_tok"])[:8]}
|
||||
Path(a.out).mkdir(parents=True, exist_ok=True)
|
||||
(Path(a.out) / f"lookup_sim_w{FIRST_READ}.json").write_text(json.dumps(res, indent=1))
|
||||
print(json.dumps({k: v for k, v in res.items() if k not in ("worst_regressions", "best_wins")}, indent=1))
|
||||
print("best wins:", json.dumps(res["best_wins"][:4], indent=1)); print("worst regressions:", json.dumps(res["worst_regressions"][:4], indent=1))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,152 @@
|
||||
"""Runtime benchmarks for one tree. Usage: python runtime_bench.py <tree> <label> [reps]
|
||||
|
||||
Each probe runs in a FRESH subprocess with an isolated HERMES_HOME so nothing is cached across reps.
|
||||
Reports medians + min over reps. Writes <label>.runtime.json.
|
||||
"""
|
||||
import json, os, statistics, subprocess, sys, tempfile, time, shutil
|
||||
|
||||
TREE, LABEL = sys.argv[1], sys.argv[2]
|
||||
REPS = int(sys.argv[3]) if len(sys.argv) > 3 else 7
|
||||
PY = os.environ.get("NAV_PY", sys.executable)
|
||||
HOME = tempfile.mkdtemp(prefix=f"hh_{LABEL}_")
|
||||
os.makedirs(f"{HOME}/skills", exist_ok=True)
|
||||
open(f"{HOME}/config.yaml", "w", encoding="utf-8").write("model:\n default: openai/gpt-4o-mini\n provider: openrouter\nterminal:\n backend: local\n")
|
||||
ENV = {**os.environ, "HERMES_HOME": HOME, "PYTHONPATH": TREE, "PYTHONDONTWRITEBYTECODE": "0", "OPENROUTER_API_KEY": "sk-bench-placeholder",
|
||||
"HERMES_SKIP_UPDATE_CHECK": "1", "NO_COLOR": "1", "TERM": "dumb", "COLUMNS": "120"}
|
||||
for k in list(ENV):
|
||||
if k.startswith(("HERMES_SESSION", "HERMES_PROFILE")): ENV.pop(k)
|
||||
|
||||
def run(argv, code=None, timeout=300):
|
||||
t0 = time.perf_counter()
|
||||
r = subprocess.run([PY, *argv] if code is None else [PY, "-c", code], cwd=TREE, env=ENV, capture_output=True, text=True, timeout=timeout, stdin=subprocess.DEVNULL)
|
||||
return time.perf_counter() - t0, r
|
||||
|
||||
def warm_pyc():
|
||||
subprocess.run([PY, "-m", "compileall", "-q", "-j", "16", TREE], cwd=TREE, env=ENV, capture_output=True)
|
||||
|
||||
def med(xs): return round(statistics.median(xs), 4)
|
||||
|
||||
results = {"label": LABEL, "tree": TREE, "reps": REPS}
|
||||
|
||||
# 0. bytecode on disk (after compileall) — proxy for "how much code gets loaded"
|
||||
warm_pyc()
|
||||
pyc_bytes = 0; pyc_n = 0
|
||||
for dp, dns, fns in os.walk(TREE):
|
||||
if any(s in dp for s in ("/node_modules", "/apps/", "/.git", "/tests", "/website", "/build", "/venv")): dns[:] = []; continue
|
||||
for f in fns:
|
||||
if f.endswith(".pyc"): pyc_bytes += os.path.getsize(os.path.join(dp, f)); pyc_n += 1
|
||||
results["pyc_files"] = pyc_n; results["pyc_bytes"] = pyc_bytes
|
||||
|
||||
# 1. import-time probes: fresh interpreter, measure wall + module count + RSS
|
||||
IMPORT_TARGETS = ["run_agent", "cli", "hermes_cli.main", "gateway.run", "tools.registry", "hermes_state", "tui_gateway.server", "hermes_cli.web_server", "model_tools", "agent.prompt_builder"]
|
||||
probe = r'''
|
||||
import sys, time, os, resource, json
|
||||
t0=time.perf_counter()
|
||||
n0=len(sys.modules)
|
||||
import importlib
|
||||
try:
|
||||
importlib.import_module(sys.argv[1]); err=None
|
||||
except Exception as e:
|
||||
err=repr(e)[:200]
|
||||
dt=time.perf_counter()-t0
|
||||
rss=resource.getrusage(resource.RUSAGE_SELF).ru_maxrss
|
||||
tree=os.getcwd(); foreign=[m for m,mod in list(sys.modules.items()) if getattr(mod,"__file__",None) and "hermes-agent" in mod.__file__ and not mod.__file__.startswith(tree) and "site-packages" not in mod.__file__]
|
||||
if foreign: err=(err or "")+f" FOREIGN_MODULES:{foreign[:3]}"
|
||||
print(json.dumps({"dt":dt,"mods":len(sys.modules)-n0,"rss_kb":rss,"err":err}))
|
||||
'''
|
||||
imp = {}
|
||||
for tgt in IMPORT_TARGETS:
|
||||
rows = []
|
||||
for _ in range(REPS):
|
||||
_, r = run([], code=None) if False else (None, None)
|
||||
r = subprocess.run([PY, "-c", probe, tgt], cwd=TREE, env=ENV, capture_output=True, text=True, timeout=300)
|
||||
try: rows.append(json.loads([l for l in (r.stdout + "\n" + r.stderr).splitlines() if l.lstrip().startswith("{") and "\"dt\"" in l][-1].strip()))
|
||||
except Exception: rows.append({"dt": None, "mods": None, "rss_kb": None, "err": (r.stderr or r.stdout)[-300:]})
|
||||
ok = [x for x in rows if x["dt"] is not None and not x["err"]]
|
||||
imp[tgt] = {"dt_med": med([x["dt"] for x in ok]) if ok else None, "dt_min": round(min(x["dt"] for x in ok), 4) if ok else None,
|
||||
"mods": ok[0]["mods"] if ok else None, "rss_mb": round(med([x["rss_kb"] for x in ok]) / 1024, 1) if ok else None,
|
||||
"err": next((x["err"] for x in rows if x["err"]), None)}
|
||||
results["import"] = imp
|
||||
|
||||
# 2. CLI end-to-end startup: `hermes --version`, `hermes --help`, `hermes doctor --help`, `hermes config get model` (no network)
|
||||
CLI = {"version": ["hermes_cli/main.py", "--version"], "help": ["hermes_cli/main.py", "--help"], "config_get": ["hermes_cli/main.py", "config", "get", "model"], "tools_list": ["hermes_cli/main.py", "tools", "--help"], "skills_help": ["hermes_cli/main.py", "skills", "--help"]}
|
||||
cli = {}
|
||||
for name, argv in CLI.items():
|
||||
ts = []; rc = None; last = ""
|
||||
for _ in range(REPS):
|
||||
dt, r = run(argv); ts.append(dt); rc = r.returncode; last = (r.stderr or r.stdout)[-200:]
|
||||
cli[name] = {"med": med(ts), "min": round(min(ts), 4), "rc": rc, "tail": last if rc else ""}
|
||||
results["cli"] = cli
|
||||
|
||||
# 3. in-process hot paths (single fresh interpreter, timeit inside): tool schema assembly, system prompt build, state db ops
|
||||
hot = r'''
|
||||
import time, json, os, sys, timeit, importlib
|
||||
out={}
|
||||
def T(name, fn, n):
|
||||
fn() # warm
|
||||
t=timeit.Timer(fn).repeat(repeat=5, number=n)
|
||||
out[name]=round(min(t)/n*1000, 4) # ms per call
|
||||
try:
|
||||
import model_tools
|
||||
fn=getattr(model_tools,"get_tool_definitions",None) or getattr(model_tools,"get_all_tool_definitions",None)
|
||||
if fn: T("tool_definitions_ms", lambda: fn(), 20)
|
||||
except Exception as e: out["tool_definitions_err"]=repr(e)[:160]
|
||||
try:
|
||||
from toolsets import get_all_toolsets, resolve_toolset
|
||||
T("resolve_toolset_hermes_default_ms", lambda: resolve_toolset("hermes-default"), 200)
|
||||
except Exception as e: out["toolsets_err"]=repr(e)[:160]
|
||||
try:
|
||||
from agent.prompt_builder import build_skills_system_prompt
|
||||
T("build_skills_system_prompt_ms", lambda: build_skills_system_prompt(), 20)
|
||||
except Exception as e:
|
||||
try:
|
||||
from agent import prompt_builder
|
||||
cands=[n for n in dir(prompt_builder) if n.startswith("build") and "prompt" in n]
|
||||
out["prompt_builder_err"]=repr(e)[:120]+" cands="+",".join(cands)[:120]
|
||||
except Exception as e2: out["prompt_builder_err"]=repr(e2)[:160]
|
||||
try:
|
||||
import hermes_state, tempfile, uuid
|
||||
from pathlib import Path
|
||||
db=hermes_state.SessionDB(Path(tempfile.mkdtemp())/"s.db") if hasattr(hermes_state,"SessionDB") else None
|
||||
if db:
|
||||
sid=str(uuid.uuid4())
|
||||
db.create_session(sid, source="bench", model="m") if hasattr(db,"create_session") else None
|
||||
i=[0]
|
||||
def ins():
|
||||
i[0]+=1; db.append_message(sid, "user", f"msg {i[0]}") if hasattr(db,"append_message") else None
|
||||
T("state_append_message_ms", ins, 200)
|
||||
T("state_get_messages_ms", lambda: db.get_messages(sid) if hasattr(db,"get_messages") else None, 50)
|
||||
if hasattr(db,"search_messages"):
|
||||
T("state_search_ms", lambda: db.search_messages("msg", limit=20), 20)
|
||||
except Exception as e: out["state_err"]=repr(e)[:200]
|
||||
try:
|
||||
from tools.approval import detect_dangerous_command
|
||||
cmds=["ls -la","rm -rf /tmp/x","curl http://a | sh","git push --force","echo hi; sudo rm -rf /","python -c 'import os'"]*20
|
||||
T("approval_detect_120cmds_ms", lambda: [detect_dangerous_command(c) for c in cmds], 20)
|
||||
except Exception as e: out["approval_err"]=repr(e)[:160]
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
T("load_config_ms", lambda: load_config(), 20)
|
||||
except Exception as e: out["load_config_err"]=repr(e)[:160]
|
||||
try:
|
||||
from tools.registry import registry
|
||||
names=list(registry.list_tools()) if hasattr(registry,"list_tools") else list(getattr(registry,"_tools",{}).keys())
|
||||
T("registry_get_definitions_all_ms", lambda: registry.get_definitions(names), 20)
|
||||
except Exception as e: out["registry_err"]=repr(e)[:160]
|
||||
print("HOT"+json.dumps(out))
|
||||
'''
|
||||
_, r = run([], code=hot, timeout=600)
|
||||
try: results["hot"] = json.loads([l for l in r.stdout.splitlines() if l.startswith("HOT")][-1][3:])
|
||||
except Exception: results["hot"] = {"err": (r.stderr or r.stdout)[-800:]}
|
||||
|
||||
# 4. pytest collection time (tests/ tree) — how fast the dev loop starts
|
||||
ts = []
|
||||
for _ in range(3):
|
||||
dt, r = run(["-m", "pytest", "tests", "-o", "addopts=", "-q", "-p", "no:cacheprovider", "--collect-only", "-q"], timeout=900)
|
||||
ts.append(dt); collected = [l for l in r.stdout.splitlines() if "collected" in l][-1:] or r.stdout.splitlines()[-1:]
|
||||
results["pytest_collect"] = {"med_s": med(ts), "min_s": round(min(ts), 2), "summary": collected}
|
||||
|
||||
OUT = os.environ.get("NAV_OUT", "."); os.makedirs(OUT, exist_ok=True)
|
||||
json.dump(results, open(os.path.join(OUT, f"{LABEL}.runtime.json"), "w", encoding="utf-8"), indent=1)
|
||||
shutil.rmtree(HOME, ignore_errors=True)
|
||||
print(f"{LABEL}: import run_agent={imp['run_agent']['dt_med']}s mods={imp['run_agent']['mods']} rss={imp['run_agent']['rss_mb']}MB | cli --version={cli['version']['med']}s help={cli['help']['med']}s | collect={results['pytest_collect']['med_s']}s | pyc={pyc_bytes/1e6:.1f}MB")
|
||||
@@ -0,0 +1,157 @@
|
||||
"""Static code metrics for one tree. Usage: python static_metrics.py <tree> <label> -> writes <label>.static.json
|
||||
|
||||
First-party Python only (excludes tests/, node_modules, apps/, website, build, .venv, skills md).
|
||||
"""
|
||||
import ast, collections, io, json, os, shutil, subprocess, sys, time, tokenize
|
||||
|
||||
TREE, LABEL = sys.argv[1], sys.argv[2]
|
||||
SKIP = {".git", "node_modules", "apps", "website", "build", ".venv", "venv", "MagicMock", "__pycache__", ".worktrees", "dist", "evals", "skills", "optional-skills", "docs"}
|
||||
|
||||
def py_files(root, include_tests):
|
||||
for dp, dns, fns in os.walk(root):
|
||||
rel = os.path.relpath(dp, root)
|
||||
top = rel.split(os.sep)[0]
|
||||
if top in SKIP: dns[:] = []; continue
|
||||
if (top == "tests") != include_tests and rel != ".": dns[:] = []; continue
|
||||
if rel == "." and not include_tests: pass
|
||||
for f in fns:
|
||||
if f.endswith(".py"):
|
||||
p = os.path.join(dp, f)
|
||||
if include_tests and not os.path.relpath(p, root).startswith("tests"): continue
|
||||
if not include_tests and os.path.relpath(p, root).startswith("tests"): continue
|
||||
yield p
|
||||
|
||||
def code_lines(src):
|
||||
"""Non-blank, non-comment, non-docstring logical source lines (pygount-style 'code')."""
|
||||
try:
|
||||
toks = list(tokenize.generate_tokens(io.StringIO(src).readline))
|
||||
except Exception:
|
||||
return sum(1 for l in src.splitlines() if l.strip() and not l.strip().startswith("#")), 0, 0
|
||||
code_rows, comment_rows, doc_rows = set(), set(), set()
|
||||
# docstrings: STRING tokens that are the first statement of a module/def/class → approximate via ast
|
||||
try:
|
||||
tree = ast.parse(src)
|
||||
for n in ast.walk(tree):
|
||||
if isinstance(n, (ast.Module, ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)) and n.body and isinstance(n.body[0], ast.Expr) and isinstance(getattr(n.body[0], "value", None), ast.Constant) and isinstance(n.body[0].value.value, str):
|
||||
for r in range(n.body[0].lineno, n.body[0].end_lineno + 1): doc_rows.add(r)
|
||||
except Exception:
|
||||
pass
|
||||
for t in toks:
|
||||
if t.type == tokenize.COMMENT: comment_rows.add(t.start[0])
|
||||
elif t.type not in (tokenize.NL, tokenize.NEWLINE, tokenize.INDENT, tokenize.DEDENT, tokenize.ENDMARKER, tokenize.ENCODING):
|
||||
for r in range(t.start[0], t.end[0] + 1):
|
||||
if r not in doc_rows: code_rows.add(r)
|
||||
return len(code_rows - comment_rows), len(comment_rows - code_rows), len(doc_rows)
|
||||
|
||||
def analyse(files):
|
||||
m = {}; per_file = []; funcs = []; if_chains = []; nest = []
|
||||
mods = {}; edges = collections.defaultdict(set)
|
||||
t0 = time.perf_counter()
|
||||
for p in files:
|
||||
try: src = open(p, encoding="utf-8", errors="replace").read()
|
||||
except Exception: continue
|
||||
lines = src.count("\n") + (0 if src.endswith("\n") else 1)
|
||||
code, comm, doc = code_lines(src)
|
||||
m["files"] = m.get("files", 0) + 1; m["lines"] = m.get("lines", 0) + lines; m["code"] = m.get("code", 0) + code; m["comment"] = m.get("comment", 0) + comm; m["docstring"] = m.get("docstring", 0) + doc; m["bytes"] = m.get("bytes", 0) + len(src.encode())
|
||||
per_file.append((lines, p))
|
||||
try: tree = ast.parse(src)
|
||||
except Exception: m["unparsable"] = m.get("unparsable", 0) + 1; continue
|
||||
rel = os.path.relpath(p, TREE)[:-3].replace(os.sep, ".")
|
||||
if rel.endswith(".__init__"): rel = rel[:-9]
|
||||
mods[rel] = p
|
||||
for n in ast.walk(tree):
|
||||
if isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
||||
L = n.end_lineno - n.lineno + 1; funcs.append(L)
|
||||
m["functions"] = m.get("functions", 0) + 1
|
||||
# nesting depth
|
||||
def depth(node, d=0):
|
||||
best = d
|
||||
for c in ast.iter_child_nodes(node):
|
||||
if isinstance(c, (ast.If, ast.For, ast.While, ast.With, ast.Try, ast.AsyncFor, ast.AsyncWith, ast.Match)):
|
||||
best = max(best, depth(c, d + 1))
|
||||
else:
|
||||
best = max(best, depth(c, d))
|
||||
return best
|
||||
nest.append(depth(n))
|
||||
elif isinstance(n, ast.ClassDef): m["classes"] = m.get("classes", 0) + 1
|
||||
elif isinstance(n, ast.If):
|
||||
# count elif chain length (If whose orelse is a single If, recursively)
|
||||
k, cur = 1, n
|
||||
while len(cur.orelse) == 1 and isinstance(cur.orelse[0], ast.If):
|
||||
k += 1; cur = cur.orelse[0]
|
||||
if k >= 2: if_chains.append(k)
|
||||
if isinstance(n, ast.Import):
|
||||
for a in n.names: edges[rel].add(a.name)
|
||||
elif isinstance(n, ast.ImportFrom) and n.module and n.level == 0:
|
||||
edges[rel].add(n.module)
|
||||
m["parse_all_s"] = round(time.perf_counter() - t0, 2)
|
||||
funcs.sort(); per_file.sort(reverse=True)
|
||||
def pct(a, q): return a[int(len(a) * q) - 1] if a else 0
|
||||
m.update({
|
||||
"files_gt_1000": sum(1 for L, _ in per_file if L > 1000), "files_gt_2000": sum(1 for L, _ in per_file if L > 2000), "files_gt_5000": sum(1 for L, _ in per_file if L > 5000),
|
||||
"largest_files": [(L, os.path.relpath(p, TREE)) for L, p in per_file[:10]],
|
||||
"median_file_lines": sorted(L for L, _ in per_file)[len(per_file) // 2] if per_file else 0,
|
||||
"funcs_gt_100": sum(1 for L in funcs if L > 100), "funcs_gt_300": sum(1 for L in funcs if L > 300), "func_len_p50": pct(funcs, .5), "func_len_p95": pct(funcs, .95), "func_len_max": funcs[-1] if funcs else 0,
|
||||
"elif_chains_ge4": sum(1 for k in if_chains if k >= 4), "elif_chains_ge8": sum(1 for k in if_chains if k >= 8), "longest_elif_chain": max(if_chains) if if_chains else 0,
|
||||
"nesting_ge5": sum(1 for d in nest if d >= 5), "nesting_max": max(nest) if nest else 0,
|
||||
})
|
||||
# first-party import graph
|
||||
fp = set(mods)
|
||||
def resolve(name):
|
||||
while name and name not in fp: name = name.rpartition(".")[0]
|
||||
return name
|
||||
E = {a: {resolve(b) for b in bs} - {"", a} for a, bs in edges.items() if a in fp}
|
||||
E = {a: {b for b in bs if b in fp} for a, bs in E.items()}
|
||||
fan_out = [len(v) for v in E.values()]; fan_in = collections.Counter(b for bs in E.values() for b in bs)
|
||||
# SCCs (Tarjan) for cycles
|
||||
index = {}; low = {}; onstack = set(); stack = []; sccs = []; counter = [0]
|
||||
sys.setrecursionlimit(20000)
|
||||
def strong(v):
|
||||
index[v] = low[v] = counter[0]; counter[0] += 1; stack.append(v); onstack.add(v)
|
||||
for w in E.get(v, ()):
|
||||
if w not in index: strong(w); low[v] = min(low[v], low[w])
|
||||
elif w in onstack: low[v] = min(low[v], index[w])
|
||||
if low[v] == index[v]:
|
||||
comp = []
|
||||
while True:
|
||||
w = stack.pop(); onstack.discard(w); comp.append(w)
|
||||
if w == v: break
|
||||
sccs.append(comp)
|
||||
for v in list(E):
|
||||
if v not in index: strong(v)
|
||||
cyc = [c for c in sccs if len(c) > 1]
|
||||
m.update({"modules": len(fp), "import_edges": sum(fan_out), "fan_out_avg": round(sum(fan_out) / max(1, len(fan_out)), 2), "fan_out_max": max(fan_out) if fan_out else 0,
|
||||
"fan_in_max": max(fan_in.values()) if fan_in else 0, "fan_in_top": fan_in.most_common(5), "import_cycles": len(cyc), "largest_cycle": max((len(c) for c in cyc), default=0), "modules_in_cycles": sum(len(c) for c in cyc)})
|
||||
return dict(m)
|
||||
|
||||
def radon(files):
|
||||
"""radon cc + mi over the file list (JSON), aggregated."""
|
||||
R = shutil.which("radon") or os.path.join(os.path.dirname(sys.executable), "radon")
|
||||
import tempfile
|
||||
out = {}
|
||||
lst = "\n".join(files)
|
||||
# radon can't take a list file; run per-directory roots instead
|
||||
roots = sorted({os.path.relpath(f, TREE).split(os.sep)[0] if os.sep in os.path.relpath(f, TREE) else os.path.relpath(f, TREE) for f in files})
|
||||
roots = [r for r in roots if r != "tests"]
|
||||
cc = subprocess.run([R, "cc", "-j", "-s", "-e", "tests/*,tests/**", *roots], cwd=TREE, capture_output=True, text=True).stdout
|
||||
mi = subprocess.run([R, "mi", "-j", "-e", "tests/*,tests/**", *roots], cwd=TREE, capture_output=True, text=True).stdout
|
||||
try:
|
||||
ccj = json.loads(cc); blocks = [b for v in ccj.values() if isinstance(v, list) for b in v if isinstance(b, dict) and "complexity" in b]
|
||||
cs = sorted(b["complexity"] for b in blocks)
|
||||
out.update({"cc_blocks": len(cs), "cc_avg": round(sum(cs) / max(1, len(cs)), 2), "cc_p95": cs[int(len(cs) * .95) - 1] if cs else 0, "cc_max": cs[-1] if cs else 0,
|
||||
"cc_gt_15": sum(1 for c in cs if c > 15), "cc_gt_30": sum(1 for c in cs if c > 30), "cc_gt_50": sum(1 for c in cs if c > 50),
|
||||
"cc_worst": sorted(((b["complexity"], f, b["name"]) for f, v in ccj.items() if isinstance(v, list) for b in v if isinstance(b, dict) and "complexity" in b), reverse=True)[:8]})
|
||||
except Exception as e: out["cc_error"] = repr(e)[:200]
|
||||
try:
|
||||
mij = json.loads(mi); vals = [v["mi"] for v in mij.values() if isinstance(v, dict) and "mi" in v]
|
||||
out.update({"mi_files": len(vals), "mi_avg": round(sum(vals) / max(1, len(vals)), 2), "mi_lt_20": sum(1 for v in vals if v < 20), "mi_lt_10": sum(1 for v in vals if v < 10)})
|
||||
except Exception as e: out["mi_error"] = repr(e)[:200]
|
||||
return out
|
||||
|
||||
src_files = sorted(py_files(TREE, False)); test_files = sorted(py_files(TREE, True))
|
||||
res = {"label": LABEL, "tree": TREE, "source": analyse(src_files), "tests": analyse(test_files)}
|
||||
res["source"].update({"radon_" + k: v for k, v in radon(src_files).items()})
|
||||
OUT = os.environ.get("NAV_OUT", "."); os.makedirs(OUT, exist_ok=True)
|
||||
json.dump(res, open(os.path.join(OUT, f"{LABEL}.static.json"), "w", encoding="utf-8"), indent=1, default=str)
|
||||
s = res["source"]; t = res["tests"]
|
||||
print(f"{LABEL}: src files={s['files']} lines={s['lines']} code={s['code']} funcs={s['functions']} >300={s['funcs_gt_300']} files>2k={s['files_gt_2000']} >5k={s['files_gt_5000']} cycles={s['import_cycles']} cc_avg={s.get('radon_cc_avg')} mi_avg={s.get('radon_mi_avg')} | tests files={t['files']} lines={t['lines']}")
|
||||
@@ -0,0 +1,55 @@
|
||||
"""Guardrail for the navigability eval: its symbol resolver must follow the facade/sibling layout.
|
||||
|
||||
Two invariants the harness relies on (and that a future refactor could silently break):
|
||||
1. a name imported from a facade that lives in a sibling resolves to the SIBLING (through the facade's
|
||||
top-level `from sibling import name` or its PLUGIN-COMPAT lazy table);
|
||||
2. a name defined in the facade itself resolves to the facade.
|
||||
Both are checked against real modules on the current tree, so they also pin the layout the eval documents.
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from evals.codebase_navigability import bench
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def mods():
|
||||
return bench.source_modules(ROOT)
|
||||
|
||||
|
||||
def test_resolver_follows_reexport_to_defining_sibling(mods):
|
||||
cache: dict = {}
|
||||
# gateway.run re-exports names it no longer defines; each must resolve to a gateway.run_* sibling
|
||||
src = (ROOT / "gateway" / "run.py").read_text(encoding="utf-8")
|
||||
rex = bench._reexports(src)
|
||||
moved = [(n, m) for n, (m, _) in rex.items() if m.startswith("gateway.run_")]
|
||||
assert moved, "expected gateway.run to re-export from gateway.run_* siblings"
|
||||
for name, expected in moved[:20]:
|
||||
got = bench.resolve_definer(mods, name, "gateway.run", cache)
|
||||
assert got == expected, (name, got, expected)
|
||||
|
||||
|
||||
def test_resolver_keeps_facade_defined_names_on_facade(mods):
|
||||
cache: dict = {}
|
||||
src = (ROOT / "hermes_state.py").read_text(encoding="utf-8")
|
||||
own = [n for n, (_, _, node) in bench.top_level_defs(src).items() if not bench._is_alias(node)]
|
||||
assert own, "hermes_state.py should still define something at top level"
|
||||
for name in own[:20]:
|
||||
assert bench.resolve_definer(mods, name, "hermes_state", cache) == "hermes_state", name
|
||||
|
||||
|
||||
def test_tokenizer_falls_back_without_crashing(monkeypatch):
|
||||
import builtins
|
||||
real_import = builtins.__import__
|
||||
|
||||
def no_tiktoken(name, *a, **k):
|
||||
if name == "tiktoken":
|
||||
raise ImportError
|
||||
return real_import(name, *a, **k)
|
||||
|
||||
monkeypatch.setattr(builtins, "__import__", no_tiktoken)
|
||||
tok = bench._tokenizer()
|
||||
assert tok("abcdefgh") == 2
|
||||
Reference in New Issue
Block a user