a8ca904922
evals/postmortem/ turns the one-off audit behind tracking issue #103563 into something anyone with a Hermes state.db copy (and optionally rotated agent.log*) can run on their own fan-out: forensics/ common.py discovers the run tree (root = most descendants, compression-rollover children excluded so cost buckets stay disjoint), fits pricing from estimated_cost_usd, and five lanes recompute the OBSERVED figures: tokens (buckets, depth/duration shares, context reconstruction, excess-cache-write proxy, cap replay), logcalls (per-call cache behaviour from agent.log with coverage printed first; strict and loose plateau definitions reported separately), delegation (timeouts, orphaned children, polling hours, batch-join withheld child-hours, truncated summaries), tools (hardline blocks, foreground refusals, whole-file rewrites), goal_loop (nudges, parked barrier), rework (public-surface drop at PR open + post-open commit inventory). Every figure is labeled OBSERVED or MODELED. live_ab/ the per-PR A/Bs (real code paths, fake providers, temp HERMES_HOME), paths from argv. review_probes/ the independent /review's probes, credited and adapted; each reproduced a round-1 defect and the fixed head must pass it. run.py runs the offline probes against one or two checkouts and prints PASS/FAIL side by side (--live adds the ones that spend cents). tests/ synthetic-DB smoke test for the lanes and runner. On the run's DB the lanes reproduce the tracking issue's population exactly (1,394 sessions, 93,284 calls, $19,302.59; cache_write $11,159.76) and on main vs an integration checkout of the 13 PRs the runner shows every probe FAIL -> PASS (two guard-only probes pass on both, noted in run.py). The trajectories are deliberately not shipped: the DB holds 51,956 home paths, 5,341 e-mails, private IPs, chat ids and real-shaped credentials in tool output. The lane reports and recomputed JSON are in a secret gist linked from #103563.
50 lines
3.2 KiB
Python
50 lines
3.2 KiB
Python
"""Live A/B for the thinking-strip cache miss (F0). Runs a short real tool loop through AIAgent on
|
|
Fable 5.1 via Nous and prints per-call cache hit ratios from agent.log. ~10 calls, well under $1.
|
|
Arm A = current code. Arm B = HERMES_KEEP_ALL_THINKING=1 monkeypatch of _manage_thinking_signatures
|
|
that passes thinking blocks back unchanged for the Nous/Anthropic route."""
|
|
import os, sys, re, time, subprocess, json
|
|
# LIVE: real provider calls (cents). Usage: python cache_prefix_live.py <repo_root> <A|B>
|
|
os.environ.setdefault("HERMES_HOME", os.path.expanduser("~/.hermes"))
|
|
sys.path.insert(0, sys.argv[1])
|
|
arm = sys.argv[2] if len(sys.argv) > 2 else "A"
|
|
import agent.anthropic_message_convert as amc
|
|
if arm == "B":
|
|
_orig = amc._manage_thinking_signatures
|
|
def _keep_all(result, base_url, model):
|
|
# keep-all model on a signature-validating route: pass every thinking block back unchanged,
|
|
# only drop cache_control on thinking blocks and the internal flag.
|
|
for idx, m in amc._assistant_block_lists(result):
|
|
for b in m["content"]:
|
|
if amc._block_type(b) in amc._THINKING_TYPES:
|
|
b.pop("cache_control", None)
|
|
m.pop("_thinking_signature_invalidated", None)
|
|
amc._manage_thinking_signatures = _keep_all
|
|
# the adapter imports the name at call time via module attribute? verify binding
|
|
import agent.anthropic_adapter as ad
|
|
if hasattr(ad, "_manage_thinking_signatures"):
|
|
ad._manage_thinking_signatures = _keep_all
|
|
from run_agent import AIAgent
|
|
from hermes_cli.runtime_provider import resolve_runtime_provider
|
|
rt = resolve_runtime_provider(requested="nous", target_model="anthropic/claude-fable-5.1")
|
|
sid = f"f0ab_{arm}_{int(time.time())}"
|
|
ag = AIAgent(model="anthropic/claude-fable-5.1", provider="nous", base_url=rt.get("base_url"), api_key=rt.get("api_key"),
|
|
api_mode=rt.get("api_mode"), session_id=sid, quiet_mode=True,
|
|
enabled_toolsets=["file", "terminal"], platform="cli", max_iterations=12,
|
|
skip_context_files=True, skip_memory=True, reasoning_config={"enabled": True, "effort": "medium"})
|
|
task = ("In /tmp/f0ab_work (create it), do these steps ONE tool call at a time, no parallel calls: "
|
|
"1) write a.txt with 'alpha', 2) write b.txt with 'beta', 3) read a.txt, 4) read b.txt, "
|
|
"5) run `ls -la /tmp/f0ab_work`, 6) run `wc -c /tmp/f0ab_work/*`, 7) run `cat /tmp/f0ab_work/a.txt`, "
|
|
"then reply with one line: DONE.")
|
|
t0 = time.time()
|
|
r = ag.run_conversation(task)
|
|
print("final:", (r.get("final_response") or "")[:80], "| wall", round(time.time() - t0, 1), "s")
|
|
time.sleep(1)
|
|
log = subprocess.run(f"grep -h '\\[{sid}\\]' ~/.hermes/logs/agent.log | grep 'API call #'", shell=True, capture_output=True).stdout.decode("utf-8", "replace")
|
|
rows = re.findall(r"API call #(\d+): .*in=(\d+) out=(\d+) .*cache=(\d+)/(\d+) \((\d+)%\)", log)
|
|
tot_in = tot_c = 0
|
|
for n, i, o, c, ct, p in rows:
|
|
i, c = int(i), int(c); tot_in += i; tot_c += c
|
|
print(f" call {n:>2} in={i:>7} out={o:>5} cached={c:>7} ({p}%) uncached={i-c}")
|
|
print(f"ARM {arm}: calls={len(rows)} input={tot_in} cached={tot_c} uncached={tot_in-tot_c} hit={100*tot_c/max(tot_in,1):.1f}%")
|
|
print(json.dumps({"arm": arm, "calls": len(rows), "input": tot_in, "cached": tot_c}))
|