Merge origin/main into feat/plugin-catalog — reconcile with landed index/manifest-v2 tracks

This commit is contained in:
Teknium
2026-08-27 21:29:42 -07:00
7926 changed files with 1138960 additions and 479522 deletions
+18
View File
@@ -0,0 +1,18 @@
# CodeRabbit configuration.
#
# Auto-review is DISABLED repo-wide: the app was enabled at the org level on
# 2026-08-14 and immediately began reviewing every opened PR. This repo's
# merge gate is CI ("All required checks pass") plus maintainer review —
# CodeRabbit reviews carry no merge weight (protect-main ruleset requires
# 0 approving reviews), so auto-firing on the full PR firehose adds comment
# noise without gating value.
#
# The bot stays installed and summonable on demand: comment
# `@coderabbitai review` on any PR to request a one-off review, or
# `@coderabbitai ignore` to mute it on a PR it has already joined.
#
# To re-enable auto-review, flip `enabled: true` below (or delete this file —
# the app default is on).
reviews:
auto_review:
enabled: false
+11 -4
View File
@@ -40,6 +40,10 @@ ui-tui/packages/hermes-ink/dist/
# Environment files
.env
.env.*
# ...but keep the template: docker/stage2-hook.sh seeds $HERMES_HOME/.env from
# /opt/hermes/.env.example on first boot (OOF-285 — excluding it silently broke
# first-boot .env seeding and the API_SERVER_KEY generation that depends on it).
!.env.example
# IDE
.vscode/
@@ -97,12 +101,15 @@ packaging/
plans/
.plans/
# ACP registry manifest (icon + agent.json) — not consumed at runtime
acp_registry/
# Repo-level dotfiles that are git-only or dev-tooling config
.env.example
.envrc
.gitattributes
.hadolint.yaml
.mailmap
# Repo-root debug/export artifacts — must never reach image layers (COPY . .)
/log.txt
/sqlite_leak_fix.png
/*.png.bak
/default.tar.gz
/*.tar.gz
+8
View File
@@ -104,6 +104,14 @@
# $10/month subscription. Get your key at: https://opencode.ai/auth
# OPENCODE_GO_API_KEY=
# =============================================================================
# LLM PROVIDER (OpenCode Free)
# =============================================================================
# OpenCode Free provides keyless free models (Ox Alpha / x-preview-f-free,
# big-pickle, etc.). NO env var and NO account needed — requests are sent
# anonymously (the free tier rejects any unrecognized Authorization header).
# Select it with `hermes model` or `/model free`.
# =============================================================================
# LLM PROVIDER (Hugging Face Inference Providers)
# =============================================================================
+24
View File
@@ -8,3 +8,27 @@ web/package-lock.json linguist-generated=true
Dockerfile text eol=lf
*.dockerfile text eol=lf
docker/entrypoint.sh text eol=lf
# Enforce LF for all source/text files. Windows editors and tools default to
# CRLF; without normalization a Windows contributor's edit turns into a
# whole-file phantom diff (every line "changed" by its ending), breaks
# string-match patch tooling, and pollutes review. `text` normalizes to LF
# at check-in; `eol=lf` also checks out as LF so the working tree matches
# the index on every platform. PowerShell files are the deliberate
# exception (PS 5.1 tooling expects CRLF).
*.py text eol=lf
*.ts text eol=lf
*.tsx text eol=lf
*.js text eol=lf
*.mjs text eol=lf
*.cjs text eol=lf
*.jsx text eol=lf
*.json text eol=lf
*.yaml text eol=lf
*.yml text eol=lf
*.toml text eol=lf
*.md text eol=lf
*.css text eol=lf
*.html text eol=lf
*.svg text eol=lf
*.ps1 text eol=crlf
+9
View File
@@ -0,0 +1,9 @@
# actionlint knows only GitHub-hosted runner labels. An org admin names the
# larger runners. Each one therefore reads as "unknown runner label" and hides
# the real findings, unless this file declares it.
self-hosted-runner:
labels:
- ubuntu-latest-96-core
- ubuntu-latest-32-core
- ubuntu-latest-32-arm-core
- windows-latest-32-core
+21
View File
@@ -15,12 +15,21 @@ outputs:
python:
description: Run Python tests / ruff / ty / windows-footguns.
value: ${{ steps.classify.outputs.python }}
python_prod:
description: Python changes outside tests/ — gates product jobs (Desktop E2E, Docker).
value: ${{ steps.classify.outputs.python_prod }}
frontend:
description: Run the TypeScript testing matrix + desktop build.
value: ${{ steps.classify.outputs.frontend }}
docker_meta:
description: Docker setup and meta files have changed.
value: ${{ steps.classify.outputs.docker_meta }}
docker:
description: Files included in the docker image have changed.
value: ${{ steps.classify.outputs.docker }}
nix:
description: Run `nix flake check` (flake inputs, or any product Python change).
value: ${{ steps.classify.outputs.nix }}
site:
description: Build the Docusaurus docs site.
value: ${{ steps.classify.outputs.site }}
@@ -30,15 +39,27 @@ outputs:
deps:
description: Check pyproject.toml dependency upper bounds.
value: ${{ steps.classify.outputs.deps }}
uv_lock:
description: Run `uv lock --check` (pyproject.toml / uv.lock changes only).
value: ${{ steps.classify.outputs.uv_lock }}
npm_lock:
description: Post/update the semantic package-lock.json diff PR comment.
value: ${{ steps.classify.outputs.npm_lock }}
installer:
description: Run the PowerShell installer tests on a Windows runner.
value: ${{ steps.classify.outputs.installer }}
rust:
description: Run `cargo test` for the Tauri bootstrap installer.
value: ${{ steps.classify.outputs.rust }}
mcp_catalog:
description: Require MCP catalog security review label.
value: ${{ steps.classify.outputs.mcp_catalog }}
ci_review:
description: Require CI-sensitive file review label.
value: ${{ steps.classify.outputs.ci_review }}
ci_review_files:
description: JSON list of CI-sensitive files changed by the pull request.
value: ${{ steps.classify.outputs.ci_review_files }}
runs:
using: composite
+18 -8
View File
@@ -5,24 +5,32 @@ description: >-
5,000 req/hr per installation (vs 1,000 for the default GITHUB_TOKEN)
and are scoped to the App's installation permissions, not a user account.
Falls back to the built-in GITHUB_TOKEN when APP_CLIENT_ID is not set —
this happens on fork PRs where repo secrets are unavailable. The fallback
ensures classification, timings, and review comments still work on
forks (with the lower GITHUB_TOKEN rate limit).
Callers must source App credentials from a protected, main-only environment.
Never pass an App private key to a pull_request job, a local action, or a
reusable workflow resolved from an untrusted PR ref. The fallback keeps a
trusted caller functional when its protected environment is misconfigured.
Composite actions cannot access the secrets context directly, so the
calling workflow must pass secrets.APP_CLIENT_ID and secrets.APP_PRIVATE_KEY
as inputs. When both are empty (fork PRs), the fallback fires.
Composite actions cannot access contexts directly, so callers pass the
public vars.APP_CLIENT_ID and protected secrets.APP_PRIVATE_KEY as inputs.
When the private key is empty, the fallback fires.
inputs:
client-id:
description: GitHub App Client ID. Pass secrets.APP_CLIENT_ID from the calling workflow.
description: GitHub App Client ID. Pass vars.APP_CLIENT_ID from the calling workflow.
required: false
default: ''
private-key:
description: GitHub App private key PEM. Pass secrets.APP_PRIVATE_KEY from the calling workflow.
required: false
default: ''
owner:
description: GitHub App installation owner. Empty scopes the token to the current repository.
required: false
default: ''
repositories:
description: Comma- or newline-separated repositories to scope within the installation owner.
required: false
default: ''
outputs:
token:
@@ -51,6 +59,8 @@ runs:
with:
client-id: ${{ inputs.client-id }}
private-key: ${{ inputs.private-key }}
owner: ${{ inputs.owner }}
repositories: ${{ inputs.repositories }}
- name: Fall back to GITHUB_TOKEN
id: fallback
Binary file not shown.

Before

Width:  |  Height:  |  Size: 36 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 40 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 33 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 36 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 138 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 148 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 428 KiB

+141
View File
@@ -0,0 +1,141 @@
// Run every workspace check at the same time and report all failures.
//
// The unit of work is a CHECK, and not a workspace. A package that declares
// `check:*` sub-scripts gives one unit for each sub-script. A package with a
// plain `check` gives that. This is the same selection rule the old CI matrix
// used, so the set of commands is unchanged. Only the schedule is different.
//
// This is not `npm run --ws check`, because that command is serial and stops
// at the first workspace that fails. This runs every unit and fails at the
// end with the full list.
//
// The output of each unit goes to a buffer and prints on completion inside a
// group that collapses. Children that write to one stdout together interleave
// their lines, and a failure is then hard to read.
//
// This also runs on a laptop: `node .github/scripts/run-workspace-checks.mjs`.
// `--concurrency N` sets the limit. `--list` prints the units and exits.
import { execFileSync, spawn } from 'node:child_process'
import { availableParallelism } from 'node:os'
const IS_CI = Boolean(process.env.GITHUB_ACTIONS)
const NPM = process.platform === 'win32' ? 'npm.cmd' : 'npm'
/** @returns {{pkg: string, script: string}[]} */
function discoverUnits() {
const raw = execFileSync(NPM, ['query', '.workspace'], {
encoding: 'utf-8',
shell: process.platform === 'win32',
})
/** @type {{location: string, scripts?: Record<string,string>}[]} */
const pkgs = JSON.parse(raw)
/** @type {{pkg: string, script: string}[]} */
const units = []
for (const pkg of pkgs) {
const scripts = pkg.scripts || {}
const subs = Object.keys(scripts).filter((s) => /^check:.+$/.test(s))
if (subs.length > 0) {
for (const script of subs) units.push({ pkg: pkg.location, script })
} else if (scripts.check) {
units.push({ pkg: pkg.location, script: 'check' })
}
}
return units
}
/** @param {{pkg: string, script: string}} unit */
function runUnit(unit) {
return new Promise((resolve) => {
const started = Date.now()
const child = spawn(NPM, ['run', '--prefix', unit.pkg, unit.script], {
// Buffer, and do not inherit. Children that share one stdout
// interleave their lines, and a failure is then hard to read.
stdio: ['ignore', 'pipe', 'pipe'],
shell: process.platform === 'win32',
})
/** @type {Buffer[]} */
const chunks = []
child.stdout.on('data', (c) => chunks.push(c))
child.stderr.on('data', (c) => chunks.push(c))
child.on('error', (err) => {
chunks.push(Buffer.from(`failed to spawn: ${err.message}\n`))
resolve({ unit, code: 1, output: Buffer.concat(chunks).toString('utf-8'), ms: Date.now() - started })
})
child.on('close', (code) => {
resolve({
unit,
code: code ?? 1,
output: Buffer.concat(chunks).toString('utf-8'),
ms: Date.now() - started,
})
})
})
}
async function main() {
const argv = process.argv.slice(2)
const units = discoverUnits()
if (units.length === 0) {
console.error(
'::error::No workspace package declares a check script — refusing to report green having run nothing.',
)
process.exit(1)
}
if (argv.includes('--list')) {
for (const u of units) console.log(`${u.pkg} :: ${u.script}`)
return
}
const flagIdx = argv.indexOf('--concurrency')
const concurrency = Math.max(
1,
flagIdx !== -1 ? Number(argv[flagIdx + 1]) : Math.min(units.length, availableParallelism()),
)
console.log(`running ${units.length} checks, up to ${concurrency} at a time:`)
for (const u of units) console.log(` ${u.pkg} :: ${u.script}`)
console.log('')
const queue = [...units]
/** @type {{unit: {pkg: string, script: string}, code: number, output: string, ms: number}[]} */
const results = []
async function worker() {
for (;;) {
const unit = queue.shift()
if (!unit) return
const res = await runUnit(unit)
results.push(res)
const label = `${res.unit.pkg} :: ${res.unit.script}`
const secs = (res.ms / 1000).toFixed(1)
const status = res.code === 0 ? 'PASS' : 'FAIL'
if (IS_CI) console.log(`::group::${status} ${label} (${secs}s)`)
else console.log(`----- ${status} ${label} (${secs}s) -----`)
process.stdout.write(res.output.endsWith('\n') ? res.output : res.output + '\n')
if (IS_CI) console.log('::endgroup::')
}
}
await Promise.all(Array.from({ length: Math.min(concurrency, units.length) }, worker))
const failed = results.filter((r) => r.code !== 0)
console.log('\n=== summary ===')
for (const r of [...results].sort((a, b) => b.ms - a.ms)) {
console.log(
` ${r.code === 0 ? 'pass' : 'FAIL'} ${(r.ms / 1000).toFixed(1).padStart(6)}s ${r.unit.pkg} :: ${r.unit.script}`,
)
}
if (failed.length > 0) {
for (const r of failed) console.error(`::error::${r.unit.pkg} :: ${r.unit.script} failed`)
console.error(`::error::${failed.length} of ${results.length} checks failed`)
process.exit(1)
}
console.log(`\nall ${results.length} checks passed`)
}
await main()
+80
View File
@@ -0,0 +1,80 @@
name: CI review comment
# Live-updating PR review comment.
#
# The poller runs for up to 40 minutes.
# This run lives in its own workflow.
# A run stays in progress until its last job ends, and GitHub refuses
# ``gh run rerun`` on a run that is in progress.
#
# ``workflow_run`` starts this when CI starts. It always reads the workflow
# and the scripts from the default branch, never from the PR head.
# It makes a write token safe here.
#
# The poller reads job results through the API. Thus it watches the CI run
# and the separate docker run, and it depends on neither.
on:
workflow_run:
workflows: [CI]
# ``in_progress``, not ``requested``: a first-time contributor's run
# sits in ``action_required`` until a maintainer approves it, and
# ``requested`` fires at creation — the poller would wait out its
# whole timeout on a run that never starts. ``in_progress`` fires
# when the run actually starts, and it also fires on re-runs, which
# ``requested`` does not.
types: [in_progress]
permissions:
contents: read
actions: read
pull-requests: write
# One poller per CI run. A new push starts a new CI run, and its poller
# cancels the poller of the run that GitHub superseded. The group keys
# on the head repository too: fork PRs often share a branch name
# (``main``, ``patch-1``), and two PRs must not cancel each other.
concurrency:
group: ci-review-comment-${{ github.event.workflow_run.head_repository.full_name }}-${{ github.event.workflow_run.head_branch }}
cancel-in-progress: true
jobs:
comment:
name: CI review comment (live)
# Fork PRs get no comment: the poller needs a write token, and the
# ``pull_requests`` payload is empty for a fork run.
if: >-
github.event.workflow_run.event == 'pull_request' &&
github.event.workflow_run.head_repository.full_name == github.repository
runs-on: ubuntu-latest
timeout-minutes: 60
steps:
- name: Checkout trusted default branch
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
ref: ${{ github.event.repository.default_branch }}
persist-credentials: false
- name: Run live comment poller
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
# The CI run to report on — not this run. The name must not be
# GITHUB_RUN_ID: the runner sets the GITHUB_* defaults itself and
# ignores an env: override, so that name silently resolves to THIS
# run. The poller then watches itself, which stays in_progress for
# as long as the poller runs, and it waits out its whole timeout.
CI_RUN_ID: ${{ github.event.workflow_run.id }}
# Sibling runs for the same commit that the comment also covers,
# one workflow name per line (a name can contain a comma).
# The poller resolves each name to its runs through the API.
WATCH_WORKFLOWS: |
Docker Build, Test, and Publish
PR_NUMBER: ${{ github.event.workflow_run.pull_requests[0].number }}
RUN_URL: ${{ github.event.workflow_run.html_url }}
# Commit info for the review comment header.
COMMIT_SHA: ${{ github.event.workflow_run.head_sha }}
COMMIT_MESSAGE: ${{ github.event.workflow_run.head_commit.message }}
run: |
python3 -u scripts/ci/live_comment.py \
--interval 15 \
--timeout 3000
@@ -9,6 +9,10 @@ name: CI
# definitions, matrices, and concurrency settings. They no longer have
# ``push:`` / ``pull_request:`` triggers of their own — everything flows
# through this file.
#
# SECURITY: this workflow runs PR-controlled actions, workflows, and code.
# Do not add ``secrets: inherit`` or GitHub App credentials here. Trusted
# main-only automation uses protected environments in its own workflows.
on:
pull_request:
@@ -20,7 +24,6 @@ permissions:
pull-requests: write # needed by lint (PR comment) + supply-chain review_status
actions: read # needed by osv-scanner (SARIF upload)
security-events: write # needed by osv-scanner (SARIF upload)
packages: write # needed by docker build
concurrency:
group: ci-${{ github.ref }}
@@ -35,33 +38,32 @@ jobs:
detect:
name: Detect affected areas
runs-on: ubuntu-latest
timeout-minutes: 10
timeout-minutes: 1
outputs:
python: ${{ steps.classify.outputs.python }}
python_prod: ${{ steps.classify.outputs.python_prod }}
frontend: ${{ steps.classify.outputs.frontend }}
site: ${{ steps.classify.outputs.site }}
scan: ${{ steps.classify.outputs.scan }}
deps: ${{ steps.classify.outputs.deps }}
uv_lock: ${{ steps.classify.outputs.uv_lock }}
npm_lock: ${{ steps.classify.outputs.npm_lock }}
installer: ${{ steps.classify.outputs.installer }}
rust: ${{ steps.classify.outputs.rust }}
docker_meta: ${{ steps.classify.outputs.docker_meta }}
mcp_catalog: ${{ steps.classify.outputs.mcp_catalog }}
ci_review: ${{ steps.classify.outputs.ci_review }}
ci_review_files: ${{ steps.classify.outputs.ci_review_files }}
event_name: ${{ github.event_name }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Get GitHub App token
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ secrets.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- name: Detect affected areas
id: classify
uses: ./.github/actions/detect-changes
with:
# The get-app-token composite action falls back to GITHUB_TOKEN
# on fork PRs where APP_ID is unavailable.
github-token: ${{ steps.app-token.outputs.token }}
sparse-checkout: scripts/ci/classify_changes.py
sparse-checkout-cone-mode: false
github-token: ${{ github.token }}
# ─────────────────────────────────────────────────────────────────────
# Lane-gated sub-workflows. Each runs in parallel after detect finishes.
@@ -72,9 +74,16 @@ jobs:
needs: detect
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/tests.yml
with:
slice_count: 8
secrets: inherit
# macOS + Windows lanes. The main `tests` lane above is Linux-only, and
# the OS-marked tests it collects are skipped there by design (see the
# `_OS_MARKS` comment in tests/conftest.py) — this is where they run.
# Same `python` lane gate: if no Python changed, neither runs.
tests-os:
name: OS-specific tests
needs: detect
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/tests-os.yml
lint:
name: Python lints
@@ -83,19 +92,44 @@ jobs:
uses: ./.github/workflows/lint.yml
with:
event_name: ${{ needs.detect.outputs.event_name }}
secrets: inherit
js-tests:
name: JS & TS checks
needs: detect
if: needs.detect.outputs.frontend == 'true'
uses: ./.github/workflows/js-tests.yml
secrets: inherit
installer-tests:
name: Installer tests
needs: detect
# Windows-only, and only for PRs that touch install.ps1 or its tests.
if: needs.detect.outputs.installer == 'true'
uses: ./.github/workflows/installer-tests.yml
rust-tests:
name: Rust tests
needs: detect
# Only for PRs that touch a Rust crate. `.rs` is under apps/, so these
# changes used to run the TypeScript matrix and nothing that compiles them.
if: needs.detect.outputs.rust == 'true'
uses: ./.github/workflows/rust-tests.yml
e2e-desktop:
name: Desktop E2E
needs: detect
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true'
# python_prod (not python): the Playwright suite exercises the built app
# + `hermes serve` backend, which never import anything under tests/.
# Tests-only PRs (~17% of commits) skip this 5-minute job — the longest
# single job in the workflow — while still running the full pytest lanes.
#
# ⛔ TEMPORARILY DISABLED (Aug 2, 2026, Teknium) — the suite is red on
# every PR and on main itself since the Aug 1 night engines/npm churn
# (#76499 → #76562 → #76575): the mock-backend Electron window never
# gets a title, so boot/chat/setup/interim specs all fail identically
# regardless of the PR's diff (verified on #76573 and the docs-only
# #76582). Tracking issue: #76627 (assigned: Ari). To re-enable,
# delete the `false &&` below — nothing else changed.
if: ${{ false && (needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true') }}
uses: ./.github/workflows/e2e-desktop.yml
docs-site:
@@ -103,48 +137,46 @@ jobs:
needs: detect
if: needs.detect.outputs.site == 'true'
uses: ./.github/workflows/docs-site-checks.yml
secrets: inherit
history-check:
name: Deny unrelated histories
needs: detect
if: needs.detect.outputs.event_name == 'pull_request'
uses: ./.github/workflows/history-check.yml
secrets: inherit
contributor-check:
name: Check contributors
needs: detect
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/contributor-check.yml
secrets: inherit
uv-lockfile:
name: Check uv.lock
needs: detect
# Gated: `uv lock --check` re-resolves the whole dependency graph against
# PyPI, so on every PR it spent a network round-trip — and, on a registry
# blip, a blocking red X — for diffs that cannot desync the lockfile
# (docs, frontend, prose). Only pyproject.toml / uv.lock can. A
# `.github/` change still forces it on via the classifier's fail-open.
if: needs.detect.outputs.uv_lock == 'true'
uses: ./.github/workflows/uv-lockfile-check.yml
secrets: inherit
infographic-check:
name: Check no committed infographics
needs: detect
uses: ./.github/workflows/infographic-check.yml
lockfile-diff:
name: package-lock.json diff
needs: detect
if: needs.detect.outputs.event_name == 'pull_request' && needs.detect.outputs.npm_lock == 'true'
uses: ./.github/workflows/lockfile-diff.yml
secrets: inherit
docker-lint:
name: Lint Docker scripts
needs: detect
if: needs.detect.outputs.docker_meta == 'true'
uses: ./.github/workflows/docker-lint.yml
secrets: inherit
docker:
name: Build&Test Docker image
needs: detect
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true' || needs.detect.outputs.docker_meta == 'true'
uses: ./.github/workflows/docker.yml
secrets: inherit
supply-chain:
name: Supply-chain scan
@@ -163,89 +195,13 @@ jobs:
uses: ./.github/workflows/review-labels.yml
with:
ci_review: ${{ needs.detect.outputs.ci_review == 'true' }}
ci_review_files: ${{ needs.detect.outputs.ci_review_files }}
mcp_catalog: ${{ needs.detect.outputs.mcp_catalog == 'true' }}
supply_chain: ${{ needs.supply-chain.outputs.critical_findings == 'true' }}
secrets: inherit
osv-scanner:
name: OSV scan
uses: ./.github/workflows/osv-scanner.yml
secrets: inherit
# ─────────────────────────────────────────────────────────────────────
# Live-updating PR review comment.
#
# A single ``comment-live`` job polls the GitHub Actions API every 15s
# for job statuses in this run, re-assembles the review comment from
# whatever results are available, and upserts it via the
# ``<!-- hermes-ci-review-bot -->`` marker.
#
# The poller exits when all non-infra jobs are completed (or on
# timeout). ci-timings' review_status is picked up automatically when
# its artifact becomes available — the poller downloads and merges it.
# ─────────────────────────────────────────────────────────────────────
comment-live:
name: CI review comment (live)
needs: [detect, review-labels, lockfile-diff, supply-chain, osv-scanner, uv-lockfile, history-check, contributor-check]
if: always() && github.event_name == 'pull_request' && github.event.pull_request.head.repo.fork != true
runs-on: ubuntu-latest
timeout-minutes: 40
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Run live comment poller
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
GITHUB_REPOSITORY: ${{ github.repository }}
GITHUB_RUN_ID: ${{ github.run_id }}
PR_NUMBER: ${{ github.event.pull_request.number }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
# Commit info for the review comment header.
COMMIT_SHA: ${{ github.event.pull_request.head.sha }}
COMMIT_MESSAGE: ${{ github.event.pull_request.head.commit.message }}
COMMIT_URL: ${{ github.server_url }}/${{ github.repository }}/pull/${{ github.event.pull_request.number }}/commits/${{ github.event.pull_request.head.sha }}
# Structured review statuses from workflow_call jobs.
# Each job outputs a JSON array of {source, results: [...]} objects
# that the assembler renders directly — no hardcoded job-name
# matching. We merge all available outputs into one array.
REVIEW_STATUSES: ${{ toJSON(needs.*.outputs.review_status) }}
run: |
set -uo pipefail
# REVIEW_STATUSES is a JSON array of strings (some may be empty
# when a job was skipped). Parse each string and merge into one
# flat array for the assembler.
python3 - <<'PYEOF'
import json, os, sys
raw = os.environ.get("REVIEW_STATUSES", "")
merged = []
if raw:
try:
arr = json.loads(raw)
except (json.JSONDecodeError, TypeError):
arr = []
for item in arr:
if not item:
continue
try:
statuses = json.loads(item)
except (json.JSONDecodeError, TypeError):
continue
if isinstance(statuses, list):
merged.extend(statuses)
# Write merged array to a temp file the poller reads.
with open("/tmp/review_statuses.json", "w") as f:
json.dump(merged, f)
print(f"Merged {len(merged)} review status entries")
PYEOF
python3 scripts/ci/live_comment.py \
--interval 15 \
--timeout 2100 \
--review-statuses-file /tmp/review_statuses.json
# ─────────────────────────────────────────────────────────────────────
# Gate: runs after everything. ``if: always()`` ensures it reports a
@@ -262,8 +218,11 @@ jobs:
needs:
- detect
- tests
- tests-os
- lint
- js-tests
- installer-tests
- rust-tests
- e2e-desktop
- docs-site
- history-check
@@ -274,9 +233,10 @@ jobs:
- supply-chain
- review-labels
- osv-scanner
# comment-live is a polling job — it doesn't block the gate.
# we don't require docker to pass rn because it's so slow lol
# - docker
# The image build runs in its own workflow (docker.yml) and reports
# its own check. It was never required here, because it is too slow
# to block a merge. A separate run also stops it from holding this
# run open. That is what blocked ``gh run rerun``.
if: always()
runs-on: ubuntu-latest
timeout-minutes: 10
@@ -313,12 +273,13 @@ jobs:
# report with a gantt chart + per-step breakdown. The report is uploaded
# as an artifact and a markdown summary is written to $GITHUB_STEP_SUMMARY.
#
# The live comment poller picks up ci-timings' completion automatically —
# it reads review-status.json from the artifact when the job finishes.
# The live comment poller dynamically fetches all review-status-* artifacts
# across the orchestrator and sub-workflow runs every cycle, so its link
# points straight at that report.
# ─────────────────────────────────────────────────────────────────────
ci-timings:
name: CI timing report
needs: [all-checks-pass, docker]
needs: [all-checks-pass]
if: always()
runs-on: ubuntu-latest
timeout-minutes: 10
@@ -326,13 +287,6 @@ jobs:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Get GitHub App token
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ secrets.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- name: Restore baseline cache (PR only)
if: github.event_name == 'pull_request'
uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
@@ -346,38 +300,52 @@ jobs:
- name: Collect timings and generate report
env:
# Forks get no repo secrets (AUTOFIX_BOT_PAT is empty); fall back to
# the built-in read-only token so the timings API read still works
# there instead of hard-failing this advisory job on every fork PR.
# The get-app-token composite action falls back to GITHUB_TOKEN
# on fork PRs where APP_ID is unavailable.
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
GITHUB_TOKEN: ${{ github.token }}
run: |
python3 scripts/ci/timings_report.py \
--baseline ci-timings-baseline.json \
--output ci-timings-report.html \
--json-out ci-timings.json \
--summary-out ci-timings-summary.md \
--review-status-out review-status.json
--summary-out ci-timings-summary.md
- name: Upload HTML report + review status
- name: Upload HTML report
# Advisory report — artifact-service blips must not fail the job.
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
id: ci-timings-artifact
id: ci-timings-html
with:
name: ci-timings-report
path: |
ci-timings-report.html
review-status.json
path: ci-timings-report.html
retention-days: 14
- name: Build linked review status
if: hashFiles('ci-timings.json') != ''
env:
CI_TIMINGS_REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url }}
run: |
python3 scripts/ci/timings_report.py \
--from-json ci-timings.json \
--baseline ci-timings-baseline.json \
--review-status-out review-status.json \
--review-status-only
- name: Upload review status
if: hashFiles('review-status.json') != ''
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: review-status-ci-timings
path: review-status.json
retention-days: 14
- name: Output summary
env:
REPORT_URL: ${{ steps.ci-timings-artifact.outputs.artifact-url}}
REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url}}
run: |
echo "# CI Timing report" >> "$GITHUB_STEP_SUMMARY"
echo "[View the full interactive report]($REPORT_URL)" >> "$GITHUB_STEP_SUMMARY"
{
echo "# CI Timing report"
echo "[View the full interactive report]($REPORT_URL)"
} >> "$GITHUB_STEP_SUMMARY"
cat ci-timings-summary.md >> "$GITHUB_STEP_SUMMARY"
- name: Save baseline cache (main only)
+18 -1
View File
@@ -33,6 +33,7 @@ jobs:
if [ -z "$NEW_EMAILS" ]; then
echo "No new commits to check."
echo "review_status=[]" >> "$GITHUB_OUTPUT"
echo "review_status=[]" > review-status.json
exit 0
fi
@@ -67,6 +68,8 @@ jobs:
echo -e "$MISSING"
echo ""
echo "Add a mapping file (do NOT edit AUTHOR_MAP in release.py):"
echo " python3 scripts/audit_pr_attribution.py --fix # auto-resolve + create files"
echo "or manually:"
echo -e "$MISSING" | while read -r line; do
email=$(echo "$line" | sed 's/^ *//' | cut -d' ' -f1)
[ -z "$email" ] && continue
@@ -78,14 +81,28 @@ jobs:
# Emit review_status for unmapped emails
DETAIL=$(echo -e "$MISSING" | sed '/^$/d; s/^ //')
HOW_TO_FIX=$'Add mappings to scripts/release.py AUTHOR_MAP:\n```\n"<email>": "<github-username>",\n```\nTo find the GitHub username for an email:\n```\ngh api \'search/users?q=EMAIL+in:email\' --jq \'.items[0].login\'\n```\n'
HOW_TO_FIX=$'Run from the PR branch:\n```\npython3 scripts/audit_pr_attribution.py --fix\ngit add contributors && git commit -m "chore: map contributor emails" && git push\n```\nOr map one email manually (do NOT edit AUTHOR_MAP in release.py):\n```\npython3 scripts/add_contributor.py <email> <github-username>\n```\nTo find the GitHub username for an email:\n```\ngh api \'search/users?q=EMAIL+in:email\' --jq \'.items[0].login\'\n```\n'
REVIEW_STATUS=$(jq -nc \
--arg detail "$DETAIL" \
--arg how_to_fix "$HOW_TO_FIX" \
'[{"source":"contributor attribution","results":[{"kind":"action_required","title":"Unmapped contributor email(s)","summary":"New contributor email(s) are not in AUTHOR_MAP.","detail":$detail,"how_to_fix":$how_to_fix}]}]')
echo "review_status=$REVIEW_STATUS" >> "$GITHUB_OUTPUT"
echo "review_status=$REVIEW_STATUS" > review-status.json
exit 1
else
echo "✅ All contributor emails are mapped."
echo "review_status=[]" >> "$GITHUB_OUTPUT"
echo "review_status=[]" > review-status.json
fi
- name: Upload review status artifact
if: always() && steps.check-emails.outcome != 'skipped'
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: review-status-contributor-check
path: review-status.json
retention-days: 1
overwrite: true
if-no-files-found: ignore
+61 -2
View File
@@ -60,15 +60,19 @@ jobs:
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ secrets.APP_CLIENT_ID }}
client-id: ${{ vars.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
cache-dependency-path: website/package-lock.json
- name: grab npm 12
run: |
npm i -g npm@12
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
with:
python-version: '3.11'
@@ -188,6 +192,61 @@ jobs:
cp website/build/llms-full.txt _site/llms-full.txt
fi
# Pages serves exactly the newest artifact, so each deploy used to delete
# the previous build's content-hashed JS/CSS while edge caches (Vercel →
# Fastly, max-age=300 + stale-while-revalidate=3600) kept serving HTML
# that referenced it — every asset request 404'd for up to ~65 minutes
# after each deploy. With push-triggered deploys landing every ~15-30
# minutes, the docs were in that broken window most of the day (search,
# being pure client JS, died first). Fix: keep a rolling pool of prior
# builds' hashed assets and union-merge it into every artifact so stale
# HTML keeps resolving. Hashed filenames are content-addressed, so a
# collision is by definition the identical file — the merge never
# overwrites current-build output (cp --update=none).
- name: Restore asset retention pool
uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
with:
path: _asset_retention
key: docs-asset-retention-${{ github.run_id }}
restore-keys: |
docs-asset-retention-
- name: Merge retained assets from previous deploys
run: |
set -euo pipefail
ASSET_DIRS="assets zh-Hans/assets"
mkdir -p _asset_retention
# 1) Add this build's hashed assets to the pool (fresh mtimes, so
# assets still shipped by current builds never age out).
for d in $ASSET_DIRS; do
if [ -d "_site/docs/$d" ]; then
mkdir -p "_asset_retention/$d"
cp -a "_site/docs/$d/." "_asset_retention/$d/"
fi
done
# 2) Drop pool entries no build has produced for 14 days — far
# beyond any edge-cache or open-tab horizon.
find _asset_retention -type f -mtime +14 -delete
find _asset_retention -type d -empty -delete
# 3) Union-merge the pool into the artifact; --update=none keeps the
# current build authoritative for any path it produced.
for d in $ASSET_DIRS; do
if [ -d "_asset_retention/$d" ]; then
mkdir -p "_site/docs/$d"
cp -R --update=none "_asset_retention/$d/." "_site/docs/$d/"
fi
done
echo "retention pool:" && du -sh _asset_retention
echo "deployed assets:" && du -sh _site/docs/assets
- name: Save asset retention pool
# Always save under a fresh key (caches are immutable); restore-keys
# prefix matching picks the newest on the next run.
uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
with:
path: _asset_retention
key: docs-asset-retention-${{ github.run_id }}
- name: Upload artifact
uses: actions/upload-pages-artifact@56afc609e74202658d3ffba0e8f6dda462b719fa # v3
with:
+167 -51
View File
@@ -1,9 +1,19 @@
name: Docker Build, Test, and Publish
on:
# This workflow owns its own triggers. ci.yml does not call it.
# A reusable-workflow call eeps the caller run in progress for that full time.
# GitHub refuses ``gh run rerun`` on a run that is still in progress.
# Thus one slow advisory job blocked every rerun of the fast required jobs. A separate
# run reruns and cancels independently.
#
# Trusted main pushes resolve the environment-scoped Docker Hub secrets in
# this same workflow, never across a workflow boundary.
pull_request:
push:
branches: [main]
release:
types: [published]
workflow_call:
permissions:
contents: read
@@ -20,20 +30,60 @@ env:
IMAGE_NAME: nousresearch/hermes-agent
jobs:
# Build, test, and optionally push the image for each architecture.
# Classify the PR's changed files. ci.yml used to gate the docker call on
# its own ``detect`` outputs; now that this workflow triggers itself, it
# runs the same composite action. On push and release the classifier fails
# open (every lane true), so post-merge validation is never weakened.
detect:
name: Detect affected areas
runs-on: ubuntu-latest
timeout-minutes: 10
outputs:
build: ${{ steps.gate.outputs.build }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Detect affected areas
id: classify
uses: ./.github/actions/detect-changes
with:
github-token: ${{ github.token }}
- name: Decide whether to build
id: gate
env:
# The docker lane derives from python_prod (not python: the image
# copies installed code, never tests/, so tests-only PRs skip the
# build), frontend and docker_meta. classify_changes.py owns the
# formula so this gate and the nix lane cannot drift apart.
DOCKER: ${{ steps.classify.outputs.docker }}
run: |
set -euo pipefail
if [ "$DOCKER" = "true" ]; then
echo "build=true" >> "$GITHUB_OUTPUT"
else
echo "build=false" >> "$GITHUB_OUTPUT"
fi
# Build and test the image for each architecture. This job runs PR code,
# so it must remain secret-free. Publishing happens in the separate,
# protected publish job after these tests pass.
build:
if: github.repository == 'NousResearch/hermes-agent'
needs: [detect]
if: github.repository == 'NousResearch/hermes-agent' && needs.detect.outputs.build == 'true'
strategy:
fail-fast: false
matrix:
include:
- arch: amd64
runner: ubuntu-latest
runner: ubuntu-latest-32-core
platform: linux/amd64
cache-from: type=gha,scope=docker-amd64
cache-to: type=gha,mode=max,scope=docker-amd64
# arm64 builds on the native arm64 larger runner. A build of
# linux/arm64 on an x64 host uses emulation.
- arch: arm64
runner: ubuntu-24.04-arm
runner: ubuntu-latest-32-arm-core
platform: linux/arm64
cache-from: type=gha,scope=docker-arm64
cache-to: type=gha,mode=max,scope=docker-arm64
@@ -44,7 +94,19 @@ jobs:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Retry once on transient Docker Hub / buildkit pull failures
# (connection reset, auth token timeout, rate limiting). The action
# generates a unique builder name per invocation so the retry doesn't
# collide with the failed first attempt. A genuine persistent failure
# still fails the job — only the first attempt has continue-on-error.
# Refs: docker/setup-buildx-action#510
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
# Build once, load into the local daemon for testing. Cached
@@ -62,49 +124,6 @@ jobs:
cache-from: ${{ matrix.cache-from }}
cache-to: ${{ (github.event_name != 'pull_request') && matrix.cache-to || '' }}
- name: Log in to Docker Hub
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
username: ${{ secrets.DOCKERHUB_USERNAME }}
password: ${{ secrets.DOCKERHUB_TOKEN }}
# Push by digest only (no tag). The merge job assembles the
# tagged manifest list. `push-by-digest=true` is docker's recommended
# pattern for multi-runner multi-platform builds.
- name: Push ${{ matrix.arch }} by digest
id: push
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
with:
context: .
file: Dockerfile
platforms: ${{ matrix.platform }}
labels: |
org.opencontainers.image.revision=${{ github.sha }}
build-args: |
HERMES_GIT_SHA=${{ github.sha }}
outputs: type=image,name=${{ env.IMAGE_NAME }},push-by-digest=true,name-canonical=true,push=true
cache-from: ${{ matrix.cache-from }}
cache-to: ${{ matrix.cache-to }}
# Write the digest to a file and upload it as an artifact so the
# merge job can stitch both per-arch digests into a manifest list.
- name: Export digest
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
run: |
mkdir -p /tmp/digests
digest="${{ steps.push.outputs.digest }}"
touch "/tmp/digests/${digest#sha256:}"
- name: Upload digest artifact
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: digest-${{ matrix.arch }}
path: /tmp/digests/*
if-no-files-found: error
retention-days: 1
# Run the docker-integration test suite against the freshly-built
# image already loaded into the local daemon (`:test`).
@@ -122,9 +141,16 @@ jobs:
# ---------------------------------------------------------------------
- name: Install uv (for docker tests)
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
# raw.githubusercontent.com every job; transient fetch failures
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
version: "0.9.28"
- name: Set up Python 3.11 (for docker tests)
run: uv python install 3.11
uses: ./.github/actions/retry
with:
command: uv python install 3.11
- name: Install Python dependencies (for docker tests)
# ``dev`` extra pulls in pytest, pytest-asyncio —
@@ -145,7 +171,88 @@ jobs:
OPENAI_API_KEY: ""
NOUS_API_KEY: ""
run: |
scripts/run_tests.sh tests/docker/ --file-timeout 600
# Each of these tests drives a container, so the docker daemon sets
# the limit and not the processor. This caps the workers. The
# default from run_tests.sh is cpu_count*2, which starts 64
# containers together on the 32-core amd64 runner.
HERMES_TEST_WORKERS=$(nproc) scripts/run_tests.sh tests/docker/ --file-timeout 600
# ---------------------------------------------------------------------------
# Rebuild and push each architecture only after the unprivileged build/test
# matrix passes. This job is the sole Docker Hub credential boundary.
# ---------------------------------------------------------------------------
publish:
if: github.repository == 'NousResearch/hermes-agent' && (github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release')
needs: [build]
environment: container-publish
strategy:
fail-fast: false
matrix:
include:
- arch: amd64
runner: ubuntu-latest-32-core
platform: linux/amd64
cache-from: type=gha,scope=docker-amd64
cache-to: type=gha,mode=max,scope=docker-amd64
# Native arm64 for the same reason as the build matrix above.
- arch: arm64
runner: ubuntu-latest-32-arm-core
platform: linux/arm64
cache-from: type=gha,scope=docker-arm64
cache-to: type=gha,mode=max,scope=docker-arm64
runs-on: ${{ matrix.runner }}
timeout-minutes: 30
steps:
- name: Checkout trusted source
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Retry once on transient Docker Hub / buildkit pull failures.
# See build job for rationale; same pattern.
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Log in to Docker Hub
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
username: ${{ secrets.DOCKERHUB_USERNAME }}
password: ${{ secrets.DOCKERHUB_TOKEN }}
# Push by digest only (no tag). The merge job assembles the tagged
# manifest list after both architecture publishers complete.
- name: Push ${{ matrix.arch }} by digest
id: push
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
with:
context: .
file: Dockerfile
platforms: ${{ matrix.platform }}
labels: |
org.opencontainers.image.revision=${{ github.sha }}
build-args: |
HERMES_GIT_SHA=${{ github.sha }}
outputs: type=image,name=${{ env.IMAGE_NAME }},push-by-digest=true,name-canonical=true,push=true
cache-from: ${{ matrix.cache-from }}
cache-to: ${{ matrix.cache-to }}
- name: Export digest
run: |
mkdir -p /tmp/digests
digest="${{ steps.push.outputs.digest }}"
touch "/tmp/digests/${digest#sha256:}"
- name: Upload digest artifact
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: digest-${{ matrix.arch }}
path: /tmp/digests/*
if-no-files-found: error
retention-days: 1
# ---------------------------------------------------------------------------
# Stitch both per-arch digests into a single tagged multi-arch manifest.
@@ -158,8 +265,9 @@ jobs:
merge:
if: github.repository == 'NousResearch/hermes-agent' && (github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release')
runs-on: ubuntu-latest
needs: [build]
needs: [publish]
timeout-minutes: 10
environment: container-publish
steps:
- name: Download digests
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
@@ -168,7 +276,15 @@ jobs:
pattern: digest-*
merge-multiple: true
# Retry once on transient Docker Hub / buildkit pull failures.
# See build job for rationale; same pattern.
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Log in to Docker Hub
+9 -2
View File
@@ -15,10 +15,14 @@ jobs:
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
cache-dependency-path: website/package-lock.json
- name: grab npm 12
run: |
npm i -g npm@12
- name: Install website dependencies
uses: ./.github/actions/retry
with:
@@ -45,5 +49,8 @@ jobs:
working-directory: website
- name: Build Docusaurus
run: npm run build
# Build only the default (en) locale in CI — the full bilingual
# build runs in deploy-site.yml on push-to-main / release.
# Same pattern Docusaurus uses internally (build:fast --locale en).
run: npm run build:fast
working-directory: website
+103 -31
View File
@@ -2,6 +2,10 @@ name: E2E Desktop
on:
workflow_call:
outputs:
review_status:
description: Screenshot and visual-diff status for the CI review comment.
value: ${{ jobs.e2e.outputs.review_status }}
permissions:
contents: read
@@ -13,8 +17,13 @@ concurrency:
jobs:
e2e:
name: Playwright E2E (Linux)
runs-on: ubuntu-latest
# This job builds the renderer and the electron bundle, then drives a real
# Electron app under xvfb. vite, tsc and the Playwright workers all scale
# with the core count.
runs-on: ubuntu-latest-32-core
timeout-minutes: 20
outputs:
review_status: ${{ steps.review-status.outputs.review_status }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
@@ -33,8 +42,13 @@ jobs:
# ── Node ───────────────────────────────────────────────────────────
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
- name: grab npm 12
run: |
npm i -g npm@12
# Full npm ci (not --ignore-scripts): electron's postinstall
# downloads the binary we launch, and node-pty's native build is
# needed for the terminal pane.
@@ -46,19 +60,29 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pin the uv version: unpinned, setup-uv resolves "latest" by
# fetching a manifest from raw.githubusercontent.com on EVERY job —
# a transient fetch failure fails the whole job (2026-07-28 slice-5
# incident). Pinned, the binary downloads directly; no manifest hop.
version: '0.9.28'
enable-cache: true
cache-dependency-glob: |
pyproject.toml
uv.lock
- name: Set up Python 3.11
run: uv python install 3.11
uses: ./.github/actions/retry
with:
command: uv python install 3.11
- name: Install Python dependencies
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev
# ── Build desktop app ─────────────────────────────────────────────
- run: npm run --prefix apps/desktop build
# The Playwright step below runs `npm run build` before testing so
# dist/ is always fresh — no separate build step needed here.
# ── Restore visual baseline screenshots from main ──────────────────
# Baselines are generated on main (via --update-snapshots) and cached.
@@ -79,24 +103,26 @@ jobs:
# xvfb runs at a fixed 1280x1024 screen so the 1220x800 Electron
# window always has a consistent viewport for screenshot comparison.
# On main, we run with --update-snapshots to generate baselines.
# `npm run test:e2e` builds dist/ as a pretest hook so the renderer
# is always fresh — no separate build step needed.
- name: Run Playwright E2E tests
working-directory: apps/desktop
run: |
if [ "${{ github.ref_name }}" = "main" ]; then
echo "On main — generating/updating baseline screenshots"
xvfb-run -a --server-args="-screen 0 1280x1024x24" \
npm run build && xvfb-run -a --server-args="-screen 0 1280x1024x24" \
npx playwright test --reporter=list --update-snapshots
else
echo "On PR — comparing against cached baselines"
xvfb-run -a --server-args="-screen 0 1280x1024x24" \
npm run build && xvfb-run -a --server-args="-screen 0 1280x1024x24" \
npx playwright test --reporter=list
fi
env:
CI: "true"
CI: 'true'
# Ensure no real API keys leak into the test env.
OPENROUTER_API_KEY: ""
OPENAI_API_KEY: ""
NOUS_API_KEY: ""
OPENROUTER_API_KEY: ''
OPENAI_API_KEY: ''
NOUS_API_KEY: ''
# ── Save updated baselines to cache (main only) ───────────────────
- name: Save updated baselines to cache
@@ -143,6 +169,50 @@ jobs:
overwrite: true
if-no-files-found: ignore
- name: Build screenshot review status
id: review-status
if: always()
working-directory: apps/desktop
env:
RESULTS_URL: ${{ steps.upload-results.outputs.artifact-url }}
run: |
python3 ../../scripts/ci/e2e_screenshot_status.py \
--results-dir test-results \
--manifest-output /tmp/e2e-screenshot-manifest.json \
--evidence-dir /tmp/e2e-evidence \
--artifact-url "$RESULTS_URL" \
--output /tmp/e2e-review-status.json
{
echo 'review_status<<__E2E_REVIEW_STATUS__'
cat /tmp/e2e-review-status.json
echo '__E2E_REVIEW_STATUS__'
} >> "$GITHUB_OUTPUT"
cp /tmp/e2e-review-status.json review-status.json
- name: Upload review status artifact
if: always() && steps.review-status.outcome != 'skipped'
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: review-status-e2e-desktop
path: apps/desktop/review-status.json
retention-days: 1
overwrite: true
if-no-files-found: ignore
# The trusted workflow_run publisher consumes only this flat, bounded
# artifact. It turns selected images into GitHub attachment URLs; it
# never checks out or runs this PR's code.
- name: Upload inline E2E evidence
if: always() && github.ref_name != 'main'
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: e2e-evidence-${{ github.sha }}
path: /tmp/e2e-evidence
retention-days: 14
overwrite: true
if-no-files-found: error
# ── Generate step summary with visual diff info ───────────────────
# Parse the JSON report + scan for diff images, then post a summary
# to the GitHub Actions step output so reviewers can see what changed
@@ -156,49 +226,50 @@ jobs:
RESULTS_URL: ${{ steps.upload-results.outputs.artifact-url }}
DIFFS_URL: ${{ steps.upload-diffs.outputs.artifact-url }}
run: |
echo "## Desktop E2E — Visual Diff Report" >> "$GITHUB_STEP_SUMMARY"
echo "" >> "$GITHUB_STEP_SUMMARY"
{
echo "## Desktop E2E — Visual Diff Report"
echo ""
# Count diff images (playwright writes *-diff.png on mismatch)
DIFF_COUNT=$(find test-results -name '*-diff.png' 2>/dev/null | wc -l)
ACTUAL_COUNT=$(find test-results -name '*-actual.png' 2>/dev/null | wc -l)
if [ "$DIFF_COUNT" -eq 0 ]; then
echo "✅ All $ACTUAL_COUNT screenshot(s) matched their baselines (or no baselines existed yet)." >> "$GITHUB_STEP_SUMMARY"
echo "✅ All $ACTUAL_COUNT screenshot(s) matched their baselines (or no baselines existed yet)."
else
echo "📸 **$DIFF_COUNT of $ACTUAL_COUNT screenshot(s) differ from baseline:**" >> "$GITHUB_STEP_SUMMARY"
echo "" >> "$GITHUB_STEP_SUMMARY"
echo "| Test | Diff | Actual | Expected |" >> "$GITHUB_STEP_SUMMARY"
echo "|------|------|--------|----------|" >> "$GITHUB_STEP_SUMMARY"
echo "📸 **$DIFF_COUNT of $ACTUAL_COUNT screenshot(s) differ from baseline:**"
echo ""
echo "| Test | Diff | Actual | Expected |"
echo "|------|------|--------|----------|"
# List each diff image with a link to the artifact
for diff in $(find test-results -name '*-diff.png' 2>/dev/null | sort); do
base=$(echo "$diff" | sed 's/-diff\.png$//')
base=${diff%-diff.png}
test_name=$(basename "$base")
echo "| $test_name | [diff]($diff) | [actual](${base}-actual.png) | [expected](${base}-expected.png) |" >> "$GITHUB_STEP_SUMMARY"
echo "| $test_name | [diff]($diff) | [actual](${base}-actual.png) | [expected](${base}-expected.png) |"
done
fi
echo "" >> "$GITHUB_STEP_SUMMARY"
echo "📥 **Artifacts:**" >> "$GITHUB_STEP_SUMMARY"
echo "" >> "$GITHUB_STEP_SUMMARY"
echo ""
echo "📥 **Artifacts:**"
echo ""
if [ -n "$RESULTS_URL" ]; then
echo "- [playwright-test-results]($RESULTS_URL) — all screenshots (actual + expected + diff) + traces" >> "$GITHUB_STEP_SUMMARY"
echo "- [playwright-test-results]($RESULTS_URL) — all screenshots (actual + expected + diff) + traces"
fi
if [ -n "$REPORT_URL" ]; then
echo "- [playwright-report]($REPORT_URL) — interactive HTML report" >> "$GITHUB_STEP_SUMMARY"
echo "- [playwright-report]($REPORT_URL) — interactive HTML report"
fi
if [ -n "$DIFFS_URL" ]; then
echo "- [visual-diffs]($DIFFS_URL) — just the diffed screenshots (small, fast to review)" >> "$GITHUB_STEP_SUMMARY"
echo "- [visual-diffs]($DIFFS_URL) — just the diffed screenshots (small, fast to review)"
fi
echo "" >> "$GITHUB_STEP_SUMMARY"
echo "**To update baselines:** merge to main (baselines auto-update on main runs) or run \`npx playwright test --update-snapshots\` locally." >> "$GITHUB_STEP_SUMMARY"
echo ""
echo "**To update baselines:** merge to main (baselines auto-update on main runs) or run \`npx playwright test --update-snapshots\` locally."
# Also parse the JSON report for pass/fail counts
if [ -f playwright-report/results.json ]; then
echo "" >> "$GITHUB_STEP_SUMMARY"
echo "### Test Results" >> "$GITHUB_STEP_SUMMARY"
echo "" >> "$GITHUB_STEP_SUMMARY"
echo ""
echo "### Test Results"
echo ""
node -e "
const r = require('./playwright-report/results.json');
const stats = r.stats || {};
@@ -208,5 +279,6 @@ jobs:
console.log('| ❌ Failed | ' + (stats.unexpected || 0) + ' |');
console.log('| ⏭️ Skipped | ' + (stats.skipped || 0) + ' |');
console.log('| 🔄 Flaky | ' + (stats.flaky || 0) + ' |');
" >> "$GITHUB_STEP_SUMMARY" 2>/dev/null || true
" 2>/dev/null || true
fi
} >> "$GITHUB_STEP_SUMMARY"
+13
View File
@@ -44,6 +44,7 @@ jobs:
if ! BASE=$(git merge-base origin/main HEAD 2>/dev/null) || [ -z "$BASE" ]; then
STATUS='[{"source":"unrelated histories","results":[{"kind":"action_required","title":"Unrelated histories","summary":"This PR has no common ancestor with main.","detail":"","how_to_fix":"Rebase your changes onto current main:\n```\ngit fetch origin main\ngit checkout -b fix-branch origin/main\n# re-apply your changes (cherry-pick, copy files, etc.)\ngit push -f origin fix-branch\n```\n"}]}]'
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
echo "review_status=${STATUS}" > review-status.json
echo ""
echo "::error::This PR has no common ancestor with main."
echo ""
@@ -66,3 +67,15 @@ jobs:
fi
echo "::notice::Common ancestor with main: $BASE"
echo "review_status=[]" >> "$GITHUB_OUTPUT"
echo "review_status=[]" > review-status.json
- name: Upload review status artifact
if: always() && steps.merge-base-check.outcome != 'skipped'
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: review-status-history-check
path: review-status.json
retention-days: 1
overwrite: true
if-no-files-found: ignore
+78
View File
@@ -0,0 +1,78 @@
name: Infographic Check
# Rejects PRs that commit PR-infographic images into the repo.
#
# PR infographics are rendered to an image-provider URL (fal.media) and
# embedded in the PR *description*. The PR body is the archive; the binary
# never belongs in git history.
#
# This has now leaked twice. PR #48261 removed the first batch, PR #54564
# removed a second batch and added `infographic/` to `.gitignore` — but
# `.gitignore` only stops *accidental* `git add`. It does nothing against
# `git add -f`, and it does nothing for a path that does not literally match
# the ignore pattern. Nine more PNGs (~14MB) were committed in the four
# weeks AFTER that rule landed, plus PR #70552 caught an `infograficos/`
# spelling that sidestepped the pattern entirely.
#
# A passive ignore rule cannot enforce a policy. This check can.
on:
workflow_call:
outputs:
review_status:
description: "JSON array of review_status objects for the synthesizer."
value: ${{ jobs.check-no-committed-infographics.outputs.review_status }}
permissions:
contents: read
jobs:
check-no-committed-infographics:
runs-on: ubuntu-latest
timeout-minutes: 10
outputs:
review_status: ${{ steps.infographic-check.outputs.review_status }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- id: infographic-check
name: Reject committed PR-infographic images
run: |
# Match on the IMAGE, not on a directory name. Keying this to
# `infographic/` is what let `infograficos/` through in #70552 —
# any localized or typo'd directory would sidestep it again.
# Instead: find tracked raster images whose path contains an
# infographic-ish segment, in any spelling, at any depth.
#
# `docs/assets` and `website/` legitimately hold product imagery
# and are excluded; those are referenced from shipped docs pages.
OFFENDERS=$(git ls-files -z \
| tr '\0' '\n' \
| grep -iE '(^|/)(infograph|infograf)[^/]*/' \
| grep -iE '\.(png|jpe?g|webp|gif)$' \
|| true)
if [ -n "$OFFENDERS" ]; then
COUNT=$(printf '%s\n' "$OFFENDERS" | wc -l | tr -d ' ')
STATUS='[{"source":"committed infographics","results":[{"kind":"action_required","title":"PR infographic committed to the repo","summary":"Infographic images belong in the PR description, never in git.","detail":"","how_to_fix":"Untrack the image and reference the provider URL from the PR body instead:\n```\ngit rm --cached <path-to-image>\n```\nThen put it in the PR description:\n```\n## Infographic\n\n![slug](https://<provider-url>)\n```\n"}]}]'
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
echo ""
echo "::error::${COUNT} PR-infographic image(s) are tracked in git."
echo ""
printf '%s\n' "$OFFENDERS" | sed 's/^/ /'
echo ""
echo "PR infographics are rendered to an image-provider URL and"
echo "embedded in the PR DESCRIPTION. The PR body is the archive —"
echo "the binary never enters git history."
echo ""
echo "This rule has been re-established twice already (#48261,"
echo "#54564) and leaked both times, because .gitignore cannot stop"
echo "'git add -f' or a differently-spelled directory (#70552)."
echo ""
echo "To fix:"
echo " git rm --cached <path> # keeps your local copy"
echo " # then embed the provider URL in the PR description"
exit 1
fi
echo "::notice::No committed PR-infographic images."
echo "review_status=[]" >> "$GITHUB_OUTPUT"
+122
View File
@@ -0,0 +1,122 @@
name: Install & Update E2E (reusable)
# Runs ONE update route against ONE starting commit, in the dev sandbox, with a
# real install (uv, a managed Python, Node, the venv) behind it.
#
# Reusable so callers can fan out over the combinations that matter -- update
# from the tip vs. from an older release, `hermes update` vs. re-running the
# installer -- without duplicating the runner setup. Each leg is independent:
# its own sandbox, its own install, nothing rewound or shared.
#
# Call it:
#
# jobs:
# tip:
# uses: ./.github/workflows/install-e2e-run.yml
# with:
# route: update
# install-ref: refs/heads/main
on:
workflow_call:
inputs:
route:
description: 'Update path to exercise: update (hermes update) or installer (re-run install.sh).'
required: true
type: string
install-ref:
description: 'What to install before updating: a branch, a tag (v2026.7.7), or a SHA reachable from main.'
required: false
type: string
default: refs/heads/main
runner:
description: 'Runner label.'
required: false
type: string
default: ubuntu-latest
timeout-minutes:
description: 'Job timeout. A cold run installs real toolchains twice.'
required: false
type: number
default: 45
permissions:
contents: read
jobs:
e2e:
name: ${{ inputs.route }} from ${{ inputs.install-ref }}
runs-on: ${{ inputs.runner }}
timeout-minutes: ${{ inputs.timeout-minutes }}
steps:
# Full history: the sandbox fetches the starting commit and the test
# compares against this commit, so a shallow clone is not enough.
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
fetch-depth: 0
# bubblewrap + slirp4netns are what the sandbox is built on; util-linux
# supplies the `unshare` that builds the multi-uid userns for the
# user-level (non-root) install.
- name: Install sandbox dependencies
run: |
set -euo pipefail
sudo apt-get update -qq
sudo apt-get install -y -qq bubblewrap slirp4netns uidmap util-linux
# Ubuntu 24.04 restricts unprivileged user namespaces through AppArmor,
# which is exactly what bwrap needs. Report the state before touching it
# so a future runner-image change is visible in the log rather than
# silently altering what this job proves.
- name: Permit unprivileged user namespaces
run: |
set -euo pipefail
echo "--- kernel userns settings (before)"
sysctl kernel.unprivileged_userns_clone 2>/dev/null || echo " (sysctl absent)"
sysctl kernel.apparmor_restrict_unprivileged_userns 2>/dev/null || echo " (sysctl absent)"
if sysctl -n kernel.apparmor_restrict_unprivileged_userns >/dev/null 2>&1; then
sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
fi
echo "--- subuid/subgid for $(id -un)"
grep "^$(id -un):" /etc/subuid /etc/subgid || echo " (none — sandbox will say so)"
- name: Run install + update E2E
run: |
set -euo pipefail
tests/install/install-update-e2e.sh \
--route '${{ inputs.route }}' \
--install-ref '${{ inputs.install-ref }}'
env:
# Outside the workspace on purpose: the script creates this directory
# up front, and an untracked dir inside the repo makes the worktree
# dirty -- which dev-sandbox reacts to by snapshotting the working
# copy into a fresh fake-main commit on every invocation, moving the
# update target mid-run.
HERMES_E2E_LOG_DIR: ${{ runner.temp }}/e2e-logs
# Artifact names cannot contain '/', and install-ref may be a full ref
# like refs/heads/main. GitHub Actions expressions have no string-replace
# function, so build the safe name here. Runs even on failure -- that is
# exactly when the logs are wanted.
- name: Build artifact name
if: always()
id: artifact
run: |
set -euo pipefail
safe_ref='${{ inputs.install-ref }}'
safe_ref="${safe_ref//\//-}"
echo "name=install-e2e-${{ inputs.route }}-${safe_ref}" >> "$GITHUB_OUTPUT"
# The installer's own transcripts say far more than the assertion that
# tripped when a real install breaks.
- name: Upload installer logs
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
# Unique per leg: a matrix over releases runs this workflow several
# times per route, and same-named artifacts collide.
name: ${{ steps.artifact.outputs.name }}-${{ github.sha }}
path: ${{ runner.temp }}/e2e-logs
retention-days: 14
if-no-files-found: ignore
+110
View File
@@ -0,0 +1,110 @@
name: Install & Update E2E
# Can a user on a released version get to this commit?
#
# For each release we sample, a leg installs that release through the real
# `curl | install.sh` one-liner (uv, a managed Python, Node, the venv) inside
# scripts/dev-sandbox.sh, then applies one update route and requires the
# checkout to land on this commit with a working `hermes`.
#
# The starting versions are chosen at runtime from the repo's release tags
# (scripts/sandbox/pick-release-tags.sh): newest, oldest, and a spread between.
# A hardcoded list would stop covering the newest release the day after it
# ships, and would pin an "oldest" that nobody still runs.
#
# Triggers:
# * every 12 hours, so upstream drift (a new uv, a Node bump, a PyPI change)
# surfaces on a schedule rather than in someone's review cycle;
# * when a release tag is created -- the moment the set of versions users can
# update FROM changes, and the moment a broken updater would strand them;
# * manually, where you can pick the route and how many releases to sample.
#
# Deliberately NOT on pull_request: a leg takes ~11 minutes of real toolchain
# installation, and the matrix multiplies that. Updating is release-shaped work,
# so it is gated on releases and the clock instead.
on:
workflow_dispatch:
inputs:
route:
description: 'Which update route to exercise.'
required: false
type: choice
default: both
options: [both, update, installer]
tag-count:
description: 'How many release tags to sample (newest, oldest, and a spread between).'
required: false
type: string
default: '5'
schedule:
# Every 12 hours, off the hour to avoid the top-of-hour runner crunch.
- cron: '20 7,19 * * *'
push:
tags:
# Release tags only: the repo also carries backup/* and one-off tags.
- 'v[0-9]+.[0-9]+.[0-9]+'
- 'v[0-9]+.[0-9]+.[0-9]+.[0-9]+'
permissions:
contents: read
concurrency:
group: install-e2e-${{ github.ref }}
cancel-in-progress: true
jobs:
# Which released versions do we test updating FROM? Resolved once and shared
# by both route matrices, so the two routes cover the same set.
pick-releases:
name: Pick release tags
runs-on: ubuntu-latest
timeout-minutes: 5
outputs:
tags: ${{ steps.pick.outputs.tags }}
steps:
# This job only reads tag names and runs one script, so take the cheap
# checkout: no blobs (filter), no other files (sparse), but DO fetch tags
# -- they are the whole input, and the default shallow checkout has none.
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
filter: blob:none
fetch-tags: true
sparse-checkout: scripts/sandbox/pick-release-tags.sh
sparse-checkout-cone-mode: false
- id: pick
run: |
set -euo pipefail
tags="$(scripts/sandbox/pick-release-tags.sh --count '${{ inputs.tag-count || 5 }}')"
echo "Testing updates from: $tags"
echo "tags=$tags" >> "$GITHUB_OUTPUT"
# `hermes update` -- the route most users take.
update:
if: github.event_name != 'workflow_dispatch' || inputs.route != 'installer'
needs: pick-releases
strategy:
# One release breaking is worth knowing about even if another already
# failed, so let every leg report.
fail-fast: false
matrix:
install-ref: ${{ fromJSON(needs.pick-releases.outputs.tags) }}
uses: ./.github/workflows/install-e2e-run.yml
with:
route: update
install-ref: ${{ matrix.install-ref }}
# Re-running the curl one-liner over an existing checkout: autostash + pull
# rather than the updater's own git handling.
installer:
if: github.event_name != 'workflow_dispatch' || inputs.route != 'update'
needs: pick-releases
strategy:
fail-fast: false
max-parallel: 3
matrix:
install-ref: ${{ fromJSON(needs.pick-releases.outputs.tags) }}
uses: ./.github/workflows/install-e2e-run.yml
with:
route: installer
install-ref: ${{ matrix.install-ref }}
+38
View File
@@ -0,0 +1,38 @@
name: Installer tests
# scripts/install.ps1's PowerShell tests. They exercise the installer as a real
# subprocess, and every path contract they assert (8.3 short-name aliases,
# Git Bash layouts, provider-cmdlet behavior) is Windows-specific — so they need
# a Windows runner. Before this workflow existed the files were in the tree but
# nothing ever ran them.
on:
workflow_call:
permissions:
contents: read
concurrency:
group: installer-tests-${{ github.ref }}
cancel-in-progress: true
jobs:
powershell:
name: PowerShell installer tests
runs-on: windows-latest
timeout-minutes: 15
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Windows PowerShell 5.1 as well as pwsh 7: install.ps1 is delivered via
# `irm | iex` into whatever shell the user already has, and 5.1 is what
# ships with Windows. A construct that only parses under 7 is a broken
# installer for most of the people hitting it.
- name: 8.3 short-path normalization (pwsh 7)
shell: pwsh
run: pwsh -NoProfile -ExecutionPolicy Bypass -File scripts/tests/test-install-ps1-longpath.ps1
- name: 8.3 short-path normalization (Windows PowerShell 5.1)
shell: powershell
run: powershell -NoProfile -ExecutionPolicy Bypass -File scripts/tests/test-install-ps1-longpath.ps1
+23 -5
View File
@@ -67,9 +67,13 @@ jobs:
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
- name: grab npm 12
run: |
npm i -g npm@12
# --ignore-scripts: eslint only needs TS sources + eslint packages.
- uses: ./.github/actions/retry
with:
@@ -85,20 +89,33 @@ jobs:
- name: Produce patch
id: produce-patch
run: |
if git diff --quiet; then
# Exclude every file that the dep-version-gate ruleset guards
# (package manifests, eslint configs, workflow files). A patch
# that contains one of these files makes the bot PR wait for a
# team review, and auto-merge then stops. The check step in
# typecheck.yml still reports their lint errors.
# The (glob) magic makes "**/" also match files at the repo
# root, which plain pathspec wildcards do not.
EXCLUDES=(
':(exclude,glob)**/package.json'
':(exclude,glob)**/package-lock.json'
':(exclude,glob)**/eslint.config.*'
':(exclude,glob).github/**'
)
if git diff --quiet -- . "${EXCLUDES[@]}"; then
echo "No fixes needed."
echo "has-fixes=false" >> "$GITHUB_OUTPUT"
# Empty patch signals "nothing to do" to apply-patch.
: > js-fix.patch
else
git diff > js-fix.patch
git diff -- . "${EXCLUDES[@]}" > js-fix.patch
echo "has-fixes=true" >> "$GITHUB_OUTPUT"
echo "Patch size: $(wc -c < js-fix.patch) bytes"
# Reject patches that touch anything outside JS/TS/JSON sources.
# `npm run fix` should only ever modify those; anything else means
# eslint/prettier or a plugin went rogue and we refuse to ship it.
BAD=$(git diff --name-only | grep -vE '\.(js|cjs|mjs|ts|tsx|json)$' || true)
BAD=$(git diff --name-only -- . "${EXCLUDES[@]}" | grep -vE '\.(js|cjs|mjs|ts|tsx|json)$' || true)
if [ -n "$BAD" ]; then
echo "::error::Refusing to upload patch — touches disallowed files:"
echo "$BAD"
@@ -122,6 +139,7 @@ jobs:
if: needs.generate-patch.outputs.has-fixes == 'true'
runs-on: ubuntu-latest
timeout-minutes: 15
environment: trusted-automation
permissions:
contents: write # needed to push to bot/js-autofix
pull-requests: write # needed for PR creation + auto-merge
@@ -132,7 +150,7 @@ jobs:
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ secrets.APP_CLIENT_ID }}
client-id: ${{ vars.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- name: Download patch
+67 -35
View File
@@ -5,47 +5,79 @@ on:
workflow_call:
jobs:
workspaces:
name: List npm workspaces
runs-on: ubuntu-latest
timeout-minutes: 20
outputs:
packages: ${{ steps.set-matrix.outputs.packages }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
cache: npm
- uses: ./.github/actions/retry
with:
command: npm ci --ignore-scripts
- id: set-matrix
run: |
PACKAGES=$(npm query .workspace | jq -c '[.[].location]')
if [ "$PACKAGES" = "[]" ] || [ -z "$PACKAGES" ]; then
echo "::error::Workspace discovery produced an empty package list — refusing to emit a zero-length matrix (would skip all JS/TS checks silently)."
exit 1
fi
echo "packages=$PACKAGES" >> "$GITHUB_OUTPUT"
check:
name: Typecheck & Test
needs: workspaces
runs-on: ubuntu-latest
timeout-minutes: 20
strategy:
matrix:
package: ${{ fromJson(needs.workspaces.outputs.packages) }}
fail-fast: false # report all failures, not just the first one
name: JS & TS checks
# One 32-core job replaces a 14-leg matrix. The matrix spread about 612s
# of check payload over 4-core runners. It paid about 371s of repeated
# setup to do it: 14 checkouts, 14 node installs, 14 node_modules
# restores.
#
# One larger runner installs one time. vitest, tsc and eslint each size
# their own worker pool from the core count.
runs-on: ubuntu-latest-32-core
timeout-minutes: 30
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
node-version: 26
cache: npm
- name: grab npm 12
run: |
# No-op once the bundled npm is already 12.x — saves ~5-15s/job and
# keeps the installed major aligned with the npm12 cache-key tag.
npm --version | grep -q '^12\.' || npm i -g npm@12
# The ``cache: npm`` option of ``setup-node`` caches only the ~/.npm
# tarball cache. The job then extracts the full workspace node_modules
# again and runs the postinstalls again, which includes the Electron
# binary fetch. This caches the installed tree itself, keyed on the
# lockfile, and skips ``npm ci`` on an exact hit. There are no
# restore-keys: a partial hit leaves a stale tree, so anything other
# than an exact lockfile match reinstalls from the start.
#
# This install runs WITH scripts, so the tree holds the postinstall
# artifacts. The postinstall of electron unpacks its binary into
# node_modules/electron/dist, which is inside the cached tree.
#
# The ~/.cache/electron download cache stays out of the key on purpose.
# ``npm ci`` is skipped on a hit, so nothing reads that cache. It only
# makes the archive larger.
- name: Restore node_modules
id: node-modules-cache
uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4
with:
path: |
node_modules
apps/*/node_modules
ui-tui/node_modules
ui-tui/packages/*/node_modules
tests-js/node_modules
web/node_modules
key: node-modules-scripts-${{ runner.os }}-node26-npm12-${{ hashFiles('package-lock.json') }}
- uses: ./.github/actions/retry
if: steps.node-modules-cache.outputs.cache-hit != 'true'
with:
command: npm ci
- run: npm run --prefix ${{ matrix.package }} check
- run: npm run --prefix ${{ matrix.package }} fix
# Every check runs at the same time. The step fails only after all of
# them finish. There are two reasons this is not ``npm run --ws check``.
#
# * ``--ws`` is serial and stops at the first workspace that fails. A
# run then reports one failure, where the matrix this replaced
# reported every failure together.
# * The unit of work is a CHECK, and not a workspace. apps/desktop is
# most of the payload, and its own ``check`` is a serial && chain.
# A spread across workspaces alone leaves that chain as the long
# pole. This expands the ``check:*`` sub-scripts of a package, so
# its lint, ui, electron and plugin suites all run together. That
# is the same selection rule the old matrix job used.
#
# Discovery is ``npm query .workspace``. A new package or a new
# ``check:*`` script needs no change here. An empty list is an error and
# not an empty run, because an empty run reports green after it checks
# nothing.
- name: Run all workspace checks
run: node .github/scripts/run-workspace-checks.mjs
+18 -12
View File
@@ -2,13 +2,16 @@ name: Label rerun
# When the ``ci-reviewed`` label is added to a PR, rerun all failed jobs in
# the latest CI run. This re-evaluates ``review-labels`` (which now sees the
# label) and GitHub automatically reruns dependent jobs (``comment-live``,
# ``all-checks-pass``) — so the review comment gets updated too.
# label) and GitHub reruns the dependent ``all-checks-pass`` gate.
#
# If the CI run is still in progress when the label is added, we wait for it
# to finish before rerunning (``gh run rerun`` only works on completed runs).
# The wait can be long (20+ min for a full CI run), but it's better than
# silently failing and leaving the reviewer stuck.
# ``gh run rerun`` only works on a completed run. Thus this waits when the
# run is still in progress. The wait is now short. The two slowest jobs are
# the 40-minute comment poller and the 45-minute image build. Each one moved
# to its own workflow, so a CI run ends when its required jobs end.
#
# The review comment updates without help. The poller in
# ci-review-comment.yml watches the CI run through the API. It gets the
# rerun results.
on:
pull_request:
@@ -38,22 +41,25 @@ jobs:
set -uo pipefail
# Find the latest CI run for this PR's head SHA.
RUN_ID=$(gh run list \
RUN_INFO=$(gh run list \
--repo "$REPO" \
--commit "$HEAD_SHA" \
--workflow ci.yml \
--workflow ci.yaml \
--limit 1 \
--json databaseId,status \
--jq '.[0] | "\(.databaseId) \(.status)"' 2>/dev/null || true)
if [ -z "$RUN_ID" ]; then
if [ -z "$RUN_INFO" ]; then
echo "No CI run found for this PR — nothing to rerun."
exit 0
fi
# Split "RUN_ID STATUS" into two vars.
RUN_ID="${RUN_ID%% *}"
STATUS="${RUN_ID##* }"
# Split "RUN_ID STATUS" into two vars. Read STATUS from RUN_INFO,
# not from the truncated RUN_ID. Both values came from the same
# var before, which made STATUS the run id. Thus the wait branch
# always ran.
RUN_ID="${RUN_INFO%% *}"
STATUS="${RUN_INFO##* }"
echo "Latest CI run: $RUN_ID (status: $STATUS)"
+21 -3
View File
@@ -40,6 +40,11 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
# raw.githubusercontent.com every job; transient fetch failures
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
version: "0.9.28"
- name: Install ruff + ty
uses: ./.github/actions/retry
@@ -129,6 +134,11 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
# raw.githubusercontent.com every job; transient fetch failures
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
version: "0.9.28"
- name: Install ruff
uses: ./.github/actions/retry
@@ -153,10 +163,18 @@ jobs:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Set up Python
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v5
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
python-version: "3.11"
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
# raw.githubusercontent.com every job; transient fetch failures
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
version: "0.9.28"
- name: Set up Python 3.11
uses: ./.github/actions/retry
with:
command: uv python install 3.11
- name: Run footgun checker
run: python scripts/check-windows-footguns.py --all
+13 -1
View File
@@ -79,12 +79,24 @@ jobs:
if [ "$CHANGED" = "true" ]; then
CONTENT=$(cat /tmp/lockfile-diff.md | python3 -c "import sys,json; print(json.dumps(sys.stdin.read()))")
STATUS="[{\"source\":\"lockfile-diff\",\"results\":[{\"kind\":\"action_required\",\"title\":\"package-lock.json\",\"summary\":\"Locked npm dependency versions changed.\",\"detail\":${CONTENT},\"how_to_fix\":\"Add the \`ci-reviewed\` label after verifying the version changes are expected.\"}]}"
STATUS="[{\"source\":\"lockfile-diff\",\"results\":[{\"kind\":\"action_required\",\"title\":\"package-lock.json\",\"summary\":\"Locked npm dependency versions changed.\",\"detail\":${CONTENT},\"how_to_fix\":\"Add the \`ci-reviewed\` label after verifying the version changes are expected.\"}]}]"
else
STATUS="[]"
fi
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
echo "review_status=${STATUS}" > review-status.json
- name: Upload review status artifact
if: always() && steps.emit-status.outcome != 'skipped'
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: review-status-lockfile-diff
path: review-status.json
retention-days: 1
overwrite: true
if-no-files-found: ignore
- name: Upload diff artifact
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
+119
View File
@@ -0,0 +1,119 @@
name: Nix flake check
# Builds every output of the flake: the package, the devShell, and the 21
# checks under nix/checks.nix — module evaluation, option parity, .env
# assembly, service argv, and the rest.
#
# This workflow owns its triggers and ci.yml does not call it, for the reason
# docker.yml gives: a reusable-workflow call holds the caller run in progress
# for the full build, and GitHub refuses `gh run rerun` on a run that is still
# in progress. One slow advisory job in the CI lane blocks every rerun of the
# fast required jobs beside it. A separate run reruns and cancels on its own.
on:
pull_request:
push:
branches: [main]
permissions:
contents: read
# PR runs collapse to the newest commit. A push to main is never cancelled:
# each one saves the store cache that later PRs restore from, so cancelling a
# merge would leave the next PR to build from nothing.
concurrency:
group: nix-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
jobs:
# A `paths:` filter cannot gate this workflow correctly. The flake packages
# the product, and nine of the checks then run the built binary, so a change
# to hermes_cli/ alone can fail `nix flake check` without touching one file
# under nix/. The `nix` lane therefore follows python_prod as well as the
# flake inputs. On push the classifier fails open and every lane is true.
detect:
name: Detect affected areas
runs-on: ubuntu-latest
timeout-minutes: 10
outputs:
nix: ${{ steps.classify.outputs.nix }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Detect affected areas
id: classify
uses: ./.github/actions/detect-changes
with:
github-token: ${{ github.token }}
flake-check:
name: nix flake check
needs: [detect]
if: needs.detect.outputs.nix == 'true'
# The build compiles the package and its whole dependency closure, so this
# takes minutes and not seconds when the cache misses. `nix flake check`
# builds 21 checks, and --max-jobs defaults to the core count. It uses the
# wider runner with no more configuration.
runs-on: ubuntu-latest-32-core
timeout-minutes: 60
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Install Nix
uses: cachix/install-nix-action@630ae543ea3a38a9a4166f03376c02c50f408342 # v31.11.0
with:
extra_nix_config: |
experimental-features = nix-command flakes
# A store path that does not substitute is a cache miss and not a
# build failure. Build it here instead.
fallback = true
# Each source archive is fetched one time in a run, and not one
# time for each evaluation.
tarball-ttl = 3600
# Restores /nix/store from the GitHub Actions cache. The store holds the
# whole dependency closure, so a hit turns a build of several minutes
# into a short evaluation.
#
# The Magic Nix Cache is not an option here. Its free tier ended in
# February 2025 with the GitHub cache API that it was built on. This
# action uses the current API and needs no account and no secret.
- name: Restore and save the Nix store
uses: nix-community/cache-nix-action@7df957e333c1e5da7721f60227dbba6d06080569 # v7
with:
# The closure changes when the flake inputs change or when the
# dependencies of the project change. The key hashes both, so an
# edit to the source alone keeps the hit.
primary-key: nix-${{ runner.os }}-${{ hashFiles('flake.lock', 'nix/**', 'pyproject.toml', 'uv.lock') }}
# On a miss, restore the newest store for this runner. Most of the
# closure — Python, node, each transitive library — survives a bump
# of the lockfile, so an old store still removes most of the work.
restore-prefixes-first-match: nix-${{ runner.os }}-
# Save from main only. A cache that a PR writes is visible to that
# PR alone and never to another branch, so a save there spends the
# 10 GB quota of the repository and helps no later run. A PR still
# restores: it reads the cache that the merge to main wrote. This is
# the same rule that docker.yml applies to `cache-to`.
save: ${{ github.event_name != 'pull_request' }}
# Collect garbage before the save, so the store stays inside the
# 10 GB quota of the repository. Without a limit the store grows at
# each merge until GitHub removes the entry, and the next PR then
# gets nothing. This number is the size of the store and not the
# size of the compressed archive.
gc-max-store-size-linux: 5G
# Delete the caches that this key replaces. GitHub removes caches by
# least recent use across the whole repository, so a Nix store that
# is never purged pushes out the caches of the other workflows.
purge: true
purge-prefixes: nix-${{ runner.os }}-
purge-created: 0
purge-primary-key: never
- name: nix flake check
# --print-build-logs: a check that fails then prints the assertion
# that failed, and not only the derivation that failed to build.
run: nix flake check --print-build-logs
+19 -2
View File
@@ -43,11 +43,16 @@ jobs:
uses: google/osv-scanner-action/.github/workflows/osv-scanner-reusable.yml@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8
with:
# Scan explicit lockfiles rather than recursing, so we only look at
# the three sources of truth and skip vendored / test / worktree dirs.
# the five sources of truth and skip vendored / test / worktree dirs.
scan-args: |-
--lockfile=uv.lock
--lockfile=package-lock.json
--lockfile=website/package-lock.json
--lockfile=plugins/platforms/photon/sidecar/package-lock.json
--lockfile=scripts/whatsapp-bridge/package-lock.json
# The upstream reusable workflow uploads this exact file under its
# fixed artifact name, which the wrapper downloads below.
results-file-name: osv-results.sarif
fail-on-vuln: false
emit-status:
@@ -64,7 +69,7 @@ jobs:
- name: Download SARIF result
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
with:
name: osv-results
name: OSV Scanner SARIF file
path: /tmp/osv-results
continue-on-error: true
@@ -122,3 +127,15 @@ jobs:
fi
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
echo "review_status=${STATUS}" > review-status.json
- name: Upload review status artifact
if: always() && steps.emit.outcome != 'skipped'
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: review-status-osv-scanner
path: review-status.json
retention-days: 1
overwrite: true
if-no-files-found: ignore
@@ -0,0 +1,80 @@
name: Publish E2E evidence
# This runs only from the default branch after CI completes. It intentionally
# checks out main, never the PR ref, and treats the downloaded artifact as
# untrusted input before uploading validated GitHub attachments.
on:
workflow_run:
workflows: [CI]
types: [completed]
permissions:
actions: read
contents: read
pull-requests: write
concurrency:
group: publish-e2e-evidence-${{ github.event.workflow_run.id }}
cancel-in-progress: false
jobs:
publish:
name: Publish inline E2E evidence
if: github.event.workflow_run.event == 'pull_request'
runs-on: ubuntu-latest
timeout-minutes: 10
environment: gh-image
steps:
- name: Check out trusted publisher
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
ref: ${{ github.event.repository.default_branch }}
persist-credentials: false
# v1.2.0 resolves to 44f4b93ecbbe22de6c45fa2f62f519aee564ca8c.
- name: Install gh-image
env:
GH_TOKEN: ${{ github.token }}
run: gh extension install drogers0/gh-image --pin v1.2.0
- name: Download and attach evidence
env:
GH_TOKEN: ${{ github.token }}
GITHUB_TOKEN: ${{ github.token }}
GH_SESSION_TOKEN: ${{ secrets.GH_IMAGE_SESSION_TOKEN }}
SOURCE_REPO: ${{ github.repository }}
SOURCE_RUN_ID: ${{ github.event.workflow_run.id }}
HEAD_OWNER: ${{ github.event.workflow_run.head_repository.owner.login }}
HEAD_BRANCH: ${{ github.event.workflow_run.head_branch }}
HEAD_SHA: ${{ github.event.workflow_run.head_sha }}
run: |
set -euo pipefail
# The run's own ``pull_requests`` payload is always empty for a
# fork PR, so resolve the PR from its head reference instead.
# The head-SHA match skips runs that a newer push superseded.
PR_NUMBER=$(gh api -X GET "repos/$SOURCE_REPO/pulls" \
-f head="$HEAD_OWNER:$HEAD_BRANCH" -f state=open \
--jq '.[] | select(.head.sha == $ENV.HEAD_SHA) | .number' \
| head -n1)
if [ -z "$PR_NUMBER" ]; then
echo "No open pull request has head $HEAD_OWNER:$HEAD_BRANCH at $HEAD_SHA (CI run $SOURCE_RUN_ID)."
exit 0
fi
ARTIFACT_NAME=$(gh api "repos/$SOURCE_REPO/actions/runs/$SOURCE_RUN_ID/artifacts" \
--jq '.artifacts[] | select(.expired == false and (.name | startswith("e2e-evidence-"))) | .name' \
| python3 -c 'import sys; print(next(iter(sys.stdin), "").strip())')
if [ -z "$ARTIFACT_NAME" ]; then
echo "No E2E evidence artifact was produced for CI run $SOURCE_RUN_ID."
exit 0
fi
EVIDENCE_DIR="$RUNNER_TEMP/e2e-evidence"
mkdir -p "$EVIDENCE_DIR"
gh run download "$SOURCE_RUN_ID" --repo "$SOURCE_REPO" --name "$ARTIFACT_NAME" --dir "$EVIDENCE_DIR"
python3 scripts/ci/publish_e2e_evidence.py \
--evidence-dir "$EVIDENCE_DIR" \
--source-repo "$SOURCE_REPO" \
--pr-number "$PR_NUMBER"
+26 -1
View File
@@ -23,6 +23,10 @@ on:
description: Whether CI-sensitive files (eslint config, workflows, actions) changed.
type: boolean
default: false
ci_review_files:
description: JSON list of CI-sensitive files changed by the pull request.
type: string
default: '[]'
mcp_catalog:
description: Whether the MCP catalog / installer changed.
type: boolean
@@ -78,18 +82,39 @@ jobs:
id: build-status
env:
CI_REVIEW: ${{ inputs.ci_review }}
CI_REVIEW_FILES: ${{ inputs.ci_review_files }}
MCP_CATALOG: ${{ inputs.mcp_catalog }}
SUPPLY_CHAIN: ${{ inputs.supply_chain }}
LABEL_PRESENT: ${{ steps.label-check.outputs.ci_reviewed }}
REPO_URL: ${{ github.server_url }}/${{ github.repository }}
BASE_SHA: ${{ github.event.pull_request.base.sha }}
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
run: |
set -euo pipefail
args=()
if [ "$CI_REVIEW" = "true" ]; then args+=(--ci-review); fi
args+=(--ci-review-files "$CI_REVIEW_FILES")
if [ "$MCP_CATALOG" = "true" ]; then args+=(--mcp-catalog); fi
if [ "$SUPPLY_CHAIN" = "true" ]; then args+=(--supply-chain); fi
if [ "$LABEL_PRESENT" = "true" ]; then args+=(--label-present); fi
python3 scripts/ci/emit_review_status.py "${args[@]}" --output "$GITHUB_OUTPUT"
# Write to both $GITHUB_OUTPUT and review-status.json for the
# live comment poller to pick up as an artifact.
python3 scripts/ci/emit_review_status.py "${args[@]}" \
--repo-url "$REPO_URL" --base-sha "$BASE_SHA" --head-sha "$HEAD_SHA" \
--output "$GITHUB_OUTPUT"
grep '^review_status=' "$GITHUB_OUTPUT" > review-status.json
- name: Upload review status artifact
if: always() && steps.build-status.outcome != 'skipped'
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: review-status-review-labels
path: review-status.json
retention-days: 1
overwrite: true
if-no-files-found: ignore
- name: Fail on missing label
if: steps.label-check.outputs.ci_reviewed != 'true'
+76
View File
@@ -0,0 +1,76 @@
# .github/workflows/rust-tests.yml
name: Rust tests
# `cargo test` for the Tauri bootstrap installer (Hermes-Setup). Nothing in CI
# compiled this crate before: `.rs` lives under `apps/`, so the change
# classifier matched it as `frontend` and ran the TypeScript matrix, which
# cannot notice a Rust error. The crate's unit tests existed in the tree and had
# never run.
#
# Linux runner on purpose. The pipe-drain tests in src/powershell.rs need a real
# process tree whose grandchild inherits the parent's stdout, and their fixture
# is `#[cfg(unix)]`; the Windows half of that same contract is covered by
# `-SelfTestPipeDrain` in scripts/desktop-update/windows.ps1 on the Windows
# lane. A Windows runner here would compile them out and report green over zero
# coverage.
on:
workflow_call:
permissions:
contents: read
concurrency:
group: rust-tests-${{ github.ref }}
cancel-in-progress: true
jobs:
bootstrap-installer:
name: cargo test (bootstrap installer)
# cargo builds codegen units and test binaries in parallel across the
# cores. This lane also builds the crate from the start when Cargo.toml
# changes.
runs-on: ubuntu-latest-32-core
timeout-minutes: 30
defaults:
run:
working-directory: apps/bootstrap-installer/src-tauri
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Tauri links against the system webkit2gtk on Linux, so the crate does
# not compile without these even for `cargo test --lib`.
- name: Install Tauri system dependencies
working-directory: .
run: |
sudo apt-get update
sudo apt-get install --no-install-recommends -y \
libwebkit2gtk-4.1-dev \
libappindicator3-dev \
librsvg2-dev \
libxdo-dev \
libssl-dev \
patchelf
# Keyed on Cargo.toml, not Cargo.lock: apps/bootstrap-installer/.gitignore
# excludes the lockfile (a create-tauri-app scaffold default), so there is
# nothing pinned to hash and `--locked` cannot be used. That also means
# this crate re-resolves its whole dependency graph on every build, which
# is a real gap for a signed installer given the pinning policy in
# AGENTS.md — tracking separately rather than widening this PR.
#
# No restore-keys: a partial hit leaves a stale target dir, and cargo
# re-resolves correctly on top of a Cargo.toml-keyed hit anyway.
- name: Restore cargo cache
uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4
with:
path: |
~/.cargo/registry
~/.cargo/git
apps/bootstrap-installer/src-tauri/target
key: cargo-${{ runner.os }}-${{ hashFiles('apps/bootstrap-installer/src-tauri/Cargo.toml') }}
# --lib only: the integration/bin targets would need a built frontend
# (vite dist) that this lane deliberately does not produce.
- name: cargo test
run: cargo test --lib
+10 -1
View File
@@ -21,7 +21,16 @@ jobs:
if: github.repository == 'NousResearch/hermes-agent'
runs-on: ubuntu-latest
timeout-minutes: 10
environment: trusted-automation
steps:
# `Get GitHub App token` below is a LOCAL composite action
# (./.github/actions/get-app-token), so the repository must be on disk
# before it can be resolved. Without this checkout the job dies with
# "Can't find 'action.yml' ... under .github/actions/get-app-token"
# every time the probe is non-ok — i.e. exactly when the watchdog is
# supposed to file its issue.
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Probe live index
id: probe
run: |
@@ -113,7 +122,7 @@ jobs:
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ secrets.APP_CLIENT_ID }}
client-id: ${{ vars.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- name: Open issue on degraded / failed probe
+11 -2
View File
@@ -21,6 +21,7 @@ jobs:
if: github.repository == 'NousResearch/hermes-agent'
runs-on: ubuntu-latest
timeout-minutes: 15
environment: trusted-automation
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
@@ -28,7 +29,7 @@ jobs:
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ secrets.APP_CLIENT_ID }}
client-id: ${{ vars.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
@@ -60,13 +61,21 @@ jobs:
if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest
timeout-minutes: 15
environment: trusted-automation
steps:
# Required: `Get GitHub App token` is a LOCAL composite action
# (./.github/actions/get-app-token) and cannot resolve without the repo
# checked out. `build-index` above already does this; this job did not,
# so the scheduled deploy re-trigger never fired.
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Get GitHub App token
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ secrets.APP_CLIENT_ID }}
client-id: ${{ vars.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- name: Trigger Deploy Site workflow
env:
GH_TOKEN: ${{ steps.app-token.outputs.token }}
+23 -11
View File
@@ -65,17 +65,11 @@ jobs:
with:
fetch-depth: 0
- name: Get GitHub App token
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ secrets.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- name: Scan diff for critical patterns
id: scan
env:
GH_TOKEN: ${{ steps.app-token.outputs.token }}
GH_TOKEN: ${{ github.token }}
CI_REVIEWED: ${{ contains(github.event.pull_request.labels.*.name, 'ci-reviewed') }}
run: |
set -euo pipefail
@@ -93,7 +87,7 @@ jobs:
# --- .pth files (auto-execute on Python startup) ---
# The exact mechanism used in the litellm supply chain attack:
# https://github.com/BerriAI/litellm/issues/24512
PTH_FILES=$(git diff --name-only "$BASE"..."$HEAD" | grep '\.pth$' || true)
PTH_FILES=$(git diff --diff-filter=d --name-only "$BASE"..."$HEAD" | grep '\.pth$' || true)
if [ -n "$PTH_FILES" ]; then
FINDINGS="${FINDINGS}
### 🚨 CRITICAL: .pth file added or modified
@@ -141,8 +135,11 @@ jobs:
# auto-loaded by the interpreter via site.py. Any nested file with the
# same name (e.g. hermes_cli/setup.py — the CLI setup wizard) is unrelated
# and produced false positives that trained reviewers to ignore the scanner.
SETUP_HITS=$(git diff --name-only "$BASE"..."$HEAD" | grep -E '^(setup\.py|setup\.cfg|sitecustomize\.py|usercustomize\.py|__init__\.pth)$' || true)
if [ -n "$SETUP_HITS" ]; then
SETUP_HITS=$(git diff --diff-filter=d --name-only "$BASE"..."$HEAD" | grep -E '^(setup\.py|setup\.cfg|sitecustomize\.py|usercustomize\.py|__init__\.pth)$' || true)
# A maintainer-applied ci-reviewed label records the manual review
# required for intentional changes to an install hook. The scanner
# still blocks every unreviewed addition or modification.
if [ -n "$SETUP_HITS" ] && [ "$CI_REVIEWED" != "true" ]; then
FINDINGS="${FINDINGS}
### 🚨 CRITICAL: Install-hook file added or modified
These files can execute code during package installation or interpreter startup.
@@ -293,4 +290,19 @@ jobs:
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as f:
f.write(f"review_status={json.dumps(merged)}\n")
f.write("critical_findings=" + os.environ.get("CRITICAL_FINDINGS", "false") + "\n")
# Write review-status.json for the live comment poller artifact.
with open("review-status.json", "w", encoding="utf-8") as f:
f.write(f"review_status={json.dumps(merged)}\n")
PYEOF
- name: Upload review status artifact
if: always() && steps.merge.outcome != 'skipped'
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: review-status-supply-chain
path: review-status.json
retention-days: 1
overwrite: true
if-no-files-found: ignore
+154
View File
@@ -0,0 +1,154 @@
name: OS-specific tests
# Runs the tests that can only be trusted on their own host OS.
#
# The main Python suite (.github/workflows/tests.yml) runs on
# ubuntu-latest and covers everything that is either platform-agnostic or
# genuinely Linux-specific. Tests whose subject is macOS- or
# Windows-specific behaviour carry a marker (see the ``_OS_MARKS`` block
# comment in tests/conftest.py) and are SKIPPED on Linux, because faking
# ``sys.platform`` on a Linux runner selects the branch under test without
# reproducing any of the OS behaviour that branch exists for. This workflow
# is where those markers actually execute:
#
# macos → ``-m macos_only`` on macos-latest
# windows → ``-m windows_only`` on windows-latest
#
# Deliberately NOT sliced. The marked set is small (tens of tests, not
# thousands), so one plain ``pytest`` process per OS is both faster and far
# less machinery than the per-file parallel runner the Linux lane uses.
# If either lane grows past its timeout, that is the signal to reach for
# scripts/run_tests.sh here too.
#
# Each lane FAILS when it selects zero tests (pytest exit code 5). Without
# that guard, a renamed marker or a bad selector would report a green job
# that ran nothing — the exact silent-coverage-loss failure this workflow
# exists to prevent.
on:
workflow_call:
permissions:
contents: read
concurrency:
group: tests-os-${{ github.ref }}
cancel-in-progress: true
jobs:
os-tests:
name: ${{ matrix.name }}
runs-on: ${{ matrix.runner }}
timeout-minutes: 30
strategy:
fail-fast: false
matrix:
include:
- name: macOS-only tests
runner: macos-latest
marker: macos_only
- name: Windows-only tests
runner: windows-latest-32-core
marker: windows_only
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pinned for the same reason as the Linux lane: unpinned, setup-uv
# resolves "latest" by fetching a manifest on every job and a
# transient fetch failure fails the whole job.
version: "0.9.28"
enable-cache: true
cache-dependency-glob: |
pyproject.toml
uv.lock
- name: Set up Python 3.11
uses: ./.github/actions/retry
with:
command: uv python install 3.11
- name: Install dependencies
# Same extras as the Linux test lane so an OS-marked test can import
# anything its Linux siblings can. ``[all]`` is deliberately
# Windows/macOS-installable (see the policy comment on the extra in
# pyproject.toml — matrix/python-olm was removed from it precisely
# because it could not build here).
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
run: uv cache prune --ci
- name: Run ${{ matrix.marker }} tests
# Two-step selection:
#
# 1. scripts/ci/list_os_marked_tests.py narrows WHICH FILES are
# imported. ``-m`` filters after collection, and collection
# imports every module under tests/ — on this host that would
# drag ~900 unrelated test modules through import, where a
# single unrelated ImportError would fail a job whose own
# subject is fine. The helper exits non-zero if the marker
# matches no file at all.
# 2. ``-m`` decides WHICH TESTS run, and stays authoritative.
# Passing it on the command line REPLACES pyproject's
# ``-m 'not integration'`` addopts (same option, last wins) —
# hence repeating ``not integration``, or the integration
# suite would return through the side door.
#
# ``--timeout-method`` needs no override: tests/conftest.py's
# pytest_configure already downgrades the signal-based timer on
# Windows, which has no SIGALRM.
shell: bash
run: |
set -uo pipefail
LIST="${RUNNER_TEMP:-.}/selected-tests.txt"
# Process substitution would hide the helper's exit status, so write
# to a file and check it explicitly.
if ! uv run --no-sync python scripts/ci/list_os_marked_tests.py \
"${{ matrix.marker }}" > "$LIST"; then
echo "::error::could not enumerate ${{ matrix.marker }} test files"
exit 1
fi
if [ ! -s "$LIST" ]; then
echo "::error::empty ${{ matrix.marker }} file list"
exit 1
fi
# Deliberately NOT `mapfile`: that is a bash 4 builtin and the macOS
# runner's /bin/bash is 3.2. Word-splitting is safe here because the
# helper emits repo-relative test paths, which contain no spaces.
# shellcheck disable=SC2046
set -- $(cat "$LIST")
echo "selected $# file(s) for ${{ matrix.marker }}:"
cat "$LIST"
# ``shell: bash`` runs this script with ``-e`` injected, which
# ``set -uo pipefail`` above does not clear. A bare pytest call
# would therefore abort the script on any non-zero exit and the
# exit-5 branch below would be unreachable dead code — the job
# would still fail red, but the diagnostic would never print.
status=0
uv run --no-sync python -m pytest \
"$@" \
-m "${{ matrix.marker }} and not integration" \
-v --tb=short || status=$?
if [ "$status" -eq 5 ]; then
echo "::error::No tests matched -m ${{ matrix.marker }}. Either the" \
"marker was renamed/dropped or selection is broken — this job" \
"must never pass without running its OS's tests."
exit 1
fi
exit "$status"
env:
# Belt-and-suspenders with tests/conftest.py's env blanking: no
# test may reach a real provider API.
OPENROUTER_API_KEY: ""
OPENAI_API_KEY: ""
NOUS_API_KEY: ""
+63 -97
View File
@@ -2,11 +2,6 @@ name: Tests
on:
workflow_call:
inputs:
slice_count:
description: Number of parallel test slices
type: number
default: 8
permissions:
contents: read
@@ -17,42 +12,17 @@ concurrency:
cancel-in-progress: true
jobs:
generate:
name: "Generate slices"
runs-on: ubuntu-latest
timeout-minutes: 10
outputs:
matrix: ${{ steps.matrix.outputs.matrix }}
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Restore duration cache
uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
with:
path: test_durations.json
key: test-durations
# Saves use test-durations-${run_id}, so the exact key above never
# matches — without this prefix fallback the cache ALWAYS missed,
# LPT slicing ran on no data, and unbalanced slices pushed heavy
# files toward the per-file timeout under load.
restore-keys: |
test-durations-
- name: Generate test slices
id: matrix
run: |
MATRIX=$(python3 scripts/run_tests_parallel.py --generate-slices ${{ inputs.slice_count }})
echo "matrix=$MATRIX" >> "$GITHUB_OUTPUT"
test:
name: Run tests slice ${{ matrix.slice.index }}/${{ inputs.slice_count }}
needs: generate
runs-on: ubuntu-latest
name: Run tests
# One 96-core runner for the whole suite. There is no slicing. Slicing
# existed to spread the suite over 4-core runners. It cost a matrix job, a
# duration cache, a per-slice artifact and a merge job to do it.
#
# 96 cores clear the floor that the slowest single test file sets (about
# 82s). A second slice divides work that is already at that floor, and
# adds a second setup.
runs-on: ubuntu-latest-96-core
timeout-minutes: 30
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.generate.outputs.matrix) }}
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
@@ -74,6 +44,11 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pin the uv version: unpinned, setup-uv resolves "latest" by
# fetching a manifest from raw.githubusercontent.com on EVERY job —
# a transient fetch failure fails the whole job (2026-07-28 slice-5
# incident). Pinned, the binary downloads directly; no manifest hop.
version: "0.9.28"
# Persist uv's download/wheel cache (~/.cache/uv) across runs.
# Keyed on the dependency manifests, so the cache is reused until
# pyproject.toml or uv.lock changes. `uv sync` still runs every
@@ -85,85 +60,71 @@ jobs:
uv.lock
- name: Set up Python 3.11
run: uv python install 3.11
uses: ./.github/actions/retry
with:
command: uv python install 3.11
- name: Install dependencies
# `uv sync --locked` installs the exact pinned set from uv.lock (and
# fails if the lock is out of sync with pyproject.toml), giving a
# reproducible env. It also creates .venv itself, so no separate
# `uv venv` step is needed.
#
# The trailing extras beyond all/dev are the lazy-install features
# (tools/lazy_deps.py) that tests exercise for real: provider.anthropic,
# stt/tts.mistral, image.fal, terminal.modal, terminal.daytona,
# memory.hindsight, search.parallel. The hermetic test env forbids
# mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
# tests/conftest.py), so the SDKs those tests need must be in the
# venv up front — resolved from uv.lock like everything else, which
# also honors the exact supply-chain pins these extras carry.
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
# Optimized for CI: prunes pre-built wheels that are cheap to
# re-download, keeping the persisted cache small and fast to restore.
run: uv cache prune --ci
- name: Run tests (slice ${{ matrix.slice.index }}/${{ inputs.slice_count }})
- name: Run tests
# Per-file isolation via scripts/run_tests.sh: each test file runs
# in its own freshly-spawned `python -m pytest <file>` subprocess
# with bounded parallelism. No xdist, no shared workers, no
# module-level state leakage between files.
#
# File list is pre-computed by the generate job (--generate-slices)
# which runs LPT distribution once and passes the file list to each
# matrix job via --files. Previously each job re-discovered files and
# re-ran LPT independently — redundant N times.
# No --files: the runner discovers the suite itself. The discovered
# set is identical to the list the removed matrix job used to pass in.
run: |
source .venv/bin/activate
scripts/run_tests.sh --files '${{ matrix.slice.files }}'
scripts/run_tests.sh
env:
# This is the maximum number of test FILES that run together.
# run_tests_parallel.py starts one pytest subprocess for each file
# from a single ThreadPoolExecutor, so this value IS the limit. The
# default is cpu_count*2, which is 192 here.
#
# Measured on this runner (96-core EPYC 7763, 377GB). Whole suite,
# two repetitions for each value. See run 32549672063:
#
# workers x cores mean
# 48 0.5x 138s
# 96 1.0x 126s <- fastest
# 144 1.5x 132s
# 192 2.0x 132s
# 240 2.5x 140s
# 288 3.0x 142s
#
# One worker for each core wins. The curve is shallow: 126s to 142s
# across a 6x range. The suite has sufficient concurrency at this
# size. The remaining time is the slowest files plus the setup.
# Workers above the core count only add contention.
HERMES_TEST_WORKERS: 96
# Ensure tests don't accidentally call real APIs
OPENROUTER_API_KEY: ""
OPENAI_API_KEY: ""
NOUS_API_KEY: ""
- name: Upload per-slice durations
# Advisory artifact (feeds slice balancing) — a transient artifact-
# service blip must not fail an otherwise-green test slice.
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: test-durations-slice-${{ matrix.slice.index }}
path: test_durations.json
retention-days: 1
# Merge per-slice duration data into a single cache, so future runs
# (including PRs) get balanced slicing.
save-durations:
needs: test
if: needs.test.result == 'success' && github.ref == 'refs/heads/main'
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- name: Download all slice durations
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
pattern: test-durations-slice-*
path: durations
merge-multiple: true
- name: Merge into single durations file
run: |
python3 -c "
import json, glob, os
merged = {}
for f in glob.glob('durations/*test_durations.json'):
with open(f) as fh:
merged.update(json.load(fh))
with open('test_durations.json', 'w') as fh:
json.dump(merged, fh, indent=2, sort_keys=True)
print(f'Merged {len(merged)} file durations')
"
- name: Save merged duration cache
uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
with:
path: test_durations.json
key: test-durations-${{ github.run_id }}
e2e:
runs-on: ubuntu-latest
timeout-minutes: 15
@@ -188,6 +149,11 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pin the uv version: unpinned, setup-uv resolves "latest" by
# fetching a manifest from raw.githubusercontent.com on EVERY job —
# a transient fetch failure fails the whole job (2026-07-28 slice-5
# incident). Pinned, the binary downloads directly; no manifest hop.
version: "0.9.28"
# Persist uv's download/wheel cache (~/.cache/uv) across runs.
# Keyed on the dependency manifests, so the cache is reused until
# pyproject.toml or uv.lock changes. `uv sync` still runs every
@@ -206,20 +172,20 @@ jobs:
# fails if the lock is out of sync with pyproject.toml), giving a
# reproducible env. It also creates .venv itself, so no separate
# `uv venv` step is needed.
#
# Same extras as the test job's sync above: the hermetic test env
# forbids mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
# tests/conftest.py), so lazy-install SDKs exercised by tests must be
# in the venv up front.
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
# Optimized for CI: prunes pre-built wheels that are cheap to
# re-download, keeping the persisted cache small and fast to restore.
run: uv cache prune --ci
- name: Packaged-wheel i18n smoke test
run: |
source .venv/bin/activate
python -m pytest -m integration tests/test_wheel_locales_e2e.py -v
- name: Run e2e tests
run: |
source .venv/bin/activate
-188
View File
@@ -1,188 +0,0 @@
name: Publish to PyPI
# Triggered by CalVer tag pushes from scripts/release.py (e.g. v2026.5.15)
# Can also be triggered manually from the Actions tab as an escape hatch.
on:
push:
tags:
- "v20*" # CalVer tags: v2026.5.15, v2026.5.15.2, etc.
workflow_dispatch:
inputs:
confirm_tag:
description: "Tag to publish (e.g. v2026.5.15). Must already exist."
required: true
type: string
# Restrict default token to read-only; each job escalates as needed.
permissions:
contents: read
# Prevent overlapping publishes (e.g. two same-day tags pushed quickly).
concurrency:
group: pypi-publish
cancel-in-progress: false
jobs:
build:
name: Build distribution 📦
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
# On workflow_dispatch, check out the confirmed tag.
ref: ${{ inputs.confirm_tag || github.ref }}
fetch-tags: true
- name: Validate tag exists
if: github.event_name == 'workflow_dispatch'
run: |
if ! git tag -l "${{ inputs.confirm_tag }}" | grep -q .; then
echo "::error::Tag '${{ inputs.confirm_tag }}' does not exist in the repo"
exit 1
fi
- name: Set up Python
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
with:
python-version: "3.13"
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
- name: Set up Node.js
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: "22"
- name: Build web dashboard
uses: ./.github/actions/retry
with:
command: npm ci
working-directory: web
- name: Compile web dashboard
run: npm run build
working-directory: web
- name: Build TUI bundle
uses: ./.github/actions/retry
with:
command: npm ci
working-directory: ui-tui
- name: Compile TUI bundle
run: npm run build
working-directory: ui-tui
- name: Bundle TUI into hermes_cli
run: |
mkdir -p hermes_cli/tui_dist
cp ui-tui/dist/entry.js hermes_cli/tui_dist/entry.js
- name: Verify frontend assets exist
run: |
test -f hermes_cli/web_dist/index.html || { echo "ERROR: web_dist not built"; exit 1; }
test -f hermes_cli/tui_dist/entry.js || { echo "ERROR: tui_dist not built"; exit 1; }
- name: Bundle install scripts into wheel
run: |
mkdir -p hermes_cli/scripts
cp scripts/install.sh hermes_cli/scripts/install.sh
cp scripts/install.ps1 hermes_cli/scripts/install.ps1
- name: Build wheel and sdist
run: uv build --sdist --wheel
- name: Upload distribution artifacts
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: python-package-distributions
path: dist/
publish:
name: Publish to PyPI
needs: build
runs-on: ubuntu-latest
timeout-minutes: 30
environment:
name: pypi
url: https://pypi.org/p/hermes-agent
permissions:
id-token: write # OIDC trusted publishing
steps:
- name: Download distribution artifacts
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
with:
name: python-package-distributions
path: dist/
- name: Publish to PyPI
uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # v1.14.0
with:
skip-existing: true
sign:
name: Sign and attach to GitHub Release
# Only runs on tag pushes — release.py creates the GitHub Release,
# and workflow_dispatch won't have a matching release to attach to.
if: startsWith(github.ref, 'refs/tags/')
needs: publish
runs-on: ubuntu-latest
timeout-minutes: 30
permissions:
contents: write # attach assets to the existing release
id-token: write # sigstore signing
steps:
- name: Download distribution artifacts
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
with:
name: python-package-distributions
path: dist/
- name: Get GitHub App token
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ secrets.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- name: Wait for GitHub Release to exist
env:
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
# release.py creates the GitHub Release after pushing the tag,
# but this workflow starts from the tag push — wait for it.
run: |
for i in $(seq 1 30); do
if gh release view "$GITHUB_REF_NAME" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then
echo "Release $GITHUB_REF_NAME found"
exit 0
fi
echo "Waiting for release... ($i/30)"
sleep 10
done
echo "::warning::Release $GITHUB_REF_NAME not found after 5 minutes — skipping signature upload"
echo "skip_sign=true" >> "$GITHUB_ENV"
- name: Sign with Sigstore
if: env.skip_sign != 'true'
uses: sigstore/gh-action-sigstore-python@04cffa1d795717b140764e8b640de88853c92acc # v3.3.0
with:
inputs: >-
./dist/*.tar.gz
./dist/*.whl
- name: Attach signed artifacts to GitHub Release
if: env.skip_sign != 'true'
env:
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
# release.py already created the GitHub Release — just upload
# the Sigstore signatures alongside the existing assets.
run: >-
gh release upload
"$GITHUB_REF_NAME" dist/*.sigstore.json
--repo "$GITHUB_REPOSITORY"
--clobber
+51 -3
View File
@@ -70,6 +70,11 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
# raw.githubusercontent.com every job; transient fetch failures
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
version: "0.9.28"
# `uv lock --check` re-resolves the project from pyproject.toml and
# compares the result to uv.lock, exiting non-zero if they disagree.
@@ -84,16 +89,46 @@ jobs:
# uv lock --check re-resolves against PyPI (network). Retry so a
# registry blip doesn't read as "lockfile stale". A genuinely stale
# lockfile fails all attempts (deterministic), costing only seconds.
#
# Backoff rather than a flat 10s: three attempts inside ~20s all
# land in the same blip. 5/15/45s spans ~65s instead.
#
# Network failure and a real desync are also reported differently.
# uv says "Request failed after N retries" when it can't reach the
# registry, versus "lockfile needs to be updated" when the lock is
# genuinely stale. Only the second is the contributor's to fix, so
# an unreachable registry says so instead of sending them to
# `uv lock` with nothing to regenerate.
ok=false
net_fail=false
delays=(5 15 45)
for i in 1 2 3; do
if uv lock --check; then
if uv lock --check >lock-check.log 2>&1; then
ok=true
net_fail=false
cat lock-check.log
break
fi
cat lock-check.log
if grep -qiE "Request failed after|Failed to fetch|error sending request|connection (reset|closed)|timed out" lock-check.log; then
net_fail=true
else
# Deterministic failure (a real desync) — retrying just prints
# the same error twice more.
net_fail=false
break
fi
[ "$i" = 3 ] && break
echo "::warning::uv lock --check failed (attempt $i); retrying in 10s"
sleep 10
echo "::warning::uv lock --check could not reach the registry (attempt $i); retrying in ${delays[$((i-1))]}s"
sleep "${delays[$((i-1))]}"
done
if [ "$ok" != true ] && [ "$net_fail" = true ]; then
echo "::error title=uv.lock check could not reach PyPI::Registry unreachable after 3 attempts — infrastructure failure, not a stale lockfile. Re-run the job."
review_status='[{"source":"uv.lock check","results":[{"kind":"action_required","title":"uv.lock check could not reach PyPI","summary":"`uv lock --check` could not reach the package registry after 3 attempts. This is an infrastructure failure, not a stale lockfile.","how_to_fix":"Re-run the failed job. No change to `uv.lock` is needed."}]}]'
echo "review_status=${review_status}" >> "$GITHUB_OUTPUT"
echo "review_status=${review_status}" > review-status.json
exit 1
fi
if [ "$ok" != true ]; then
cat <<'EOF' >> "$GITHUB_STEP_SUMMARY"
## ❌ uv.lock is out of sync with pyproject.toml
@@ -126,7 +161,20 @@ jobs:
echo "::error title=uv.lock out of sync::Run \`uv lock\` locally and commit the result. If on a PR, sync with main first."
review_status='[{"source":"uv.lock check","results":[{"kind":"action_required","title":"uv.lock out of sync","summary":"uv.lock is out of sync with pyproject.toml.","how_to_fix":"Run `uv lock` locally and commit the result. If on a PR, sync with main first:\n```\ngit fetch origin main\ngit rebase origin/main\nuv lock\ngit add uv.lock\ngit commit -m \"chore: refresh uv.lock\"\n```\n"}]}]'
echo "review_status=${review_status}" >> "$GITHUB_OUTPUT"
echo "review_status=${review_status}" > review-status.json
exit 1
fi
review_status='[]'
echo "review_status=${review_status}" >> "$GITHUB_OUTPUT"
echo "review_status=${review_status}" > review-status.json
- name: Upload review status artifact
if: always() && steps.verify.outcome != 'skipped'
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: review-status-uv-lockfile
path: review-status.json
retention-days: 1
overwrite: true
if-no-files-found: ignore
+61
View File
@@ -0,0 +1,61 @@
name: Windows venv-holder live E2E
# ON-DEMAND ONLY (fleet-update #91277, venv-holder consolidation work).
#
# Runs the live venv-holder E2E suite on a real windows-latest runner:
# spawns actual processes with realistic Hermes argv shapes and drives the
# REAL detection/classification/exemption code against the live process
# table — the coverage that cannot exist on the Linux lanes and that the
# maintainer cannot exercise locally before the work reaches main.
#
# Deliberately NOT wired to pull_request/main: it fires only on pushes to
# wine2e/** working branches, so it costs nothing on normal PRs. Delete or
# keep dormant after the venv-holder work lands.
on:
push:
branches:
- "wine2e/**"
permissions:
contents: read
concurrency:
group: windows-venv-e2e-${{ github.ref }}
cancel-in-progress: true
jobs:
venv-holder-e2e:
name: venv-holder live E2E (windows-latest)
runs-on: windows-latest
timeout-minutes: 25
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
version: "0.9.28"
enable-cache: true
cache-dependency-glob: |
pyproject.toml
uv.lock
- name: Set up Python 3.11
uses: ./.github/actions/retry
with:
command: uv python install 3.11
- name: Install dependencies
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra dev
- name: Run venv-holder live E2E
shell: bash
run: |
set -uo pipefail
uv run --no-sync python -m pytest \
tests/hermes_cli/test_venv_holder_windows_live.py \
-o addopts= -v -p no:cacheprovider
+42
View File
@@ -1,6 +1,9 @@
.DS_Store
/venv/
/venv.old/
/venv.stale.runtime-*/
/bin/
/.hermes-runtime/
/_pycache/
*.pyc*
__pycache__/
@@ -29,6 +32,10 @@ __pycache__/model_tools.cpython-310.pyc
__pycache__/web_tools.cpython-310.pyc
logs/
data/
# Bundled community plugin index seed (shipped as package data) — the bare
# `data/` pattern above would otherwise swallow it.
!hermes_cli/data/
!hermes_cli/data/plugin_index.json
.pytest_cache/
test_durations.json
.pytest-cache/
@@ -85,8 +92,18 @@ apps/desktop/dist/
apps/desktop/src/**/*.js
apps/desktop/src/**/*.js.map
apps/desktop/src/**/*.d.ts
# EXCEPT bundled plain-ESM plugin entries (adopted SDK-consumer plugins,
# e.g. hermes-bots): plugin.js IS the source, not tsc output. No .tsx
# sibling exists, so the stale-shadow hazard above cannot apply.
!apps/desktop/src/plugins/*/plugin.js
!apps/desktop/src/global.d.ts
!apps/desktop/src/vite-env.d.ts
# Repo-root build/debug artifacts that must never be committed
/log.txt
/sqlite_leak_fix.png
/*.png.bak
/default.tar.gz
apps/shared/src/**/*.js
apps/shared/src/**/*.js.map
apps/shared/src/**/*.d.ts
@@ -147,6 +164,9 @@ docs/superpowers/*
# Persistent dev sandbox dir (scripts/dev-sandbox.sh --persistent)
.hermes-sandbox/
# Sandbox dirs used by the install/update E2E (tests/install/). The suffix is
# the route name, so each route gets its own tree and two can run at once.
.hermes-sandbox-e2e*/
# Interrupted-update breadcrumb + recovery lock written next to the shared venv
# by `hermes update` / launch-time self-heal. Runtime state, never a code change
@@ -154,6 +174,16 @@ docs/superpowers/*
.update-incomplete
.update-incomplete.lock
# Checkout fingerprint the __pycache__ tree was last validated against
# (launch-time stale-bytecode sweep). Runtime state, never a code change.
.bytecode-fingerprint
.bytecode-fingerprint.tmp
# Installer-written method stamp in the managed checkout root (scripts/install.sh).
# Runtime metadata only — never a code change. Ignore so `git status` stays clean
# and `hermes update`'s untracked autostash does not treat it as a local edit (#66189 / #54855).
/.install_method
# Tool Search live-test harness output — non-deterministic model transcripts,
# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo.
scripts/out/
@@ -172,5 +202,17 @@ apps/desktop/demo/
# image-provider (fal.media) URL — they are NEVER committed to the repo. The
# PR body is the archive. See the hermes-agent-dev skill's
# pr-infographic-workflow reference (storage rule + lapse #8 / #COMMIT-1).
#
# Spelling variants are listed because a single `infographic/` pattern was
# sidestepped by an `infograficos/` directory (#70552). .gitignore is only
# the first line of defence and cannot stop `git add -f` at all — the
# infographic-check CI job is what actually enforces this.
infographic/
infographics/
infograficos/
infografico/
native/fts5_cjk/*.so
# Runtime marker written by hermes update when a lazy dependency refresh is
# interrupted; consumed by launch-time recovery. Never commit it (was tracked
# by accident via 3a69e34702, removed in the #72002 salvage).
.lazy-refresh-incomplete
+2 -1
View File
@@ -18,6 +18,7 @@ Teknium <127238744+teknium1@users.noreply.github.com> <teknium@nousresearch.com>
# Format: Canonical Name <GH-noreply> <commit-email>
# Verified via GH API email search
kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> <kshitijkapoorr@gmail.com>
luyao618 <364939526@qq.com> <364939526@qq.com>
ethernet8023 <arilotter@gmail.com> <arilotter@gmail.com>
nicoloboschi <boschi1997@gmail.com> <boschi1997@gmail.com>
@@ -86,7 +87,7 @@ yongtenglei <yongtenglei@gmail.com> <yongtenglei@gmail.com>
# Nous Research team
benbarclay <ben@nousresearch.com> <ben@nousresearch.com>
jquesnelle <jonny@nousresearch.com> <jonny@nousresearch.com>
yoniebans <jonny@nousresearch.com> <jonny@nousresearch.com>
# GH contributor list verified
spideystreet <dhicham.pro@gmail.com> <dhicham.pro@gmail.com>
+63
View File
@@ -0,0 +1,63 @@
# needed to prevent bad npm that has min-release-age but not exclude
engine-strict=true
min-release-age=14
# allow assistant-ui packages & a couple specific deps since they update a LOT.
# remove this when we stabilize (or we haven't updated in 2 wks)
min-release-age-exclude[]=@assistant-ui/*
min-release-age-exclude[]=assistant-cloud
min-release-age-exclude[]=assistant-stream
min-release-age-exclude[]=@radix-ui/*
min-release-age-exclude[]=radix-ui
min-release-age-exclude[]=safe-content-frame
# react-router 8.3.0 includes fixes for vulns. remove this when 8.3.0 is > 2wks old.
min-release-age-exclude[]=react-router
# eslint 10.8.0 includes fixes for vulns. remove this when 10.8.0 is > 2wks old.
min-release-age-exclude[]=eslint
min-release-age-exclude[]=@eslint/*
# tar 7.5.21 includes fixes for vulns. remove this when 7.5.21 is > 2wks old
min-release-age-exclude[]=tar
# concurrently 10.0.4 includes fixes for vulns. remove this when 10.0.4 is > 2wks old
min-release-age-exclude[]=concurrently
# fast-uri 3.1.4 includes fixes for vulns. remove this when 3.1.4 is > 2wks old
min-release-age-exclude[]=fast-uri
# minimatch 10.2.6 includes fixes for vulns. remove this when 10.2.6 is > 2wks old
min-release-age-exclude[]=minimatch
# brace-expansion 5.0.9 includes fixes for vulns. remove this when 5.0.9 is > 2wks old
min-release-age-exclude[]=brace-expansion
# js-yaml 4.3.1 includes fixes for GHSA-5p4m-2wfm-xmqj. remove when > 2wks old (rel 2026-07-31)
min-release-age-exclude[]=js-yaml
# nanoid 3.3.17 includes fixes for GHSA-2v37-7h3g-55p8. remove when > 2wks old (rel 2026-08-03)
min-release-age-exclude[]=nanoid
# mermaid 11.16.1 includes fixes for 5 GHSAs. remove when > 2wks old (rel 2026-08-04)
min-release-age-exclude[]=mermaid
# dompurify 3.4.13 includes fixes for GHSA-55q2-fjhq-7xh7. remove when > 2wks old (rel 2026-08-03)
min-release-age-exclude[]=dompurify
# vite 8.2.0 is the first release depending on rolldown >= 1.2.1, which fixes
# a rolldown panic that breaks `npm run build` in apps/desktop
# (rolldown/rolldown#10337 — a regression in 1.1.5, the version vite 8.1.5
# pins as ~1.1.5). @oxc-project/types is here because rolldown 1.2.1 pins it
# as `=0.142.0` — an exact pin, so no older release satisfies it and the age
# gate would fail the whole install with ETARGET.
# remove these once vite 8.2.0 is > 2wks old.
min-release-age-exclude[]=vite
min-release-age-exclude[]=rolldown
min-release-age-exclude[]=@rolldown/*
min-release-age-exclude[]=@oxc-project/types
# ink needs
min-release-age-exclude[]=lightningcss
min-release-age-exclude[]=postcss
+1
View File
@@ -0,0 +1 @@
26
-291
View File
@@ -1,291 +0,0 @@
# OpenAI-Compatible API Server for Hermes Agent
## Motivation
Every major chat frontend (Open WebUI 126k★, LobeChat 73k★, LibreChat 34k★,
AnythingLLM 56k★, NextChat 87k★, ChatBox 39k★, Jan 26k★, HF Chat-UI 8k★,
big-AGI 7k★) connects to backends via the OpenAI-compatible REST API with
SSE streaming. By exposing this endpoint, hermes-agent becomes instantly
usable as a backend for all of them — no custom adapters needed.
## What It Enables
```
┌──────────────────┐
│ Open WebUI │──┐
│ LobeChat │ │ POST /v1/chat/completions
│ LibreChat │ ├──► Authorization: Bearer <key> ┌─────────────────┐
│ AnythingLLM │ │ {"messages": [...]} │ hermes-agent │
│ NextChat │ │ │ gateway │
│ Any OAI client │──┘ ◄── SSE streaming response │ (API server) │
└──────────────────┘ └─────────────────┘
```
A user would:
1. Set `API_SERVER_ENABLED=true` in `~/.hermes/.env`
2. Run `hermes gateway` (API server starts alongside Telegram/Discord/etc.)
3. Point Open WebUI (or any frontend) at `http://localhost:8642/v1`
4. Chat with hermes-agent through any OpenAI-compatible UI
## Endpoints
| Method | Path | Purpose |
|--------|------|---------|
| POST | `/v1/chat/completions` | Chat with the agent (streaming + non-streaming) |
| GET | `/v1/models` | List available "models" (returns hermes-agent as a model) |
| GET | `/health` | Health check |
## Architecture
### Option A: Gateway Platform Adapter (recommended)
Create `gateway/platforms/api_server.py` as a new platform adapter that
extends `BasePlatformAdapter`. This is the cleanest approach because:
- Reuses all gateway infrastructure (session management, auth, context building)
- Runs in the same async loop as other adapters
- Gets message handling, interrupt support, and session persistence for free
- Follows the established pattern (like Telegram, Discord, etc.)
- Uses `aiohttp.web` (already a dependency) for the HTTP server
The adapter would start an `aiohttp.web.Application` server in `connect()`
and route incoming HTTP requests through the standard `handle_message()` pipeline.
### Option B: Standalone Component
A separate HTTP server class in `gateway/api_server.py` that creates its own
AIAgent instances directly. Simpler but duplicates session/auth logic.
**Recommendation: Option A** — fits the existing architecture, less code to
maintain, gets all gateway features for free.
## Request/Response Format
### Chat Completions (non-streaming)
```
POST /v1/chat/completions
Authorization: Bearer hermes-api-key-here
Content-Type: application/json
{
"model": "hermes-agent",
"messages": [
{"role": "system", "content": "You are a helpful assistant."},
{"role": "user", "content": "What files are in the current directory?"}
],
"stream": false,
"temperature": 0.7
}
```
Response:
```json
{
"id": "chatcmpl-abc123",
"object": "chat.completion",
"created": 1710000000,
"model": "hermes-agent",
"choices": [{
"index": 0,
"message": {
"role": "assistant",
"content": "Here are the files in the current directory:\n..."
},
"finish_reason": "stop"
}],
"usage": {
"prompt_tokens": 50,
"completion_tokens": 200,
"total_tokens": 250
}
}
```
### Chat Completions (streaming)
Same request with `"stream": true`. Response is SSE:
```
data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"role":"assistant"},"finish_reason":null}]}
data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"content":"Here "},"finish_reason":null}]}
data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"content":"are "},"finish_reason":null}]}
data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}
data: [DONE]
```
### Models List
```
GET /v1/models
Authorization: Bearer hermes-api-key-here
```
Response:
```json
{
"object": "list",
"data": [{
"id": "hermes-agent",
"object": "model",
"created": 1710000000,
"owned_by": "hermes-agent"
}]
}
```
## Key Design Decisions
### 1. Session Management
The OpenAI API is stateless — each request includes the full conversation.
But hermes-agent sessions have persistent state (memory, skills, tool context).
**Approach: Hybrid**
- Default: Stateless. Each request is independent. The `messages` array IS
the conversation. No session persistence between requests.
- Opt-in persistent sessions via `X-Session-ID` header. When provided, the
server maintains session state across requests (conversation history,
memory context, tool state). This enables richer agent behavior.
- The session ID also enables interrupt support — a subsequent request with
the same session ID while one is running triggers an interrupt.
### 2. Streaming
The agent's `run_conversation()` is synchronous and returns the full response.
For real SSE streaming, we need to emit chunks as they're generated.
**Phase 1 (MVP):** Run agent in a thread, return the complete response as
a single SSE chunk + `[DONE]`. This works with all frontends — they just see
a fast single-chunk response. Not true streaming but functional.
**Phase 2:** Add a response callback to AIAgent that emits text chunks as the
LLM generates them. The API server captures these via a queue and streams them
as SSE events. This gives real token-by-token streaming.
**Phase 3:** Stream tool execution progress too — emit tool call/result events
as the agent works, giving frontends visibility into what the agent is doing.
### 3. Tool Transparency
Two modes:
- **Opaque (default):** Frontends see only the final response. Tool calls
happen server-side and are invisible. Best for general-purpose UIs.
- **Transparent (opt-in via header):** Tool calls are emitted as OpenAI-format
tool_call/tool_result messages in the stream. Useful for agent-aware frontends.
### 4. Authentication
- Bearer token via `Authorization: Bearer <key>` header
- Token configured via `API_SERVER_KEY` env var
- Optional: allow unauthenticated local-only access (127.0.0.1 bind)
- Follows the same pattern as other platform adapters
### 5. Model Mapping
Frontends send `"model": "hermes-agent"` (or whatever). The actual LLM model
used is configured server-side in config.yaml. The API server maps any
requested model name to the configured hermes-agent model.
Optionally, allow model passthrough: if the frontend sends
`"model": "anthropic/claude-sonnet-4"`, the agent uses that model. Controlled
by a config flag.
## Configuration
```yaml
# In config.yaml
api_server:
enabled: true
port: 8642
host: "127.0.0.1" # localhost only by default
key: "your-secret-key" # or via API_SERVER_KEY env var
allow_model_override: false # let clients choose the model
max_concurrent: 5 # max simultaneous requests
```
Environment variables:
```bash
API_SERVER_ENABLED=true
API_SERVER_PORT=8642
API_SERVER_HOST=127.0.0.1
API_SERVER_KEY=your-secret-key
```
## Implementation Plan
### Phase 1: MVP (non-streaming) — PR
1. `gateway/platforms/api_server.py` — new adapter
- aiohttp.web server with endpoints:
- `POST /v1/chat/completions` — Chat Completions API (universal compat)
- `POST /v1/responses` — Responses API (server-side state, tool preservation)
- `GET /v1/models` — list available models
- `GET /health` — health check
- Bearer token auth middleware
- Non-streaming responses (run agent, return full result)
- Chat Completions: stateless, messages array is the conversation
- Responses API: server-side conversation storage via previous_response_id
- Store full internal conversation (including tool calls) keyed by response ID
- On subsequent requests, reconstruct full context from stored chain
- Frontend system prompt layered on top of hermes-agent's core prompt
2. `gateway/config.py` — add `Platform.API_SERVER` enum + config
3. `gateway/run.py` — register adapter in `_create_adapter()`
4. Tests in `tests/gateway/test_api_server.py`
### Phase 2: SSE Streaming
1. Add response streaming to both endpoints
- Chat Completions: `choices[0].delta.content` SSE format
- Responses API: semantic events (response.output_text.delta, etc.)
- Run agent in thread, collect output via callback queue
- Handle client disconnect (cancel agent)
2. Add `stream_callback` parameter to `AIAgent.run_conversation()`
### Phase 3: Enhanced Features
1. Tool call transparency mode (opt-in)
2. Model passthrough/override
3. Concurrent request limiting
4. Usage tracking / rate limiting
5. CORS headers for browser-based frontends
6. GET /v1/responses/{id} — retrieve stored response
7. DELETE /v1/responses/{id} — delete stored response
## Files Changed
| File | Change |
|------|--------|
| `gateway/platforms/api_server.py` | NEW — main adapter (~300 lines) |
| `gateway/config.py` | Add Platform.API_SERVER + config (~20 lines) |
| `gateway/run.py` | Register adapter in _create_adapter() (~10 lines) |
| `tests/gateway/test_api_server.py` | NEW — tests (~200 lines) |
| `cli-config.yaml.example` | Add api_server section |
| `README.md` | Mention API server in platform list |
## Compatibility Matrix
Once implemented, hermes-agent works as a drop-in backend for:
| Frontend | Stars | How to Connect |
|----------|-------|---------------|
| Open WebUI | 126k | Settings → Connections → Add OpenAI API, URL: `http://localhost:8642/v1` |
| NextChat | 87k | BASE_URL env var |
| LobeChat | 73k | Custom provider endpoint |
| AnythingLLM | 56k | LLM Provider → Generic OpenAI |
| Oobabooga | 42k | Already a backend, not a frontend |
| ChatBox | 39k | API Host setting |
| LibreChat | 34k | librechat.yaml custom endpoint |
| Chatbot UI | 29k | Custom API endpoint |
| Jan | 26k | Remote model config |
| AionUI | 18k | Custom API endpoint |
| HF Chat-UI | 8k | OPENAI_BASE_URL env var |
| big-AGI | 7k | Custom endpoint |
-705
View File
@@ -1,705 +0,0 @@
# Streaming LLM Response Support for Hermes Agent
## Overview
Add token-by-token streaming of LLM responses across all platforms. When enabled,
users see the response typing out live instead of waiting for the full generation.
Streaming is opt-in via config, defaults to off, and all existing non-streaming
code paths remain intact as the default.
## Design Principles
1. **Feature-flagged**: `streaming.enabled: true` in config.yaml. Off by default.
When off, all existing code paths are unchanged — zero risk to current behavior.
2. **Callback-based**: A simple `stream_callback(text_delta: str)` function injected
into AIAgent. The agent doesn't know or care what the consumer does with tokens.
3. **Graceful degradation**: If the provider doesn't support streaming, or streaming
fails for any reason, silently fall back to the non-streaming path.
4. **Platform-agnostic core**: The streaming mechanism in AIAgent works the same
regardless of whether the consumer is CLI, Telegram, Discord, or the API server.
---
## Architecture
```
stream_callback(delta)
│
┌─────────────┐ ┌─────────────▼──────────────┐
│ LLM API │ │ queue.Queue() │
│ (stream) │───►│ thread-safe bridge between │
│ │ │ agent thread & consumer │
└─────────────┘ └─────────────┬──────────────┘
│
┌──────────────┼──────────────┐
│ │ │
┌─────▼─────┐ ┌─────▼─────┐ ┌─────▼─────┐
│ CLI │ │ Gateway │ │ API Server│
│ print to │ │ edit msg │ │ SSE event │
│ terminal │ │ on Tg/Dc │ │ to client │
└───────────┘ └───────────┘ └───────────┘
```
The agent runs in a thread. The callback puts tokens into a thread-safe queue.
Each consumer reads the queue in its own context (async task, main thread, etc.).
---
## Configuration
### config.yaml
```yaml
streaming:
enabled: false # Master switch. Default off.
# Per-platform overrides (optional):
# cli: true # Override for CLI only
# telegram: true # Override for Telegram only
# discord: false # Keep Discord non-streaming
# api_server: true # Override for API server
```
### Environment variables
```
HERMES_STREAMING_ENABLED=true # Master switch via env
```
### How the flag is read
- **CLI**: `load_cli_config()` reads `streaming.enabled`, sets env var. AIAgent
checks at init time.
- **Gateway**: `_run_agent()` reads config, decides whether to pass
`stream_callback` to the AIAgent constructor.
- **API server**: For Chat Completions `stream=true` requests, always uses streaming
regardless of config (the client is explicitly requesting it). For non-stream
requests, uses config.
### Precedence
1. API server: client's `stream` field overrides everything
2. Per-platform config override (e.g., `streaming.telegram: true`)
3. Master `streaming.enabled` flag
4. Default: off
---
## Implementation Plan
### Phase 1: Core streaming infrastructure in AIAgent
**File: run_agent.py**
#### 1a. Add stream_callback parameter to __init__ (~5 lines)
```python
def __init__(self, ..., stream_callback: callable = None, ...):
self.stream_callback = stream_callback
```
No other init changes. The callback is optional — when None, everything
works exactly as before.
#### 1b. Add _run_streaming_chat_completion() method (~65 lines)
New method for Chat Completions API streaming:
```python
def _run_streaming_chat_completion(self, api_kwargs: dict):
"""Stream a chat completion, emitting text tokens via stream_callback.
Returns a fake response object compatible with the non-streaming code path.
Falls back to non-streaming on any error.
"""
stream_kwargs = dict(api_kwargs)
stream_kwargs["stream"] = True
stream_kwargs["stream_options"] = {"include_usage": True}
accumulated_content = []
accumulated_tool_calls = {} # index -> {id, name, arguments}
final_usage = None
try:
stream = self.client.chat.completions.create(**stream_kwargs)
for chunk in stream:
if not chunk.choices:
# Usage-only chunk (final)
if chunk.usage:
final_usage = chunk.usage
continue
delta = chunk.choices[0].delta
# Text content — emit via callback
if delta.content:
accumulated_content.append(delta.content)
if self.stream_callback:
try:
self.stream_callback(delta.content)
except Exception:
pass
# Tool call deltas — accumulate silently
if delta.tool_calls:
for tc_delta in delta.tool_calls:
idx = tc_delta.index
if idx not in accumulated_tool_calls:
accumulated_tool_calls[idx] = {
"id": tc_delta.id or "",
"name": "", "arguments": ""
}
if tc_delta.function:
if tc_delta.function.name:
accumulated_tool_calls[idx]["name"] = tc_delta.function.name
if tc_delta.function.arguments:
accumulated_tool_calls[idx]["arguments"] += tc_delta.function.arguments
# Build fake response compatible with existing code
tool_calls = []
for idx in sorted(accumulated_tool_calls):
tc = accumulated_tool_calls[idx]
if tc["name"]:
tool_calls.append(SimpleNamespace(
id=tc["id"], type="function",
function=SimpleNamespace(name=tc["name"], arguments=tc["arguments"]),
))
return SimpleNamespace(
choices=[SimpleNamespace(
message=SimpleNamespace(
content="".join(accumulated_content) or "",
tool_calls=tool_calls or None,
role="assistant",
),
finish_reason="tool_calls" if tool_calls else "stop",
)],
usage=final_usage,
model=self.model,
)
except Exception as e:
logger.debug("Streaming failed, falling back to non-streaming: %s", e)
return self.client.chat.completions.create(**api_kwargs)
```
#### 1c. Modify _run_codex_stream() for Responses API (~10 lines)
The method already iterates the stream. Add callback emission:
```python
def _run_codex_stream(self, api_kwargs: dict):
with self.client.responses.stream(**api_kwargs) as stream:
for event in stream:
# Emit text deltas if streaming callback is set
if self.stream_callback and hasattr(event, 'type'):
if event.type == 'response.output_text.delta':
try:
self.stream_callback(event.delta)
except Exception:
pass
return stream.get_final_response()
```
#### 1d. Modify _interruptible_api_call() (~5 lines)
Add the streaming branch:
```python
def _call():
try:
if self.api_mode == "codex_responses":
result["response"] = self._run_codex_stream(api_kwargs)
elif self.stream_callback is not None:
result["response"] = self._run_streaming_chat_completion(api_kwargs)
else:
result["response"] = self.client.chat.completions.create(**api_kwargs)
except Exception as e:
result["error"] = e
```
#### 1e. Signal end-of-stream to consumers (~5 lines)
After the API call returns, signal the callback that streaming is done
so consumers can finalize (remove cursor, close SSE, etc.):
```python
# In run_conversation(), after _interruptible_api_call returns:
if self.stream_callback:
try:
self.stream_callback(None) # None = end of stream signal
except Exception:
pass
```
Consumers check: `if delta is None: finalize()`
**Tests for Phase 1:** (~150 lines)
- Test _run_streaming_chat_completion with mocked stream
- Test fallback to non-streaming on error
- Test tool_call accumulation during streaming
- Test stream_callback receives correct deltas
- Test None signal at end of stream
- Test streaming disabled when callback is None
---
### Phase 2: Gateway consumers (Telegram, Discord, etc.)
**File: gateway/run.py**
#### 2a. Read streaming config (~15 lines)
In `_run_agent()`, before creating the AIAgent:
```python
# Read streaming config
_streaming_enabled = False
try:
# Check per-platform override first
platform_key = source.platform.value if source.platform else ""
_stream_cfg = {} # loaded from config.yaml streaming section
if _stream_cfg.get(platform_key) is not None:
_streaming_enabled = bool(_stream_cfg[platform_key])
else:
_streaming_enabled = bool(_stream_cfg.get("enabled", False))
except Exception:
pass
# Env var override
if os.getenv("HERMES_STREAMING_ENABLED", "").lower() in ("true", "1", "yes"):
_streaming_enabled = True
```
#### 2b. Set up queue + callback (~15 lines)
```python
_stream_q = None
_stream_done = None
_stream_msg_id = [None] # mutable ref for the async task
if _streaming_enabled:
import queue as _q
_stream_q = _q.Queue()
_stream_done = threading.Event()
def _on_token(delta):
if delta is None:
_stream_done.set()
else:
_stream_q.put(delta)
```
Pass `stream_callback=_on_token` to the AIAgent constructor.
#### 2c. Telegram/Discord stream preview task (~50 lines)
```python
async def stream_preview():
"""Progressively edit a message with streaming tokens."""
if not _stream_q:
return
adapter = self.adapters.get(source.platform)
if not adapter:
return
accumulated = []
token_count = 0
last_edit = 0.0
MIN_TOKENS = 20 # Don't show until enough context
EDIT_INTERVAL = 1.5 # Respect Telegram rate limits
try:
while not _stream_done.is_set():
try:
chunk = _stream_q.get(timeout=0.1)
accumulated.append(chunk)
token_count += 1
except queue.Empty:
continue
now = time.monotonic()
if token_count >= MIN_TOKENS and (now - last_edit) >= EDIT_INTERVAL:
preview = "".join(accumulated) + " ▌"
if _stream_msg_id[0] is None:
r = await adapter.send(
chat_id=source.chat_id,
content=preview,
metadata=_thread_metadata,
)
if r.success and r.message_id:
_stream_msg_id[0] = r.message_id
else:
await adapter.edit_message(
chat_id=source.chat_id,
message_id=_stream_msg_id[0],
content=preview,
)
last_edit = now
# Drain remaining tokens
while not _stream_q.empty():
accumulated.append(_stream_q.get_nowait())
# Final edit — remove cursor, show complete text
if _stream_msg_id[0] and accumulated:
await adapter.edit_message(
chat_id=source.chat_id,
message_id=_stream_msg_id[0],
content="".join(accumulated),
)
except asyncio.CancelledError:
# Clean up on cancel
if _stream_msg_id[0] and accumulated:
try:
await adapter.edit_message(
chat_id=source.chat_id,
message_id=_stream_msg_id[0],
content="".join(accumulated),
)
except Exception:
pass
except Exception as e:
logger.debug("stream_preview error: %s", e)
```
#### 2d. Skip final send if already streamed (~10 lines)
In `_process_message_background()` (base.py), after getting the response,
if streaming was active and `_stream_msg_id[0]` is set, the final response
was already delivered via progressive edits. Skip the normal `self.send()`
call to avoid duplicating the message.
This is the most delicate integration point — we need to communicate from
the gateway's `_run_agent` back to the base adapter's response sender that
the response was already delivered. Options:
- **Option A**: Return a special marker in the result dict:
`result["_streamed_msg_id"] = _stream_msg_id[0]`
The base adapter checks this and skips `send()`.
- **Option B**: Edit the already-sent message with the final response
(which may differ slightly from accumulated tokens due to think-block
stripping, etc.) and don't send a new one.
- **Option C**: The stream preview task handles the FULL final response
(including any post-processing), and the handler returns None to skip
the normal send path.
Recommended: **Option A** — cleanest separation. The result dict already
carries metadata; adding one more field is low-risk.
**Platform-specific considerations:**
| Platform | Edit support | Rate limits | Streaming approach |
|----------|-------------|-------------|-------------------|
| Telegram | ✅ edit_message_text | ~20 edits/min | Edit every 1.5s |
| Discord | ✅ message.edit | 5 edits/5s per message | Edit every 1.2s |
| Slack | ✅ chat.update | Tier 3 (~50/min) | Edit every 1.5s |
| WhatsApp | ❌ no edit support | N/A | Skip streaming, use normal path |
| HomeAssistant | ❌ no edit | N/A | Skip streaming |
| API Server | ✅ SSE native | No limit | Real SSE events |
WhatsApp and HomeAssistant fall back to non-streaming automatically because
they don't support message editing.
**Tests for Phase 2:** (~100 lines)
- Test stream_preview sends/edits correctly
- Test skip-final-send when streaming delivered
- Test WhatsApp/HA graceful fallback
- Test streaming disabled per-platform config
- Test thread_id metadata forwarded in stream messages
---
### Phase 3: CLI streaming
**File: cli.py**
#### 3a. Set up callback in the CLI chat loop (~20 lines)
In `_chat_once()` or wherever the agent is invoked:
```python
if streaming_enabled:
_stream_q = queue.Queue()
_stream_done = threading.Event()
def _cli_stream_callback(delta):
if delta is None:
_stream_done.set()
else:
_stream_q.put(delta)
agent.stream_callback = _cli_stream_callback
```
#### 3b. Token display thread/task (~30 lines)
Start a thread that reads the queue and prints tokens:
```python
def _stream_display():
"""Print tokens to terminal as they arrive."""
first_token = True
while not _stream_done.is_set():
try:
delta = _stream_q.get(timeout=0.1)
except queue.Empty:
continue
if first_token:
# Print response box top border
_cprint(f"\n{top}")
first_token = False
sys.stdout.write(delta)
sys.stdout.flush()
# Drain remaining
while not _stream_q.empty():
sys.stdout.write(_stream_q.get_nowait())
sys.stdout.flush()
# Print bottom border
_cprint(f"\n\n{bot}")
```
**Integration challenge: prompt_toolkit**
The CLI uses prompt_toolkit which controls the terminal. Writing directly
to stdout while prompt_toolkit is active can cause display corruption.
The existing KawaiiSpinner already solves this by using prompt_toolkit's
`patch_stdout` context. The streaming display would need to do the same.
Alternative: use `_cprint()` for each token chunk (routes through
prompt_toolkit's renderer). But this might be slow for individual tokens.
Recommended approach: accumulate tokens in small batches (e.g., every 50ms)
and `_cprint()` the batch. This balances display responsiveness with
prompt_toolkit compatibility.
**Tests for Phase 3:** (~50 lines)
- Test CLI streaming callback setup
- Test response box borders with streaming
- Test fallback when streaming disabled
---
### Phase 4: API Server real streaming
**File: gateway/platforms/api_server.py**
Replace the pseudo-streaming `_write_sse_chat_completion()` with real
token-by-token SSE when the agent supports it.
#### 4a. Wire streaming callback for stream=true requests (~20 lines)
```python
if stream:
_stream_q = queue.Queue()
def _api_stream_callback(delta):
_stream_q.put(delta) # None = done
# Pass callback to _run_agent
result, usage = await self._run_agent(
..., stream_callback=_api_stream_callback,
)
```
#### 4b. Real SSE writer (~40 lines)
```python
async def _write_real_sse(self, request, completion_id, model, stream_q):
response = web.StreamResponse(
headers={"Content-Type": "text/event-stream", "Cache-Control": "no-cache"},
)
await response.prepare(request)
# Role chunk
await response.write(...)
# Stream content chunks as they arrive
while True:
try:
delta = await asyncio.get_event_loop().run_in_executor(
None, lambda: stream_q.get(timeout=0.1)
)
except queue.Empty:
continue
if delta is None: # End of stream
break
chunk = {"id": completion_id, "object": "chat.completion.chunk", ...
"choices": [{"delta": {"content": delta}, ...}]}
await response.write(f"data: {json.dumps(chunk)}\n\n".encode())
# Finish + [DONE]
await response.write(...)
await response.write(b"data: [DONE]\n\n")
return response
```
**Challenge: concurrent execution**
The agent runs in a thread executor. SSE writing happens in the async event
loop. The queue bridges them. But `_run_agent()` currently awaits the full
result before returning. For real streaming, we need to start the agent in
the background and stream tokens while it runs:
```python
# Start agent in background
agent_task = asyncio.create_task(self._run_agent_async(...))
# Stream tokens while agent runs
await self._write_real_sse(request, ..., stream_q)
# Agent is done by now (stream_q received None)
result, usage = await agent_task
```
This requires splitting `_run_agent` into an async version that doesn't
block waiting for the result, or running it in a separate task.
**Responses API SSE format:**
For `/v1/responses` with `stream=true`, the SSE events are different:
```
event: response.output_text.delta
data: {"type":"response.output_text.delta","delta":"Hello"}
event: response.completed
data: {"type":"response.completed","response":{...}}
```
This needs a separate SSE writer that emits Responses API format events.
**Tests for Phase 4:** (~80 lines)
- Test real SSE streaming with mocked agent
- Test SSE event format (Chat Completions vs Responses)
- Test client disconnect during streaming
- Test fallback to pseudo-streaming when callback not available
---
## Integration Issues & Edge Cases
### 1. Tool calls during streaming
When the model returns tool calls instead of text, no text tokens are emitted.
The stream_callback is simply never called with text. After tools execute, the
next API call may produce the final text response — streaming picks up again.
The stream preview task needs to handle this: if no tokens arrive during a
tool-call round, don't send/edit any message. The tool progress messages
continue working as before.
### 2. Duplicate messages
The biggest risk: the agent sends the final response normally (via the
existing send path) AND the stream preview already showed it. The user
sees the response twice.
Prevention: when streaming is active and tokens were delivered, the final
response send must be suppressed. The `result["_streamed_msg_id"]` marker
tells the base adapter to skip its normal send.
### 3. Response post-processing
The final response may differ from the accumulated streamed tokens:
- Think block stripping (`<think>...</think>` removed)
- Trailing whitespace cleanup
- Tool result media tag appending
The stream preview shows raw tokens. The final edit should use the
post-processed version. This means the final edit (removing the cursor)
should use the post-processed `final_response`, not just the accumulated
stream text.
### 4. Context compression during streaming
If the agent triggers context compression mid-conversation, the streaming
tokens from BEFORE compression are from a different context than those
after. This isn't a problem in practice — compression happens between
API calls, not during streaming.
### 5. Interrupt during streaming
User sends a new message while streaming → interrupt. The stream is killed
(HTTP connection closed), accumulated tokens are shown as-is (no cursor),
and the interrupt message is processed normally. This is already handled by
`_interruptible_api_call` closing the client.
### 6. Multi-model / fallback
If the primary model fails and the agent falls back to a different model,
streaming state resets. The fallback call may or may not support streaming.
The graceful fallback in `_run_streaming_chat_completion` handles this.
### 7. Rate limiting on edits
Telegram: ~20 edits/minute (~1 every 3 seconds to be safe)
Discord: 5 edits per 5 seconds per message
Slack: ~50 API calls/minute
The 1.5s edit interval is conservative enough for all platforms. If we get
429 rate limit errors on edits, just skip that edit cycle and try next time.
---
## Files Changed Summary
| File | Phase | Changes |
|------|-------|---------|
| `run_agent.py` | 1 | +stream_callback param, +_run_streaming_chat_completion(), modify _run_codex_stream(), modify _interruptible_api_call() |
| `gateway/run.py` | 2 | +streaming config reader, +queue/callback setup, +stream_preview task, +skip-final-send logic |
| `gateway/platforms/base.py` | 2 | +check for _streamed_msg_id in response handler |
| `cli.py` | 3 | +streaming setup, +token display, +response box integration |
| `gateway/platforms/api_server.py` | 4 | +real SSE writer, +streaming callback wiring |
| `hermes_cli/config.py` | 1 | +streaming config defaults |
| `cli-config.yaml.example` | 1 | +streaming section |
| `tests/test_streaming.py` | 1-4 | NEW — ~380 lines of tests |
**Total new code**: ~500 lines across all phases
**Total test code**: ~380 lines
---
## Rollout Plan
1. **Phase 1** (core): Merge to main. Streaming disabled by default.
Zero impact on existing behavior. Can be tested with env var.
2. **Phase 2** (gateway): Merge to main. Test on Telegram manually.
Enable per-platform: `streaming.telegram: true` in config.
3. **Phase 3** (CLI): Merge to main. Test in terminal.
Enable: `streaming.cli: true` or `streaming.enabled: true`.
4. **Phase 4** (API server): Merge to main. Test with Open WebUI.
Auto-enabled when client sends `stream: true`.
Each phase is independently mergeable and testable. Streaming stays
off by default throughout. Once all phases are stable, consider
changing the default to enabled.
---
## Config Reference (final state)
```yaml
# config.yaml
streaming:
enabled: false # Master switch (default: off)
cli: true # Per-platform override
telegram: true
discord: true
slack: true
api_server: true # API server always streams when client requests it
edit_interval: 1.5 # Seconds between message edits (default: 1.5)
min_tokens: 20 # Tokens before first display (default: 20)
```
```bash
# Environment variable override
HERMES_STREAMING_ENABLED=true
```
+1
View File
@@ -0,0 +1 @@
3.11
+370 -20
View File
@@ -210,6 +210,45 @@ backends, providers, notifiers), don't merge them one at a time — design an
ABC + orchestrator, wrap the existing built-in as the first provider, and turn
the competing PRs into plugins against that interface.
### Surface capability is a property of the SESSION, never of the process env
A tool that only works because of *who is on the other end of the connection* —
the desktop app's panes, the in-app browser, message reactions, Projects — must
resolve its availability from the **session's own source**, not from an env var
on the backend process.
The client and the backend are separate machines on separate clocks. The
desktop app can be driving a backend Electron spawned locally, one over SSH,
one behind a plain URL + token, or Hermes Cloud. Only the first two are spawned
by us and carry `HERMES_DESKTOP=1`. Every env-keyed GUI gate is therefore a
silent no-op on the other half of the topologies, and the failure is invisible:
the tool is stripped from the schema before the model ever sees it, on the same
backend whose platform hint is telling the model it's *"chatting inside the
Hermes desktop app."*
The pattern that works:
- **The toolset is the surface gate.** Keep the tools off `_HERMES_CORE_TOOLS`
(nobody else should pay their schema) and put them in a named toolset —
`desktop_ui`, `project`. The GUI gateway's `_load_enabled_toolsets(platform)`
folds that toolset in when the session's platform says GUI. One resolver,
every topology.
- **`check_fn` answers reachability or user opt-in, not surface.** "Is the
renderer bridge wired?", "did the user enable reactions?" — fine. "Was I
spawned by Electron?" — not fine. `check_fn` results are also TTL-cached
process-wide (`tools/registry.py`), so a per-session answer does not belong
there at all: one process serves many sessions.
- **Ask which identity you actually mean.** `HERMES_DESKTOP=1` legitimately
marks *"this backend process was spawned by the app"* — it gates the cron
ticker and web-dist handling correctly. It does NOT mean "a GUI is watching",
and the embedded terminal pane (`hermes --tui` against that same backend) is
the standing counterexample.
Same test both ways: if the capability would still make sense with the client
on another machine, it is session-scoped. Cover it with a test that asserts the
GUI session gets the tool **with the env var absent** — that's the assertion
the original gate could never have passed.
## Development Environment
```bash
@@ -325,7 +364,7 @@ class AIAgent:
provider: str = None,
api_mode: str = None, # "chat_completions" | "codex_responses" | ...
model: str = "", # empty → resolved from config/provider later
max_iterations: int = 90, # tool-calling iterations (shared with subagents)
max_iterations: int = 500, # tool-calling iterations (shared with subagents)
enabled_toolsets: list = None,
disabled_toolsets: list = None,
quiet_mode: bool = False,
@@ -758,12 +797,45 @@ as a side effect of importing `model_tools.py`. Code paths that read plugin
state without importing `model_tools.py` first must call `discover_plugins()`
explicitly (it's idempotent).
#### Native plugin compatibility policy
The canonical contract and deprecation policy live in
`website/docs/developer-guide/plugins/index.md#native-plugin-compatibility-contract`.
Compatibility is enforced as a behavior contract, not through a monolithic
`PLUGIN_API_VERSION`, a manifest-wide native `api:` match, or version literals
on unrelated payloads. Keep documented plugin surfaces additive:
- add hook payload data as keyword fields; signature-inspect callbacks so old
narrow signatures receive only fields they declare, while `**kwargs`
callbacks receive the complete payload;
- do not remove or rename `PluginContext` methods; make new parameters optional
with defaults and keyword-only where possible;
- ignore unknown native manifest fields;
- give new provider methods default implementations, and signature-inspect
optional callback kwargs rather than forwarding them unconditionally;
- use a local schema version only for a capability with a wire or persisted
contract, and preserve old state/config/session replay or ship a migration.
Deprecations require a once-per-process warning, a documented replacement and
migration note, and at least two subsequent minor releases before removal.
Compatibility tests must load frozen plugins through the real discovery path
and assert outcomes. Do not replace these with exact registry/catalog counts,
source-reading tests, or assertions that a global version literal changed.
### Memory-provider plugins (`plugins/memory/<name>/`)
Separate discovery system for pluggable memory backends. Current built-in
providers include **honcho, mem0, supermemory, byterover, hindsight,
holographic, openviking, retaindb**.
Discovery covers the same four sources as the general `PluginManager` —
bundled, `$HERMES_HOME/plugins/`, `./.hermes/plugins/` (opt-in via
`HERMES_ENABLE_PROJECT_PLUGINS`), and `hermes_agent.memory_providers` entry
points — but with **bundled-first** precedence, the reverse of the general
system's later-wins order: a memory provider is activated by name, so a
dropped-in directory must not be able to shadow a shipped one. Discovery
enumerates without importing; nothing runs until `memory.provider` names it.
Each provider implements the `MemoryProvider` ABC (see `agent/memory_provider.py`)
and is orchestrated by `agent/memory_manager.py`. Lifecycle hooks include
`sync_turn(turn_messages)`, `prefetch(query)`, `shutdown()`, and optional
@@ -848,6 +920,72 @@ plug into `agent/context_engine.py`; image-gen providers into
[`hermes-example-plugins`](https://github.com/NousResearch/hermes-example-plugins)
companion repo, not in this tree.
### Bot Mode (`apps/desktop/src/plugins/hermes-bots/`)
The desktop "Bots" experience ships bundled in-tree. Each bot is a Hermes
agent **profile** with a persistent identity. Its design rests on one settled
invariant that has been regressed repeatedly, cost users real conversation
history each time, and is not open for re-litigation in a routine PR:
**One bot = ONE canonical forever-chat, identified by NAME.** The chat's one
and only identity is **(profile, session titled exactly "Bot Chat")** — the
state DB's UNIQUE(title) index makes that pair an exact registry of at most
one row. The full lifecycle when a bot row is clicked:
1. **Resolve the registry, every time.** Look up the profile's `Bot Chat`
session by exact title via `session.list {title, include_hidden: true}`
(indexed, window-free; hidden rows resolve because canonical chats are
always hidden; compression lineages resolve to the live tip). Row exists →
open it. That is the entire happy path.
2. **No row → create it,** titled `Bot Chat`, born hidden, kicked off with
the bot's intro. Creation adopts-before-minting: it re-runs the registry
lookup first, so a concurrent or pre-existing row is opened, never forked.
(`set_session_title` silently drops conflicting titles — returns 0 rows —
which is how the 2026-08 infinite fork loop started; adopt-before-mint is
what kills it.)
**There is NO session-id pin.** The previous design stored a pointer in
`ui_meta['hermes-bots'].chat` and verified it per click; five hardening
waves (#88690, #90732, #90751, the #91791 revert, #92042) each guarded a new
way that pointer dangled or got stolen — rows[0] steals, `last_session`
adoptions, transient clears, drifted-title welds (a pin re-anchored onto a
cron session passed every guard). Name-as-identity removes the failure class:
a name cannot dangle, and a corrupted historical pointer simply never gets
read. Legacy `chat` keys in ui_meta are ignored and dropped from merges.
Why recency must never win (the #91791 → #92042 lesson): canonical Bot
Chats are **unconditionally hidden** from the Sessions sidebar, so the bot
row is the ONLY door to the forever-chat. A "newest visible session wins"
preference doesn't re-order two equivalent entry points — it walls the
entire relationship off behind a row that previews one session and opens
another, and any stray draft that catches a prompt captures the row.
Side-chats started via "New chat with this agent" are not plumbing-titled,
stay visible in the Sessions sidebar, and are reachable there; they are
never the bot row's target.
Corollaries for reviewers:
- There is no per-bot session browser, by explicit design (removed in
#90732). Do not add one back.
- Reject any PR that reintroduces a stored session-id pointer as canonical
identity — including "as a fallback tier" or "for verification". The
registry lookup is the whole contract; pointers are how every prior
incident started.
- Reject any PR that consults recency, visibility, or "where the user left
off" for the bot row's target — reports that motivate such a change are
almost always about side-chats, and the fix belongs in the Sessions
sidebar (hide-sweep false positives), not in the bot row's target.
- The gateway reports the registry row per profile as `canonical_session`
on `profiles.list` (resolved server-side by title); roster preview,
activity signals, and the `/new`→`/compact` guard all read it, so preview
identity and click identity are the same row by construction.
Regression tests encoding this contract:
`tests/canonical-chat-registry.test.mjs` (includes a tripwire asserting the
open path never reads or writes a stored pointer),
`tests/canonical-chat-creation.test.mjs`, `tests/hide-bot-chats.test.mjs`,
and `tests/tui_gateway/test_profiles_list_canonical_session.py`.
---
## Skills
@@ -1096,15 +1234,16 @@ kanban task.
- **CLI:** `hermes_cli/kanban.py` wires `hermes kanban` with verbs
`init`, `create`, `list` (alias `ls`), `show`, `assign`, `link`,
`unlink`, `comment`, `attach`, `attachments`, `attach-rm`, `complete`,
`block`, `unblock`, `archive`, `tail`, plus less-commonly-used `watch`,
`stats`, `runs`, `log`, `assignees`, `heartbeat`, `notify-*`,
`dispatch`, `daemon`, `gc`.
`request-review`, `request-changes`, `reopen-review`, `block`, `unblock`, `archive`,
`tail`, plus less-commonly-used `watch`, `stats`, `runs`, `log`,
`assignees`, `heartbeat`, `notify-*`, `dispatch`, `daemon`, `gc`.
- **Worker/orchestrator toolset:** `tools/kanban_tools.py` exposes
`kanban_show`, `kanban_complete`, `kanban_block`, `kanban_heartbeat`,
`kanban_comment`, `kanban_create`, `kanban_link`, `kanban_attach`,
`kanban_attach_url`, `kanban_attachments`; profiles that explicitly
enable the `kanban` toolset outside a dispatcher-spawned task also get
`kanban_list` and `kanban_unblock` for board routing.
`kanban_show`, `kanban_complete`, `kanban_request_review`,
`kanban_request_changes`, `kanban_block`,
`kanban_heartbeat`, `kanban_comment`, `kanban_create`, `kanban_link`,
`kanban_attach`, `kanban_attach_url`, `kanban_attachments`; profiles that
explicitly enable the `kanban` toolset outside a dispatcher-spawned
task also get `kanban_list` and `kanban_unblock` for board routing.
- **Dispatcher:** long-lived loop that (default every 60s) reclaims
stale claims, promotes ready tasks, atomically claims, and spawns
assigned profiles. Runs **inside the gateway** by default via
@@ -1128,6 +1267,79 @@ Full user-facing docs: `website/docs/user-guide/features/kanban.md`.
---
## Update Pipeline (`hermes update`)
The updater is transactional in shape (fleet-update campaign, #91277 —
Aug 2026). Every stage exists because its absence was a real field
failure; PRs that weaken a stage need to answer for the failure class it
guards:
```
plan → snapshot → apply → restart-per-kind → verify → report
```
- **Plan** (`hermes_cli/update_inventory.py`, `hermes update --plan`):
read-only inventory — install kind, all profiles, every live gateway
with supervisor + running code version. Deployment kinds are
first-class: `git` updates in place; `docker`/`nix`/`apt` are NOT
in-place-updatable and the updater reports the correct external
command instead of fighting the deployment model.
- **Snapshot** (`hermes_cli/backup.py`): pre-update quick snapshot for
EVERY profile (the code swap + fleet restart touch all of them), each
into its own `state-snapshots/`, identical file set + 1 GiB per-file
cap + keep=1. **Never add a partial/tiered snapshot set** — mixed
coverage creates torn-restore states across schema generations. Quick
snapshots are FILE-LOSS RECOVERY (the per-profile cron-jobs safety
net restores from them), NOT code-rollback insurance; `--backup` full
mode owns rollback.
- **Apply**: git pull, or the Windows ZIP fallback — which fires ONLY
when git itself failed (`_should_zip_fallback_on_update_error`,
argv-classified; a dependency-install failure must never trigger a
tree-clobbering re-download), REFUSES a dirty working tree
(`-uall`, plus a pre-swap TOCTOU re-check), and grafts the live
`apps/desktop/release/` into the staged swap (the GitHub source ZIP
has no built desktop app; without the graft the swap deletes it).
- **Restart-per-kind**: systemd and launchd restarts are FLEET-WIDE
(every `hermes-gateway*` unit / `ai.hermes.gateway*` LaunchAgent),
drain-first (SIGUSR1) with per-unit/per-label failure isolation.
Restarting only the invoking profile's service leaves siblings on
stale `sys.modules` until they crash — the largest dupe-PR cluster in
the repo's history came from that bug.
- **Verify**: gateways stamp their running `code_sha`/`code_version`
into `gateway_state.json` on every runtime-status write
(`gateway/status.py`); after the restart phase the updater compares
each live gateway against the fresh checkout and prints a fleet
version matrix. A provably-stale gateway fails the update (exit 1) —
automation must never treat a mixed-version fleet as healthy.
- **Report**: every run writes a machine-readable receipt to
`~/.hermes/logs/update_receipts/` (`latest.json` pointer; steps,
skips WITH reasons, restart outcome, plan, fleet snapshot).
Finalization is owned by the `cmd_update` command boundary — early
`sys.exit` paths (preflight refusals, fetch failures) still persist
a receipt with the real exit code. A begun-but-unwritten receipt is
a bug: the refused/failed runs are the ones receipts exist for.
Architecture direction: process-scan-based coordination between the
updater, serve/dashboard, and the gateway is being replaced by a
gateway-owned control socket (#92091). Do not add new scan heuristics
without checking that design; scans are the fallback layer.
### Gateway lifecycle vs. the Desktop app
`hermes serve` (control plane, desktop-spawned child) dies with the app
— by design. The messaging gateway (`gateway run`) SURVIVES the app: the
serve backend's `/api/gateway/*` endpoints spawn it detached
(`_spawn_hermes_action` — `start_new_session` / `DETACHED_PROCESS`), so
`before-quit`'s backend SIGTERM never reaches it. Bots keep running
when the user closes the app. The known breach of this contract is the
Windows shim-unlock teardown (`taskkill /T /F` on venv-shim holders,
#85265) — it exists to let updates proceed, and its replacement is
#92091's `pause-for-update`. Do not "fix" gateway-dies-with-app reports
by re-parenting the gateway under the backend, and do not "fix" update
locks by widening the tree-kill.
---
## Important Policies
### Prompt Caching Must Not Break
@@ -1151,9 +1363,10 @@ detects process completion and triggers a new agent turn. Control verbosity of b
messages with `display.background_process_notifications`
in config.yaml (or `HERMES_BACKGROUND_NOTIFICATIONS` env var):
- `all` — running-output updates + final message (default)
- `result` — only the final completion message
- `error` — only the final message when exit code != 0
- `concise` — one-line status message on completion; failures append a short output tail (default)
- `all` — running-output updates + final raw-output message
- `result` — only the final raw-output completion message
- `error` — only the final raw-output message when exit code != 0
- `off` — no watcher messages at all
---
@@ -1214,19 +1427,60 @@ automatically scope to the active profile.
This is intentional — it lets `hermes -p coder profile list` see all profiles regardless
of which one is active.
7. **Multiplex profile-scoped env reads MUST fail closed — never borrow from `os.environ`**
(`agent/secret_scope.py` contract; #72348, #86905). Under `gateway.multiplex_profiles`,
`os.environ` holds the **default profile's** values; a secondary profile's `.env` lives
only in its secret scope (installed per-turn by `_profile_runtime_scope`). Any
profile-level env config — credentials (`app_secret`, tokens) AND authorization
(`FEISHU_ALLOWED_USERS`, `{PLATFORM}_ALLOW_ALL_USERS`, `GATEWAY_ALLOW_ALL_USERS`,
`group_policy`, `allow_bots`, ...) — must be read scope-aware:
- Adapters: `_get_scoped_secret()` (canonical fail-closed copy in
`plugins/platforms/feishu/adapter.py`, #86905).
- Gateway authz: `_auth_env()` / `_platform_gate_env()` (`gateway/authz_mixin.py`).
Rules:
- Scope installed + multiplex active → a scoped miss returns the **default**.
NEVER fall through to `os.environ` — that leaks another profile's value and
silently breaks routing/admission (a leaked default allowlist skips the
allow-all check and rejects every secondary-profile sender, #86905).
- Unscoped default-profile path (`UnscopedSecretError`) and single-profile
deployments keep the `os.environ` read — there it IS the profile's own value.
- Authorization config is the sharpest edge: allowlist/allow-all leaks cause
silent rejections (or worse, fail-open) that only show up as missing replies.
- The `_get_scoped_secret` wrapper is copy-pasted across ~15 platform adapters —
when touching any of them, make sure the fail-closed semantics are present;
do not reintroduce the `except _UnscopedSecretError: val = os.getenv(...)`
fallback-after-miss shape.
## Known Pitfalls
### DO NOT infer process identity from argv substrings
The bug class behind ~10 fleet-update issues (#90778, #87594, #78089,
#76129, #91964, ...): classifying a process by `"serve" in cmdline` or
similar. `kanban --preserve-cache` contains "serve"; a flag VALUE can
equal a subcommand (`-m dashboard serve`); truncated cmdlines hide the
real subcommand. Rules:
- Use the canonical matchers: `gateway.status.looks_like_gateway_command_line`
(gateway run), `hermes_cli.update_cmd._hermes_holder_subcommand`
(top-level subcommand of any Hermes argv). Never hand-roll token scans.
- Flag sets must be DERIVED from the parser
(`_holder_value_flags()` introspects `build_top_level_parser()`), never
hand-written lists — they drift.
- Never blanket-exclude ancestors from process scans: when `/update` runs
as the gateway's child, a gateway ancestor must stay visible to the
pause machinery (#87594). Exclude interactive ancestry, carve out
gateway-shaped ancestors.
- Match on FULL cmdlines; truncate only at display time (#78089).
- Before adding any new scan heuristic, read #92091 — the gateway control
socket replaces scans as the primary coordination mechanism; scans are
the fallback layer for old/crashed processes.
### DO NOT hardcode `~/.hermes` paths
Use `get_hermes_home()` from `hermes_constants` for code paths. Use `display_hermes_home()`
for user-facing print/log messages. Hardcoding `~/.hermes` breaks profiles — each profile
has its own `HERMES_HOME` directory. This was the source of 5 bugs fixed in PR #3575.
### DO NOT introduce new `simple_term_menu` usage
Existing call sites in `hermes_cli/main.py` remain for legacy fallback only;
the preferred UI is curses (stdlib) because `simple_term_menu` has
ghost-duplication rendering bugs in tmux/iTerm2 with arrow keys. New
interactive menus must use `hermes_cli/curses_ui.py` — see
`hermes_cli/tools_config.py` for the canonical pattern.
### All CLI menu-pickers MUST use curses.
Interactive menus must use `hermes_cli/curses_ui.py`. See `hermes_cli/tools_config.py` for an example.
### DO NOT use `\033[K` (ANSI erase-to-EOL) in spinner/display code
Leaks as literal `?[K` text under `prompt_toolkit`'s `patch_stdout`. Use space-padding: `f"\r{line}{' ' * pad}"`.
@@ -1248,6 +1502,47 @@ while the agent is blocked (e.g. approval prompts) MUST bypass BOTH
guards and be dispatched inline, not via `_process_message_background()`
(which races session lifecycle).
### Streaming delivery contract (stream-is-the-message adapters) — duplicate-final class
Adapters with `draft_stream_is_message = True` (relay Slack native streaming)
keep ONE cumulative native stream per turn; the stream IS the final message.
Four invariants, each learned from a live duplicate-final incident (NS-658
canary ledger, hermes#85796 / gateway-gateway#210). Violating any of them
re-creates a duplicate or a frozen stream:
1. **Draft frames must be prefix-stable.** The connector computes append-only
deltas: frame N must be a string prefix of frame N+1. NEVER mutate draft
frames per-tick — no fence-closing (`ensure_closed_code_fences`), no cursor
suffix, no segment-state resets at tool boundaries, no mrkdwn conversion.
Any non-prefix frame triggers a whole-snapshot re-append on the platform
("stacked copies"). The finalize path may still transform the real final.
2. **The consumer declares the final; the adapter never guesses.**
`finish(final_text)` carries the completed `final_response` (verifier
footer, completion explainer included) as the authoritative finalize
payload. New post-stream response augmentation MUST ride this payload —
if it mutates `final_response` after the stream sealed, it re-opens the
#11 bug (`delivered_final_matches` mismatch → corrective duplicate send).
3. **Interim sends must carry `_interim_send` metadata.** Any consumer-side
`adapter.send()` that is NOT the turn-final (commentary, segment-tail
flushes) must set `metadata["_interim_send"] = True`, or the relay
adapter's seal-interception will seal the live stream with interim text.
Seal-interception exists at BOTH egress doors (`send()` AND
`send_for_platform()`); a new egress door needs the same two checks.
4. **Reconcile by edit, never by plain send.** Any lane that delivers a final
beside an already-sealed stream (queued follow-ups, media-accompanied
finals, future lanes) must first try `edit_message` on the consumer's
`message_id`; plain `send()` is the fallback only when no editable message
exists. A sealed native stream is a regular message — `chat.update` on it
works (live-verified).
Contract tests: `tests/gateway/test_stream_final_contract.py` (all four
invariants, mutation-checked). Slack streaming API ground truth (live-probed,
also encoded in connector comments/tests): `chat.*Stream` speaks STANDARD
markdown, not mrkdwn; `stopStream.markdown_text` APPENDS (never replaces);
`startStream`/`stopStream` are rate-limit Tier 2 (~20/min).
Guard style note: check `draft_stream_is_message` with `is True` — MagicMock
adapters in older tests auto-create truthy attributes.
### Squash merges from stale branches silently revert recent fixes
Before squash-merging a PR, ensure the branch is up to date with `main`
(`git fetch origin main && git reset --hard origin/main` in the worktree,
@@ -1284,14 +1579,15 @@ def profile_env(tmp_path, monkeypatch):
### Python
**ALWAYS use `scripts/run_tests.sh`** — do not call `pytest` directly. The script enforces
hermetic environment parity with CI (unset credential vars, TZ=UTC, LANG=C.UTF-8,
`-n auto` xdist workers, in-tree subprocess-isolation plugin). Direct `pytest`
per-file subprocess isolation via `scripts/run_tests_parallel.py` — no xdist,
worker count auto-scaled from CPU count). Direct `pytest`
on a 16+ core developer machine with API keys set diverges from CI in ways
that have caused multiple "works locally, fails in CI" incidents (and the reverse).
```bash
scripts/run_tests.sh # full suite, CI-parity
scripts/run_tests.sh tests/gateway/ # one directory
scripts/run_tests.sh tests/agent/test_foo.py::test_x # one test
scripts/run_tests.sh tests/agent/test_foo.py -k test_x # one test (file + -k; the runner is file-granular)
scripts/run_tests.sh -v --tb=long # pass-through pytest flags
```
@@ -1329,6 +1625,60 @@ Any test that reads or asserts about `package.json`,
`package-lock.json`, `tsconfig.json`, `.ts`/`.tsx`/`.js`/`.mjs`/`.cjs`
source files configuration belongs in the JS (vitest) test suite, not in `tests/*.py`.
### Don't fake the host OS
Hermes supports Linux, macOS and native Windows, and plenty of its behaviour
genuinely differs per host. Those differences are tested by running on the
host, not by patching `sys.platform`.
```python
@pytest.mark.linux_only
@pytest.mark.macos_only
@pytest.mark.windows_only
```
Things that are host-independent can stay unmarked:
- **Pure functions that take a platform as data** —
`hidden_windows_child_options(opts, is_windows=True)` is input→output, not a
fake host. (Contrast: setting a module-level `IS_WINDOWS` flag and then
calling `windows_detach_flags()` *is* a fake.)
- **Declaration/packaging invariants** — "pyproject declares `tzdata` with a
`sys_platform == 'win32'` marker" asserts about a file, not about runtime.
The line: **if the test needs the interpreter to believe it is on another OS
in order to pass, it belongs on that OS.**
When one test body walks several platforms in sequence, split it.
Keep the host-native arm on the Linux lane and move the other arm into its own marked test.
**Live Windows process-topology E2E: the `wine2e` lane.** For claims about
real Windows process behavior that mocks cannot reproduce (venv-holder
scans, process-tree parentage, launcher/worker chains, detach semantics),
there is an on-demand workflow `windows-venv-e2e.yml` that runs
`tests/hermes_cli/test_venv_holder_windows_live.py` on a real
`windows-latest` runner — spawning actual processes and driving the real
detection code, no mocked psutil. It fires ONLY on pushes to `wine2e/**`
branches (inert on PRs and main; costs nothing on normal work). The proven
workflow: write probes that pin CORRECT behavior, push to a `wine2e/`
branch to reproduce the bugs live on unfixed code, build the fix, iterate
until the lane is green, then open the PR — the live receipt on the exact
head is the Windows proof reviewers ask for. Extend the live suite when
touching that subsystem; assert against the gateway ANCESTOR found by
argv, not the direct parent (the venv shim makes every spawn a
launcher/worker chain).
**Use the marker, never a bare `skipif`.** `scripts/ci/list_os_marked_tests.py`
decides which files the macOS/Windows lanes import by grepping for the marker
*name*, and the lane then filters with `-m <marker>`. A test gated with
`@pytest.mark.skipif(sys.platform != "win32")` therefore skips on Linux AND is
never imported on the Windows lane — it runs on no host at all, silently. The
same trap catches a file-local alias (`windows_only = pytest.mark.skipif(...)`):
the grep matches the name, so the file *is* listed, but `-m windows_only`
deselects every test in it and the lane reports green over zero coverage.
Equally, don't `pytest.skip()` the non-host rows of a `@parametrize` over
platforms — split it into one marked test per OS, or only the host's row ever
executes.
### Don't write change-detector tests
A test is a **change-detector** if it fails whenever data that is **expected
+1 -1
View File
@@ -582,7 +582,7 @@ test(tools): añadir tests unitarios para file_operations
## Reportar Issues
- Usa [GitHub Issues](https://github.com/NousResearch/hermes-agent/issues)
- Incluye: SO, versión de Python, versión de Hermes (`hermes version`), traza de error completa
- Incluye: SO, versión de Python, versión de Hermes (`hermes --version`), traza de error completa
- Incluye pasos para reproducir
- Verifica los issues existentes antes de crear duplicados
- Para vulnerabilidades de seguridad, por favor reporta de forma privada
+26 -41
View File
@@ -130,7 +130,7 @@ cd "${HERMES_HOME:-$HOME/.hermes}/hermes-agent"
# Add dev/test extras on top of the standard install.
uv pip install -e ".[all,dev]"
# Optional: browser tools / docs site dependencies.
# Optional: docs site + workspace dependencies.
npm install
```
@@ -167,7 +167,7 @@ export PATH="$VIRTUAL_ENV/bin:$PATH"
# Install with all extras (messaging, cron, CLI menus, dev tools)
uv pip install -e ".[all,dev]"
# Optional: browser tools
# Optional: workspace / docs dependencies
npm install
```
@@ -201,7 +201,8 @@ ln -sf "$(pwd)/venv/bin/hermes" ~/.local/bin/hermes
### Run tests
```bash
# Preferred — matches CI (hermetic env, 4 xdist workers); see AGENTS.md
# Preferred — matches CI (hermetic `env -i`, per-file subprocess isolation
# via run_tests_parallel.py, worker count auto-scaled); see AGENTS.md
scripts/run_tests.sh
# Alternative (activate the venv first). The wrapper is still recommended
@@ -722,22 +723,9 @@ that touches the OS, assume *any* platform can hit your code path.
For process enumeration: PowerShell's `Get-CimInstance Win32_Process` is
the modern replacement for `wmic process`. See
`hermes_cli/gateway.py::_scan_gateway_pids` for the pattern.
3. **`termios` and `fcntl` are Unix-only.** Always catch both `ImportError`
and `NotImplementedError`:
```python
try:
from simple_term_menu import TerminalMenu
menu = TerminalMenu(options)
idx = menu.show()
except (ImportError, NotImplementedError):
# Fallback: numbered menu for Windows
for i, opt in enumerate(options):
print(f" {i+1}. {opt}")
idx = int(input("Choice: ")) - 1
```
4. **File encoding.** Windows may save `.env` files in `cp1252`. Always
3. **File encoding.** Windows may save `.env` files in `cp1252`. Always
handle encoding errors:
```python
try:
@@ -749,7 +737,7 @@ that touches the OS, assume *any* platform can hit your code path.
similar editors — use `encoding="utf-8-sig"` when reading files that
could have been touched by a Windows GUI editor.
5. **Process management.** `os.setsid()`, `os.killpg()`, `os.fork()`,
4. **Process management.** `os.setsid()`, `os.killpg()`, `os.fork()`,
`os.getuid()`, and POSIX signal handling differ on Windows. Guard with
`platform.system()`, `sys.platform`, or `hasattr(os, "setsid")`:
```python
@@ -773,29 +761,29 @@ that touches the OS, assume *any* platform can hit your code path.
pass
```
6. **Signals that don't exist on Windows: `SIGALRM`, `SIGCHLD`, `SIGHUP`,
5. **Signals that don't exist on Windows: `SIGALRM`, `SIGCHLD`, `SIGHUP`,
`SIGUSR1`, `SIGUSR2`, `SIGPIPE`, `SIGQUIT`, `SIGKILL`.** Python's
`signal` module raises `AttributeError` at import time if you reference
them on Windows. Use `getattr(signal, "SIGKILL", signal.SIGTERM)` or
gate the whole block behind a platform check. `loop.add_signal_handler`
raises `NotImplementedError` on Windows — always catch it.
7. **Path separators.** Use `pathlib.Path` instead of string concatenation
6. **Path separators.** Use `pathlib.Path` instead of string concatenation
with `/`. Forward slashes work almost everywhere on Windows, but
`subprocess.run(["cmd.exe", "/c", ...])` and other shell contexts can
require backslashes — convert with `str(path)` at the subprocess boundary,
not inside Python logic.
8. **Symlinks need elevated privileges on Windows** (unless Developer Mode is
7. **Symlinks need elevated privileges on Windows** (unless Developer Mode is
on). Tests that create symlinks need `@pytest.mark.skipif(sys.platform ==
"win32", reason="Symlinks require elevated privileges on Windows")`.
9. **POSIX file modes (0o600, 0o644, etc.) are NOT enforced on NTFS** by
8. **POSIX file modes (0o600, 0o644, etc.) are NOT enforced on NTFS** by
default. Tests that assert on `stat().st_mode & 0o777` must skip on
Windows — the concept doesn't translate. Use ACLs (`icacls`, `pywin32`)
for Windows secret-file protection if needed.
10. **Detached background daemons on Windows need `pythonw.exe`, NOT
9. **Detached background daemons on Windows need `pythonw.exe`, NOT
`python.exe`.** `python.exe` always allocates or attaches to a console,
which makes it vulnerable to `CTRL_C_EVENT` broadcasts from any sibling
process. `pythonw.exe` is the no-console variant. Combine with
@@ -804,38 +792,38 @@ that touches the OS, assume *any* platform can hit your code path.
See `hermes_cli/gateway_windows.py::_spawn_detached` for the reference
implementation.
11. **`subprocess.Popen` with `.cmd` or `.bat` shims needs `shutil.which`
10. **`subprocess.Popen` with `.cmd` or `.bat` shims needs `shutil.which`
to resolve.** Passing `"agent-browser"` to `Popen` on Windows finds
the extensionless POSIX shebang shim in `node_modules/.bin/`, which
`CreateProcessW` can't execute — you'll get `WinError 193 "not a valid
Win32 application"`. Use `shutil.which("agent-browser", path=local_bin)`
which honors PATHEXT and picks the `.CMD` variant on Windows.
12. **Don't use shell shebangs as a way to run Python.** `#!/usr/bin/env
11. **Don't use shell shebangs as a way to run Python.** `#!/usr/bin/env
python` only works when the file is executed through a Unix shell.
`subprocess.run(["./myscript.py"])` on Windows fails even if the file
has a shebang line. Always invoke Python explicitly:
`[sys.executable, "myscript.py"]`.
13. **Shell commands in installers.** If you change `scripts/install.sh`,
12. **Shell commands in installers.** If you change `scripts/install.sh`,
make the equivalent change in `scripts/install.ps1`. The two scripts
are the canonical example of "works on Linux does not mean works on
Windows" and have drifted multiple times — keep them in lockstep.
14. **Known paths that are OneDrive-redirected on Windows:** Desktop,
13. **Known paths that are OneDrive-redirected on Windows:** Desktop,
Documents, Pictures, Videos. The "real" path when OneDrive Backup is
enabled is `%USERPROFILE%\OneDrive\Desktop` (etc.), NOT
`%USERPROFILE%\Desktop` (which exists as an empty husk). Resolve the
real location via `ctypes` + `SHGetKnownFolderPath` or by reading the
`Shell Folders` registry key — never assume `~/Desktop`.
15. **CRLF vs LF in generated scripts.** Windows `cmd.exe` and `schtasks`
14. **CRLF vs LF in generated scripts.** Windows `cmd.exe` and `schtasks`
parse line-by-line; mixed or LF-only line endings can break multi-line
`.cmd` / `.bat` files. Use `open(path, "w", encoding="utf-8",
newline="\r\n")` — or `open(path, "wb")` + explicit bytes — when
generating scripts Windows will execute.
16. **Two different quoting schemes in one command line.** `subprocess.run
15. **Two different quoting schemes in one command line.** `subprocess.run
(["schtasks", "/TR", some_cmd])` → schtasks itself parses `/TR`, AND
the `some_cmd` string is re-parsed by `cmd.exe` when the task fires.
Different parsers, different escape rules. Use two separate quoting
@@ -845,18 +833,15 @@ that touches the OS, assume *any* platform can hit your code path.
### Testing cross-platform
Tests that use POSIX-only syscalls need a skip marker. Common ones:
- Symlinks → `@pytest.mark.skipif(sys.platform == "win32", ...)`
- `0o600` file modes → `@pytest.mark.skipif(sys.platform.startswith("win"), ...)`
- `signal.SIGALRM` → Unix-only (see `tests/conftest.py::_enforce_test_timeout`)
- `os.setsid` / `os.fork` → Unix-only
- Live Winsock / Windows-specific regression tests →
`@pytest.mark.skipif(sys.platform != "win32", reason="Windows-specific regression")`
Tests that excercise behavior on specific platforms must run on their target platforms.
If you monkeypatch `sys.platform` for cross-platform tests, also patch
`platform.system()` / `platform.release()` / `platform.mac_ver()` — each
re-reads the real OS independently, so half-patched tests still route
through the wrong branch on a Windows runner.
```python
@pytest.mark.linux_only
@pytest.mark.macos_only
@pytest.mark.windows_only
```
Avoid monkeypatching `sys.platform` unless absolutely needed, but if you do, also patch `platform.system()` / `platform.release()` / `platform.mac_ver()`.
Symlinks, 0o600 permissions, SIGALRM, os.setsid/fork are all unix-only.
---
@@ -988,7 +973,7 @@ test(tools): add unit tests for file_operations
## Reporting Issues
- Use [GitHub Issues](https://github.com/NousResearch/hermes-agent/issues)
- Include: OS, Python version, Hermes version (`hermes version`), full error traceback
- Include: OS, Python version, Hermes version (`hermes --version`), full error traceback
- Include steps to reproduce
- Check existing issues before creating duplicates
- For security vulnerabilities, please report privately
+140 -38
View File
@@ -1,12 +1,54 @@
# Debian 13 still ships SQLite 3.46.1, which contains the upstream WAL-reset
# corruption bug. Build a pinned shared library for the runtime image instead
# of relying on a distro backport that trixie does not currently provide.
# See #70480 and https://sqlite.org/wal.html#walresetbug.
FROM debian:13.4 AS sqlite_build
ARG SQLITE_AUTOCONF_VERSION=3530400
ARG SQLITE_SHA256=0e9483900e92cd5de8fd48d16bf9200145a61f7fd5be542a5ac81d8a9516eb9c
RUN apt-get -o Acquire::Retries=3 update && \
apt-get -o Acquire::Retries=3 install -y --no-install-recommends \
build-essential ca-certificates curl && \
rm -rf /var/lib/apt/lists/* && \
(curl -fsSL --retry 1 --retry-all-errors --connect-timeout 15 --max-time 60 \
-o /tmp/sqlite.tar.gz \
"https://sqlite.org/2026/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}.tar.gz" || \
curl -fsSL --retry 3 --retry-all-errors --connect-timeout 15 --max-time 120 \
-o /tmp/sqlite.tar.gz \
"https://sources.buildroot.net/sqlite/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}.tar.gz") && \
printf '%s %s\n' "${SQLITE_SHA256}" /tmp/sqlite.tar.gz > /tmp/sqlite.sha256 && \
sha256sum -c /tmp/sqlite.sha256 && \
tar -xzf /tmp/sqlite.tar.gz -C /tmp && \
cd "/tmp/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}" && \
CFLAGS="-O2 \
-DSQLITE_ENABLE_FTS3 \
-DSQLITE_ENABLE_FTS3_PARENTHESIS \
-DSQLITE_ENABLE_FTS4 \
-DSQLITE_ENABLE_FTS5 \
-DSQLITE_ENABLE_RTREE \
-DSQLITE_ENABLE_GEOPOLY \
-DSQLITE_ENABLE_COLUMN_METADATA \
-DSQLITE_ENABLE_UNLOCK_NOTIFY \
-DSQLITE_ENABLE_DBSTAT_VTAB \
-DSQLITE_ENABLE_DBPAGE_VTAB \
-DSQLITE_ENABLE_MATH_FUNCTIONS \
-DSQLITE_ENABLE_PREUPDATE_HOOK \
-DSQLITE_ENABLE_SESSION \
-DSQLITE_SECURE_DELETE \
-DSQLITE_THREADSAFE=1 \
-DSQLITE_MAX_VARIABLE_NUMBER=250000" \
./configure --prefix=/opt/sqlite-fixed --disable-static && \
make -j"$(nproc)" && \
make install
FROM ghcr.io/astral-sh/uv:0.11.6-python3.13-trixie@sha256:b3c543b6c4f23a5f2df22866bd7857e5d304b67a564f4feab6ac22044dde719b AS uv_source
# Node 22 LTS source stage. Debian trixie's bundled nodejs is pinned to 20.x
# which reached EOL in April 2026 — we copy node + npm + corepack from the
# upstream node:22 image instead so we can stay on a supported LTS without
# waiting for Debian 14 (forky, ~mid-2027). Bookworm-based slim image used
# so the produced binary links against glibc 2.36, which runs cleanly on
# our Debian 13 (trixie, glibc 2.41) runtime. Bumping to a new Node major
# is a one-line ARG change; see #4977.
FROM node:22-bookworm-slim@sha256:7af03b14a13c8cdd38e45058fd957bf00a72bbe17feac43b1c15a689c029c732 AS node_source
# Node 26 source stage. Debian trixie's bundled nodejs is pinned to 20.x
# which reached EOL in April 2026 — we copy node + npm from the upstream
# node:26 image instead (Hermes pins its toolchain to Node 26 everywhere).
# Bookworm-based slim image used so the produced binary links
# against glibc 2.36, which runs cleanly on our Debian 13 (trixie, glibc
# 2.41) runtime. Bumping to a new Node major is a one-line ARG change; see
# #4977.
FROM node:26-bookworm-slim@sha256:9e6f9357d371591e32ab6f2d8a26d63bdd0d17c29eee3f4f3e7e454d9634bf73 AS node_source
FROM debian:13.4
# Disable Python stdout buffering to ensure logs are printed immediately.
@@ -28,9 +70,26 @@ ENV PLAYWRIGHT_BROWSERS_PATH=/opt/hermes/.playwright
# hermes process, the dashboard, and per-profile gateways.
RUN apt-get -o Acquire::Retries=3 update && \
apt-get -o Acquire::Retries=3 install -y --no-install-recommends \
ca-certificates curl iputils-ping python3 python-is-python3 ripgrep ffmpeg gcc g++ make cmake python3-dev python3-venv libffi-dev libolm-dev procps git openssh-client docker-cli xz-utils && \
ca-certificates curl iputils-ping python3 python-is-python3 ripgrep ffmpeg gcc g++ make cmake python3-dev python3-venv libffi-dev libolm-dev libatomic1 procps git openssh-client docker-cli xz-utils && \
rm -rf /var/lib/apt/lists/*
# Prefer the fixed SQLite over Debian's vulnerable libsqlite3.so.0. Keep the
# public library name stable so both the system interpreter and the uv-created
# venv resolve the replacement without changing Python import paths.
COPY --from=sqlite_build /opt/sqlite-fixed/lib/libsqlite3.so.3.53.4 /usr/local/lib/
RUN ln -sf libsqlite3.so.3.53.4 /usr/local/lib/libsqlite3.so.0 && \
ln -sf libsqlite3.so.3.53.4 /usr/local/lib/libsqlite3.so && \
printf '/usr/local/lib\n' > /etc/ld.so.conf.d/000-sqlite-fixed.conf && \
ldconfig && \
python3 -c "import sqlite3, sys; \
v = sqlite3.sqlite_version_info; \
sys.exit(f'linked SQLite {sqlite3.sqlite_version} still has the WAL-reset bug') if v < (3, 51, 3) else None; \
db = sqlite3.connect(':memory:'); \
db.execute(\"CREATE VIRTUAL TABLE docs USING fts5(content, tokenize='trigram')\"); \
db.execute(\"INSERT INTO docs VALUES ('hermes')\"); \
sys.exit('SQLite FTS5 trigram self-test failed') if db.execute(\"SELECT count(*) FROM docs WHERE docs MATCH 'erm'\").fetchone()[0] != 1 else None; \
db.close()"
# ---------- s6-overlay install ----------
# s6-overlay provides supervision for the main hermes process, the dashboard,
# and per-profile gateways. /init becomes PID 1 below — see ENTRYPOINT.
@@ -92,17 +151,20 @@ RUN useradd -u 10000 -m -d /opt/data hermes
COPY --chmod=0755 --from=uv_source /usr/local/bin/uv /usr/local/bin/uvx /usr/local/bin/
# Node 22 LTS: copy the node binary plus the bundled npm + corepack JS
# installs from the upstream image. npm and npx are recreated as symlinks
# because they're symlinks in the source image (and need to live on PATH).
# Node 26: copy the node binary plus the bundled npm JS install from the
# upstream image. npm and npx are recreated as symlinks because they're
# symlinks in the source image (and need to live on PATH).
#
# No corepack: Node unbundled it upstream, so node:26 ships only npm in
# /usr/local/lib/node_modules. Nothing here needs it — no package.json
# declares a `packageManager`, and no build step shells out to yarn or pnpm.
#
# See node_source stage at the top of the file for the version-bump
# rationale (#4977).
COPY --chmod=0755 --from=node_source /usr/local/bin/node /usr/local/bin/
COPY --from=node_source /usr/local/lib/node_modules/npm /usr/local/lib/node_modules/npm
COPY --from=node_source /usr/local/lib/node_modules/corepack /usr/local/lib/node_modules/corepack
RUN ln -sf /usr/local/lib/node_modules/npm/bin/npm-cli.js /usr/local/bin/npm && \
ln -sf /usr/local/lib/node_modules/npm/bin/npx-cli.js /usr/local/bin/npx && \
ln -sf /usr/local/lib/node_modules/corepack/dist/corepack.js /usr/local/bin/corepack
ln -sf /usr/local/lib/node_modules/npm/bin/npx-cli.js /usr/local/bin/npx
WORKDIR /opt/hermes
@@ -141,6 +203,22 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
done && \
npm cache clean --force
# ---------- Photon iMessage sidecar deps (baked, NS-606) ----------
# The photon plugin's Node sidecar needs its own node_modules
# (spectrum-ts). The install tree is immutable at runtime, so a lazy
# `npm ci` on first connect would hit EROFS — bake the deps here instead
# (deterministic installs, NS-559). The patch script is copied alongside
# the manifests because package.json's postinstall runs it, which also
# means the spectrum-ts patch is applied at build time. Layer-cached:
# only re-runs when the sidecar manifests/patch change.
COPY plugins/platforms/photon/sidecar/package.json \
plugins/platforms/photon/sidecar/package-lock.json \
plugins/platforms/photon/sidecar/patch-spectrum-mixed-attachments.mjs \
plugins/platforms/photon/sidecar/
RUN cd plugins/platforms/photon/sidecar && \
npm ci --no-audit --fetch-retries=5 && \
npm cache clean --force
# ---------- Layer-cached Python dependency install ----------
# Copy only pyproject.toml + uv.lock so the Python dep resolve + wheel
# download + native-extension compile layer is cached unless those inputs
@@ -152,7 +230,7 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
# frontend stats the readme path during dep resolution, so we `touch` an
# empty placeholder — the real README is restored by `COPY . .` below.
#
# `uv sync --frozen --no-install-project --extra all --extra messaging`
# `uv sync --frozen --no-install-project --extra all --extra messaging --extra otlp`
# installs the deps reachable through the composite `[all]` extra
# (handpicked set intended for the production image — excludes `[dev]`),
# plus gateway messaging adapters that should work in the published image
@@ -165,6 +243,10 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
# so Docker users can use these providers without requiring runtime
# lazy-install access to PyPI (often blocked in containerized envs).
#
# The [otlp] extra contains the SDK/exporter imported by Hermes when Gateway
# Health export is enabled. Collector and observability-backend dependencies
# remain external and are not part of the Hermes production image.
#
# The hindsight memory provider's client (hindsight-client) is baked in
# for the same reason: it lazy-installs into /opt/hermes/.venv at first
# use, which lives inside the (immutable) image layer rather than the
@@ -182,7 +264,7 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
# The editable link is created after the source copy below.
COPY pyproject.toml uv.lock ./
RUN touch ./README.md
RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra anthropic --extra bedrock --extra azure-identity --extra hindsight --extra matrix
RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra otlp --extra anthropic --extra bedrock --extra azure-identity --extra hindsight --extra matrix
# ---------- Frontend build (cached independently from Python source) ----------
# Copy only the frontend source trees first so that Python-only changes don't
@@ -216,7 +298,7 @@ RUN uv pip install --no-cache-dir --no-deps -e "."
USER root
RUN mkdir -p /opt/hermes/bin && \
cp /opt/hermes/docker/hermes-exec-shim.sh /opt/hermes/bin/hermes && \
chmod 0755 /opt/hermes/bin/hermes && \
chmod 0755 /opt/hermes /opt/hermes/bin/hermes && \
printf 'docker\n' > /opt/hermes/.install_method
# The ``.install_method`` stamp is baked next to the running code (the install
# tree), NOT into $HERMES_HOME. $HERMES_HOME (/opt/data) is a shared data
@@ -229,7 +311,11 @@ RUN mkdir -p /opt/hermes/bin && \
# `s6-setuidgid hermes` in its run script. If HERMES_UID is unset, services
# run as the default hermes user (UID 10000).
# ---------- Bake build-time git revision ----------
# ---------- Bake image provenance + build-time git revision ----------
# The versioned, non-secret provenance marker is the authoritative runtime
# signal that this filesystem came from an immutable image. It deliberately
# lives outside both /opt/hermes (which operators sometimes bind-mount as a
# checkout) and /opt/data (the mutable HERMES_HOME volume).
# .dockerignore excludes .git, so `git rev-parse HEAD` from inside the
# container always returns nothing — meaning `hermes dump` reports
# "(unknown)" and the startup banner drops its `· upstream <sha>` suffix.
@@ -242,14 +328,18 @@ RUN mkdir -p /opt/hermes/bin && \
# banner.get_git_banner_state() try the baked SHA first, then fall back
# to live `git rev-parse` for source installs (unchanged behaviour).
#
# The arg is optional — local `docker build` without --build-arg simply
# omits the file, and the runtime falls back to live-git lookup. CI
# The arg is optional — local `docker build` without --build-arg omits the
# SHA file (and records a null provenance revision), so build-info falls back
# to live-git lookup. CI
# (.github/workflows/docker.yml) passes ${{ github.sha }} so
# every published image has it.
ARG HERMES_GIT_SHA=
RUN if [ -n "${HERMES_GIT_SHA}" ]; then \
RUN set -eu; \
if [ -n "${HERMES_GIT_SHA}" ]; then \
printf '%s\n' "${HERMES_GIT_SHA}" > /opt/hermes/.hermes_build_sha; \
fi
fi; \
mkdir -p /etc/hermes; \
HERMES_GIT_SHA="${HERMES_GIT_SHA}" python3 -c 'import json, os, pathlib, tomllib; project = tomllib.loads(pathlib.Path("/opt/hermes/pyproject.toml").read_text(encoding="utf-8"))["project"]; marker = pathlib.Path("/etc/hermes/image-provenance.json"); marker.write_text(json.dumps({"schema": 1, "deployment_kind": "image", "manager": "docker", "image": "nousresearch/hermes-agent", "version": project["version"], "revision": os.environ.get("HERMES_GIT_SHA") or None}, sort_keys=True, separators=(",", ":")) + "\n", encoding="utf-8"); marker.chmod(0o444)'
# ---------- s6-overlay service wiring ----------
# Static services declared at build time: main-hermes + dashboard.
@@ -321,6 +411,8 @@ ENV HERMES_LAZY_INSTALL_TARGET=/opt/data/lazy-packages
# Recursion is impossible because the shim exec's the venv binary by
# absolute path (/opt/hermes/.venv/bin/hermes). See the shim source for
# the opt-out env var (HERMES_DOCKER_EXEC_AS_ROOT=1).
COPY --chmod=0755 docker/hermes-exec-shim.sh /opt/hermes/bin/hermes
COPY --chmod=0755 docker/entrypoint-dispatch.sh /opt/hermes/docker/entrypoint-dispatch.sh
# Pre-s6 entrypoint.sh did `source .venv/bin/activate` which exported
# the venv bin onto PATH; Architecture B's main-wrapper.sh does the
@@ -337,27 +429,37 @@ ENV PATH="/opt/hermes/bin:/opt/hermes/.venv/bin:/opt/data/.local/bin:${PATH}"
RUN mkdir -p /opt/data
VOLUME [ "/opt/data" ]
# s6-overlay's /init is PID 1. It sets up the supervision tree, runs
# /etc/cont-init.d/* (our stage2 hook), starts s6-rc services
# declared in /etc/s6-overlay/s6-rc.d/, then exec's its remaining
# argv as the container's "main program" with stdin/stdout/stderr
# inherited (this is what makes interactive --tui work). When the
# main program exits, /init begins stage 3 shutdown and the container
# exits with the program's exit code. Replaces tini — see Phase 2 of
# docs/plans/2026-05-07-s6-overlay-dynamic-subagent-gateways.md.
# The image ENTRYPOINT is a tiny dispatcher rather than `/init` directly.
# When the image really owns PID 1 (normal Docker / Podman), the dispatcher
# execs `/init` and preserves the full s6 supervision tree. When a platform
# wraps the image entrypoint under its own PID-1 init (Fly Machines,
# `docker run --init`, some schedulers), `/init` would abort with
# `can only run as pid 1`; in that case the dispatcher falls back to
# `stage2-hook.sh` + `main-wrapper.sh` directly so foreground commands still
# work. See #38349.
#
# On the PID-1 path, s6-overlay's /init sets up the supervision tree, runs
# /etc/cont-init.d/* (our stage2 hook), starts s6-rc services declared in
# /etc/s6-overlay/s6-rc.d/, then exec's its remaining argv as the container's
# "main program" with stdin/stdout/stderr inherited (this is what makes
# interactive --tui work). When the main program exits, /init begins stage 3
# shutdown and the container exits with the program's exit code. Replaces
# tini — see Phase 2 of docs/plans/2026-05-07-s6-overlay-dynamic-subagent-gateways.md.
#
# We use the ENTRYPOINT+CMD split rather than CMD alone so the
# wrapper is prepended to user-supplied args automatically:
#
# docker run <image> → /init main-wrapper.sh (CMD default)
# docker run <image> chat -q "hi" → /init main-wrapper.sh chat -q hi
# docker run <image> sleep infinity → /init main-wrapper.sh sleep infinity
# docker run <image> --tui → /init main-wrapper.sh --tui
# docker run <image> → entrypoint-dispatch.sh (CMD default)
# docker run <image> chat -q "hi" → entrypoint-dispatch.sh chat -q hi
# docker run <image> sleep infinity → entrypoint-dispatch.sh sleep infinity
# docker run <image> --tui → entrypoint-dispatch.sh --tui
#
# main-wrapper.sh handles arg routing (bare-exec vs. hermes
# subcommand vs. no-args), drops to the hermes user via s6-setuidgid,
# and exec's the final program so its exit code becomes the container
# exit code. Without the wrapper-as-ENTRYPOINT, leading-dash args
# like `--version` would be intercepted by /init's POSIX shell.
ENTRYPOINT [ "/init", "/opt/hermes/docker/main-wrapper.sh" ]
# exit code. The dispatcher preserves that contract across both the
# supervised PID-1 path and the non-PID-1 fallback path. Without the
# wrapper-as-ENTRYPOINT, leading-dash args like `--version` would be
# intercepted by /init's POSIX shell.
ENTRYPOINT [ "/opt/hermes/docker/entrypoint-dispatch.sh" ]
CMD [ ]
-14
View File
@@ -1,14 +0,0 @@
graft skills
graft optional-skills
graft optional-mcps
graft hermes_cli/web_dist
graft locales
# Bundled plugin manifests (plugin.yaml / plugin.yml). Without these the
# PluginManager scan (hermes_cli/plugins.py) finds zero plugins on installs
# built from the sdist (e.g. Homebrew, downstream packagers). package-data
# below covers the wheel; this covers the sdist. See #34034 / #28149.
recursive-include plugins plugin.yaml plugin.yml
# Gateway assets include images plus YAML catalogs such as status_phrases.yaml.
recursive-include gateway/assets *
global-exclude __pycache__
global-exclude *.py[cod]
+1 -1
View File
@@ -26,7 +26,7 @@ Use any model you want — [Nous Portal](https://portal.nousresearch.com), OpenR
<tr><td><b>A closed learning loop</b></td><td>Agent-curated memory with periodic nudges. Autonomous skill creation after complex tasks. Skills self-improve during use. FTS5 session search with LLM summarization for cross-session recall. <a href="https://github.com/plastic-labs/honcho">Honcho</a> dialectic user modeling. Compatible with the <a href="https://agentskills.io">agentskills.io</a> open standard.</td></tr>
<tr><td><b>Scheduled automations</b></td><td>Built-in cron scheduler with delivery to any platform. Daily reports, nightly backups, weekly audits — all in natural language, running unattended.</td></tr>
<tr><td><b>Delegates and parallelizes</b></td><td>Spawn isolated subagents for parallel workstreams. Write Python scripts that call tools via RPC, collapsing multi-step pipelines into zero-context-cost turns.</td></tr>
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Six terminal backends — local, Docker, SSH, Singularity, Modal, and Daytona. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Seven terminal backends — local, Docker, SSH, Singularity, Modal, Daytona, and Vercel Sandbox. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
<tr><td><b>Research-ready</b></td><td>Batch trajectory generation, trajectory compression for training the next generation of tool-calling models.</td></tr>
</table>
+8 -4
View File
@@ -16,7 +16,7 @@ Un informe útil incluye:
- Una descripción concisa y evaluación de severidad.
- El componente afectado, identificado por ruta de archivo y rango de líneas
(ej. `path/to/file.py:120-145`).
- Detalles del entorno (`hermes version`, SHA del commit, SO, versión de Python).
- Detalles del entorno (`hermes --version`, SHA del commit, SO, versión de Python).
- Una reproducción contra `main` o el último release.
- Una declaración de qué límite de confianza del §2 se cruza.
@@ -173,9 +173,13 @@ modelo de autorización, pero las reglas a continuación se aplican uniformement
**Superficies en Hermes Agent:**
- **Adaptadores de plataforma del gateway.** Integraciones de mensajería en
`gateway/platforms/` (Telegram, Discord, Slack, email, SMS, etc.)
y adaptadores análogos incluidos como plugins.
- **Adaptadores de plataforma del gateway.** La mayoría de las integraciones
de mensajería se distribuyen como plugins empaquetados en
`plugins/platforms/<name>/` (Telegram, Discord, Slack, email, SMS, etc.).
Los tipos base compartidos y un conjunto menor de adaptadores
legacy/directos viven en `gateway/platforms/` (`base.py`, Signal, servidor
API, webhooks, …), con descubrimiento y carga diferida vía
`gateway/platform_registry.py`.
- **Superficies HTTP expuestas en red.** El adaptador del servidor API, el
plugin del dashboard, los endpoints HTTP del plugin kanban, y cualquier
otro plugin que vincule un socket de escucha.
+7 -4
View File
@@ -16,7 +16,7 @@ A useful report includes:
- A concise description and severity assessment.
- The affected component, identified by file path and line range
(e.g. `path/to/file.py:120-145`).
- Environment details (`hermes version`, commit SHA, OS, Python
- Environment details (`hermes --version`, commit SHA, OS, Python
version).
- A reproduction against `main` or the latest release.
- A statement of which trust boundary in §2 is crossed.
@@ -177,9 +177,12 @@ authorization model, but the rules below apply uniformly.
**Surfaces in Hermes Agent:**
- **Gateway platform adapters.** Messaging integrations in
`gateway/platforms/` (Telegram, Discord, Slack, email, SMS, etc.)
and analogous adapters shipped as plugins.
- **Gateway platform adapters.** Most messaging integrations ship as
bundled plugins under `plugins/platforms/<name>/` (Telegram, Discord,
Slack, email, SMS, etc.). Shared base types and a smaller set of
legacy/direct adapters live under `gateway/platforms/`
(`base.py`, Signal, API server, webhooks, …), with discovery and
deferred loading via `gateway/platform_registry.py`.
- **Network-exposed HTTP surfaces.** The API server adapter, the
dashboard plugin, the kanban plugin's HTTP endpoints, and any
other plugin that binds a listening socket.
+23 -12
View File
@@ -32,6 +32,7 @@ else:
import argparse
import asyncio
import logging
import os
import sys
from pathlib import Path
from hermes_constants import get_hermes_home
@@ -79,9 +80,11 @@ class _BenignProbeMethodFilter(logging.Filter):
def _setup_logging() -> None:
"""Route all logging to stderr so stdout stays clean for ACP stdio."""
from agent.redact import RedactingFormatter
handler = logging.StreamHandler(sys.stderr)
handler.setFormatter(
logging.Formatter(
RedactingFormatter(
"%(asctime)s [%(levelname)s] %(name)s: %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
)
@@ -190,7 +193,7 @@ def _run_setup_browser(assume_yes: bool = False) -> int:
"""Bootstrap agent-browser + Chromium.
Routes through dep_ensure -> install.{sh,ps1} --ensure, sharing code
with ``hermes postinstall`` and the runtime lazy installer.
with the runtime lazy installer.
Returns 0 on success, 1 on failure.
"""
@@ -246,16 +249,24 @@ def main(argv: list[str] | None = None) -> None:
import acp
from .server import HermesACPAgent
# MCP tool discovery from config.yaml — run before asyncio.run() so
# it's safe to use blocking waits. (ACP also registers per-session
# MCP servers dynamically via asyncio.to_thread inside the event
# loop; that path is unaffected.) Moved from model_tools.py module
# scope to avoid freezing the gateway's loop on lazy import (#16856).
try:
from tools.mcp_tool import discover_mcp_tools
discover_mcp_tools()
except Exception:
logger.debug("MCP tool discovery failed at ACP startup", exc_info=True)
# MCP tool discovery from config.yaml — fire-and-forget in a
# background daemon thread so the ACP server becomes responsive
# immediately while MCP servers connect. Previously this blocked
# asyncio.run() for 2-5 s. (ACP also registers per-session MCP
# servers dynamically via asyncio.to_thread inside the event loop;
# that path is unaffected.) Moved from model_tools.py module scope
# to avoid freezing the gateway's loop on lazy import (#16856).
# Metadata-only hosts can opt out of unrelated global MCP startup.
if os.environ.get("HERMES_ACP_SKIP_CONFIGURED_MCP", "").strip() != "1":
try:
from hermes_cli.mcp_startup import start_background_mcp_discovery
start_background_mcp_discovery(
logger=logger,
thread_name="acp-mcp-discovery",
)
except Exception:
logger.debug("MCP tool discovery failed at ACP startup", exc_info=True)
agent = HermesACPAgent()
try:
+21 -6
View File
@@ -39,13 +39,19 @@ def _permission_option_supports_kind(kind: str) -> bool:
def _build_permission_options(
*, allow_permanent: bool, smart_denied: bool = False,
*, allow_permanent: bool, allow_session: bool = True,
smart_denied: bool = False,
) -> list[PermissionOption]:
"""Return ACP options that match Hermes approval semantics."""
# A gate that re-asks every time (allow_session=False, e.g. protected
# agent-instruction writes) collapses to the same two options as a
# Smart DENY override — the editor must not offer a scope Hermes
# discards, or every subsequent write re-prompts (#81887).
once_only = smart_denied or not allow_session
options = [PermissionOption(
option_id="allow_once", kind="allow_once", name="Allow once",
)]
if not smart_denied:
if not once_only:
options.append(PermissionOption(
option_id="allow_session",
# ACP has no session-scoped kind, so use the closest persistent
@@ -53,7 +59,7 @@ def _build_permission_options(
kind="allow_always",
name="Allow for session",
))
if allow_permanent and not smart_denied:
if allow_permanent and not once_only:
options.append(
PermissionOption(
option_id="allow_always",
@@ -62,7 +68,7 @@ def _build_permission_options(
),
)
options.append(PermissionOption(option_id="deny", kind="reject_once", name="Deny"))
if not smart_denied and _permission_option_supports_kind("reject_always"):
if not once_only and _permission_option_supports_kind("reject_always"):
options.append(
PermissionOption(
option_id="deny_always",
@@ -132,6 +138,7 @@ def make_approval_callback(
description: str,
*,
allow_permanent: bool = True,
allow_session: bool = True,
smart_denied: bool = False,
**_: object,
) -> str:
@@ -139,6 +146,7 @@ def make_approval_callback(
options = _build_permission_options(
allow_permanent=allow_permanent,
allow_session=allow_session,
smart_denied=smart_denied,
)
@@ -158,9 +166,16 @@ def make_approval_callback(
try:
response = future.result(timeout=timeout)
except (FutureTimeout, Exception) as exc:
except FutureTimeout:
future.cancel()
logger.warning("Permission request timed out or failed: %s", exc)
logger.warning("Permission request timed out after %ss", timeout)
# Distinct from an explicit deny: the client never answered.
# tools.approval callers report this as "timed out without user
# response" instead of a user denial.
return "timeout"
except Exception as exc:
future.cancel()
logger.warning("Permission request failed: %s", exc)
return "deny"
if response is None:
+639 -91
View File
@@ -74,6 +74,11 @@ from acp_adapter.permissions import make_approval_callback
from acp_adapter.provenance import session_provenance_meta
from acp_adapter.session import SessionManager, SessionState, _expand_acp_enabled_toolsets
from acp_adapter.tools import build_tool_complete, build_tool_start
from agent.context_compressor import (
COMPRESSED_SUMMARY_METADATA_KEY,
ContextCompressor,
)
from agent.interrupt_compat import request_hard_interrupt
from tools.approval import (
reset_hermes_interactive_context,
set_hermes_interactive_context,
@@ -81,6 +86,142 @@ from tools.approval import (
logger = logging.getLogger(__name__)
def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, str]]]]:
"""Return ``(slug, label, [(model_id, description), ...])`` for named endpoints.
Covers both the v12 ``providers:`` mapping and the legacy
``custom_providers:`` list. These endpoints never appear in canonical
provider enumeration, so without this the ACP model selector hides every
named endpoint that the TUI ``/model`` picker already renders (#47039
implemented named-endpoint rows for the TUI surface only).
Model lists come from the entry's declared models (``default_model`` +
``models``), refreshed from the endpoint's live ``/models`` listing when a
credential is available and ``discover_models`` is not disabled. Declared
models are kept even when live discovery fails — some OpenAI-compatible
endpoints (e.g. Bedrock Mantle Responses) expose no ``/models`` route at
all yet serve the declared models fine.
Slugs use the ``custom:<name>`` shape that ``parse_model_input`` and
``resolve_runtime_provider`` already resolve, so encoded choice ids
(``custom:<name>:<model>``) round-trip through ``set_session_model``
unchanged.
"""
try:
from hermes_cli.config import (
get_compatible_custom_providers,
is_provider_enabled,
load_config,
)
from hermes_cli.model_switch import (
_NativePickerModelList,
_declared_model_ids,
_entry_models_discovered,
_fetch_picker_live_models,
_models_config_is_allowlist,
)
from hermes_cli.models import should_use_ollama_native_catalog
from hermes_cli.providers import custom_provider_slug
except ImportError:
return []
try:
cfg = load_config()
entries = get_compatible_custom_providers(cfg)
except Exception:
logger.debug("Could not load named custom providers", exc_info=True)
return []
# ``get_compatible_custom_providers`` drops the ``enabled`` flag during
# normalization, so collect explicitly disabled provider keys from the
# raw config and skip their entries below.
disabled_keys: set[str] = set()
raw_providers = cfg.get("providers") if isinstance(cfg, dict) else None
if isinstance(raw_providers, dict):
for raw_key, raw_entry in raw_providers.items():
if isinstance(raw_entry, dict) and not is_provider_enabled(raw_entry):
disabled_keys.add(str(raw_key).strip().lower())
catalogs: list[tuple[str, str, list[tuple[str, str]]]] = []
for entry in entries:
if not isinstance(entry, dict):
continue
provider_key = str(entry.get("provider_key", "") or "").strip()
if provider_key.lower() in disabled_keys:
continue
name = str(entry.get("name", "") or "").strip()
base_url = str(entry.get("base_url", "") or "").strip()
if not name or not base_url:
continue
slug = custom_provider_slug(name, provider_key)
api_key = str(entry.get("api_key", "") or "").strip()
if not api_key:
key_env = str(
entry.get("key_env") or entry.get("api_key_env") or ""
).strip()
api_key = os.environ.get(key_env, "").strip() if key_env else ""
declared: list[str] = []
default_model = str(entry.get("model", "") or "").strip()
if default_model:
declared.append(default_model)
models_cfg = entry.get("models")
for mid in _declared_model_ids(models_cfg):
if mid not in declared:
declared.append(mid)
native_headers = entry.get("extra_headers") or None
native_catalog_provider = (
provider_key
if provider_key.lower() in {"ollama", "custom:ollama"}
else "custom"
)
is_native_ollama = should_use_ollama_native_catalog(
native_catalog_provider, base_url, headers=native_headers
)
explicit_catalog = _models_config_is_allowlist(
models_cfg, _entry_models_discovered(entry)
)
if not api_key and not declared and not is_native_ollama:
# No credential to discover with and nothing declared:
# not addressable from the selector.
continue
model_ids = list(declared)
discover = entry.get("discover_models", True)
if isinstance(discover, str):
discover = discover.lower() not in {"false", "no", "0"}
native_catalog_provider = native_catalog_provider if is_native_ollama else "custom"
live = None
if discover and (api_key or is_native_ollama):
try:
live = _fetch_picker_live_models(
api_key,
base_url,
native_catalog_provider,
explicit_catalog,
headers=native_headers,
timeout=1.5,
api_mode=entry.get("api_mode"),
)
except Exception:
live = None
if live is not None:
if isinstance(live, _NativePickerModelList):
model_ids = list(live)
else:
model_ids = declared + [m for m in live if m not in declared]
if not model_ids:
if isinstance(live if "live" in locals() else None, _NativePickerModelList):
catalogs.append((slug, name, []))
continue
catalogs.append((slug, name, [(mid, "") for mid in model_ids]))
return catalogs
try:
from hermes_cli import __version__ as HERMES_VERSION
except Exception:
@@ -93,6 +234,13 @@ _executor = ThreadPoolExecutor(max_workers=4, thread_name_prefix="acp-agent")
# does not expose a client-side limit, so this is a fixed cap that clients
# paginate against using `cursor` / `next_cursor`.
_LIST_SESSIONS_PAGE_SIZE = 50
# Per-provider cap for the ACP model selector. ACP clients (Zed, Buzz) render
# the whole `availableModels` array in one dropdown, so an unbounded
# cross-provider catalog degrades the picker. Mirrors the cap the MoA picker
# already uses (`hermes_cli/moa_cmd.py`). This bounds each provider's row, not
# the total; aggregator providers stay intentionally uncapped inside the shared
# inventory, and the current model is always kept via the fallback insert below.
ACP_MAX_MODELS_PER_PROVIDER = 200
_MAX_ACP_RESOURCE_BYTES = 512 * 1024
_TEXT_RESOURCE_MIME_PREFIXES = ("text/",)
_TEXT_RESOURCE_MIME_TYPES = {
@@ -581,54 +729,240 @@ class HermesACPAgent(acp.Agent):
return f"{raw_provider}:{raw_model}"
def _build_model_state(self, state: SessionState) -> SessionModelState | None:
"""Return the ACP model selector payload for editors like Zed."""
"""Return authenticated providers and their models for ACP clients.
The shared Hermes inventory is also used by ``hermes model``, the TUI,
and the dashboard. Keeping ACP on that substrate prevents its selector
from silently collapsing to the current provider's curated list.
"""
model = str(state.model or getattr(state.agent, "model", "") or "").strip()
provider = getattr(state.agent, "provider", None) or detect_provider() or "openrouter"
try:
from hermes_cli.models import curated_models_for_provider, normalize_provider, provider_label
from hermes_cli.inventory import build_models_payload, load_picker_context
from hermes_cli.models import normalize_provider, provider_label
normalized_provider = normalize_provider(provider)
provider_name = provider_label(normalized_provider)
context = load_picker_context().with_overrides(
current_provider=normalized_provider,
current_model=model,
current_base_url=str(getattr(state.agent, "base_url", "") or ""),
)
payload = build_models_payload(
context,
explicit_only=True,
include_unconfigured=False,
picker_hints=False,
canonical_order=True,
pricing=False,
capabilities=False,
refresh=False,
probe_custom_providers=False,
probe_current_custom_provider=False,
max_models=ACP_MAX_MODELS_PER_PROVIDER,
)
available_models: list[ModelInfo] = []
seen_ids: set[str] = set()
current_choice_provider = str(provider or "").strip().lower()
if current_choice_provider == "ollama":
current_choice_provider = "custom:ollama"
current_base_url = str(
getattr(state.agent, "base_url", "") or ""
).strip().rstrip("/").lower()
for model_id, description in curated_models_for_provider(normalized_provider):
rendered_model = str(model_id or "").strip()
if not rendered_model:
def semantic_provider(provider_id: str) -> str:
raw = str(provider_id or "").strip().lower()
if raw in {"ollama", "custom:ollama"}:
return "ollama"
if raw.startswith("custom:"):
return raw
return normalize_provider(raw)
seen_semantic_ids: set[str] = set()
native_empty_rows: set[str] = set()
current_identity_resolved = current_choice_provider not in {"", "custom"}
for row in payload.get("providers") or []:
raw_row_provider = str(row.get("slug") or "").strip().lower()
row_provider = normalize_provider(raw_row_provider)
row_base_url = str(row.get("api_url") or "").strip().rstrip("/").lower()
if row.get("native_catalog_empty"):
native_empty_rows.add(raw_row_provider)
if (
not current_identity_resolved
and raw_row_provider in {"ollama", "custom:ollama"}
and current_base_url
and row_base_url == current_base_url
):
current_choice_provider = "custom:ollama"
current_identity_resolved = True
if not row_provider:
continue
choice_id = self._encode_model_choice(normalized_provider, rendered_model)
if choice_id in seen_ids:
continue
desc_parts = [f"Provider: {provider_name}"]
if description:
desc_parts.append(str(description).strip())
if rendered_model == model:
desc_parts.append("current")
available_models.append(
ModelInfo(
model_id=choice_id,
name=rendered_model,
description=" • ".join(part for part in desc_parts if part),
)
provider_name = str(row.get("name") or "").strip() or provider_label(
row_provider
)
seen_ids.add(choice_id)
row_models = row.get("models")
if not isinstance(row_models, (list, tuple)):
continue
for model_entry in row_models:
if isinstance(model_entry, dict):
rendered_model = str(
model_entry.get("id")
or model_entry.get("model")
or model_entry.get("name")
or ""
).strip()
else:
rendered_model = str(model_entry or "").strip()
if not rendered_model:
continue
encoded_provider = (
"custom:ollama"
if raw_row_provider == "ollama"
else raw_row_provider
if raw_row_provider == "custom:ollama"
else raw_row_provider
if raw_row_provider.startswith("custom:")
else row_provider
)
choice_id = self._encode_model_choice(
encoded_provider, rendered_model
)
semantic_id = f"{semantic_provider(encoded_provider)}:{rendered_model}"
if choice_id in seen_ids or semantic_id in seen_semantic_ids:
continue
is_current = (
semantic_provider(encoded_provider)
== semantic_provider(current_choice_provider)
and rendered_model == model
)
description = f"Provider: {provider_name}"
if is_current:
description += " • current"
available_models.append(
ModelInfo(
model_id=choice_id,
name=f"{provider_name} · {rendered_model}",
description=description,
)
)
seen_ids.add(choice_id)
seen_semantic_ids.add(semantic_id)
current_model_id = self._encode_model_choice(normalized_provider, model)
if current_model_id and current_model_id not in seen_ids:
# Named user-defined endpoints (providers: / custom_providers:)
# are invisible to canonical provider enumeration — append them
# so editor clients can select them like the TUI /model picker.
named_empty_authoritative: set[str] = set(native_empty_rows)
for named_slug, named_label, named_catalog in _named_custom_provider_catalogs():
if not named_catalog:
named_empty_authoritative.add(str(named_slug).strip().lower())
continue
for named_model, named_desc in named_catalog:
named_choice = self._encode_model_choice(named_slug, named_model)
named_semantic_id = (
f"{semantic_provider(named_slug)}:{named_model}"
)
if (
not named_choice
or named_choice in seen_ids
or named_semantic_id in seen_semantic_ids
):
continue
named_parts = [f"Provider: {named_label}"]
if named_desc:
named_parts.append(str(named_desc).strip())
if named_slug == normalized_provider and named_model == model:
named_parts.append("current")
available_models.append(
ModelInfo(
model_id=named_choice,
name=named_model,
description=" • ".join(part for part in named_parts if part),
)
)
seen_ids.add(named_choice)
seen_semantic_ids.add(named_semantic_id)
def empty_catalog_applies(provider_id: str) -> bool:
raw = str(provider_id or "").strip().lower()
normalized = normalize_provider(raw)
if normalized == "custom":
return any(
candidate == raw
or f"custom:{candidate}" == raw
or (raw == "custom" and candidate == "custom")
for candidate in named_empty_authoritative
)
return any(
candidate == raw
or candidate == f"custom:{normalized}"
or candidate == f"custom:{raw}"
or normalize_provider(candidate) == normalized
for candidate in named_empty_authoritative
)
def choice_provider(model_id: str) -> str:
parts = model_id.split(":")
if parts[:1] == ["custom"] and len(parts) > 1:
from hermes_cli.models import _configured_custom_provider_ids
lowered = model_id.lower()
for candidate in sorted(
(
provider_id
for provider_id in _configured_custom_provider_ids()
if provider_id.startswith("custom:")
),
key=len,
reverse=True,
):
if lowered.startswith(candidate + ":"):
return candidate
return "custom"
return parts[0]
if named_empty_authoritative:
available_models = [
item
for item in available_models
if not empty_catalog_applies(choice_provider(item.model_id))
]
seen_ids = {item.model_id for item in available_models}
current_is_empty = empty_catalog_applies(current_choice_provider)
if current_is_empty:
available_models = [
item
for item in available_models
if " • current" not in str(item.description or "")
]
seen_ids = {item.model_id for item in available_models}
current_model_id = (
"" if current_is_empty else self._encode_model_choice(current_choice_provider, model)
)
if (
current_model_id
and current_model_id not in seen_ids
and not current_is_empty
):
provider_name = provider_label(normalized_provider)
available_models.insert(
0,
ModelInfo(
model_id=current_model_id,
name=model,
name=f"{provider_name} · {model}",
description=f"Provider: {provider_name} • current",
),
)
if not available_models and current_is_empty:
return SessionModelState(available_models=[], current_model_id="")
if available_models:
return SessionModelState(
available_models=available_models,
current_model_id=current_model_id or available_models[0].model_id,
current_model_id=current_model_id
if current_model_id or current_is_empty
else available_models[0].model_id,
)
except Exception:
logger.debug("Could not build ACP model state", exc_info=True)
@@ -860,6 +1194,102 @@ class HermesACPAgent(acp.Agent):
exc_info=True,
)
def _schedule_mcp_late_refresh(self, state: SessionState) -> None:
"""Refresh the agent's tool snapshot when background MCP discovery lands late.
ACP entry.py starts MCP tool discovery in a background daemon thread so a
slow/dead configured server can't block ``asyncio.run()``. ``_make_agent``
briefly joins that thread (``wait_for_mcp_discovery``, bounded ~1.5s) so
already-spawning fast servers land in the snapshot — but a server slower
than the bound lands *after* the agent is built, leaving its tools absent
for the whole session.
This schedules an off-critical-path daemon that waits for discovery to
finish (bounded 30s), then rebuilds the snapshot via the shared
``refresh_agent_mcp_tools`` helper — the same rebuild ``/reload-mcp``
performs, but automatic. Mirrors the TUI late-refresh (PR #48403).
Cache safety: the rebuild only runs while the session is still
pre-first-turn (no API call made yet → nothing cached to invalidate).
Once the user has sent a message we leave the snapshot frozen rather
than break the cached prompt prefix mid-conversation; servers that land
later are picked up cache-safely by the between-turns prologue refresh
(``agent/turn_context.py``) at the next turn boundary. The marginal
value of this pre-first-turn daemon is therefore freshness in the
window [session created → first message] — e.g. the "Available tools"
listing a client may request before the first prompt.
No-op when discovery already finished, when the join times out, when the
registry was unchanged, or when the session was closed while waiting.
"""
try:
from hermes_cli.mcp_startup import mcp_discovery_in_flight
except Exception:
return
if not mcp_discovery_in_flight():
return
import threading
agent = state.agent
session_id = state.session_id
def _wait_then_refresh() -> None:
try:
from hermes_cli.mcp_startup import join_mcp_discovery
if not join_mcp_discovery(timeout=30.0):
return
# Session may have been closed while we waited. In-memory-only
# lookup on purpose: ``get_session()`` falls through to a DB
# restore that builds a whole new AIAgent as a side effect just
# to decide "no-op" here (the TUI equivalent also checks its
# in-memory dict only).
with self.session_manager._lock:
current = self.session_manager._sessions.get(session_id)
if current is None or current.agent is not agent:
return
# Cache safety: never rebuild the tool list once the conversation
# has started — that would invalidate the cached prompt prefix.
# Serialized with turn start: ``prompt()`` flips ``is_running``
# under ``runtime_lock`` before dispatching, so holding it here
# (and bailing when a turn is already running) closes the window
# where the guard passes but the first prompt starts before the
# refresh publishes — which would swap ``tools=`` mid-turn and
# break the just-created cache prefix.
with current.runtime_lock:
if current.is_running:
return
if (
int(getattr(agent, "_user_turn_count", 0) or 0) > 0
or int(getattr(agent, "_api_call_count", 0) or 0) > 0
):
return
from tools.mcp_tool import refresh_agent_mcp_tools
added = refresh_agent_mcp_tools(agent, quiet_mode=True)
if added:
logger.info(
"Session %s: late MCP refresh added %d tools: %s",
session_id,
len(added),
", ".join(sorted(added)),
)
except Exception:
logger.debug(
"Session %s: late MCP refresh failed",
session_id,
exc_info=True,
)
threading.Thread(
target=_wait_then_refresh,
name=f"acp-mcp-late-refresh-{session_id}",
daemon=True,
).start()
# ---- ACP lifecycle ------------------------------------------------------
async def initialize(
@@ -969,11 +1399,49 @@ class HermesACPAgent(acp.Agent):
return text
return ""
@staticmethod
def _history_summary_meta(message: dict[str, Any], text: str) -> dict[str, Any] | None:
"""Build the ``_meta`` payload for a replayed compaction summary.
Compaction summaries are persisted as ordinary history messages —
standalone handoffs under ``role="user"`` OR ``role="assistant"``
(the compressor picks whichever role keeps alternation valid), and
merge-into-tail messages where the summary is appended after the
first preserved tail message's real content. Without a wire flag,
ACP frontends render all of these as ordinary turns.
Two distinct keys under ``_meta.hermes`` (ACP's extensibility
channel), so clients cannot accidentally hide real content:
* ``compactionSummary: true`` — the entire chunk is the handoff
summary. Safe to restyle or collapse wholesale.
* ``containsCompactionSummary: true`` — a merged-tail message: real
preserved turn content followed by the summary. Clients may style
it, but collapsing the whole chunk would hide the preserved
content, hence the separate key.
Detection honors the in-process ``_compressed_summary`` flag and
falls back to content classification, so it also works for a
DB-reloaded session that lost the in-memory flag.
"""
kind = ContextCompressor.classify_summary_content(text)
if kind is None and message.get(COMPRESSED_SUMMARY_METADATA_KEY):
# Flagged in-process but content didn't classify (e.g. future
# prefix drift): treat as a standalone summary — the flag is only
# ever set on summary-bearing messages.
kind = "standalone"
if kind == "standalone":
return {"hermes": {"compactionSummary": True}}
if kind == "merged":
return {"hermes": {"containsCompactionSummary": True}}
return None
@staticmethod
def _history_message_update(
*,
role: str,
text: str,
field_meta: dict[str, Any] | None = None,
) -> UserMessageChunk | AgentMessageChunk | None:
"""Build an ACP history replay update for a user/assistant message."""
block = TextContentBlock(type="text", text=text)
@@ -981,11 +1449,13 @@ class HermesACPAgent(acp.Agent):
return UserMessageChunk(
session_update="user_message_chunk",
content=block,
field_meta=field_meta,
)
if role == "assistant":
return AgentMessageChunk(
session_update="agent_message_chunk",
content=block,
field_meta=field_meta,
)
return None
@@ -1056,7 +1526,11 @@ class HermesACPAgent(acp.Agent):
if role == "user":
text = self._history_message_text(message)
if text:
update = self._history_message_update(role=role, text=text)
update = self._history_message_update(
role=role,
text=text,
field_meta=self._history_summary_meta(message, text),
)
if update is not None and not await _send(update):
return
continue
@@ -1068,7 +1542,11 @@ class HermesACPAgent(acp.Agent):
text = self._history_message_text(message)
if text:
update = self._history_message_update(role=role, text=text)
update = self._history_message_update(
role=role,
text=text,
field_meta=self._history_summary_meta(message, text),
)
if update is not None and not await _send(update):
return
@@ -1118,6 +1596,7 @@ class HermesACPAgent(acp.Agent):
) -> NewSessionResponse:
state = self.session_manager.create_session(cwd=cwd)
await self._register_session_mcp_servers(state, mcp_servers)
self._schedule_mcp_late_refresh(state)
logger.info("New session %s (cwd=%s)", state.session_id, cwd)
self._schedule_available_commands_update(state.session_id)
self._schedule_usage_update(state)
@@ -1142,6 +1621,7 @@ class HermesACPAgent(acp.Agent):
logger.warning("load_session: session %s not found", session_id)
return None
await self._register_session_mcp_servers(state, mcp_servers)
self._schedule_mcp_late_refresh(state)
logger.info("Loaded session %s", session_id)
# Per ACP spec, `session/load` must stream the prior conversation back
# to the client via `session/update` notifications BEFORE responding,
@@ -1189,6 +1669,7 @@ class HermesACPAgent(acp.Agent):
logger.warning("resume_session: session %s not found, creating new", session_id)
state = self.session_manager.create_session(cwd=cwd)
await self._register_session_mcp_servers(state, mcp_servers)
self._schedule_mcp_late_refresh(state)
logger.info("Resumed session %s", state.session_id)
# See `load_session` above for the spec rationale — replay must
# complete before the response so clients receive the full transcript
@@ -1218,12 +1699,19 @@ class HermesACPAgent(acp.Agent):
with state.runtime_lock:
if state.is_running and state.current_prompt_text:
state.interrupted_prompt_text = state.current_prompt_text
state.cancel_event.set()
try:
if getattr(state, "agent", None) and hasattr(state.agent, "interrupt"):
state.agent.interrupt()
except Exception:
logger.debug("Failed to interrupt ACP session %s", session_id, exc_info=True)
# Publish cancellation and hard-stop the agent before another
# prompt can acquire this lock and mistake the turn for
# redirectable work.
state.cancel_event.set()
try:
if getattr(state, "agent", None):
request_hard_interrupt(state.agent)
except Exception:
logger.debug(
"Failed to interrupt ACP session %s",
session_id,
exc_info=True,
)
logger.info("Cancelled session %s", session_id)
async def fork_session(
@@ -1352,6 +1840,26 @@ class HermesACPAgent(acp.Agent):
elif rewrite_idle:
user_text = steer_text
user_content = steer_text
elif (
text_only_prompt
and isinstance(user_content, str)
and not user_text.startswith("/")
):
# Some ACP clients implement "stop and send" as two protocol calls:
# cancel the active prompt, then submit plain correction text. Keep
# the cancelled request attached so deictic follow-ups ("not that
# file") still have an explicit target.
interrupted_prompt = ""
with state.runtime_lock:
if not state.is_running and state.interrupted_prompt_text:
interrupted_prompt = state.interrupted_prompt_text
state.interrupted_prompt_text = ""
if interrupted_prompt:
user_text = (
f"{interrupted_prompt}\n\n"
f"User correction/guidance after interrupt: {user_text}"
)
user_content = user_text
# Intercept slash commands — handle locally without calling the LLM.
# Slash commands are text-only; if the client included images/resources,
@@ -1366,23 +1874,54 @@ class HermesACPAgent(acp.Agent):
await self._send_usage_update(state)
return PromptResponse(stop_reason="end_turn")
# If Zed sends another regular prompt while the same ACP session is
# still running, queue it instead of racing two AIAgent loops against
# the same state.history. /steer and /queue are handled above and can
# land immediately.
# If the client sends another regular text prompt while this ACP session
# is running, route it through the core active-turn redirect. Rich media
# and older runtimes retain the proven next-turn queue fallback.
redirected = False
queued_depth: int | None = None
with state.runtime_lock:
if state.is_running:
queued_text = user_text or "[Image attachment]"
state.queued_prompts.append(queued_text)
depth = len(state.queued_prompts)
if self._conn:
update = acp.update_agent_message_text(
f"Queued for the next turn. ({depth} queued)"
if (
text_only_prompt
and isinstance(user_content, str)
and getattr(
state.agent,
"_supports_active_turn_redirect",
False,
)
await self._conn.session_update(session_id, update)
return PromptResponse(stop_reason="end_turn")
state.is_running = True
state.current_prompt_text = user_text or "[Image attachment]"
is True
and hasattr(state.agent, "redirect")
):
try:
redirected = bool(state.agent.redirect(user_content))
except Exception:
logger.debug(
"ACP active-turn redirect failed for %s",
session_id,
exc_info=True,
)
if not redirected:
queued_text = user_text or "[Image attachment]"
state.queued_prompts.append(queued_text)
queued_depth = len(state.queued_prompts)
else:
state.is_running = True
state.current_prompt_text = user_text or "[Image attachment]"
if redirected:
if self._conn:
update = acp.update_agent_message_text(
"Redirected the active turn with your correction."
)
await self._conn.session_update(session_id, update)
return PromptResponse(stop_reason="end_turn")
if queued_depth is not None:
if self._conn:
update = acp.update_agent_message_text(
f"Queued for the next turn. ({queued_depth} queued)"
)
await self._conn.session_update(session_id, update)
return PromptResponse(stop_reason="end_turn")
logger.info("Prompt on session %s: %s", session_id, user_text[:100])
@@ -1478,7 +2017,19 @@ class HermesACPAgent(acp.Agent):
clear_session_vars,
set_session_vars,
)
session_tokens = set_session_vars(session_key=session_id)
# ``cwd`` pins the logical working directory for this context,
# which is what the system prompt's "Current working directory"
# line reports (agent/prompt_builder.py -> resolve_agent_cwd).
# Without it the prompt advertises the global Hermes workspace
# while the tools are rooted at the client's project, so the
# model emits absolute paths under ~/.hermes/workspace and the
# edit silently lands outside the editor's workspace.
# cron_session="" explicitly marks this as a non-cron context,
# masking any leaked process-global HERMES_CRON_SESSION (#37968).
session_tokens = set_session_vars(
session_key=session_id, session_id=session_id, cwd=state.cwd,
cron_session="",
)
except Exception:
session_tokens = None
clear_session_vars = None # type: ignore[assignment]
@@ -1509,6 +2060,17 @@ class HermesACPAgent(acp.Agent):
# never leaks one session's id into the next session's tools.
previous_session_id = os.environ.get("HERMES_SESSION_ID")
os.environ["HERMES_SESSION_ID"] = session_id
# Auto-titling fires inside the turn prologue now; give the agent
# this session's notifier so a new title reaches the client as a
# session-info update instead of waiting for the next one.
def _notify_title_update(_title: str, _source: str) -> None:
if conn:
loop.call_soon_threadsafe(
asyncio.create_task,
self._send_session_info_update(session_id),
)
agent._on_session_title = _notify_title_update
try:
result = agent.run_conversation(
user_message=user_content,
@@ -1606,43 +2168,6 @@ class HermesACPAgent(acp.Agent):
suppress_interrupt_response = interrupted and final_response.startswith(
INTERRUPT_WAITING_FOR_MODEL_PREFIX
)
if final_response and not suppress_interrupt_response:
try:
from agent.title_generator import maybe_auto_title
def _notify_title_update(_title: str) -> None:
if conn:
loop.call_soon_threadsafe(
asyncio.create_task,
self._send_session_info_update(session_id),
)
# Snapshot the runtime identity; the validator lets the
# background titler skip its LLM call if the session's model
# changed before it fires (#19027).
_title_model = getattr(state.agent, "model", None)
_title_provider = getattr(state.agent, "provider", None)
maybe_auto_title(
self.session_manager._get_db(),
session_id,
user_text,
final_response,
state.history,
main_runtime={
"model": getattr(state.agent, "model", None),
"provider": getattr(state.agent, "provider", None),
"base_url": getattr(state.agent, "base_url", None),
"api_key": getattr(state.agent, "api_key", None),
"api_mode": getattr(state.agent, "api_mode", None),
},
runtime_validator=lambda: (
getattr(state.agent, "model", None) == _title_model
and getattr(state.agent, "provider", None) == _title_provider
),
title_callback=_notify_title_update,
)
except Exception:
logger.debug("Failed to auto-title ACP session %s", session_id, exc_info=True)
if (
final_response
and conn
@@ -1765,8 +2290,26 @@ class HermesACPAgent(acp.Agent):
if handler is None:
return None # not a known command — let the LLM handle it
try:
# Slash handlers run on the event-loop thread, OUTSIDE the per-turn
# contextvars.copy_context() that pins the session cwd for the agent
# call. ``/compress`` and ``/model`` reach code that REBUILDS the
# system prompt (agent._build_system_prompt -> resolve_agent_cwd), so
# an unpinned handler bakes the Hermes install tree into the session's
# cached prompt — persisted, and therefore poisoning every later turn
# even though the turn itself is pinned. Pin inside a fresh context so
# the write can't leak into other concurrent ACP sessions and needs no
# teardown.
def _dispatch() -> str | None:
try:
from agent.runtime_cwd import set_session_cwd
set_session_cwd(state.cwd)
except Exception:
logger.debug("Could not pin ACP session cwd for slash command", exc_info=True)
return handler(args, state)
try:
return contextvars.copy_context().run(_dispatch)
except Exception as e:
logger.error("Slash command /%s error: %s", cmd, e, exc_info=True)
return f"Error executing /{cmd}: {e}"
@@ -1826,8 +2369,8 @@ class HermesACPAgent(acp.Agent):
return "No tools available."
lines = [f"Available tools ({len(tools)}):"]
for t in tools:
name = t.get("function", {}).get("name", "?")
desc = t.get("function", {}).get("description", "")
name = (t.get("function") or {}).get("name", "?")
desc = (t.get("function") or {}).get("description", "")
# Truncate long descriptions
if len(desc) > 80:
desc = desc[:77] + "..."
@@ -1911,7 +2454,10 @@ class HermesACPAgent(acp.Agent):
lines.append(f"Compression threshold: ~{threshold_tokens:,} tokens")
if getattr(agent, "compression_enabled", True) is False:
lines.append("Compression is disabled for this agent.")
lines.append(
"Auto-compaction is disabled (compression.enabled: false); "
"/compress still compresses manually."
)
else:
lines.append("Tip: run /compress to compress manually before the threshold.")
@@ -1938,8 +2484,9 @@ class HermesACPAgent(acp.Agent):
return "Nothing to compress — conversation is empty."
try:
agent = state.agent
if not getattr(agent, "compression_enabled", True):
return "Context compression is disabled for this agent."
# No compression_enabled gate: the flag disables *automatic*
# compaction only; manual /compress must keep working (matches
# the CLI /compress and gateway handlers).
if not hasattr(agent, "_compress_context"):
return "Context compression not available for this agent."
@@ -1964,6 +2511,7 @@ class HermesACPAgent(acp.Agent):
getattr(agent, "_cached_system_prompt", "") or "",
approx_tokens=approx_tokens,
task_id=state.session_id,
force=True,
)
finally:
agent._session_db = original_session_db
+46 -10
View File
@@ -56,7 +56,18 @@ def _normalize_cwd_for_compare(cwd: str | None) -> str:
elif re.match(r"^/mnt/[A-Za-z]/", expanded):
expanded = f"/mnt/{expanded[5].lower()}/{expanded[7:]}"
return os.path.normpath(expanded)
# Resolve symlink aliases so equivalent spellings of the same directory
# compare equal — macOS reports editor workspaces as ``/var/...`` while
# sessions get stored under ``/private/var/...`` (and ``/tmp`` vs
# ``/private/tmp``), which made ACP history filters silently drop a
# workspace's own sessions. ``os.path.realpath`` is lexical for missing
# paths (strict=False), so cwds that don't exist on this host — e.g.
# WSL-translated Windows drives — keep the previous normpath behavior.
# Ported from PrimeIntellect-ai/prime-agent#628.
try:
return os.path.realpath(expanded)
except OSError:
return os.path.normpath(expanded)
def _build_session_title(title: Any, preview: Any, cwd: str | None) -> str:
@@ -480,16 +491,17 @@ class SessionManager:
# fresh agent with _session_db_created=False (so the check above
# is False) yet leave the durable archived transcript in place.
# A full-history replace would DELETE those archived rows just
# like the owned-agent case. Guard against it: when archived
# rows exist, replace ONLY the live (active=1) set and leave the
# archived turns untouched; otherwise the destructive replace is
# safe (fresh create/fork with no archived history to lose).
try:
has_archived = db.has_archived_messages(state.session_id)
except Exception:
has_archived = False
# like the owned-agent case. Guard against it by replacing ONLY
# the live (active=1) set unconditionally: on a fresh
# create/fork every row is active=1, so active-only replace is
# behaviorally identical to the full replace — and when archived
# rows DO exist they survive. An existence probe here
# (has_archived_messages) would fail OPEN into the destructive
# replace on any DB error and can race a concurrent
# archive_and_compact — the same probe failure mode #80216's
# /retry fix (gateway/slash_commands.py) deliberately avoids.
db.replace_messages(
state.session_id, state.history, active_only=has_archived
state.session_id, state.history, active_only=True
)
except Exception:
logger.warning("Failed to persist ACP session %s", state.session_id, exc_info=True)
@@ -648,6 +660,30 @@ class SessionManager:
logger.debug("ACP session falling back to default provider resolution", exc_info=True)
_register_task_cwd(session_id, cwd)
# Bounded wait for background MCP discovery so already-spawning fast
# servers land in the agent's tool snapshot. ACP entry.py fires
# discovery in a background daemon thread (start_background_mcp_discovery);
# the agent snapshots tools once at build (run_agent/agent_init) and
# never re-reads the registry, so without this join a reachable-but-
# slow configured server would be invisible for the whole session.
# ``ensure_mcp_discovery_before_agent_build`` also (re)starts discovery
# when the entry.py spawn never ran or exited with zero connected
# servers (the retry-after-zero-connected allowance), making this
# construction site self-sufficient. Bounded by
# ``mcp_discovery_timeout`` (config.yaml, default ~1.5s) so a dead
# server can't block — servers that miss the bound are picked up by
# the automatic late-refresh (see HermesACPAgent._schedule_mcp_late_refresh).
try:
from hermes_cli.mcp_startup import ensure_mcp_discovery_before_agent_build
ensure_mcp_discovery_before_agent_build(
logger=logger,
thread_name="acp-mcp-discovery",
)
except Exception:
logger.debug("ACP: bounded MCP discovery wait failed", exc_info=True)
agent = AIAgent(**kwargs)
# Codex app-server sessions are spawned lazily on the first turn. Stamp
# the ACP workspace onto the agent so the Codex runtime starts from the
+2 -1
View File
@@ -75,7 +75,8 @@ _POLISHED_TOOLS = {
"feishu_doc_read", "feishu_drive_list_comments", "feishu_drive_list_comment_replies",
"feishu_drive_reply_comment", "feishu_drive_add_comment",
"kanban_create", "kanban_show", "kanban_comment", "kanban_complete",
"kanban_block", "kanban_link", "kanban_heartbeat",
"kanban_block", "kanban_request_review", "kanban_request_changes",
"kanban_link", "kanban_heartbeat",
"yb_query_group_info", "yb_query_group_members", "yb_search_sticker",
"yb_send_dm", "yb_send_sticker",
}
-16
View File
@@ -1,16 +0,0 @@
{
"id": "hermes-agent",
"name": "Hermes Agent",
"version": "0.19.0",
"description": "Self-improving open-source AI agent by Nous Research with ACP editor integration, persistent memory, skills, and rich tool support.",
"repository": "https://github.com/NousResearch/hermes-agent",
"website": "https://hermes-agent.nousresearch.com/docs/user-guide/features/acp",
"authors": ["Nous Research"],
"license": "MIT",
"distribution": {
"uvx": {
"package": "hermes-agent[acp]==0.19.0",
"args": ["hermes-acp"]
}
}
}
-8
View File
@@ -1,8 +0,0 @@
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 16 16" width="16" height="16" fill="none">
<path d="M8 1.5v13" stroke="currentColor" stroke-width="1.5" stroke-linecap="round"/>
<path d="M8 3.25c-2.35-1.4-4.7-.95-6.25.35 1.85-.2 3.8.2 5.55 1.55" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
<path d="M8 3.25c2.35-1.4 4.7-.95 6.25.35-1.85-.2-3.8.2-5.55 1.55" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
<path d="M8 13.25c-2.3-1-3.05-2.65-1.35-4.15-2 .8-2.35 2.95-.35 4" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
<path d="M8 13.25c2.3-1 3.05-2.65 1.35-4.15 2 .8 2.35 2.95.35 4" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
<circle cx="8" cy="1.8" r="1.1" fill="currentColor"/>
</svg>

Before

Width:  |  Height:  |  Size: 882 B

+12
View File
@@ -701,6 +701,18 @@ def redeem_codex_reset_credit(
remaining = max(0, available - 1)
plural = "s" if remaining != 1 else ""
if code == "reset":
# The redeemed reset restores the account's quota upstream — lift any
# persisted pool cooldowns so Hermes doesn't keep the credential
# frozen behind the now-stale ``last_error_reset_at`` (issue #43747).
try:
from hermes_cli.auth import clear_codex_pool_quota_cooldowns
clear_codex_pool_quota_cooldowns()
except Exception:
logger.debug(
"Failed to clear Codex pool cooldowns after reset redemption",
exc_info=True,
)
return CodexResetRedeemResult(
status="reset",
message=(
+287
View File
@@ -0,0 +1,287 @@
"""OpenAI-shape bridge shared by Hermes' ACP clients.
An ACP agent (``copilot --acp``, and the ACP CLIs that reach Hermes as
providers) speaks the Agent Client Protocol, which has no OpenAI-style
``tools``/``tool_calls`` channel: a prompt is text, and a response is text plus
the agent's *own* tool notifications. Hermes' agentic surface — ``memory``,
``todo``, ``skill_manage`` and friends — is dispatched from OpenAI-shaped
``tool_calls``, so on an ACP provider it can only work if the schemas travel
*into* the prompt as text and the calls are parsed back *out* of the response
text.
``agent/copilot_acp_client.py`` already carried a private copy of that bridge.
This module is that code, lifted verbatim into one place so every ACP client
shares it instead of re-deriving the wire contract:
* :func:`render_tool_bridge_sections` — prompt sections describing the
forwarded tools and the ``<tool_call>{...}</tool_call>`` contract.
* :func:`extract_tool_calls_from_text` — parse those blocks back into
``ChatCompletionMessageToolCall`` objects and return the response text with
the blocks stripped.
* :func:`completion_to_stream_chunks` — re-shape a one-shot ACP response as
OpenAI stream chunks for callers that asked for ``stream=True`` (an ACP turn
is inherently one-shot from Hermes' perspective).
The one axis clients differ on is *which* tools they forward, so
``render_tool_bridge_sections`` takes an optional allowlist. A CLI with no tools
of its own (Copilot) forwards everything Hermes offers; a CLI that is an
autonomous agent with its own read/edit/execute tools must forward only Hermes'
agent-level tools, because re-offering the overlapping ones makes Hermes re-run
work the agent already finished.
"""
from __future__ import annotations
import json
import re
from types import SimpleNamespace
from typing import Any, Iterable
from openai.types.chat.chat_completion_message_tool_call import (
ChatCompletionMessageToolCall,
Function,
)
TOOL_CALL_BLOCK_RE = re.compile(r"<tool_call>\s*(\{.*?\})\s*</tool_call>", re.DOTALL)
TOOL_CALL_JSON_RE = re.compile(
r"\{\s*\"id\"\s*:\s*\"[^\"]+\"\s*,\s*\"type\"\s*:\s*\"function\"\s*,\s*\"function\"\s*:\s*\{.*?\}\s*\}",
re.DOTALL,
)
# The contract sentence shared by every ACP client: how to emit a call.
TOOL_CALL_CONTRACT = (
"Available tools (OpenAI function schema). "
"When using a tool, emit ONLY <tool_call>{...}</tool_call> with one JSON object "
"containing id/type/function{name,arguments}. arguments must be a JSON string."
)
__all__ = [
"TOOL_CALL_BLOCK_RE",
"TOOL_CALL_JSON_RE",
"TOOL_CALL_CONTRACT",
"StreamChunks",
"build_openai_tool_call",
"tool_specs_from_openai_tools",
"render_tool_bridge_sections",
"extract_tool_calls_from_text",
"completion_to_stream_chunks",
]
class StreamChunks(list):
"""Stream chunks that can still carry response-level attributes.
Hermes reads provider-level extras off the object returned by
``chat.completions.create`` (e.g. ``hermes_projected_messages``, consumed by
``agent/provider_projection.py``). A plain list of chunks would silently drop
them on the ``stream=True`` path, so ACP clients return this instead and copy
the extras onto it.
"""
def completion_to_stream_chunks(completion: SimpleNamespace) -> StreamChunks:
"""Convert a one-shot ACP response into OpenAI-style stream chunks.
Response-level attributes other than ``choices``/``usage``/``model`` are
copied onto the returned object so nothing a caller reads off the completion
is lost when it asked to stream.
"""
choice = completion.choices[0]
message = choice.message
tool_call_deltas = None
if message.tool_calls:
tool_call_deltas = []
for index, tool_call in enumerate(message.tool_calls):
tool_call_deltas.append(
SimpleNamespace(
index=index,
id=getattr(tool_call, "id", None),
type=getattr(tool_call, "type", "function"),
function=SimpleNamespace(
name=getattr(tool_call.function, "name", None),
arguments=getattr(tool_call.function, "arguments", None),
),
)
)
delta = SimpleNamespace(
role="assistant",
content=message.content or None,
tool_calls=tool_call_deltas,
reasoning_content=getattr(message, "reasoning_content", None),
reasoning=getattr(message, "reasoning", None),
)
data_chunk = SimpleNamespace(
choices=[
SimpleNamespace(
index=0,
delta=delta,
finish_reason=choice.finish_reason,
)
],
model=completion.model,
usage=None,
)
usage_chunk = SimpleNamespace(
choices=[],
model=completion.model,
usage=completion.usage,
)
chunks = StreamChunks([data_chunk, usage_chunk])
for key, value in vars(completion).items():
if key not in ("choices", "usage", "model"):
setattr(chunks, key, value)
return chunks
def build_openai_tool_call(
*,
call_id: str,
name: str,
arguments: str,
) -> ChatCompletionMessageToolCall:
"""Build an OpenAI-compatible tool-call object for downstream handling."""
return ChatCompletionMessageToolCall(
id=call_id,
call_id=call_id,
response_item_id=None,
type="function",
function=Function(name=name, arguments=arguments),
)
def tool_specs_from_openai_tools(
tools: list[dict[str, Any]] | None,
*,
allowlist: Iterable[str] | None = None,
) -> list[dict[str, Any]]:
"""Flatten OpenAI ``tools`` into ``{name, description, parameters}`` specs.
Malformed entries are skipped. When ``allowlist`` is given, only tools whose
name is in it survive — that is how a client forwards just Hermes'
agent-level tools instead of the whole toolset.
"""
allowed = {str(n).strip() for n in allowlist} if allowlist is not None else None
specs: list[dict[str, Any]] = []
for t in tools or []:
if not isinstance(t, dict):
continue
fn = t.get("function") or {}
if not isinstance(fn, dict):
continue
name = fn.get("name")
if not isinstance(name, str) or not name.strip():
continue
name = name.strip()
if allowed is not None and name not in allowed:
continue
specs.append(
{
"name": name,
"description": fn.get("description", ""),
"parameters": fn.get("parameters", {}),
}
)
return specs
def render_tool_bridge_sections(
tools: list[dict[str, Any]] | None,
tool_choice: Any = None,
*,
allowlist: Iterable[str] | None = None,
) -> list[str]:
"""Prompt sections that carry the forwarded tool schemas + choice hint.
Returns an empty list when no tool survives filtering and no choice hint was
requested, so callers can splice the result into their section list
unconditionally.
"""
specs = tool_specs_from_openai_tools(tools, allowlist=allowlist)
sections: list[str] = []
if specs:
sections.append(
TOOL_CALL_CONTRACT + "\n" + json.dumps(specs, ensure_ascii=False)
)
if tool_choice is not None:
sections.append(f"Tool choice hint: {json.dumps(tool_choice, ensure_ascii=False)}")
return sections
def extract_tool_calls_from_text(
text: str,
) -> tuple[list[ChatCompletionMessageToolCall], str]:
"""Pull ``<tool_call>`` blocks out of an ACP response.
Returns ``(tool_calls, cleaned_text)`` where ``cleaned_text`` is the
response with the consumed blocks removed, so the assistant message doesn't
show raw JSON to the user.
"""
if not isinstance(text, str) or not text.strip():
return [], ""
extracted: list[ChatCompletionMessageToolCall] = []
consumed_spans: list[tuple[int, int]] = []
def _try_add_tool_call(raw_json: str) -> None:
try:
obj = json.loads(raw_json)
except Exception:
return
if not isinstance(obj, dict):
return
fn = obj.get("function")
if not isinstance(fn, dict):
return
fn_name = fn.get("name")
if not isinstance(fn_name, str) or not fn_name.strip():
return
fn_args = fn.get("arguments", "{}")
if not isinstance(fn_args, str):
fn_args = json.dumps(fn_args, ensure_ascii=False)
call_id = obj.get("id")
if not isinstance(call_id, str) or not call_id.strip():
call_id = f"acp_call_{len(extracted)+1}"
extracted.append(
build_openai_tool_call(
call_id=call_id,
name=fn_name.strip(),
arguments=fn_args,
)
)
for m in TOOL_CALL_BLOCK_RE.finditer(text):
raw = m.group(1)
_try_add_tool_call(raw)
consumed_spans.append((m.start(), m.end()))
# Only try bare-JSON fallback when no XML blocks were found.
if not extracted:
for m in TOOL_CALL_JSON_RE.finditer(text):
raw = m.group(0)
_try_add_tool_call(raw)
consumed_spans.append((m.start(), m.end()))
if not consumed_spans:
return extracted, text.strip()
consumed_spans.sort()
merged: list[tuple[int, int]] = []
for start, end in consumed_spans:
if not merged or start > merged[-1][1]:
merged.append((start, end))
else:
merged[-1] = (merged[-1][0], max(merged[-1][1], end))
parts: list[str] = []
cursor = 0
for start, end in merged:
if cursor < start:
parts.append(text[cursor:start])
cursor = max(cursor, end)
if cursor < len(text):
parts.append(text[cursor:])
cleaned = "\n".join(p.strip() for p in parts if p and p.strip()).strip()
return extracted, cleaned
+637 -91
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+511 -53
View File
@@ -23,7 +23,26 @@ from urllib.parse import urlparse
from hermes_constants import get_hermes_home
from typing import Any, Dict, List, Optional, Tuple
from utils import base_url_host_matches, normalize_proxy_env_vars
from utils import base_url_host_matches, base_url_hostname, normalize_proxy_env_vars
from agent.secret_scope import get_secret as _get_secret
try:
import hermes_cli as _hermes_cli
_HERMES_VERSION = str(_hermes_cli.__version__)
except Exception:
_HERMES_VERSION = "0.0.0"
def _getenv(name: str, default: str = "") -> str:
"""Profile-scoped replacement for os.getenv on credential reads.
Routes through the secret scope (Workstream A): identical to os.getenv
when multiplexing is off, scope-aware (and fail-closed on an unscoped
read) when on. Mirrors the same wrapper in hermes_cli/runtime_provider.py.
"""
val = _get_secret(name, default)
return val if val is not None else default
# NOTE: `import anthropic` is deliberately NOT at module top — the SDK pulls
# ~220 ms of imports (anthropic.types, anthropic.lib.tools._beta_runner, etc.)
@@ -113,6 +132,17 @@ _NO_XHIGH_CLAUDE_SUBSTRINGS = (
"claude-sonnet-4-6", "claude-sonnet-4.6",
)
# Adaptive Claude families that REJECT a thinking disable — thinking is
# mandatory and ``thinking: {"type": "disabled"}`` answers HTTP 400. The Portal
# catalog flags the same families with ``reasoning.mandatory``.
#
# Unlike the two lists above, the failure here is asymmetric: a missing entry
# 400s the turn, while a spurious one only leaves thinking on. When in doubt,
# add the family.
_MANDATORY_THINKING_CLAUDE_SUBSTRINGS = (
"claude-fable",
)
def _is_claude_model(model: str | None) -> bool:
return "claude" in (model or "").lower()
@@ -279,6 +309,32 @@ def _supports_xhigh_effort(model: str) -> bool:
return not any(v in m for v in _NO_XHIGH_CLAUDE_SUBSTRINGS)
def _accepts_thinking_disable(model: str) -> bool:
"""Return True when *model* accepts an explicit thinking disable.
Adaptive Claude models default to thinking ON, so "thinking off" only
takes effect if we actively send ``thinking: {"type": "disabled"}`` —
omitting the parameter leaves the upstream default in place and the model
thinks anyway. Reasoning-mandatory families reject the disable outright
with an HTTP 400, so they keep the omit-everything behavior.
Legacy manual-thinking Claude models are excluded because they need no
disable: thinking is opt-in there via ``budget_tokens``, so not sending
the block already means off.
Scoped to Claude deliberately. Kimi/Moonshot endpoints also speak the
adaptive contract, but their documented disable behavior is omission
(#13848) and they are not part of this bug; sending them a new parameter
on the strength of Claude's contract would be a guess.
"""
if not _is_claude_model(model):
return False
if not _supports_adaptive_thinking(model):
return False
m = model.lower()
return not any(v in m for v in _MANDATORY_THINKING_CLAUDE_SUBSTRINGS)
def _forbids_sampling_params(model: str) -> bool:
"""Return True for models that 400 on any non-default temperature/top_p/top_k.
@@ -368,7 +424,7 @@ def _detect_claude_code_version() -> str:
try:
result = _sp.run(
[cmd, "--version"],
capture_output=True, text=True, timeout=5,
capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5,
)
if result.returncode == 0 and result.stdout.strip():
# Output is like "2.1.74 (Claude Code)" or just "2.1.74"
@@ -455,6 +511,11 @@ def _is_kimi_coding_endpoint(base_url: str | None) -> bool:
return normalized.rstrip("/").lower().startswith("https://api.kimi.com/coding")
def _is_opencode_endpoint(base_url: str | None) -> bool:
"""Return True for OpenCode's Zen/Go relay (opencode.ai)."""
return base_url_host_matches(base_url or "", "opencode.ai")
# Model-name prefixes that identify the Kimi / Moonshot family. Covers
# - official slugs: ``kimi-k2.5``, ``kimi_thinking``, ``moonshot-v1-8k``
# - common release lines: ``k1.5-...``, ``k2-thinking``, ``k25-...``, ``k2.5-...``,
@@ -546,15 +607,49 @@ def _is_deepseek_anthropic_endpoint(base_url: str | None) -> bool:
return "/anthropic" in normalized.rstrip("/").lower()
def _is_nous_portal_endpoint(base_url: str | None) -> bool:
"""Return True for Nous Portal's Anthropic Messages route.
Portal serves its ``anthropic/*`` catalog natively at
``https://inference-api.nousresearch.com/v1/messages``. Portal-specific
behaviours key off this: Bearer JWT auth, verbatim catalog model ids,
and native thinking-signature replay.
Trusted hosts only:
1. Prod hostname ``inference-api.nousresearch.com``
2. The operator-set ``NOUS_INFERENCE_BASE_URL`` hostname (staging/preview)
Lookalikes such as ``inference-api.nousresearch.com.attacker.test`` are
rejected (hostname match, not substring).
"""
if base_url_host_matches(base_url or "", "inference-api.nousresearch.com"):
return True
try:
from hermes_cli.auth import _nous_inference_env_override
override = _nous_inference_env_override()
except Exception:
return False
if not override:
return False
# Exact host equality (not subdomain) so the env override can't broaden
# into sibling hosts the operator did not set.
override_host = base_url_hostname(override)
return bool(override_host) and base_url_hostname(base_url or "") == override_host
def _requires_bearer_auth(base_url: str | None) -> bool:
"""Return True for Anthropic-compatible providers that require Bearer auth.
Some third-party /anthropic endpoints implement Anthropic's Messages API but
require Authorization: Bearer instead of Anthropic's native x-api-key header.
MiniMax's global and China Anthropic-compatible endpoints, Azure AI
Foundry's Anthropic-style endpoint, and Palantir Foundry's LLM proxy
follow this pattern.
Foundry's Anthropic-style endpoint, Palantir Foundry's LLM proxy, and Nous
Portal's Messages route follow this pattern.
"""
if _is_nous_portal_endpoint(base_url):
return True
normalized = _normalize_base_url_text(base_url)
if not normalized:
return False
@@ -567,6 +662,10 @@ def _requires_bearer_auth(base_url: str | None) -> bool:
# Hostname match (not substring) so e.g. evil.com/palantirfoundry
# paths don't trigger Bearer auth.
or base_url_host_matches(normalized, "palantirfoundry.com")
# CommandCode's /provider/v1/messages endpoint uses Bearer auth,
# not Anthropic's native x-api-key header. Hostname match for the
# same reason as above.
or base_url_host_matches(normalized, "api.commandcode.ai")
)
@@ -721,7 +820,11 @@ def _build_anthropic_client_with_bearer_hook(
if common_betas:
kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)}
return _anthropic_sdk.Anthropic(**kwargs)
client = _anthropic_sdk.Anthropic(**kwargs)
# Same env-inference trap as build_anthropic_client: auth_token-only
# construction would otherwise also send ANTHROPIC_API_KEY as X-Api-Key.
client.api_key = None
return client
def build_anthropic_client(
@@ -807,12 +910,18 @@ def build_anthropic_client(
)
if _is_kimi_coding_endpoint(base_url):
# Kimi's /coding endpoint requires User-Agent: claude-code/0.1.0
# to be recognized as a valid Coding Agent. Without it, returns 403.
# Check this BEFORE _requires_bearer_auth since both match api.kimi.com/coding.
# Kimi's /coding endpoint requires a non-empty User-Agent to be
# recognized as a valid Coding Agent. Originally we sent
# ``claude-code/0.1.0`` (the minimum that avoided a 403), but the Kimi
# team asked us to identify ourselves properly so they can attribute
# traffic correctly. Send the same attribution header set we send to
# OpenRouter, Vercel AI Gateway, and Fireworks:
# HTTP-Referer + X-Title + HermesAgent User-Agent.
kwargs["api_key"] = api_key
kwargs["default_headers"] = {
"User-Agent": "claude-code/0.1.0",
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
"X-Title": "Hermes Agent",
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
**( {"anthropic-beta": ",".join(common_betas)} if common_betas else {} )
}
elif _requires_bearer_auth(normalized_base_url):
@@ -850,7 +959,28 @@ def build_anthropic_client(
if common_betas:
kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)}
return _anthropic_sdk.Anthropic(**kwargs)
if _is_opencode_endpoint(base_url):
# OpenCode identifies clients by request headers, like OpenRouter does.
# The OpenAI-wire paths pick these up from profile.default_headers
# (plugins/model-providers/opencode-zen), but the Anthropic Messages
# route builds its client right here and never sees the profile. Merge
# the same set on top of whatever auth branch ran above.
headers = dict(kwargs.get("default_headers") or {})
headers.setdefault("HTTP-Referer", "https://hermes-agent.nousresearch.com")
headers.setdefault("X-Title", "Hermes Agent")
headers.setdefault("User-Agent", f"HermesAgent/{_HERMES_VERSION}")
kwargs["default_headers"] = headers
client = _anthropic_sdk.Anthropic(**kwargs)
# Bearer-only construction leaves ``api_key`` unset, so the SDK fills it
# from ``ANTHROPIC_API_KEY`` (Hermes loads that into the process env from
# ``~/.hermes/.env``). The result is dual auth —
# ``X-Api-Key: sk-ant-…`` *and* ``Authorization: Bearer <portal-jwt>`` —
# on every Portal / MiniMax / OAuth Messages request. Clear the env-filled
# key whenever we intentionally authenticated via auth_token alone.
if "auth_token" in kwargs and "api_key" not in kwargs:
client.api_key = None
return client
def build_anthropic_bedrock_client(region: str):
@@ -914,7 +1044,7 @@ def _read_claude_code_credentials_from_keychain() -> Optional[Dict[str, Any]]:
"-s", "Claude Code-credentials",
"-w"],
capture_output=True,
text=True,
text=True, encoding='utf-8', errors='replace',
timeout=5,
stdin=subprocess.DEVNULL,
)
@@ -1275,7 +1405,7 @@ def _resolve_anthropic_pool_token() -> Optional[str]:
# to auth.json or trigger a network refresh from a bare resolve. select()
# is deliberately NOT used — it runs clear_expired=True, refresh=True,
# which would violate this read-only contract.
entries = pool._available_entries(clear_expired=False, refresh=False)
entries, _pending = pool._available_entries(clear_expired=False, refresh=False)
except Exception:
logger.debug("Failed to read Anthropic credential_pool", exc_info=True)
return None
@@ -1301,47 +1431,55 @@ def resolve_anthropic_token() -> Optional[str]:
Priority:
1. ANTHROPIC_TOKEN env var (OAuth/setup token saved by Hermes)
2. CLAUDE_CODE_OAUTH_TOKEN env var
3. Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json)
3. ANTHROPIC_API_KEY env var (explicit regular API key)
4. Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json)
— with automatic refresh if expired and a refresh token is available
4. Anthropic credential_pool OAuth entry (~/.hermes/auth.json)
5. ANTHROPIC_API_KEY env var (regular API key, or legacy fallback)
5. Anthropic credential_pool OAuth entry (~/.hermes/auth.json)
Returns the token string or None.
"""
creds = read_claude_code_credentials()
creds: Optional[Dict[str, Any]] = None
creds_loaded = False
def _read_creds() -> Optional[Dict[str, Any]]:
nonlocal creds, creds_loaded
if not creds_loaded:
creds = read_claude_code_credentials()
creds_loaded = True
return creds
# 1. Hermes-managed OAuth/setup token env var
token = os.getenv("ANTHROPIC_TOKEN", "").strip()
token = _getenv("ANTHROPIC_TOKEN").strip()
if token:
preferred = _prefer_refreshable_claude_code_token(token, creds)
preferred = _prefer_refreshable_claude_code_token(token, _read_creds())
if preferred:
return preferred
return token
# 2. CLAUDE_CODE_OAUTH_TOKEN (used by Claude Code for setup-tokens)
cc_token = os.getenv("CLAUDE_CODE_OAUTH_TOKEN", "").strip()
cc_token = _getenv("CLAUDE_CODE_OAUTH_TOKEN").strip()
if cc_token:
preferred = _prefer_refreshable_claude_code_token(cc_token, creds)
preferred = _prefer_refreshable_claude_code_token(cc_token, _read_creds())
if preferred:
return preferred
return cc_token
# 3. Claude Code credential file
resolved_claude_token = _resolve_claude_code_token_from_credentials(creds)
# 3. Regular API key. An explicit user-configured key must not be shadowed
# by auto-discovered Claude Code or credential-pool OAuth credentials.
api_key = _getenv("ANTHROPIC_API_KEY").strip()
if api_key:
return api_key
# 4. Claude Code credential file
resolved_claude_token = _resolve_claude_code_token_from_credentials(_read_creds())
if resolved_claude_token:
return resolved_claude_token
# 4. Hermes credential_pool OAuth entry.
# 5. Hermes credential_pool OAuth entry.
resolved_pool_token = _resolve_anthropic_pool_token()
if resolved_pool_token:
return resolved_pool_token
# 5. Regular API key, or a legacy OAuth token saved in ANTHROPIC_API_KEY.
# This remains as a compatibility fallback for pre-migration Hermes configs.
api_key = os.getenv("ANTHROPIC_API_KEY", "").strip()
if api_key:
return api_key
return None
@@ -1381,7 +1519,7 @@ def run_oauth_setup_token() -> Optional[str]:
# Check env vars that may have been set
for env_var in ("CLAUDE_CODE_OAUTH_TOKEN", "ANTHROPIC_TOKEN"):
val = os.getenv(env_var, "").strip()
val = _getenv(env_var).strip()
if val:
return val
@@ -1800,7 +1938,16 @@ def _to_plain_data(value: Any, *, _depth: int = 0, _path: Optional[set] = None)
if hasattr(value, "model_dump"):
_path.add(obj_id)
result = _to_plain_data(value.model_dump(), _depth=_depth + 1, _path=_path)
try:
# warnings=False: content blocks from the streaming accumulator
# (ParsedTextBlock et al.) trip pydantic's serializer-mismatch
# UserWarning against the generic Message union; the dump itself
# is correct, and the warning leaks to the user's terminal.
dumped = value.model_dump(warnings=False)
except TypeError:
# Duck-typed model_dump without pydantic's signature.
dumped = value.model_dump()
result = _to_plain_data(dumped, _depth=_depth + 1, _path=_path)
_path.discard(obj_id)
return result
if isinstance(value, dict):
@@ -1881,6 +2028,28 @@ def _content_parts_to_anthropic_blocks(parts: Any) -> List[Dict[str, Any]]:
return out
_EMPTY_TEXT_PLACEHOLDER = "(empty)"
def _safe_text(text: Any) -> str:
"""Return ``text`` if it's non-whitespace, else a non-whitespace placeholder.
The Anthropic Messages API rejects requests where a text content block is
empty or whitespace-only (HTTP 400 "text content blocks must contain
non-whitespace text"). When such a block gets stored in session history —
e.g. produced by context compression — it is replayed verbatim on every
subsequent turn, permanently wedging the session. Coercing to a
non-whitespace placeholder is self-healing: the next API call recovers.
Mirrors ``bedrock_adapter._safe_text`` (#9486); ref #69512.
"""
if text is None:
return _EMPTY_TEXT_PLACEHOLDER
if not isinstance(text, str):
text = str(text)
return text if text.strip() else _EMPTY_TEXT_PLACEHOLDER
def _sanitize_replay_block(b: Dict[str, Any]) -> Optional[Dict[str, Any]]:
"""Strip output-only fields from a stored Anthropic content block so it is
valid as REQUEST input on replay.
@@ -1898,7 +2067,18 @@ def _sanitize_replay_block(b: Dict[str, Any]) -> Optional[Dict[str, Any]]:
return None
btype = b.get("type")
if btype == "text":
out: Dict[str, Any] = {"type": "text", "text": b.get("text", "")}
text_val = b.get("text", "")
# Bedrock and strict Anthropic-compatible endpoints reject text
# blocks where "text" is empty or whitespace-only (#69512). Drop the
# blank block (the caller relocates any cache_control it carried and
# falls back to a non-whitespace placeholder when nothing survives)
# rather than coercing in place — a coerced "(empty)" block would be
# model-visible noise next to surviving thinking/tool_use blocks.
# Type-safe: captured blocks can carry text=None from an invalid
# upstream payload, which a bare .strip() would crash on.
if not isinstance(text_val, str) or not text_val.strip():
return None
out: Dict[str, Any] = {"type": "text", "text": text_val}
# citations is input-valid ONLY when it's a non-empty list; the SDK
# emits citations=None on responses, which the input schema rejects.
cits = b.get("citations")
@@ -1986,9 +2166,17 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
parsed_args = {}
redacted_input_by_id[_sanitize_tool_id(tc.get("id", ""))] = parsed_args
replayed: List[Dict[str, Any]] = []
_relocated_replay_cache_control = None
_dropped_blank_text = False
for b in ordered_blocks:
clean = _sanitize_replay_block(b)
if clean is None:
if isinstance(b, dict) and b.get("type") == "text":
_dropped_blank_text = True
if isinstance(b, dict) and isinstance(b.get("cache_control"), dict):
# A dropped blank text block can still carry the cache
# breakpoint marker -- relocate it rather than losing it.
_relocated_replay_cache_control = b["cache_control"]
continue
if clean.get("type") == "tool_use":
# Override raw (un-redacted) input with the redacted copy when
@@ -1998,20 +2186,90 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
if redacted is not None:
clean["input"] = redacted
replayed.append(clean)
# When every text block was blank and nothing cacheable survived
# (e.g. signed thinking + a blank text block, or a SOLE blank
# cache-marked block), emit the non-whitespace placeholder so the
# replayed message stays schema-valid (#69512) and a relocated cache
# marker still has a carrier instead of being silently lost.
_has_cacheable_replay = any(
isinstance(b, dict) and b.get("type") in {"text", "tool_use"}
for b in replayed
)
if not _has_cacheable_replay and (
_dropped_blank_text or _relocated_replay_cache_control is not None
):
replayed.append({"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER})
if replayed:
if _relocated_replay_cache_control is not None:
_apply_assistant_cache_control_to_last_cacheable_block(
replayed, _relocated_replay_cache_control
)
_apply_assistant_cache_control_to_last_cacheable_block(
replayed, m.get("cache_control")
)
# apply_anthropic_cache_control marks an assistant turn with
# non-empty text by writing cache_control INTO ``content`` (see
# _apply_cache_marker's list branch), not at the top level. This
# branch rebuilds the message from ordered_blocks and never reads
# ``content``, so that marker would be dropped -- and because
# _can_carry_marker already counted this message as a carrier, the
# breakpoint is burned rather than relocated. #56195 covered the
# complementary shape (blank content -> top-level marker); this is
# the interleaved thinking + preamble-text + tool_use shape.
_inline_cc = None
_msg_content = m.get("content")
if isinstance(_msg_content, list):
for _blk in _msg_content:
if isinstance(_blk, dict) and isinstance(
_blk.get("cache_control"), dict
):
_inline_cc = _blk["cache_control"]
break
if _inline_cc is not None:
_apply_assistant_cache_control_to_last_cacheable_block(
replayed, _inline_cc
)
return {"role": "assistant", "content": replayed}
blocks = _extract_preserved_thinking_blocks(m)
# Cache markers dropped along with a blank block are relocated onto the
# last surviving cacheable block below (via
# _apply_assistant_cache_control_to_last_cacheable_block), rather than
# lost -- prompt_caching.py's _apply_cache_marker() sets cache_control
# directly on content[-1] for list content, so if that last part happens
# to be blank text, dropping it silently would lose the breakpoint.
_relocated_cache_control = None
if content:
if isinstance(content, list):
converted_content = _convert_content_to_anthropic(content)
if isinstance(converted_content, list):
blocks.extend(converted_content)
# Bedrock and strict Anthropic-compatible endpoints reject
# text blocks where "text" is empty or whitespace-only. The
# ordered-replay path enforces the same invariant via
# _sanitize_replay_block(). Type-safe against ANY invalid
# "text" value from an upstream payload -- None, or a
# truthy non-string like an int -- not just None: checking
# isinstance() first (rather than `blk.get("text") or ""`)
# means a non-string value is treated as blank/invalid
# instead of reaching .strip() and raising AttributeError.
for blk in converted_content:
_blk_text = blk.get("text") if isinstance(blk, dict) else None
if (
isinstance(blk, dict)
and blk.get("type") == "text"
and (not isinstance(_blk_text, str) or not _blk_text.strip())
):
if isinstance(blk.get("cache_control"), dict):
_relocated_cache_control = blk["cache_control"]
continue
blocks.append(blk)
else:
blocks.append({"type": "text", "text": str(content)})
# Scalar (non-list) content: a whitespace-only string is the
# same invalid-payload case as an empty list block -- drop it
# rather than emitting a blank text block.
text_str = str(content)
if text_str.strip():
blocks.append({"type": "text", "text": text_str})
for tc in m.get("tool_calls", []):
if not tc or not isinstance(tc, dict):
continue
@@ -2027,9 +2285,6 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
"name": fn.get("name", ""),
"input": parsed_args,
})
_apply_assistant_cache_control_to_last_cacheable_block(
blocks, m.get("cache_control")
)
# Kimi's /coding endpoint (Anthropic protocol) requires assistant
# tool-call messages to carry reasoning_content when thinking is
# enabled server-side. Preserve it as a thinking block so Kimi
@@ -2055,10 +2310,26 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
)
if isinstance(reasoning_content, str) and not _already_has_thinking:
blocks.insert(0, {"type": "thinking", "thinking": reasoning_content})
# Anthropic rejects empty assistant content
effective = blocks or content
if not effective or effective == "":
effective = [{"type": "text", "text": "(empty)"}]
# Anthropic rejects empty assistant content. IMPORTANT: fall back only
# to the placeholder, never to the raw `content` variable -- `content`
# is the UNFILTERED original message content, and can itself be exactly
# the blank/whitespace-only payload the filtering above just removed
# (a sole blank text block, or scalar whitespace with no tool_calls).
# `blocks or content` there would silently restore the invalid provider
# payload this function exists to prevent (#69512).
effective = blocks if blocks else [{"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}]
# Applied here (after the empty-fallback resolution) rather than
# earlier against `blocks` directly, so a cache_control relocated from
# a dropped blank block that was the ONLY block still lands on the
# (empty) placeholder instead of being silently lost when blocks was
# empty at the point the marker would otherwise have been applied.
if _relocated_cache_control is not None:
_apply_assistant_cache_control_to_last_cacheable_block(
effective, _relocated_cache_control
)
_apply_assistant_cache_control_to_last_cacheable_block(
effective, m.get("cache_control")
)
return {"role": "assistant", "content": effective}
@@ -2128,13 +2399,14 @@ def _convert_user_message(content: Any) -> Dict[str, Any]:
"""Validate and convert a user message to anthropic format."""
if isinstance(content, list):
converted_blocks = _convert_content_to_anthropic(content)
if not converted_blocks or all(
(b.get("text") or "").strip() == ""
for b in converted_blocks
if isinstance(b, dict) and b.get("type") == "text"
):
converted_blocks = [{"type": "text", "text": "(empty message)"}]
return {"role": "user", "content": converted_blocks}
kept_blocks = _fix_blank_text_blocks_in_list(
converted_blocks,
placeholder_text="(empty message)",
msg_index=-1,
role="user",
location="_convert_user_message",
)
return {"role": "user", "content": kept_blocks}
else:
if not content or (isinstance(content, str) and not content.strip()):
content = "(empty message)"
@@ -2292,10 +2564,22 @@ def _manage_thinking_signatures(
replayed assistant tool-call messages. See hermes-agent#13848 (Kimi) and
hermes-agent#16748 (DeepSeek).
Nous Portal's ``/v1/messages`` route is the exception among third-party
hosts: it proxies Claude to Anthropic/Vertex/Bedrock and validates the
same signed thinking blocks. Sticky ``session_id`` keeps a conversation
on one upstream instance so those signatures stay warm — stripping them
here would 400 the first tool-loop turn ("thinking must be passed back").
Portal therefore takes the native Anthropic replay path below.
Mutates ``result`` in place.
"""
_THINKING_TYPES = frozenset(("thinking", "redacted_thinking"))
_is_third_party = _is_third_party_anthropic_endpoint(base_url)
# Portal speaks Anthropic's thinking contract end-to-end; do not treat it
# as a signature-blind proxy even though the host is not anthropic.com.
_is_third_party = (
_is_third_party_anthropic_endpoint(base_url)
and not _is_nous_portal_endpoint(base_url)
)
last_assistant_idx = None
for i in range(len(result) - 1, -1, -1):
@@ -2425,9 +2709,114 @@ def _ensure_leading_user_turn(result: List[Dict[str, Any]]) -> None:
Mirror the Bedrock Converse adapter, which unconditionally prepends a
minimal user turn when the first message is not user
(convert_messages_to_converse).
The inserted text block must be non-whitespace: Anthropic separately
rejects any text content block whose text is empty or whitespace-only
("text content blocks must contain non-whitespace text"), so a single
space here traded the "leading assistant turn" 400 for that one (#69512
class). Uses the same placeholder as every other synthesized filler
block in this module for consistency.
"""
if result and result[0].get("role") != "user":
result.insert(0, {"role": "user", "content": [{"type": "text", "text": " "}]})
result.insert(
0, {"role": "user", "content": [{"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}]}
)
def _fix_blank_text_blocks_in_list(
blocks: List[Any],
*,
placeholder_text: str,
msg_index: int,
role: Any,
location: str,
) -> List[Any]:
"""Drop blank/whitespace-only text blocks from ``blocks``, in place logic.
Non-text blocks (tool_use, tool_result, image, document, thinking, …)
and the relative order of everything else are left untouched. A
cache_control marker riding on a dropped block is relocated onto the
last surviving text/tool_use block so a breakpoint is never silently
lost. If nothing survives, a single non-blank placeholder text block
takes the dropped blocks' place (carrying the relocated cache_control,
if any) so the message never has empty content.
Returns a new list; does not mutate ``blocks``.
"""
kept: List[Any] = []
relocated_cache_control = None
for block_index, blk in enumerate(blocks):
if (
isinstance(blk, dict)
and blk.get("type") == "text"
and not (isinstance(blk.get("text"), str) and blk["text"].strip())
):
if isinstance(blk.get("cache_control"), dict):
relocated_cache_control = blk["cache_control"]
logger.warning(
"Pre-call sanitizer: dropped blank text content block "
"(message_index=%d role=%s location=%s block_index=%d "
"block_type=text)",
msg_index,
role,
location,
block_index,
)
continue
kept.append(blk)
if not kept:
placeholder: Dict[str, Any] = {"type": "text", "text": placeholder_text}
if relocated_cache_control is not None:
placeholder["cache_control"] = relocated_cache_control
kept.append(placeholder)
elif relocated_cache_control is not None:
_apply_assistant_cache_control_to_last_cacheable_block(kept, relocated_cache_control)
return kept
def _scrub_blank_text_blocks(result: List[Dict[str, Any]]) -> None:
"""Final provider-boundary guard against blank Anthropic text blocks.
Anthropic rejects any text content block whose ``text`` is empty or
whitespace-only with HTTP 400 ("text content blocks must contain
non-whitespace text"). ``_convert_assistant_message``,
``_convert_user_message`` and ``_ensure_leading_user_turn`` already
avoid emitting these for the paths that build them, but this pass runs
last — after every other transform in ``convert_messages_to_anthropic``
— so a blank block from any current or future producer (including one
nested inside a ``tool_result``'s own content list) never reaches the
wire. Diagnostics are structural only: message index, role, content
location, block index/type. Never logs message text, tool arguments,
tokens, or credentials. Mutates ``result`` in place.
"""
for msg_index, msg in enumerate(result):
if not isinstance(msg, dict):
continue
role = msg.get("role")
content = msg.get("content")
if not isinstance(content, list) or not content:
continue
placeholder_text = _EMPTY_TEXT_PLACEHOLDER if role == "assistant" else "(empty message)"
new_content = _fix_blank_text_blocks_in_list(
content,
placeholder_text=placeholder_text,
msg_index=msg_index,
role=role,
location="content",
)
for blk in new_content:
if not isinstance(blk, dict) or blk.get("type") != "tool_result":
continue
inner = blk.get("content")
if isinstance(inner, list) and inner:
blk["content"] = _fix_blank_text_blocks_in_list(
inner,
placeholder_text="(no output)",
msg_index=msg_index,
role=role,
location="tool_result",
)
msg["content"] = new_content
def convert_messages_to_anthropic(
@@ -2466,7 +2855,26 @@ def convert_messages_to_anthropic(
p.get("cache_control") for p in content if isinstance(p, dict)
)
if has_cache:
system = [p for p in content if isinstance(p, dict)]
# Copy blocks before coercing so the caller's message
# dicts are never mutated, then replace blank/whitespace
# text with the shared non-whitespace placeholder —
# Anthropic rejects a blank system text block with the
# same HTTP 400 as message blocks ("text content blocks
# must contain non-whitespace text"), and a blank block
# carrying a cache_control breakpoint cannot simply be
# dropped (#70909).
system = []
for p in content:
if not isinstance(p, dict):
continue
if (
p.get("type") == "text"
and isinstance(p.get("text"), str)
and not p["text"].strip()
):
p = dict(p)
p["text"] = _EMPTY_TEXT_PLACEHOLDER
system.append(p)
else:
system = "\n".join(
p["text"] for p in content if p.get("type") == "text"
@@ -2491,6 +2899,7 @@ def convert_messages_to_anthropic(
_ensure_leading_user_turn(result)
_manage_thinking_signatures(result, base_url, model)
_evict_old_screenshots(result)
_scrub_blank_text_blocks(result)
return system, result
@@ -2552,7 +2961,12 @@ def build_anthropic_kwargs(
)
anthropic_tools = convert_tools_to_anthropic(tools) if tools else []
model = normalize_model_name(model, preserve_dots=preserve_dots)
# Nous Portal routes on its own catalog ids (``anthropic/claude-opus-4.8``);
# normalizing to the bare Anthropic slug would make the model unresolvable
# there. Skipping the call preserves the prefix AND the dots, so
# ``preserve_dots`` stays irrelevant for Portal.
if not _is_nous_portal_endpoint(base_url):
model = normalize_model_name(model, preserve_dots=preserve_dots)
# effective_max_tokens = output cap for this call (≠ total context window)
# Use the resolver helper so non-positive values (negative ints,
# fractional floats, NaN, non-numeric) fail locally with a clear error
@@ -2676,7 +3090,15 @@ def build_anthropic_kwargs(
# request "summarized" so the reasoning blocks stay populated — matching
# 4.6 behavior and preserving the activity-feed UX during long tool runs.
if reasoning_config and isinstance(reasoning_config, dict):
if reasoning_config.get("enabled") is not False and "haiku" not in model.lower():
if reasoning_config.get("enabled") is False:
# "Thinking off". Adaptive models think by DEFAULT, so omitting the
# parameter is not a disable — it silently leaves thinking on and
# the user keeps paying for it. Send the disable explicitly.
# Mandatory-thinking models reject it with a 400, so they keep the
# omission: a silently-ignored disable beats a dead turn.
if _accepts_thinking_disable(model):
kwargs["thinking"] = {"type": "disabled"}
elif "haiku" not in model.lower():
effort = str(reasoning_config.get("effort", "medium")).lower()
budget = THINKING_BUDGET.get(effort, 8000)
if _supports_adaptive_thinking(model):
@@ -2791,6 +3213,8 @@ def create_anthropic_message(
*,
log_prefix: str = "",
prefer_stream: bool = True,
on_stream_event=None,
on_response=None,
) -> Any:
"""Create an Anthropic message, aggregating via stream when available.
@@ -2800,6 +3224,20 @@ def create_anthropic_message(
crash on ``.content``. Prefer ``messages.stream().get_final_message()`` to
match the main turn path, falling back to ``create()`` only for providers
that explicitly do not support streaming, such as restricted Bedrock roles.
``on_stream_event``: optional callable invoked once per streamed event
(best-effort, exceptions swallowed). Lets callers report forward progress
to liveness watchdogs — e.g. the auxiliary compression path ticking its
progress hook so a slow-but-generating summary model isn't treated as
hung. Only fires on the streaming path; the ``create()`` fallback has no
events to report.
``on_response``: optional callable invoked once with the underlying httpx
response before the message is aggregated (best-effort, exceptions
swallowed). Response *headers* carry out-of-band provider state that the
parsed ``Message`` drops — Nous Portal's ``x-nous-credits-*`` balance family
in particular. Only fires on the streaming path, which is the one the main
turn loop takes.
"""
sanitize_anthropic_kwargs(api_kwargs, log_prefix=log_prefix)
@@ -2810,6 +3248,26 @@ def create_anthropic_message(
stream_kwargs.pop("stream", None)
try:
with stream_fn(**stream_kwargs) as stream:
if callable(on_response):
try:
on_response(getattr(stream, "response", None))
except Exception:
logger.debug(
"%son_response callback failed",
log_prefix, exc_info=True,
)
if callable(on_stream_event):
# Consume the event stream manually so each event can
# tick the caller's progress callback; get_final_message
# then returns the accumulated snapshot.
for _event in stream:
try:
on_stream_event(_event)
except Exception:
logger.debug(
"%son_stream_event callback failed",
log_prefix, exc_info=True,
)
return stream.get_final_message()
except Exception as exc:
if not _is_stream_unavailable_error(exc):
+3177 -314
View File
File diff suppressed because it is too large Load Diff
+18 -2
View File
@@ -367,11 +367,27 @@ def describe_active_credential(config: Optional[EntraIdentityConfig] = None,
info["tenant_id_env"] = os.environ["AZURE_TENANT_ID"].strip()
# Surface which env-var sources are present without minting yet.
# Credential-bearing vars (AZURE_CLIENT_SECRET, AZURE_FEDERATED_TOKEN_FILE)
# are read through the profile secret scope so a multiplexed profile's
# diagnostics don't report another profile's env-bridged credentials;
# unscoped CLI probes keep the legacy env read (Slack pattern).
def _scoped_env(name: str) -> str:
try:
from agent.secret_scope import UnscopedSecretError, get_secret
try:
return (get_secret(name) or "").strip()
except UnscopedSecretError:
pass
except Exception:
pass
return os.environ.get(name, "").strip()
env_sources = []
if os.environ.get("AZURE_FEDERATED_TOKEN_FILE", "").strip():
if _scoped_env("AZURE_FEDERATED_TOKEN_FILE"):
env_sources.append("WorkloadIdentityCredential (AZURE_FEDERATED_TOKEN_FILE)")
if (os.environ.get("AZURE_CLIENT_ID", "").strip()
and os.environ.get("AZURE_CLIENT_SECRET", "").strip()
and _scoped_env("AZURE_CLIENT_SECRET")
and os.environ.get("AZURE_TENANT_ID", "").strip()):
env_sources.append("EnvironmentCredential (client secret)")
if os.environ.get("IDENTITY_ENDPOINT", "").strip() or os.environ.get("MSI_ENDPOINT", "").strip():
+204
View File
@@ -0,0 +1,204 @@
"""Single owner for backend identity and failure-scoped skip decisions.
Every fallback / dedup / skip / quarantine decision in Hermes ultimately asks
one question: **"is this candidate the same backend as the one that failed,
along the axis that failure invalidated?"** Before this module, that
question was re-implemented inline at six call sites across four subsystems,
each comparing whatever string was locally convenient (provider label,
provider+model, base_url+model, ...). Each incident fixed one site while the
others kept the bug: #22548 (same-shim aliases), #70893 (xai-oauth vs xai —
same host, distinct credential), #59561 (aux chain skipped sibling models),
#72468 (aux main-model safety net, same bug three weeks later), #62984 /
#54250 / #57584 (dedup ignoring base_url strands multi-endpoint pools).
The root insight: "provider" conflates three independent identity axes, and
each failure class invalidates a different one:
* **credential surface** — auth 401 / payment 402 kill everything sharing the
credential (every model, every host reached with that key/token).
* **endpoint** — DNS failure / connection refused kill everything behind the
URL, regardless of model or credential.
* **model deployment** — timeout / overload / rate limit / model-incompatible
kill ONE model's deployment. A sibling model behind the same URL is an
independent deployment (real incident: aux ``glm-5.2`` hung and timed out
while main ``macaron-v1-venti`` on the identical endpoint was serving
448K-token turns).
Call sites should build :class:`BackendIdentity` values, classify the failure
with :func:`classify_failure_scope`, and ask :func:`should_skip_candidate`.
Do not re-implement any comparison inline — extend THIS module instead.
"""
from __future__ import annotations
import logging
from dataclasses import dataclass
from enum import Enum
from typing import Optional
logger = logging.getLogger(__name__)
class FailureScope(Enum):
"""Which identity axis a failure invalidates."""
#: Timeout, overload/429, connection blip, model-incompatible, invalid
#: response: evidence against ONE model deployment only.
MODEL = "model"
#: Auth 401 / payment 402: evidence against the shared credential —
#: every model reached with it is equally dead.
CREDENTIAL = "credential"
#: DNS / connection-refused / unreachable host: evidence against the
#: endpoint — every model behind the URL is equally dead.
ENDPOINT = "endpoint"
#: Reason strings already used by auxiliary_client's except-chain, mapped to
#: scopes. Unknown reasons default to MODEL — the least-invalidating scope —
#: so an unrecognized failure never over-skips viable candidates.
_REASON_SCOPES = {
"auth error": FailureScope.CREDENTIAL,
"payment error": FailureScope.CREDENTIAL,
"rate limit": FailureScope.MODEL,
"model incompatible with route": FailureScope.MODEL,
"invalid provider response": FailureScope.MODEL,
"connection error": FailureScope.MODEL,
"timeout": FailureScope.MODEL,
}
def classify_failure_scope(reason: Optional[str]) -> FailureScope:
"""Map a human-readable failure reason to the identity axis it kills."""
return _REASON_SCOPES.get((reason or "").strip().lower(), FailureScope.MODEL)
def _norm_provider(value: Optional[str]) -> str:
return (value or "").strip().lower()
def _norm_model(value: Optional[str]) -> str:
return (value or "").strip().lower()
def _norm_base_url(value: Optional[str]) -> str:
return (value or "").strip().rstrip("/").lower()
@dataclass(frozen=True)
class BackendIdentity:
"""Normalized identity of one (provider, model, endpoint) deployment.
Empty fields mean "unknown" — comparisons treat an unknown axis as
non-distinguishing (it can neither prove sameness nor difference on its
own; the remaining axes decide).
"""
provider: str = ""
model: str = ""
base_url: str = ""
@classmethod
def build(
cls,
provider: Optional[str] = None,
model: Optional[str] = None,
base_url: Optional[str] = None,
) -> "BackendIdentity":
return cls(
provider=_norm_provider(provider),
model=_norm_model(model),
base_url=_norm_base_url(base_url),
)
def _both_first_class(a: BackendIdentity, b: BackendIdentity) -> bool:
"""True when both providers are distinct registered first-class providers.
Two different registry providers have distinct credential surfaces even
when they share an inference host (xai-oauth vs xai, openai-codex vs
openai-api) — #70893. Custom/shim aliases are NOT in the registry, so
two aliases pointing at one URL still count as the same backend (#22548).
"""
if not a.provider or not b.provider or a.provider == b.provider:
return False
try:
from hermes_cli.auth import PROVIDER_REGISTRY
return a.provider in PROVIDER_REGISTRY and b.provider in PROVIDER_REGISTRY
except Exception:
return False
def same_credential_surface(a: BackendIdentity, b: BackendIdentity) -> bool:
"""Do two identities share the credential a 401/402 just invalidated?
Conservative on purpose: an unprovable axis must answer "different"
(try the candidate — worst case one wasted RTT) rather than "same"
(skip — worst case stranded failover). Two distinct custom labels at
one URL may carry different per-entry api_keys, so a shared URL alone
never proves a shared credential; it is only used as a weak signal
when a provider label is missing entirely.
"""
if a.provider and b.provider:
# Same label = same configured credential. Different labels =
# different credential config (first-class registry providers
# explicitly so — #70893; custom entries can each carry their own
# api_key, so sameness is unprovable and we must not skip).
return a.provider == b.provider
# Provider unknown on a side: same explicit URL is the best signal left.
return bool(a.base_url and a.base_url == b.base_url)
def same_endpoint(a: BackendIdentity, b: BackendIdentity) -> bool:
"""Do two identities sit behind the endpoint that just went unreachable?"""
if a.base_url and b.base_url:
return a.base_url == b.base_url
# An unknown base_url inherits the provider default → same provider
# label implies the same default endpoint.
return bool(a.provider and a.provider == b.provider)
def same_deployment(a: BackendIdentity, b: BackendIdentity) -> bool:
"""Are these the exact same model deployment (the thing a timeout kills)?
Provider+model must match; the base_url axis distinguishes only when BOTH
sides carry an explicit URL (#62984: same provider+model on two different
explicit URLs is two deployments — a pool). A side with an unknown URL
inherits the provider default and cannot prove difference.
"""
if not (a.provider and b.provider and a.provider == b.provider):
# Same-host different-label shims: same URL + same model IS the same
# deployment even when the alias labels differ (#22548) — unless both
# labels are first-class registry providers (#70893).
if (
a.base_url
and a.base_url == b.base_url
and a.model
and a.model == b.model
and not _both_first_class(a, b)
):
return True
return False
if not (a.model and b.model and a.model == b.model):
return False
if a.base_url and b.base_url and a.base_url != b.base_url:
return False # distinct explicit endpoints — a pool, not a dup
return True
def should_skip_candidate(
candidate: BackendIdentity,
failed: BackendIdentity,
scope: FailureScope = FailureScope.MODEL,
) -> bool:
"""THE skip predicate: would trying ``candidate`` just repeat the failure?
True when the candidate is the same backend as ``failed`` along the axis
``scope`` says the failure invalidated. Every fallback/dedup/skip site
must call this instead of comparing labels inline.
"""
if scope is FailureScope.CREDENTIAL:
return same_credential_surface(candidate, failed)
if scope is FailureScope.ENDPOINT:
return same_endpoint(candidate, failed)
return same_deployment(candidate, failed)
File diff suppressed because it is too large Load Diff
+276 -19
View File
@@ -33,6 +33,9 @@ import os
import re
from types import SimpleNamespace
from typing import Any, Dict, List, Optional, Tuple
from urllib.parse import urlparse
import httpx
logger = logging.getLogger(__name__)
@@ -57,6 +60,25 @@ except Exception:
_bedrock_runtime_client_cache: Dict[str, Any] = {}
_bedrock_control_client_cache: Dict[str, Any] = {}
# Bedrock-hosted OpenAI GPT-5.5 is not exposed through the native Converse
# runtime. AWS serves it from the Bedrock Mantle OpenAI-compatible Responses
# endpoint instead (https://bedrock-mantle.<region>.api.aws/openai/v1).
# Keep the allowlist intentionally narrow so OpenAI GPT-OSS models that are
# Converse-capable continue to use the native Bedrock path.
BEDROCK_OPENAI_RESPONSES_MODEL_IDS: Tuple[str, ...] = (
"openai.gpt-5.5",
# GPT-5.6 family (GA on Bedrock 2026-07-13): Sol (frontier), Terra
# (balanced), Luna (fast/affordable). All are Mantle-only — the model
# cards list bedrock-runtime/Converse as unsupported.
# https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html
"openai.gpt-5.6-sol",
"openai.gpt-5.6-terra",
"openai.gpt-5.6-luna",
)
_BEDROCK_OPENAI_HOST_RE = re.compile(
r"^bedrock-mantle\.([a-z0-9-]+)\.api\.aws$", re.IGNORECASE
)
_MIN_BOTO3_VERSION = (1, 34, 59)
@@ -133,6 +155,143 @@ def invalidate_runtime_client(region: str) -> bool:
return existed
# ---------------------------------------------------------------------------
# Bedrock Mantle / OpenAI Responses support
# ---------------------------------------------------------------------------
def is_openai_bedrock_model(model_id: str) -> bool:
"""Return True for Bedrock-hosted OpenAI models that require Mantle.
Bedrock's GPT-OSS models are Converse-capable and intentionally do not
match this helper. The allowlist tracks models served by the OpenAI
Responses-compatible ``bedrock-mantle`` route.
"""
normalized = str(model_id or "").strip().lower()
return normalized in {m.lower() for m in BEDROCK_OPENAI_RESPONSES_MODEL_IDS}
def merge_bedrock_openai_model_ids(model_ids: List[str]) -> List[str]:
"""Append Bedrock OpenAI Responses models to a discovered Bedrock list.
The Bedrock control plane's ListFoundationModels/ListInferenceProfiles
discovery covers Converse models but does not enumerate Mantle-only
OpenAI Responses models. The picker needs both surfaces under AWS Bedrock.
"""
merged = list(model_ids or [])
seen = {str(m).lower() for m in merged}
for model_id in BEDROCK_OPENAI_RESPONSES_MODEL_IDS:
if model_id.lower() not in seen:
merged.append(model_id)
seen.add(model_id.lower())
return merged
def bedrock_openai_base_url(region: str) -> str:
"""Return Bedrock Mantle's OpenAI-compatible base URL for *region*."""
resolved = (region or "").strip() or resolve_bedrock_runtime_region()
return f"https://bedrock-mantle.{resolved}.api.aws/openai/v1"
def bedrock_openai_region_from_base_url(base_url: str) -> Optional[str]:
"""Extract the AWS region from a Bedrock Mantle OpenAI base URL."""
host = urlparse(str(base_url or "")).hostname or ""
match = _BEDROCK_OPENAI_HOST_RE.match(host)
return match.group(1) if match else None
def is_bedrock_openai_base_url(base_url: str) -> bool:
"""Return True for Bedrock Mantle OpenAI-compatible endpoints."""
parsed = urlparse(str(base_url or ""))
host = parsed.hostname or ""
if not _BEDROCK_OPENAI_HOST_RE.match(host):
return False
# The OpenAI GPT-5.5 Bedrock route lives under /openai/v1. Accept a bare
# host too so callers can normalize before appending the path.
path = (parsed.path or "").rstrip("/").lower()
return path in {"", "/openai", "/openai/v1"}
def resolve_bedrock_bearer_token(env: Optional[Dict[str, str]] = None) -> str:
"""Return AWS_BEARER_TOKEN_BEDROCK when Bedrock API-key auth is configured."""
env = env if env is not None else os.environ
return (env.get("AWS_BEARER_TOKEN_BEDROCK", "") or "").strip()
class BedrockOpenAISigV4Auth(httpx.Auth):
"""httpx auth hook that SigV4-signs Bedrock Mantle OpenAI requests."""
requires_request_body = True
def __init__(self, region: str, service: str = "bedrock"):
self.region = (region or "").strip() or resolve_bedrock_runtime_region()
self.service = service
def auth_flow(self, request): # pragma: no cover - exercised by live call
import botocore.session
from botocore.auth import SigV4Auth
from botocore.awsrequest import AWSRequest
credentials = botocore.session.get_session().get_credentials()
if credentials is None:
raise RuntimeError(
"No AWS credentials available for Bedrock OpenAI Responses. "
"Configure AWS_ACCESS_KEY_ID/AWS_SECRET_ACCESS_KEY, AWS_PROFILE, "
"SSO, or an instance/task role."
)
frozen = credentials.get_frozen_credentials()
# Drop the OpenAI SDK's placeholder bearer header before signing; SigV4
# must own Authorization. Keep all other SDK headers so AWS receives
# content-type, accept, request IDs, etc.
headers = {
str(k): str(v)
for k, v in request.headers.items()
if str(k).lower() not in {"authorization", "x-amz-date", "x-amz-security-token"}
}
aws_request = AWSRequest(
method=request.method,
url=str(request.url),
data=request.content or b"",
headers=headers,
)
SigV4Auth(frozen, self.service, self.region).add_auth(aws_request)
request.headers.update(dict(aws_request.headers.items()))
yield request
def build_bedrock_openai_http_client(region: str, *, timeout: Optional[float] = None):
"""Build an httpx client that SigV4-signs Bedrock OpenAI requests."""
import httpx
kwargs: Dict[str, Any] = {"auth": BedrockOpenAISigV4Auth(region)}
if isinstance(timeout, (int, float)) and not isinstance(timeout, bool) and timeout > 0:
kwargs["timeout"] = timeout
return httpx.Client(**kwargs)
def configure_bedrock_openai_client_kwargs(
client_kwargs: Dict[str, Any],
*,
timeout: Optional[float] = None,
) -> Dict[str, Any]:
"""Install SigV4 auth on OpenAI SDK kwargs for Bedrock Mantle.
``AWS_BEARER_TOKEN_BEDROCK``/explicit Bedrock API keys continue to use the
SDK's normal bearer auth. The special ``aws-sdk`` placeholder means IAM
credential-chain auth, so we attach a per-request SigV4 httpx client.
"""
base_url = str(client_kwargs.get("base_url") or "")
if not is_bedrock_openai_base_url(base_url):
return client_kwargs
api_key = client_kwargs.get("api_key")
if isinstance(api_key, str) and api_key.strip() and api_key not in {"aws-sdk", "no-key-required"}:
return client_kwargs
region = bedrock_openai_region_from_base_url(base_url) or resolve_bedrock_runtime_region()
client_kwargs["api_key"] = "aws-sdk"
client_kwargs["http_client"] = build_bedrock_openai_http_client(region, timeout=timeout)
return client_kwargs
# ---------------------------------------------------------------------------
# Stale-connection detection
# ---------------------------------------------------------------------------
@@ -384,6 +543,36 @@ def resolve_bedrock_region(env: Optional[Dict[str, str]] = None) -> str:
return "us-east-1"
def resolve_bedrock_runtime_region(config: Optional[Dict[str, Any]] = None) -> str:
"""Resolve the Bedrock region with the same priority as the main runtime.
Priority (matches the runtime provider resolver in
``hermes_cli/runtime_provider.py``):
1. ``bedrock.region`` in config.yaml
2. ``resolve_bedrock_region()`` (AWS_REGION / AWS_DEFAULT_REGION /
botocore profile / us-east-1)
Callers that already hold a loaded config dict should pass it to avoid a
disk read; when *config* is None the config is loaded read-only. Every
non-runtime call site that constructs a Bedrock endpoint (auxiliary
client resolution, model discovery for the picker) must use this helper —
using bare ``resolve_bedrock_region()`` there lets auxiliary calls leave
the primary runtime's configured region when ``bedrock.region`` and the
ambient AWS env/profile disagree.
"""
if config is None:
try:
from hermes_cli.config import load_config_readonly
config = load_config_readonly()
except Exception:
config = {}
bedrock_cfg = (config or {}).get("bedrock") or {}
cfg_region = str(bedrock_cfg.get("region") or "").strip()
if cfg_region:
return cfg_region
return resolve_bedrock_region()
def bedrock_model_ids_or_none() -> Optional[List[str]]:
"""Live-discover Bedrock model IDs for the active region.
@@ -396,9 +585,9 @@ def bedrock_model_ids_or_none() -> Optional[List[str]]:
``list_authenticated_providers`` section 2, and section 3.
"""
try:
discovered = discover_bedrock_models(resolve_bedrock_region())
discovered = discover_bedrock_models(resolve_bedrock_runtime_region())
if discovered:
return [m["id"] for m in discovered]
return merge_bedrock_openai_model_ids([m["id"] for m in discovered])
except Exception:
pass
return None
@@ -433,6 +622,29 @@ def _model_supports_tool_use(model_id: str) -> bool:
return not any(pattern in model_lower for pattern in _NON_TOOL_CALLING_PATTERNS)
# ---------------------------------------------------------------------------
# Prompt-cache capability detection (Converse API cachePoint)
# ---------------------------------------------------------------------------
# Claude on Bedrock already gets prompt caching through the AnthropicBedrock
# SDK path (see is_anthropic_bedrock_model / runtime_provider.py's dual-path
# routing) — it never reaches build_converse_kwargs unless bearer-token auth
# forces the Converse path (#28156). This allowlist covers the Converse API
# itself: sending an unsupported model a cachePoint block raises a
# ValidationException, so — like _model_supports_tool_use but inverted —
# unknown models default to NOT receiving cache markers until confirmed.
# Ref: https://docs.aws.amazon.com/bedrock/latest/userguide/prompt-caching.html
_CACHE_POINT_PATTERNS = [
"anthropic.claude", # bearer-token fallback path
"amazon.nova",
]
def _model_supports_prompt_cache(model_id: str) -> bool:
"""Return True if the model accepts a Converse API cachePoint block."""
model_lower = model_id.lower()
return any(pattern in model_lower for pattern in _CACHE_POINT_PATTERNS)
def is_anthropic_bedrock_model(model_id: str) -> bool:
"""Return True if the model is an Anthropic Claude model on Bedrock.
@@ -764,14 +976,22 @@ def normalize_converse_response(response: Dict) -> SimpleNamespace:
reasoning_content="\n\n".join(reasoning_parts) if reasoning_parts else None,
)
# Build usage stats
# Build usage stats. Converse's inputTokens excludes cache read/write
# tokens (unlike OpenAI's prompt_tokens, which includes them) — restore
# the OpenAI-style "total includes cache" convention here so downstream
# normalize_usage() can subtract them back out consistently, and surface
# the Anthropic-named fields it already falls back to for cache reads.
usage_data = response.get("usage", {})
input_tokens = usage_data.get("inputTokens", 0)
cache_read_tokens = usage_data.get("cacheReadInputTokens", 0)
cache_write_tokens = usage_data.get("cacheWriteInputTokens", 0)
output_tokens = usage_data.get("outputTokens", 0)
usage = SimpleNamespace(
prompt_tokens=usage_data.get("inputTokens", 0),
completion_tokens=usage_data.get("outputTokens", 0),
total_tokens=(
usage_data.get("inputTokens", 0) + usage_data.get("outputTokens", 0)
),
prompt_tokens=input_tokens + cache_read_tokens + cache_write_tokens,
completion_tokens=output_tokens,
total_tokens=input_tokens + cache_read_tokens + cache_write_tokens + output_tokens,
cache_read_input_tokens=cache_read_tokens,
cache_creation_input_tokens=cache_write_tokens,
)
finish_reason = _converse_stop_reason_to_openai(stop_reason)
@@ -936,6 +1156,8 @@ def stream_converse_with_callbacks(
usage_data = {
"inputTokens": meta_usage.get("inputTokens", 0),
"outputTokens": meta_usage.get("outputTokens", 0),
"cacheReadInputTokens": meta_usage.get("cacheReadInputTokens", 0),
"cacheWriteInputTokens": meta_usage.get("cacheWriteInputTokens", 0),
}
# Flush remaining text
@@ -949,12 +1171,16 @@ def stream_converse_with_callbacks(
reasoning_content="\n\n".join(reasoning_parts) if reasoning_parts else None,
)
input_tokens = usage_data.get("inputTokens", 0)
cache_read_tokens = usage_data.get("cacheReadInputTokens", 0)
cache_write_tokens = usage_data.get("cacheWriteInputTokens", 0)
output_tokens = usage_data.get("outputTokens", 0)
usage = SimpleNamespace(
prompt_tokens=usage_data.get("inputTokens", 0),
completion_tokens=usage_data.get("outputTokens", 0),
total_tokens=(
usage_data.get("inputTokens", 0) + usage_data.get("outputTokens", 0)
),
prompt_tokens=input_tokens + cache_read_tokens + cache_write_tokens,
completion_tokens=output_tokens,
total_tokens=input_tokens + cache_read_tokens + cache_write_tokens + output_tokens,
cache_read_input_tokens=cache_read_tokens,
cache_creation_input_tokens=cache_write_tokens,
)
finish_reason = _converse_stop_reason_to_openai(stop_reason)
@@ -982,7 +1208,7 @@ def build_converse_kwargs(
model: str,
messages: List[Dict],
tools: Optional[List[Dict]] = None,
max_tokens: int = 4096,
max_tokens: Optional[int] = 4096,
temperature: Optional[float] = None,
top_p: Optional[float] = None,
stop_sequences: Optional[List[str]] = None,
@@ -991,18 +1217,29 @@ def build_converse_kwargs(
"""Build kwargs for ``bedrock-runtime.converse()`` or ``converse_stream()``.
Converts OpenAI-format inputs to Converse API parameters.
``max_tokens=None`` omits ``inferenceConfig.maxTokens`` entirely, in which
case Bedrock defaults to the model's maximum allowed output — the Converse
field is optional per the AWS API reference. The default stays 4096 so
existing callers are unaffected; callers that want the model's full output
budget (e.g. uncapped auxiliary vision calls) pass ``None`` explicitly.
"""
system_prompt, converse_messages = convert_messages_to_converse(messages)
cache_enabled = _model_supports_prompt_cache(model)
inference_config: Dict[str, Any] = {}
if max_tokens is not None:
inference_config["maxTokens"] = max_tokens
kwargs: Dict[str, Any] = {
"modelId": model,
"messages": converse_messages,
"inferenceConfig": {
"maxTokens": max_tokens,
},
"inferenceConfig": inference_config,
}
if system_prompt:
if cache_enabled:
system_prompt = system_prompt + [{"cachePoint": {"type": "default"}}]
kwargs["system"] = system_prompt
from agent.anthropic_adapter import _forbids_sampling_params
@@ -1026,6 +1263,8 @@ def build_converse_kwargs(
# Strip tools for known non-tool-calling models and warn the user.
# Ref: PR #7920 feedback from @ptlally, pattern from PR #4346.
if _model_supports_tool_use(model):
if cache_enabled:
converse_tools = converse_tools + [{"cachePoint": {"type": "default"}}]
kwargs["toolConfig"] = {"tools": converse_tools}
else:
logger.warning(
@@ -1033,9 +1272,21 @@ def build_converse_kwargs(
"The agent will operate in text-only mode.", model
)
if cache_enabled and len(converse_messages) >= 2:
# Checkpoint everything up to (not including) the newest turn, so the
# marker survives unchanged across requests as only the tail grows —
# mirroring the Anthropic system_and_3 strategy in prompt_caching.py.
content = converse_messages[-2].get("content")
if isinstance(content, list) and content:
content.append({"cachePoint": {"type": "default"}})
if guardrail_config:
kwargs["guardrailConfig"] = guardrail_config
if not kwargs["inferenceConfig"]:
# inferenceConfig is optional on the wire; don't send an empty object.
del kwargs["inferenceConfig"]
return kwargs
@@ -1044,7 +1295,7 @@ def call_converse(
model: str,
messages: List[Dict],
tools: Optional[List[Dict]] = None,
max_tokens: int = 4096,
max_tokens: Optional[int] = 4096,
temperature: Optional[float] = None,
top_p: Optional[float] = None,
stop_sequences: Optional[List[str]] = None,
@@ -1085,7 +1336,7 @@ def call_converse_stream(
model: str,
messages: List[Dict],
tools: Optional[List[Dict]] = None,
max_tokens: int = 4096,
max_tokens: Optional[int] = 4096,
temperature: Optional[float] = None,
top_p: Optional[float] = None,
stop_sequences: Optional[List[str]] = None,
@@ -1388,6 +1639,12 @@ BEDROCK_CONTEXT_LENGTHS: Dict[str, int] = {
"mistral.mistral-large": 128_000,
# DeepSeek
"deepseek.v3": 128_000,
# OpenAI on Bedrock (Mantle/Responses route)
# https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html
"openai.gpt-5.5": 272_000,
"openai.gpt-5.6-sol": 272_000,
"openai.gpt-5.6-terra": 272_000,
"openai.gpt-5.6-luna": 272_000,
}
# Default for unknown Bedrock models
+124
View File
@@ -0,0 +1,124 @@
"""Provider-agnostic billing/credit recovery links.
Maps a billing-classified failure onto a recovery link + label. *Detection*
is not done here — that is :mod:`agent.error_classifier`
(``FailoverReason.billing``), the single source of truth for "credit wall vs.
rate limit / auth / transport". The resulting :class:`BillingBlock` rides the
turn result and the gateway ``message.complete`` event so every surface (CLI,
TUI, desktop) renders one structured signal instead of re-parsing error text.
"""
from __future__ import annotations
from dataclasses import asdict, dataclass
from typing import Optional
from utils import base_url_host_matches
@dataclass
class BillingBlock:
"""Structured billing-wall descriptor shared across every surface.
``is_nous`` is the routing bit: Nous has a first-class in-app billing surface
(desktop Settings → Billing, TUI/CLI ``/topup``), so surfaces prefer that over
``billing_url``; third-party providers have no in-app flow, so ``billing_url``
is the deep link the user actually needs.
"""
provider: str
provider_label: str
model: str
billing_url: Optional[str]
is_nous: bool
message: str
def to_dict(self) -> dict:
return asdict(self)
@dataclass(frozen=True)
class _Provider:
label: str
url: str
slugs: tuple[str, ...]
hosts: tuple[str, ...] = ()
# Single source of truth: internal slug(s) + base_url host(s) → billing page.
# Curated "add credits / manage billing" landing pages, not marketing homes.
# Hosts back the OpenAI-compatible fallback where the slug is a generic bucket
# (e.g. "openai_compatible") but base_url reveals the real upstream. An unknown
# provider degrades to a readable label with no invented URL.
_PROVIDERS: tuple[_Provider, ...] = (
_Provider("OpenAI", "https://platform.openai.com/settings/organization/billing", ("openai",), ("api.openai.com",)),
_Provider("Anthropic", "https://console.anthropic.com/settings/billing", ("anthropic",), ("api.anthropic.com",)),
_Provider("OpenRouter", "https://openrouter.ai/settings/credits", ("openrouter",), ("openrouter.ai",)),
_Provider("xAI", "https://console.x.ai/team/default/billing", ("xai", "xai-oauth"), ("api.x.ai",)),
_Provider("DeepSeek", "https://platform.deepseek.com/top_up", ("deepseek",), ("api.deepseek.com",)),
_Provider("Groq", "https://console.groq.com/settings/billing", ("groq",), ("api.groq.com",)),
_Provider("Mistral", "https://console.mistral.ai/billing", ("mistral",), ("api.mistral.ai",)),
_Provider("Together AI", "https://api.together.ai/settings/billing", ("together",), ("api.together.ai", "api.together.xyz")),
_Provider("Fireworks AI", "https://fireworks.ai/account/billing", ("fireworks",), ("fireworks.ai",)),
_Provider("Perplexity", "https://www.perplexity.ai/settings/api", ("perplexity",), ("perplexity.ai",)),
_Provider("Google AI", "https://aistudio.google.com/app/billing", ("google", "gemini"), ("generativelanguage.googleapis.com",)),
_Provider("Cohere", "https://dashboard.cohere.com/billing", ("cohere",)),
_Provider("Moonshot AI", "https://platform.moonshot.ai/console/pay", ("moonshot",)),
_Provider("NVIDIA", "https://build.nvidia.com/settings/billing", ("nvidia",)),
)
_BY_SLUG: dict[str, _Provider] = {slug: p for p in _PROVIDERS for slug in p.slugs}
def is_nous_inference_route(provider: str, base_url: str) -> bool:
"""True when the failing route is the Nous-managed inference gateway."""
if (provider or "").strip().lower() == "nous":
return True
return base_url_host_matches(str(base_url or ""), "inference-api.nousresearch.com")
def _nous_billing_url() -> Optional[str]:
"""Best-effort Nous portal billing URL (text-surface fallback; Nous prefers the in-app flow)."""
try:
from hermes_cli.nous_account import nous_portal_billing_url
return nous_portal_billing_url(None)
except Exception:
return "https://portal.nousresearch.com/billing"
def _resolve_provider_link(slug: str, base_url: str) -> tuple[str, Optional[str]]:
"""Resolve ``(label, url)``: exact slug → base_url host → readable-label fallback."""
hit = _BY_SLUG.get(slug)
if hit:
return hit.label, hit.url
base = str(base_url or "")
for p in _PROVIDERS:
if any(base_url_host_matches(base, host) for host in p.hosts):
return p.label, p.url
return slug.replace("_", " ").replace("-", " ").strip().title() or "your provider", None
def build_billing_block(
*,
provider: str,
base_url: str,
model: str,
message: str = "",
) -> BillingBlock:
"""Build the billing descriptor for a billing-classified failure.
``message`` is the guidance already assembled by the agent loop
(:func:`agent.conversation_loop._billing_or_entitlement_message`), carried
through unchanged so every surface shows identical copy.
"""
slug = (provider or "").strip().lower()
model = (model or "").strip()
if is_nous_inference_route(slug, base_url):
return BillingBlock(slug or "nous", "Nous Portal", model, _nous_billing_url(), True, message or "")
label, url = _resolve_provider_link(slug, base_url)
return BillingBlock(slug, label, model, url, False, message or "")
+1 -1
View File
@@ -34,7 +34,7 @@ from __future__ import annotations
import logging
import math
import os
from dataclasses import dataclass, field
from dataclasses import dataclass
from typing import Any, Optional
logger = logging.getLogger(__name__)
+54 -1
View File
@@ -17,7 +17,7 @@ from __future__ import annotations
import logging
import os
import uuid
from dataclasses import dataclass, field
from dataclasses import dataclass
from decimal import Decimal, InvalidOperation
from typing import Any, Optional
@@ -107,6 +107,22 @@ class CardInfo:
return f"{self.masked} — {label}" if label else self.masked
@dataclass(frozen=True)
class PaymentMethodInfo:
"""The payment method on file. `kind` is "card", "link", or "unknown"
— anything else is normalised to "unknown" at parse time, so consumers
only ever see fields that belong to the kind they are looking at."""
kind: str
brand: Optional[str] = None
last4: Optional[str] = None
wallet: Optional[str] = None
email: Optional[str] = None
resolved_via: Optional[str] = None
#: What the server called it, when we did not recognise the kind.
raw_kind: Optional[str] = None
@dataclass(frozen=True)
class MonthlyCap:
limit_usd: Optional[Decimal] = None
@@ -150,6 +166,7 @@ class BillingState:
min_usd: Optional[Decimal] = None
max_usd: Optional[Decimal] = None
card: Optional[CardInfo] = None
payment_method: Optional[PaymentMethodInfo] = None
monthly_cap: Optional[MonthlyCap] = None
auto_reload: Optional[AutoReload] = None
portal_url: Optional[str] = None
@@ -201,6 +218,41 @@ def _parse_card(raw: Any) -> Optional[CardInfo]:
return CardInfo(brand=brand, last4=last4, resolved_via=resolved_via)
def _parse_payment_method(raw: Any) -> Optional[PaymentMethodInfo]:
if not isinstance(raw, dict):
return None
kind = raw.get("kind")
if not isinstance(kind, str):
return None
def _optional_string(key: str) -> Optional[str]:
value = raw.get(key)
return value if isinstance(value, str) else None
resolved_via = _optional_string("resolvedVia")
brand = _optional_string("brand")
last4 = _optional_string("last4")
# Settle the kind here, the way _parse_card settles a card, so nothing
# downstream has to re-check which fields this kind is allowed to have.
if kind == "card" and brand and last4:
return PaymentMethodInfo(
kind="card",
brand=brand,
last4=last4,
wallet=_optional_string("wallet"),
resolved_via=resolved_via,
)
if kind == "link":
return PaymentMethodInfo(
kind="link",
email=_optional_string("email"),
resolved_via=resolved_via,
)
return PaymentMethodInfo(
kind="unknown", raw_kind=kind, resolved_via=resolved_via
)
def _parse_monthly_cap(raw: Any) -> Optional[MonthlyCap]:
if not isinstance(raw, dict):
return None
@@ -274,6 +326,7 @@ def billing_state_from_payload(
min_usd=parse_money(bounds.get("minUsd")),
max_usd=parse_money(bounds.get("maxUsd")),
card=_parse_card(payload.get("card")),
payment_method=_parse_payment_method(payload.get("paymentMethod")),
monthly_cap=_parse_monthly_cap(payload.get("monthlyCap")),
auto_reload=_parse_auto_reload(payload.get("autoReload")),
portal_url=portal_url,
+4 -2
View File
@@ -26,6 +26,7 @@ Session metadata contract (preserved from the legacy ``CloudBrowserProvider``)::
"session_name": str, # unique name for agent-browser --session
"bb_session_id": str, # provider session ID (for close/cleanup)
"cdp_url": str, # CDP websocket URL
"expires_at": str, # optional provider-authoritative ISO timestamp
"features": dict, # feature flags that were enabled
"external_call_id": str, # optional, managed-gateway billing key
}
@@ -38,7 +39,7 @@ which provider is in use.
from __future__ import annotations
import abc
from typing import Any, Dict
from typing import Any, Dict, Optional
# ---------------------------------------------------------------------------
@@ -96,6 +97,7 @@ class BrowserProvider(abc.ABC):
"session_name": str, # unique name for agent-browser --session
"bb_session_id": str, # provider session ID (for close/cleanup)
"cdp_url": str, # CDP websocket URL
"expires_at": str, # optional provider-authoritative ISO timestamp
"features": dict, # feature flags that were enabled
}
@@ -124,7 +126,7 @@ class BrowserProvider(abc.ABC):
credentials, network errors, etc. — log and move on. Must not raise.
"""
def get_setup_schema(self) -> Dict[str, Any]:
def get_setup_schema(self) -> Optional[Dict[str, Any]]:
"""Return provider metadata for the ``hermes tools`` picker.
Used by :mod:`hermes_cli.tools_config` to inject this provider as a
+70 -9
View File
@@ -41,15 +41,19 @@ import threading
from typing import Dict, List, Optional
from agent.browser_provider import BrowserProvider
from hermes_constants import hermes_home_key
logger = logging.getLogger(__name__)
_providers: Dict[str, BrowserProvider] = {}
_scoped_providers: Dict[str, Dict[str, BrowserProvider]] = {}
_generation = 0
_scoped_generations: Dict[str, int] = {}
_lock = threading.Lock()
def register_provider(provider: BrowserProvider) -> None:
def register_provider(provider: BrowserProvider, *, scope: Optional[str] = None) -> None:
"""Register a cloud browser provider.
Re-registration (same ``name``) overwrites the previous entry and logs
@@ -61,12 +65,19 @@ def register_provider(provider: BrowserProvider) -> None:
f"register_provider() expects a BrowserProvider instance, "
f"got {type(provider).__name__}"
)
name = provider.name
if not isinstance(name, str) or not name.strip():
raw_name = provider.name
if not isinstance(raw_name, str) or not raw_name.strip():
raise ValueError("Browser provider .name must be a non-empty string")
name = raw_name.strip()
global _generation
with _lock:
existing = _providers.get(name)
_providers[name] = provider
target = _providers if scope is None else _scoped_providers.setdefault(scope, {})
existing = target.get(name)
target[name] = provider
if scope is None:
_generation += 1
else:
_scoped_generations[scope] = _scoped_generations.get(scope, 0) + 1
if existing is not None:
logger.debug(
"Browser provider '%s' re-registered (was %r)",
@@ -79,19 +90,64 @@ def register_provider(provider: BrowserProvider) -> None:
)
def list_providers() -> List[BrowserProvider]:
def list_providers(*, scope: Optional[str] = None) -> List[BrowserProvider]:
"""Return all registered providers, sorted by name."""
with _lock:
items = list(_providers.values())
merged = dict(_providers)
merged.update(_scoped_providers.get(scope or hermes_home_key(), {}))
items = list(merged.values())
return sorted(items, key=lambda p: p.name)
def get_provider(name: str) -> Optional[BrowserProvider]:
def get_provider(name: str, *, scope: Optional[str] = None) -> Optional[BrowserProvider]:
"""Return the provider registered under *name*, or None."""
if not isinstance(name, str):
return None
with _lock:
return _providers.get(name.strip())
key = name.strip()
return _scoped_providers.get(scope or hermes_home_key(), {}).get(key) or _providers.get(key)
def snapshot_registration(
name: str, *, scope: Optional[str] = None
) -> Optional[BrowserProvider]:
with _lock:
target = _providers if scope is None else _scoped_providers.get(scope, {})
return target.get(name.strip())
def registry_generation(*, scope: Optional[str] = None) -> tuple[int, int]:
"""Return a cache fingerprint for the global base and one profile."""
active_scope = scope or hermes_home_key()
with _lock:
return _generation, _scoped_generations.get(active_scope, 0)
def restore_registration(
name: str,
current: BrowserProvider,
previous: Optional[BrowserProvider],
*,
scope: Optional[str] = None,
) -> bool:
"""Restore a plugin registration only when *current* is still installed."""
key = name.strip()
global _generation
with _lock:
target = _providers if scope is None else _scoped_providers.setdefault(scope, {})
if target.get(key) is not current:
return False
if previous is None:
target.pop(key, None)
else:
target[key] = previous
if scope is None:
_generation += 1
else:
_scoped_generations[scope] = _scoped_generations.get(scope, 0) + 1
if not target:
_scoped_providers.pop(scope, None)
return True
# ---------------------------------------------------------------------------
@@ -145,6 +201,7 @@ def _resolve(configured: Optional[str]) -> Optional[BrowserProvider]:
"""
with _lock:
snapshot = dict(_providers)
snapshot.update(_scoped_providers.get(hermes_home_key(), {}))
def _is_available_safe(p: BrowserProvider) -> bool:
"""Wrap ``is_available()`` so a buggy provider doesn't kill resolution."""
@@ -188,5 +245,9 @@ def _resolve(configured: Optional[str]) -> Optional[BrowserProvider]:
def _reset_for_tests() -> None:
"""Clear the registry. **Test-only.**"""
global _generation
with _lock:
_providers.clear()
_scoped_providers.clear()
_scoped_generations.clear()
_generation += 1
File diff suppressed because it is too large Load Diff
+435 -30
View File
@@ -14,10 +14,12 @@ import hashlib
import json
import logging
import re
import unicodedata
import uuid
from types import SimpleNamespace
from typing import Any, Dict, List, Optional
from typing import Any, Dict, List, NamedTuple, Optional
from agent.message_sanitization import deterministic_call_id
from agent.prompt_builder import DEFAULT_AGENT_IDENTITY
logger = logging.getLogger(__name__)
@@ -72,6 +74,79 @@ _TOOL_CALL_LEAK_PATTERN = re.compile(
)
# The ChatGPT Codex backend reserves these Harmony wire tokens. If their
# literal spellings are replayed anywhere in request text, the backend rejects
# the request before inference with ``invalid_prompt: Request blocked.``.
# Category-Cf handling covers persisted sessions from an earlier U+200B weak
# defang; fullwidth bars survive format-character stripping while keeping the
# inspected source legible.
_HARMONY_CONTROL_TOKEN_RE = re.compile(
r"<\|(start|end|channel|message|constrain|return|call)\|>"
)
_FULLWIDTH_PIPE = "\uff5c"
def _neutralize_harmony_tokens(text: str) -> str:
"""Keep Harmony source readable without emitting reserved wire tokens."""
if not text or "<" not in text or "|" not in text:
return text
replacement = rf"<{_FULLWIDTH_PIPE}\1{_FULLWIDTH_PIPE}>"
if not any(unicodedata.category(char) == "Cf" for char in text):
return _HARMONY_CONTROL_TOKEN_RE.sub(replacement, text)
# U+200B is confirmed to be stripped by the Codex backend before its
# reserved-token check. Treat every Unicode format control equivalently so
# moving the character elsewhere in the token (or swapping in another Cf)
# cannot recreate the same visually hidden form.
visible_chars: List[str] = []
original_positions: List[int] = []
for index, char in enumerate(text):
if unicodedata.category(char) == "Cf":
continue
visible_chars.append(char)
original_positions.append(index)
visible_text = "".join(visible_chars)
matches = list(_HARMONY_CONTROL_TOKEN_RE.finditer(visible_text))
if not matches:
return text
result: List[str] = []
original_cursor = 0
for match in matches:
original_start = original_positions[match.start()]
original_end = original_positions[match.end() - 1] + 1
result.append(text[original_cursor:original_start])
result.append(f"<{_FULLWIDTH_PIPE}{match.group(1)}{_FULLWIDTH_PIPE}>")
original_cursor = original_end
result.append(text[original_cursor:])
return "".join(result)
def _neutralize_harmony_structure(value: Any) -> Any:
"""Neutralize JSON-like values; normalize tuples and reject unsafe keys.
Rewriting an object key could desynchronize a tool schema from the executor
contract, so a reserved token there is rejected explicitly instead.
"""
if isinstance(value, str):
return _neutralize_harmony_tokens(value)
if isinstance(value, (list, tuple)):
return [_neutralize_harmony_structure(item) for item in value]
if isinstance(value, dict):
normalized = {}
for key, item in value.items():
if isinstance(key, str) and _neutralize_harmony_tokens(key) != key:
raise ValueError(
"Reserved Harmony tokens in a JSON object key cannot be "
"neutralized without changing its contract."
)
normalized[key] = _neutralize_harmony_structure(item)
return normalized
return value
# ---------------------------------------------------------------------------
# Multimodal content helpers
# ---------------------------------------------------------------------------
@@ -182,15 +257,92 @@ def _summarize_user_message_for_log(content: Any, *, sep: str = " ") -> str:
def _deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
"""Generate a deterministic call_id from tool call content.
Used as a fallback when the API doesn't provide a call_id.
Thin wrapper over the single policy owner
``agent.message_sanitization.deterministic_call_id`` (audit F4) — kept
as a module-level name because run_agent and tests import it from here.
Deterministic IDs prevent cache invalidation — random UUIDs would
make every API call's prefix unique, breaking OpenAI's prompt cache.
"""
seed = f"{fn_name}:{arguments}:{index}"
digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12]
return deterministic_call_id(fn_name, arguments, index)
def _clamp_responses_call_id(call_id: str) -> str:
"""Keep a ``call_id`` within the Responses API's 64-char limit (#73492).
The codex app-server namespaces MCP tool call ids as
``codex_mcp__<server>__<tool>_<codex_call_id>``; with an ``exec-<uuid>``
component the built-in ``hermes-tools`` server already overflows 64 chars,
and the Responses API rejects the whole payload with a non-retryable HTTP
400 that then replays every turn — permanently bricking the session.
Sibling defect to #10788 (which clamped ``input[*].id``), applied here to
``call_id``. The surrogate is a pure, deterministic function of the
original, so the ``function_call`` and its matching ``function_call_output``
— which carry the same original id — map to the same surrogate and stay
paired without correlating the two items. Short ids pass through unchanged,
preserving prompt-cache prefixes.
"""
if len(call_id) <= _MAX_RESPONSES_ITEM_ID_LENGTH:
return call_id
digest = hashlib.sha256(call_id.encode("utf-8", errors="replace")).hexdigest()[:32]
return f"call_{digest}"
# The Responses API enforces the same 64-char cap on function names as on
# input item ids (_MAX_RESPONSES_ITEM_ID_LENGTH) — names over the cap are
# rejected with the same non-retryable 400 as pattern violations.
_VALID_RESPONSES_FN_NAME_RE = re.compile(r"[a-zA-Z0-9_-]{1,64}")
def _sanitize_replayed_fn_name(name: str) -> str:
"""Coerce a *replayed* function_call name to the Responses API contract.
The Responses API requires ``function_call.name`` to match
``^[a-zA-Z0-9_-]+$`` and rejects the whole request with a non-retryable
HTTP 400 otherwise (issue #31666). A name with invalid characters (dots,
spaces, unicode — e.g. from an earlier model degeneration) stored in
conversation history therefore bricks every subsequent turn of the
session: the 400 replays forever until the user manually starts a new
conversation.
Invalid characters are replaced with ``_`` (runs collapsed) rather than
stripped, so an all-invalid name degrades to the ``"fn"`` placeholder
instead of an empty string — an empty name would just trade one
non-retryable 400 for a preflight ValueError. Valid names pass through
unchanged, preserving prompt-cache prefixes.
Apply this ONLY to replayed function_call input items, never to live
tool definitions: tool schema names must match the dispatch registry
exactly. Pairing with function_call_output is by call_id, so renaming
a replayed function_call is safe.
"""
if not isinstance(name, str):
return "fn"
if _VALID_RESPONSES_FN_NAME_RE.fullmatch(name):
return name
coerced = re.sub(r"[^A-Za-z0-9_-]", "_", name.strip())
coerced = re.sub(r"_+", "_", coerced).strip("_")
return coerced[:64] or "fn"
def _canonical_call_id_from_fc(response_item_id: Any) -> Optional[str]:
"""Map an ``fc_…`` response-item id to its canonical ``call_<suffix>``.
Both sides of a replayed pair — the assistant ``function_call`` and the
tool ``function_call_output`` — must derive the SAME call_id from an
fc_-only stored id, or an oversized pair clamps to two different
surrogates and the API rejects the output as unmatched. Keep every
caller on this single helper.
"""
if (
isinstance(response_item_id, str)
and response_item_id.startswith("fc_")
and len(response_item_id) > len("fc_")
):
return f"call_{response_item_id[len('fc_'):]}"
return None
def _split_responses_tool_id(raw_id: Any) -> tuple[Optional[str], Optional[str]]:
"""Split a stored tool id into (call_id, response_item_id)."""
if not isinstance(raw_id, str):
@@ -317,6 +469,7 @@ def _chat_messages_to_responses_input(
is_github_responses: bool = False,
replay_encrypted_reasoning: bool = True,
current_issuer_kind: Optional[str] = None,
native_compaction_eligible: bool = False,
) -> List[Dict[str, Any]]:
"""Convert internal chat-style messages to Responses input items.
@@ -361,8 +514,32 @@ def _chat_messages_to_responses_input(
``replay_encrypted_reasoning=False`` is the session-wide kill switch
(drops ALL replay); ``current_issuer_kind`` is the per-item filter
that runs only when replay is still enabled.
``native_compaction_eligible`` mirrors, for THIS request, the decision
made by ``native_compaction.native_compaction_context_management`` — it
is True only when that gate returned a payload, i.e. when the request
actually carries ``context_management``. It controls two things that
must never outlive the gate: replaying ``type: "compaction"`` checkpoint
items, and restructuring the wire around them
(``prune_pre_checkpoint_items``). Checkpoints are persisted in the
``codex_reasoning_items`` sidecar and survive a mid-session model swap,
a ``compression.enabled: false`` flip, the rejection kill switch and a
resumed session; without this flag a single captured checkpoint would
keep deleting every pre-checkpoint item from every later request, on a
model that cannot decrypt the blob (#85914). Default False = pre-feature
wire, which is also correct for every caller that never sends
``context_management`` (auxiliary/compression client, ad-hoc
``convert_messages``). Dropping the checkpoint costs nothing: Hermes'
local history is never truncated by native compaction, so the full
conversation is still on the wire.
"""
items: List[Dict[str, Any]] = []
# Parallel to `items`: the raw chat message each converted item came
# from. Pruning needs this to read a canonical summary carrier's
# up-to-date, provenance-tagged content directly — the converted `item`
# can be a lossy shape (stale exact-replay, or a typed
# `function_call_output` wrapper) that no longer carries it (#90976).
item_sources: List[Optional[Dict[str, Any]]] = []
seen_item_ids: set = set()
for msg in messages:
@@ -402,6 +579,20 @@ def _chat_messages_to_responses_input(
item_id = ri.get("id")
if item_id and item_id in seen_item_ids:
continue
# Native-compaction gate: a checkpoint is only
# meaningful to the endpoint/model that minted it
# AND only while this request still asks for
# server-side compaction. Once the gate closes
# (model swapped out of the gpt-5.6 family,
# compression disabled, rejection kill switch),
# the persisted checkpoint must not be replayed —
# replaying it is what makes the wire restructure
# below erase pre-checkpoint history forever.
if (
ri.get("type") == "compaction"
and not native_compaction_eligible
):
continue
# Cross-issuer guard: drop reasoning blocks that
# were minted by a different Responses endpoint.
# The current endpoint cannot decrypt foreign
@@ -437,6 +628,7 @@ def _chat_messages_to_responses_input(
if k not in ("id", "_issuer_kind")
}
items.append(replay_item)
item_sources.append(msg)
if item_id:
seen_item_ids.add(item_id)
has_codex_reasoning = True
@@ -493,14 +685,17 @@ def _chat_messages_to_responses_input(
if isinstance(phase, str) and phase.strip():
replay_item["phase"] = phase.strip()
items.append(replay_item)
item_sources.append(msg)
replayed_message_items += 1
if replayed_message_items > 0:
pass
elif content_parts:
items.append({"role": "assistant", "content": content_parts})
item_sources.append(msg)
elif content_text.strip():
items.append({"role": "assistant", "content": content_text})
item_sources.append(msg)
elif has_codex_reasoning:
# The Responses API requires a following item after each
# reasoning item (otherwise: missing_following_item error).
@@ -508,6 +703,7 @@ def _chat_messages_to_responses_input(
# content, emit an empty assistant message as the required
# following item.
items.append({"role": "assistant", "content": ""})
item_sources.append(msg)
tool_calls = msg.get("tool_calls")
if isinstance(tool_calls, list):
@@ -526,13 +722,8 @@ def _chat_messages_to_responses_input(
if not isinstance(call_id, str) or not call_id.strip():
call_id = embedded_call_id
if not isinstance(call_id, str) or not call_id.strip():
if (
isinstance(embedded_response_item_id, str)
and embedded_response_item_id.startswith("fc_")
and len(embedded_response_item_id) > len("fc_")
):
call_id = f"call_{embedded_response_item_id[len('fc_'):]}"
else:
call_id = _canonical_call_id_from_fc(embedded_response_item_id)
if call_id is None:
_raw_args = str(fn.get("arguments", "{}"))
call_id = _deterministic_call_id(fn_name, _raw_args, len(items))
call_id = call_id.strip()
@@ -546,10 +737,11 @@ def _chat_messages_to_responses_input(
items.append({
"type": "function_call",
"call_id": call_id,
"name": fn_name,
"call_id": _clamp_responses_call_id(call_id),
"name": _sanitize_replayed_fn_name(fn_name),
"arguments": arguments,
})
item_sources.append(msg)
continue
# Non-assistant (user) role: emit multimodal parts when present,
@@ -558,13 +750,18 @@ def _chat_messages_to_responses_input(
items.append({"role": role, "content": content_parts})
else:
items.append({"role": role, "content": content_text})
item_sources.append(msg)
continue
if role == "tool":
raw_tool_call_id = msg.get("tool_call_id")
call_id, _ = _split_responses_tool_id(raw_tool_call_id)
call_id, tool_response_item_id = _split_responses_tool_id(raw_tool_call_id)
if not isinstance(call_id, str) or not call_id.strip():
if isinstance(raw_tool_call_id, str) and raw_tool_call_id.strip():
# Legacy fc_-only stored ids: canonicalize to the same
# ``call_<suffix>`` the assistant branch synthesizes above, so
# a >64-char pair clamps to the SAME surrogate on both sides.
call_id = _canonical_call_id_from_fc(tool_response_item_id)
if call_id is None and isinstance(raw_tool_call_id, str) and raw_tool_call_id.strip():
call_id = raw_tool_call_id.strip()
if not isinstance(call_id, str) or not call_id.strip():
continue
@@ -589,11 +786,158 @@ def _chat_messages_to_responses_input(
items.append({
"type": "function_call_output",
"call_id": call_id,
"call_id": _clamp_responses_call_id(call_id),
"output": output_value,
})
item_sources.append(msg)
return items
# Native server-side compaction: when a replayed checkpoint is present,
# restructure the wire around it. The server renders nothing placed
# before a compaction item (live-verified Aug 2026), so pre-checkpoint
# history is dead upload weight and — worse — the user's plaintext asks,
# and any local-compression summary already merged into that history,
# silently vanish from the model's view. Keep the newest checkpoint
# first, retain pre-checkpoint USER messages and compression-SUMMARY
# messages (whole, never byte-sliced) verbatim within a token budget
# each (Codex CLI parity for the user side), and leave the
# post-checkpoint tail untouched. Gated on the CURRENT request's native
# eligibility, not merely on the presence of a checkpoint: a persisted
# checkpoint outlives the gate, and pruning for a request that carries no
# ``context_management`` deletes history the server never compacted.
#
# ``item_sources`` (parallel to ``items``) carries the raw chat message
# each converted item came from. A canonical summary carrier's content
# can be lost or gone stale by the time it becomes a Responses item — a
# merge-into-tail tool-result carrier becomes a typed
# ``function_call_output`` (no ``content``/``role`` at all), and a
# merge-into-tail assistant carrier can be shadowed by a stale exact
# ``codex_message_items`` replay from before the merge rewrote its
# content. Pruning reads the source message's own up-to-date,
# provenance-tagged content directly instead of trying to recover it
# from whatever shape the conversion produced (#90976).
if not native_compaction_eligible:
return items
from agent.native_compaction import prune_pre_checkpoint_items
return prune_pre_checkpoint_items(items, item_sources=item_sources)
class ResponsesRouteFlags(NamedTuple):
"""Which special Responses-API route an agent is talking to.
Single owner of the codex/xai/github route predicates. Every site that
needs these flags (request kwargs build, preflight estimation, silent-
reject hints) must call :func:`classify_responses_route` instead of
re-implementing the string comparisons inline — inline copies drift
(backend-identity class: #22548/#70893/#59561/#72468).
"""
is_codex_backend: bool
is_xai_responses: bool
is_github_responses: bool
def classify_responses_route(agent: Any) -> ResponsesRouteFlags:
"""Classify the agent's Responses route from provider + base URL.
Host checks are exact-host-or-subdomain (``base_url_hostname``
semantics), never substring matching — ``https://evil.com/models.github.ai``
must not classify as a GitHub route.
"""
from utils import base_url_hostname
provider = getattr(agent, "provider", None)
base_url = str(getattr(agent, "base_url", "") or "")
hostname = str(getattr(agent, "_base_url_hostname", "") or "").lower()
if not hostname:
hostname = base_url_hostname(base_url)
lower = str(getattr(agent, "_base_url_lower", "") or base_url).lower()
def _host_is(domain: str) -> bool:
return hostname == domain or hostname.endswith("." + domain)
is_codex_backend = provider == "openai-codex" or (
_host_is("chatgpt.com") and "/backend-api/codex" in lower
)
is_github_responses = _host_is("models.github.ai") or _host_is("githubcopilot.com")
is_xai_responses = provider in {"xai", "xai-oauth"} or hostname == "api.x.ai"
return ResponsesRouteFlags(
is_codex_backend=is_codex_backend,
is_xai_responses=is_xai_responses,
is_github_responses=is_github_responses,
)
def estimate_native_responses_preflight_tokens(
agent: Any,
messages: List[Dict[str, Any]],
*,
system_prompt: str = "",
tools: Optional[List[Dict[str, Any]]] = None,
) -> Optional[int]:
"""Estimate tokens for the checkpoint-pruned Responses payload.
Automatic preflight previously counted the full durable transcript.
On a natively compacted Codex session that overstates the wire by
several times and fires local compression against history the main
request will never send (#96155).
Returns None when native compaction is not proven eligible for this
request, or when conversion fails — the caller must then use the
generic durable-transcript estimate (conservative).
"""
if getattr(agent, "api_mode", None) != "codex_responses":
return None
if not isinstance(messages, list):
return None
is_codex_backend, is_xai_responses, is_github_responses = classify_responses_route(agent)
from agent.native_compaction import native_compaction_context_management
context_management = native_compaction_context_management(
agent,
is_codex_backend=is_codex_backend,
is_xai_responses=is_xai_responses,
is_github_responses=is_github_responses,
)
if not context_management:
return None
try:
items = _chat_messages_to_responses_input(
messages,
is_xai_responses=is_xai_responses,
is_github_responses=is_github_responses,
replay_encrypted_reasoning=bool(
getattr(agent, "_codex_reasoning_replay_enabled", True)
),
current_issuer_kind=_classify_responses_issuer(
is_xai_responses=is_xai_responses,
is_github_responses=is_github_responses,
is_codex_backend=is_codex_backend,
base_url=getattr(agent, "base_url", None),
),
native_compaction_eligible=True,
)
except Exception:
logger.debug(
"native Responses preflight conversion failed; falling back to generic estimate",
exc_info=True,
)
return None
if not isinstance(items, list):
return None
from agent.model_metadata import estimate_request_tokens_rough
return estimate_request_tokens_rough(
items,
system_prompt=system_prompt or "",
tools=tools,
)
# ---------------------------------------------------------------------------
@@ -604,10 +948,16 @@ def _preflight_codex_input_items(
raw_items: Any,
*,
is_github_responses: bool = False,
sanitize_harmony_tokens: bool = False,
) -> List[Dict[str, Any]]:
if not isinstance(raw_items, list):
raise ValueError("Codex Responses input must be a list of input items.")
sanitize_text = (
_neutralize_harmony_tokens
if sanitize_harmony_tokens
else lambda text: text
)
normalized: List[Dict[str, Any]] = []
seen_ids: set = set()
for idx, item in enumerate(raw_items):
@@ -628,13 +978,13 @@ def _preflight_codex_input_items(
arguments = json.dumps(arguments, ensure_ascii=False)
elif not isinstance(arguments, str):
arguments = str(arguments)
arguments = arguments.strip() or "{}"
arguments = sanitize_text(arguments.strip() or "{}")
normalized.append(
{
"type": "function_call",
"call_id": call_id.strip(),
"name": name.strip(),
"name": _sanitize_replayed_fn_name(name),
"arguments": arguments,
}
)
@@ -662,7 +1012,7 @@ def _preflight_codex_input_items(
if ptype == "input_text":
text = part.get("text")
if isinstance(text, str) and text:
cleaned.append({"type": "input_text", "text": text})
cleaned.append({"type": "input_text", "text": sanitize_text(text)})
elif ptype == "input_image":
url = part.get("image_url")
if isinstance(url, str) and url:
@@ -686,7 +1036,7 @@ def _preflight_codex_input_items(
{
"type": "function_call_output",
"call_id": call_id.strip(),
"output": output,
"output": sanitize_text(output),
}
)
continue
@@ -699,19 +1049,37 @@ def _preflight_codex_input_items(
if item_id in seen_ids:
continue
seen_ids.add(item_id)
reasoning_item = {"type": "reasoning", "encrypted_content": encrypted}
reasoning_item: Dict[str, Any] = {
"type": "reasoning",
"encrypted_content": encrypted,
}
# Do NOT include the "id" in the outgoing item — with
# store=False (our default) the API tries to resolve the
# id server-side and returns 404. The id is still used
# above for local deduplication via seen_ids.
summary = item.get("summary")
if isinstance(summary, list):
reasoning_item["summary"] = summary
reasoning_item["summary"] = (
_neutralize_harmony_structure(summary)
if sanitize_harmony_tokens
else summary
)
else:
reasoning_item["summary"] = []
normalized.append(reasoning_item)
continue
if item_type == "compaction":
# Replayed native server-side compaction checkpoint (gpt-5.6,
# direct OpenAI/Codex routes). Opaque, issuer-sealed; forward
# only the fields the API defines.
encrypted = item.get("encrypted_content")
if isinstance(encrypted, str) and encrypted:
normalized.append(
{"type": "compaction", "encrypted_content": encrypted}
)
continue
if item_type == "message":
role = item.get("role")
if role != "assistant":
@@ -735,7 +1103,7 @@ def _preflight_codex_input_items(
text = ""
if not isinstance(text, str):
text = str(text)
normalized_content.append({"type": "output_text", "text": text})
normalized_content.append({"type": "output_text", "text": sanitize_text(text)})
if not normalized_content:
raise ValueError(f"Codex Responses input[{idx}] message item must contain at least one text part.")
normalized_item: Dict[str, Any] = {
@@ -775,7 +1143,7 @@ def _preflight_codex_input_items(
for part_idx, part in enumerate(content):
if isinstance(part, str):
if part:
validated.append({"type": text_type, "text": part})
validated.append({"type": text_type, "text": sanitize_text(part)})
continue
if not isinstance(part, dict):
raise ValueError(
@@ -786,7 +1154,7 @@ def _preflight_codex_input_items(
text = part.get("text", "")
if not isinstance(text, str):
text = str(text or "")
validated.append({"type": text_type, "text": text})
validated.append({"type": text_type, "text": sanitize_text(text)})
elif ptype in {"input_image", "image_url"}:
image_ref = part.get("image_url", "")
detail = part.get("detail")
@@ -810,7 +1178,7 @@ def _preflight_codex_input_items(
if not isinstance(content, str):
content = str(content)
normalized.append({"role": role, "content": content})
normalized.append({"role": role, "content": sanitize_text(content)})
continue
raise ValueError(
@@ -825,6 +1193,7 @@ def _preflight_codex_api_kwargs(
*,
allow_stream: bool = False,
is_github_responses: bool = False,
sanitize_harmony_tokens: bool = False,
) -> Dict[str, Any]:
if not isinstance(api_kwargs, dict):
raise ValueError("Codex Responses request must be a dict.")
@@ -845,10 +1214,13 @@ def _preflight_codex_api_kwargs(
if not isinstance(instructions, str):
instructions = str(instructions)
instructions = instructions.strip() or DEFAULT_AGENT_IDENTITY
if sanitize_harmony_tokens:
instructions = _neutralize_harmony_tokens(instructions)
normalized_input = _preflight_codex_input_items(
api_kwargs.get("input"),
is_github_responses=is_github_responses,
sanitize_harmony_tokens=sanitize_harmony_tokens,
)
tools = api_kwargs.get("tools")
@@ -905,6 +1277,9 @@ def _preflight_codex_api_kwargs(
}
)
if sanitize_harmony_tokens and normalized_tools is not None:
normalized_tools = _neutralize_harmony_structure(normalized_tools)
store = api_kwargs.get("store", False)
if store is not False:
raise ValueError("Codex Responses contract requires 'store' to be false.")
@@ -912,7 +1287,8 @@ def _preflight_codex_api_kwargs(
allowed_keys = {
"model", "instructions", "input", "tools", "store",
"reasoning", "include", "max_output_tokens", "temperature",
"tool_choice", "parallel_tool_calls", "prompt_cache_key", "service_tier",
"tool_choice", "parallel_tool_calls", "prompt_cache_key",
"prompt_cache_retention", "service_tier", "context_management",
"extra_headers", "extra_body", "timeout",
}
normalized: Dict[str, Any] = {
@@ -950,12 +1326,24 @@ def _preflight_codex_api_kwargs(
if isinstance(temperature, (int, float)):
normalized["temperature"] = float(temperature)
# Pass through tool_choice, parallel_tool_calls, prompt_cache_key
for passthrough_key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key"):
# Pass through cache routing/retention and tool-dispatch hints.
for passthrough_key in (
"tool_choice",
"parallel_tool_calls",
"prompt_cache_key",
"prompt_cache_retention",
):
val = api_kwargs.get(passthrough_key)
if val is not None:
normalized[passthrough_key] = val
# Native server-side compaction directive (gpt-5.6 on direct OpenAI /
# Codex routes — eligibility already resolved upstream in
# agent/native_compaction.py; the preflight only preserves the shape).
context_management = api_kwargs.get("context_management")
if isinstance(context_management, list) and context_management:
normalized["context_management"] = context_management
extra_headers = api_kwargs.get("extra_headers")
if extra_headers is not None:
if not isinstance(extra_headers, dict):
@@ -1293,6 +1681,23 @@ def _normalize_codex_response(
raw_summary.append({"type": "summary_text", "text": text})
raw_item["summary"] = raw_summary
reasoning_items_raw.append(raw_item)
elif item_type == "compaction":
# Native server-side compaction checkpoint (gpt-5.6 on direct
# OpenAI/Codex routes). The encrypted blob stands in for the
# pruned older context on subsequent requests. It rides the
# codex_reasoning_items sidecar so it inherits persistence
# (state.db), session replay, the cross-issuer guard, and the
# invalid-encrypted-content kill switch without new state.
encrypted = getattr(item, "encrypted_content", None)
if isinstance(encrypted, str) and encrypted:
raw_item = {"type": "compaction", "encrypted_content": encrypted}
if issuer_kind:
raw_item["_issuer_kind"] = issuer_kind
reasoning_items_raw.append(raw_item)
logger.info(
"Native Responses compaction item captured (%d chars encrypted).",
len(encrypted),
)
elif item_type == "function_call":
if item_status in {"queued", "in_progress", "incomplete"}:
continue
+574 -42
View File
@@ -28,6 +28,65 @@ from agent.stream_single_writer import claim_stream_writer, stream_writer_is_cur
logger = logging.getLogger(__name__)
def _codex_request_failure_details(error: BaseException) -> tuple[int | None, str]:
"""Return the serialized request size and exception class chain.
OpenAI connection exceptions retain the final ``httpx.Request``. Reading
its already-buffered content gives us the exact byte count handed to the
transport without logging any request content. The class-only chain keeps
the underlying transport failure visible without exposing URLs or payloads
from exception messages.
"""
request_body_bytes: int | None = None
exception_classes: list[str] = []
current: BaseException | None = error
seen: set[int] = set()
while current is not None and id(current) not in seen and len(seen) < 8:
seen.add(id(current))
exception_classes.append(type(current).__name__)
if request_body_bytes is None:
try:
request = getattr(current, "request", None)
except Exception:
request = None
if request is not None:
try:
content = request.content
except Exception:
content = None
if isinstance(content, str):
request_body_bytes = len(content.encode("utf-8"))
elif isinstance(content, (bytes, bytearray, memoryview)):
request_body_bytes = len(content)
cause = current.__cause__
if cause is None and not current.__suppress_context__:
cause = current.__context__
current = cause
return request_body_bytes, " <- ".join(exception_classes)
def _log_codex_request_failure(
agent: Any,
error: BaseException,
*,
stream_opened: bool,
) -> None:
request_body_bytes, exception_chain = _codex_request_failure_details(error)
logger.warning(
"Codex Responses request failed: "
"serialized_request_body_bytes=%s stream_opened=%s "
"exception_chain=%s model=%s",
request_body_bytes if request_body_bytes is not None else "unknown",
str(stream_opened).lower(),
exception_chain,
getattr(agent, "model", "unknown"),
)
def _coerce_usage_int(value: Any) -> int:
if isinstance(value, bool):
return 0
@@ -74,7 +133,10 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
try:
if not agent._session_db_created:
agent._ensure_db_session()
agent._session_db.update_token_counts(
# Enqueued for the SessionDB background writer — keeps the
# per-call accounting write off the turn thread (see
# conversation_loop's queue_token_counts call).
agent._session_db.queue_token_counts(
agent.session_id,
model=agent.model,
billing_provider=agent.provider,
@@ -154,7 +216,8 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
try:
if not agent._session_db_created:
agent._ensure_db_session()
agent._session_db.update_token_counts(
# Enqueued for the SessionDB background writer (see above).
agent._session_db.queue_token_counts(
agent.session_id,
input_tokens=canonical_usage.input_tokens,
output_tokens=canonical_usage.output_tokens,
@@ -628,6 +691,20 @@ def run_codex_app_server_turn(
Called from run_conversation() when agent.api_mode == "codex_app_server".
Returns the same dict shape as the chat_completions path.
"""
# Defense in depth for compression.checkpoint_required: agent init
# already refuses this combination, but api_mode is a plain attribute a
# future code path could mutate on a live agent. Fail closed before the
# codex agent can compact its thread — once run_turn() executes, a
# codex-owned compaction may already have happened with no pre-compress
# checkpoint. Explicit-True check matches the compress_context() gate.
if getattr(agent, "compression_checkpoint_required", False) is True:
from agent.conversation_compression import _checkpoint_blocked
raise _checkpoint_blocked(
"codex_app_server owns the authoritative thread and compacts it "
"without a truthful pre-compaction transcript boundary"
)
from agent.transports.codex_app_server_session import (
CodexAppServerSession,
_ServerRequestRouting,
@@ -702,6 +779,16 @@ def run_codex_app_server_turn(
except Exception:
pass
agent._codex_session = None
_user_interrupted = bool(
getattr(agent, "_interrupt_requested", False)
)
_interrupt_message = (
getattr(agent, "_interrupt_message", None)
if _user_interrupted
else None
)
if _user_interrupted:
agent.clear_interrupt()
return {
"final_response": (
f"Codex app-server turn failed: {exc}. "
@@ -711,9 +798,27 @@ def run_codex_app_server_turn(
"api_calls": 0,
"completed": False,
"partial": True,
"interrupted": _user_interrupted,
**(
{"interrupt_message": _interrupt_message}
if _interrupt_message
else {}
),
"error": str(exc),
}
# This runtime bypasses the normal conversation-loop finalizer. Mirror its
# interrupt handoff/cleanup so a hard stop cannot poison the next turn and a
# message-bearing compatibility interrupt can still be replayed by callers.
_user_interrupted = bool(
turn.interrupted and getattr(agent, "_interrupt_requested", False)
)
_interrupt_message = (
getattr(agent, "_interrupt_message", None) if _user_interrupted else None
)
if _user_interrupted:
agent.clear_interrupt()
# If the turn signalled the underlying client is wedged (deadline
# blown, post-tool watchdog tripped, OAuth refresh died, subprocess
# exited), retire the session so the next turn respawns codex
@@ -734,7 +839,10 @@ def run_codex_app_server_turn(
# standard {role, content, tool_calls, tool_call_id} entries, which
# is exactly what curator.py / sessions DB expect.
if turn.projected_messages:
messages.extend(turn.projected_messages)
from agent.message_metadata import append_message
for projected_message in turn.projected_messages:
append_message(messages, projected_message)
# Persist the newly-projected assistant/tool messages ourselves.
# This path is an early return that bypasses conversation_loop, whose
@@ -750,12 +858,27 @@ def run_codex_app_server_turn(
# the already-flushed user turn). See gateway/run.py agent_persisted.
if getattr(agent, "_session_db", None) is not None:
try:
agent._flush_messages_to_session_db(messages)
_codex_flush_ok = agent._flush_messages_to_session_db(messages)
except Exception:
logger.debug(
_codex_flush_ok = False
logger.warning(
"codex app-server projected-message flush failed",
exc_info=True,
)
if _codex_flush_ok is False:
# Unlike the chat-completions loop (which fails closed BEFORE
# projection — see conversation_loop session_persistence_failed),
# codex output has already streamed to the user by the time this
# flush runs, so there is nothing left to withhold. We cannot
# flip agent_persisted=False either: the gateway fallback write
# would re-INSERT the already-flushed user turn (#860/#42039).
# Surface the durability gap loudly instead of a silent debug.
logger.warning(
"codex app-server turn was delivered but could NOT be "
"persisted to the session DB (session=%s) — this turn "
"will be missing after restart/resume",
getattr(agent, "session_id", None),
)
# Counter ticks for the agent-improvement loop.
@@ -819,6 +942,12 @@ def run_codex_app_server_turn(
"api_calls": api_calls,
"completed": not turn.interrupted and turn.error is None,
"partial": turn.interrupted or turn.error is not None,
"interrupted": _user_interrupted,
**(
{"interrupt_message": _interrupt_message}
if _interrupt_message
else {}
),
"error": turn.error,
# The codex app-server runtime IS an early-return path that bypasses
# conversation_loop, but we flush the projected assistant/tool messages
@@ -970,17 +1099,41 @@ def _consume_codex_event_stream(
* ``interrupt_check()`` — returns True to break the loop early.
"""
collected_output_items: List[Any] = []
# output_index of each collected_output_items entry, appended in lockstep
# so settled pending calls can be merged back in stream order.
collected_output_indexes: List[Any] = []
collected_output_sequences: List[int] = []
collected_text_deltas: List[str] = []
has_tool_calls = False
# Function calls announced via output_item.added but not yet confirmed by
# output_item.done, keyed by item id. Some OpenAI-compatible backends omit
# per-item done events on a successful completion (upstream evidence:
# anomalyco/opencode#37159); these are settled from accumulated stream
# state at the terminal event so the tool call executes instead of being
# silently dropped.
pending_function_calls: Dict[str, Dict[str, Any]] = {}
# First-observed (sequence, output_index) per announced item id, so items
# confirmed later via output_item.done keep their announced stream
# position when merged with settled pending calls.
announced_output_order: Dict[str, tuple] = {}
first_delta_fired = False
active_message_phase: str | None = None
commentary_text_deltas: List[str] = []
# Last reasoning summary_index seen. The Responses stream delimits summary
# parts by this index and gives each part no separator of its own, so a
# change of index is where the blank line belongs.
active_summary_index: Any = None
terminal_status: str = "completed"
terminal_usage: Any = None
terminal_response_id: str = None
terminal_incomplete_details: Any = None
terminal_error: Any = None
saw_terminal = False
# Settlement of pending calls requires an actually observed successful
# terminal frame. ``terminal_status`` defaults to "completed", so it
# cannot distinguish a real response.completed from EOF/interruption.
saw_response_completed = False
next_output_sequence = 0
for event in event_iter:
if on_event is not None:
@@ -1023,8 +1176,32 @@ def _consume_codex_event_stream(
commentary_text_deltas = []
else:
active_message_phase = None
# First-observed ordering metadata for EVERY announced item (not
# just function calls): when this item later lands via
# output_item.done, the done path must reuse the announced
# sequence/index instead of allocating a fresh tail position, or
# a mixed announced/pending stream without output_index values
# reorders the calls (review P1 on PR #92767).
item_id = str(_item_field(item, "id", ""))
if item_id and item_id not in announced_output_order:
announced_output_order[item_id] = (
next_output_sequence,
_event_field(event, "output_index", None),
)
next_output_sequence += 1
if "function_call" in str(item_type):
has_tool_calls = True
if item_id:
announced_sequence, announced_index = announced_output_order[item_id]
# Seed from the announced item's own arguments when the
# backend attaches them up front, and remember the stream
# position so a settled call keeps its place in the output.
pending_function_calls[item_id] = {
"item": item,
"arguments": str(_item_field(item, "arguments", "") or ""),
"output_index": announced_index,
"sequence": announced_sequence,
}
continue
if "output_text.delta" in event_type or event_type == "response.output_text.delta":
@@ -1063,11 +1240,42 @@ def _consume_codex_event_stream(
if "function_call" in event_type:
has_tool_calls = True
# fall through — function_call items still get added on output_item.done
# Accumulate streamed argument deltas for calls announced via
# output_item.added, so a stream that completes without per-item
# done events can still be settled from accumulated state.
if "delta" in event_type:
delta_args = _event_field(event, "delta", "")
pending = pending_function_calls.get(str(_event_field(event, "item_id", "")))
if pending is not None and delta_args:
pending["arguments"] += delta_args
continue
if event_type.endswith("function_call_arguments.done"):
done_args = _event_field(event, "arguments", None)
pending = pending_function_calls.get(str(_event_field(event, "item_id", "")))
if pending is not None and done_args is not None:
# Per-item arguments.done is authoritative for the
# accumulated string when the item itself never lands.
# An explicit empty string (zero-argument call) counts as
# authoritative; only a missing field leaves the streamed
# deltas in place.
pending["arguments"] = str(done_args)
continue
# other function_call frames fall through — function_call items still get added on output_item.done
if "reasoning" in event_type and "delta" in event_type:
reasoning_text = _event_field(event, "delta", "")
if reasoning_text and on_reasoning_delta is not None:
# Summary parts stream one after another with no separator of
# their own; summary_index is the boundary the wire gives us.
summary_index = _event_field(event, "summary_index")
if (
summary_index is not None
and active_summary_index is not None
and summary_index != active_summary_index
):
reasoning_text = f"\n\n{reasoning_text}"
if summary_index is not None:
active_summary_index = summary_index
try:
on_reasoning_delta(reasoning_text)
except Exception:
@@ -1078,6 +1286,26 @@ def _consume_codex_event_stream(
done_item = _event_field(event, "item")
if done_item is not None:
collected_output_items.append(done_item)
# Reuse the first-observed position when this item was
# announced earlier via output_item.added; a fresh tail
# sequence is allocated only for genuinely unannounced items.
# The .done event's own output_index wins when present, with
# the announced index as its fallback.
done_id = str(_item_field(done_item, "id", ""))
announced_sequence, announced_index = announced_output_order.get(
done_id, (None, None)
)
done_index = _event_field(event, "output_index", None)
if done_index is None:
done_index = announced_index
if announced_sequence is None:
announced_sequence = next_output_sequence
next_output_sequence += 1
collected_output_indexes.append(done_index)
collected_output_sequences.append(announced_sequence)
# Confirmed by the authoritative per-item done event; remove
# from pending so it is not settled twice.
pending_function_calls.pop(done_id, None)
done_phase = _item_field(done_item, "phase", None)
done_phase = done_phase.strip().lower() if isinstance(done_phase, str) else None
if done_phase == "commentary" and on_commentary_message is not None:
@@ -1126,6 +1354,7 @@ def _consume_codex_event_stream(
if terminal_error is None and isinstance(resp_obj, dict):
terminal_error = resp_obj.get("error")
if event_type == "response.completed":
saw_response_completed = True
terminal_status = terminal_status or "completed"
elif event_type == "response.incomplete":
terminal_status = terminal_status or "incomplete"
@@ -1150,6 +1379,56 @@ def _consume_codex_event_stream(
else:
output = []
# Settle function calls that were announced via output_item.added and
# streamed argument deltas but never confirmed by output_item.done: some
# OpenAI-compatible backends omit per-item done events on a successful
# completion (anomalyco/opencode#37159). Done items stay authoritative;
# this only fills the gap so the call executes instead of vanishing.
if pending_function_calls and saw_response_completed:
# Assemble settled calls and .done items in output_index order instead
# of appending at the tail: a pending call that streamed before a later
# .done item must keep its position, or dependent side effects invert.
indexed = [
(index, sequence, position, item)
for position, (index, sequence, item) in enumerate(
zip(
collected_output_indexes,
collected_output_sequences,
collected_output_items,
)
)
]
for position, pending in enumerate(pending_function_calls.values(), start=len(indexed)):
item = pending["item"]
# Canonicalize empty/whitespace arguments so zero-delta calls stay
# executable; malformed non-empty JSON passes through untouched and
# stays rejected by downstream argument parsing.
arguments = (pending["arguments"] or "").strip() or "{}"
indexed.append((pending.get("output_index"), pending["sequence"], position, SimpleNamespace(
type="function_call",
id=_item_field(item, "id", None),
call_id=_item_field(item, "call_id", None),
name=_item_field(item, "name", None),
arguments=arguments,
status="completed",
)))
# output_index is optional in compatible Responses streams. A partial
# ordering (sorting indexed entries while interleaving unindexed ones)
# is not well-defined and can produce contradictory comparisons. Keep
# the observed wire order whenever any index is missing; use the
# protocol ordering only when every entry provides an index.
if all(entry[0] is not None for entry in indexed):
try:
indexed.sort(key=lambda entry: entry[0])
except TypeError:
# Preserve wire order if a backend sends non-comparable index
# values instead of integers.
pass
else:
indexed.sort(key=lambda entry: entry[1])
output = [entry[3] for entry in indexed]
# If the stream ended without any terminal event AND produced no usable
# content (no items, no text deltas), surface that as a RuntimeError so
# callers can distinguish "stream truncated mid-flight / provider rejected
@@ -1176,6 +1455,134 @@ def _consume_codex_event_stream(
return final
def _sanitize_consumer_codex_request(
agent: Any,
request: dict[str, Any],
) -> dict[str, Any]:
"""Drop fields the ChatGPT OAuth Codex endpoint does not accept.
This guard intentionally lives at the final wire boundary, after Relay or
other request middleware has had a chance to transform the request. The
normal transport builder already omits ``prompt_cache_retention`` for this
endpoint, but a late mutation must not be allowed to turn a valid tool
follow-up into a non-retryable HTTP 400.
Explicit ``request_overrides`` are subject to the same endpoint contract:
unsupported retention is dropped with a warning instead of being sent and
rejected by the provider. The check covers both the top-level kwarg and a
nested ``extra_body`` entry — the OpenAI SDK merges ``extra_body`` into
the outgoing JSON body, so either shape reaches the endpoint.
"""
sanitized = dict(request)
# Resolved defensively on purpose: run_codex_stream is also driven with
# lightweight stand-in agents that carry only the attributes a given path
# needs (see tests/agent/test_codex_request_transport_diagnostics.py), so a
# bare agent._is_codex_backend() here would raise AttributeError on them.
backend_predicate = getattr(agent, "_is_codex_backend", None)
is_consumer_codex = (
bool(backend_predicate()) if callable(backend_predicate) else False
)
if not is_consumer_codex:
return sanitized
dropped_from: list[str] = []
if "prompt_cache_retention" in sanitized:
sanitized.pop("prompt_cache_retention")
dropped_from.append("top-level")
# The OpenAI SDK merges ``extra_body`` into the outgoing JSON body, so a
# nested ``extra_body.prompt_cache_retention`` reaches the endpoint just
# like the top-level field would. Copy before editing — the caller's
# mapping must not be mutated — and drop the mapping when it empties.
extra_body = sanitized.get("extra_body")
if isinstance(extra_body, dict) and "prompt_cache_retention" in extra_body:
extra_body = dict(extra_body)
extra_body.pop("prompt_cache_retention")
if extra_body:
sanitized["extra_body"] = extra_body
else:
sanitized.pop("extra_body")
dropped_from.append("extra_body")
if dropped_from:
logger.warning(
"Dropped unsupported prompt_cache_retention at consumer Codex "
"wire boundary (model=%s, via %s).",
sanitized.get("model", getattr(agent, "model", "unknown")),
", ".join(dropped_from),
)
return sanitized
# Bulk request fields that carry the conversation payload. Everything else in
# the request is scalar configuration the SDK transform handles in microseconds.
_SDK_TRANSFORM_BYPASS_FIELDS = ("input", "tools")
def _is_plain_json_data(value: Any) -> bool:
"""True when ``value`` is composed purely of JSON wire types.
The SDK's request transform exists to convert typed params (TypedDict
key aliases, pydantic models, ``PropertyInfo`` formats) into wire
format. Hermes assembles Codex payloads from JSON round-trips, so they
are already wire format — but that is only provable when every node is
a plain JSON type. Anything else must keep the typed SDK path.
"""
if value is None or isinstance(value, (str, int, float, bool)):
return True
if isinstance(value, dict):
return all(
isinstance(key, str) and _is_plain_json_data(item)
for key, item in value.items()
)
if isinstance(value, list):
return all(_is_plain_json_data(item) for item in value)
return False
def _bypass_sdk_request_transform(stream_kwargs: dict) -> dict:
"""Route bulk payload fields around the SDK's ``maybe_transform`` (#93650).
``responses.create`` re-walks the entire request body against the
``ResponseCreateParams`` union graph before any byte leaves the process.
That walk runs with the GIL held, and #93650 documents it wedging for
12+ hours on a ~1.4 MB conversation — starving every other thread,
including the TTFB/stale watchdogs whose job is to rescue this exact
call. Because the hang is client-side and pre-network, no socket kill
can unblock it.
The SDK merges ``extra_body`` into the JSON body *after* the transform
(``_base_client._build_request``), so moving the already-wire-format
bulk fields there skips the walk entirely and produces a byte-identical
request. Fields containing anything that is not plain JSON data (e.g.
pydantic models, generators) stay on the typed path, which still needs
the transform. Set HERMES_CODEX_SDK_TRANSFORM=1 to restore the pre-fix
behavior.
"""
if os.environ.get("HERMES_CODEX_SDK_TRANSFORM", "").strip().lower() in {
"1", "true", "yes", "on"
}:
return stream_kwargs
moved = {
field: stream_kwargs[field]
for field in _SDK_TRANSFORM_BYPASS_FIELDS
if isinstance(stream_kwargs.get(field), (dict, list))
and _is_plain_json_data(stream_kwargs[field])
}
if not moved:
return stream_kwargs
bypassed = {
key: value for key, value in stream_kwargs.items() if key not in moved
}
extra_body = bypassed.get("extra_body")
merged = dict(extra_body) if isinstance(extra_body, dict) else {}
for field, value in moved.items():
# An explicit caller-provided extra_body entry keeps precedence,
# matching what the SDK's post-transform merge would have done.
merged.setdefault(field, value)
bypassed["extra_body"] = merged
return bypassed
def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta=None):
"""Execute one streaming Responses API request and return the final response.
@@ -1186,6 +1593,9 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
the terminal event's ``output`` field.
"""
import httpx as _httpx
from openai import APIConnectionError as _APIConnectionError
from agent import relay_llm
active_client = client or agent._ensure_primary_openai_client(reason="codex_stream_direct")
max_stream_retries = 1
@@ -1211,48 +1621,104 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
if agent._interrupt_requested:
raise InterruptedError("Agent interrupted before Codex stream retry")
stream_kwargs = dict(api_kwargs)
stream_kwargs["stream"] = True
intercepted_events = []
writer_token = {"value": None}
try:
event_stream = active_client.responses.create(**stream_kwargs)
except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc:
if attempt < max_stream_retries:
logger.debug(
"Codex Responses stream connect failed (attempt %s/%s); retrying. %s error=%s",
attempt + 1, max_stream_retries + 1,
agent._client_log_context(), exc,
)
continue
raise
def _open_codex_stream(next_api_kwargs: dict[str, Any]):
stream_kwargs = _sanitize_consumer_codex_request(
agent,
next_api_kwargs,
)
stream_kwargs["stream"] = True
stream_kwargs = _bypass_sdk_request_transform(stream_kwargs)
return active_client.responses.create(**stream_kwargs)
# Claim the delta sink for THIS attempt (#65991) — parity with the
# chat_completions/anthropic/bedrock paths. If a prior attempt's
# stream is somehow still alive, this claim supersedes it so its
# late deltas are fenced out of the turn; conversely, a newer
# attempt supersedes us and the interrupt_check below stops our
# consumption immediately.
_writer_token = claim_stream_writer(agent)
def _codex_stream_created(_raw_stream: Any) -> None:
# Claim the delta sink for THIS physical attempt. A newer attempt
# supersedes this token and fences late deltas out of the turn.
writer_token["value"] = claim_stream_writer(agent)
def _interrupt_or_superseded(_tok=_writer_token) -> bool:
if agent._interrupt_requested:
return True
if not stream_writer_is_current(agent, _tok):
logger.warning(
"Codex streaming attempt superseded by a newer stream; "
"stopping consumption to preserve the single-writer "
"invariant (model=%s).",
api_kwargs.get("model", "unknown"),
)
def _accept_codex_chunk(_chunk: Any) -> bool:
token = writer_token["value"]
if token is None or stream_writer_is_current(agent, token):
return True
logger.warning(
"Codex streaming attempt superseded by a newer stream; "
"stopping consumption to preserve the single-writer "
"invariant (model=%s).",
api_kwargs.get("model", "unknown"),
)
return False
try:
# Compatibility: some mocks/providers return a concrete response
# instead of an iterable. Pass it straight through.
if hasattr(event_stream, "output") and not hasattr(event_stream, "__iter__"):
return event_stream
def _finalize_codex_stream() -> Any:
return _consume_codex_event_stream(
list(intercepted_events),
model=api_kwargs.get("model"),
)
try:
event_stream = relay_llm.stream(
dict(api_kwargs),
_open_codex_stream,
session_id=str(getattr(agent, "session_id", "") or ""),
name=str(getattr(agent, "provider", "") or "codex"),
model_name=str(api_kwargs.get("model") or ""),
finalizer=_finalize_codex_stream,
on_stream_created=_codex_stream_created,
on_chunk=intercepted_events.append,
chunk_adapter=lambda chunk: chunk,
accept_chunk=_accept_codex_chunk,
completed_response_predicate=lambda response: bool(
hasattr(response, "output") and not hasattr(response, "__iter__")
),
metadata={
"api_mode": "codex_responses",
"api_request_id": getattr(agent, "_current_api_request_id", None),
"call_role": (
"delegated"
if getattr(agent, "is_subagent", False)
else "fallback"
if int(getattr(agent, "_fallback_index", 0) or 0) > 0
else "primary"
),
"retry_count": attempt,
},
defer_logical_completion=True,
)
except (
_httpx.RemoteProtocolError,
_httpx.ReadTimeout,
_httpx.ConnectError,
ConnectionError,
) as exc:
if attempt < max_stream_retries:
logger.debug(
"Codex Responses stream connect failed (attempt %s/%s); "
"retrying. %s error=%s",
attempt + 1,
max_stream_retries + 1,
agent._client_log_context(),
exc,
)
continue
_log_codex_request_failure(
agent,
exc,
stream_opened=writer_token["value"] is not None,
)
raise
except _APIConnectionError as exc:
_log_codex_request_failure(
agent,
exc,
stream_opened=writer_token["value"] is not None,
)
raise
def _interrupt_or_superseded() -> bool:
return bool(agent._interrupt_requested)
try:
try:
final = _consume_codex_event_stream(
event_stream,
@@ -1280,7 +1746,60 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
agent._client_log_context(), exc,
)
continue
_log_codex_request_failure(
agent,
exc,
stream_opened=writer_token["value"] is not None,
)
raise
except RuntimeError:
if event_stream.final_response is not None:
return event_stream.final_response
raise
except _APIConnectionError as exc:
_log_codex_request_failure(
agent,
exc,
stream_opened=writer_token["value"] is not None,
)
raise
# A terminal response has already been assembled at this point
# (``final`` is built), so a transport error while draining the
# rest of the iterator — done only to let Relay run its response
# finalizer — must NOT discard it or trigger a new physical
# request. Record it as a non-fatal finalization warning and
# still return the already-completed, already-billed response.
if not agent._interrupt_requested:
try:
for _ignored in event_stream:
pass
except (
_httpx.RemoteProtocolError,
_httpx.ReadTimeout,
_httpx.ConnectError,
ConnectionError,
) as exc:
logger.warning(
"Codex Responses stream transport finalization failed "
"after a terminal response was already received; "
"returning the completed response instead of "
"retrying. %s error=%s",
agent._client_log_context(), exc,
)
except _APIConnectionError as exc:
_log_codex_request_failure(
agent,
exc,
stream_opened=writer_token["value"] is not None,
)
logger.warning(
"Codex Responses stream transport finalization failed "
"after a terminal response was already received; "
"returning the completed response instead of "
"retrying. %s error=%s",
agent._client_log_context(), exc,
)
if final.status in {"incomplete", "failed"}:
logger.warning(
@@ -1298,7 +1817,20 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
try:
close_fn()
except Exception:
pass
# A failed close can leave this response's connection
# checked out of the httpx pool while the caller's finally
# reports a reuse-reason close (e.g. interrupt_check broke
# the event loop with collected output) — caching the
# client with the leaked connection. Poison the slot so
# that close really closes the pool (owner-thread abort;
# mirrors the chat-streaming interrupt-break handling).
# ``client is None`` means the shared primary client,
# which is never reuse-cached and must not have its
# sockets force-shut here.
if client is not None:
agent._abort_request_openai_client(
active_client, reason="codex_stream_close_failed"
)
def run_codex_create_stream_fallback(agent, api_kwargs: dict, client: Any = None):
+55 -10
View File
@@ -337,9 +337,9 @@ def _coding_mode(config: Optional[dict[str, Any]]) -> str:
"""Return the normalized ``agent.coding_context`` mode (auto/focus/on/off)."""
if config is None:
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
config = load_config()
config = load_config_readonly()
except Exception:
config = {}
raw = ((config or {}).get("agent", {}) or {}).get("coding_context", "auto")
@@ -520,30 +520,61 @@ class RuntimeMode:
return None
return [self.profile.toolset, *_enabled_mcp_servers(config)]
def system_blocks(self) -> list[str]:
"""Stable system-prompt blocks for this posture (brief + workspace).
def system_prompt_parts(
self, valid_tool_names=None
) -> tuple[list[str], list[str], list[str]]:
"""Return prefix, workspace, and trailing posture blocks separately.
The operating brief carries a model-family edit-format nudge appended
to it (one cached string, not a separate block) so the model is steered
toward the `patch` mode it handles best — see ``_edit_format_line``.
``valid_tool_names`` (when provided) tailors the brief to the session's
toolset: the ``todo`` tracking sentence is dropped when the todo tool
isn't loaded (e.g. Blank Slate), so the brief never references a tool
the model can't call. The toolset is fixed at session construction,
so the rendered brief is deterministic per session — cache-safe.
The three lists preserve the historical flat prompt order: the brief,
the live workspace snapshot, then configured operator instructions.
Prompt assembly can therefore put a cache boundary before the snapshot
without changing the persisted system-prompt bytes.
"""
if not self.is_coding:
return []
blocks: list[str] = []
return [], [], []
prefix: list[str] = []
workspace_parts: list[str] = []
trailing: list[str] = []
if self.profile.guidance:
brief = self.profile.guidance
if valid_tool_names is not None and "todo" not in valid_tool_names:
brief = brief.replace(
"- Track multi-step work with `todo`. Reference code as "
"`path:line` instead of pasting whole files.",
"- Reference code as `path:line` instead of pasting "
"whole files.",
)
edit_line = _edit_format_line(self.model)
if edit_line:
brief = f"{brief}\n{edit_line}"
blocks.append(brief)
prefix.append(brief)
workspace = build_coding_workspace_block(self.cwd)
if workspace:
blocks.append(workspace)
workspace_parts.append(workspace)
# Operator instructions ride their own block so the brief (block 0) stays
# byte-stable and cache-keyed independently of user config.
if self.instructions:
blocks.append(f"Operator instructions (from config):\n{self.instructions}")
return blocks
trailing.append(f"Operator instructions (from config):\n{self.instructions}")
return prefix, workspace_parts, trailing
def system_blocks(self) -> list[str]:
"""Return posture blocks in their historical display order.
``system_prompt_parts`` is the cache-aware API. This compatibility
helper retains the public flat list for callers outside prompt assembly.
"""
prefix, workspace, trailing = self.system_prompt_parts()
return [*prefix, *workspace, *trailing]
def compact_skill_categories(self) -> frozenset[str]:
"""Skill categories to demote to names-only in the prompt's skill index.
@@ -644,6 +675,20 @@ def coding_system_blocks(
).system_blocks()
def coding_system_prompt_parts(
*,
platform: Optional[str] = None,
cwd: Optional[str | Path] = None,
config: Optional[dict[str, Any]] = None,
model: Optional[str] = None,
valid_tool_names=None,
) -> tuple[list[str], list[str], list[str]]:
"""Return coding prefix, workspace snapshot, and trailing guidance."""
return resolve_runtime_mode(
platform=platform, cwd=cwd, config=config, model=model
).system_prompt_parts(valid_tool_names=valid_tool_names)
def coding_compact_skill_categories(
*,
platform: Optional[str] = None,
+190
View File
@@ -0,0 +1,190 @@
"""Mint a provider API key by running a command (``key_cmd``).
Static API keys are the exception at enterprise gateways: SSO/OIDC brokers,
cloud IAM, and internal auth proxies all issue SHORT-LIVED bearers instead.
A key copied into ``.env`` (``key_env``) is stale within the hour, so every
request after that 401s and the user has to restart the session.
``key_cmd`` names a command that PRINTS a token, so the credential is derived
rather than stored::
providers:
my-gateway:
base_url: https://gateway.internal.example.com/v1
api_mode: chat_completions
key_cmd: my-auth-cli print-token --profile prod
This is the established pattern for agent tooling — Claude Code's
``apiKeyHelper``, the ``gcloud auth print-access-token`` / ``aws ecr
get-login-password`` idiom, and vendor helpers such as ``databricks auth
token`` all expose exactly this contract. Hermes already accepts a callable
API key on both wire clients (the Entra ID / Azure identity path) and invokes
it per request, so nothing downstream changes: the token is simply always
fresh. It is cached until shortly before expiry, so the command runs about
once per token lifetime rather than once per request.
Output contract: print ONLY the token on stdout, either bare or as JSON with
an ``access_token`` field (``expires_in`` is honoured when present) — the
shape OAuth 2.0 token endpoints and the helpers above already emit.
Precedence: an explicit ``--api-key`` still wins (the one-off recovery escape
hatch); otherwise ``key_cmd`` is preferred over a static ``api_key`` /
``key_env`` on the same entry.
"""
from __future__ import annotations
import json
import logging
import subprocess
import threading
import time
from typing import Callable, Optional
logger = logging.getLogger(__name__)
# Treat a cached token as spent slightly before its stated expiry, so a request
# can't be signed with a token that dies in flight. 60s matches the leeway used
# by comparable OAuth token caches.
_TOKEN_REFRESH_LEEWAY_SECONDS = 60.0
# A token helper reads a local credential cache and should answer in
# milliseconds; anything approaching this budget is hung, not slow.
_MINT_TIMEOUT_SECONDS = 15
# When a helper advertises NO expiry, the token cannot be cached for the life
# of the process: nothing in the request path re-mints on 401 (the SDK retries
# 429/5xx only), so an expired no-TTL token would 401 every request until
# restart. Re-mint on a bounded window instead — the helper answers from a
# local credential cache in milliseconds, so a periodic re-run is cheap, and a
# helper that wants a longer cache can simply advertise its real expiry.
_NO_TTL_REFRESH_SECONDS = 900.0
class CommandTokenError(RuntimeError):
"""A ``key_cmd`` failed to produce a usable token."""
def _mint(command: str, label: str) -> tuple[str, Optional[float]]:
"""Run *command*, returning ``(token, ttl_seconds_or_None)``."""
try:
completed = subprocess.run(
command,
shell=True,
capture_output=True,
text=True,
timeout=_MINT_TIMEOUT_SECONDS,
)
except subprocess.TimeoutExpired as exc:
raise CommandTokenError(
f"key_cmd for provider {label!r} timed out after "
f"{_MINT_TIMEOUT_SECONDS}s"
) from exc
except OSError as exc:
raise CommandTokenError(
f"key_cmd for provider {label!r} could not be executed: {exc}"
) from exc
if completed.returncode != 0:
# NEVER include stdout/stderr: a partially-successful auth helper can
# print a token or refresh secret there. The command STRING is also
# withheld — a key_cmd can legitimately embed a secret
# (`print-token --client-secret=…`), so echoing it back would leak the
# very credential this module exists to protect. Name the provider so
# the user knows which config entry to run by hand.
raise CommandTokenError(
f"key_cmd for provider {label!r} exited {completed.returncode}. "
f"Run that provider's key_cmd manually to see why "
f"(e.g. `databricks auth login` if its OAuth session expired)."
)
stdout = completed.stdout or ""
if not stdout.strip():
raise CommandTokenError(f"key_cmd for provider {label!r} produced no output")
# JSON payload — the shape `databricks auth token --output json` prints.
# Token extraction mirrors databricks/ucode's get_databricks_token:
# json.loads(result.stdout or "{}").get("access_token", "")
if stdout.lstrip().startswith("{"):
try:
payload = json.loads(stdout)
except json.JSONDecodeError:
payload = None
if isinstance(payload, dict):
token = str(payload.get("access_token") or "").strip()
if not token:
raise CommandTokenError(
f"key_cmd for provider {label!r} returned JSON without an "
"'access_token' field"
)
ttl = payload.get("expires_in")
if isinstance(ttl, (int, float)) and ttl > 0:
return token, float(ttl)
# A relative lifetime is the OAuth 2.0 field, but CLI token helpers
# commonly print an absolute ISO 8601 deadline instead. Treating
# that as "no TTL advertised" caches the token for the life of the
# process, so every request 401s once the deadline passes.
# Imported lazily: hermes_cli.auth imports from agent.* at module
# level, so a top-level import here would risk a cycle.
from hermes_cli.auth import _parse_iso_timestamp
for field in ("expiry", "expiresOn"):
deadline = _parse_iso_timestamp(payload.get(field))
if deadline is not None:
remaining = deadline - time.time()
if remaining > 0:
return token, remaining
return token, None
# Bare token. The contract every comparable helper documents is "stdout
# carries the token and nothing else" — extra output would be consumed as
# part of the credential. Strip surrounding whitespace and take the rest
# verbatim; do NOT silently keep one line of several, which converts a
# misconfigured helper (banner, warning, two tokens) into a corrupt-key 401
# that is far harder to diagnose than an explicit refusal.
token = stdout.strip()
if "\n" in token:
raise CommandTokenError(
f"key_cmd for provider {label!r} printed multiple lines; it must "
"print only the token (or JSON with an 'access_token' field)"
)
return token, None
class CommandTokenSource:
"""Callable returning a bearer token, cached until shortly before expiry."""
def __init__(self, command: str, label: str = "custom") -> None:
self._command = command
self._label = label or "custom"
self._lock = threading.Lock()
self._token = ""
self._expires_at: float = 0.0
def __call__(self) -> str:
with self._lock:
if self._token and time.monotonic() < self._expires_at:
return self._token
token, ttl = _mint(self._command, self._label)
self._token = token
self._expires_at = (
time.monotonic() + max(ttl - _TOKEN_REFRESH_LEEWAY_SECONDS, 5.0)
if ttl
# No advertised TTL: bounded cache (see _NO_TTL_REFRESH_SECONDS)
# — there is no 401-driven re-mint hook to fall back on.
else time.monotonic() + _NO_TTL_REFRESH_SECONDS
)
logger.debug(
"Minted key_cmd token for provider %s (ttl=%s)",
self._label, f"{int(ttl)}s" if ttl else "unknown",
)
return token
def build_command_token_provider(
key_cmd: str,
provider_label: str = "custom",
) -> Optional[Callable[[], str]]:
"""A per-request token provider for *key_cmd*, or ``None`` when unset."""
command = str(key_cmd or "").strip()
if not command:
return None
return CommandTokenSource(command, provider_label)
+47
View File
@@ -0,0 +1,47 @@
"""Client-facing projection helpers for model-only compaction carriers."""
from __future__ import annotations
from typing import Any, Dict, Optional
from agent.context_compressor import (
ContextCompressor,
is_compaction_summary_message,
)
_COMPACTION_INTERNAL_FIELDS = (
"tool_calls",
"finish_reason",
"reasoning",
"reasoning_content",
"reasoning_details",
"codex_reasoning_items",
"codex_message_items",
)
def project_compaction_message_for_display(
message: Dict[str, Any],
) -> Optional[Dict[str, Any]]:
"""Return authentic transcript content, or ``None`` for a pure handoff.
Model-facing recovery history retains the complete carrier. Display
projections instead remove the handoff, inherited tool state, and internal
reasoning while preserving any real prior-tail content or live user ask
embedded in the carrier.
"""
if not isinstance(message, dict):
return None
if not is_compaction_summary_message(message):
return message.copy()
projected = ContextCompressor._strip_context_summary_handoff_message(message)
if projected is None:
return None
projected = projected.copy()
for key in _COMPACTION_INTERNAL_FIELDS:
projected.pop(key, None)
projected.pop("display_kind", None)
return projected
+204
View File
@@ -154,3 +154,207 @@ def compute_session_context_breakdown(
"estimated_total": estimated_total,
"model": getattr(agent, "model", "") or "",
}
# ── /context rendering (CLI + gateway) ──────────────────────────────────────
#
# Pure text renderers over the payload above. The CLI shows a glyph block-grid
# plus a category table; the gateway uses the same table without the grid
# (proportional monospace is not guaranteed on messaging platforms).
_CATEGORY_GLYPHS = {
"system_prompt": "■",
"tool_definitions": "▣",
"rules": "▩",
"skills": "▤",
"mcp": "▥",
"subagent_definitions": "▦",
"memory": "▧",
"conversation": "▨",
}
_FREE_GLYPH = "·"
_GRID_COLUMNS = 20
_GRID_ROWS = 5 # 100 cells → 1 cell per percent of the context window
# Human-readable tables cap the expanded listings; nothing is dropped from
# the underlying data.
_DETAILS_TABLE_LIMIT = 15
def _bytes_to_tokens(size: Optional[int]) -> Optional[int]:
if size is None:
return None
return (int(size) + 3) // 4
def compute_context_details(agent: Any) -> Dict[str, Any]:
"""Expanded per-skill / per-toolset cost listing for ``/context all``.
Reuses the ``hermes prompt-size`` attribution mechanism (PR #66656):
per-skill index-line bytes parsed from the live ``<available_skills>``
block, and per-toolset schema bytes attributed via the tool registry's
canonical tool→toolset map. Byte figures are converted to the same
chars/4 token heuristic the categories above use.
"""
from hermes_cli.prompt_size import (
_compute_skills_breakdown,
_compute_toolsets_breakdown,
)
from agent.system_prompt import build_system_prompt_parts
parts = build_system_prompt_parts(agent)
stable = parts.get("stable", "") or ""
skills_match = _SKILLS_BLOCK_RE.search(stable)
skills_block = skills_match.group(0) if skills_match else ""
skills: List[Dict[str, Any]] = []
if skills_block:
for entry in _compute_skills_breakdown(skills_block):
skills.append({
"name": entry.get("name", ""),
"index_tokens": _bytes_to_tokens(entry.get("index_line_bytes")) or 0,
"skill_md_tokens": _bytes_to_tokens(entry.get("skill_md_bytes")),
})
toolsets: List[Dict[str, Any]] = []
tools = list(getattr(agent, "tools", None) or [])
if tools:
for group in _compute_toolsets_breakdown(tools):
toolsets.append({
"toolset": group.get("toolset", ""),
"tool_count": int(group.get("tool_count", 0) or 0),
"schema_tokens": _bytes_to_tokens(group.get("json_bytes")) or 0,
})
return {"skills": skills, "toolsets": toolsets}
def render_context_grid(payload: Dict[str, Any]) -> List[str]:
"""Render the payload as a Claude Code-style glyph block grid.
100 cells (5×20), each one percent of the model context window. Categories
fill in declaration order; the remainder renders as free space.
"""
context_max = int(payload.get("context_max") or 0)
categories = payload.get("categories") or []
total_cells = _GRID_COLUMNS * _GRID_ROWS
cells: List[str] = []
if context_max > 0:
for cat in categories:
tokens = int(cat.get("tokens") or 0)
n = round(tokens / context_max * total_cells)
if tokens > 0 and n == 0:
n = 1 # never render a nonzero category as invisible
glyph = _CATEGORY_GLYPHS.get(str(cat.get("id") or ""), "▪")
cells.extend([glyph] * n)
cells = cells[:total_cells]
cells.extend([_FREE_GLYPH] * (total_cells - len(cells)))
return [
" ".join(cells[row * _GRID_COLUMNS:(row + 1) * _GRID_COLUMNS])
for row in range(_GRID_ROWS)
]
def render_context_category_lines(payload: Dict[str, Any]) -> List[str]:
"""Render the 'Estimated usage by category' table as plain-text lines."""
categories = payload.get("categories") or []
context_max = int(payload.get("context_max") or 0)
estimated_total = int(payload.get("estimated_total") or 0)
denom = context_max or estimated_total
lines = ["Estimated usage by category"]
if not categories:
lines.append(" (no data yet — send a message first)")
return lines
width = max(len(str(cat.get("label") or "")) for cat in categories)
width = max(width, len("Free space"))
for cat in categories:
tokens = int(cat.get("tokens") or 0)
glyph = _CATEGORY_GLYPHS.get(str(cat.get("id") or ""), "▪")
pct = tokens / denom * 100 if denom else 0.0
label = str(cat.get("label") or cat.get("id") or "")
lines.append(f"{glyph} {label:<{width}} {tokens:>9,} tokens {pct:>5.1f}%")
if context_max > 0:
free = max(0, context_max - estimated_total)
pct = free / context_max * 100
lines.append(f"{_FREE_GLYPH} {'Free space':<{width}} {free:>9,} tokens {pct:>5.1f}%")
return lines
def render_context_details_lines(details: Dict[str, Any]) -> List[str]:
"""Render the expanded ``/context all`` per-skill / per-toolset tables."""
lines: List[str] = []
toolsets = details.get("toolsets") or []
if toolsets:
lines.append("Toolsets by schema cost (largest first)")
for group in toolsets[:_DETAILS_TABLE_LIMIT]:
lines.append(
f" {group['toolset']:<24} {group['tool_count']:>3} tools"
f" {group['schema_tokens']:>8,} tokens"
)
remaining = len(toolsets) - _DETAILS_TABLE_LIMIT
if remaining > 0:
lines.append(f" … and {remaining} more")
skills = details.get("skills") or []
if skills:
if lines:
lines.append("")
lines.append("Skills by cost (index = always-on; SKILL.md = cost when loaded)")
for entry in skills[:_DETAILS_TABLE_LIMIT]:
name = str(entry.get("name") or "")
if len(name) > 28:
name = name[:27] + "…"
md = entry.get("skill_md_tokens")
md_str = f"{md:>8,}" if md is not None else f"{'n/a':>8}"
lines.append(
f" {name:<28} index {entry['index_tokens']:>6,}"
f" SKILL.md {md_str} tokens"
)
remaining = len(skills) - _DETAILS_TABLE_LIMIT
if remaining > 0:
lines.append(f" … and {remaining} more")
return lines
def render_context_breakdown_lines(
payload: Dict[str, Any],
*,
details: Optional[Dict[str, Any]] = None,
grid: bool = True,
) -> List[str]:
"""Render the full /context view as plain-text lines.
``grid=True`` (CLI) prepends the glyph block grid; the gateway passes
``grid=False`` and keeps its own gauge. ``details`` (from
:func:`compute_context_details`) appends the expanded listings.
"""
lines: List[str] = []
if grid:
lines.extend(render_context_grid(payload))
lines.append("")
lines.extend(render_context_category_lines(payload))
context_max = int(payload.get("context_max") or 0)
context_used = int(payload.get("context_used") or 0)
if context_max > 0:
pct = int(payload.get("context_percent") or 0)
lines.append("")
lines.append(
f"Context window: {context_used:,} / {context_max:,} tokens ({pct}%)"
)
if details is not None:
detail_lines = render_context_details_lines(details)
if detail_lines:
lines.append("")
lines.extend(detail_lines)
else:
lines.append("")
lines.append("Use /context all for per-skill and per-toolset costs.")
return lines
+4459 -335
View File
File diff suppressed because it is too large Load Diff
+211
View File
@@ -53,6 +53,39 @@ def sanitize_memory_context(memory_context: str) -> str:
)
def automatic_compaction_status_message(
engine: Any,
*,
phase: str,
default_message: str,
**context: Any,
) -> str | None:
"""Resolve host-visible status for an automatic compaction event.
Engines can suppress routine automatic status with
``emit_automatic_compaction_status = False`` or customize it by defining
``get_automatic_compaction_status_message(...)``. Empty strings and
``None`` mean "do not emit a lifecycle status".
"""
if not getattr(engine, "emit_automatic_compaction_status", True):
return None
formatter = getattr(engine, "get_automatic_compaction_status_message", None)
if callable(formatter):
message = formatter(
phase=phase,
default_message=default_message,
**context,
)
else:
message = default_message
if message is None:
return None
message = str(message).strip()
return message or None
class ContextEngine(ABC):
"""Base class all context engines must implement."""
@@ -89,6 +122,12 @@ class ContextEngine(ABC):
protect_first_n: int = 3
protect_last_n: int = 6
# User-visible lifecycle status for automatic host-triggered compaction.
# Alternative engines that treat compaction as routine background
# maintenance can set this false to keep successful automatic passes silent;
# warnings, errors, and explicit manual commands should still surface.
emit_automatic_compaction_status: bool = True
# -- Core interface ----------------------------------------------------
@abstractmethod
@@ -107,6 +146,19 @@ class ContextEngine(ABC):
def should_compress(self, prompt_tokens: int = None) -> bool:
"""Return True if compaction should fire this turn."""
def should_compress_info(self, prompt_tokens: int = None) -> "tuple[bool, str | None]":
"""Return ``(should_compress, reason)``.
The base implementation is backward-compatible: engines that only
implement ``should_compress`` get ``(should_compress(prompt_tokens),
None)``. Concrete engines with richer block reasons (e.g. a
summary-LLM cooldown or an anti-thrashing guard) override this to
surface a human-readable reason so callers can warn the user instead
of silently skipping compression. Added for the silent-overflow
warning fix (#62625) so plugin engines don't raise AttributeError.
"""
return self.should_compress(prompt_tokens), None
@abstractmethod
def compress(
self,
@@ -137,6 +189,144 @@ class ContextEngine(ABC):
host filters unsupported optional arguments by signature.
"""
# -- Optional: proactive tool-result prune -----------------------------
def prune_tool_results_only(
self,
messages: List[Dict[str, Any]],
current_tokens: int | None = None,
) -> tuple[List[Dict[str, Any]], int]:
"""Deterministically trim old tool-result payloads without an LLM call.
Runs on a low, cost-oriented trigger independent of ``should_compress``
so large-window engines can reclaim re-sent tool output long before full
compaction would fire. Returns ``(messages, n_pruned)``.
Default is a safe no-op: the list is returned unchanged with ``0``
pruned. Engines that don't implement a cheap prune — and any engine that
predates this hook — inherit this default, so the agent loop's
post-tool-call prune path never raises ``AttributeError`` on them. The
built-in ContextCompressor overrides this with the real implementation.
"""
return messages, 0
# -- Optional: per-turn context selection (distinct from compression) --
def select_context(
self,
request_messages: List[Dict[str, Any]],
*,
conversation_messages: List[Dict[str, Any]] = None,
incoming_message: Dict[str, Any] = None,
budget_tokens: int = 0,
) -> List[Dict[str, Any]]:
"""Optionally choose/replace the context for THIS request, pre-generation.
Called every turn after the request message list is assembled and
before it is dispatched to the provider — independent of
``should_compress()``. This lets an engine *select* which context
enters the prompt (retrieval, topic routing, role/branch switching)
rather than *shrink* context that is already there. The two verbs are
orthogonal:
- ``compress()`` : context is too long -> make it shorter.
- ``select_context()``: this turn belongs to a different context
-> use that one instead.
Without this hook, engines that need per-turn access to the message
list have to force ``should_compress()`` to return ``True`` so that
``compress()`` is invoked every turn purely as a callback — which
conflates selection with compression and degrades behaviour when the
engine's backend is unavailable. ``select_context()`` removes the need
for that workaround.
The returned list is request-only: it replaces the messages sent to
the provider for this single call and MUST NOT be treated as persisted
transcript state. The conversation history in the session DB is left
untouched, so nothing leaks across turns. Return ``None`` to leave the
request unchanged.
Unlike the ``pre_llm_call`` plugin hook (which appends to the user
message and intentionally never rewrites the list, to preserve the
cache prefix), ``select_context()`` may *replace* the message list.
Ordering / cache contract: the host runs this hook **before** prompt
cache-control and **before** every request sanitizer (orphaned-tool
cleanup, thinking-only/role normalization, whitespace/JSON
normalization). So (a) whatever the hook returns still passes through
the same validation as any request — a malformed replacement cannot
reach the provider — and (b) prompt-cache stability (an AGENTS.md
invariant) is preserved: the default no-op leaves the request
byte-identical, so cache behaviour is unchanged for the built-in
compressor and any non-implementing engine. An engine that *does*
replace the list changes its own cache prefix by definition; that is
the engine's concern, and cache-control breakpoints are re-derived on
the selected list. The hook is evaluated per provider request (so it
re-runs on retries within a turn), consistent with "select the context
for THIS request".
Args:
request_messages: The assembled request message list (system
prompt + history + any ephemeral prefill), in OpenAI format.
conversation_messages: The unmodified persisted conversation
history, for reference only (do not mutate).
incoming_message: The current turn's user message, if available.
budget_tokens: The active model's context length, or 0 if unknown.
Default returns ``None`` (no-op) — zero impact on the built-in
compressor or any existing engine.
"""
return None
def on_turn_complete(
self,
messages: List[Dict[str, Any]],
usage: Dict[str, Any] = None,
**kwargs: Any,
) -> None:
"""Observe a finished user turn (post-turn ingestion / observation).
Called from the standard turn-finalization path once the assistant/tool
loop completes, with the finalized in-memory transcript snapshot. This
is the complement to ``select_context()``: selection happens *before*
the request, while observation happens *after* the turn. It lets an
engine ingest, index, summarize, or update routing / topic / session
state from what actually happened — so the next ``select_context()``
can act on it.
Coverage: this fires from the normal finalization seam. Some abnormal
early-return paths in the loop (e.g. a content-policy block or a
provider terminal failure) persist and return without routing through
finalization, and therefore do not currently emit this hook. Treat it
as a best-effort post-turn observation for completed turns, not a
guaranteed callback for every possible early exit; unifying all
terminal paths behind one finalization seam is a separate follow-up.
Together the two hooks remove the need to abuse ``should_compress()`` /
``compress()`` as a generic per-turn callback just to observe history,
and they cover the case where a turn finishes and there may be no next
request from which to infer the previous turn.
``messages`` is a shallow copy and should be treated as read-only:
return values are ignored and this hook must not rely on transcript
mutation for persistence. ``kwargs`` may include ``turn_id``,
``task_id``, ``api_call_count``, ``interrupted``, ``failed``, and
``turn_exit_reason``.
``usage`` carries the completed turn's canonical token usage (the same
dict shape passed to ``update_from_response`` — ``prompt_tokens`` /
``completion_tokens`` / ``total_tokens`` plus the canonical
``input_tokens`` / ``output_tokens`` / ``cache_read_tokens`` /
``cache_write_tokens`` / ``reasoning_tokens`` buckets) so an engine can
weigh how large/expensive the selected context actually was when
deciding the next ``select_context()``. It is ``None`` on finalized
turns that never reached a provider response (e.g. interrupt); engines
must treat it as optional.
Default is a no-op.
"""
return None
# -- Optional: pre-flight check ----------------------------------------
def should_compress_preflight(self, messages: List[Dict[str, Any]]) -> bool:
@@ -156,6 +346,27 @@ class ContextEngine(ABC):
"""
return False
def get_automatic_compaction_status_message(
self,
*,
phase: str,
default_message: str,
**context: Any,
) -> str | None:
"""Return user-visible status for automatic host-triggered compaction.
Return ``None`` to suppress successful automatic lifecycle status for
this compaction event. ``phase`` identifies the host call site (for
example ``"preflight"`` or ``"compress"``). ``context`` contains
best-effort fields such as ``approx_tokens`` and ``threshold_tokens``.
This hook does not control warning/error messages or explicit manual
commands such as ``/compress``.
"""
if not self.emit_automatic_compaction_status:
return None
return default_message
# -- Optional: manual /compress preflight ------------------------------
def has_content_to_compress(self, messages: List[Dict[str, Any]]) -> bool:
+148 -26
View File
@@ -13,12 +13,82 @@ from typing import Awaitable, Callable
from agent.model_metadata import estimate_tokens_rough
from hermes_cli._subprocess_compat import IS_WINDOWS, windows_hide_flags
from hermes_cli.sizefmt import format_bytes
from abc import ABC, abstractmethod
# ---------------------------------------------------------------------------
# Plugin context-reference provider API (Issue #26193)
# ---------------------------------------------------------------------------
BUILTIN_PREFIXES = frozenset({"diff", "staged", "file", "folder", "git", "url"})
_context_reference_providers: dict[str, "ContextReferenceProvider"] = {}
class ContextCompletionItem:
"""A single autocomplete result from a context reference provider."""
__slots__ = ("text", "display", "meta")
def __init__(self, text: str, display: str = "", meta: str = "") -> None:
self.text = text
self.display = display or text
self.meta = meta
class ContextReferenceProvider(ABC):
"""Base class for plugin-registered @-prefix context reference providers.
Plugins subclass this and register via
``PluginContext.register_context_reference()``.
"""
prefix: str = "" # e.g. "issue", "channel", "doc"
description: str = "" # shown in autocomplete meta column
@abstractmethod
async def autocomplete(self, query: str, *, limit: int = 10) -> list[ContextCompletionItem]:
"""Return autocomplete items for the given query string."""
...
@abstractmethod
async def expand(self, target: str) -> str | None:
"""Expand *target* to prompt content. Return ``None`` to skip."""
...
def register_context_reference_provider(provider: ContextReferenceProvider) -> None:
"""Register a plugin context reference provider."""
if not isinstance(provider, ContextReferenceProvider):
raise TypeError("provider must be a ContextReferenceProvider instance")
prefix = provider.prefix.lower().strip()
if not prefix:
raise ValueError("prefix must be a non-empty string")
if prefix in BUILTIN_PREFIXES:
raise ValueError(f"prefix '{prefix}' is reserved for built-in references")
if prefix in _context_reference_providers:
raise ValueError(f"prefix '{prefix}' is already registered")
_context_reference_providers[prefix] = provider
def get_context_reference_providers() -> dict[str, ContextReferenceProvider]:
"""Return a snapshot of all registered plugin providers."""
return dict(_context_reference_providers)
_QUOTED_REFERENCE_VALUE = r'(?:`[^`\n]+`|"[^"\n]+"|\'[^\'\n]+\')'
REFERENCE_PATTERN = re.compile(
rf"(?<![\w/])@(?:(?P<simple>diff|staged)\b|(?P<kind>file|folder|git|url):(?P<value>{_QUOTED_REFERENCE_VALUE}(?::\d+(?:-\d+)?)?|\S+))"
)
# Plugin fallback pattern – catches any @<word>:<value> not handled by the
# built-in regex so that plugin-registered prefixes can be resolved.
_PLUGIN_REFERENCE_PATTERN = re.compile(
rf"(?<![\w/])@(?P<kind>[a-zA-Z][a-zA-Z0-9_-]*):(?P<value>{_QUOTED_REFERENCE_VALUE}(?::\d+(?:-\d+)?)?|\S+)"
)
TRAILING_PUNCTUATION = ",.;!?"
_NEEDS_QUOTING = re.compile(r"""[\s()\[\]{}<>"'`]""")
_SENSITIVE_HOME_DIRS = (".ssh", ".aws", ".gnupg", ".kube", ".docker", ".azure", ".config/gh")
_SENSITIVE_HERMES_DIRS = (Path("skills") / ".hub",)
_SENSITIVE_HOME_FILES = (
@@ -60,6 +130,21 @@ class ContextReferenceResult:
blocked: bool = False
def format_reference_value(value: str) -> str:
"""Quote a reference value so ``REFERENCE_PATTERN`` reads it back whole.
The unquoted alternative in the pattern is ``\\S+``, so a path containing a
space parses as a truncated ref with the tail left behind as loose text.
Mirrors ``formatRefValue`` in the desktop's directive-text.tsx.
"""
if not _NEEDS_QUOTING.search(value):
return value
for quote in ("`", '"', "'"):
if quote not in value:
return f"{quote}{value}{quote}"
return value
def parse_context_references(message: str) -> list[ContextReference]:
refs: list[ContextReference] = []
if not message:
@@ -100,6 +185,27 @@ def parse_context_references(message: str) -> list[ContextReference]:
)
)
# Second pass: resolve plugin-registered prefixes the built-in pattern missed
if _context_reference_providers:
for match in _PLUGIN_REFERENCE_PATTERN.finditer(message):
kind = match.group("kind")
if kind in BUILTIN_PREFIXES:
continue
# Skip if already captured by the built-in pattern
if any(r.kind == kind and r.start == match.start() for r in refs):
continue
if kind in _context_reference_providers:
value = _strip_trailing_punctuation(match.group("value") or "")
refs.append(
ContextReference(
raw=match.group(0),
kind=kind,
target=_strip_reference_wrappers(value),
start=match.start(),
end=match.end(),
)
)
return refs
@@ -197,8 +303,12 @@ async def preprocess_context_references_async(
f"@ context injection warning: {injected_tokens} tokens exceeds the 25% soft limit ({soft_limit})."
)
stripped = _remove_reference_tokens(message, refs)
final = stripped
# Leave the `@file:`/`@folder:` tokens where the user typed them. The token
# IS the reference, not scaffolding around it: clients render each one as an
# inline chip, so stripping them left a sentence with a hole in it ("review
# and ship") and made the desktop re-derive the refs from the attached block
# to show them as a detached list above the prose.
final = message
if warnings:
final = f"{final}\n\n--- Context Warnings ---\n" + "\n".join(f"- {warning}" for warning in warnings)
if blocks:
@@ -242,6 +352,16 @@ async def _expand_reference(
except Exception as exc:
return f"{ref.raw}: {exc}", None
# Plugin-provided context references
provider = _context_reference_providers.get(ref.kind)
if provider is not None:
try:
plugin_content = await provider.expand(ref.target)
if plugin_content is not None:
return None, f"📌 {ref.raw} ({estimate_tokens_rough(plugin_content)} tokens)\n{plugin_content}"
except Exception as exc:
return f"{ref.raw}: plugin expansion error: {exc}", None
return f"{ref.raw}: unsupported reference type", None
@@ -308,7 +428,7 @@ def _expand_git_reference(
["git", *args],
cwd=cwd,
capture_output=True,
text=True,
text=True, encoding='utf-8', errors='replace',
timeout=30,
stdin=subprocess.DEVNULL,
**_popen_kwargs,
@@ -457,19 +577,6 @@ def _parse_file_reference_value(value: str) -> tuple[str, int | None, int | None
return _strip_reference_wrappers(value), None, None
def _remove_reference_tokens(message: str, refs: list[ContextReference]) -> str:
pieces: list[str] = []
cursor = 0
for ref in refs:
pieces.append(message[cursor:ref.start])
cursor = ref.end
pieces.append(message[cursor:])
text = "".join(pieces)
text = re.sub(r"\s{2,}", " ", text)
text = re.sub(r"\s+([,.;:!?])", r"\1", text)
return text.strip()
def _is_binary_file(path: Path) -> bool:
mime, _ = mimetypes.guess_type(path.name)
if mime and not mime.startswith("text/") and not any(
@@ -534,7 +641,7 @@ def _rg_files(path: Path, cwd: Path, limit: int) -> list[Path] | None:
["rg", "--files", str(path.relative_to(cwd))],
cwd=cwd,
capture_output=True,
text=True,
text=True, encoding='utf-8', errors='replace',
timeout=10,
stdin=subprocess.DEVNULL,
**_popen_kwargs,
@@ -547,25 +654,40 @@ def _rg_files(path: Path, cwd: Path, limit: int) -> list[Path] | None:
return files[:limit]
def _human_bytes(n: int) -> str:
size = float(n)
for unit in ("B", "KB", "MB", "GB"):
if size < 1024 or unit == "GB":
return f"{int(size)} {unit}" if unit == "B" else f"{size:.1f} {unit}"
size /= 1024
return f"{size:.1f} GB"
def _agent_visible_path(path: Path) -> str:
"""Map a host path to the path the agent's tools can read in the active backend.
Under a container backend (docker) the gateway host path dangles inside the
sandbox — the container has its own filesystem and the host path is not
mounted. Files staged into an auto-mounted cache dir (``images/``,
``attachments/``, ...) are translated to their in-container path via the
existing ``tools.credential_files`` machinery (#76577). Falls back to the
host path when the backend is local or translation is unavailable.
"""
try:
# Desktop/in-process gateways may not have bridged ``terminal.*``
# config into ``TERMINAL_ENV`` at startup; run the idempotent bridge so
# the credential_files translation gate sees the active backend.
from tools.terminal_tool import _ensure_terminal_env_bridged
_ensure_terminal_env_bridged()
from tools.credential_files import to_agent_visible_cache_path
return to_agent_visible_cache_path(str(path))
except Exception:
return str(path)
def _binary_reference_block(ref: ContextReference, path: Path) -> str:
mime, _ = mimetypes.guess_type(path.name)
mime = mime or "application/octet-stream"
try:
size = _human_bytes(path.stat().st_size)
size = format_bytes(path.stat().st_size)
except OSError:
size = "unknown size"
return (
f"📎 {ref.raw} ({mime}, {size}) — binary file, not inlined as text. "
f"It is available on disk at `{path}`. Use your tools to work with it "
f"It is available on disk at `{_agent_visible_path(path)}`. Use your tools to work with it "
f"(read or convert it, extract its text, or view/render it as needed); "
f"do not tell the user the file type is unsupported."
)
File diff suppressed because it is too large Load Diff
+2972 -250
View File
File diff suppressed because it is too large Load Diff
+100 -173
View File
@@ -21,21 +21,18 @@ from pathlib import Path
from types import SimpleNamespace
from typing import Any
from openai.types.chat.chat_completion_message_tool_call import (
ChatCompletionMessageToolCall,
Function,
from agent.acp_openai_bridge import (
completion_to_stream_chunks as _completion_to_stream_chunks,
extract_tool_calls_from_text as _extract_tool_calls_from_text,
render_tool_bridge_sections as _render_tool_bridge_sections,
)
from agent.file_safety import get_read_block_error, get_write_denied_error
from agent.file_safety import get_read_block_error, get_write_denied_error, is_write_approval_required
from agent.redact import redact_sensitive_text
from tools.environments.local import hermes_subprocess_env
ACP_MARKER_BASE_URL = "acp://copilot"
_DEFAULT_TIMEOUT_SECONDS = 900.0
_TOOL_CALL_BLOCK_RE = re.compile(r"<tool_call>\s*(\{.*?\})\s*</tool_call>", re.DOTALL)
_TOOL_CALL_JSON_RE = re.compile(r"\{\s*\"id\"\s*:\s*\"[^\"]+\"\s*,\s*\"type\"\s*:\s*\"function\"\s*,\s*\"function\"\s*:\s*\{.*?\}\s*\}", re.DOTALL)
# Stderr fingerprint of the deprecated `gh copilot` CLI extension
# (https://github.blog/changelog/2025-09-25-upcoming-deprecation-of-gh-copilot-cli-extension).
# We require BOTH the literal product name ("gh-copilot") AND a deprecation
@@ -74,6 +71,60 @@ def _resolve_args() -> list[str]:
return shlex.split(raw)
# Probe verdicts cached per binary path so repeated prompts against a
# CLI that supports --acp pay the ~50ms --help cost exactly once per
# process. Only definitive verdicts (True/False) are cached; an
# inconclusive probe (binary missing, --help crashed or timed out) is
# not cached so a CLI installed mid-session is picked up.
_ACP_PROBE_CACHE: dict[str, bool] = {}
def _acp_supported(command: str, args: list[str]) -> bool | None:
"""Tri-state probe: does ``command`` accept the ACP args we'd pass?
Different CLI versions support different transports. The GitHub
Copilot CLI (`@github/copilot`, late 2025+) ships with ``--acp``;
older releases (and Claude Code v2.x as of Aug 2026) do not.
Spawning a CLI that doesn't recognize the flag silently exits
with code 1 and ``error: unknown option '--acp'`` on stderr,
after which every delegate_task call hangs the parent for
``child_timeout_seconds`` (default 600s) waiting for stdout
that never arrives.
Returns:
- ``True`` — help text advertises ``--acp``; safe to spawn.
- ``False`` — help ran cleanly but ``--acp`` is absent; spawning
would hang, so the caller should fast-fail with a clear error.
- ``None`` — inconclusive (binary missing, --help failed or
timed out). The caller must fall through to the normal spawn
path, which surfaces the existing "Could not start Copilot ACP
command" error with full context.
Only probes when ``--acp`` is actually among ``args``: a custom
HERMES_COPILOT_ACP_ARGS transport is the operator's business.
"""
if "--acp" not in args:
return True
cached = _ACP_PROBE_CACHE.get(command)
if cached is not None:
return cached
try:
probe = subprocess.run(
[command, "--help"],
capture_output=True, text=True, timeout=5,
)
except (FileNotFoundError, subprocess.TimeoutExpired, OSError):
return None
if probe.returncode != 0:
# --help itself failed; can't tell anything about --acp.
return None
# Match ``--acp`` as a flag in the help text; tolerate spacing and
# variants like ``[--acp]``.
verdict = bool(re.search(r"(?:^|[\s\[])--acp(?:[\s=\],]|$)", probe.stdout, re.MULTILINE))
_ACP_PROBE_CACHE[command] = verdict
return verdict
def _resolve_home_dir() -> str:
"""Return a stable HOME for child ACP processes."""
home = os.environ.get("HOME", "").strip()
@@ -149,34 +200,9 @@ def _format_messages_as_prompt(
if model:
sections.append(f"Hermes requested model hint: {model}")
if isinstance(tools, list) and tools:
tool_specs: list[dict[str, Any]] = []
for t in tools:
if not isinstance(t, dict):
continue
fn = t.get("function") or {}
if not isinstance(fn, dict):
continue
name = fn.get("name")
if not isinstance(name, str) or not name.strip():
continue
tool_specs.append(
{
"name": name.strip(),
"description": fn.get("description", ""),
"parameters": fn.get("parameters", {}),
}
)
if tool_specs:
sections.append(
"Available tools (OpenAI function schema). "
"When using a tool, emit ONLY <tool_call>{...}</tool_call> with one JSON object "
"containing id/type/function{name,arguments}. arguments must be a JSON string.\n"
+ json.dumps(tool_specs, ensure_ascii=False)
)
if tool_choice is not None:
sections.append(f"Tool choice hint: {json.dumps(tool_choice, ensure_ascii=False)}")
# Copilot has no tools of its own that would collide with Hermes', so it
# forwards the whole toolset (no allowlist).
sections.extend(_render_tool_bridge_sections(tools, tool_choice))
transcript: list[str] = []
for message in messages:
@@ -233,140 +259,6 @@ def _render_message_content(content: Any) -> str:
return str(content).strip()
def _build_openai_tool_call(
*,
call_id: str,
name: str,
arguments: str,
) -> ChatCompletionMessageToolCall:
"""Build an OpenAI-compatible tool-call object for downstream handling."""
return ChatCompletionMessageToolCall(
id=call_id,
call_id=call_id,
response_item_id=None,
type="function",
function=Function(name=name, arguments=arguments),
)
def _completion_to_stream_chunks(completion: SimpleNamespace) -> list[SimpleNamespace]:
"""Convert a one-shot ACP response into OpenAI-style stream chunks."""
choice = completion.choices[0]
message = choice.message
tool_call_deltas = None
if message.tool_calls:
tool_call_deltas = []
for index, tool_call in enumerate(message.tool_calls):
tool_call_deltas.append(
SimpleNamespace(
index=index,
id=getattr(tool_call, "id", None),
type=getattr(tool_call, "type", "function"),
function=SimpleNamespace(
name=getattr(tool_call.function, "name", None),
arguments=getattr(tool_call.function, "arguments", None),
),
)
)
delta = SimpleNamespace(
role="assistant",
content=message.content or None,
tool_calls=tool_call_deltas,
reasoning_content=message.reasoning_content,
reasoning=message.reasoning,
)
data_chunk = SimpleNamespace(
choices=[
SimpleNamespace(
index=0,
delta=delta,
finish_reason=choice.finish_reason,
)
],
model=completion.model,
usage=None,
)
usage_chunk = SimpleNamespace(
choices=[],
model=completion.model,
usage=completion.usage,
)
return [data_chunk, usage_chunk]
def _extract_tool_calls_from_text(text: str) -> tuple[list[ChatCompletionMessageToolCall], str]:
if not isinstance(text, str) or not text.strip():
return [], ""
extracted: list[ChatCompletionMessageToolCall] = []
consumed_spans: list[tuple[int, int]] = []
def _try_add_tool_call(raw_json: str) -> None:
try:
obj = json.loads(raw_json)
except Exception:
return
if not isinstance(obj, dict):
return
fn = obj.get("function")
if not isinstance(fn, dict):
return
fn_name = fn.get("name")
if not isinstance(fn_name, str) or not fn_name.strip():
return
fn_args = fn.get("arguments", "{}")
if not isinstance(fn_args, str):
fn_args = json.dumps(fn_args, ensure_ascii=False)
call_id = obj.get("id")
if not isinstance(call_id, str) or not call_id.strip():
call_id = f"acp_call_{len(extracted)+1}"
extracted.append(
_build_openai_tool_call(
call_id=call_id,
name=fn_name.strip(),
arguments=fn_args,
)
)
for m in _TOOL_CALL_BLOCK_RE.finditer(text):
raw = m.group(1)
_try_add_tool_call(raw)
consumed_spans.append((m.start(), m.end()))
# Only try bare-JSON fallback when no XML blocks were found.
if not extracted:
for m in _TOOL_CALL_JSON_RE.finditer(text):
raw = m.group(0)
_try_add_tool_call(raw)
consumed_spans.append((m.start(), m.end()))
if not consumed_spans:
return extracted, text.strip()
consumed_spans.sort()
merged: list[tuple[int, int]] = []
for start, end in consumed_spans:
if not merged or start > merged[-1][1]:
merged.append((start, end))
else:
merged[-1] = (merged[-1][0], max(merged[-1][1], end))
parts: list[str] = []
cursor = 0
for start, end in merged:
if cursor < start:
parts.append(text[cursor:start])
cursor = max(cursor, end)
if cursor < len(text):
parts.append(text[cursor:])
cleaned = "\n".join(p.strip() for p in parts if p and p.strip()).strip()
return extracted, cleaned
def _ensure_path_within_cwd(path_text: str, cwd: str) -> Path:
candidate = Path(path_text)
if not candidate.is_absolute():
@@ -502,16 +394,43 @@ class CopilotACPClient:
return completion
def _run_prompt(self, prompt_text: str, *, timeout_seconds: float) -> tuple[str, str]:
# Fast-fail when the CLI doesn't support the ACP args we'd pass.
# Without this guard, a CLI like Claude Code v2.x exits with
# ``error: unknown option '--acp'`` immediately, then the parent
# ACP loop waits the full ``child_timeout_seconds`` (default 600s)
# for stdout that never arrives. The probe costs ~50ms and turns
# a 600s silent hang into a 280ms clear error.
# ``None`` (inconclusive probe — e.g. binary missing) falls
# through to the spawn below, which raises the established
# "Could not start Copilot ACP command" error.
if _acp_supported(self._acp_command, self._acp_args) is False:
preview = " ".join(self._acp_args[:3]) if self._acp_args else "(none)"
raise RuntimeError(
f"ACP transport not supported by '{self._acp_command}': "
f"`{preview}` is rejected as an unknown option. "
f"This usually means the CLI is an older release (e.g. "
f"Claude Code v2.x) or a different tool than expected. "
f"Either install a CLI that ships with --acp support "
f"(e.g. `@github/copilot` late 2025+), or set "
f"HERMES_COPILOT_ACP_COMMAND / HERMES_COPILOT_ACP_ARGS "
f"to a working pair."
)
try:
# Hide the console the CLI child would otherwise flash on Windows
# (#56747). Hide-only — stdio pipes stay intact for the ACP wire.
from hermes_cli._subprocess_compat import windows_hide_flags
proc = subprocess.Popen(
[self._acp_command] + self._acp_args,
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
text=True, encoding='utf-8', errors='replace',
bufsize=1,
cwd=self._acp_cwd,
env=_build_subprocess_env(),
creationflags=windows_hide_flags(),
)
except FileNotFoundError as exc:
raise RuntimeError(
@@ -703,7 +622,7 @@ class CopilotACPClient:
if block_error:
raise PermissionError(block_error)
try:
content = path.read_text()
content = path.read_text(encoding="utf-8")
except FileNotFoundError:
content = ""
line = params.get("line")
@@ -730,8 +649,16 @@ class CopilotACPClient:
denied = get_write_denied_error(str(path))
if denied:
raise PermissionError(denied)
# Approval-gated paths (e.g. ~/.ssh/config) are not hard-denied
# for interactive tools, but the ACP shim has no human channel
# to confirm the write — fail closed here.
if is_write_approval_required(str(path)):
raise PermissionError(
f"Write denied: '{path}' requires interactive approval "
"and cannot be written through the ACP file bridge."
)
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(str(params.get("content") or ""))
path.write_text(str(params.get("content") or ""), encoding="utf-8")
response = {
"jsonrpc": "2.0",
"id": message_id,
+896 -240
View File
File diff suppressed because it is too large Load Diff
+9 -1
View File
@@ -162,9 +162,17 @@ def _remove_env_source(provider: str, removed) -> RemovalResult:
try:
env_path = get_env_path()
if env_path.exists():
# Read the .env as UTF-8 with BOM tolerance, matching the
# canonical reader in hermes_cli/config.py. read_text() with no
# encoding falls back to the system locale (cp1252/GBK on Windows)
# and never strips a BOM, so a Notepad-edited .env (BOM + non-ASCII
# values) would make the first line fail the startswith() check —
# misreporting a .env-backed var as a shell export.
env_in_dotenv = any(
line.strip().startswith(f"{env_var}=")
for line in env_path.read_text(errors="replace").splitlines()
for line in env_path.read_text(
encoding="utf-8-sig", errors="replace"
).splitlines()
)
except OSError:
pass
+73 -7
View File
@@ -170,6 +170,27 @@ CREDITS_USAGE_BANDS: tuple[tuple[float, str, int], ...] = (
)
CREDITS_USAGE_KEY = "credits.usage" # single key for the escalating usage notice
# Minimum subscription balance that counts as "grant not yet spent" for the
# grant_spent crossing gate (see evaluate_credits_notices). 1¢: portal-seeded
# states derive micros from float dollars and can carry sub-cent residue where
# the inference headers report exactly 0 — without this floor such a seed
# opens the gate and the first header re-creates the at-open nag.
GRANT_UNSPENT_MIN_MICROS = 10_000
def new_credits_latch() -> dict:
"""Fresh notice latch in the shape :func:`evaluate_credits_notices` expects.
The policy owns this schema — every producer (agent build, lazy re-init,
tests) must build the latch through here so a new gate key lands everywhere
at once instead of drifting across hand-rolled literals."""
return {
"active": set(),
"seen_below_90": False,
"usage_band": None,
"seen_grant_unspent": False,
}
# ── AgentNotice (out-of-band notice payload; driver-agnostic) ────────────────
@@ -205,12 +226,15 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool:
1. The ``:free`` suffix — the canonical Nous free SKU marker (e.g.
``nvidia/nemotron-3-ultra:free``). Free by construction on the API side
(spend is forced to 0 for ``:free`` ids).
2. A peek into the in-process pricing cache in ``hermes_cli.models``
2. The ``stealth/`` prefix — Nous stealth-preview SKUs (e.g.
``stealth/ox-alpha``) are free-tier but carry no ``:free`` suffix. Spend
is forced to zero server-side, so these are also free by construction.
3. A peek into the in-process pricing cache in ``hermes_cli.models``
(populated when the model picker fetched ``/v1/models`` pricing for
*base_url*). PEEK ONLY — a cache miss never triggers a fetch. This is
CLI/TUI-session best-effort: gateway sessions never run the picker's
pricing fetch, so suppression there rests entirely on the ``:free``
suffix (which all Nous free SKUs carry).
suffix and ``stealth/`` prefix.
Fail-open to False (the depleted notice still shows) on any error: wrongly
showing the warning is recoverable noise; wrongly hiding it on a paid model
@@ -220,6 +244,11 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool:
return False
if model.endswith(":free"):
return True
# Stealth-preview SKUs are free-tier but carry no ``:free`` suffix (see
# docstring point 2). Naming-convention trust: if a PAID model ever shipped
# under ``stealth/`` this would wrongly suppress the banner on it.
if model.startswith("stealth/"):
return True
if not base_url:
return False
try:
@@ -250,7 +279,8 @@ def evaluate_credits_notices(
) -> tuple[list[AgentNotice], list[str]]:
"""Reconcile credits notices against the latch. Mutates ``latch`` IN PLACE.
latch = {"active": set[str], "seen_below_90": bool, "usage_band": Optional[int]}.
latch = {"active": set[str], "seen_below_90": bool, "usage_band": Optional[int],
"seen_grant_unspent": bool}.
``model_is_free``: True when the session's active model is a Nous free-tier
model (see :func:`is_free_tier_model`). Suppresses the ``credits.depleted``
@@ -277,6 +307,18 @@ def evaluate_credits_notices(
if uf is not None and uf < _lowest_band:
latch["seen_below_90"] = True # gate opened: usage-band notices may now fire
# Grant-spent crossing gate: grant_spent may fire only after this session
# has OBSERVED the grant meaningfully unspent (≥1¢ left — see
# GRANT_UNSPENT_MIN_MICROS). Opening at grant-spent is a steady STATE, not
# an event — /usage carries it; only a live in-session crossing announces.
# Unlike seen_below_90, seeds must NOT prime this gate.
if (
uf is not None
and uf < 1.0
and state.subscription_micros >= GRANT_UNSPENT_MIN_MICROS
):
latch["seen_grant_unspent"] = True
active = latch["active"]
# ── Conditions ───────────────────────────────────────────────────────────
@@ -316,12 +358,21 @@ def evaluate_credits_notices(
active.discard(CREDITS_USAGE_KEY)
if target_band is not None:
# Belt-and-suspenders: a producer could set subscription_limit_micros
# without subscription_limit_usd. Render "$? cap" rather than "$None cap".
# without subscription_limit_usd. Render "$?" rather than "$None".
_cap_usd = state.subscription_limit_usd or "?"
_level = current_band[1] # type: ignore[index] (current_band set when target_band set)
# Report absolute dollars used, not a bare "N% used": the percentage is
# only meaningful against a Nous subscription cap (no cap → never fires),
# so dollars are clearer and don't imply a universal %. Used = cap −
# remaining (micros, money-safe), clamped to [0, cap]. Re-emits on band
# change (50 → 75 → 90), not every turn — a snapshot, not a live ticker.
_lim = state.subscription_limit_micros or 0
_used_micros = max(0, min(_lim, _lim - state.subscription_micros))
_used_usd = f"{_used_micros / 1_000_000:.2f}" if _lim else "?"
_glyph = "⚠" if _level == "warn" else "•"
to_show.append(
AgentNotice(
text=f"{'⚠' if _level == 'warn' else '•'} Credits {target_band}% used · ${_cap_usd} cap",
text=f"{_glyph} You've used ${_used_usd} of your ${_cap_usd} cap",
level=_level,
kind=CREDITS_NOTICE_KIND,
key=CREDITS_USAGE_KEY,
@@ -332,7 +383,17 @@ def evaluate_credits_notices(
latch["usage_band"] = target_band
# ── grant_spent ──────────────────────────────────────────────────────────
if grant_cond and "credits.grant_spent" not in active:
# The crossing gate guards only the SHOW and is CONSUMED by it — one
# announcement per crossing. A header flicker (uf → None → back to 1.0)
# clears the sticky line via grant_cond but cannot re-announce; only a
# renewal that re-opens the gate (a fresh ≥1¢ observation) arms the next
# announcement. .get(): default closed for any hand-built latch missing
# the key, so a first observation can never fire this notice.
if (
grant_cond
and "credits.grant_spent" not in active
and latch.get("seen_grant_unspent", False)
):
to_show.append(
AgentNotice(
text=f"• Grant spent · ${state.purchased_usd} top-up left",
@@ -343,6 +404,7 @@ def evaluate_credits_notices(
)
)
active.add("credits.grant_spent")
latch["seen_grant_unspent"] = False
elif "credits.grant_spent" in active and not grant_cond:
to_clear.append("credits.grant_spent")
active.discard("credits.grant_spent")
@@ -618,7 +680,8 @@ _DEV_FIXTURES: dict[str, dict] = {
subscription_limit_micros=20_000_000, subscription_limit_usd="20.00",
denominator_kind="subscription_cap", paid_access=True,
),
"grant_exhausted": dict( # used_fraction == 1.0 + purchased>0 → credits.grant_spent
"grant_exhausted": dict( # uf == 1.0 + purchased>0 → SILENT at open (crossing-gated);
# flip healthy → grant_exhausted via the fixture-file path to see credits.grant_spent
remaining_micros=12_340_000, remaining_usd="12.34",
subscription_micros=0, subscription_usd="0.00",
subscription_limit_micros=20_000_000, subscription_limit_usd="20.00",
@@ -732,6 +795,9 @@ def _hydrate_seed_state(agent, state) -> None:
agent._credits_session_start_micros = state.remaining_micros
_latch = getattr(agent, "_credits_latch", None)
if isinstance(_latch, dict) and state.used_fraction is not None:
# Prime ONLY seen_below_90 (open-high band warnings are wanted at open).
# Never prime seen_grant_unspent here: a seed observing grant-spent is a
# steady state, and priming it would revive the every-session nag.
_latch["seen_below_90"] = True
emit = getattr(agent, "_emit_credits_notices", None)
if callable(emit):
+42 -17
View File
@@ -138,8 +138,8 @@ def is_paused() -> bool:
def _load_config() -> Dict[str, Any]:
"""Read curator.* config from ~/.hermes/config.yaml. Tolerates missing file."""
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e:
logger.debug("Failed to load config for curator: %s", e)
return {}
@@ -325,7 +325,7 @@ def apply_automatic_transitions(now: Optional[datetime] = None) -> Dict[str, int
counts = {"marked_stale": 0, "archived": 0, "reactivated": 0, "checked": 0, "seeded": 0}
for row in _u.agent_created_report():
for row in _u.curated_report():
counts["checked"] += 1
name = row["name"]
if row.get("pinned"):
@@ -369,7 +369,22 @@ def apply_automatic_transitions(now: Optional[datetime] = None) -> Dict[str, int
continue
if anchor <= archive_cutoff and current != _u.STATE_ARCHIVED:
ok, _msg = _u.archive_skill(name)
# Tag the ledger entry with the curator actor: this archive is an
# autonomous curator transition, not a foreground agent/user call.
try:
from tools.skill_ledger import reset_ledger_actor, set_ledger_actor
_tok = set_ledger_actor("curator")
except Exception:
_tok = None
reset_ledger_actor = None # type: ignore[assignment]
try:
ok, _msg = _u.archive_skill(name)
finally:
if _tok is not None and reset_ledger_actor is not None:
try:
reset_ledger_actor(_tok)
except Exception:
pass
if ok:
counts["archived"] += 1
elif anchor <= stale_cutoff and current == _u.STATE_ACTIVE:
@@ -422,7 +437,9 @@ CURATOR_REVIEW_PROMPT = (
"INSTRUCTIONS AND EXPERIENTIAL KNOWLEDGE. A collection of hundreds of "
"narrow skills where each one captures one session's specific bug is "
"a FAILURE of the library — not a feature. An agent searching skills "
"matches on descriptions, not on exact names; one broad umbrella "
"matches on descriptions, not on exact names (note: long descriptions "
"are truncated to 57 chars in the system prompt skill index — keep the "
"trigger class in that window). One broad umbrella "
"skill with labeled subsections beats five narrow siblings for "
"discoverability, not the other way around.\n\n"
"The right target shape is CLASS-LEVEL skills with rich SKILL.md "
@@ -522,6 +539,13 @@ CURATOR_REVIEW_PROMPT = (
"merges.\n\n"
"Your toolset:\n"
" - skills_list, skill_view — read the current landscape\n"
" READ BEFORE WRITE — enforced, not advisory. Before skill_manage "
"action=patch, action=edit, action=write_file on a file that already "
"exists, or action=remove_file, call skill_view on that SAME target in "
"this review turn — skill_view(name) for SKILL.md, "
"skill_view(name, file_path=...) for a supporting file — and build the "
"write from the content it just returned. A write without that read is "
"REFUSED and nothing is saved.\n"
" - skill_manage action=patch — add sections to the umbrella\n"
" - skill_manage action=create — create a new umbrella SKILL.md\n"
" - skill_manage action=write_file — add a references/, templates/, "
@@ -900,7 +924,6 @@ def _reconcile_classification(
Every removed skill is placed in exactly one bucket.
"""
heur_cons = {e["name"]: e for e in heuristic.get("consolidated", [])}
heur_pruned = {e["name"] for e in heuristic.get("pruned", [])}
model_cons = {e["from"]: e for e in model_block.get("consolidations", [])}
model_pruned = {e["name"]: e for e in model_block.get("prunings", [])}
@@ -1470,15 +1493,16 @@ def _render_report_markdown(p: Dict[str, Any]) -> str:
# ---------------------------------------------------------------------------
def _render_candidate_list() -> str:
"""Human/agent-readable list of agent-created skills with usage stats."""
rows = skill_usage.agent_created_report()
"""Human/agent-readable list of curator-managed skills with usage stats."""
rows = skill_usage.curated_report()
if not rows:
return "No agent-created skills to review."
return "No curator-managed skills to review."
cron_referenced = _cron_referenced_skills()
lines = [f"Agent-created skills ({len(rows)}):\n"]
lines = [f"Curator-managed skills ({len(rows)}):\n"]
for r in rows:
lines.append(
f"- {r['name']} "
f"provenance={r.get('provenance', 'agent')} "
f"state={r['state']} "
f"pinned={'yes' if r.get('pinned') else 'no'} "
f"cron={'yes' if r['name'] in cron_referenced else 'no'} "
@@ -1531,7 +1555,7 @@ def run_curator_review(
if dry_run:
# Count candidates without mutating state.
try:
report = skill_usage.agent_created_report()
report = skill_usage.curated_report()
counts = {
"checked": len(report),
"marked_stale": 0,
@@ -1584,7 +1608,7 @@ def run_curator_review(
nonlocal auto_summary
# Snapshot skill state BEFORE the LLM pass so the report can diff.
try:
before_report = skill_usage.agent_created_report()
before_report = skill_usage.curated_report()
except Exception:
before_report = []
before_names = {r.get("name") for r in before_report if isinstance(r, dict)}
@@ -1610,7 +1634,7 @@ def run_curator_review(
state2["last_run_duration_seconds"] = elapsed
state2["last_run_summary"] = final_summary
try:
after_report = skill_usage.agent_created_report()
after_report = skill_usage.curated_report()
except Exception:
after_report = []
try:
@@ -1697,7 +1721,7 @@ def run_curator_review(
try:
rename_lines = _build_rename_summary(
before_names=before_names,
after_report=skill_usage.agent_created_report(),
after_report=skill_usage.curated_report(),
tool_calls=llm_meta.get("tool_calls", []) or [],
model_final=llm_meta.get("final", "") or "",
)
@@ -1715,7 +1739,7 @@ def run_curator_review(
# reporting bug never breaks the curator itself. Report path is
# recorded in state so `hermes curator status` can point at it.
try:
after_report = skill_usage.agent_created_report()
after_report = skill_usage.curated_report()
except Exception:
after_report = []
try:
@@ -1873,9 +1897,9 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
_acp_args = None
_model_name = ""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
from hermes_cli.runtime_provider import resolve_runtime_provider
_cfg = load_config()
_cfg = load_config_readonly()
_binding = _resolve_review_runtime(_cfg)
_provider, _model_name = _binding.provider, _binding.model
_rp = resolve_runtime_provider(
@@ -1921,6 +1945,7 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
credential_pool=_credential_pool,
request_overrides=_request_overrides,
**_agent_kwargs,
enabled_toolsets=["skills", "terminal"],
# Umbrella-building over a large skill collection is worth a
# high iteration ceiling — the pass typically takes 50-100
# API calls against hundreds of candidate skills. The
+61 -19
View File
@@ -50,6 +50,7 @@ from typing import Any, Dict, List, Optional, Set, Tuple
from hermes_constants import get_hermes_home
from agent.skill_utils import is_excluded_skill_path
from hermes_cli.sizefmt import format_bytes
logger = logging.getLogger(__name__)
@@ -147,8 +148,8 @@ def _utc_id(now: Optional[datetime] = None) -> str:
def _load_config() -> Dict[str, Any]:
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e:
logger.debug("Failed to load config for curator backup: %s", e)
return {}
@@ -541,6 +542,33 @@ def _restore_cron_skill_links(snapshot_dir: Path) -> Dict[str, Any]:
def _unstage(moved: List[Tuple[Path, Path]]) -> List[str]:
"""Move staged entries back to their original paths.
``shutil.move`` moves *into* an existing destination directory rather than
replacing it, so a partially-completed extract leaves debris that would
otherwise bury the user's real skill one level deeper
(``skills/foo/foo/``) while the tree still looks populated. Clear whatever
the failed extract created at each original path first. The staged copy is
authoritative, and the pre-rollback safety snapshot is the undo handle for
the extract's own output.
Returns the names that could not be restored, so the caller can report an
incomplete recovery instead of claiming the state was restored.
"""
failed: List[str] = []
for orig, dest in moved:
try:
if orig.is_dir() and not orig.is_symlink():
shutil.rmtree(orig)
elif orig.exists() or orig.is_symlink():
orig.unlink()
shutil.move(str(dest), str(orig))
except OSError:
failed.append(orig.name)
return failed
def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]]:
"""Restore ``~/.hermes/skills/`` from a snapshot.
@@ -582,12 +610,19 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]
# Protect the target from this snapshot's prune step: at the steady
# keep limit, pruning the oldest snapshot would otherwise delete the
# very snapshot we are about to extract from.
snapshot_skills(
safety_snapshot = snapshot_skills(
reason=f"pre-rollback to {target.name}",
protect_ids={target.name},
)
except Exception as e:
return (False, f"pre-rollback safety snapshot failed: {e}", None)
if safety_snapshot is None:
return (
False,
"pre-rollback safety snapshot failed; backups may be disabled "
"or unavailable, and current skills were not changed",
None,
)
# Additionally move current entries into an internal staging dir so
# the extract happens into an empty skills tree (predictable result).
@@ -609,11 +644,7 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]
moved.append((entry, dest))
except OSError as e:
# Best-effort rollback of the move
for orig, dest in moved:
try:
shutil.move(str(dest), str(orig))
except OSError:
pass
_unstage(moved)
try:
shutil.rmtree(staged, ignore_errors=True)
except OSError:
@@ -638,12 +669,30 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]
# Python < 3.12 — no filter kwarg
tf.extractall(str(skills))
except (OSError, tarfile.TarError) as e:
# Best-effort recover: move staged contents back
for orig, dest in moved:
# Best-effort recover. A partial extract can leave entries the
# original tree never had, so drop those first, otherwise the
# "restored" tree is the user's skills plus a slice of the snapshot.
staged_names = {orig.name for orig, _ in moved}
for entry in list(skills.iterdir()):
if entry.name in _EXCLUDE_TOP_LEVEL or entry.name in staged_names:
continue
try:
shutil.move(str(dest), str(orig))
if entry.is_dir() and not entry.is_symlink():
shutil.rmtree(entry)
else:
entry.unlink()
except OSError:
pass
unrestored = _unstage(moved)
if unrestored:
# Do not claim a clean restore we did not achieve, and keep the
# staging dir so the entries can be recovered by hand.
return (
False,
f"snapshot extract failed: {e} - could not restore "
f"{', '.join(sorted(unrestored))}; staged copies kept at {staged}",
None,
)
try:
shutil.rmtree(staged, ignore_errors=True)
except OSError:
@@ -692,13 +741,6 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]
# Human-readable summary for CLI
# ---------------------------------------------------------------------------
def format_size(n: int) -> str:
for unit in ("B", "KB", "MB", "GB"):
if n < 1024 or unit == "GB":
return f"{n:.1f} {unit}" if unit != "B" else f"{n} B"
n /= 1024
return f"{n:.1f} GB"
def summarize_backups() -> str:
rows = list_backups()
@@ -711,6 +753,6 @@ def summarize_backups() -> str:
f"{r.get('id','?'):<24} "
f"{(r.get('reason','?') or '?')[:40]:<40} "
f"{r.get('skill_files', 0):>6} "
f"{format_size(int(r.get('archive_bytes', 0))):>8}"
f"{format_bytes(int(r.get('archive_bytes', 0))):>8}"
)
return "\n".join(lines)

Some files were not shown because too many files have changed in this diff Show More