Merge origin/main into feat/plugin-catalog — reconcile with landed index/manifest-v2 tracks
@@ -0,0 +1,18 @@
|
||||
# CodeRabbit configuration.
|
||||
#
|
||||
# Auto-review is DISABLED repo-wide: the app was enabled at the org level on
|
||||
# 2026-08-14 and immediately began reviewing every opened PR. This repo's
|
||||
# merge gate is CI ("All required checks pass") plus maintainer review —
|
||||
# CodeRabbit reviews carry no merge weight (protect-main ruleset requires
|
||||
# 0 approving reviews), so auto-firing on the full PR firehose adds comment
|
||||
# noise without gating value.
|
||||
#
|
||||
# The bot stays installed and summonable on demand: comment
|
||||
# `@coderabbitai review` on any PR to request a one-off review, or
|
||||
# `@coderabbitai ignore` to mute it on a PR it has already joined.
|
||||
#
|
||||
# To re-enable auto-review, flip `enabled: true` below (or delete this file —
|
||||
# the app default is on).
|
||||
reviews:
|
||||
auto_review:
|
||||
enabled: false
|
||||
@@ -40,6 +40,10 @@ ui-tui/packages/hermes-ink/dist/
|
||||
# Environment files
|
||||
.env
|
||||
.env.*
|
||||
# ...but keep the template: docker/stage2-hook.sh seeds $HERMES_HOME/.env from
|
||||
# /opt/hermes/.env.example on first boot (OOF-285 — excluding it silently broke
|
||||
# first-boot .env seeding and the API_SERVER_KEY generation that depends on it).
|
||||
!.env.example
|
||||
|
||||
# IDE
|
||||
.vscode/
|
||||
@@ -97,12 +101,15 @@ packaging/
|
||||
plans/
|
||||
.plans/
|
||||
|
||||
# ACP registry manifest (icon + agent.json) — not consumed at runtime
|
||||
acp_registry/
|
||||
|
||||
# Repo-level dotfiles that are git-only or dev-tooling config
|
||||
.env.example
|
||||
.envrc
|
||||
.gitattributes
|
||||
.hadolint.yaml
|
||||
.mailmap
|
||||
|
||||
# Repo-root debug/export artifacts — must never reach image layers (COPY . .)
|
||||
/log.txt
|
||||
/sqlite_leak_fix.png
|
||||
/*.png.bak
|
||||
/default.tar.gz
|
||||
/*.tar.gz
|
||||
|
||||
@@ -104,6 +104,14 @@
|
||||
# $10/month subscription. Get your key at: https://opencode.ai/auth
|
||||
# OPENCODE_GO_API_KEY=
|
||||
|
||||
# =============================================================================
|
||||
# LLM PROVIDER (OpenCode Free)
|
||||
# =============================================================================
|
||||
# OpenCode Free provides keyless free models (Ox Alpha / x-preview-f-free,
|
||||
# big-pickle, etc.). NO env var and NO account needed — requests are sent
|
||||
# anonymously (the free tier rejects any unrecognized Authorization header).
|
||||
# Select it with `hermes model` or `/model free`.
|
||||
|
||||
# =============================================================================
|
||||
# LLM PROVIDER (Hugging Face Inference Providers)
|
||||
# =============================================================================
|
||||
|
||||
@@ -8,3 +8,27 @@ web/package-lock.json linguist-generated=true
|
||||
Dockerfile text eol=lf
|
||||
*.dockerfile text eol=lf
|
||||
docker/entrypoint.sh text eol=lf
|
||||
|
||||
# Enforce LF for all source/text files. Windows editors and tools default to
|
||||
# CRLF; without normalization a Windows contributor's edit turns into a
|
||||
# whole-file phantom diff (every line "changed" by its ending), breaks
|
||||
# string-match patch tooling, and pollutes review. `text` normalizes to LF
|
||||
# at check-in; `eol=lf` also checks out as LF so the working tree matches
|
||||
# the index on every platform. PowerShell files are the deliberate
|
||||
# exception (PS 5.1 tooling expects CRLF).
|
||||
*.py text eol=lf
|
||||
*.ts text eol=lf
|
||||
*.tsx text eol=lf
|
||||
*.js text eol=lf
|
||||
*.mjs text eol=lf
|
||||
*.cjs text eol=lf
|
||||
*.jsx text eol=lf
|
||||
*.json text eol=lf
|
||||
*.yaml text eol=lf
|
||||
*.yml text eol=lf
|
||||
*.toml text eol=lf
|
||||
*.md text eol=lf
|
||||
*.css text eol=lf
|
||||
*.html text eol=lf
|
||||
*.svg text eol=lf
|
||||
*.ps1 text eol=crlf
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
# actionlint knows only GitHub-hosted runner labels. An org admin names the
|
||||
# larger runners. Each one therefore reads as "unknown runner label" and hides
|
||||
# the real findings, unless this file declares it.
|
||||
self-hosted-runner:
|
||||
labels:
|
||||
- ubuntu-latest-96-core
|
||||
- ubuntu-latest-32-core
|
||||
- ubuntu-latest-32-arm-core
|
||||
- windows-latest-32-core
|
||||
@@ -15,12 +15,21 @@ outputs:
|
||||
python:
|
||||
description: Run Python tests / ruff / ty / windows-footguns.
|
||||
value: ${{ steps.classify.outputs.python }}
|
||||
python_prod:
|
||||
description: Python changes outside tests/ — gates product jobs (Desktop E2E, Docker).
|
||||
value: ${{ steps.classify.outputs.python_prod }}
|
||||
frontend:
|
||||
description: Run the TypeScript testing matrix + desktop build.
|
||||
value: ${{ steps.classify.outputs.frontend }}
|
||||
docker_meta:
|
||||
description: Docker setup and meta files have changed.
|
||||
value: ${{ steps.classify.outputs.docker_meta }}
|
||||
docker:
|
||||
description: Files included in the docker image have changed.
|
||||
value: ${{ steps.classify.outputs.docker }}
|
||||
nix:
|
||||
description: Run `nix flake check` (flake inputs, or any product Python change).
|
||||
value: ${{ steps.classify.outputs.nix }}
|
||||
site:
|
||||
description: Build the Docusaurus docs site.
|
||||
value: ${{ steps.classify.outputs.site }}
|
||||
@@ -30,15 +39,27 @@ outputs:
|
||||
deps:
|
||||
description: Check pyproject.toml dependency upper bounds.
|
||||
value: ${{ steps.classify.outputs.deps }}
|
||||
uv_lock:
|
||||
description: Run `uv lock --check` (pyproject.toml / uv.lock changes only).
|
||||
value: ${{ steps.classify.outputs.uv_lock }}
|
||||
npm_lock:
|
||||
description: Post/update the semantic package-lock.json diff PR comment.
|
||||
value: ${{ steps.classify.outputs.npm_lock }}
|
||||
installer:
|
||||
description: Run the PowerShell installer tests on a Windows runner.
|
||||
value: ${{ steps.classify.outputs.installer }}
|
||||
rust:
|
||||
description: Run `cargo test` for the Tauri bootstrap installer.
|
||||
value: ${{ steps.classify.outputs.rust }}
|
||||
mcp_catalog:
|
||||
description: Require MCP catalog security review label.
|
||||
value: ${{ steps.classify.outputs.mcp_catalog }}
|
||||
ci_review:
|
||||
description: Require CI-sensitive file review label.
|
||||
value: ${{ steps.classify.outputs.ci_review }}
|
||||
ci_review_files:
|
||||
description: JSON list of CI-sensitive files changed by the pull request.
|
||||
value: ${{ steps.classify.outputs.ci_review_files }}
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
|
||||
@@ -5,24 +5,32 @@ description: >-
|
||||
5,000 req/hr per installation (vs 1,000 for the default GITHUB_TOKEN)
|
||||
and are scoped to the App's installation permissions, not a user account.
|
||||
|
||||
Falls back to the built-in GITHUB_TOKEN when APP_CLIENT_ID is not set —
|
||||
this happens on fork PRs where repo secrets are unavailable. The fallback
|
||||
ensures classification, timings, and review comments still work on
|
||||
forks (with the lower GITHUB_TOKEN rate limit).
|
||||
Callers must source App credentials from a protected, main-only environment.
|
||||
Never pass an App private key to a pull_request job, a local action, or a
|
||||
reusable workflow resolved from an untrusted PR ref. The fallback keeps a
|
||||
trusted caller functional when its protected environment is misconfigured.
|
||||
|
||||
Composite actions cannot access the secrets context directly, so the
|
||||
calling workflow must pass secrets.APP_CLIENT_ID and secrets.APP_PRIVATE_KEY
|
||||
as inputs. When both are empty (fork PRs), the fallback fires.
|
||||
Composite actions cannot access contexts directly, so callers pass the
|
||||
public vars.APP_CLIENT_ID and protected secrets.APP_PRIVATE_KEY as inputs.
|
||||
When the private key is empty, the fallback fires.
|
||||
|
||||
inputs:
|
||||
client-id:
|
||||
description: GitHub App Client ID. Pass secrets.APP_CLIENT_ID from the calling workflow.
|
||||
description: GitHub App Client ID. Pass vars.APP_CLIENT_ID from the calling workflow.
|
||||
required: false
|
||||
default: ''
|
||||
private-key:
|
||||
description: GitHub App private key PEM. Pass secrets.APP_PRIVATE_KEY from the calling workflow.
|
||||
required: false
|
||||
default: ''
|
||||
owner:
|
||||
description: GitHub App installation owner. Empty scopes the token to the current repository.
|
||||
required: false
|
||||
default: ''
|
||||
repositories:
|
||||
description: Comma- or newline-separated repositories to scope within the installation owner.
|
||||
required: false
|
||||
default: ''
|
||||
|
||||
outputs:
|
||||
token:
|
||||
@@ -51,6 +59,8 @@ runs:
|
||||
with:
|
||||
client-id: ${{ inputs.client-id }}
|
||||
private-key: ${{ inputs.private-key }}
|
||||
owner: ${{ inputs.owner }}
|
||||
repositories: ${{ inputs.repositories }}
|
||||
|
||||
- name: Fall back to GITHUB_TOKEN
|
||||
id: fallback
|
||||
|
||||
|
Before Width: | Height: | Size: 36 KiB |
|
Before Width: | Height: | Size: 40 KiB |
|
Before Width: | Height: | Size: 33 KiB |
|
Before Width: | Height: | Size: 36 KiB |
|
Before Width: | Height: | Size: 138 KiB |
|
Before Width: | Height: | Size: 148 KiB |
|
Before Width: | Height: | Size: 428 KiB |
@@ -0,0 +1,141 @@
|
||||
// Run every workspace check at the same time and report all failures.
|
||||
//
|
||||
// The unit of work is a CHECK, and not a workspace. A package that declares
|
||||
// `check:*` sub-scripts gives one unit for each sub-script. A package with a
|
||||
// plain `check` gives that. This is the same selection rule the old CI matrix
|
||||
// used, so the set of commands is unchanged. Only the schedule is different.
|
||||
//
|
||||
// This is not `npm run --ws check`, because that command is serial and stops
|
||||
// at the first workspace that fails. This runs every unit and fails at the
|
||||
// end with the full list.
|
||||
//
|
||||
// The output of each unit goes to a buffer and prints on completion inside a
|
||||
// group that collapses. Children that write to one stdout together interleave
|
||||
// their lines, and a failure is then hard to read.
|
||||
//
|
||||
// This also runs on a laptop: `node .github/scripts/run-workspace-checks.mjs`.
|
||||
// `--concurrency N` sets the limit. `--list` prints the units and exits.
|
||||
|
||||
import { execFileSync, spawn } from 'node:child_process'
|
||||
import { availableParallelism } from 'node:os'
|
||||
|
||||
const IS_CI = Boolean(process.env.GITHUB_ACTIONS)
|
||||
const NPM = process.platform === 'win32' ? 'npm.cmd' : 'npm'
|
||||
|
||||
/** @returns {{pkg: string, script: string}[]} */
|
||||
function discoverUnits() {
|
||||
const raw = execFileSync(NPM, ['query', '.workspace'], {
|
||||
encoding: 'utf-8',
|
||||
shell: process.platform === 'win32',
|
||||
})
|
||||
/** @type {{location: string, scripts?: Record<string,string>}[]} */
|
||||
const pkgs = JSON.parse(raw)
|
||||
|
||||
/** @type {{pkg: string, script: string}[]} */
|
||||
const units = []
|
||||
for (const pkg of pkgs) {
|
||||
const scripts = pkg.scripts || {}
|
||||
const subs = Object.keys(scripts).filter((s) => /^check:.+$/.test(s))
|
||||
if (subs.length > 0) {
|
||||
for (const script of subs) units.push({ pkg: pkg.location, script })
|
||||
} else if (scripts.check) {
|
||||
units.push({ pkg: pkg.location, script: 'check' })
|
||||
}
|
||||
}
|
||||
return units
|
||||
}
|
||||
|
||||
/** @param {{pkg: string, script: string}} unit */
|
||||
function runUnit(unit) {
|
||||
return new Promise((resolve) => {
|
||||
const started = Date.now()
|
||||
const child = spawn(NPM, ['run', '--prefix', unit.pkg, unit.script], {
|
||||
// Buffer, and do not inherit. Children that share one stdout
|
||||
// interleave their lines, and a failure is then hard to read.
|
||||
stdio: ['ignore', 'pipe', 'pipe'],
|
||||
shell: process.platform === 'win32',
|
||||
})
|
||||
/** @type {Buffer[]} */
|
||||
const chunks = []
|
||||
child.stdout.on('data', (c) => chunks.push(c))
|
||||
child.stderr.on('data', (c) => chunks.push(c))
|
||||
child.on('error', (err) => {
|
||||
chunks.push(Buffer.from(`failed to spawn: ${err.message}\n`))
|
||||
resolve({ unit, code: 1, output: Buffer.concat(chunks).toString('utf-8'), ms: Date.now() - started })
|
||||
})
|
||||
child.on('close', (code) => {
|
||||
resolve({
|
||||
unit,
|
||||
code: code ?? 1,
|
||||
output: Buffer.concat(chunks).toString('utf-8'),
|
||||
ms: Date.now() - started,
|
||||
})
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const argv = process.argv.slice(2)
|
||||
const units = discoverUnits()
|
||||
|
||||
if (units.length === 0) {
|
||||
console.error(
|
||||
'::error::No workspace package declares a check script — refusing to report green having run nothing.',
|
||||
)
|
||||
process.exit(1)
|
||||
}
|
||||
|
||||
if (argv.includes('--list')) {
|
||||
for (const u of units) console.log(`${u.pkg} :: ${u.script}`)
|
||||
return
|
||||
}
|
||||
|
||||
const flagIdx = argv.indexOf('--concurrency')
|
||||
const concurrency = Math.max(
|
||||
1,
|
||||
flagIdx !== -1 ? Number(argv[flagIdx + 1]) : Math.min(units.length, availableParallelism()),
|
||||
)
|
||||
|
||||
console.log(`running ${units.length} checks, up to ${concurrency} at a time:`)
|
||||
for (const u of units) console.log(` ${u.pkg} :: ${u.script}`)
|
||||
console.log('')
|
||||
|
||||
const queue = [...units]
|
||||
/** @type {{unit: {pkg: string, script: string}, code: number, output: string, ms: number}[]} */
|
||||
const results = []
|
||||
|
||||
async function worker() {
|
||||
for (;;) {
|
||||
const unit = queue.shift()
|
||||
if (!unit) return
|
||||
const res = await runUnit(unit)
|
||||
results.push(res)
|
||||
const label = `${res.unit.pkg} :: ${res.unit.script}`
|
||||
const secs = (res.ms / 1000).toFixed(1)
|
||||
const status = res.code === 0 ? 'PASS' : 'FAIL'
|
||||
if (IS_CI) console.log(`::group::${status} ${label} (${secs}s)`)
|
||||
else console.log(`----- ${status} ${label} (${secs}s) -----`)
|
||||
process.stdout.write(res.output.endsWith('\n') ? res.output : res.output + '\n')
|
||||
if (IS_CI) console.log('::endgroup::')
|
||||
}
|
||||
}
|
||||
|
||||
await Promise.all(Array.from({ length: Math.min(concurrency, units.length) }, worker))
|
||||
|
||||
const failed = results.filter((r) => r.code !== 0)
|
||||
console.log('\n=== summary ===')
|
||||
for (const r of [...results].sort((a, b) => b.ms - a.ms)) {
|
||||
console.log(
|
||||
` ${r.code === 0 ? 'pass' : 'FAIL'} ${(r.ms / 1000).toFixed(1).padStart(6)}s ${r.unit.pkg} :: ${r.unit.script}`,
|
||||
)
|
||||
}
|
||||
|
||||
if (failed.length > 0) {
|
||||
for (const r of failed) console.error(`::error::${r.unit.pkg} :: ${r.unit.script} failed`)
|
||||
console.error(`::error::${failed.length} of ${results.length} checks failed`)
|
||||
process.exit(1)
|
||||
}
|
||||
console.log(`\nall ${results.length} checks passed`)
|
||||
}
|
||||
|
||||
await main()
|
||||
@@ -0,0 +1,80 @@
|
||||
name: CI review comment
|
||||
|
||||
# Live-updating PR review comment.
|
||||
#
|
||||
# The poller runs for up to 40 minutes.
|
||||
# This run lives in its own workflow.
|
||||
# A run stays in progress until its last job ends, and GitHub refuses
|
||||
# ``gh run rerun`` on a run that is in progress.
|
||||
#
|
||||
# ``workflow_run`` starts this when CI starts. It always reads the workflow
|
||||
# and the scripts from the default branch, never from the PR head.
|
||||
# It makes a write token safe here.
|
||||
#
|
||||
# The poller reads job results through the API. Thus it watches the CI run
|
||||
# and the separate docker run, and it depends on neither.
|
||||
|
||||
on:
|
||||
workflow_run:
|
||||
workflows: [CI]
|
||||
# ``in_progress``, not ``requested``: a first-time contributor's run
|
||||
# sits in ``action_required`` until a maintainer approves it, and
|
||||
# ``requested`` fires at creation — the poller would wait out its
|
||||
# whole timeout on a run that never starts. ``in_progress`` fires
|
||||
# when the run actually starts, and it also fires on re-runs, which
|
||||
# ``requested`` does not.
|
||||
types: [in_progress]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
actions: read
|
||||
pull-requests: write
|
||||
|
||||
# One poller per CI run. A new push starts a new CI run, and its poller
|
||||
# cancels the poller of the run that GitHub superseded. The group keys
|
||||
# on the head repository too: fork PRs often share a branch name
|
||||
# (``main``, ``patch-1``), and two PRs must not cancel each other.
|
||||
concurrency:
|
||||
group: ci-review-comment-${{ github.event.workflow_run.head_repository.full_name }}-${{ github.event.workflow_run.head_branch }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
comment:
|
||||
name: CI review comment (live)
|
||||
# Fork PRs get no comment: the poller needs a write token, and the
|
||||
# ``pull_requests`` payload is empty for a fork run.
|
||||
if: >-
|
||||
github.event.workflow_run.event == 'pull_request' &&
|
||||
github.event.workflow_run.head_repository.full_name == github.repository
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: Checkout trusted default branch
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
ref: ${{ github.event.repository.default_branch }}
|
||||
persist-credentials: false
|
||||
|
||||
- name: Run live comment poller
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
# The CI run to report on — not this run. The name must not be
|
||||
# GITHUB_RUN_ID: the runner sets the GITHUB_* defaults itself and
|
||||
# ignores an env: override, so that name silently resolves to THIS
|
||||
# run. The poller then watches itself, which stays in_progress for
|
||||
# as long as the poller runs, and it waits out its whole timeout.
|
||||
CI_RUN_ID: ${{ github.event.workflow_run.id }}
|
||||
# Sibling runs for the same commit that the comment also covers,
|
||||
# one workflow name per line (a name can contain a comma).
|
||||
# The poller resolves each name to its runs through the API.
|
||||
WATCH_WORKFLOWS: |
|
||||
Docker Build, Test, and Publish
|
||||
PR_NUMBER: ${{ github.event.workflow_run.pull_requests[0].number }}
|
||||
RUN_URL: ${{ github.event.workflow_run.html_url }}
|
||||
# Commit info for the review comment header.
|
||||
COMMIT_SHA: ${{ github.event.workflow_run.head_sha }}
|
||||
COMMIT_MESSAGE: ${{ github.event.workflow_run.head_commit.message }}
|
||||
run: |
|
||||
python3 -u scripts/ci/live_comment.py \
|
||||
--interval 15 \
|
||||
--timeout 3000
|
||||
@@ -9,6 +9,10 @@ name: CI
|
||||
# definitions, matrices, and concurrency settings. They no longer have
|
||||
# ``push:`` / ``pull_request:`` triggers of their own — everything flows
|
||||
# through this file.
|
||||
#
|
||||
# SECURITY: this workflow runs PR-controlled actions, workflows, and code.
|
||||
# Do not add ``secrets: inherit`` or GitHub App credentials here. Trusted
|
||||
# main-only automation uses protected environments in its own workflows.
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
@@ -20,7 +24,6 @@ permissions:
|
||||
pull-requests: write # needed by lint (PR comment) + supply-chain review_status
|
||||
actions: read # needed by osv-scanner (SARIF upload)
|
||||
security-events: write # needed by osv-scanner (SARIF upload)
|
||||
packages: write # needed by docker build
|
||||
|
||||
concurrency:
|
||||
group: ci-${{ github.ref }}
|
||||
@@ -35,33 +38,32 @@ jobs:
|
||||
detect:
|
||||
name: Detect affected areas
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
timeout-minutes: 1
|
||||
outputs:
|
||||
python: ${{ steps.classify.outputs.python }}
|
||||
python_prod: ${{ steps.classify.outputs.python_prod }}
|
||||
frontend: ${{ steps.classify.outputs.frontend }}
|
||||
site: ${{ steps.classify.outputs.site }}
|
||||
scan: ${{ steps.classify.outputs.scan }}
|
||||
deps: ${{ steps.classify.outputs.deps }}
|
||||
uv_lock: ${{ steps.classify.outputs.uv_lock }}
|
||||
npm_lock: ${{ steps.classify.outputs.npm_lock }}
|
||||
installer: ${{ steps.classify.outputs.installer }}
|
||||
rust: ${{ steps.classify.outputs.rust }}
|
||||
docker_meta: ${{ steps.classify.outputs.docker_meta }}
|
||||
mcp_catalog: ${{ steps.classify.outputs.mcp_catalog }}
|
||||
ci_review: ${{ steps.classify.outputs.ci_review }}
|
||||
ci_review_files: ${{ steps.classify.outputs.ci_review_files }}
|
||||
event_name: ${{ github.event_name }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- name: Get GitHub App token
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ secrets.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
- name: Detect affected areas
|
||||
id: classify
|
||||
uses: ./.github/actions/detect-changes
|
||||
with:
|
||||
# The get-app-token composite action falls back to GITHUB_TOKEN
|
||||
# on fork PRs where APP_ID is unavailable.
|
||||
github-token: ${{ steps.app-token.outputs.token }}
|
||||
sparse-checkout: scripts/ci/classify_changes.py
|
||||
sparse-checkout-cone-mode: false
|
||||
github-token: ${{ github.token }}
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
# Lane-gated sub-workflows. Each runs in parallel after detect finishes.
|
||||
@@ -72,9 +74,16 @@ jobs:
|
||||
needs: detect
|
||||
if: needs.detect.outputs.python == 'true'
|
||||
uses: ./.github/workflows/tests.yml
|
||||
with:
|
||||
slice_count: 8
|
||||
secrets: inherit
|
||||
|
||||
# macOS + Windows lanes. The main `tests` lane above is Linux-only, and
|
||||
# the OS-marked tests it collects are skipped there by design (see the
|
||||
# `_OS_MARKS` comment in tests/conftest.py) — this is where they run.
|
||||
# Same `python` lane gate: if no Python changed, neither runs.
|
||||
tests-os:
|
||||
name: OS-specific tests
|
||||
needs: detect
|
||||
if: needs.detect.outputs.python == 'true'
|
||||
uses: ./.github/workflows/tests-os.yml
|
||||
|
||||
lint:
|
||||
name: Python lints
|
||||
@@ -83,19 +92,44 @@ jobs:
|
||||
uses: ./.github/workflows/lint.yml
|
||||
with:
|
||||
event_name: ${{ needs.detect.outputs.event_name }}
|
||||
secrets: inherit
|
||||
|
||||
js-tests:
|
||||
name: JS & TS checks
|
||||
needs: detect
|
||||
if: needs.detect.outputs.frontend == 'true'
|
||||
uses: ./.github/workflows/js-tests.yml
|
||||
secrets: inherit
|
||||
|
||||
installer-tests:
|
||||
name: Installer tests
|
||||
needs: detect
|
||||
# Windows-only, and only for PRs that touch install.ps1 or its tests.
|
||||
if: needs.detect.outputs.installer == 'true'
|
||||
uses: ./.github/workflows/installer-tests.yml
|
||||
|
||||
rust-tests:
|
||||
name: Rust tests
|
||||
needs: detect
|
||||
# Only for PRs that touch a Rust crate. `.rs` is under apps/, so these
|
||||
# changes used to run the TypeScript matrix and nothing that compiles them.
|
||||
if: needs.detect.outputs.rust == 'true'
|
||||
uses: ./.github/workflows/rust-tests.yml
|
||||
|
||||
e2e-desktop:
|
||||
name: Desktop E2E
|
||||
needs: detect
|
||||
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true'
|
||||
# python_prod (not python): the Playwright suite exercises the built app
|
||||
# + `hermes serve` backend, which never import anything under tests/.
|
||||
# Tests-only PRs (~17% of commits) skip this 5-minute job — the longest
|
||||
# single job in the workflow — while still running the full pytest lanes.
|
||||
#
|
||||
# ⛔ TEMPORARILY DISABLED (Aug 2, 2026, Teknium) — the suite is red on
|
||||
# every PR and on main itself since the Aug 1 night engines/npm churn
|
||||
# (#76499 → #76562 → #76575): the mock-backend Electron window never
|
||||
# gets a title, so boot/chat/setup/interim specs all fail identically
|
||||
# regardless of the PR's diff (verified on #76573 and the docs-only
|
||||
# #76582). Tracking issue: #76627 (assigned: Ari). To re-enable,
|
||||
# delete the `false &&` below — nothing else changed.
|
||||
if: ${{ false && (needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true') }}
|
||||
uses: ./.github/workflows/e2e-desktop.yml
|
||||
|
||||
docs-site:
|
||||
@@ -103,48 +137,46 @@ jobs:
|
||||
needs: detect
|
||||
if: needs.detect.outputs.site == 'true'
|
||||
uses: ./.github/workflows/docs-site-checks.yml
|
||||
secrets: inherit
|
||||
|
||||
history-check:
|
||||
name: Deny unrelated histories
|
||||
needs: detect
|
||||
if: needs.detect.outputs.event_name == 'pull_request'
|
||||
uses: ./.github/workflows/history-check.yml
|
||||
secrets: inherit
|
||||
|
||||
contributor-check:
|
||||
name: Check contributors
|
||||
needs: detect
|
||||
if: needs.detect.outputs.python == 'true'
|
||||
uses: ./.github/workflows/contributor-check.yml
|
||||
secrets: inherit
|
||||
|
||||
uv-lockfile:
|
||||
name: Check uv.lock
|
||||
needs: detect
|
||||
# Gated: `uv lock --check` re-resolves the whole dependency graph against
|
||||
# PyPI, so on every PR it spent a network round-trip — and, on a registry
|
||||
# blip, a blocking red X — for diffs that cannot desync the lockfile
|
||||
# (docs, frontend, prose). Only pyproject.toml / uv.lock can. A
|
||||
# `.github/` change still forces it on via the classifier's fail-open.
|
||||
if: needs.detect.outputs.uv_lock == 'true'
|
||||
uses: ./.github/workflows/uv-lockfile-check.yml
|
||||
secrets: inherit
|
||||
|
||||
infographic-check:
|
||||
name: Check no committed infographics
|
||||
needs: detect
|
||||
uses: ./.github/workflows/infographic-check.yml
|
||||
|
||||
lockfile-diff:
|
||||
name: package-lock.json diff
|
||||
needs: detect
|
||||
if: needs.detect.outputs.event_name == 'pull_request' && needs.detect.outputs.npm_lock == 'true'
|
||||
uses: ./.github/workflows/lockfile-diff.yml
|
||||
secrets: inherit
|
||||
|
||||
docker-lint:
|
||||
name: Lint Docker scripts
|
||||
needs: detect
|
||||
if: needs.detect.outputs.docker_meta == 'true'
|
||||
uses: ./.github/workflows/docker-lint.yml
|
||||
secrets: inherit
|
||||
|
||||
docker:
|
||||
name: Build&Test Docker image
|
||||
needs: detect
|
||||
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true' || needs.detect.outputs.docker_meta == 'true'
|
||||
uses: ./.github/workflows/docker.yml
|
||||
secrets: inherit
|
||||
|
||||
supply-chain:
|
||||
name: Supply-chain scan
|
||||
@@ -163,89 +195,13 @@ jobs:
|
||||
uses: ./.github/workflows/review-labels.yml
|
||||
with:
|
||||
ci_review: ${{ needs.detect.outputs.ci_review == 'true' }}
|
||||
ci_review_files: ${{ needs.detect.outputs.ci_review_files }}
|
||||
mcp_catalog: ${{ needs.detect.outputs.mcp_catalog == 'true' }}
|
||||
supply_chain: ${{ needs.supply-chain.outputs.critical_findings == 'true' }}
|
||||
secrets: inherit
|
||||
|
||||
osv-scanner:
|
||||
name: OSV scan
|
||||
uses: ./.github/workflows/osv-scanner.yml
|
||||
secrets: inherit
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
# Live-updating PR review comment.
|
||||
#
|
||||
# A single ``comment-live`` job polls the GitHub Actions API every 15s
|
||||
# for job statuses in this run, re-assembles the review comment from
|
||||
# whatever results are available, and upserts it via the
|
||||
# ``<!-- hermes-ci-review-bot -->`` marker.
|
||||
#
|
||||
# The poller exits when all non-infra jobs are completed (or on
|
||||
# timeout). ci-timings' review_status is picked up automatically when
|
||||
# its artifact becomes available — the poller downloads and merges it.
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
comment-live:
|
||||
name: CI review comment (live)
|
||||
needs: [detect, review-labels, lockfile-diff, supply-chain, osv-scanner, uv-lockfile, history-check, contributor-check]
|
||||
if: always() && github.event_name == 'pull_request' && github.event.pull_request.head.repo.fork != true
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 40
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Run live comment poller
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GITHUB_REPOSITORY: ${{ github.repository }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
PR_NUMBER: ${{ github.event.pull_request.number }}
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
# Commit info for the review comment header.
|
||||
COMMIT_SHA: ${{ github.event.pull_request.head.sha }}
|
||||
COMMIT_MESSAGE: ${{ github.event.pull_request.head.commit.message }}
|
||||
COMMIT_URL: ${{ github.server_url }}/${{ github.repository }}/pull/${{ github.event.pull_request.number }}/commits/${{ github.event.pull_request.head.sha }}
|
||||
# Structured review statuses from workflow_call jobs.
|
||||
# Each job outputs a JSON array of {source, results: [...]} objects
|
||||
# that the assembler renders directly — no hardcoded job-name
|
||||
# matching. We merge all available outputs into one array.
|
||||
REVIEW_STATUSES: ${{ toJSON(needs.*.outputs.review_status) }}
|
||||
run: |
|
||||
set -uo pipefail
|
||||
|
||||
# REVIEW_STATUSES is a JSON array of strings (some may be empty
|
||||
# when a job was skipped). Parse each string and merge into one
|
||||
# flat array for the assembler.
|
||||
python3 - <<'PYEOF'
|
||||
import json, os, sys
|
||||
|
||||
raw = os.environ.get("REVIEW_STATUSES", "")
|
||||
merged = []
|
||||
if raw:
|
||||
try:
|
||||
arr = json.loads(raw)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
arr = []
|
||||
for item in arr:
|
||||
if not item:
|
||||
continue
|
||||
try:
|
||||
statuses = json.loads(item)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
continue
|
||||
if isinstance(statuses, list):
|
||||
merged.extend(statuses)
|
||||
|
||||
# Write merged array to a temp file the poller reads.
|
||||
with open("/tmp/review_statuses.json", "w") as f:
|
||||
json.dump(merged, f)
|
||||
print(f"Merged {len(merged)} review status entries")
|
||||
PYEOF
|
||||
|
||||
python3 scripts/ci/live_comment.py \
|
||||
--interval 15 \
|
||||
--timeout 2100 \
|
||||
--review-statuses-file /tmp/review_statuses.json
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
# Gate: runs after everything. ``if: always()`` ensures it reports a
|
||||
@@ -262,8 +218,11 @@ jobs:
|
||||
needs:
|
||||
- detect
|
||||
- tests
|
||||
- tests-os
|
||||
- lint
|
||||
- js-tests
|
||||
- installer-tests
|
||||
- rust-tests
|
||||
- e2e-desktop
|
||||
- docs-site
|
||||
- history-check
|
||||
@@ -274,9 +233,10 @@ jobs:
|
||||
- supply-chain
|
||||
- review-labels
|
||||
- osv-scanner
|
||||
# comment-live is a polling job — it doesn't block the gate.
|
||||
# we don't require docker to pass rn because it's so slow lol
|
||||
# - docker
|
||||
# The image build runs in its own workflow (docker.yml) and reports
|
||||
# its own check. It was never required here, because it is too slow
|
||||
# to block a merge. A separate run also stops it from holding this
|
||||
# run open. That is what blocked ``gh run rerun``.
|
||||
if: always()
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
@@ -313,12 +273,13 @@ jobs:
|
||||
# report with a gantt chart + per-step breakdown. The report is uploaded
|
||||
# as an artifact and a markdown summary is written to $GITHUB_STEP_SUMMARY.
|
||||
#
|
||||
# The live comment poller picks up ci-timings' completion automatically —
|
||||
# it reads review-status.json from the artifact when the job finishes.
|
||||
# The live comment poller dynamically fetches all review-status-* artifacts
|
||||
# across the orchestrator and sub-workflow runs every cycle, so its link
|
||||
# points straight at that report.
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
ci-timings:
|
||||
name: CI timing report
|
||||
needs: [all-checks-pass, docker]
|
||||
needs: [all-checks-pass]
|
||||
if: always()
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
@@ -326,13 +287,6 @@ jobs:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Get GitHub App token
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ secrets.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- name: Restore baseline cache (PR only)
|
||||
if: github.event_name == 'pull_request'
|
||||
uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
@@ -346,38 +300,52 @@ jobs:
|
||||
|
||||
- name: Collect timings and generate report
|
||||
env:
|
||||
# Forks get no repo secrets (AUTOFIX_BOT_PAT is empty); fall back to
|
||||
# the built-in read-only token so the timings API read still works
|
||||
# there instead of hard-failing this advisory job on every fork PR.
|
||||
# The get-app-token composite action falls back to GITHUB_TOKEN
|
||||
# on fork PRs where APP_ID is unavailable.
|
||||
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
python3 scripts/ci/timings_report.py \
|
||||
--baseline ci-timings-baseline.json \
|
||||
--output ci-timings-report.html \
|
||||
--json-out ci-timings.json \
|
||||
--summary-out ci-timings-summary.md \
|
||||
--review-status-out review-status.json
|
||||
--summary-out ci-timings-summary.md
|
||||
|
||||
- name: Upload HTML report + review status
|
||||
- name: Upload HTML report
|
||||
# Advisory report — artifact-service blips must not fail the job.
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
id: ci-timings-artifact
|
||||
id: ci-timings-html
|
||||
with:
|
||||
name: ci-timings-report
|
||||
path: |
|
||||
ci-timings-report.html
|
||||
review-status.json
|
||||
path: ci-timings-report.html
|
||||
retention-days: 14
|
||||
|
||||
- name: Build linked review status
|
||||
if: hashFiles('ci-timings.json') != ''
|
||||
env:
|
||||
CI_TIMINGS_REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url }}
|
||||
run: |
|
||||
python3 scripts/ci/timings_report.py \
|
||||
--from-json ci-timings.json \
|
||||
--baseline ci-timings-baseline.json \
|
||||
--review-status-out review-status.json \
|
||||
--review-status-only
|
||||
|
||||
- name: Upload review status
|
||||
if: hashFiles('review-status.json') != ''
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: review-status-ci-timings
|
||||
path: review-status.json
|
||||
retention-days: 14
|
||||
|
||||
- name: Output summary
|
||||
env:
|
||||
REPORT_URL: ${{ steps.ci-timings-artifact.outputs.artifact-url}}
|
||||
REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url}}
|
||||
run: |
|
||||
echo "# CI Timing report" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "[View the full interactive report]($REPORT_URL)" >> "$GITHUB_STEP_SUMMARY"
|
||||
{
|
||||
echo "# CI Timing report"
|
||||
echo "[View the full interactive report]($REPORT_URL)"
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
cat ci-timings-summary.md >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
- name: Save baseline cache (main only)
|
||||
@@ -33,6 +33,7 @@ jobs:
|
||||
if [ -z "$NEW_EMAILS" ]; then
|
||||
echo "No new commits to check."
|
||||
echo "review_status=[]" >> "$GITHUB_OUTPUT"
|
||||
echo "review_status=[]" > review-status.json
|
||||
exit 0
|
||||
fi
|
||||
|
||||
@@ -67,6 +68,8 @@ jobs:
|
||||
echo -e "$MISSING"
|
||||
echo ""
|
||||
echo "Add a mapping file (do NOT edit AUTHOR_MAP in release.py):"
|
||||
echo " python3 scripts/audit_pr_attribution.py --fix # auto-resolve + create files"
|
||||
echo "or manually:"
|
||||
echo -e "$MISSING" | while read -r line; do
|
||||
email=$(echo "$line" | sed 's/^ *//' | cut -d' ' -f1)
|
||||
[ -z "$email" ] && continue
|
||||
@@ -78,14 +81,28 @@ jobs:
|
||||
|
||||
# Emit review_status for unmapped emails
|
||||
DETAIL=$(echo -e "$MISSING" | sed '/^$/d; s/^ //')
|
||||
HOW_TO_FIX=$'Add mappings to scripts/release.py AUTHOR_MAP:\n```\n"<email>": "<github-username>",\n```\nTo find the GitHub username for an email:\n```\ngh api \'search/users?q=EMAIL+in:email\' --jq \'.items[0].login\'\n```\n'
|
||||
HOW_TO_FIX=$'Run from the PR branch:\n```\npython3 scripts/audit_pr_attribution.py --fix\ngit add contributors && git commit -m "chore: map contributor emails" && git push\n```\nOr map one email manually (do NOT edit AUTHOR_MAP in release.py):\n```\npython3 scripts/add_contributor.py <email> <github-username>\n```\nTo find the GitHub username for an email:\n```\ngh api \'search/users?q=EMAIL+in:email\' --jq \'.items[0].login\'\n```\n'
|
||||
REVIEW_STATUS=$(jq -nc \
|
||||
--arg detail "$DETAIL" \
|
||||
--arg how_to_fix "$HOW_TO_FIX" \
|
||||
'[{"source":"contributor attribution","results":[{"kind":"action_required","title":"Unmapped contributor email(s)","summary":"New contributor email(s) are not in AUTHOR_MAP.","detail":$detail,"how_to_fix":$how_to_fix}]}]')
|
||||
echo "review_status=$REVIEW_STATUS" >> "$GITHUB_OUTPUT"
|
||||
echo "review_status=$REVIEW_STATUS" > review-status.json
|
||||
|
||||
exit 1
|
||||
else
|
||||
echo "✅ All contributor emails are mapped."
|
||||
echo "review_status=[]" >> "$GITHUB_OUTPUT"
|
||||
echo "review_status=[]" > review-status.json
|
||||
fi
|
||||
|
||||
- name: Upload review status artifact
|
||||
if: always() && steps.check-emails.outcome != 'skipped'
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: review-status-contributor-check
|
||||
path: review-status.json
|
||||
retention-days: 1
|
||||
overwrite: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
@@ -60,15 +60,19 @@ jobs:
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ secrets.APP_CLIENT_ID }}
|
||||
client-id: ${{ vars.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 22
|
||||
node-version: 26
|
||||
cache: npm
|
||||
cache-dependency-path: website/package-lock.json
|
||||
|
||||
- name: grab npm 12
|
||||
run: |
|
||||
npm i -g npm@12
|
||||
|
||||
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
with:
|
||||
python-version: '3.11'
|
||||
@@ -188,6 +192,61 @@ jobs:
|
||||
cp website/build/llms-full.txt _site/llms-full.txt
|
||||
fi
|
||||
|
||||
# Pages serves exactly the newest artifact, so each deploy used to delete
|
||||
# the previous build's content-hashed JS/CSS while edge caches (Vercel →
|
||||
# Fastly, max-age=300 + stale-while-revalidate=3600) kept serving HTML
|
||||
# that referenced it — every asset request 404'd for up to ~65 minutes
|
||||
# after each deploy. With push-triggered deploys landing every ~15-30
|
||||
# minutes, the docs were in that broken window most of the day (search,
|
||||
# being pure client JS, died first). Fix: keep a rolling pool of prior
|
||||
# builds' hashed assets and union-merge it into every artifact so stale
|
||||
# HTML keeps resolving. Hashed filenames are content-addressed, so a
|
||||
# collision is by definition the identical file — the merge never
|
||||
# overwrites current-build output (cp --update=none).
|
||||
- name: Restore asset retention pool
|
||||
uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
with:
|
||||
path: _asset_retention
|
||||
key: docs-asset-retention-${{ github.run_id }}
|
||||
restore-keys: |
|
||||
docs-asset-retention-
|
||||
|
||||
- name: Merge retained assets from previous deploys
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ASSET_DIRS="assets zh-Hans/assets"
|
||||
mkdir -p _asset_retention
|
||||
# 1) Add this build's hashed assets to the pool (fresh mtimes, so
|
||||
# assets still shipped by current builds never age out).
|
||||
for d in $ASSET_DIRS; do
|
||||
if [ -d "_site/docs/$d" ]; then
|
||||
mkdir -p "_asset_retention/$d"
|
||||
cp -a "_site/docs/$d/." "_asset_retention/$d/"
|
||||
fi
|
||||
done
|
||||
# 2) Drop pool entries no build has produced for 14 days — far
|
||||
# beyond any edge-cache or open-tab horizon.
|
||||
find _asset_retention -type f -mtime +14 -delete
|
||||
find _asset_retention -type d -empty -delete
|
||||
# 3) Union-merge the pool into the artifact; --update=none keeps the
|
||||
# current build authoritative for any path it produced.
|
||||
for d in $ASSET_DIRS; do
|
||||
if [ -d "_asset_retention/$d" ]; then
|
||||
mkdir -p "_site/docs/$d"
|
||||
cp -R --update=none "_asset_retention/$d/." "_site/docs/$d/"
|
||||
fi
|
||||
done
|
||||
echo "retention pool:" && du -sh _asset_retention
|
||||
echo "deployed assets:" && du -sh _site/docs/assets
|
||||
|
||||
- name: Save asset retention pool
|
||||
# Always save under a fresh key (caches are immutable); restore-keys
|
||||
# prefix matching picks the newest on the next run.
|
||||
uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
with:
|
||||
path: _asset_retention
|
||||
key: docs-asset-retention-${{ github.run_id }}
|
||||
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-pages-artifact@56afc609e74202658d3ffba0e8f6dda462b719fa # v3
|
||||
with:
|
||||
|
||||
@@ -1,9 +1,19 @@
|
||||
name: Docker Build, Test, and Publish
|
||||
|
||||
on:
|
||||
# This workflow owns its own triggers. ci.yml does not call it.
|
||||
# A reusable-workflow call eeps the caller run in progress for that full time.
|
||||
# GitHub refuses ``gh run rerun`` on a run that is still in progress.
|
||||
# Thus one slow advisory job blocked every rerun of the fast required jobs. A separate
|
||||
# run reruns and cancels independently.
|
||||
#
|
||||
# Trusted main pushes resolve the environment-scoped Docker Hub secrets in
|
||||
# this same workflow, never across a workflow boundary.
|
||||
pull_request:
|
||||
push:
|
||||
branches: [main]
|
||||
release:
|
||||
types: [published]
|
||||
workflow_call:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -20,20 +30,60 @@ env:
|
||||
IMAGE_NAME: nousresearch/hermes-agent
|
||||
|
||||
jobs:
|
||||
# Build, test, and optionally push the image for each architecture.
|
||||
# Classify the PR's changed files. ci.yml used to gate the docker call on
|
||||
# its own ``detect`` outputs; now that this workflow triggers itself, it
|
||||
# runs the same composite action. On push and release the classifier fails
|
||||
# open (every lane true), so post-merge validation is never weakened.
|
||||
detect:
|
||||
name: Detect affected areas
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
outputs:
|
||||
build: ${{ steps.gate.outputs.build }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Detect affected areas
|
||||
id: classify
|
||||
uses: ./.github/actions/detect-changes
|
||||
with:
|
||||
github-token: ${{ github.token }}
|
||||
|
||||
- name: Decide whether to build
|
||||
id: gate
|
||||
env:
|
||||
# The docker lane derives from python_prod (not python: the image
|
||||
# copies installed code, never tests/, so tests-only PRs skip the
|
||||
# build), frontend and docker_meta. classify_changes.py owns the
|
||||
# formula so this gate and the nix lane cannot drift apart.
|
||||
DOCKER: ${{ steps.classify.outputs.docker }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ "$DOCKER" = "true" ]; then
|
||||
echo "build=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "build=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# Build and test the image for each architecture. This job runs PR code,
|
||||
# so it must remain secret-free. Publishing happens in the separate,
|
||||
# protected publish job after these tests pass.
|
||||
build:
|
||||
if: github.repository == 'NousResearch/hermes-agent'
|
||||
needs: [detect]
|
||||
if: github.repository == 'NousResearch/hermes-agent' && needs.detect.outputs.build == 'true'
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- arch: amd64
|
||||
runner: ubuntu-latest
|
||||
runner: ubuntu-latest-32-core
|
||||
platform: linux/amd64
|
||||
cache-from: type=gha,scope=docker-amd64
|
||||
cache-to: type=gha,mode=max,scope=docker-amd64
|
||||
# arm64 builds on the native arm64 larger runner. A build of
|
||||
# linux/arm64 on an x64 host uses emulation.
|
||||
- arch: arm64
|
||||
runner: ubuntu-24.04-arm
|
||||
runner: ubuntu-latest-32-arm-core
|
||||
platform: linux/arm64
|
||||
cache-from: type=gha,scope=docker-arm64
|
||||
cache-to: type=gha,mode=max,scope=docker-arm64
|
||||
@@ -44,7 +94,19 @@ jobs:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
# Retry once on transient Docker Hub / buildkit pull failures
|
||||
# (connection reset, auth token timeout, rate limiting). The action
|
||||
# generates a unique builder name per invocation so the retry doesn't
|
||||
# collide with the failed first attempt. A genuine persistent failure
|
||||
# still fails the job — only the first attempt has continue-on-error.
|
||||
# Refs: docker/setup-buildx-action#510
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
continue-on-error: true
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
- name: Set up Docker Buildx (retry)
|
||||
if: steps.buildx.outcome == 'failure'
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
# Build once, load into the local daemon for testing. Cached
|
||||
@@ -62,49 +124,6 @@ jobs:
|
||||
cache-from: ${{ matrix.cache-from }}
|
||||
cache-to: ${{ (github.event_name != 'pull_request') && matrix.cache-to || '' }}
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
|
||||
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
# Push by digest only (no tag). The merge job assembles the
|
||||
# tagged manifest list. `push-by-digest=true` is docker's recommended
|
||||
# pattern for multi-runner multi-platform builds.
|
||||
- name: Push ${{ matrix.arch }} by digest
|
||||
id: push
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
|
||||
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
|
||||
with:
|
||||
context: .
|
||||
file: Dockerfile
|
||||
platforms: ${{ matrix.platform }}
|
||||
labels: |
|
||||
org.opencontainers.image.revision=${{ github.sha }}
|
||||
build-args: |
|
||||
HERMES_GIT_SHA=${{ github.sha }}
|
||||
outputs: type=image,name=${{ env.IMAGE_NAME }},push-by-digest=true,name-canonical=true,push=true
|
||||
cache-from: ${{ matrix.cache-from }}
|
||||
cache-to: ${{ matrix.cache-to }}
|
||||
|
||||
# Write the digest to a file and upload it as an artifact so the
|
||||
# merge job can stitch both per-arch digests into a manifest list.
|
||||
- name: Export digest
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
|
||||
run: |
|
||||
mkdir -p /tmp/digests
|
||||
digest="${{ steps.push.outputs.digest }}"
|
||||
touch "/tmp/digests/${digest#sha256:}"
|
||||
|
||||
- name: Upload digest artifact
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: digest-${{ matrix.arch }}
|
||||
path: /tmp/digests/*
|
||||
if-no-files-found: error
|
||||
retention-days: 1
|
||||
|
||||
# Run the docker-integration test suite against the freshly-built
|
||||
# image already loaded into the local daemon (`:test`).
|
||||
@@ -122,9 +141,16 @@ jobs:
|
||||
# ---------------------------------------------------------------------
|
||||
- name: Install uv (for docker tests)
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
|
||||
# raw.githubusercontent.com every job; transient fetch failures
|
||||
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
|
||||
version: "0.9.28"
|
||||
|
||||
- name: Set up Python 3.11 (for docker tests)
|
||||
run: uv python install 3.11
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv python install 3.11
|
||||
|
||||
- name: Install Python dependencies (for docker tests)
|
||||
# ``dev`` extra pulls in pytest, pytest-asyncio —
|
||||
@@ -145,7 +171,88 @@ jobs:
|
||||
OPENAI_API_KEY: ""
|
||||
NOUS_API_KEY: ""
|
||||
run: |
|
||||
scripts/run_tests.sh tests/docker/ --file-timeout 600
|
||||
# Each of these tests drives a container, so the docker daemon sets
|
||||
# the limit and not the processor. This caps the workers. The
|
||||
# default from run_tests.sh is cpu_count*2, which starts 64
|
||||
# containers together on the 32-core amd64 runner.
|
||||
HERMES_TEST_WORKERS=$(nproc) scripts/run_tests.sh tests/docker/ --file-timeout 600
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Rebuild and push each architecture only after the unprivileged build/test
|
||||
# matrix passes. This job is the sole Docker Hub credential boundary.
|
||||
# ---------------------------------------------------------------------------
|
||||
publish:
|
||||
if: github.repository == 'NousResearch/hermes-agent' && (github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release')
|
||||
needs: [build]
|
||||
environment: container-publish
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- arch: amd64
|
||||
runner: ubuntu-latest-32-core
|
||||
platform: linux/amd64
|
||||
cache-from: type=gha,scope=docker-amd64
|
||||
cache-to: type=gha,mode=max,scope=docker-amd64
|
||||
# Native arm64 for the same reason as the build matrix above.
|
||||
- arch: arm64
|
||||
runner: ubuntu-latest-32-arm-core
|
||||
platform: linux/arm64
|
||||
cache-from: type=gha,scope=docker-arm64
|
||||
cache-to: type=gha,mode=max,scope=docker-arm64
|
||||
runs-on: ${{ matrix.runner }}
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout trusted source
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
# Retry once on transient Docker Hub / buildkit pull failures.
|
||||
# See build job for rationale; same pattern.
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
continue-on-error: true
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
- name: Set up Docker Buildx (retry)
|
||||
if: steps.buildx.outcome == 'failure'
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
# Push by digest only (no tag). The merge job assembles the tagged
|
||||
# manifest list after both architecture publishers complete.
|
||||
- name: Push ${{ matrix.arch }} by digest
|
||||
id: push
|
||||
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
|
||||
with:
|
||||
context: .
|
||||
file: Dockerfile
|
||||
platforms: ${{ matrix.platform }}
|
||||
labels: |
|
||||
org.opencontainers.image.revision=${{ github.sha }}
|
||||
build-args: |
|
||||
HERMES_GIT_SHA=${{ github.sha }}
|
||||
outputs: type=image,name=${{ env.IMAGE_NAME }},push-by-digest=true,name-canonical=true,push=true
|
||||
cache-from: ${{ matrix.cache-from }}
|
||||
cache-to: ${{ matrix.cache-to }}
|
||||
|
||||
- name: Export digest
|
||||
run: |
|
||||
mkdir -p /tmp/digests
|
||||
digest="${{ steps.push.outputs.digest }}"
|
||||
touch "/tmp/digests/${digest#sha256:}"
|
||||
|
||||
- name: Upload digest artifact
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: digest-${{ matrix.arch }}
|
||||
path: /tmp/digests/*
|
||||
if-no-files-found: error
|
||||
retention-days: 1
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stitch both per-arch digests into a single tagged multi-arch manifest.
|
||||
@@ -158,8 +265,9 @@ jobs:
|
||||
merge:
|
||||
if: github.repository == 'NousResearch/hermes-agent' && (github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release')
|
||||
runs-on: ubuntu-latest
|
||||
needs: [build]
|
||||
needs: [publish]
|
||||
timeout-minutes: 10
|
||||
environment: container-publish
|
||||
steps:
|
||||
- name: Download digests
|
||||
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
|
||||
@@ -168,7 +276,15 @@ jobs:
|
||||
pattern: digest-*
|
||||
merge-multiple: true
|
||||
|
||||
# Retry once on transient Docker Hub / buildkit pull failures.
|
||||
# See build job for rationale; same pattern.
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
continue-on-error: true
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
- name: Set up Docker Buildx (retry)
|
||||
if: steps.buildx.outcome == 'failure'
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
|
||||
@@ -15,10 +15,14 @@ jobs:
|
||||
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 22
|
||||
node-version: 26
|
||||
cache: npm
|
||||
cache-dependency-path: website/package-lock.json
|
||||
|
||||
- name: grab npm 12
|
||||
run: |
|
||||
npm i -g npm@12
|
||||
|
||||
- name: Install website dependencies
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
@@ -45,5 +49,8 @@ jobs:
|
||||
working-directory: website
|
||||
|
||||
- name: Build Docusaurus
|
||||
run: npm run build
|
||||
# Build only the default (en) locale in CI — the full bilingual
|
||||
# build runs in deploy-site.yml on push-to-main / release.
|
||||
# Same pattern Docusaurus uses internally (build:fast --locale en).
|
||||
run: npm run build:fast
|
||||
working-directory: website
|
||||
|
||||
@@ -2,6 +2,10 @@ name: E2E Desktop
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
outputs:
|
||||
review_status:
|
||||
description: Screenshot and visual-diff status for the CI review comment.
|
||||
value: ${{ jobs.e2e.outputs.review_status }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -13,8 +17,13 @@ concurrency:
|
||||
jobs:
|
||||
e2e:
|
||||
name: Playwright E2E (Linux)
|
||||
runs-on: ubuntu-latest
|
||||
# This job builds the renderer and the electron bundle, then drives a real
|
||||
# Electron app under xvfb. vite, tsc and the Playwright workers all scale
|
||||
# with the core count.
|
||||
runs-on: ubuntu-latest-32-core
|
||||
timeout-minutes: 20
|
||||
outputs:
|
||||
review_status: ${{ steps.review-status.outputs.review_status }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
@@ -33,8 +42,13 @@ jobs:
|
||||
# ── Node ───────────────────────────────────────────────────────────
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 22
|
||||
node-version: 26
|
||||
cache: npm
|
||||
|
||||
- name: grab npm 12
|
||||
run: |
|
||||
npm i -g npm@12
|
||||
|
||||
# Full npm ci (not --ignore-scripts): electron's postinstall
|
||||
# downloads the binary we launch, and node-pty's native build is
|
||||
# needed for the terminal pane.
|
||||
@@ -46,19 +60,29 @@ jobs:
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pin the uv version: unpinned, setup-uv resolves "latest" by
|
||||
# fetching a manifest from raw.githubusercontent.com on EVERY job —
|
||||
# a transient fetch failure fails the whole job (2026-07-28 slice-5
|
||||
# incident). Pinned, the binary downloads directly; no manifest hop.
|
||||
version: '0.9.28'
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
pyproject.toml
|
||||
uv.lock
|
||||
|
||||
- name: Set up Python 3.11
|
||||
run: uv python install 3.11
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv python install 3.11
|
||||
|
||||
- name: Install Python dependencies
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev
|
||||
|
||||
# ── Build desktop app ─────────────────────────────────────────────
|
||||
- run: npm run --prefix apps/desktop build
|
||||
# The Playwright step below runs `npm run build` before testing so
|
||||
# dist/ is always fresh — no separate build step needed here.
|
||||
|
||||
# ── Restore visual baseline screenshots from main ──────────────────
|
||||
# Baselines are generated on main (via --update-snapshots) and cached.
|
||||
@@ -79,24 +103,26 @@ jobs:
|
||||
# xvfb runs at a fixed 1280x1024 screen so the 1220x800 Electron
|
||||
# window always has a consistent viewport for screenshot comparison.
|
||||
# On main, we run with --update-snapshots to generate baselines.
|
||||
# `npm run test:e2e` builds dist/ as a pretest hook so the renderer
|
||||
# is always fresh — no separate build step needed.
|
||||
- name: Run Playwright E2E tests
|
||||
working-directory: apps/desktop
|
||||
run: |
|
||||
if [ "${{ github.ref_name }}" = "main" ]; then
|
||||
echo "On main — generating/updating baseline screenshots"
|
||||
xvfb-run -a --server-args="-screen 0 1280x1024x24" \
|
||||
npm run build && xvfb-run -a --server-args="-screen 0 1280x1024x24" \
|
||||
npx playwright test --reporter=list --update-snapshots
|
||||
else
|
||||
echo "On PR — comparing against cached baselines"
|
||||
xvfb-run -a --server-args="-screen 0 1280x1024x24" \
|
||||
npm run build && xvfb-run -a --server-args="-screen 0 1280x1024x24" \
|
||||
npx playwright test --reporter=list
|
||||
fi
|
||||
env:
|
||||
CI: "true"
|
||||
CI: 'true'
|
||||
# Ensure no real API keys leak into the test env.
|
||||
OPENROUTER_API_KEY: ""
|
||||
OPENAI_API_KEY: ""
|
||||
NOUS_API_KEY: ""
|
||||
OPENROUTER_API_KEY: ''
|
||||
OPENAI_API_KEY: ''
|
||||
NOUS_API_KEY: ''
|
||||
|
||||
# ── Save updated baselines to cache (main only) ───────────────────
|
||||
- name: Save updated baselines to cache
|
||||
@@ -143,6 +169,50 @@ jobs:
|
||||
overwrite: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
- name: Build screenshot review status
|
||||
id: review-status
|
||||
if: always()
|
||||
working-directory: apps/desktop
|
||||
env:
|
||||
RESULTS_URL: ${{ steps.upload-results.outputs.artifact-url }}
|
||||
run: |
|
||||
python3 ../../scripts/ci/e2e_screenshot_status.py \
|
||||
--results-dir test-results \
|
||||
--manifest-output /tmp/e2e-screenshot-manifest.json \
|
||||
--evidence-dir /tmp/e2e-evidence \
|
||||
--artifact-url "$RESULTS_URL" \
|
||||
--output /tmp/e2e-review-status.json
|
||||
{
|
||||
echo 'review_status<<__E2E_REVIEW_STATUS__'
|
||||
cat /tmp/e2e-review-status.json
|
||||
echo '__E2E_REVIEW_STATUS__'
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
cp /tmp/e2e-review-status.json review-status.json
|
||||
|
||||
- name: Upload review status artifact
|
||||
if: always() && steps.review-status.outcome != 'skipped'
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: review-status-e2e-desktop
|
||||
path: apps/desktop/review-status.json
|
||||
retention-days: 1
|
||||
overwrite: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
# The trusted workflow_run publisher consumes only this flat, bounded
|
||||
# artifact. It turns selected images into GitHub attachment URLs; it
|
||||
# never checks out or runs this PR's code.
|
||||
- name: Upload inline E2E evidence
|
||||
if: always() && github.ref_name != 'main'
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: e2e-evidence-${{ github.sha }}
|
||||
path: /tmp/e2e-evidence
|
||||
retention-days: 14
|
||||
overwrite: true
|
||||
if-no-files-found: error
|
||||
|
||||
# ── Generate step summary with visual diff info ───────────────────
|
||||
# Parse the JSON report + scan for diff images, then post a summary
|
||||
# to the GitHub Actions step output so reviewers can see what changed
|
||||
@@ -156,49 +226,50 @@ jobs:
|
||||
RESULTS_URL: ${{ steps.upload-results.outputs.artifact-url }}
|
||||
DIFFS_URL: ${{ steps.upload-diffs.outputs.artifact-url }}
|
||||
run: |
|
||||
echo "## Desktop E2E — Visual Diff Report" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "" >> "$GITHUB_STEP_SUMMARY"
|
||||
{
|
||||
echo "## Desktop E2E — Visual Diff Report"
|
||||
echo ""
|
||||
|
||||
# Count diff images (playwright writes *-diff.png on mismatch)
|
||||
DIFF_COUNT=$(find test-results -name '*-diff.png' 2>/dev/null | wc -l)
|
||||
ACTUAL_COUNT=$(find test-results -name '*-actual.png' 2>/dev/null | wc -l)
|
||||
|
||||
if [ "$DIFF_COUNT" -eq 0 ]; then
|
||||
echo "✅ All $ACTUAL_COUNT screenshot(s) matched their baselines (or no baselines existed yet)." >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "✅ All $ACTUAL_COUNT screenshot(s) matched their baselines (or no baselines existed yet)."
|
||||
else
|
||||
echo "📸 **$DIFF_COUNT of $ACTUAL_COUNT screenshot(s) differ from baseline:**" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "| Test | Diff | Actual | Expected |" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "|------|------|--------|----------|" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "📸 **$DIFF_COUNT of $ACTUAL_COUNT screenshot(s) differ from baseline:**"
|
||||
echo ""
|
||||
echo "| Test | Diff | Actual | Expected |"
|
||||
echo "|------|------|--------|----------|"
|
||||
|
||||
# List each diff image with a link to the artifact
|
||||
for diff in $(find test-results -name '*-diff.png' 2>/dev/null | sort); do
|
||||
base=$(echo "$diff" | sed 's/-diff\.png$//')
|
||||
base=${diff%-diff.png}
|
||||
test_name=$(basename "$base")
|
||||
echo "| $test_name | [diff]($diff) | [actual](${base}-actual.png) | [expected](${base}-expected.png) |" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "| $test_name | [diff]($diff) | [actual](${base}-actual.png) | [expected](${base}-expected.png) |"
|
||||
done
|
||||
fi
|
||||
|
||||
echo "" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "📥 **Artifacts:**" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo ""
|
||||
echo "📥 **Artifacts:**"
|
||||
echo ""
|
||||
if [ -n "$RESULTS_URL" ]; then
|
||||
echo "- [playwright-test-results]($RESULTS_URL) — all screenshots (actual + expected + diff) + traces" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "- [playwright-test-results]($RESULTS_URL) — all screenshots (actual + expected + diff) + traces"
|
||||
fi
|
||||
if [ -n "$REPORT_URL" ]; then
|
||||
echo "- [playwright-report]($REPORT_URL) — interactive HTML report" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "- [playwright-report]($REPORT_URL) — interactive HTML report"
|
||||
fi
|
||||
if [ -n "$DIFFS_URL" ]; then
|
||||
echo "- [visual-diffs]($DIFFS_URL) — just the diffed screenshots (small, fast to review)" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "- [visual-diffs]($DIFFS_URL) — just the diffed screenshots (small, fast to review)"
|
||||
fi
|
||||
echo "" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "**To update baselines:** merge to main (baselines auto-update on main runs) or run \`npx playwright test --update-snapshots\` locally." >> "$GITHUB_STEP_SUMMARY"
|
||||
echo ""
|
||||
echo "**To update baselines:** merge to main (baselines auto-update on main runs) or run \`npx playwright test --update-snapshots\` locally."
|
||||
|
||||
# Also parse the JSON report for pass/fail counts
|
||||
if [ -f playwright-report/results.json ]; then
|
||||
echo "" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "### Test Results" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo ""
|
||||
echo "### Test Results"
|
||||
echo ""
|
||||
node -e "
|
||||
const r = require('./playwright-report/results.json');
|
||||
const stats = r.stats || {};
|
||||
@@ -208,5 +279,6 @@ jobs:
|
||||
console.log('| ❌ Failed | ' + (stats.unexpected || 0) + ' |');
|
||||
console.log('| ⏭️ Skipped | ' + (stats.skipped || 0) + ' |');
|
||||
console.log('| 🔄 Flaky | ' + (stats.flaky || 0) + ' |');
|
||||
" >> "$GITHUB_STEP_SUMMARY" 2>/dev/null || true
|
||||
" 2>/dev/null || true
|
||||
fi
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
@@ -44,6 +44,7 @@ jobs:
|
||||
if ! BASE=$(git merge-base origin/main HEAD 2>/dev/null) || [ -z "$BASE" ]; then
|
||||
STATUS='[{"source":"unrelated histories","results":[{"kind":"action_required","title":"Unrelated histories","summary":"This PR has no common ancestor with main.","detail":"","how_to_fix":"Rebase your changes onto current main:\n```\ngit fetch origin main\ngit checkout -b fix-branch origin/main\n# re-apply your changes (cherry-pick, copy files, etc.)\ngit push -f origin fix-branch\n```\n"}]}]'
|
||||
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
|
||||
echo "review_status=${STATUS}" > review-status.json
|
||||
echo ""
|
||||
echo "::error::This PR has no common ancestor with main."
|
||||
echo ""
|
||||
@@ -66,3 +67,15 @@ jobs:
|
||||
fi
|
||||
echo "::notice::Common ancestor with main: $BASE"
|
||||
echo "review_status=[]" >> "$GITHUB_OUTPUT"
|
||||
echo "review_status=[]" > review-status.json
|
||||
|
||||
- name: Upload review status artifact
|
||||
if: always() && steps.merge-base-check.outcome != 'skipped'
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: review-status-history-check
|
||||
path: review-status.json
|
||||
retention-days: 1
|
||||
overwrite: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
name: Infographic Check
|
||||
|
||||
# Rejects PRs that commit PR-infographic images into the repo.
|
||||
#
|
||||
# PR infographics are rendered to an image-provider URL (fal.media) and
|
||||
# embedded in the PR *description*. The PR body is the archive; the binary
|
||||
# never belongs in git history.
|
||||
#
|
||||
# This has now leaked twice. PR #48261 removed the first batch, PR #54564
|
||||
# removed a second batch and added `infographic/` to `.gitignore` — but
|
||||
# `.gitignore` only stops *accidental* `git add`. It does nothing against
|
||||
# `git add -f`, and it does nothing for a path that does not literally match
|
||||
# the ignore pattern. Nine more PNGs (~14MB) were committed in the four
|
||||
# weeks AFTER that rule landed, plus PR #70552 caught an `infograficos/`
|
||||
# spelling that sidestepped the pattern entirely.
|
||||
#
|
||||
# A passive ignore rule cannot enforce a policy. This check can.
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
outputs:
|
||||
review_status:
|
||||
description: "JSON array of review_status objects for the synthesizer."
|
||||
value: ${{ jobs.check-no-committed-infographics.outputs.review_status }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
check-no-committed-infographics:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
outputs:
|
||||
review_status: ${{ steps.infographic-check.outputs.review_status }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- id: infographic-check
|
||||
name: Reject committed PR-infographic images
|
||||
run: |
|
||||
# Match on the IMAGE, not on a directory name. Keying this to
|
||||
# `infographic/` is what let `infograficos/` through in #70552 —
|
||||
# any localized or typo'd directory would sidestep it again.
|
||||
# Instead: find tracked raster images whose path contains an
|
||||
# infographic-ish segment, in any spelling, at any depth.
|
||||
#
|
||||
# `docs/assets` and `website/` legitimately hold product imagery
|
||||
# and are excluded; those are referenced from shipped docs pages.
|
||||
OFFENDERS=$(git ls-files -z \
|
||||
| tr '\0' '\n' \
|
||||
| grep -iE '(^|/)(infograph|infograf)[^/]*/' \
|
||||
| grep -iE '\.(png|jpe?g|webp|gif)$' \
|
||||
|| true)
|
||||
|
||||
if [ -n "$OFFENDERS" ]; then
|
||||
COUNT=$(printf '%s\n' "$OFFENDERS" | wc -l | tr -d ' ')
|
||||
STATUS='[{"source":"committed infographics","results":[{"kind":"action_required","title":"PR infographic committed to the repo","summary":"Infographic images belong in the PR description, never in git.","detail":"","how_to_fix":"Untrack the image and reference the provider URL from the PR body instead:\n```\ngit rm --cached <path-to-image>\n```\nThen put it in the PR description:\n```\n## Infographic\n\n\n```\n"}]}]'
|
||||
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
|
||||
echo ""
|
||||
echo "::error::${COUNT} PR-infographic image(s) are tracked in git."
|
||||
echo ""
|
||||
printf '%s\n' "$OFFENDERS" | sed 's/^/ /'
|
||||
echo ""
|
||||
echo "PR infographics are rendered to an image-provider URL and"
|
||||
echo "embedded in the PR DESCRIPTION. The PR body is the archive —"
|
||||
echo "the binary never enters git history."
|
||||
echo ""
|
||||
echo "This rule has been re-established twice already (#48261,"
|
||||
echo "#54564) and leaked both times, because .gitignore cannot stop"
|
||||
echo "'git add -f' or a differently-spelled directory (#70552)."
|
||||
echo ""
|
||||
echo "To fix:"
|
||||
echo " git rm --cached <path> # keeps your local copy"
|
||||
echo " # then embed the provider URL in the PR description"
|
||||
exit 1
|
||||
fi
|
||||
echo "::notice::No committed PR-infographic images."
|
||||
echo "review_status=[]" >> "$GITHUB_OUTPUT"
|
||||
@@ -0,0 +1,122 @@
|
||||
name: Install & Update E2E (reusable)
|
||||
|
||||
# Runs ONE update route against ONE starting commit, in the dev sandbox, with a
|
||||
# real install (uv, a managed Python, Node, the venv) behind it.
|
||||
#
|
||||
# Reusable so callers can fan out over the combinations that matter -- update
|
||||
# from the tip vs. from an older release, `hermes update` vs. re-running the
|
||||
# installer -- without duplicating the runner setup. Each leg is independent:
|
||||
# its own sandbox, its own install, nothing rewound or shared.
|
||||
#
|
||||
# Call it:
|
||||
#
|
||||
# jobs:
|
||||
# tip:
|
||||
# uses: ./.github/workflows/install-e2e-run.yml
|
||||
# with:
|
||||
# route: update
|
||||
# install-ref: refs/heads/main
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
route:
|
||||
description: 'Update path to exercise: update (hermes update) or installer (re-run install.sh).'
|
||||
required: true
|
||||
type: string
|
||||
install-ref:
|
||||
description: 'What to install before updating: a branch, a tag (v2026.7.7), or a SHA reachable from main.'
|
||||
required: false
|
||||
type: string
|
||||
default: refs/heads/main
|
||||
runner:
|
||||
description: 'Runner label.'
|
||||
required: false
|
||||
type: string
|
||||
default: ubuntu-latest
|
||||
timeout-minutes:
|
||||
description: 'Job timeout. A cold run installs real toolchains twice.'
|
||||
required: false
|
||||
type: number
|
||||
default: 45
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
e2e:
|
||||
name: ${{ inputs.route }} from ${{ inputs.install-ref }}
|
||||
runs-on: ${{ inputs.runner }}
|
||||
timeout-minutes: ${{ inputs.timeout-minutes }}
|
||||
|
||||
steps:
|
||||
# Full history: the sandbox fetches the starting commit and the test
|
||||
# compares against this commit, so a shallow clone is not enough.
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
# bubblewrap + slirp4netns are what the sandbox is built on; util-linux
|
||||
# supplies the `unshare` that builds the multi-uid userns for the
|
||||
# user-level (non-root) install.
|
||||
- name: Install sandbox dependencies
|
||||
run: |
|
||||
set -euo pipefail
|
||||
sudo apt-get update -qq
|
||||
sudo apt-get install -y -qq bubblewrap slirp4netns uidmap util-linux
|
||||
|
||||
# Ubuntu 24.04 restricts unprivileged user namespaces through AppArmor,
|
||||
# which is exactly what bwrap needs. Report the state before touching it
|
||||
# so a future runner-image change is visible in the log rather than
|
||||
# silently altering what this job proves.
|
||||
- name: Permit unprivileged user namespaces
|
||||
run: |
|
||||
set -euo pipefail
|
||||
echo "--- kernel userns settings (before)"
|
||||
sysctl kernel.unprivileged_userns_clone 2>/dev/null || echo " (sysctl absent)"
|
||||
sysctl kernel.apparmor_restrict_unprivileged_userns 2>/dev/null || echo " (sysctl absent)"
|
||||
if sysctl -n kernel.apparmor_restrict_unprivileged_userns >/dev/null 2>&1; then
|
||||
sudo sysctl -w kernel.apparmor_restrict_unprivileged_userns=0
|
||||
fi
|
||||
echo "--- subuid/subgid for $(id -un)"
|
||||
grep "^$(id -un):" /etc/subuid /etc/subgid || echo " (none — sandbox will say so)"
|
||||
|
||||
- name: Run install + update E2E
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tests/install/install-update-e2e.sh \
|
||||
--route '${{ inputs.route }}' \
|
||||
--install-ref '${{ inputs.install-ref }}'
|
||||
env:
|
||||
# Outside the workspace on purpose: the script creates this directory
|
||||
# up front, and an untracked dir inside the repo makes the worktree
|
||||
# dirty -- which dev-sandbox reacts to by snapshotting the working
|
||||
# copy into a fresh fake-main commit on every invocation, moving the
|
||||
# update target mid-run.
|
||||
HERMES_E2E_LOG_DIR: ${{ runner.temp }}/e2e-logs
|
||||
|
||||
# Artifact names cannot contain '/', and install-ref may be a full ref
|
||||
# like refs/heads/main. GitHub Actions expressions have no string-replace
|
||||
# function, so build the safe name here. Runs even on failure -- that is
|
||||
# exactly when the logs are wanted.
|
||||
- name: Build artifact name
|
||||
if: always()
|
||||
id: artifact
|
||||
run: |
|
||||
set -euo pipefail
|
||||
safe_ref='${{ inputs.install-ref }}'
|
||||
safe_ref="${safe_ref//\//-}"
|
||||
echo "name=install-e2e-${{ inputs.route }}-${safe_ref}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The installer's own transcripts say far more than the assertion that
|
||||
# tripped when a real install breaks.
|
||||
- name: Upload installer logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
# Unique per leg: a matrix over releases runs this workflow several
|
||||
# times per route, and same-named artifacts collide.
|
||||
name: ${{ steps.artifact.outputs.name }}-${{ github.sha }}
|
||||
path: ${{ runner.temp }}/e2e-logs
|
||||
retention-days: 14
|
||||
if-no-files-found: ignore
|
||||
@@ -0,0 +1,110 @@
|
||||
name: Install & Update E2E
|
||||
|
||||
# Can a user on a released version get to this commit?
|
||||
#
|
||||
# For each release we sample, a leg installs that release through the real
|
||||
# `curl | install.sh` one-liner (uv, a managed Python, Node, the venv) inside
|
||||
# scripts/dev-sandbox.sh, then applies one update route and requires the
|
||||
# checkout to land on this commit with a working `hermes`.
|
||||
#
|
||||
# The starting versions are chosen at runtime from the repo's release tags
|
||||
# (scripts/sandbox/pick-release-tags.sh): newest, oldest, and a spread between.
|
||||
# A hardcoded list would stop covering the newest release the day after it
|
||||
# ships, and would pin an "oldest" that nobody still runs.
|
||||
#
|
||||
# Triggers:
|
||||
# * every 12 hours, so upstream drift (a new uv, a Node bump, a PyPI change)
|
||||
# surfaces on a schedule rather than in someone's review cycle;
|
||||
# * when a release tag is created -- the moment the set of versions users can
|
||||
# update FROM changes, and the moment a broken updater would strand them;
|
||||
# * manually, where you can pick the route and how many releases to sample.
|
||||
#
|
||||
# Deliberately NOT on pull_request: a leg takes ~11 minutes of real toolchain
|
||||
# installation, and the matrix multiplies that. Updating is release-shaped work,
|
||||
# so it is gated on releases and the clock instead.
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
route:
|
||||
description: 'Which update route to exercise.'
|
||||
required: false
|
||||
type: choice
|
||||
default: both
|
||||
options: [both, update, installer]
|
||||
tag-count:
|
||||
description: 'How many release tags to sample (newest, oldest, and a spread between).'
|
||||
required: false
|
||||
type: string
|
||||
default: '5'
|
||||
schedule:
|
||||
# Every 12 hours, off the hour to avoid the top-of-hour runner crunch.
|
||||
- cron: '20 7,19 * * *'
|
||||
push:
|
||||
tags:
|
||||
# Release tags only: the repo also carries backup/* and one-off tags.
|
||||
- 'v[0-9]+.[0-9]+.[0-9]+'
|
||||
- 'v[0-9]+.[0-9]+.[0-9]+.[0-9]+'
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: install-e2e-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
# Which released versions do we test updating FROM? Resolved once and shared
|
||||
# by both route matrices, so the two routes cover the same set.
|
||||
pick-releases:
|
||||
name: Pick release tags
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
outputs:
|
||||
tags: ${{ steps.pick.outputs.tags }}
|
||||
steps:
|
||||
# This job only reads tag names and runs one script, so take the cheap
|
||||
# checkout: no blobs (filter), no other files (sparse), but DO fetch tags
|
||||
# -- they are the whole input, and the default shallow checkout has none.
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
filter: blob:none
|
||||
fetch-tags: true
|
||||
sparse-checkout: scripts/sandbox/pick-release-tags.sh
|
||||
sparse-checkout-cone-mode: false
|
||||
- id: pick
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tags="$(scripts/sandbox/pick-release-tags.sh --count '${{ inputs.tag-count || 5 }}')"
|
||||
echo "Testing updates from: $tags"
|
||||
echo "tags=$tags" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# `hermes update` -- the route most users take.
|
||||
update:
|
||||
if: github.event_name != 'workflow_dispatch' || inputs.route != 'installer'
|
||||
needs: pick-releases
|
||||
strategy:
|
||||
# One release breaking is worth knowing about even if another already
|
||||
# failed, so let every leg report.
|
||||
fail-fast: false
|
||||
matrix:
|
||||
install-ref: ${{ fromJSON(needs.pick-releases.outputs.tags) }}
|
||||
uses: ./.github/workflows/install-e2e-run.yml
|
||||
with:
|
||||
route: update
|
||||
install-ref: ${{ matrix.install-ref }}
|
||||
|
||||
# Re-running the curl one-liner over an existing checkout: autostash + pull
|
||||
# rather than the updater's own git handling.
|
||||
installer:
|
||||
if: github.event_name != 'workflow_dispatch' || inputs.route != 'update'
|
||||
needs: pick-releases
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: 3
|
||||
matrix:
|
||||
install-ref: ${{ fromJSON(needs.pick-releases.outputs.tags) }}
|
||||
uses: ./.github/workflows/install-e2e-run.yml
|
||||
with:
|
||||
route: installer
|
||||
install-ref: ${{ matrix.install-ref }}
|
||||
@@ -0,0 +1,38 @@
|
||||
name: Installer tests
|
||||
|
||||
# scripts/install.ps1's PowerShell tests. They exercise the installer as a real
|
||||
# subprocess, and every path contract they assert (8.3 short-name aliases,
|
||||
# Git Bash layouts, provider-cmdlet behavior) is Windows-specific — so they need
|
||||
# a Windows runner. Before this workflow existed the files were in the tree but
|
||||
# nothing ever ran them.
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: installer-tests-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
powershell:
|
||||
name: PowerShell installer tests
|
||||
runs-on: windows-latest
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
# Windows PowerShell 5.1 as well as pwsh 7: install.ps1 is delivered via
|
||||
# `irm | iex` into whatever shell the user already has, and 5.1 is what
|
||||
# ships with Windows. A construct that only parses under 7 is a broken
|
||||
# installer for most of the people hitting it.
|
||||
- name: 8.3 short-path normalization (pwsh 7)
|
||||
shell: pwsh
|
||||
run: pwsh -NoProfile -ExecutionPolicy Bypass -File scripts/tests/test-install-ps1-longpath.ps1
|
||||
|
||||
- name: 8.3 short-path normalization (Windows PowerShell 5.1)
|
||||
shell: powershell
|
||||
run: powershell -NoProfile -ExecutionPolicy Bypass -File scripts/tests/test-install-ps1-longpath.ps1
|
||||
@@ -67,9 +67,13 @@ jobs:
|
||||
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 22
|
||||
node-version: 26
|
||||
cache: npm
|
||||
|
||||
- name: grab npm 12
|
||||
run: |
|
||||
npm i -g npm@12
|
||||
|
||||
# --ignore-scripts: eslint only needs TS sources + eslint packages.
|
||||
- uses: ./.github/actions/retry
|
||||
with:
|
||||
@@ -85,20 +89,33 @@ jobs:
|
||||
- name: Produce patch
|
||||
id: produce-patch
|
||||
run: |
|
||||
if git diff --quiet; then
|
||||
# Exclude every file that the dep-version-gate ruleset guards
|
||||
# (package manifests, eslint configs, workflow files). A patch
|
||||
# that contains one of these files makes the bot PR wait for a
|
||||
# team review, and auto-merge then stops. The check step in
|
||||
# typecheck.yml still reports their lint errors.
|
||||
# The (glob) magic makes "**/" also match files at the repo
|
||||
# root, which plain pathspec wildcards do not.
|
||||
EXCLUDES=(
|
||||
':(exclude,glob)**/package.json'
|
||||
':(exclude,glob)**/package-lock.json'
|
||||
':(exclude,glob)**/eslint.config.*'
|
||||
':(exclude,glob).github/**'
|
||||
)
|
||||
if git diff --quiet -- . "${EXCLUDES[@]}"; then
|
||||
echo "No fixes needed."
|
||||
echo "has-fixes=false" >> "$GITHUB_OUTPUT"
|
||||
# Empty patch signals "nothing to do" to apply-patch.
|
||||
: > js-fix.patch
|
||||
else
|
||||
git diff > js-fix.patch
|
||||
git diff -- . "${EXCLUDES[@]}" > js-fix.patch
|
||||
echo "has-fixes=true" >> "$GITHUB_OUTPUT"
|
||||
echo "Patch size: $(wc -c < js-fix.patch) bytes"
|
||||
|
||||
# Reject patches that touch anything outside JS/TS/JSON sources.
|
||||
# `npm run fix` should only ever modify those; anything else means
|
||||
# eslint/prettier or a plugin went rogue and we refuse to ship it.
|
||||
BAD=$(git diff --name-only | grep -vE '\.(js|cjs|mjs|ts|tsx|json)$' || true)
|
||||
BAD=$(git diff --name-only -- . "${EXCLUDES[@]}" | grep -vE '\.(js|cjs|mjs|ts|tsx|json)$' || true)
|
||||
if [ -n "$BAD" ]; then
|
||||
echo "::error::Refusing to upload patch — touches disallowed files:"
|
||||
echo "$BAD"
|
||||
@@ -122,6 +139,7 @@ jobs:
|
||||
if: needs.generate-patch.outputs.has-fixes == 'true'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
environment: trusted-automation
|
||||
permissions:
|
||||
contents: write # needed to push to bot/js-autofix
|
||||
pull-requests: write # needed for PR creation + auto-merge
|
||||
@@ -132,7 +150,7 @@ jobs:
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ secrets.APP_CLIENT_ID }}
|
||||
client-id: ${{ vars.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- name: Download patch
|
||||
|
||||
@@ -5,47 +5,79 @@ on:
|
||||
workflow_call:
|
||||
|
||||
jobs:
|
||||
workspaces:
|
||||
name: List npm workspaces
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
outputs:
|
||||
packages: ${{ steps.set-matrix.outputs.packages }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 22
|
||||
cache: npm
|
||||
- uses: ./.github/actions/retry
|
||||
with:
|
||||
command: npm ci --ignore-scripts
|
||||
- id: set-matrix
|
||||
run: |
|
||||
PACKAGES=$(npm query .workspace | jq -c '[.[].location]')
|
||||
if [ "$PACKAGES" = "[]" ] || [ -z "$PACKAGES" ]; then
|
||||
echo "::error::Workspace discovery produced an empty package list — refusing to emit a zero-length matrix (would skip all JS/TS checks silently)."
|
||||
exit 1
|
||||
fi
|
||||
echo "packages=$PACKAGES" >> "$GITHUB_OUTPUT"
|
||||
|
||||
check:
|
||||
name: Typecheck & Test
|
||||
needs: workspaces
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
strategy:
|
||||
matrix:
|
||||
package: ${{ fromJson(needs.workspaces.outputs.packages) }}
|
||||
fail-fast: false # report all failures, not just the first one
|
||||
name: JS & TS checks
|
||||
# One 32-core job replaces a 14-leg matrix. The matrix spread about 612s
|
||||
# of check payload over 4-core runners. It paid about 371s of repeated
|
||||
# setup to do it: 14 checkouts, 14 node installs, 14 node_modules
|
||||
# restores.
|
||||
#
|
||||
# One larger runner installs one time. vitest, tsc and eslint each size
|
||||
# their own worker pool from the core count.
|
||||
runs-on: ubuntu-latest-32-core
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 22
|
||||
node-version: 26
|
||||
cache: npm
|
||||
|
||||
- name: grab npm 12
|
||||
run: |
|
||||
# No-op once the bundled npm is already 12.x — saves ~5-15s/job and
|
||||
# keeps the installed major aligned with the npm12 cache-key tag.
|
||||
npm --version | grep -q '^12\.' || npm i -g npm@12
|
||||
|
||||
# The ``cache: npm`` option of ``setup-node`` caches only the ~/.npm
|
||||
# tarball cache. The job then extracts the full workspace node_modules
|
||||
# again and runs the postinstalls again, which includes the Electron
|
||||
# binary fetch. This caches the installed tree itself, keyed on the
|
||||
# lockfile, and skips ``npm ci`` on an exact hit. There are no
|
||||
# restore-keys: a partial hit leaves a stale tree, so anything other
|
||||
# than an exact lockfile match reinstalls from the start.
|
||||
#
|
||||
# This install runs WITH scripts, so the tree holds the postinstall
|
||||
# artifacts. The postinstall of electron unpacks its binary into
|
||||
# node_modules/electron/dist, which is inside the cached tree.
|
||||
#
|
||||
# The ~/.cache/electron download cache stays out of the key on purpose.
|
||||
# ``npm ci`` is skipped on a hit, so nothing reads that cache. It only
|
||||
# makes the archive larger.
|
||||
- name: Restore node_modules
|
||||
id: node-modules-cache
|
||||
uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4
|
||||
with:
|
||||
path: |
|
||||
node_modules
|
||||
apps/*/node_modules
|
||||
ui-tui/node_modules
|
||||
ui-tui/packages/*/node_modules
|
||||
tests-js/node_modules
|
||||
web/node_modules
|
||||
key: node-modules-scripts-${{ runner.os }}-node26-npm12-${{ hashFiles('package-lock.json') }}
|
||||
|
||||
- uses: ./.github/actions/retry
|
||||
if: steps.node-modules-cache.outputs.cache-hit != 'true'
|
||||
with:
|
||||
command: npm ci
|
||||
- run: npm run --prefix ${{ matrix.package }} check
|
||||
- run: npm run --prefix ${{ matrix.package }} fix
|
||||
|
||||
# Every check runs at the same time. The step fails only after all of
|
||||
# them finish. There are two reasons this is not ``npm run --ws check``.
|
||||
#
|
||||
# * ``--ws`` is serial and stops at the first workspace that fails. A
|
||||
# run then reports one failure, where the matrix this replaced
|
||||
# reported every failure together.
|
||||
# * The unit of work is a CHECK, and not a workspace. apps/desktop is
|
||||
# most of the payload, and its own ``check`` is a serial && chain.
|
||||
# A spread across workspaces alone leaves that chain as the long
|
||||
# pole. This expands the ``check:*`` sub-scripts of a package, so
|
||||
# its lint, ui, electron and plugin suites all run together. That
|
||||
# is the same selection rule the old matrix job used.
|
||||
#
|
||||
# Discovery is ``npm query .workspace``. A new package or a new
|
||||
# ``check:*`` script needs no change here. An empty list is an error and
|
||||
# not an empty run, because an empty run reports green after it checks
|
||||
# nothing.
|
||||
- name: Run all workspace checks
|
||||
run: node .github/scripts/run-workspace-checks.mjs
|
||||
|
||||
@@ -2,13 +2,16 @@ name: Label rerun
|
||||
|
||||
# When the ``ci-reviewed`` label is added to a PR, rerun all failed jobs in
|
||||
# the latest CI run. This re-evaluates ``review-labels`` (which now sees the
|
||||
# label) and GitHub automatically reruns dependent jobs (``comment-live``,
|
||||
# ``all-checks-pass``) — so the review comment gets updated too.
|
||||
# label) and GitHub reruns the dependent ``all-checks-pass`` gate.
|
||||
#
|
||||
# If the CI run is still in progress when the label is added, we wait for it
|
||||
# to finish before rerunning (``gh run rerun`` only works on completed runs).
|
||||
# The wait can be long (20+ min for a full CI run), but it's better than
|
||||
# silently failing and leaving the reviewer stuck.
|
||||
# ``gh run rerun`` only works on a completed run. Thus this waits when the
|
||||
# run is still in progress. The wait is now short. The two slowest jobs are
|
||||
# the 40-minute comment poller and the 45-minute image build. Each one moved
|
||||
# to its own workflow, so a CI run ends when its required jobs end.
|
||||
#
|
||||
# The review comment updates without help. The poller in
|
||||
# ci-review-comment.yml watches the CI run through the API. It gets the
|
||||
# rerun results.
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
@@ -38,22 +41,25 @@ jobs:
|
||||
set -uo pipefail
|
||||
|
||||
# Find the latest CI run for this PR's head SHA.
|
||||
RUN_ID=$(gh run list \
|
||||
RUN_INFO=$(gh run list \
|
||||
--repo "$REPO" \
|
||||
--commit "$HEAD_SHA" \
|
||||
--workflow ci.yml \
|
||||
--workflow ci.yaml \
|
||||
--limit 1 \
|
||||
--json databaseId,status \
|
||||
--jq '.[0] | "\(.databaseId) \(.status)"' 2>/dev/null || true)
|
||||
|
||||
if [ -z "$RUN_ID" ]; then
|
||||
if [ -z "$RUN_INFO" ]; then
|
||||
echo "No CI run found for this PR — nothing to rerun."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Split "RUN_ID STATUS" into two vars.
|
||||
RUN_ID="${RUN_ID%% *}"
|
||||
STATUS="${RUN_ID##* }"
|
||||
# Split "RUN_ID STATUS" into two vars. Read STATUS from RUN_INFO,
|
||||
# not from the truncated RUN_ID. Both values came from the same
|
||||
# var before, which made STATUS the run id. Thus the wait branch
|
||||
# always ran.
|
||||
RUN_ID="${RUN_INFO%% *}"
|
||||
STATUS="${RUN_INFO##* }"
|
||||
|
||||
echo "Latest CI run: $RUN_ID (status: $STATUS)"
|
||||
|
||||
|
||||
@@ -40,6 +40,11 @@ jobs:
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
|
||||
# raw.githubusercontent.com every job; transient fetch failures
|
||||
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
|
||||
version: "0.9.28"
|
||||
|
||||
- name: Install ruff + ty
|
||||
uses: ./.github/actions/retry
|
||||
@@ -129,6 +134,11 @@ jobs:
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
|
||||
# raw.githubusercontent.com every job; transient fetch failures
|
||||
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
|
||||
version: "0.9.28"
|
||||
|
||||
- name: Install ruff
|
||||
uses: ./.github/actions/retry
|
||||
@@ -153,10 +163,18 @@ jobs:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v5
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
python-version: "3.11"
|
||||
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
|
||||
# raw.githubusercontent.com every job; transient fetch failures
|
||||
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
|
||||
version: "0.9.28"
|
||||
|
||||
- name: Set up Python 3.11
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv python install 3.11
|
||||
|
||||
- name: Run footgun checker
|
||||
run: python scripts/check-windows-footguns.py --all
|
||||
|
||||
@@ -79,12 +79,24 @@ jobs:
|
||||
|
||||
if [ "$CHANGED" = "true" ]; then
|
||||
CONTENT=$(cat /tmp/lockfile-diff.md | python3 -c "import sys,json; print(json.dumps(sys.stdin.read()))")
|
||||
STATUS="[{\"source\":\"lockfile-diff\",\"results\":[{\"kind\":\"action_required\",\"title\":\"package-lock.json\",\"summary\":\"Locked npm dependency versions changed.\",\"detail\":${CONTENT},\"how_to_fix\":\"Add the \`ci-reviewed\` label after verifying the version changes are expected.\"}]}"
|
||||
STATUS="[{\"source\":\"lockfile-diff\",\"results\":[{\"kind\":\"action_required\",\"title\":\"package-lock.json\",\"summary\":\"Locked npm dependency versions changed.\",\"detail\":${CONTENT},\"how_to_fix\":\"Add the \`ci-reviewed\` label after verifying the version changes are expected.\"}]}]"
|
||||
else
|
||||
STATUS="[]"
|
||||
fi
|
||||
|
||||
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
|
||||
echo "review_status=${STATUS}" > review-status.json
|
||||
|
||||
- name: Upload review status artifact
|
||||
if: always() && steps.emit-status.outcome != 'skipped'
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: review-status-lockfile-diff
|
||||
path: review-status.json
|
||||
retention-days: 1
|
||||
overwrite: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
- name: Upload diff artifact
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
name: Nix flake check
|
||||
|
||||
# Builds every output of the flake: the package, the devShell, and the 21
|
||||
# checks under nix/checks.nix — module evaluation, option parity, .env
|
||||
# assembly, service argv, and the rest.
|
||||
#
|
||||
# This workflow owns its triggers and ci.yml does not call it, for the reason
|
||||
# docker.yml gives: a reusable-workflow call holds the caller run in progress
|
||||
# for the full build, and GitHub refuses `gh run rerun` on a run that is still
|
||||
# in progress. One slow advisory job in the CI lane blocks every rerun of the
|
||||
# fast required jobs beside it. A separate run reruns and cancels on its own.
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches: [main]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# PR runs collapse to the newest commit. A push to main is never cancelled:
|
||||
# each one saves the store cache that later PRs restore from, so cancelling a
|
||||
# merge would leave the next PR to build from nothing.
|
||||
concurrency:
|
||||
group: nix-${{ github.event.pull_request.number || github.ref }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
jobs:
|
||||
# A `paths:` filter cannot gate this workflow correctly. The flake packages
|
||||
# the product, and nine of the checks then run the built binary, so a change
|
||||
# to hermes_cli/ alone can fail `nix flake check` without touching one file
|
||||
# under nix/. The `nix` lane therefore follows python_prod as well as the
|
||||
# flake inputs. On push the classifier fails open and every lane is true.
|
||||
detect:
|
||||
name: Detect affected areas
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
outputs:
|
||||
nix: ${{ steps.classify.outputs.nix }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Detect affected areas
|
||||
id: classify
|
||||
uses: ./.github/actions/detect-changes
|
||||
with:
|
||||
github-token: ${{ github.token }}
|
||||
|
||||
flake-check:
|
||||
name: nix flake check
|
||||
needs: [detect]
|
||||
if: needs.detect.outputs.nix == 'true'
|
||||
# The build compiles the package and its whole dependency closure, so this
|
||||
# takes minutes and not seconds when the cache misses. `nix flake check`
|
||||
# builds 21 checks, and --max-jobs defaults to the core count. It uses the
|
||||
# wider runner with no more configuration.
|
||||
runs-on: ubuntu-latest-32-core
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Install Nix
|
||||
uses: cachix/install-nix-action@630ae543ea3a38a9a4166f03376c02c50f408342 # v31.11.0
|
||||
with:
|
||||
extra_nix_config: |
|
||||
experimental-features = nix-command flakes
|
||||
# A store path that does not substitute is a cache miss and not a
|
||||
# build failure. Build it here instead.
|
||||
fallback = true
|
||||
# Each source archive is fetched one time in a run, and not one
|
||||
# time for each evaluation.
|
||||
tarball-ttl = 3600
|
||||
|
||||
# Restores /nix/store from the GitHub Actions cache. The store holds the
|
||||
# whole dependency closure, so a hit turns a build of several minutes
|
||||
# into a short evaluation.
|
||||
#
|
||||
# The Magic Nix Cache is not an option here. Its free tier ended in
|
||||
# February 2025 with the GitHub cache API that it was built on. This
|
||||
# action uses the current API and needs no account and no secret.
|
||||
- name: Restore and save the Nix store
|
||||
uses: nix-community/cache-nix-action@7df957e333c1e5da7721f60227dbba6d06080569 # v7
|
||||
with:
|
||||
# The closure changes when the flake inputs change or when the
|
||||
# dependencies of the project change. The key hashes both, so an
|
||||
# edit to the source alone keeps the hit.
|
||||
primary-key: nix-${{ runner.os }}-${{ hashFiles('flake.lock', 'nix/**', 'pyproject.toml', 'uv.lock') }}
|
||||
# On a miss, restore the newest store for this runner. Most of the
|
||||
# closure — Python, node, each transitive library — survives a bump
|
||||
# of the lockfile, so an old store still removes most of the work.
|
||||
restore-prefixes-first-match: nix-${{ runner.os }}-
|
||||
|
||||
# Save from main only. A cache that a PR writes is visible to that
|
||||
# PR alone and never to another branch, so a save there spends the
|
||||
# 10 GB quota of the repository and helps no later run. A PR still
|
||||
# restores: it reads the cache that the merge to main wrote. This is
|
||||
# the same rule that docker.yml applies to `cache-to`.
|
||||
save: ${{ github.event_name != 'pull_request' }}
|
||||
|
||||
# Collect garbage before the save, so the store stays inside the
|
||||
# 10 GB quota of the repository. Without a limit the store grows at
|
||||
# each merge until GitHub removes the entry, and the next PR then
|
||||
# gets nothing. This number is the size of the store and not the
|
||||
# size of the compressed archive.
|
||||
gc-max-store-size-linux: 5G
|
||||
|
||||
# Delete the caches that this key replaces. GitHub removes caches by
|
||||
# least recent use across the whole repository, so a Nix store that
|
||||
# is never purged pushes out the caches of the other workflows.
|
||||
purge: true
|
||||
purge-prefixes: nix-${{ runner.os }}-
|
||||
purge-created: 0
|
||||
purge-primary-key: never
|
||||
|
||||
- name: nix flake check
|
||||
# --print-build-logs: a check that fails then prints the assertion
|
||||
# that failed, and not only the derivation that failed to build.
|
||||
run: nix flake check --print-build-logs
|
||||
@@ -43,11 +43,16 @@ jobs:
|
||||
uses: google/osv-scanner-action/.github/workflows/osv-scanner-reusable.yml@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8
|
||||
with:
|
||||
# Scan explicit lockfiles rather than recursing, so we only look at
|
||||
# the three sources of truth and skip vendored / test / worktree dirs.
|
||||
# the five sources of truth and skip vendored / test / worktree dirs.
|
||||
scan-args: |-
|
||||
--lockfile=uv.lock
|
||||
--lockfile=package-lock.json
|
||||
--lockfile=website/package-lock.json
|
||||
--lockfile=plugins/platforms/photon/sidecar/package-lock.json
|
||||
--lockfile=scripts/whatsapp-bridge/package-lock.json
|
||||
# The upstream reusable workflow uploads this exact file under its
|
||||
# fixed artifact name, which the wrapper downloads below.
|
||||
results-file-name: osv-results.sarif
|
||||
fail-on-vuln: false
|
||||
|
||||
emit-status:
|
||||
@@ -64,7 +69,7 @@ jobs:
|
||||
- name: Download SARIF result
|
||||
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
|
||||
with:
|
||||
name: osv-results
|
||||
name: OSV Scanner SARIF file
|
||||
path: /tmp/osv-results
|
||||
continue-on-error: true
|
||||
|
||||
@@ -122,3 +127,15 @@ jobs:
|
||||
fi
|
||||
|
||||
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
|
||||
echo "review_status=${STATUS}" > review-status.json
|
||||
|
||||
- name: Upload review status artifact
|
||||
if: always() && steps.emit.outcome != 'skipped'
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: review-status-osv-scanner
|
||||
path: review-status.json
|
||||
retention-days: 1
|
||||
overwrite: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
name: Publish E2E evidence
|
||||
|
||||
# This runs only from the default branch after CI completes. It intentionally
|
||||
# checks out main, never the PR ref, and treats the downloaded artifact as
|
||||
# untrusted input before uploading validated GitHub attachments.
|
||||
on:
|
||||
workflow_run:
|
||||
workflows: [CI]
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
pull-requests: write
|
||||
|
||||
concurrency:
|
||||
group: publish-e2e-evidence-${{ github.event.workflow_run.id }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
name: Publish inline E2E evidence
|
||||
if: github.event.workflow_run.event == 'pull_request'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
environment: gh-image
|
||||
steps:
|
||||
- name: Check out trusted publisher
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
ref: ${{ github.event.repository.default_branch }}
|
||||
persist-credentials: false
|
||||
|
||||
# v1.2.0 resolves to 44f4b93ecbbe22de6c45fa2f62f519aee564ca8c.
|
||||
- name: Install gh-image
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: gh extension install drogers0/gh-image --pin v1.2.0
|
||||
|
||||
- name: Download and attach evidence
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
GH_SESSION_TOKEN: ${{ secrets.GH_IMAGE_SESSION_TOKEN }}
|
||||
SOURCE_REPO: ${{ github.repository }}
|
||||
SOURCE_RUN_ID: ${{ github.event.workflow_run.id }}
|
||||
HEAD_OWNER: ${{ github.event.workflow_run.head_repository.owner.login }}
|
||||
HEAD_BRANCH: ${{ github.event.workflow_run.head_branch }}
|
||||
HEAD_SHA: ${{ github.event.workflow_run.head_sha }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
# The run's own ``pull_requests`` payload is always empty for a
|
||||
# fork PR, so resolve the PR from its head reference instead.
|
||||
# The head-SHA match skips runs that a newer push superseded.
|
||||
PR_NUMBER=$(gh api -X GET "repos/$SOURCE_REPO/pulls" \
|
||||
-f head="$HEAD_OWNER:$HEAD_BRANCH" -f state=open \
|
||||
--jq '.[] | select(.head.sha == $ENV.HEAD_SHA) | .number' \
|
||||
| head -n1)
|
||||
if [ -z "$PR_NUMBER" ]; then
|
||||
echo "No open pull request has head $HEAD_OWNER:$HEAD_BRANCH at $HEAD_SHA (CI run $SOURCE_RUN_ID)."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
ARTIFACT_NAME=$(gh api "repos/$SOURCE_REPO/actions/runs/$SOURCE_RUN_ID/artifacts" \
|
||||
--jq '.artifacts[] | select(.expired == false and (.name | startswith("e2e-evidence-"))) | .name' \
|
||||
| python3 -c 'import sys; print(next(iter(sys.stdin), "").strip())')
|
||||
if [ -z "$ARTIFACT_NAME" ]; then
|
||||
echo "No E2E evidence artifact was produced for CI run $SOURCE_RUN_ID."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
EVIDENCE_DIR="$RUNNER_TEMP/e2e-evidence"
|
||||
mkdir -p "$EVIDENCE_DIR"
|
||||
gh run download "$SOURCE_RUN_ID" --repo "$SOURCE_REPO" --name "$ARTIFACT_NAME" --dir "$EVIDENCE_DIR"
|
||||
|
||||
python3 scripts/ci/publish_e2e_evidence.py \
|
||||
--evidence-dir "$EVIDENCE_DIR" \
|
||||
--source-repo "$SOURCE_REPO" \
|
||||
--pr-number "$PR_NUMBER"
|
||||
@@ -23,6 +23,10 @@ on:
|
||||
description: Whether CI-sensitive files (eslint config, workflows, actions) changed.
|
||||
type: boolean
|
||||
default: false
|
||||
ci_review_files:
|
||||
description: JSON list of CI-sensitive files changed by the pull request.
|
||||
type: string
|
||||
default: '[]'
|
||||
mcp_catalog:
|
||||
description: Whether the MCP catalog / installer changed.
|
||||
type: boolean
|
||||
@@ -78,18 +82,39 @@ jobs:
|
||||
id: build-status
|
||||
env:
|
||||
CI_REVIEW: ${{ inputs.ci_review }}
|
||||
CI_REVIEW_FILES: ${{ inputs.ci_review_files }}
|
||||
MCP_CATALOG: ${{ inputs.mcp_catalog }}
|
||||
SUPPLY_CHAIN: ${{ inputs.supply_chain }}
|
||||
LABEL_PRESENT: ${{ steps.label-check.outputs.ci_reviewed }}
|
||||
REPO_URL: ${{ github.server_url }}/${{ github.repository }}
|
||||
BASE_SHA: ${{ github.event.pull_request.base.sha }}
|
||||
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
args=()
|
||||
if [ "$CI_REVIEW" = "true" ]; then args+=(--ci-review); fi
|
||||
args+=(--ci-review-files "$CI_REVIEW_FILES")
|
||||
if [ "$MCP_CATALOG" = "true" ]; then args+=(--mcp-catalog); fi
|
||||
if [ "$SUPPLY_CHAIN" = "true" ]; then args+=(--supply-chain); fi
|
||||
if [ "$LABEL_PRESENT" = "true" ]; then args+=(--label-present); fi
|
||||
|
||||
python3 scripts/ci/emit_review_status.py "${args[@]}" --output "$GITHUB_OUTPUT"
|
||||
# Write to both $GITHUB_OUTPUT and review-status.json for the
|
||||
# live comment poller to pick up as an artifact.
|
||||
python3 scripts/ci/emit_review_status.py "${args[@]}" \
|
||||
--repo-url "$REPO_URL" --base-sha "$BASE_SHA" --head-sha "$HEAD_SHA" \
|
||||
--output "$GITHUB_OUTPUT"
|
||||
grep '^review_status=' "$GITHUB_OUTPUT" > review-status.json
|
||||
|
||||
- name: Upload review status artifact
|
||||
if: always() && steps.build-status.outcome != 'skipped'
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: review-status-review-labels
|
||||
path: review-status.json
|
||||
retention-days: 1
|
||||
overwrite: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
- name: Fail on missing label
|
||||
if: steps.label-check.outputs.ci_reviewed != 'true'
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
# .github/workflows/rust-tests.yml
|
||||
name: Rust tests
|
||||
|
||||
# `cargo test` for the Tauri bootstrap installer (Hermes-Setup). Nothing in CI
|
||||
# compiled this crate before: `.rs` lives under `apps/`, so the change
|
||||
# classifier matched it as `frontend` and ran the TypeScript matrix, which
|
||||
# cannot notice a Rust error. The crate's unit tests existed in the tree and had
|
||||
# never run.
|
||||
#
|
||||
# Linux runner on purpose. The pipe-drain tests in src/powershell.rs need a real
|
||||
# process tree whose grandchild inherits the parent's stdout, and their fixture
|
||||
# is `#[cfg(unix)]`; the Windows half of that same contract is covered by
|
||||
# `-SelfTestPipeDrain` in scripts/desktop-update/windows.ps1 on the Windows
|
||||
# lane. A Windows runner here would compile them out and report green over zero
|
||||
# coverage.
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: rust-tests-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
bootstrap-installer:
|
||||
name: cargo test (bootstrap installer)
|
||||
# cargo builds codegen units and test binaries in parallel across the
|
||||
# cores. This lane also builds the crate from the start when Cargo.toml
|
||||
# changes.
|
||||
runs-on: ubuntu-latest-32-core
|
||||
timeout-minutes: 30
|
||||
defaults:
|
||||
run:
|
||||
working-directory: apps/bootstrap-installer/src-tauri
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
# Tauri links against the system webkit2gtk on Linux, so the crate does
|
||||
# not compile without these even for `cargo test --lib`.
|
||||
- name: Install Tauri system dependencies
|
||||
working-directory: .
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install --no-install-recommends -y \
|
||||
libwebkit2gtk-4.1-dev \
|
||||
libappindicator3-dev \
|
||||
librsvg2-dev \
|
||||
libxdo-dev \
|
||||
libssl-dev \
|
||||
patchelf
|
||||
|
||||
# Keyed on Cargo.toml, not Cargo.lock: apps/bootstrap-installer/.gitignore
|
||||
# excludes the lockfile (a create-tauri-app scaffold default), so there is
|
||||
# nothing pinned to hash and `--locked` cannot be used. That also means
|
||||
# this crate re-resolves its whole dependency graph on every build, which
|
||||
# is a real gap for a signed installer given the pinning policy in
|
||||
# AGENTS.md — tracking separately rather than widening this PR.
|
||||
#
|
||||
# No restore-keys: a partial hit leaves a stale target dir, and cargo
|
||||
# re-resolves correctly on top of a Cargo.toml-keyed hit anyway.
|
||||
- name: Restore cargo cache
|
||||
uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
apps/bootstrap-installer/src-tauri/target
|
||||
key: cargo-${{ runner.os }}-${{ hashFiles('apps/bootstrap-installer/src-tauri/Cargo.toml') }}
|
||||
|
||||
# --lib only: the integration/bin targets would need a built frontend
|
||||
# (vite dist) that this lane deliberately does not produce.
|
||||
- name: cargo test
|
||||
run: cargo test --lib
|
||||
@@ -21,7 +21,16 @@ jobs:
|
||||
if: github.repository == 'NousResearch/hermes-agent'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
environment: trusted-automation
|
||||
steps:
|
||||
# `Get GitHub App token` below is a LOCAL composite action
|
||||
# (./.github/actions/get-app-token), so the repository must be on disk
|
||||
# before it can be resolved. Without this checkout the job dies with
|
||||
# "Can't find 'action.yml' ... under .github/actions/get-app-token"
|
||||
# every time the probe is non-ok — i.e. exactly when the watchdog is
|
||||
# supposed to file its issue.
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Probe live index
|
||||
id: probe
|
||||
run: |
|
||||
@@ -113,7 +122,7 @@ jobs:
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ secrets.APP_CLIENT_ID }}
|
||||
client-id: ${{ vars.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- name: Open issue on degraded / failed probe
|
||||
|
||||
@@ -21,6 +21,7 @@ jobs:
|
||||
if: github.repository == 'NousResearch/hermes-agent'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
environment: trusted-automation
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
@@ -28,7 +29,7 @@ jobs:
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ secrets.APP_CLIENT_ID }}
|
||||
client-id: ${{ vars.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
@@ -60,13 +61,21 @@ jobs:
|
||||
if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
environment: trusted-automation
|
||||
steps:
|
||||
# Required: `Get GitHub App token` is a LOCAL composite action
|
||||
# (./.github/actions/get-app-token) and cannot resolve without the repo
|
||||
# checked out. `build-index` above already does this; this job did not,
|
||||
# so the scheduled deploy re-trigger never fired.
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Get GitHub App token
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ secrets.APP_CLIENT_ID }}
|
||||
client-id: ${{ vars.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- name: Trigger Deploy Site workflow
|
||||
env:
|
||||
GH_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
|
||||
@@ -65,17 +65,11 @@ jobs:
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Get GitHub App token
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ secrets.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- name: Scan diff for critical patterns
|
||||
id: scan
|
||||
env:
|
||||
GH_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
CI_REVIEWED: ${{ contains(github.event.pull_request.labels.*.name, 'ci-reviewed') }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
@@ -93,7 +87,7 @@ jobs:
|
||||
# --- .pth files (auto-execute on Python startup) ---
|
||||
# The exact mechanism used in the litellm supply chain attack:
|
||||
# https://github.com/BerriAI/litellm/issues/24512
|
||||
PTH_FILES=$(git diff --name-only "$BASE"..."$HEAD" | grep '\.pth$' || true)
|
||||
PTH_FILES=$(git diff --diff-filter=d --name-only "$BASE"..."$HEAD" | grep '\.pth$' || true)
|
||||
if [ -n "$PTH_FILES" ]; then
|
||||
FINDINGS="${FINDINGS}
|
||||
### 🚨 CRITICAL: .pth file added or modified
|
||||
@@ -141,8 +135,11 @@ jobs:
|
||||
# auto-loaded by the interpreter via site.py. Any nested file with the
|
||||
# same name (e.g. hermes_cli/setup.py — the CLI setup wizard) is unrelated
|
||||
# and produced false positives that trained reviewers to ignore the scanner.
|
||||
SETUP_HITS=$(git diff --name-only "$BASE"..."$HEAD" | grep -E '^(setup\.py|setup\.cfg|sitecustomize\.py|usercustomize\.py|__init__\.pth)$' || true)
|
||||
if [ -n "$SETUP_HITS" ]; then
|
||||
SETUP_HITS=$(git diff --diff-filter=d --name-only "$BASE"..."$HEAD" | grep -E '^(setup\.py|setup\.cfg|sitecustomize\.py|usercustomize\.py|__init__\.pth)$' || true)
|
||||
# A maintainer-applied ci-reviewed label records the manual review
|
||||
# required for intentional changes to an install hook. The scanner
|
||||
# still blocks every unreviewed addition or modification.
|
||||
if [ -n "$SETUP_HITS" ] && [ "$CI_REVIEWED" != "true" ]; then
|
||||
FINDINGS="${FINDINGS}
|
||||
### 🚨 CRITICAL: Install-hook file added or modified
|
||||
These files can execute code during package installation or interpreter startup.
|
||||
@@ -293,4 +290,19 @@ jobs:
|
||||
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as f:
|
||||
f.write(f"review_status={json.dumps(merged)}\n")
|
||||
f.write("critical_findings=" + os.environ.get("CRITICAL_FINDINGS", "false") + "\n")
|
||||
|
||||
# Write review-status.json for the live comment poller artifact.
|
||||
with open("review-status.json", "w", encoding="utf-8") as f:
|
||||
f.write(f"review_status={json.dumps(merged)}\n")
|
||||
PYEOF
|
||||
|
||||
- name: Upload review status artifact
|
||||
if: always() && steps.merge.outcome != 'skipped'
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: review-status-supply-chain
|
||||
path: review-status.json
|
||||
retention-days: 1
|
||||
overwrite: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
name: OS-specific tests
|
||||
|
||||
# Runs the tests that can only be trusted on their own host OS.
|
||||
#
|
||||
# The main Python suite (.github/workflows/tests.yml) runs on
|
||||
# ubuntu-latest and covers everything that is either platform-agnostic or
|
||||
# genuinely Linux-specific. Tests whose subject is macOS- or
|
||||
# Windows-specific behaviour carry a marker (see the ``_OS_MARKS`` block
|
||||
# comment in tests/conftest.py) and are SKIPPED on Linux, because faking
|
||||
# ``sys.platform`` on a Linux runner selects the branch under test without
|
||||
# reproducing any of the OS behaviour that branch exists for. This workflow
|
||||
# is where those markers actually execute:
|
||||
#
|
||||
# macos → ``-m macos_only`` on macos-latest
|
||||
# windows → ``-m windows_only`` on windows-latest
|
||||
#
|
||||
# Deliberately NOT sliced. The marked set is small (tens of tests, not
|
||||
# thousands), so one plain ``pytest`` process per OS is both faster and far
|
||||
# less machinery than the per-file parallel runner the Linux lane uses.
|
||||
# If either lane grows past its timeout, that is the signal to reach for
|
||||
# scripts/run_tests.sh here too.
|
||||
#
|
||||
# Each lane FAILS when it selects zero tests (pytest exit code 5). Without
|
||||
# that guard, a renamed marker or a bad selector would report a green job
|
||||
# that ran nothing — the exact silent-coverage-loss failure this workflow
|
||||
# exists to prevent.
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: tests-os-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
os-tests:
|
||||
name: ${{ matrix.name }}
|
||||
runs-on: ${{ matrix.runner }}
|
||||
timeout-minutes: 30
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- name: macOS-only tests
|
||||
runner: macos-latest
|
||||
marker: macos_only
|
||||
- name: Windows-only tests
|
||||
runner: windows-latest-32-core
|
||||
marker: windows_only
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pinned for the same reason as the Linux lane: unpinned, setup-uv
|
||||
# resolves "latest" by fetching a manifest on every job and a
|
||||
# transient fetch failure fails the whole job.
|
||||
version: "0.9.28"
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
pyproject.toml
|
||||
uv.lock
|
||||
|
||||
- name: Set up Python 3.11
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv python install 3.11
|
||||
|
||||
- name: Install dependencies
|
||||
# Same extras as the Linux test lane so an OS-marked test can import
|
||||
# anything its Linux siblings can. ``[all]`` is deliberately
|
||||
# Windows/macOS-installable (see the policy comment on the extra in
|
||||
# pyproject.toml — matrix/python-olm was removed from it precisely
|
||||
# because it could not build here).
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
|
||||
|
||||
- name: Minimize uv cache
|
||||
run: uv cache prune --ci
|
||||
|
||||
- name: Run ${{ matrix.marker }} tests
|
||||
# Two-step selection:
|
||||
#
|
||||
# 1. scripts/ci/list_os_marked_tests.py narrows WHICH FILES are
|
||||
# imported. ``-m`` filters after collection, and collection
|
||||
# imports every module under tests/ — on this host that would
|
||||
# drag ~900 unrelated test modules through import, where a
|
||||
# single unrelated ImportError would fail a job whose own
|
||||
# subject is fine. The helper exits non-zero if the marker
|
||||
# matches no file at all.
|
||||
# 2. ``-m`` decides WHICH TESTS run, and stays authoritative.
|
||||
# Passing it on the command line REPLACES pyproject's
|
||||
# ``-m 'not integration'`` addopts (same option, last wins) —
|
||||
# hence repeating ``not integration``, or the integration
|
||||
# suite would return through the side door.
|
||||
#
|
||||
# ``--timeout-method`` needs no override: tests/conftest.py's
|
||||
# pytest_configure already downgrades the signal-based timer on
|
||||
# Windows, which has no SIGALRM.
|
||||
shell: bash
|
||||
run: |
|
||||
set -uo pipefail
|
||||
|
||||
LIST="${RUNNER_TEMP:-.}/selected-tests.txt"
|
||||
|
||||
# Process substitution would hide the helper's exit status, so write
|
||||
# to a file and check it explicitly.
|
||||
if ! uv run --no-sync python scripts/ci/list_os_marked_tests.py \
|
||||
"${{ matrix.marker }}" > "$LIST"; then
|
||||
echo "::error::could not enumerate ${{ matrix.marker }} test files"
|
||||
exit 1
|
||||
fi
|
||||
if [ ! -s "$LIST" ]; then
|
||||
echo "::error::empty ${{ matrix.marker }} file list"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Deliberately NOT `mapfile`: that is a bash 4 builtin and the macOS
|
||||
# runner's /bin/bash is 3.2. Word-splitting is safe here because the
|
||||
# helper emits repo-relative test paths, which contain no spaces.
|
||||
# shellcheck disable=SC2046
|
||||
set -- $(cat "$LIST")
|
||||
echo "selected $# file(s) for ${{ matrix.marker }}:"
|
||||
cat "$LIST"
|
||||
|
||||
# ``shell: bash`` runs this script with ``-e`` injected, which
|
||||
# ``set -uo pipefail`` above does not clear. A bare pytest call
|
||||
# would therefore abort the script on any non-zero exit and the
|
||||
# exit-5 branch below would be unreachable dead code — the job
|
||||
# would still fail red, but the diagnostic would never print.
|
||||
status=0
|
||||
uv run --no-sync python -m pytest \
|
||||
"$@" \
|
||||
-m "${{ matrix.marker }} and not integration" \
|
||||
-v --tb=short || status=$?
|
||||
if [ "$status" -eq 5 ]; then
|
||||
echo "::error::No tests matched -m ${{ matrix.marker }}. Either the" \
|
||||
"marker was renamed/dropped or selection is broken — this job" \
|
||||
"must never pass without running its OS's tests."
|
||||
exit 1
|
||||
fi
|
||||
exit "$status"
|
||||
env:
|
||||
# Belt-and-suspenders with tests/conftest.py's env blanking: no
|
||||
# test may reach a real provider API.
|
||||
OPENROUTER_API_KEY: ""
|
||||
OPENAI_API_KEY: ""
|
||||
NOUS_API_KEY: ""
|
||||
@@ -2,11 +2,6 @@ name: Tests
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
slice_count:
|
||||
description: Number of parallel test slices
|
||||
type: number
|
||||
default: 8
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -17,42 +12,17 @@ concurrency:
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
generate:
|
||||
name: "Generate slices"
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
outputs:
|
||||
matrix: ${{ steps.matrix.outputs.matrix }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Restore duration cache
|
||||
uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
with:
|
||||
path: test_durations.json
|
||||
key: test-durations
|
||||
# Saves use test-durations-${run_id}, so the exact key above never
|
||||
# matches — without this prefix fallback the cache ALWAYS missed,
|
||||
# LPT slicing ran on no data, and unbalanced slices pushed heavy
|
||||
# files toward the per-file timeout under load.
|
||||
restore-keys: |
|
||||
test-durations-
|
||||
|
||||
- name: Generate test slices
|
||||
id: matrix
|
||||
run: |
|
||||
MATRIX=$(python3 scripts/run_tests_parallel.py --generate-slices ${{ inputs.slice_count }})
|
||||
echo "matrix=$MATRIX" >> "$GITHUB_OUTPUT"
|
||||
|
||||
test:
|
||||
name: Run tests slice ${{ matrix.slice.index }}/${{ inputs.slice_count }}
|
||||
needs: generate
|
||||
runs-on: ubuntu-latest
|
||||
name: Run tests
|
||||
# One 96-core runner for the whole suite. There is no slicing. Slicing
|
||||
# existed to spread the suite over 4-core runners. It cost a matrix job, a
|
||||
# duration cache, a per-slice artifact and a merge job to do it.
|
||||
#
|
||||
# 96 cores clear the floor that the slowest single test file sets (about
|
||||
# 82s). A second slice divides work that is already at that floor, and
|
||||
# adds a second setup.
|
||||
runs-on: ubuntu-latest-96-core
|
||||
timeout-minutes: 30
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJSON(needs.generate.outputs.matrix) }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
@@ -74,6 +44,11 @@ jobs:
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pin the uv version: unpinned, setup-uv resolves "latest" by
|
||||
# fetching a manifest from raw.githubusercontent.com on EVERY job —
|
||||
# a transient fetch failure fails the whole job (2026-07-28 slice-5
|
||||
# incident). Pinned, the binary downloads directly; no manifest hop.
|
||||
version: "0.9.28"
|
||||
# Persist uv's download/wheel cache (~/.cache/uv) across runs.
|
||||
# Keyed on the dependency manifests, so the cache is reused until
|
||||
# pyproject.toml or uv.lock changes. `uv sync` still runs every
|
||||
@@ -85,85 +60,71 @@ jobs:
|
||||
uv.lock
|
||||
|
||||
- name: Set up Python 3.11
|
||||
run: uv python install 3.11
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv python install 3.11
|
||||
|
||||
- name: Install dependencies
|
||||
# `uv sync --locked` installs the exact pinned set from uv.lock (and
|
||||
# fails if the lock is out of sync with pyproject.toml), giving a
|
||||
# reproducible env. It also creates .venv itself, so no separate
|
||||
# `uv venv` step is needed.
|
||||
#
|
||||
# The trailing extras beyond all/dev are the lazy-install features
|
||||
# (tools/lazy_deps.py) that tests exercise for real: provider.anthropic,
|
||||
# stt/tts.mistral, image.fal, terminal.modal, terminal.daytona,
|
||||
# memory.hindsight, search.parallel. The hermetic test env forbids
|
||||
# mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
|
||||
# tests/conftest.py), so the SDKs those tests need must be in the
|
||||
# venv up front — resolved from uv.lock like everything else, which
|
||||
# also honors the exact supply-chain pins these extras carry.
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
|
||||
|
||||
- name: Minimize uv cache
|
||||
# Optimized for CI: prunes pre-built wheels that are cheap to
|
||||
# re-download, keeping the persisted cache small and fast to restore.
|
||||
run: uv cache prune --ci
|
||||
|
||||
- name: Run tests (slice ${{ matrix.slice.index }}/${{ inputs.slice_count }})
|
||||
- name: Run tests
|
||||
# Per-file isolation via scripts/run_tests.sh: each test file runs
|
||||
# in its own freshly-spawned `python -m pytest <file>` subprocess
|
||||
# with bounded parallelism. No xdist, no shared workers, no
|
||||
# module-level state leakage between files.
|
||||
#
|
||||
# File list is pre-computed by the generate job (--generate-slices)
|
||||
# which runs LPT distribution once and passes the file list to each
|
||||
# matrix job via --files. Previously each job re-discovered files and
|
||||
# re-ran LPT independently — redundant N times.
|
||||
# No --files: the runner discovers the suite itself. The discovered
|
||||
# set is identical to the list the removed matrix job used to pass in.
|
||||
run: |
|
||||
source .venv/bin/activate
|
||||
scripts/run_tests.sh --files '${{ matrix.slice.files }}'
|
||||
scripts/run_tests.sh
|
||||
env:
|
||||
# This is the maximum number of test FILES that run together.
|
||||
# run_tests_parallel.py starts one pytest subprocess for each file
|
||||
# from a single ThreadPoolExecutor, so this value IS the limit. The
|
||||
# default is cpu_count*2, which is 192 here.
|
||||
#
|
||||
# Measured on this runner (96-core EPYC 7763, 377GB). Whole suite,
|
||||
# two repetitions for each value. See run 32549672063:
|
||||
#
|
||||
# workers x cores mean
|
||||
# 48 0.5x 138s
|
||||
# 96 1.0x 126s <- fastest
|
||||
# 144 1.5x 132s
|
||||
# 192 2.0x 132s
|
||||
# 240 2.5x 140s
|
||||
# 288 3.0x 142s
|
||||
#
|
||||
# One worker for each core wins. The curve is shallow: 126s to 142s
|
||||
# across a 6x range. The suite has sufficient concurrency at this
|
||||
# size. The remaining time is the slowest files plus the setup.
|
||||
# Workers above the core count only add contention.
|
||||
HERMES_TEST_WORKERS: 96
|
||||
# Ensure tests don't accidentally call real APIs
|
||||
OPENROUTER_API_KEY: ""
|
||||
OPENAI_API_KEY: ""
|
||||
NOUS_API_KEY: ""
|
||||
|
||||
- name: Upload per-slice durations
|
||||
# Advisory artifact (feeds slice balancing) — a transient artifact-
|
||||
# service blip must not fail an otherwise-green test slice.
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: test-durations-slice-${{ matrix.slice.index }}
|
||||
path: test_durations.json
|
||||
retention-days: 1
|
||||
|
||||
# Merge per-slice duration data into a single cache, so future runs
|
||||
# (including PRs) get balanced slicing.
|
||||
save-durations:
|
||||
needs: test
|
||||
if: needs.test.result == 'success' && github.ref == 'refs/heads/main'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Download all slice durations
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
pattern: test-durations-slice-*
|
||||
path: durations
|
||||
merge-multiple: true
|
||||
|
||||
- name: Merge into single durations file
|
||||
run: |
|
||||
python3 -c "
|
||||
import json, glob, os
|
||||
merged = {}
|
||||
for f in glob.glob('durations/*test_durations.json'):
|
||||
with open(f) as fh:
|
||||
merged.update(json.load(fh))
|
||||
with open('test_durations.json', 'w') as fh:
|
||||
json.dump(merged, fh, indent=2, sort_keys=True)
|
||||
print(f'Merged {len(merged)} file durations')
|
||||
"
|
||||
|
||||
- name: Save merged duration cache
|
||||
uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
||||
with:
|
||||
path: test_durations.json
|
||||
key: test-durations-${{ github.run_id }}
|
||||
|
||||
e2e:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
@@ -188,6 +149,11 @@ jobs:
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pin the uv version: unpinned, setup-uv resolves "latest" by
|
||||
# fetching a manifest from raw.githubusercontent.com on EVERY job —
|
||||
# a transient fetch failure fails the whole job (2026-07-28 slice-5
|
||||
# incident). Pinned, the binary downloads directly; no manifest hop.
|
||||
version: "0.9.28"
|
||||
# Persist uv's download/wheel cache (~/.cache/uv) across runs.
|
||||
# Keyed on the dependency manifests, so the cache is reused until
|
||||
# pyproject.toml or uv.lock changes. `uv sync` still runs every
|
||||
@@ -206,20 +172,20 @@ jobs:
|
||||
# fails if the lock is out of sync with pyproject.toml), giving a
|
||||
# reproducible env. It also creates .venv itself, so no separate
|
||||
# `uv venv` step is needed.
|
||||
#
|
||||
# Same extras as the test job's sync above: the hermetic test env
|
||||
# forbids mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
|
||||
# tests/conftest.py), so lazy-install SDKs exercised by tests must be
|
||||
# in the venv up front.
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
|
||||
|
||||
- name: Minimize uv cache
|
||||
# Optimized for CI: prunes pre-built wheels that are cheap to
|
||||
# re-download, keeping the persisted cache small and fast to restore.
|
||||
run: uv cache prune --ci
|
||||
|
||||
- name: Packaged-wheel i18n smoke test
|
||||
run: |
|
||||
source .venv/bin/activate
|
||||
python -m pytest -m integration tests/test_wheel_locales_e2e.py -v
|
||||
|
||||
- name: Run e2e tests
|
||||
run: |
|
||||
source .venv/bin/activate
|
||||
|
||||
@@ -1,188 +0,0 @@
|
||||
name: Publish to PyPI
|
||||
|
||||
# Triggered by CalVer tag pushes from scripts/release.py (e.g. v2026.5.15)
|
||||
# Can also be triggered manually from the Actions tab as an escape hatch.
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- "v20*" # CalVer tags: v2026.5.15, v2026.5.15.2, etc.
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
confirm_tag:
|
||||
description: "Tag to publish (e.g. v2026.5.15). Must already exist."
|
||||
required: true
|
||||
type: string
|
||||
|
||||
# Restrict default token to read-only; each job escalates as needed.
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# Prevent overlapping publishes (e.g. two same-day tags pushed quickly).
|
||||
concurrency:
|
||||
group: pypi-publish
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: Build distribution 📦
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
# On workflow_dispatch, check out the confirmed tag.
|
||||
ref: ${{ inputs.confirm_tag || github.ref }}
|
||||
fetch-tags: true
|
||||
|
||||
- name: Validate tag exists
|
||||
if: github.event_name == 'workflow_dispatch'
|
||||
run: |
|
||||
if ! git tag -l "${{ inputs.confirm_tag }}" | grep -q .; then
|
||||
echo "::error::Tag '${{ inputs.confirm_tag }}' does not exist in the repo"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
with:
|
||||
python-version: "3.13"
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: "22"
|
||||
|
||||
- name: Build web dashboard
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: npm ci
|
||||
working-directory: web
|
||||
|
||||
- name: Compile web dashboard
|
||||
run: npm run build
|
||||
working-directory: web
|
||||
|
||||
- name: Build TUI bundle
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: npm ci
|
||||
working-directory: ui-tui
|
||||
|
||||
- name: Compile TUI bundle
|
||||
run: npm run build
|
||||
working-directory: ui-tui
|
||||
|
||||
- name: Bundle TUI into hermes_cli
|
||||
run: |
|
||||
mkdir -p hermes_cli/tui_dist
|
||||
cp ui-tui/dist/entry.js hermes_cli/tui_dist/entry.js
|
||||
|
||||
- name: Verify frontend assets exist
|
||||
run: |
|
||||
test -f hermes_cli/web_dist/index.html || { echo "ERROR: web_dist not built"; exit 1; }
|
||||
test -f hermes_cli/tui_dist/entry.js || { echo "ERROR: tui_dist not built"; exit 1; }
|
||||
|
||||
- name: Bundle install scripts into wheel
|
||||
run: |
|
||||
mkdir -p hermes_cli/scripts
|
||||
cp scripts/install.sh hermes_cli/scripts/install.sh
|
||||
cp scripts/install.ps1 hermes_cli/scripts/install.ps1
|
||||
|
||||
- name: Build wheel and sdist
|
||||
run: uv build --sdist --wheel
|
||||
|
||||
- name: Upload distribution artifacts
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: python-package-distributions
|
||||
path: dist/
|
||||
|
||||
publish:
|
||||
name: Publish to PyPI
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
environment:
|
||||
name: pypi
|
||||
url: https://pypi.org/p/hermes-agent
|
||||
permissions:
|
||||
id-token: write # OIDC trusted publishing
|
||||
|
||||
steps:
|
||||
- name: Download distribution artifacts
|
||||
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
|
||||
with:
|
||||
name: python-package-distributions
|
||||
path: dist/
|
||||
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # v1.14.0
|
||||
with:
|
||||
skip-existing: true
|
||||
|
||||
sign:
|
||||
name: Sign and attach to GitHub Release
|
||||
# Only runs on tag pushes — release.py creates the GitHub Release,
|
||||
# and workflow_dispatch won't have a matching release to attach to.
|
||||
if: startsWith(github.ref, 'refs/tags/')
|
||||
needs: publish
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: write # attach assets to the existing release
|
||||
id-token: write # sigstore signing
|
||||
|
||||
steps:
|
||||
- name: Download distribution artifacts
|
||||
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
|
||||
with:
|
||||
name: python-package-distributions
|
||||
path: dist/
|
||||
|
||||
- name: Get GitHub App token
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ secrets.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- name: Wait for GitHub Release to exist
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
# release.py creates the GitHub Release after pushing the tag,
|
||||
# but this workflow starts from the tag push — wait for it.
|
||||
run: |
|
||||
for i in $(seq 1 30); do
|
||||
if gh release view "$GITHUB_REF_NAME" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then
|
||||
echo "Release $GITHUB_REF_NAME found"
|
||||
exit 0
|
||||
fi
|
||||
echo "Waiting for release... ($i/30)"
|
||||
sleep 10
|
||||
done
|
||||
echo "::warning::Release $GITHUB_REF_NAME not found after 5 minutes — skipping signature upload"
|
||||
echo "skip_sign=true" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Sign with Sigstore
|
||||
if: env.skip_sign != 'true'
|
||||
uses: sigstore/gh-action-sigstore-python@04cffa1d795717b140764e8b640de88853c92acc # v3.3.0
|
||||
with:
|
||||
inputs: >-
|
||||
./dist/*.tar.gz
|
||||
./dist/*.whl
|
||||
|
||||
- name: Attach signed artifacts to GitHub Release
|
||||
if: env.skip_sign != 'true'
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
# release.py already created the GitHub Release — just upload
|
||||
# the Sigstore signatures alongside the existing assets.
|
||||
run: >-
|
||||
gh release upload
|
||||
"$GITHUB_REF_NAME" dist/*.sigstore.json
|
||||
--repo "$GITHUB_REPOSITORY"
|
||||
--clobber
|
||||
@@ -70,6 +70,11 @@ jobs:
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
|
||||
# raw.githubusercontent.com every job; transient fetch failures
|
||||
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
|
||||
version: "0.9.28"
|
||||
|
||||
# `uv lock --check` re-resolves the project from pyproject.toml and
|
||||
# compares the result to uv.lock, exiting non-zero if they disagree.
|
||||
@@ -84,16 +89,46 @@ jobs:
|
||||
# uv lock --check re-resolves against PyPI (network). Retry so a
|
||||
# registry blip doesn't read as "lockfile stale". A genuinely stale
|
||||
# lockfile fails all attempts (deterministic), costing only seconds.
|
||||
#
|
||||
# Backoff rather than a flat 10s: three attempts inside ~20s all
|
||||
# land in the same blip. 5/15/45s spans ~65s instead.
|
||||
#
|
||||
# Network failure and a real desync are also reported differently.
|
||||
# uv says "Request failed after N retries" when it can't reach the
|
||||
# registry, versus "lockfile needs to be updated" when the lock is
|
||||
# genuinely stale. Only the second is the contributor's to fix, so
|
||||
# an unreachable registry says so instead of sending them to
|
||||
# `uv lock` with nothing to regenerate.
|
||||
ok=false
|
||||
net_fail=false
|
||||
delays=(5 15 45)
|
||||
for i in 1 2 3; do
|
||||
if uv lock --check; then
|
||||
if uv lock --check >lock-check.log 2>&1; then
|
||||
ok=true
|
||||
net_fail=false
|
||||
cat lock-check.log
|
||||
break
|
||||
fi
|
||||
cat lock-check.log
|
||||
if grep -qiE "Request failed after|Failed to fetch|error sending request|connection (reset|closed)|timed out" lock-check.log; then
|
||||
net_fail=true
|
||||
else
|
||||
# Deterministic failure (a real desync) — retrying just prints
|
||||
# the same error twice more.
|
||||
net_fail=false
|
||||
break
|
||||
fi
|
||||
[ "$i" = 3 ] && break
|
||||
echo "::warning::uv lock --check failed (attempt $i); retrying in 10s"
|
||||
sleep 10
|
||||
echo "::warning::uv lock --check could not reach the registry (attempt $i); retrying in ${delays[$((i-1))]}s"
|
||||
sleep "${delays[$((i-1))]}"
|
||||
done
|
||||
if [ "$ok" != true ] && [ "$net_fail" = true ]; then
|
||||
echo "::error title=uv.lock check could not reach PyPI::Registry unreachable after 3 attempts — infrastructure failure, not a stale lockfile. Re-run the job."
|
||||
review_status='[{"source":"uv.lock check","results":[{"kind":"action_required","title":"uv.lock check could not reach PyPI","summary":"`uv lock --check` could not reach the package registry after 3 attempts. This is an infrastructure failure, not a stale lockfile.","how_to_fix":"Re-run the failed job. No change to `uv.lock` is needed."}]}]'
|
||||
echo "review_status=${review_status}" >> "$GITHUB_OUTPUT"
|
||||
echo "review_status=${review_status}" > review-status.json
|
||||
exit 1
|
||||
fi
|
||||
if [ "$ok" != true ]; then
|
||||
cat <<'EOF' >> "$GITHUB_STEP_SUMMARY"
|
||||
## ❌ uv.lock is out of sync with pyproject.toml
|
||||
@@ -126,7 +161,20 @@ jobs:
|
||||
echo "::error title=uv.lock out of sync::Run \`uv lock\` locally and commit the result. If on a PR, sync with main first."
|
||||
review_status='[{"source":"uv.lock check","results":[{"kind":"action_required","title":"uv.lock out of sync","summary":"uv.lock is out of sync with pyproject.toml.","how_to_fix":"Run `uv lock` locally and commit the result. If on a PR, sync with main first:\n```\ngit fetch origin main\ngit rebase origin/main\nuv lock\ngit add uv.lock\ngit commit -m \"chore: refresh uv.lock\"\n```\n"}]}]'
|
||||
echo "review_status=${review_status}" >> "$GITHUB_OUTPUT"
|
||||
echo "review_status=${review_status}" > review-status.json
|
||||
exit 1
|
||||
fi
|
||||
review_status='[]'
|
||||
echo "review_status=${review_status}" >> "$GITHUB_OUTPUT"
|
||||
echo "review_status=${review_status}" > review-status.json
|
||||
|
||||
- name: Upload review status artifact
|
||||
if: always() && steps.verify.outcome != 'skipped'
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: review-status-uv-lockfile
|
||||
path: review-status.json
|
||||
retention-days: 1
|
||||
overwrite: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
name: Windows venv-holder live E2E
|
||||
|
||||
# ON-DEMAND ONLY (fleet-update #91277, venv-holder consolidation work).
|
||||
#
|
||||
# Runs the live venv-holder E2E suite on a real windows-latest runner:
|
||||
# spawns actual processes with realistic Hermes argv shapes and drives the
|
||||
# REAL detection/classification/exemption code against the live process
|
||||
# table — the coverage that cannot exist on the Linux lanes and that the
|
||||
# maintainer cannot exercise locally before the work reaches main.
|
||||
#
|
||||
# Deliberately NOT wired to pull_request/main: it fires only on pushes to
|
||||
# wine2e/** working branches, so it costs nothing on normal PRs. Delete or
|
||||
# keep dormant after the venv-holder work lands.
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- "wine2e/**"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: windows-venv-e2e-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
venv-holder-e2e:
|
||||
name: venv-holder live E2E (windows-latest)
|
||||
runs-on: windows-latest
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
version: "0.9.28"
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
pyproject.toml
|
||||
uv.lock
|
||||
|
||||
- name: Set up Python 3.11
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv python install 3.11
|
||||
|
||||
- name: Install dependencies
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv sync --locked --python 3.11 --extra dev
|
||||
|
||||
- name: Run venv-holder live E2E
|
||||
shell: bash
|
||||
run: |
|
||||
set -uo pipefail
|
||||
uv run --no-sync python -m pytest \
|
||||
tests/hermes_cli/test_venv_holder_windows_live.py \
|
||||
-o addopts= -v -p no:cacheprovider
|
||||
@@ -1,6 +1,9 @@
|
||||
.DS_Store
|
||||
/venv/
|
||||
/venv.old/
|
||||
/venv.stale.runtime-*/
|
||||
/bin/
|
||||
/.hermes-runtime/
|
||||
/_pycache/
|
||||
*.pyc*
|
||||
__pycache__/
|
||||
@@ -29,6 +32,10 @@ __pycache__/model_tools.cpython-310.pyc
|
||||
__pycache__/web_tools.cpython-310.pyc
|
||||
logs/
|
||||
data/
|
||||
# Bundled community plugin index seed (shipped as package data) — the bare
|
||||
# `data/` pattern above would otherwise swallow it.
|
||||
!hermes_cli/data/
|
||||
!hermes_cli/data/plugin_index.json
|
||||
.pytest_cache/
|
||||
test_durations.json
|
||||
.pytest-cache/
|
||||
@@ -85,8 +92,18 @@ apps/desktop/dist/
|
||||
apps/desktop/src/**/*.js
|
||||
apps/desktop/src/**/*.js.map
|
||||
apps/desktop/src/**/*.d.ts
|
||||
# EXCEPT bundled plain-ESM plugin entries (adopted SDK-consumer plugins,
|
||||
# e.g. hermes-bots): plugin.js IS the source, not tsc output. No .tsx
|
||||
# sibling exists, so the stale-shadow hazard above cannot apply.
|
||||
!apps/desktop/src/plugins/*/plugin.js
|
||||
!apps/desktop/src/global.d.ts
|
||||
!apps/desktop/src/vite-env.d.ts
|
||||
|
||||
# Repo-root build/debug artifacts that must never be committed
|
||||
/log.txt
|
||||
/sqlite_leak_fix.png
|
||||
/*.png.bak
|
||||
/default.tar.gz
|
||||
apps/shared/src/**/*.js
|
||||
apps/shared/src/**/*.js.map
|
||||
apps/shared/src/**/*.d.ts
|
||||
@@ -147,6 +164,9 @@ docs/superpowers/*
|
||||
|
||||
# Persistent dev sandbox dir (scripts/dev-sandbox.sh --persistent)
|
||||
.hermes-sandbox/
|
||||
# Sandbox dirs used by the install/update E2E (tests/install/). The suffix is
|
||||
# the route name, so each route gets its own tree and two can run at once.
|
||||
.hermes-sandbox-e2e*/
|
||||
|
||||
# Interrupted-update breadcrumb + recovery lock written next to the shared venv
|
||||
# by `hermes update` / launch-time self-heal. Runtime state, never a code change
|
||||
@@ -154,6 +174,16 @@ docs/superpowers/*
|
||||
.update-incomplete
|
||||
.update-incomplete.lock
|
||||
|
||||
# Checkout fingerprint the __pycache__ tree was last validated against
|
||||
# (launch-time stale-bytecode sweep). Runtime state, never a code change.
|
||||
.bytecode-fingerprint
|
||||
.bytecode-fingerprint.tmp
|
||||
|
||||
# Installer-written method stamp in the managed checkout root (scripts/install.sh).
|
||||
# Runtime metadata only — never a code change. Ignore so `git status` stays clean
|
||||
# and `hermes update`'s untracked autostash does not treat it as a local edit (#66189 / #54855).
|
||||
/.install_method
|
||||
|
||||
# Tool Search live-test harness output — non-deterministic model transcripts,
|
||||
# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo.
|
||||
scripts/out/
|
||||
@@ -172,5 +202,17 @@ apps/desktop/demo/
|
||||
# image-provider (fal.media) URL — they are NEVER committed to the repo. The
|
||||
# PR body is the archive. See the hermes-agent-dev skill's
|
||||
# pr-infographic-workflow reference (storage rule + lapse #8 / #COMMIT-1).
|
||||
#
|
||||
# Spelling variants are listed because a single `infographic/` pattern was
|
||||
# sidestepped by an `infograficos/` directory (#70552). .gitignore is only
|
||||
# the first line of defence and cannot stop `git add -f` at all — the
|
||||
# infographic-check CI job is what actually enforces this.
|
||||
infographic/
|
||||
infographics/
|
||||
infograficos/
|
||||
infografico/
|
||||
native/fts5_cjk/*.so
|
||||
# Runtime marker written by hermes update when a lazy dependency refresh is
|
||||
# interrupted; consumed by launch-time recovery. Never commit it (was tracked
|
||||
# by accident via 3a69e34702, removed in the #72002 salvage).
|
||||
.lazy-refresh-incomplete
|
||||
|
||||
@@ -18,6 +18,7 @@ Teknium <127238744+teknium1@users.noreply.github.com> <teknium@nousresearch.com>
|
||||
# Format: Canonical Name <GH-noreply> <commit-email>
|
||||
|
||||
# Verified via GH API email search
|
||||
kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> <kshitijkapoorr@gmail.com>
|
||||
luyao618 <364939526@qq.com> <364939526@qq.com>
|
||||
ethernet8023 <arilotter@gmail.com> <arilotter@gmail.com>
|
||||
nicoloboschi <boschi1997@gmail.com> <boschi1997@gmail.com>
|
||||
@@ -86,7 +87,7 @@ yongtenglei <yongtenglei@gmail.com> <yongtenglei@gmail.com>
|
||||
|
||||
# Nous Research team
|
||||
benbarclay <ben@nousresearch.com> <ben@nousresearch.com>
|
||||
jquesnelle <jonny@nousresearch.com> <jonny@nousresearch.com>
|
||||
yoniebans <jonny@nousresearch.com> <jonny@nousresearch.com>
|
||||
|
||||
# GH contributor list verified
|
||||
spideystreet <dhicham.pro@gmail.com> <dhicham.pro@gmail.com>
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
# needed to prevent bad npm that has min-release-age but not exclude
|
||||
engine-strict=true
|
||||
|
||||
min-release-age=14
|
||||
# allow assistant-ui packages & a couple specific deps since they update a LOT.
|
||||
# remove this when we stabilize (or we haven't updated in 2 wks)
|
||||
min-release-age-exclude[]=@assistant-ui/*
|
||||
min-release-age-exclude[]=assistant-cloud
|
||||
min-release-age-exclude[]=assistant-stream
|
||||
min-release-age-exclude[]=@radix-ui/*
|
||||
min-release-age-exclude[]=radix-ui
|
||||
min-release-age-exclude[]=safe-content-frame
|
||||
|
||||
# react-router 8.3.0 includes fixes for vulns. remove this when 8.3.0 is > 2wks old.
|
||||
min-release-age-exclude[]=react-router
|
||||
|
||||
# eslint 10.8.0 includes fixes for vulns. remove this when 10.8.0 is > 2wks old.
|
||||
min-release-age-exclude[]=eslint
|
||||
min-release-age-exclude[]=@eslint/*
|
||||
|
||||
# tar 7.5.21 includes fixes for vulns. remove this when 7.5.21 is > 2wks old
|
||||
min-release-age-exclude[]=tar
|
||||
|
||||
# concurrently 10.0.4 includes fixes for vulns. remove this when 10.0.4 is > 2wks old
|
||||
min-release-age-exclude[]=concurrently
|
||||
|
||||
# fast-uri 3.1.4 includes fixes for vulns. remove this when 3.1.4 is > 2wks old
|
||||
min-release-age-exclude[]=fast-uri
|
||||
|
||||
# minimatch 10.2.6 includes fixes for vulns. remove this when 10.2.6 is > 2wks old
|
||||
min-release-age-exclude[]=minimatch
|
||||
|
||||
|
||||
# brace-expansion 5.0.9 includes fixes for vulns. remove this when 5.0.9 is > 2wks old
|
||||
min-release-age-exclude[]=brace-expansion
|
||||
|
||||
# js-yaml 4.3.1 includes fixes for GHSA-5p4m-2wfm-xmqj. remove when > 2wks old (rel 2026-07-31)
|
||||
min-release-age-exclude[]=js-yaml
|
||||
|
||||
# nanoid 3.3.17 includes fixes for GHSA-2v37-7h3g-55p8. remove when > 2wks old (rel 2026-08-03)
|
||||
min-release-age-exclude[]=nanoid
|
||||
|
||||
# mermaid 11.16.1 includes fixes for 5 GHSAs. remove when > 2wks old (rel 2026-08-04)
|
||||
min-release-age-exclude[]=mermaid
|
||||
|
||||
# dompurify 3.4.13 includes fixes for GHSA-55q2-fjhq-7xh7. remove when > 2wks old (rel 2026-08-03)
|
||||
min-release-age-exclude[]=dompurify
|
||||
|
||||
# vite 8.2.0 is the first release depending on rolldown >= 1.2.1, which fixes
|
||||
# a rolldown panic that breaks `npm run build` in apps/desktop
|
||||
# (rolldown/rolldown#10337 — a regression in 1.1.5, the version vite 8.1.5
|
||||
# pins as ~1.1.5). @oxc-project/types is here because rolldown 1.2.1 pins it
|
||||
# as `=0.142.0` — an exact pin, so no older release satisfies it and the age
|
||||
# gate would fail the whole install with ETARGET.
|
||||
# remove these once vite 8.2.0 is > 2wks old.
|
||||
min-release-age-exclude[]=vite
|
||||
min-release-age-exclude[]=rolldown
|
||||
min-release-age-exclude[]=@rolldown/*
|
||||
min-release-age-exclude[]=@oxc-project/types
|
||||
|
||||
# ink needs
|
||||
min-release-age-exclude[]=lightningcss
|
||||
min-release-age-exclude[]=postcss
|
||||
@@ -1,291 +0,0 @@
|
||||
# OpenAI-Compatible API Server for Hermes Agent
|
||||
|
||||
## Motivation
|
||||
|
||||
Every major chat frontend (Open WebUI 126k★, LobeChat 73k★, LibreChat 34k★,
|
||||
AnythingLLM 56k★, NextChat 87k★, ChatBox 39k★, Jan 26k★, HF Chat-UI 8k★,
|
||||
big-AGI 7k★) connects to backends via the OpenAI-compatible REST API with
|
||||
SSE streaming. By exposing this endpoint, hermes-agent becomes instantly
|
||||
usable as a backend for all of them — no custom adapters needed.
|
||||
|
||||
## What It Enables
|
||||
|
||||
```
|
||||
┌──────────────────┐
|
||||
│ Open WebUI │──┐
|
||||
│ LobeChat │ │ POST /v1/chat/completions
|
||||
│ LibreChat │ ├──► Authorization: Bearer <key> ┌─────────────────┐
|
||||
│ AnythingLLM │ │ {"messages": [...]} │ hermes-agent │
|
||||
│ NextChat │ │ │ gateway │
|
||||
│ Any OAI client │──┘ ◄── SSE streaming response │ (API server) │
|
||||
└──────────────────┘ └─────────────────┘
|
||||
```
|
||||
|
||||
A user would:
|
||||
1. Set `API_SERVER_ENABLED=true` in `~/.hermes/.env`
|
||||
2. Run `hermes gateway` (API server starts alongside Telegram/Discord/etc.)
|
||||
3. Point Open WebUI (or any frontend) at `http://localhost:8642/v1`
|
||||
4. Chat with hermes-agent through any OpenAI-compatible UI
|
||||
|
||||
## Endpoints
|
||||
|
||||
| Method | Path | Purpose |
|
||||
|--------|------|---------|
|
||||
| POST | `/v1/chat/completions` | Chat with the agent (streaming + non-streaming) |
|
||||
| GET | `/v1/models` | List available "models" (returns hermes-agent as a model) |
|
||||
| GET | `/health` | Health check |
|
||||
|
||||
## Architecture
|
||||
|
||||
### Option A: Gateway Platform Adapter (recommended)
|
||||
|
||||
Create `gateway/platforms/api_server.py` as a new platform adapter that
|
||||
extends `BasePlatformAdapter`. This is the cleanest approach because:
|
||||
|
||||
- Reuses all gateway infrastructure (session management, auth, context building)
|
||||
- Runs in the same async loop as other adapters
|
||||
- Gets message handling, interrupt support, and session persistence for free
|
||||
- Follows the established pattern (like Telegram, Discord, etc.)
|
||||
- Uses `aiohttp.web` (already a dependency) for the HTTP server
|
||||
|
||||
The adapter would start an `aiohttp.web.Application` server in `connect()`
|
||||
and route incoming HTTP requests through the standard `handle_message()` pipeline.
|
||||
|
||||
### Option B: Standalone Component
|
||||
|
||||
A separate HTTP server class in `gateway/api_server.py` that creates its own
|
||||
AIAgent instances directly. Simpler but duplicates session/auth logic.
|
||||
|
||||
**Recommendation: Option A** — fits the existing architecture, less code to
|
||||
maintain, gets all gateway features for free.
|
||||
|
||||
## Request/Response Format
|
||||
|
||||
### Chat Completions (non-streaming)
|
||||
|
||||
```
|
||||
POST /v1/chat/completions
|
||||
Authorization: Bearer hermes-api-key-here
|
||||
Content-Type: application/json
|
||||
|
||||
{
|
||||
"model": "hermes-agent",
|
||||
"messages": [
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{"role": "user", "content": "What files are in the current directory?"}
|
||||
],
|
||||
"stream": false,
|
||||
"temperature": 0.7
|
||||
}
|
||||
```
|
||||
|
||||
Response:
|
||||
```json
|
||||
{
|
||||
"id": "chatcmpl-abc123",
|
||||
"object": "chat.completion",
|
||||
"created": 1710000000,
|
||||
"model": "hermes-agent",
|
||||
"choices": [{
|
||||
"index": 0,
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": "Here are the files in the current directory:\n..."
|
||||
},
|
||||
"finish_reason": "stop"
|
||||
}],
|
||||
"usage": {
|
||||
"prompt_tokens": 50,
|
||||
"completion_tokens": 200,
|
||||
"total_tokens": 250
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Chat Completions (streaming)
|
||||
|
||||
Same request with `"stream": true`. Response is SSE:
|
||||
|
||||
```
|
||||
data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"role":"assistant"},"finish_reason":null}]}
|
||||
|
||||
data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"content":"Here "},"finish_reason":null}]}
|
||||
|
||||
data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"content":"are "},"finish_reason":null}]}
|
||||
|
||||
data: {"id":"chatcmpl-abc123","object":"chat.completion.chunk","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}
|
||||
|
||||
data: [DONE]
|
||||
```
|
||||
|
||||
### Models List
|
||||
|
||||
```
|
||||
GET /v1/models
|
||||
Authorization: Bearer hermes-api-key-here
|
||||
```
|
||||
|
||||
Response:
|
||||
```json
|
||||
{
|
||||
"object": "list",
|
||||
"data": [{
|
||||
"id": "hermes-agent",
|
||||
"object": "model",
|
||||
"created": 1710000000,
|
||||
"owned_by": "hermes-agent"
|
||||
}]
|
||||
}
|
||||
```
|
||||
|
||||
## Key Design Decisions
|
||||
|
||||
### 1. Session Management
|
||||
|
||||
The OpenAI API is stateless — each request includes the full conversation.
|
||||
But hermes-agent sessions have persistent state (memory, skills, tool context).
|
||||
|
||||
**Approach: Hybrid**
|
||||
- Default: Stateless. Each request is independent. The `messages` array IS
|
||||
the conversation. No session persistence between requests.
|
||||
- Opt-in persistent sessions via `X-Session-ID` header. When provided, the
|
||||
server maintains session state across requests (conversation history,
|
||||
memory context, tool state). This enables richer agent behavior.
|
||||
- The session ID also enables interrupt support — a subsequent request with
|
||||
the same session ID while one is running triggers an interrupt.
|
||||
|
||||
### 2. Streaming
|
||||
|
||||
The agent's `run_conversation()` is synchronous and returns the full response.
|
||||
For real SSE streaming, we need to emit chunks as they're generated.
|
||||
|
||||
**Phase 1 (MVP):** Run agent in a thread, return the complete response as
|
||||
a single SSE chunk + `[DONE]`. This works with all frontends — they just see
|
||||
a fast single-chunk response. Not true streaming but functional.
|
||||
|
||||
**Phase 2:** Add a response callback to AIAgent that emits text chunks as the
|
||||
LLM generates them. The API server captures these via a queue and streams them
|
||||
as SSE events. This gives real token-by-token streaming.
|
||||
|
||||
**Phase 3:** Stream tool execution progress too — emit tool call/result events
|
||||
as the agent works, giving frontends visibility into what the agent is doing.
|
||||
|
||||
### 3. Tool Transparency
|
||||
|
||||
Two modes:
|
||||
- **Opaque (default):** Frontends see only the final response. Tool calls
|
||||
happen server-side and are invisible. Best for general-purpose UIs.
|
||||
- **Transparent (opt-in via header):** Tool calls are emitted as OpenAI-format
|
||||
tool_call/tool_result messages in the stream. Useful for agent-aware frontends.
|
||||
|
||||
### 4. Authentication
|
||||
|
||||
- Bearer token via `Authorization: Bearer <key>` header
|
||||
- Token configured via `API_SERVER_KEY` env var
|
||||
- Optional: allow unauthenticated local-only access (127.0.0.1 bind)
|
||||
- Follows the same pattern as other platform adapters
|
||||
|
||||
### 5. Model Mapping
|
||||
|
||||
Frontends send `"model": "hermes-agent"` (or whatever). The actual LLM model
|
||||
used is configured server-side in config.yaml. The API server maps any
|
||||
requested model name to the configured hermes-agent model.
|
||||
|
||||
Optionally, allow model passthrough: if the frontend sends
|
||||
`"model": "anthropic/claude-sonnet-4"`, the agent uses that model. Controlled
|
||||
by a config flag.
|
||||
|
||||
## Configuration
|
||||
|
||||
```yaml
|
||||
# In config.yaml
|
||||
api_server:
|
||||
enabled: true
|
||||
port: 8642
|
||||
host: "127.0.0.1" # localhost only by default
|
||||
key: "your-secret-key" # or via API_SERVER_KEY env var
|
||||
allow_model_override: false # let clients choose the model
|
||||
max_concurrent: 5 # max simultaneous requests
|
||||
```
|
||||
|
||||
Environment variables:
|
||||
```bash
|
||||
API_SERVER_ENABLED=true
|
||||
API_SERVER_PORT=8642
|
||||
API_SERVER_HOST=127.0.0.1
|
||||
API_SERVER_KEY=your-secret-key
|
||||
```
|
||||
|
||||
## Implementation Plan
|
||||
|
||||
### Phase 1: MVP (non-streaming) — PR
|
||||
|
||||
1. `gateway/platforms/api_server.py` — new adapter
|
||||
- aiohttp.web server with endpoints:
|
||||
- `POST /v1/chat/completions` — Chat Completions API (universal compat)
|
||||
- `POST /v1/responses` — Responses API (server-side state, tool preservation)
|
||||
- `GET /v1/models` — list available models
|
||||
- `GET /health` — health check
|
||||
- Bearer token auth middleware
|
||||
- Non-streaming responses (run agent, return full result)
|
||||
- Chat Completions: stateless, messages array is the conversation
|
||||
- Responses API: server-side conversation storage via previous_response_id
|
||||
- Store full internal conversation (including tool calls) keyed by response ID
|
||||
- On subsequent requests, reconstruct full context from stored chain
|
||||
- Frontend system prompt layered on top of hermes-agent's core prompt
|
||||
|
||||
2. `gateway/config.py` — add `Platform.API_SERVER` enum + config
|
||||
|
||||
3. `gateway/run.py` — register adapter in `_create_adapter()`
|
||||
|
||||
4. Tests in `tests/gateway/test_api_server.py`
|
||||
|
||||
### Phase 2: SSE Streaming
|
||||
|
||||
1. Add response streaming to both endpoints
|
||||
- Chat Completions: `choices[0].delta.content` SSE format
|
||||
- Responses API: semantic events (response.output_text.delta, etc.)
|
||||
- Run agent in thread, collect output via callback queue
|
||||
- Handle client disconnect (cancel agent)
|
||||
|
||||
2. Add `stream_callback` parameter to `AIAgent.run_conversation()`
|
||||
|
||||
### Phase 3: Enhanced Features
|
||||
|
||||
1. Tool call transparency mode (opt-in)
|
||||
2. Model passthrough/override
|
||||
3. Concurrent request limiting
|
||||
4. Usage tracking / rate limiting
|
||||
5. CORS headers for browser-based frontends
|
||||
6. GET /v1/responses/{id} — retrieve stored response
|
||||
7. DELETE /v1/responses/{id} — delete stored response
|
||||
|
||||
## Files Changed
|
||||
|
||||
| File | Change |
|
||||
|------|--------|
|
||||
| `gateway/platforms/api_server.py` | NEW — main adapter (~300 lines) |
|
||||
| `gateway/config.py` | Add Platform.API_SERVER + config (~20 lines) |
|
||||
| `gateway/run.py` | Register adapter in _create_adapter() (~10 lines) |
|
||||
| `tests/gateway/test_api_server.py` | NEW — tests (~200 lines) |
|
||||
| `cli-config.yaml.example` | Add api_server section |
|
||||
| `README.md` | Mention API server in platform list |
|
||||
|
||||
## Compatibility Matrix
|
||||
|
||||
Once implemented, hermes-agent works as a drop-in backend for:
|
||||
|
||||
| Frontend | Stars | How to Connect |
|
||||
|----------|-------|---------------|
|
||||
| Open WebUI | 126k | Settings → Connections → Add OpenAI API, URL: `http://localhost:8642/v1` |
|
||||
| NextChat | 87k | BASE_URL env var |
|
||||
| LobeChat | 73k | Custom provider endpoint |
|
||||
| AnythingLLM | 56k | LLM Provider → Generic OpenAI |
|
||||
| Oobabooga | 42k | Already a backend, not a frontend |
|
||||
| ChatBox | 39k | API Host setting |
|
||||
| LibreChat | 34k | librechat.yaml custom endpoint |
|
||||
| Chatbot UI | 29k | Custom API endpoint |
|
||||
| Jan | 26k | Remote model config |
|
||||
| AionUI | 18k | Custom API endpoint |
|
||||
| HF Chat-UI | 8k | OPENAI_BASE_URL env var |
|
||||
| big-AGI | 7k | Custom endpoint |
|
||||
@@ -1,705 +0,0 @@
|
||||
# Streaming LLM Response Support for Hermes Agent
|
||||
|
||||
## Overview
|
||||
|
||||
Add token-by-token streaming of LLM responses across all platforms. When enabled,
|
||||
users see the response typing out live instead of waiting for the full generation.
|
||||
Streaming is opt-in via config, defaults to off, and all existing non-streaming
|
||||
code paths remain intact as the default.
|
||||
|
||||
## Design Principles
|
||||
|
||||
1. **Feature-flagged**: `streaming.enabled: true` in config.yaml. Off by default.
|
||||
When off, all existing code paths are unchanged — zero risk to current behavior.
|
||||
2. **Callback-based**: A simple `stream_callback(text_delta: str)` function injected
|
||||
into AIAgent. The agent doesn't know or care what the consumer does with tokens.
|
||||
3. **Graceful degradation**: If the provider doesn't support streaming, or streaming
|
||||
fails for any reason, silently fall back to the non-streaming path.
|
||||
4. **Platform-agnostic core**: The streaming mechanism in AIAgent works the same
|
||||
regardless of whether the consumer is CLI, Telegram, Discord, or the API server.
|
||||
|
||||
---
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
stream_callback(delta)
|
||||
│
|
||||
┌─────────────┐ ┌─────────────▼──────────────┐
|
||||
│ LLM API │ │ queue.Queue() │
|
||||
│ (stream) │───►│ thread-safe bridge between │
|
||||
│ │ │ agent thread & consumer │
|
||||
└─────────────┘ └─────────────┬──────────────┘
|
||||
│
|
||||
┌──────────────┼──────────────┐
|
||||
│ │ │
|
||||
┌─────▼─────┐ ┌─────▼─────┐ ┌─────▼─────┐
|
||||
│ CLI │ │ Gateway │ │ API Server│
|
||||
│ print to │ │ edit msg │ │ SSE event │
|
||||
│ terminal │ │ on Tg/Dc │ │ to client │
|
||||
└───────────┘ └───────────┘ └───────────┘
|
||||
```
|
||||
|
||||
The agent runs in a thread. The callback puts tokens into a thread-safe queue.
|
||||
Each consumer reads the queue in its own context (async task, main thread, etc.).
|
||||
|
||||
---
|
||||
|
||||
## Configuration
|
||||
|
||||
### config.yaml
|
||||
|
||||
```yaml
|
||||
streaming:
|
||||
enabled: false # Master switch. Default off.
|
||||
# Per-platform overrides (optional):
|
||||
# cli: true # Override for CLI only
|
||||
# telegram: true # Override for Telegram only
|
||||
# discord: false # Keep Discord non-streaming
|
||||
# api_server: true # Override for API server
|
||||
```
|
||||
|
||||
### Environment variables
|
||||
|
||||
```
|
||||
HERMES_STREAMING_ENABLED=true # Master switch via env
|
||||
```
|
||||
|
||||
### How the flag is read
|
||||
|
||||
- **CLI**: `load_cli_config()` reads `streaming.enabled`, sets env var. AIAgent
|
||||
checks at init time.
|
||||
- **Gateway**: `_run_agent()` reads config, decides whether to pass
|
||||
`stream_callback` to the AIAgent constructor.
|
||||
- **API server**: For Chat Completions `stream=true` requests, always uses streaming
|
||||
regardless of config (the client is explicitly requesting it). For non-stream
|
||||
requests, uses config.
|
||||
|
||||
### Precedence
|
||||
|
||||
1. API server: client's `stream` field overrides everything
|
||||
2. Per-platform config override (e.g., `streaming.telegram: true`)
|
||||
3. Master `streaming.enabled` flag
|
||||
4. Default: off
|
||||
|
||||
---
|
||||
|
||||
## Implementation Plan
|
||||
|
||||
### Phase 1: Core streaming infrastructure in AIAgent
|
||||
|
||||
**File: run_agent.py**
|
||||
|
||||
#### 1a. Add stream_callback parameter to __init__ (~5 lines)
|
||||
|
||||
```python
|
||||
def __init__(self, ..., stream_callback: callable = None, ...):
|
||||
self.stream_callback = stream_callback
|
||||
```
|
||||
|
||||
No other init changes. The callback is optional — when None, everything
|
||||
works exactly as before.
|
||||
|
||||
#### 1b. Add _run_streaming_chat_completion() method (~65 lines)
|
||||
|
||||
New method for Chat Completions API streaming:
|
||||
|
||||
```python
|
||||
def _run_streaming_chat_completion(self, api_kwargs: dict):
|
||||
"""Stream a chat completion, emitting text tokens via stream_callback.
|
||||
|
||||
Returns a fake response object compatible with the non-streaming code path.
|
||||
Falls back to non-streaming on any error.
|
||||
"""
|
||||
stream_kwargs = dict(api_kwargs)
|
||||
stream_kwargs["stream"] = True
|
||||
stream_kwargs["stream_options"] = {"include_usage": True}
|
||||
|
||||
accumulated_content = []
|
||||
accumulated_tool_calls = {} # index -> {id, name, arguments}
|
||||
final_usage = None
|
||||
|
||||
try:
|
||||
stream = self.client.chat.completions.create(**stream_kwargs)
|
||||
|
||||
for chunk in stream:
|
||||
if not chunk.choices:
|
||||
# Usage-only chunk (final)
|
||||
if chunk.usage:
|
||||
final_usage = chunk.usage
|
||||
continue
|
||||
|
||||
delta = chunk.choices[0].delta
|
||||
|
||||
# Text content — emit via callback
|
||||
if delta.content:
|
||||
accumulated_content.append(delta.content)
|
||||
if self.stream_callback:
|
||||
try:
|
||||
self.stream_callback(delta.content)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Tool call deltas — accumulate silently
|
||||
if delta.tool_calls:
|
||||
for tc_delta in delta.tool_calls:
|
||||
idx = tc_delta.index
|
||||
if idx not in accumulated_tool_calls:
|
||||
accumulated_tool_calls[idx] = {
|
||||
"id": tc_delta.id or "",
|
||||
"name": "", "arguments": ""
|
||||
}
|
||||
if tc_delta.function:
|
||||
if tc_delta.function.name:
|
||||
accumulated_tool_calls[idx]["name"] = tc_delta.function.name
|
||||
if tc_delta.function.arguments:
|
||||
accumulated_tool_calls[idx]["arguments"] += tc_delta.function.arguments
|
||||
|
||||
# Build fake response compatible with existing code
|
||||
tool_calls = []
|
||||
for idx in sorted(accumulated_tool_calls):
|
||||
tc = accumulated_tool_calls[idx]
|
||||
if tc["name"]:
|
||||
tool_calls.append(SimpleNamespace(
|
||||
id=tc["id"], type="function",
|
||||
function=SimpleNamespace(name=tc["name"], arguments=tc["arguments"]),
|
||||
))
|
||||
|
||||
return SimpleNamespace(
|
||||
choices=[SimpleNamespace(
|
||||
message=SimpleNamespace(
|
||||
content="".join(accumulated_content) or "",
|
||||
tool_calls=tool_calls or None,
|
||||
role="assistant",
|
||||
),
|
||||
finish_reason="tool_calls" if tool_calls else "stop",
|
||||
)],
|
||||
usage=final_usage,
|
||||
model=self.model,
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.debug("Streaming failed, falling back to non-streaming: %s", e)
|
||||
return self.client.chat.completions.create(**api_kwargs)
|
||||
```
|
||||
|
||||
#### 1c. Modify _run_codex_stream() for Responses API (~10 lines)
|
||||
|
||||
The method already iterates the stream. Add callback emission:
|
||||
|
||||
```python
|
||||
def _run_codex_stream(self, api_kwargs: dict):
|
||||
with self.client.responses.stream(**api_kwargs) as stream:
|
||||
for event in stream:
|
||||
# Emit text deltas if streaming callback is set
|
||||
if self.stream_callback and hasattr(event, 'type'):
|
||||
if event.type == 'response.output_text.delta':
|
||||
try:
|
||||
self.stream_callback(event.delta)
|
||||
except Exception:
|
||||
pass
|
||||
return stream.get_final_response()
|
||||
```
|
||||
|
||||
#### 1d. Modify _interruptible_api_call() (~5 lines)
|
||||
|
||||
Add the streaming branch:
|
||||
|
||||
```python
|
||||
def _call():
|
||||
try:
|
||||
if self.api_mode == "codex_responses":
|
||||
result["response"] = self._run_codex_stream(api_kwargs)
|
||||
elif self.stream_callback is not None:
|
||||
result["response"] = self._run_streaming_chat_completion(api_kwargs)
|
||||
else:
|
||||
result["response"] = self.client.chat.completions.create(**api_kwargs)
|
||||
except Exception as e:
|
||||
result["error"] = e
|
||||
```
|
||||
|
||||
#### 1e. Signal end-of-stream to consumers (~5 lines)
|
||||
|
||||
After the API call returns, signal the callback that streaming is done
|
||||
so consumers can finalize (remove cursor, close SSE, etc.):
|
||||
|
||||
```python
|
||||
# In run_conversation(), after _interruptible_api_call returns:
|
||||
if self.stream_callback:
|
||||
try:
|
||||
self.stream_callback(None) # None = end of stream signal
|
||||
except Exception:
|
||||
pass
|
||||
```
|
||||
|
||||
Consumers check: `if delta is None: finalize()`
|
||||
|
||||
**Tests for Phase 1:** (~150 lines)
|
||||
- Test _run_streaming_chat_completion with mocked stream
|
||||
- Test fallback to non-streaming on error
|
||||
- Test tool_call accumulation during streaming
|
||||
- Test stream_callback receives correct deltas
|
||||
- Test None signal at end of stream
|
||||
- Test streaming disabled when callback is None
|
||||
|
||||
---
|
||||
|
||||
### Phase 2: Gateway consumers (Telegram, Discord, etc.)
|
||||
|
||||
**File: gateway/run.py**
|
||||
|
||||
#### 2a. Read streaming config (~15 lines)
|
||||
|
||||
In `_run_agent()`, before creating the AIAgent:
|
||||
|
||||
```python
|
||||
# Read streaming config
|
||||
_streaming_enabled = False
|
||||
try:
|
||||
# Check per-platform override first
|
||||
platform_key = source.platform.value if source.platform else ""
|
||||
_stream_cfg = {} # loaded from config.yaml streaming section
|
||||
if _stream_cfg.get(platform_key) is not None:
|
||||
_streaming_enabled = bool(_stream_cfg[platform_key])
|
||||
else:
|
||||
_streaming_enabled = bool(_stream_cfg.get("enabled", False))
|
||||
except Exception:
|
||||
pass
|
||||
# Env var override
|
||||
if os.getenv("HERMES_STREAMING_ENABLED", "").lower() in ("true", "1", "yes"):
|
||||
_streaming_enabled = True
|
||||
```
|
||||
|
||||
#### 2b. Set up queue + callback (~15 lines)
|
||||
|
||||
```python
|
||||
_stream_q = None
|
||||
_stream_done = None
|
||||
_stream_msg_id = [None] # mutable ref for the async task
|
||||
|
||||
if _streaming_enabled:
|
||||
import queue as _q
|
||||
_stream_q = _q.Queue()
|
||||
_stream_done = threading.Event()
|
||||
|
||||
def _on_token(delta):
|
||||
if delta is None:
|
||||
_stream_done.set()
|
||||
else:
|
||||
_stream_q.put(delta)
|
||||
```
|
||||
|
||||
Pass `stream_callback=_on_token` to the AIAgent constructor.
|
||||
|
||||
#### 2c. Telegram/Discord stream preview task (~50 lines)
|
||||
|
||||
```python
|
||||
async def stream_preview():
|
||||
"""Progressively edit a message with streaming tokens."""
|
||||
if not _stream_q:
|
||||
return
|
||||
adapter = self.adapters.get(source.platform)
|
||||
if not adapter:
|
||||
return
|
||||
|
||||
accumulated = []
|
||||
token_count = 0
|
||||
last_edit = 0.0
|
||||
MIN_TOKENS = 20 # Don't show until enough context
|
||||
EDIT_INTERVAL = 1.5 # Respect Telegram rate limits
|
||||
|
||||
try:
|
||||
while not _stream_done.is_set():
|
||||
try:
|
||||
chunk = _stream_q.get(timeout=0.1)
|
||||
accumulated.append(chunk)
|
||||
token_count += 1
|
||||
except queue.Empty:
|
||||
continue
|
||||
|
||||
now = time.monotonic()
|
||||
if token_count >= MIN_TOKENS and (now - last_edit) >= EDIT_INTERVAL:
|
||||
preview = "".join(accumulated) + " ▌"
|
||||
if _stream_msg_id[0] is None:
|
||||
r = await adapter.send(
|
||||
chat_id=source.chat_id,
|
||||
content=preview,
|
||||
metadata=_thread_metadata,
|
||||
)
|
||||
if r.success and r.message_id:
|
||||
_stream_msg_id[0] = r.message_id
|
||||
else:
|
||||
await adapter.edit_message(
|
||||
chat_id=source.chat_id,
|
||||
message_id=_stream_msg_id[0],
|
||||
content=preview,
|
||||
)
|
||||
last_edit = now
|
||||
|
||||
# Drain remaining tokens
|
||||
while not _stream_q.empty():
|
||||
accumulated.append(_stream_q.get_nowait())
|
||||
|
||||
# Final edit — remove cursor, show complete text
|
||||
if _stream_msg_id[0] and accumulated:
|
||||
await adapter.edit_message(
|
||||
chat_id=source.chat_id,
|
||||
message_id=_stream_msg_id[0],
|
||||
content="".join(accumulated),
|
||||
)
|
||||
|
||||
except asyncio.CancelledError:
|
||||
# Clean up on cancel
|
||||
if _stream_msg_id[0] and accumulated:
|
||||
try:
|
||||
await adapter.edit_message(
|
||||
chat_id=source.chat_id,
|
||||
message_id=_stream_msg_id[0],
|
||||
content="".join(accumulated),
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as e:
|
||||
logger.debug("stream_preview error: %s", e)
|
||||
```
|
||||
|
||||
#### 2d. Skip final send if already streamed (~10 lines)
|
||||
|
||||
In `_process_message_background()` (base.py), after getting the response,
|
||||
if streaming was active and `_stream_msg_id[0]` is set, the final response
|
||||
was already delivered via progressive edits. Skip the normal `self.send()`
|
||||
call to avoid duplicating the message.
|
||||
|
||||
This is the most delicate integration point — we need to communicate from
|
||||
the gateway's `_run_agent` back to the base adapter's response sender that
|
||||
the response was already delivered. Options:
|
||||
|
||||
- **Option A**: Return a special marker in the result dict:
|
||||
`result["_streamed_msg_id"] = _stream_msg_id[0]`
|
||||
The base adapter checks this and skips `send()`.
|
||||
|
||||
- **Option B**: Edit the already-sent message with the final response
|
||||
(which may differ slightly from accumulated tokens due to think-block
|
||||
stripping, etc.) and don't send a new one.
|
||||
|
||||
- **Option C**: The stream preview task handles the FULL final response
|
||||
(including any post-processing), and the handler returns None to skip
|
||||
the normal send path.
|
||||
|
||||
Recommended: **Option A** — cleanest separation. The result dict already
|
||||
carries metadata; adding one more field is low-risk.
|
||||
|
||||
**Platform-specific considerations:**
|
||||
|
||||
| Platform | Edit support | Rate limits | Streaming approach |
|
||||
|----------|-------------|-------------|-------------------|
|
||||
| Telegram | ✅ edit_message_text | ~20 edits/min | Edit every 1.5s |
|
||||
| Discord | ✅ message.edit | 5 edits/5s per message | Edit every 1.2s |
|
||||
| Slack | ✅ chat.update | Tier 3 (~50/min) | Edit every 1.5s |
|
||||
| WhatsApp | ❌ no edit support | N/A | Skip streaming, use normal path |
|
||||
| HomeAssistant | ❌ no edit | N/A | Skip streaming |
|
||||
| API Server | ✅ SSE native | No limit | Real SSE events |
|
||||
|
||||
WhatsApp and HomeAssistant fall back to non-streaming automatically because
|
||||
they don't support message editing.
|
||||
|
||||
**Tests for Phase 2:** (~100 lines)
|
||||
- Test stream_preview sends/edits correctly
|
||||
- Test skip-final-send when streaming delivered
|
||||
- Test WhatsApp/HA graceful fallback
|
||||
- Test streaming disabled per-platform config
|
||||
- Test thread_id metadata forwarded in stream messages
|
||||
|
||||
---
|
||||
|
||||
### Phase 3: CLI streaming
|
||||
|
||||
**File: cli.py**
|
||||
|
||||
#### 3a. Set up callback in the CLI chat loop (~20 lines)
|
||||
|
||||
In `_chat_once()` or wherever the agent is invoked:
|
||||
|
||||
```python
|
||||
if streaming_enabled:
|
||||
_stream_q = queue.Queue()
|
||||
_stream_done = threading.Event()
|
||||
|
||||
def _cli_stream_callback(delta):
|
||||
if delta is None:
|
||||
_stream_done.set()
|
||||
else:
|
||||
_stream_q.put(delta)
|
||||
|
||||
agent.stream_callback = _cli_stream_callback
|
||||
```
|
||||
|
||||
#### 3b. Token display thread/task (~30 lines)
|
||||
|
||||
Start a thread that reads the queue and prints tokens:
|
||||
|
||||
```python
|
||||
def _stream_display():
|
||||
"""Print tokens to terminal as they arrive."""
|
||||
first_token = True
|
||||
while not _stream_done.is_set():
|
||||
try:
|
||||
delta = _stream_q.get(timeout=0.1)
|
||||
except queue.Empty:
|
||||
continue
|
||||
if first_token:
|
||||
# Print response box top border
|
||||
_cprint(f"\n{top}")
|
||||
first_token = False
|
||||
sys.stdout.write(delta)
|
||||
sys.stdout.flush()
|
||||
# Drain remaining
|
||||
while not _stream_q.empty():
|
||||
sys.stdout.write(_stream_q.get_nowait())
|
||||
sys.stdout.flush()
|
||||
# Print bottom border
|
||||
_cprint(f"\n\n{bot}")
|
||||
```
|
||||
|
||||
**Integration challenge: prompt_toolkit**
|
||||
|
||||
The CLI uses prompt_toolkit which controls the terminal. Writing directly
|
||||
to stdout while prompt_toolkit is active can cause display corruption.
|
||||
The existing KawaiiSpinner already solves this by using prompt_toolkit's
|
||||
`patch_stdout` context. The streaming display would need to do the same.
|
||||
|
||||
Alternative: use `_cprint()` for each token chunk (routes through
|
||||
prompt_toolkit's renderer). But this might be slow for individual tokens.
|
||||
|
||||
Recommended approach: accumulate tokens in small batches (e.g., every 50ms)
|
||||
and `_cprint()` the batch. This balances display responsiveness with
|
||||
prompt_toolkit compatibility.
|
||||
|
||||
**Tests for Phase 3:** (~50 lines)
|
||||
- Test CLI streaming callback setup
|
||||
- Test response box borders with streaming
|
||||
- Test fallback when streaming disabled
|
||||
|
||||
---
|
||||
|
||||
### Phase 4: API Server real streaming
|
||||
|
||||
**File: gateway/platforms/api_server.py**
|
||||
|
||||
Replace the pseudo-streaming `_write_sse_chat_completion()` with real
|
||||
token-by-token SSE when the agent supports it.
|
||||
|
||||
#### 4a. Wire streaming callback for stream=true requests (~20 lines)
|
||||
|
||||
```python
|
||||
if stream:
|
||||
_stream_q = queue.Queue()
|
||||
|
||||
def _api_stream_callback(delta):
|
||||
_stream_q.put(delta) # None = done
|
||||
|
||||
# Pass callback to _run_agent
|
||||
result, usage = await self._run_agent(
|
||||
..., stream_callback=_api_stream_callback,
|
||||
)
|
||||
```
|
||||
|
||||
#### 4b. Real SSE writer (~40 lines)
|
||||
|
||||
```python
|
||||
async def _write_real_sse(self, request, completion_id, model, stream_q):
|
||||
response = web.StreamResponse(
|
||||
headers={"Content-Type": "text/event-stream", "Cache-Control": "no-cache"},
|
||||
)
|
||||
await response.prepare(request)
|
||||
|
||||
# Role chunk
|
||||
await response.write(...)
|
||||
|
||||
# Stream content chunks as they arrive
|
||||
while True:
|
||||
try:
|
||||
delta = await asyncio.get_event_loop().run_in_executor(
|
||||
None, lambda: stream_q.get(timeout=0.1)
|
||||
)
|
||||
except queue.Empty:
|
||||
continue
|
||||
|
||||
if delta is None: # End of stream
|
||||
break
|
||||
|
||||
chunk = {"id": completion_id, "object": "chat.completion.chunk", ...
|
||||
"choices": [{"delta": {"content": delta}, ...}]}
|
||||
await response.write(f"data: {json.dumps(chunk)}\n\n".encode())
|
||||
|
||||
# Finish + [DONE]
|
||||
await response.write(...)
|
||||
await response.write(b"data: [DONE]\n\n")
|
||||
return response
|
||||
```
|
||||
|
||||
**Challenge: concurrent execution**
|
||||
|
||||
The agent runs in a thread executor. SSE writing happens in the async event
|
||||
loop. The queue bridges them. But `_run_agent()` currently awaits the full
|
||||
result before returning. For real streaming, we need to start the agent in
|
||||
the background and stream tokens while it runs:
|
||||
|
||||
```python
|
||||
# Start agent in background
|
||||
agent_task = asyncio.create_task(self._run_agent_async(...))
|
||||
|
||||
# Stream tokens while agent runs
|
||||
await self._write_real_sse(request, ..., stream_q)
|
||||
|
||||
# Agent is done by now (stream_q received None)
|
||||
result, usage = await agent_task
|
||||
```
|
||||
|
||||
This requires splitting `_run_agent` into an async version that doesn't
|
||||
block waiting for the result, or running it in a separate task.
|
||||
|
||||
**Responses API SSE format:**
|
||||
|
||||
For `/v1/responses` with `stream=true`, the SSE events are different:
|
||||
|
||||
```
|
||||
event: response.output_text.delta
|
||||
data: {"type":"response.output_text.delta","delta":"Hello"}
|
||||
|
||||
event: response.completed
|
||||
data: {"type":"response.completed","response":{...}}
|
||||
```
|
||||
|
||||
This needs a separate SSE writer that emits Responses API format events.
|
||||
|
||||
**Tests for Phase 4:** (~80 lines)
|
||||
- Test real SSE streaming with mocked agent
|
||||
- Test SSE event format (Chat Completions vs Responses)
|
||||
- Test client disconnect during streaming
|
||||
- Test fallback to pseudo-streaming when callback not available
|
||||
|
||||
---
|
||||
|
||||
## Integration Issues & Edge Cases
|
||||
|
||||
### 1. Tool calls during streaming
|
||||
|
||||
When the model returns tool calls instead of text, no text tokens are emitted.
|
||||
The stream_callback is simply never called with text. After tools execute, the
|
||||
next API call may produce the final text response — streaming picks up again.
|
||||
|
||||
The stream preview task needs to handle this: if no tokens arrive during a
|
||||
tool-call round, don't send/edit any message. The tool progress messages
|
||||
continue working as before.
|
||||
|
||||
### 2. Duplicate messages
|
||||
|
||||
The biggest risk: the agent sends the final response normally (via the
|
||||
existing send path) AND the stream preview already showed it. The user
|
||||
sees the response twice.
|
||||
|
||||
Prevention: when streaming is active and tokens were delivered, the final
|
||||
response send must be suppressed. The `result["_streamed_msg_id"]` marker
|
||||
tells the base adapter to skip its normal send.
|
||||
|
||||
### 3. Response post-processing
|
||||
|
||||
The final response may differ from the accumulated streamed tokens:
|
||||
- Think block stripping (`<think>...</think>` removed)
|
||||
- Trailing whitespace cleanup
|
||||
- Tool result media tag appending
|
||||
|
||||
The stream preview shows raw tokens. The final edit should use the
|
||||
post-processed version. This means the final edit (removing the cursor)
|
||||
should use the post-processed `final_response`, not just the accumulated
|
||||
stream text.
|
||||
|
||||
### 4. Context compression during streaming
|
||||
|
||||
If the agent triggers context compression mid-conversation, the streaming
|
||||
tokens from BEFORE compression are from a different context than those
|
||||
after. This isn't a problem in practice — compression happens between
|
||||
API calls, not during streaming.
|
||||
|
||||
### 5. Interrupt during streaming
|
||||
|
||||
User sends a new message while streaming → interrupt. The stream is killed
|
||||
(HTTP connection closed), accumulated tokens are shown as-is (no cursor),
|
||||
and the interrupt message is processed normally. This is already handled by
|
||||
`_interruptible_api_call` closing the client.
|
||||
|
||||
### 6. Multi-model / fallback
|
||||
|
||||
If the primary model fails and the agent falls back to a different model,
|
||||
streaming state resets. The fallback call may or may not support streaming.
|
||||
The graceful fallback in `_run_streaming_chat_completion` handles this.
|
||||
|
||||
### 7. Rate limiting on edits
|
||||
|
||||
Telegram: ~20 edits/minute (~1 every 3 seconds to be safe)
|
||||
Discord: 5 edits per 5 seconds per message
|
||||
Slack: ~50 API calls/minute
|
||||
|
||||
The 1.5s edit interval is conservative enough for all platforms. If we get
|
||||
429 rate limit errors on edits, just skip that edit cycle and try next time.
|
||||
|
||||
---
|
||||
|
||||
## Files Changed Summary
|
||||
|
||||
| File | Phase | Changes |
|
||||
|------|-------|---------|
|
||||
| `run_agent.py` | 1 | +stream_callback param, +_run_streaming_chat_completion(), modify _run_codex_stream(), modify _interruptible_api_call() |
|
||||
| `gateway/run.py` | 2 | +streaming config reader, +queue/callback setup, +stream_preview task, +skip-final-send logic |
|
||||
| `gateway/platforms/base.py` | 2 | +check for _streamed_msg_id in response handler |
|
||||
| `cli.py` | 3 | +streaming setup, +token display, +response box integration |
|
||||
| `gateway/platforms/api_server.py` | 4 | +real SSE writer, +streaming callback wiring |
|
||||
| `hermes_cli/config.py` | 1 | +streaming config defaults |
|
||||
| `cli-config.yaml.example` | 1 | +streaming section |
|
||||
| `tests/test_streaming.py` | 1-4 | NEW — ~380 lines of tests |
|
||||
|
||||
**Total new code**: ~500 lines across all phases
|
||||
**Total test code**: ~380 lines
|
||||
|
||||
---
|
||||
|
||||
## Rollout Plan
|
||||
|
||||
1. **Phase 1** (core): Merge to main. Streaming disabled by default.
|
||||
Zero impact on existing behavior. Can be tested with env var.
|
||||
|
||||
2. **Phase 2** (gateway): Merge to main. Test on Telegram manually.
|
||||
Enable per-platform: `streaming.telegram: true` in config.
|
||||
|
||||
3. **Phase 3** (CLI): Merge to main. Test in terminal.
|
||||
Enable: `streaming.cli: true` or `streaming.enabled: true`.
|
||||
|
||||
4. **Phase 4** (API server): Merge to main. Test with Open WebUI.
|
||||
Auto-enabled when client sends `stream: true`.
|
||||
|
||||
Each phase is independently mergeable and testable. Streaming stays
|
||||
off by default throughout. Once all phases are stable, consider
|
||||
changing the default to enabled.
|
||||
|
||||
---
|
||||
|
||||
## Config Reference (final state)
|
||||
|
||||
```yaml
|
||||
# config.yaml
|
||||
streaming:
|
||||
enabled: false # Master switch (default: off)
|
||||
cli: true # Per-platform override
|
||||
telegram: true
|
||||
discord: true
|
||||
slack: true
|
||||
api_server: true # API server always streams when client requests it
|
||||
edit_interval: 1.5 # Seconds between message edits (default: 1.5)
|
||||
min_tokens: 20 # Tokens before first display (default: 20)
|
||||
```
|
||||
|
||||
```bash
|
||||
# Environment variable override
|
||||
HERMES_STREAMING_ENABLED=true
|
||||
```
|
||||
@@ -0,0 +1 @@
|
||||
3.11
|
||||
@@ -210,6 +210,45 @@ backends, providers, notifiers), don't merge them one at a time — design an
|
||||
ABC + orchestrator, wrap the existing built-in as the first provider, and turn
|
||||
the competing PRs into plugins against that interface.
|
||||
|
||||
### Surface capability is a property of the SESSION, never of the process env
|
||||
|
||||
A tool that only works because of *who is on the other end of the connection* —
|
||||
the desktop app's panes, the in-app browser, message reactions, Projects — must
|
||||
resolve its availability from the **session's own source**, not from an env var
|
||||
on the backend process.
|
||||
|
||||
The client and the backend are separate machines on separate clocks. The
|
||||
desktop app can be driving a backend Electron spawned locally, one over SSH,
|
||||
one behind a plain URL + token, or Hermes Cloud. Only the first two are spawned
|
||||
by us and carry `HERMES_DESKTOP=1`. Every env-keyed GUI gate is therefore a
|
||||
silent no-op on the other half of the topologies, and the failure is invisible:
|
||||
the tool is stripped from the schema before the model ever sees it, on the same
|
||||
backend whose platform hint is telling the model it's *"chatting inside the
|
||||
Hermes desktop app."*
|
||||
|
||||
The pattern that works:
|
||||
|
||||
- **The toolset is the surface gate.** Keep the tools off `_HERMES_CORE_TOOLS`
|
||||
(nobody else should pay their schema) and put them in a named toolset —
|
||||
`desktop_ui`, `project`. The GUI gateway's `_load_enabled_toolsets(platform)`
|
||||
folds that toolset in when the session's platform says GUI. One resolver,
|
||||
every topology.
|
||||
- **`check_fn` answers reachability or user opt-in, not surface.** "Is the
|
||||
renderer bridge wired?", "did the user enable reactions?" — fine. "Was I
|
||||
spawned by Electron?" — not fine. `check_fn` results are also TTL-cached
|
||||
process-wide (`tools/registry.py`), so a per-session answer does not belong
|
||||
there at all: one process serves many sessions.
|
||||
- **Ask which identity you actually mean.** `HERMES_DESKTOP=1` legitimately
|
||||
marks *"this backend process was spawned by the app"* — it gates the cron
|
||||
ticker and web-dist handling correctly. It does NOT mean "a GUI is watching",
|
||||
and the embedded terminal pane (`hermes --tui` against that same backend) is
|
||||
the standing counterexample.
|
||||
|
||||
Same test both ways: if the capability would still make sense with the client
|
||||
on another machine, it is session-scoped. Cover it with a test that asserts the
|
||||
GUI session gets the tool **with the env var absent** — that's the assertion
|
||||
the original gate could never have passed.
|
||||
|
||||
## Development Environment
|
||||
|
||||
```bash
|
||||
@@ -325,7 +364,7 @@ class AIAgent:
|
||||
provider: str = None,
|
||||
api_mode: str = None, # "chat_completions" | "codex_responses" | ...
|
||||
model: str = "", # empty → resolved from config/provider later
|
||||
max_iterations: int = 90, # tool-calling iterations (shared with subagents)
|
||||
max_iterations: int = 500, # tool-calling iterations (shared with subagents)
|
||||
enabled_toolsets: list = None,
|
||||
disabled_toolsets: list = None,
|
||||
quiet_mode: bool = False,
|
||||
@@ -758,12 +797,45 @@ as a side effect of importing `model_tools.py`. Code paths that read plugin
|
||||
state without importing `model_tools.py` first must call `discover_plugins()`
|
||||
explicitly (it's idempotent).
|
||||
|
||||
#### Native plugin compatibility policy
|
||||
|
||||
The canonical contract and deprecation policy live in
|
||||
`website/docs/developer-guide/plugins/index.md#native-plugin-compatibility-contract`.
|
||||
Compatibility is enforced as a behavior contract, not through a monolithic
|
||||
`PLUGIN_API_VERSION`, a manifest-wide native `api:` match, or version literals
|
||||
on unrelated payloads. Keep documented plugin surfaces additive:
|
||||
|
||||
- add hook payload data as keyword fields; signature-inspect callbacks so old
|
||||
narrow signatures receive only fields they declare, while `**kwargs`
|
||||
callbacks receive the complete payload;
|
||||
- do not remove or rename `PluginContext` methods; make new parameters optional
|
||||
with defaults and keyword-only where possible;
|
||||
- ignore unknown native manifest fields;
|
||||
- give new provider methods default implementations, and signature-inspect
|
||||
optional callback kwargs rather than forwarding them unconditionally;
|
||||
- use a local schema version only for a capability with a wire or persisted
|
||||
contract, and preserve old state/config/session replay or ship a migration.
|
||||
|
||||
Deprecations require a once-per-process warning, a documented replacement and
|
||||
migration note, and at least two subsequent minor releases before removal.
|
||||
Compatibility tests must load frozen plugins through the real discovery path
|
||||
and assert outcomes. Do not replace these with exact registry/catalog counts,
|
||||
source-reading tests, or assertions that a global version literal changed.
|
||||
|
||||
### Memory-provider plugins (`plugins/memory/<name>/`)
|
||||
|
||||
Separate discovery system for pluggable memory backends. Current built-in
|
||||
providers include **honcho, mem0, supermemory, byterover, hindsight,
|
||||
holographic, openviking, retaindb**.
|
||||
|
||||
Discovery covers the same four sources as the general `PluginManager` —
|
||||
bundled, `$HERMES_HOME/plugins/`, `./.hermes/plugins/` (opt-in via
|
||||
`HERMES_ENABLE_PROJECT_PLUGINS`), and `hermes_agent.memory_providers` entry
|
||||
points — but with **bundled-first** precedence, the reverse of the general
|
||||
system's later-wins order: a memory provider is activated by name, so a
|
||||
dropped-in directory must not be able to shadow a shipped one. Discovery
|
||||
enumerates without importing; nothing runs until `memory.provider` names it.
|
||||
|
||||
Each provider implements the `MemoryProvider` ABC (see `agent/memory_provider.py`)
|
||||
and is orchestrated by `agent/memory_manager.py`. Lifecycle hooks include
|
||||
`sync_turn(turn_messages)`, `prefetch(query)`, `shutdown()`, and optional
|
||||
@@ -848,6 +920,72 @@ plug into `agent/context_engine.py`; image-gen providers into
|
||||
[`hermes-example-plugins`](https://github.com/NousResearch/hermes-example-plugins)
|
||||
companion repo, not in this tree.
|
||||
|
||||
### Bot Mode (`apps/desktop/src/plugins/hermes-bots/`)
|
||||
|
||||
The desktop "Bots" experience ships bundled in-tree. Each bot is a Hermes
|
||||
agent **profile** with a persistent identity. Its design rests on one settled
|
||||
invariant that has been regressed repeatedly, cost users real conversation
|
||||
history each time, and is not open for re-litigation in a routine PR:
|
||||
|
||||
**One bot = ONE canonical forever-chat, identified by NAME.** The chat's one
|
||||
and only identity is **(profile, session titled exactly "Bot Chat")** — the
|
||||
state DB's UNIQUE(title) index makes that pair an exact registry of at most
|
||||
one row. The full lifecycle when a bot row is clicked:
|
||||
|
||||
1. **Resolve the registry, every time.** Look up the profile's `Bot Chat`
|
||||
session by exact title via `session.list {title, include_hidden: true}`
|
||||
(indexed, window-free; hidden rows resolve because canonical chats are
|
||||
always hidden; compression lineages resolve to the live tip). Row exists →
|
||||
open it. That is the entire happy path.
|
||||
2. **No row → create it,** titled `Bot Chat`, born hidden, kicked off with
|
||||
the bot's intro. Creation adopts-before-minting: it re-runs the registry
|
||||
lookup first, so a concurrent or pre-existing row is opened, never forked.
|
||||
(`set_session_title` silently drops conflicting titles — returns 0 rows —
|
||||
which is how the 2026-08 infinite fork loop started; adopt-before-mint is
|
||||
what kills it.)
|
||||
|
||||
**There is NO session-id pin.** The previous design stored a pointer in
|
||||
`ui_meta['hermes-bots'].chat` and verified it per click; five hardening
|
||||
waves (#88690, #90732, #90751, the #91791 revert, #92042) each guarded a new
|
||||
way that pointer dangled or got stolen — rows[0] steals, `last_session`
|
||||
adoptions, transient clears, drifted-title welds (a pin re-anchored onto a
|
||||
cron session passed every guard). Name-as-identity removes the failure class:
|
||||
a name cannot dangle, and a corrupted historical pointer simply never gets
|
||||
read. Legacy `chat` keys in ui_meta are ignored and dropped from merges.
|
||||
|
||||
Why recency must never win (the #91791 → #92042 lesson): canonical Bot
|
||||
Chats are **unconditionally hidden** from the Sessions sidebar, so the bot
|
||||
row is the ONLY door to the forever-chat. A "newest visible session wins"
|
||||
preference doesn't re-order two equivalent entry points — it walls the
|
||||
entire relationship off behind a row that previews one session and opens
|
||||
another, and any stray draft that catches a prompt captures the row.
|
||||
Side-chats started via "New chat with this agent" are not plumbing-titled,
|
||||
stay visible in the Sessions sidebar, and are reachable there; they are
|
||||
never the bot row's target.
|
||||
|
||||
Corollaries for reviewers:
|
||||
|
||||
- There is no per-bot session browser, by explicit design (removed in
|
||||
#90732). Do not add one back.
|
||||
- Reject any PR that reintroduces a stored session-id pointer as canonical
|
||||
identity — including "as a fallback tier" or "for verification". The
|
||||
registry lookup is the whole contract; pointers are how every prior
|
||||
incident started.
|
||||
- Reject any PR that consults recency, visibility, or "where the user left
|
||||
off" for the bot row's target — reports that motivate such a change are
|
||||
almost always about side-chats, and the fix belongs in the Sessions
|
||||
sidebar (hide-sweep false positives), not in the bot row's target.
|
||||
- The gateway reports the registry row per profile as `canonical_session`
|
||||
on `profiles.list` (resolved server-side by title); roster preview,
|
||||
activity signals, and the `/new`→`/compact` guard all read it, so preview
|
||||
identity and click identity are the same row by construction.
|
||||
|
||||
Regression tests encoding this contract:
|
||||
`tests/canonical-chat-registry.test.mjs` (includes a tripwire asserting the
|
||||
open path never reads or writes a stored pointer),
|
||||
`tests/canonical-chat-creation.test.mjs`, `tests/hide-bot-chats.test.mjs`,
|
||||
and `tests/tui_gateway/test_profiles_list_canonical_session.py`.
|
||||
|
||||
---
|
||||
|
||||
## Skills
|
||||
@@ -1096,15 +1234,16 @@ kanban task.
|
||||
- **CLI:** `hermes_cli/kanban.py` wires `hermes kanban` with verbs
|
||||
`init`, `create`, `list` (alias `ls`), `show`, `assign`, `link`,
|
||||
`unlink`, `comment`, `attach`, `attachments`, `attach-rm`, `complete`,
|
||||
`block`, `unblock`, `archive`, `tail`, plus less-commonly-used `watch`,
|
||||
`stats`, `runs`, `log`, `assignees`, `heartbeat`, `notify-*`,
|
||||
`dispatch`, `daemon`, `gc`.
|
||||
`request-review`, `request-changes`, `reopen-review`, `block`, `unblock`, `archive`,
|
||||
`tail`, plus less-commonly-used `watch`, `stats`, `runs`, `log`,
|
||||
`assignees`, `heartbeat`, `notify-*`, `dispatch`, `daemon`, `gc`.
|
||||
- **Worker/orchestrator toolset:** `tools/kanban_tools.py` exposes
|
||||
`kanban_show`, `kanban_complete`, `kanban_block`, `kanban_heartbeat`,
|
||||
`kanban_comment`, `kanban_create`, `kanban_link`, `kanban_attach`,
|
||||
`kanban_attach_url`, `kanban_attachments`; profiles that explicitly
|
||||
enable the `kanban` toolset outside a dispatcher-spawned task also get
|
||||
`kanban_list` and `kanban_unblock` for board routing.
|
||||
`kanban_show`, `kanban_complete`, `kanban_request_review`,
|
||||
`kanban_request_changes`, `kanban_block`,
|
||||
`kanban_heartbeat`, `kanban_comment`, `kanban_create`, `kanban_link`,
|
||||
`kanban_attach`, `kanban_attach_url`, `kanban_attachments`; profiles that
|
||||
explicitly enable the `kanban` toolset outside a dispatcher-spawned
|
||||
task also get `kanban_list` and `kanban_unblock` for board routing.
|
||||
- **Dispatcher:** long-lived loop that (default every 60s) reclaims
|
||||
stale claims, promotes ready tasks, atomically claims, and spawns
|
||||
assigned profiles. Runs **inside the gateway** by default via
|
||||
@@ -1128,6 +1267,79 @@ Full user-facing docs: `website/docs/user-guide/features/kanban.md`.
|
||||
|
||||
---
|
||||
|
||||
## Update Pipeline (`hermes update`)
|
||||
|
||||
The updater is transactional in shape (fleet-update campaign, #91277 —
|
||||
Aug 2026). Every stage exists because its absence was a real field
|
||||
failure; PRs that weaken a stage need to answer for the failure class it
|
||||
guards:
|
||||
|
||||
```
|
||||
plan → snapshot → apply → restart-per-kind → verify → report
|
||||
```
|
||||
|
||||
- **Plan** (`hermes_cli/update_inventory.py`, `hermes update --plan`):
|
||||
read-only inventory — install kind, all profiles, every live gateway
|
||||
with supervisor + running code version. Deployment kinds are
|
||||
first-class: `git` updates in place; `docker`/`nix`/`apt` are NOT
|
||||
in-place-updatable and the updater reports the correct external
|
||||
command instead of fighting the deployment model.
|
||||
- **Snapshot** (`hermes_cli/backup.py`): pre-update quick snapshot for
|
||||
EVERY profile (the code swap + fleet restart touch all of them), each
|
||||
into its own `state-snapshots/`, identical file set + 1 GiB per-file
|
||||
cap + keep=1. **Never add a partial/tiered snapshot set** — mixed
|
||||
coverage creates torn-restore states across schema generations. Quick
|
||||
snapshots are FILE-LOSS RECOVERY (the per-profile cron-jobs safety
|
||||
net restores from them), NOT code-rollback insurance; `--backup` full
|
||||
mode owns rollback.
|
||||
- **Apply**: git pull, or the Windows ZIP fallback — which fires ONLY
|
||||
when git itself failed (`_should_zip_fallback_on_update_error`,
|
||||
argv-classified; a dependency-install failure must never trigger a
|
||||
tree-clobbering re-download), REFUSES a dirty working tree
|
||||
(`-uall`, plus a pre-swap TOCTOU re-check), and grafts the live
|
||||
`apps/desktop/release/` into the staged swap (the GitHub source ZIP
|
||||
has no built desktop app; without the graft the swap deletes it).
|
||||
- **Restart-per-kind**: systemd and launchd restarts are FLEET-WIDE
|
||||
(every `hermes-gateway*` unit / `ai.hermes.gateway*` LaunchAgent),
|
||||
drain-first (SIGUSR1) with per-unit/per-label failure isolation.
|
||||
Restarting only the invoking profile's service leaves siblings on
|
||||
stale `sys.modules` until they crash — the largest dupe-PR cluster in
|
||||
the repo's history came from that bug.
|
||||
- **Verify**: gateways stamp their running `code_sha`/`code_version`
|
||||
into `gateway_state.json` on every runtime-status write
|
||||
(`gateway/status.py`); after the restart phase the updater compares
|
||||
each live gateway against the fresh checkout and prints a fleet
|
||||
version matrix. A provably-stale gateway fails the update (exit 1) —
|
||||
automation must never treat a mixed-version fleet as healthy.
|
||||
- **Report**: every run writes a machine-readable receipt to
|
||||
`~/.hermes/logs/update_receipts/` (`latest.json` pointer; steps,
|
||||
skips WITH reasons, restart outcome, plan, fleet snapshot).
|
||||
Finalization is owned by the `cmd_update` command boundary — early
|
||||
`sys.exit` paths (preflight refusals, fetch failures) still persist
|
||||
a receipt with the real exit code. A begun-but-unwritten receipt is
|
||||
a bug: the refused/failed runs are the ones receipts exist for.
|
||||
|
||||
Architecture direction: process-scan-based coordination between the
|
||||
updater, serve/dashboard, and the gateway is being replaced by a
|
||||
gateway-owned control socket (#92091). Do not add new scan heuristics
|
||||
without checking that design; scans are the fallback layer.
|
||||
|
||||
### Gateway lifecycle vs. the Desktop app
|
||||
|
||||
`hermes serve` (control plane, desktop-spawned child) dies with the app
|
||||
— by design. The messaging gateway (`gateway run`) SURVIVES the app: the
|
||||
serve backend's `/api/gateway/*` endpoints spawn it detached
|
||||
(`_spawn_hermes_action` — `start_new_session` / `DETACHED_PROCESS`), so
|
||||
`before-quit`'s backend SIGTERM never reaches it. Bots keep running
|
||||
when the user closes the app. The known breach of this contract is the
|
||||
Windows shim-unlock teardown (`taskkill /T /F` on venv-shim holders,
|
||||
#85265) — it exists to let updates proceed, and its replacement is
|
||||
#92091's `pause-for-update`. Do not "fix" gateway-dies-with-app reports
|
||||
by re-parenting the gateway under the backend, and do not "fix" update
|
||||
locks by widening the tree-kill.
|
||||
|
||||
---
|
||||
|
||||
## Important Policies
|
||||
|
||||
### Prompt Caching Must Not Break
|
||||
@@ -1151,9 +1363,10 @@ detects process completion and triggers a new agent turn. Control verbosity of b
|
||||
messages with `display.background_process_notifications`
|
||||
in config.yaml (or `HERMES_BACKGROUND_NOTIFICATIONS` env var):
|
||||
|
||||
- `all` — running-output updates + final message (default)
|
||||
- `result` — only the final completion message
|
||||
- `error` — only the final message when exit code != 0
|
||||
- `concise` — one-line status message on completion; failures append a short output tail (default)
|
||||
- `all` — running-output updates + final raw-output message
|
||||
- `result` — only the final raw-output completion message
|
||||
- `error` — only the final raw-output message when exit code != 0
|
||||
- `off` — no watcher messages at all
|
||||
|
||||
---
|
||||
@@ -1214,19 +1427,60 @@ automatically scope to the active profile.
|
||||
This is intentional — it lets `hermes -p coder profile list` see all profiles regardless
|
||||
of which one is active.
|
||||
|
||||
7. **Multiplex profile-scoped env reads MUST fail closed — never borrow from `os.environ`**
|
||||
(`agent/secret_scope.py` contract; #72348, #86905). Under `gateway.multiplex_profiles`,
|
||||
`os.environ` holds the **default profile's** values; a secondary profile's `.env` lives
|
||||
only in its secret scope (installed per-turn by `_profile_runtime_scope`). Any
|
||||
profile-level env config — credentials (`app_secret`, tokens) AND authorization
|
||||
(`FEISHU_ALLOWED_USERS`, `{PLATFORM}_ALLOW_ALL_USERS`, `GATEWAY_ALLOW_ALL_USERS`,
|
||||
`group_policy`, `allow_bots`, ...) — must be read scope-aware:
|
||||
- Adapters: `_get_scoped_secret()` (canonical fail-closed copy in
|
||||
`plugins/platforms/feishu/adapter.py`, #86905).
|
||||
- Gateway authz: `_auth_env()` / `_platform_gate_env()` (`gateway/authz_mixin.py`).
|
||||
Rules:
|
||||
- Scope installed + multiplex active → a scoped miss returns the **default**.
|
||||
NEVER fall through to `os.environ` — that leaks another profile's value and
|
||||
silently breaks routing/admission (a leaked default allowlist skips the
|
||||
allow-all check and rejects every secondary-profile sender, #86905).
|
||||
- Unscoped default-profile path (`UnscopedSecretError`) and single-profile
|
||||
deployments keep the `os.environ` read — there it IS the profile's own value.
|
||||
- Authorization config is the sharpest edge: allowlist/allow-all leaks cause
|
||||
silent rejections (or worse, fail-open) that only show up as missing replies.
|
||||
- The `_get_scoped_secret` wrapper is copy-pasted across ~15 platform adapters —
|
||||
when touching any of them, make sure the fail-closed semantics are present;
|
||||
do not reintroduce the `except _UnscopedSecretError: val = os.getenv(...)`
|
||||
fallback-after-miss shape.
|
||||
|
||||
## Known Pitfalls
|
||||
|
||||
### DO NOT infer process identity from argv substrings
|
||||
The bug class behind ~10 fleet-update issues (#90778, #87594, #78089,
|
||||
#76129, #91964, ...): classifying a process by `"serve" in cmdline` or
|
||||
similar. `kanban --preserve-cache` contains "serve"; a flag VALUE can
|
||||
equal a subcommand (`-m dashboard serve`); truncated cmdlines hide the
|
||||
real subcommand. Rules:
|
||||
- Use the canonical matchers: `gateway.status.looks_like_gateway_command_line`
|
||||
(gateway run), `hermes_cli.update_cmd._hermes_holder_subcommand`
|
||||
(top-level subcommand of any Hermes argv). Never hand-roll token scans.
|
||||
- Flag sets must be DERIVED from the parser
|
||||
(`_holder_value_flags()` introspects `build_top_level_parser()`), never
|
||||
hand-written lists — they drift.
|
||||
- Never blanket-exclude ancestors from process scans: when `/update` runs
|
||||
as the gateway's child, a gateway ancestor must stay visible to the
|
||||
pause machinery (#87594). Exclude interactive ancestry, carve out
|
||||
gateway-shaped ancestors.
|
||||
- Match on FULL cmdlines; truncate only at display time (#78089).
|
||||
- Before adding any new scan heuristic, read #92091 — the gateway control
|
||||
socket replaces scans as the primary coordination mechanism; scans are
|
||||
the fallback layer for old/crashed processes.
|
||||
|
||||
### DO NOT hardcode `~/.hermes` paths
|
||||
Use `get_hermes_home()` from `hermes_constants` for code paths. Use `display_hermes_home()`
|
||||
for user-facing print/log messages. Hardcoding `~/.hermes` breaks profiles — each profile
|
||||
has its own `HERMES_HOME` directory. This was the source of 5 bugs fixed in PR #3575.
|
||||
|
||||
### DO NOT introduce new `simple_term_menu` usage
|
||||
Existing call sites in `hermes_cli/main.py` remain for legacy fallback only;
|
||||
the preferred UI is curses (stdlib) because `simple_term_menu` has
|
||||
ghost-duplication rendering bugs in tmux/iTerm2 with arrow keys. New
|
||||
interactive menus must use `hermes_cli/curses_ui.py` — see
|
||||
`hermes_cli/tools_config.py` for the canonical pattern.
|
||||
### All CLI menu-pickers MUST use curses.
|
||||
Interactive menus must use `hermes_cli/curses_ui.py`. See `hermes_cli/tools_config.py` for an example.
|
||||
|
||||
### DO NOT use `\033[K` (ANSI erase-to-EOL) in spinner/display code
|
||||
Leaks as literal `?[K` text under `prompt_toolkit`'s `patch_stdout`. Use space-padding: `f"\r{line}{' ' * pad}"`.
|
||||
@@ -1248,6 +1502,47 @@ while the agent is blocked (e.g. approval prompts) MUST bypass BOTH
|
||||
guards and be dispatched inline, not via `_process_message_background()`
|
||||
(which races session lifecycle).
|
||||
|
||||
### Streaming delivery contract (stream-is-the-message adapters) — duplicate-final class
|
||||
Adapters with `draft_stream_is_message = True` (relay Slack native streaming)
|
||||
keep ONE cumulative native stream per turn; the stream IS the final message.
|
||||
Four invariants, each learned from a live duplicate-final incident (NS-658
|
||||
canary ledger, hermes#85796 / gateway-gateway#210). Violating any of them
|
||||
re-creates a duplicate or a frozen stream:
|
||||
|
||||
1. **Draft frames must be prefix-stable.** The connector computes append-only
|
||||
deltas: frame N must be a string prefix of frame N+1. NEVER mutate draft
|
||||
frames per-tick — no fence-closing (`ensure_closed_code_fences`), no cursor
|
||||
suffix, no segment-state resets at tool boundaries, no mrkdwn conversion.
|
||||
Any non-prefix frame triggers a whole-snapshot re-append on the platform
|
||||
("stacked copies"). The finalize path may still transform the real final.
|
||||
2. **The consumer declares the final; the adapter never guesses.**
|
||||
`finish(final_text)` carries the completed `final_response` (verifier
|
||||
footer, completion explainer included) as the authoritative finalize
|
||||
payload. New post-stream response augmentation MUST ride this payload —
|
||||
if it mutates `final_response` after the stream sealed, it re-opens the
|
||||
#11 bug (`delivered_final_matches` mismatch → corrective duplicate send).
|
||||
3. **Interim sends must carry `_interim_send` metadata.** Any consumer-side
|
||||
`adapter.send()` that is NOT the turn-final (commentary, segment-tail
|
||||
flushes) must set `metadata["_interim_send"] = True`, or the relay
|
||||
adapter's seal-interception will seal the live stream with interim text.
|
||||
Seal-interception exists at BOTH egress doors (`send()` AND
|
||||
`send_for_platform()`); a new egress door needs the same two checks.
|
||||
4. **Reconcile by edit, never by plain send.** Any lane that delivers a final
|
||||
beside an already-sealed stream (queued follow-ups, media-accompanied
|
||||
finals, future lanes) must first try `edit_message` on the consumer's
|
||||
`message_id`; plain `send()` is the fallback only when no editable message
|
||||
exists. A sealed native stream is a regular message — `chat.update` on it
|
||||
works (live-verified).
|
||||
|
||||
Contract tests: `tests/gateway/test_stream_final_contract.py` (all four
|
||||
invariants, mutation-checked). Slack streaming API ground truth (live-probed,
|
||||
also encoded in connector comments/tests): `chat.*Stream` speaks STANDARD
|
||||
markdown, not mrkdwn; `stopStream.markdown_text` APPENDS (never replaces);
|
||||
`startStream`/`stopStream` are rate-limit Tier 2 (~20/min).
|
||||
|
||||
Guard style note: check `draft_stream_is_message` with `is True` — MagicMock
|
||||
adapters in older tests auto-create truthy attributes.
|
||||
|
||||
### Squash merges from stale branches silently revert recent fixes
|
||||
Before squash-merging a PR, ensure the branch is up to date with `main`
|
||||
(`git fetch origin main && git reset --hard origin/main` in the worktree,
|
||||
@@ -1284,14 +1579,15 @@ def profile_env(tmp_path, monkeypatch):
|
||||
### Python
|
||||
**ALWAYS use `scripts/run_tests.sh`** — do not call `pytest` directly. The script enforces
|
||||
hermetic environment parity with CI (unset credential vars, TZ=UTC, LANG=C.UTF-8,
|
||||
`-n auto` xdist workers, in-tree subprocess-isolation plugin). Direct `pytest`
|
||||
per-file subprocess isolation via `scripts/run_tests_parallel.py` — no xdist,
|
||||
worker count auto-scaled from CPU count). Direct `pytest`
|
||||
on a 16+ core developer machine with API keys set diverges from CI in ways
|
||||
that have caused multiple "works locally, fails in CI" incidents (and the reverse).
|
||||
|
||||
```bash
|
||||
scripts/run_tests.sh # full suite, CI-parity
|
||||
scripts/run_tests.sh tests/gateway/ # one directory
|
||||
scripts/run_tests.sh tests/agent/test_foo.py::test_x # one test
|
||||
scripts/run_tests.sh tests/agent/test_foo.py -k test_x # one test (file + -k; the runner is file-granular)
|
||||
scripts/run_tests.sh -v --tb=long # pass-through pytest flags
|
||||
```
|
||||
|
||||
@@ -1329,6 +1625,60 @@ Any test that reads or asserts about `package.json`,
|
||||
`package-lock.json`, `tsconfig.json`, `.ts`/`.tsx`/`.js`/`.mjs`/`.cjs`
|
||||
source files configuration belongs in the JS (vitest) test suite, not in `tests/*.py`.
|
||||
|
||||
### Don't fake the host OS
|
||||
|
||||
Hermes supports Linux, macOS and native Windows, and plenty of its behaviour
|
||||
genuinely differs per host. Those differences are tested by running on the
|
||||
host, not by patching `sys.platform`.
|
||||
|
||||
```python
|
||||
@pytest.mark.linux_only
|
||||
@pytest.mark.macos_only
|
||||
@pytest.mark.windows_only
|
||||
```
|
||||
|
||||
Things that are host-independent can stay unmarked:
|
||||
|
||||
- **Pure functions that take a platform as data** —
|
||||
`hidden_windows_child_options(opts, is_windows=True)` is input→output, not a
|
||||
fake host. (Contrast: setting a module-level `IS_WINDOWS` flag and then
|
||||
calling `windows_detach_flags()` *is* a fake.)
|
||||
- **Declaration/packaging invariants** — "pyproject declares `tzdata` with a
|
||||
`sys_platform == 'win32'` marker" asserts about a file, not about runtime.
|
||||
|
||||
The line: **if the test needs the interpreter to believe it is on another OS
|
||||
in order to pass, it belongs on that OS.**
|
||||
When one test body walks several platforms in sequence, split it.
|
||||
Keep the host-native arm on the Linux lane and move the other arm into its own marked test.
|
||||
|
||||
**Live Windows process-topology E2E: the `wine2e` lane.** For claims about
|
||||
real Windows process behavior that mocks cannot reproduce (venv-holder
|
||||
scans, process-tree parentage, launcher/worker chains, detach semantics),
|
||||
there is an on-demand workflow `windows-venv-e2e.yml` that runs
|
||||
`tests/hermes_cli/test_venv_holder_windows_live.py` on a real
|
||||
`windows-latest` runner — spawning actual processes and driving the real
|
||||
detection code, no mocked psutil. It fires ONLY on pushes to `wine2e/**`
|
||||
branches (inert on PRs and main; costs nothing on normal work). The proven
|
||||
workflow: write probes that pin CORRECT behavior, push to a `wine2e/`
|
||||
branch to reproduce the bugs live on unfixed code, build the fix, iterate
|
||||
until the lane is green, then open the PR — the live receipt on the exact
|
||||
head is the Windows proof reviewers ask for. Extend the live suite when
|
||||
touching that subsystem; assert against the gateway ANCESTOR found by
|
||||
argv, not the direct parent (the venv shim makes every spawn a
|
||||
launcher/worker chain).
|
||||
|
||||
**Use the marker, never a bare `skipif`.** `scripts/ci/list_os_marked_tests.py`
|
||||
decides which files the macOS/Windows lanes import by grepping for the marker
|
||||
*name*, and the lane then filters with `-m <marker>`. A test gated with
|
||||
`@pytest.mark.skipif(sys.platform != "win32")` therefore skips on Linux AND is
|
||||
never imported on the Windows lane — it runs on no host at all, silently. The
|
||||
same trap catches a file-local alias (`windows_only = pytest.mark.skipif(...)`):
|
||||
the grep matches the name, so the file *is* listed, but `-m windows_only`
|
||||
deselects every test in it and the lane reports green over zero coverage.
|
||||
Equally, don't `pytest.skip()` the non-host rows of a `@parametrize` over
|
||||
platforms — split it into one marked test per OS, or only the host's row ever
|
||||
executes.
|
||||
|
||||
### Don't write change-detector tests
|
||||
|
||||
A test is a **change-detector** if it fails whenever data that is **expected
|
||||
|
||||
@@ -582,7 +582,7 @@ test(tools): añadir tests unitarios para file_operations
|
||||
## Reportar Issues
|
||||
|
||||
- Usa [GitHub Issues](https://github.com/NousResearch/hermes-agent/issues)
|
||||
- Incluye: SO, versión de Python, versión de Hermes (`hermes version`), traza de error completa
|
||||
- Incluye: SO, versión de Python, versión de Hermes (`hermes --version`), traza de error completa
|
||||
- Incluye pasos para reproducir
|
||||
- Verifica los issues existentes antes de crear duplicados
|
||||
- Para vulnerabilidades de seguridad, por favor reporta de forma privada
|
||||
|
||||
@@ -130,7 +130,7 @@ cd "${HERMES_HOME:-$HOME/.hermes}/hermes-agent"
|
||||
# Add dev/test extras on top of the standard install.
|
||||
uv pip install -e ".[all,dev]"
|
||||
|
||||
# Optional: browser tools / docs site dependencies.
|
||||
# Optional: docs site + workspace dependencies.
|
||||
npm install
|
||||
```
|
||||
|
||||
@@ -167,7 +167,7 @@ export PATH="$VIRTUAL_ENV/bin:$PATH"
|
||||
# Install with all extras (messaging, cron, CLI menus, dev tools)
|
||||
uv pip install -e ".[all,dev]"
|
||||
|
||||
# Optional: browser tools
|
||||
# Optional: workspace / docs dependencies
|
||||
npm install
|
||||
```
|
||||
|
||||
@@ -201,7 +201,8 @@ ln -sf "$(pwd)/venv/bin/hermes" ~/.local/bin/hermes
|
||||
### Run tests
|
||||
|
||||
```bash
|
||||
# Preferred — matches CI (hermetic env, 4 xdist workers); see AGENTS.md
|
||||
# Preferred — matches CI (hermetic `env -i`, per-file subprocess isolation
|
||||
# via run_tests_parallel.py, worker count auto-scaled); see AGENTS.md
|
||||
scripts/run_tests.sh
|
||||
|
||||
# Alternative (activate the venv first). The wrapper is still recommended
|
||||
@@ -722,22 +723,9 @@ that touches the OS, assume *any* platform can hit your code path.
|
||||
For process enumeration: PowerShell's `Get-CimInstance Win32_Process` is
|
||||
the modern replacement for `wmic process`. See
|
||||
`hermes_cli/gateway.py::_scan_gateway_pids` for the pattern.
|
||||
|
||||
3. **`termios` and `fcntl` are Unix-only.** Always catch both `ImportError`
|
||||
and `NotImplementedError`:
|
||||
```python
|
||||
try:
|
||||
from simple_term_menu import TerminalMenu
|
||||
menu = TerminalMenu(options)
|
||||
idx = menu.show()
|
||||
except (ImportError, NotImplementedError):
|
||||
# Fallback: numbered menu for Windows
|
||||
for i, opt in enumerate(options):
|
||||
print(f" {i+1}. {opt}")
|
||||
idx = int(input("Choice: ")) - 1
|
||||
```
|
||||
|
||||
4. **File encoding.** Windows may save `.env` files in `cp1252`. Always
|
||||
3. **File encoding.** Windows may save `.env` files in `cp1252`. Always
|
||||
handle encoding errors:
|
||||
```python
|
||||
try:
|
||||
@@ -749,7 +737,7 @@ that touches the OS, assume *any* platform can hit your code path.
|
||||
similar editors — use `encoding="utf-8-sig"` when reading files that
|
||||
could have been touched by a Windows GUI editor.
|
||||
|
||||
5. **Process management.** `os.setsid()`, `os.killpg()`, `os.fork()`,
|
||||
4. **Process management.** `os.setsid()`, `os.killpg()`, `os.fork()`,
|
||||
`os.getuid()`, and POSIX signal handling differ on Windows. Guard with
|
||||
`platform.system()`, `sys.platform`, or `hasattr(os, "setsid")`:
|
||||
```python
|
||||
@@ -773,29 +761,29 @@ that touches the OS, assume *any* platform can hit your code path.
|
||||
pass
|
||||
```
|
||||
|
||||
6. **Signals that don't exist on Windows: `SIGALRM`, `SIGCHLD`, `SIGHUP`,
|
||||
5. **Signals that don't exist on Windows: `SIGALRM`, `SIGCHLD`, `SIGHUP`,
|
||||
`SIGUSR1`, `SIGUSR2`, `SIGPIPE`, `SIGQUIT`, `SIGKILL`.** Python's
|
||||
`signal` module raises `AttributeError` at import time if you reference
|
||||
them on Windows. Use `getattr(signal, "SIGKILL", signal.SIGTERM)` or
|
||||
gate the whole block behind a platform check. `loop.add_signal_handler`
|
||||
raises `NotImplementedError` on Windows — always catch it.
|
||||
|
||||
7. **Path separators.** Use `pathlib.Path` instead of string concatenation
|
||||
6. **Path separators.** Use `pathlib.Path` instead of string concatenation
|
||||
with `/`. Forward slashes work almost everywhere on Windows, but
|
||||
`subprocess.run(["cmd.exe", "/c", ...])` and other shell contexts can
|
||||
require backslashes — convert with `str(path)` at the subprocess boundary,
|
||||
not inside Python logic.
|
||||
|
||||
8. **Symlinks need elevated privileges on Windows** (unless Developer Mode is
|
||||
7. **Symlinks need elevated privileges on Windows** (unless Developer Mode is
|
||||
on). Tests that create symlinks need `@pytest.mark.skipif(sys.platform ==
|
||||
"win32", reason="Symlinks require elevated privileges on Windows")`.
|
||||
|
||||
9. **POSIX file modes (0o600, 0o644, etc.) are NOT enforced on NTFS** by
|
||||
8. **POSIX file modes (0o600, 0o644, etc.) are NOT enforced on NTFS** by
|
||||
default. Tests that assert on `stat().st_mode & 0o777` must skip on
|
||||
Windows — the concept doesn't translate. Use ACLs (`icacls`, `pywin32`)
|
||||
for Windows secret-file protection if needed.
|
||||
|
||||
10. **Detached background daemons on Windows need `pythonw.exe`, NOT
|
||||
9. **Detached background daemons on Windows need `pythonw.exe`, NOT
|
||||
`python.exe`.** `python.exe` always allocates or attaches to a console,
|
||||
which makes it vulnerable to `CTRL_C_EVENT` broadcasts from any sibling
|
||||
process. `pythonw.exe` is the no-console variant. Combine with
|
||||
@@ -804,38 +792,38 @@ that touches the OS, assume *any* platform can hit your code path.
|
||||
See `hermes_cli/gateway_windows.py::_spawn_detached` for the reference
|
||||
implementation.
|
||||
|
||||
11. **`subprocess.Popen` with `.cmd` or `.bat` shims needs `shutil.which`
|
||||
10. **`subprocess.Popen` with `.cmd` or `.bat` shims needs `shutil.which`
|
||||
to resolve.** Passing `"agent-browser"` to `Popen` on Windows finds
|
||||
the extensionless POSIX shebang shim in `node_modules/.bin/`, which
|
||||
`CreateProcessW` can't execute — you'll get `WinError 193 "not a valid
|
||||
Win32 application"`. Use `shutil.which("agent-browser", path=local_bin)`
|
||||
which honors PATHEXT and picks the `.CMD` variant on Windows.
|
||||
|
||||
12. **Don't use shell shebangs as a way to run Python.** `#!/usr/bin/env
|
||||
11. **Don't use shell shebangs as a way to run Python.** `#!/usr/bin/env
|
||||
python` only works when the file is executed through a Unix shell.
|
||||
`subprocess.run(["./myscript.py"])` on Windows fails even if the file
|
||||
has a shebang line. Always invoke Python explicitly:
|
||||
`[sys.executable, "myscript.py"]`.
|
||||
|
||||
13. **Shell commands in installers.** If you change `scripts/install.sh`,
|
||||
12. **Shell commands in installers.** If you change `scripts/install.sh`,
|
||||
make the equivalent change in `scripts/install.ps1`. The two scripts
|
||||
are the canonical example of "works on Linux does not mean works on
|
||||
Windows" and have drifted multiple times — keep them in lockstep.
|
||||
|
||||
14. **Known paths that are OneDrive-redirected on Windows:** Desktop,
|
||||
13. **Known paths that are OneDrive-redirected on Windows:** Desktop,
|
||||
Documents, Pictures, Videos. The "real" path when OneDrive Backup is
|
||||
enabled is `%USERPROFILE%\OneDrive\Desktop` (etc.), NOT
|
||||
`%USERPROFILE%\Desktop` (which exists as an empty husk). Resolve the
|
||||
real location via `ctypes` + `SHGetKnownFolderPath` or by reading the
|
||||
`Shell Folders` registry key — never assume `~/Desktop`.
|
||||
|
||||
15. **CRLF vs LF in generated scripts.** Windows `cmd.exe` and `schtasks`
|
||||
14. **CRLF vs LF in generated scripts.** Windows `cmd.exe` and `schtasks`
|
||||
parse line-by-line; mixed or LF-only line endings can break multi-line
|
||||
`.cmd` / `.bat` files. Use `open(path, "w", encoding="utf-8",
|
||||
newline="\r\n")` — or `open(path, "wb")` + explicit bytes — when
|
||||
generating scripts Windows will execute.
|
||||
|
||||
16. **Two different quoting schemes in one command line.** `subprocess.run
|
||||
15. **Two different quoting schemes in one command line.** `subprocess.run
|
||||
(["schtasks", "/TR", some_cmd])` → schtasks itself parses `/TR`, AND
|
||||
the `some_cmd` string is re-parsed by `cmd.exe` when the task fires.
|
||||
Different parsers, different escape rules. Use two separate quoting
|
||||
@@ -845,18 +833,15 @@ that touches the OS, assume *any* platform can hit your code path.
|
||||
|
||||
### Testing cross-platform
|
||||
|
||||
Tests that use POSIX-only syscalls need a skip marker. Common ones:
|
||||
- Symlinks → `@pytest.mark.skipif(sys.platform == "win32", ...)`
|
||||
- `0o600` file modes → `@pytest.mark.skipif(sys.platform.startswith("win"), ...)`
|
||||
- `signal.SIGALRM` → Unix-only (see `tests/conftest.py::_enforce_test_timeout`)
|
||||
- `os.setsid` / `os.fork` → Unix-only
|
||||
- Live Winsock / Windows-specific regression tests →
|
||||
`@pytest.mark.skipif(sys.platform != "win32", reason="Windows-specific regression")`
|
||||
Tests that excercise behavior on specific platforms must run on their target platforms.
|
||||
|
||||
If you monkeypatch `sys.platform` for cross-platform tests, also patch
|
||||
`platform.system()` / `platform.release()` / `platform.mac_ver()` — each
|
||||
re-reads the real OS independently, so half-patched tests still route
|
||||
through the wrong branch on a Windows runner.
|
||||
```python
|
||||
@pytest.mark.linux_only
|
||||
@pytest.mark.macos_only
|
||||
@pytest.mark.windows_only
|
||||
```
|
||||
Avoid monkeypatching `sys.platform` unless absolutely needed, but if you do, also patch `platform.system()` / `platform.release()` / `platform.mac_ver()`.
|
||||
Symlinks, 0o600 permissions, SIGALRM, os.setsid/fork are all unix-only.
|
||||
|
||||
---
|
||||
|
||||
@@ -988,7 +973,7 @@ test(tools): add unit tests for file_operations
|
||||
## Reporting Issues
|
||||
|
||||
- Use [GitHub Issues](https://github.com/NousResearch/hermes-agent/issues)
|
||||
- Include: OS, Python version, Hermes version (`hermes version`), full error traceback
|
||||
- Include: OS, Python version, Hermes version (`hermes --version`), full error traceback
|
||||
- Include steps to reproduce
|
||||
- Check existing issues before creating duplicates
|
||||
- For security vulnerabilities, please report privately
|
||||
|
||||
@@ -1,12 +1,54 @@
|
||||
# Debian 13 still ships SQLite 3.46.1, which contains the upstream WAL-reset
|
||||
# corruption bug. Build a pinned shared library for the runtime image instead
|
||||
# of relying on a distro backport that trixie does not currently provide.
|
||||
# See #70480 and https://sqlite.org/wal.html#walresetbug.
|
||||
FROM debian:13.4 AS sqlite_build
|
||||
ARG SQLITE_AUTOCONF_VERSION=3530400
|
||||
ARG SQLITE_SHA256=0e9483900e92cd5de8fd48d16bf9200145a61f7fd5be542a5ac81d8a9516eb9c
|
||||
RUN apt-get -o Acquire::Retries=3 update && \
|
||||
apt-get -o Acquire::Retries=3 install -y --no-install-recommends \
|
||||
build-essential ca-certificates curl && \
|
||||
rm -rf /var/lib/apt/lists/* && \
|
||||
(curl -fsSL --retry 1 --retry-all-errors --connect-timeout 15 --max-time 60 \
|
||||
-o /tmp/sqlite.tar.gz \
|
||||
"https://sqlite.org/2026/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}.tar.gz" || \
|
||||
curl -fsSL --retry 3 --retry-all-errors --connect-timeout 15 --max-time 120 \
|
||||
-o /tmp/sqlite.tar.gz \
|
||||
"https://sources.buildroot.net/sqlite/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}.tar.gz") && \
|
||||
printf '%s %s\n' "${SQLITE_SHA256}" /tmp/sqlite.tar.gz > /tmp/sqlite.sha256 && \
|
||||
sha256sum -c /tmp/sqlite.sha256 && \
|
||||
tar -xzf /tmp/sqlite.tar.gz -C /tmp && \
|
||||
cd "/tmp/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}" && \
|
||||
CFLAGS="-O2 \
|
||||
-DSQLITE_ENABLE_FTS3 \
|
||||
-DSQLITE_ENABLE_FTS3_PARENTHESIS \
|
||||
-DSQLITE_ENABLE_FTS4 \
|
||||
-DSQLITE_ENABLE_FTS5 \
|
||||
-DSQLITE_ENABLE_RTREE \
|
||||
-DSQLITE_ENABLE_GEOPOLY \
|
||||
-DSQLITE_ENABLE_COLUMN_METADATA \
|
||||
-DSQLITE_ENABLE_UNLOCK_NOTIFY \
|
||||
-DSQLITE_ENABLE_DBSTAT_VTAB \
|
||||
-DSQLITE_ENABLE_DBPAGE_VTAB \
|
||||
-DSQLITE_ENABLE_MATH_FUNCTIONS \
|
||||
-DSQLITE_ENABLE_PREUPDATE_HOOK \
|
||||
-DSQLITE_ENABLE_SESSION \
|
||||
-DSQLITE_SECURE_DELETE \
|
||||
-DSQLITE_THREADSAFE=1 \
|
||||
-DSQLITE_MAX_VARIABLE_NUMBER=250000" \
|
||||
./configure --prefix=/opt/sqlite-fixed --disable-static && \
|
||||
make -j"$(nproc)" && \
|
||||
make install
|
||||
|
||||
FROM ghcr.io/astral-sh/uv:0.11.6-python3.13-trixie@sha256:b3c543b6c4f23a5f2df22866bd7857e5d304b67a564f4feab6ac22044dde719b AS uv_source
|
||||
# Node 22 LTS source stage. Debian trixie's bundled nodejs is pinned to 20.x
|
||||
# which reached EOL in April 2026 — we copy node + npm + corepack from the
|
||||
# upstream node:22 image instead so we can stay on a supported LTS without
|
||||
# waiting for Debian 14 (forky, ~mid-2027). Bookworm-based slim image used
|
||||
# so the produced binary links against glibc 2.36, which runs cleanly on
|
||||
# our Debian 13 (trixie, glibc 2.41) runtime. Bumping to a new Node major
|
||||
# is a one-line ARG change; see #4977.
|
||||
FROM node:22-bookworm-slim@sha256:7af03b14a13c8cdd38e45058fd957bf00a72bbe17feac43b1c15a689c029c732 AS node_source
|
||||
# Node 26 source stage. Debian trixie's bundled nodejs is pinned to 20.x
|
||||
# which reached EOL in April 2026 — we copy node + npm from the upstream
|
||||
# node:26 image instead (Hermes pins its toolchain to Node 26 everywhere).
|
||||
# Bookworm-based slim image used so the produced binary links
|
||||
# against glibc 2.36, which runs cleanly on our Debian 13 (trixie, glibc
|
||||
# 2.41) runtime. Bumping to a new Node major is a one-line ARG change; see
|
||||
# #4977.
|
||||
FROM node:26-bookworm-slim@sha256:9e6f9357d371591e32ab6f2d8a26d63bdd0d17c29eee3f4f3e7e454d9634bf73 AS node_source
|
||||
FROM debian:13.4
|
||||
|
||||
# Disable Python stdout buffering to ensure logs are printed immediately.
|
||||
@@ -28,9 +70,26 @@ ENV PLAYWRIGHT_BROWSERS_PATH=/opt/hermes/.playwright
|
||||
# hermes process, the dashboard, and per-profile gateways.
|
||||
RUN apt-get -o Acquire::Retries=3 update && \
|
||||
apt-get -o Acquire::Retries=3 install -y --no-install-recommends \
|
||||
ca-certificates curl iputils-ping python3 python-is-python3 ripgrep ffmpeg gcc g++ make cmake python3-dev python3-venv libffi-dev libolm-dev procps git openssh-client docker-cli xz-utils && \
|
||||
ca-certificates curl iputils-ping python3 python-is-python3 ripgrep ffmpeg gcc g++ make cmake python3-dev python3-venv libffi-dev libolm-dev libatomic1 procps git openssh-client docker-cli xz-utils && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Prefer the fixed SQLite over Debian's vulnerable libsqlite3.so.0. Keep the
|
||||
# public library name stable so both the system interpreter and the uv-created
|
||||
# venv resolve the replacement without changing Python import paths.
|
||||
COPY --from=sqlite_build /opt/sqlite-fixed/lib/libsqlite3.so.3.53.4 /usr/local/lib/
|
||||
RUN ln -sf libsqlite3.so.3.53.4 /usr/local/lib/libsqlite3.so.0 && \
|
||||
ln -sf libsqlite3.so.3.53.4 /usr/local/lib/libsqlite3.so && \
|
||||
printf '/usr/local/lib\n' > /etc/ld.so.conf.d/000-sqlite-fixed.conf && \
|
||||
ldconfig && \
|
||||
python3 -c "import sqlite3, sys; \
|
||||
v = sqlite3.sqlite_version_info; \
|
||||
sys.exit(f'linked SQLite {sqlite3.sqlite_version} still has the WAL-reset bug') if v < (3, 51, 3) else None; \
|
||||
db = sqlite3.connect(':memory:'); \
|
||||
db.execute(\"CREATE VIRTUAL TABLE docs USING fts5(content, tokenize='trigram')\"); \
|
||||
db.execute(\"INSERT INTO docs VALUES ('hermes')\"); \
|
||||
sys.exit('SQLite FTS5 trigram self-test failed') if db.execute(\"SELECT count(*) FROM docs WHERE docs MATCH 'erm'\").fetchone()[0] != 1 else None; \
|
||||
db.close()"
|
||||
|
||||
# ---------- s6-overlay install ----------
|
||||
# s6-overlay provides supervision for the main hermes process, the dashboard,
|
||||
# and per-profile gateways. /init becomes PID 1 below — see ENTRYPOINT.
|
||||
@@ -92,17 +151,20 @@ RUN useradd -u 10000 -m -d /opt/data hermes
|
||||
|
||||
COPY --chmod=0755 --from=uv_source /usr/local/bin/uv /usr/local/bin/uvx /usr/local/bin/
|
||||
|
||||
# Node 22 LTS: copy the node binary plus the bundled npm + corepack JS
|
||||
# installs from the upstream image. npm and npx are recreated as symlinks
|
||||
# because they're symlinks in the source image (and need to live on PATH).
|
||||
# Node 26: copy the node binary plus the bundled npm JS install from the
|
||||
# upstream image. npm and npx are recreated as symlinks because they're
|
||||
# symlinks in the source image (and need to live on PATH).
|
||||
#
|
||||
# No corepack: Node unbundled it upstream, so node:26 ships only npm in
|
||||
# /usr/local/lib/node_modules. Nothing here needs it — no package.json
|
||||
# declares a `packageManager`, and no build step shells out to yarn or pnpm.
|
||||
#
|
||||
# See node_source stage at the top of the file for the version-bump
|
||||
# rationale (#4977).
|
||||
COPY --chmod=0755 --from=node_source /usr/local/bin/node /usr/local/bin/
|
||||
COPY --from=node_source /usr/local/lib/node_modules/npm /usr/local/lib/node_modules/npm
|
||||
COPY --from=node_source /usr/local/lib/node_modules/corepack /usr/local/lib/node_modules/corepack
|
||||
RUN ln -sf /usr/local/lib/node_modules/npm/bin/npm-cli.js /usr/local/bin/npm && \
|
||||
ln -sf /usr/local/lib/node_modules/npm/bin/npx-cli.js /usr/local/bin/npx && \
|
||||
ln -sf /usr/local/lib/node_modules/corepack/dist/corepack.js /usr/local/bin/corepack
|
||||
ln -sf /usr/local/lib/node_modules/npm/bin/npx-cli.js /usr/local/bin/npx
|
||||
|
||||
WORKDIR /opt/hermes
|
||||
|
||||
@@ -141,6 +203,22 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
|
||||
done && \
|
||||
npm cache clean --force
|
||||
|
||||
# ---------- Photon iMessage sidecar deps (baked, NS-606) ----------
|
||||
# The photon plugin's Node sidecar needs its own node_modules
|
||||
# (spectrum-ts). The install tree is immutable at runtime, so a lazy
|
||||
# `npm ci` on first connect would hit EROFS — bake the deps here instead
|
||||
# (deterministic installs, NS-559). The patch script is copied alongside
|
||||
# the manifests because package.json's postinstall runs it, which also
|
||||
# means the spectrum-ts patch is applied at build time. Layer-cached:
|
||||
# only re-runs when the sidecar manifests/patch change.
|
||||
COPY plugins/platforms/photon/sidecar/package.json \
|
||||
plugins/platforms/photon/sidecar/package-lock.json \
|
||||
plugins/platforms/photon/sidecar/patch-spectrum-mixed-attachments.mjs \
|
||||
plugins/platforms/photon/sidecar/
|
||||
RUN cd plugins/platforms/photon/sidecar && \
|
||||
npm ci --no-audit --fetch-retries=5 && \
|
||||
npm cache clean --force
|
||||
|
||||
# ---------- Layer-cached Python dependency install ----------
|
||||
# Copy only pyproject.toml + uv.lock so the Python dep resolve + wheel
|
||||
# download + native-extension compile layer is cached unless those inputs
|
||||
@@ -152,7 +230,7 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
|
||||
# frontend stats the readme path during dep resolution, so we `touch` an
|
||||
# empty placeholder — the real README is restored by `COPY . .` below.
|
||||
#
|
||||
# `uv sync --frozen --no-install-project --extra all --extra messaging`
|
||||
# `uv sync --frozen --no-install-project --extra all --extra messaging --extra otlp`
|
||||
# installs the deps reachable through the composite `[all]` extra
|
||||
# (handpicked set intended for the production image — excludes `[dev]`),
|
||||
# plus gateway messaging adapters that should work in the published image
|
||||
@@ -165,6 +243,10 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
|
||||
# so Docker users can use these providers without requiring runtime
|
||||
# lazy-install access to PyPI (often blocked in containerized envs).
|
||||
#
|
||||
# The [otlp] extra contains the SDK/exporter imported by Hermes when Gateway
|
||||
# Health export is enabled. Collector and observability-backend dependencies
|
||||
# remain external and are not part of the Hermes production image.
|
||||
#
|
||||
# The hindsight memory provider's client (hindsight-client) is baked in
|
||||
# for the same reason: it lazy-installs into /opt/hermes/.venv at first
|
||||
# use, which lives inside the (immutable) image layer rather than the
|
||||
@@ -182,7 +264,7 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
|
||||
# The editable link is created after the source copy below.
|
||||
COPY pyproject.toml uv.lock ./
|
||||
RUN touch ./README.md
|
||||
RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra anthropic --extra bedrock --extra azure-identity --extra hindsight --extra matrix
|
||||
RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra otlp --extra anthropic --extra bedrock --extra azure-identity --extra hindsight --extra matrix
|
||||
|
||||
# ---------- Frontend build (cached independently from Python source) ----------
|
||||
# Copy only the frontend source trees first so that Python-only changes don't
|
||||
@@ -216,7 +298,7 @@ RUN uv pip install --no-cache-dir --no-deps -e "."
|
||||
USER root
|
||||
RUN mkdir -p /opt/hermes/bin && \
|
||||
cp /opt/hermes/docker/hermes-exec-shim.sh /opt/hermes/bin/hermes && \
|
||||
chmod 0755 /opt/hermes/bin/hermes && \
|
||||
chmod 0755 /opt/hermes /opt/hermes/bin/hermes && \
|
||||
printf 'docker\n' > /opt/hermes/.install_method
|
||||
# The ``.install_method`` stamp is baked next to the running code (the install
|
||||
# tree), NOT into $HERMES_HOME. $HERMES_HOME (/opt/data) is a shared data
|
||||
@@ -229,7 +311,11 @@ RUN mkdir -p /opt/hermes/bin && \
|
||||
# `s6-setuidgid hermes` in its run script. If HERMES_UID is unset, services
|
||||
# run as the default hermes user (UID 10000).
|
||||
|
||||
# ---------- Bake build-time git revision ----------
|
||||
# ---------- Bake image provenance + build-time git revision ----------
|
||||
# The versioned, non-secret provenance marker is the authoritative runtime
|
||||
# signal that this filesystem came from an immutable image. It deliberately
|
||||
# lives outside both /opt/hermes (which operators sometimes bind-mount as a
|
||||
# checkout) and /opt/data (the mutable HERMES_HOME volume).
|
||||
# .dockerignore excludes .git, so `git rev-parse HEAD` from inside the
|
||||
# container always returns nothing — meaning `hermes dump` reports
|
||||
# "(unknown)" and the startup banner drops its `· upstream <sha>` suffix.
|
||||
@@ -242,14 +328,18 @@ RUN mkdir -p /opt/hermes/bin && \
|
||||
# banner.get_git_banner_state() try the baked SHA first, then fall back
|
||||
# to live `git rev-parse` for source installs (unchanged behaviour).
|
||||
#
|
||||
# The arg is optional — local `docker build` without --build-arg simply
|
||||
# omits the file, and the runtime falls back to live-git lookup. CI
|
||||
# The arg is optional — local `docker build` without --build-arg omits the
|
||||
# SHA file (and records a null provenance revision), so build-info falls back
|
||||
# to live-git lookup. CI
|
||||
# (.github/workflows/docker.yml) passes ${{ github.sha }} so
|
||||
# every published image has it.
|
||||
ARG HERMES_GIT_SHA=
|
||||
RUN if [ -n "${HERMES_GIT_SHA}" ]; then \
|
||||
RUN set -eu; \
|
||||
if [ -n "${HERMES_GIT_SHA}" ]; then \
|
||||
printf '%s\n' "${HERMES_GIT_SHA}" > /opt/hermes/.hermes_build_sha; \
|
||||
fi
|
||||
fi; \
|
||||
mkdir -p /etc/hermes; \
|
||||
HERMES_GIT_SHA="${HERMES_GIT_SHA}" python3 -c 'import json, os, pathlib, tomllib; project = tomllib.loads(pathlib.Path("/opt/hermes/pyproject.toml").read_text(encoding="utf-8"))["project"]; marker = pathlib.Path("/etc/hermes/image-provenance.json"); marker.write_text(json.dumps({"schema": 1, "deployment_kind": "image", "manager": "docker", "image": "nousresearch/hermes-agent", "version": project["version"], "revision": os.environ.get("HERMES_GIT_SHA") or None}, sort_keys=True, separators=(",", ":")) + "\n", encoding="utf-8"); marker.chmod(0o444)'
|
||||
|
||||
# ---------- s6-overlay service wiring ----------
|
||||
# Static services declared at build time: main-hermes + dashboard.
|
||||
@@ -321,6 +411,8 @@ ENV HERMES_LAZY_INSTALL_TARGET=/opt/data/lazy-packages
|
||||
# Recursion is impossible because the shim exec's the venv binary by
|
||||
# absolute path (/opt/hermes/.venv/bin/hermes). See the shim source for
|
||||
# the opt-out env var (HERMES_DOCKER_EXEC_AS_ROOT=1).
|
||||
COPY --chmod=0755 docker/hermes-exec-shim.sh /opt/hermes/bin/hermes
|
||||
COPY --chmod=0755 docker/entrypoint-dispatch.sh /opt/hermes/docker/entrypoint-dispatch.sh
|
||||
|
||||
# Pre-s6 entrypoint.sh did `source .venv/bin/activate` which exported
|
||||
# the venv bin onto PATH; Architecture B's main-wrapper.sh does the
|
||||
@@ -337,27 +429,37 @@ ENV PATH="/opt/hermes/bin:/opt/hermes/.venv/bin:/opt/data/.local/bin:${PATH}"
|
||||
RUN mkdir -p /opt/data
|
||||
VOLUME [ "/opt/data" ]
|
||||
|
||||
# s6-overlay's /init is PID 1. It sets up the supervision tree, runs
|
||||
# /etc/cont-init.d/* (our stage2 hook), starts s6-rc services
|
||||
# declared in /etc/s6-overlay/s6-rc.d/, then exec's its remaining
|
||||
# argv as the container's "main program" with stdin/stdout/stderr
|
||||
# inherited (this is what makes interactive --tui work). When the
|
||||
# main program exits, /init begins stage 3 shutdown and the container
|
||||
# exits with the program's exit code. Replaces tini — see Phase 2 of
|
||||
# docs/plans/2026-05-07-s6-overlay-dynamic-subagent-gateways.md.
|
||||
# The image ENTRYPOINT is a tiny dispatcher rather than `/init` directly.
|
||||
# When the image really owns PID 1 (normal Docker / Podman), the dispatcher
|
||||
# execs `/init` and preserves the full s6 supervision tree. When a platform
|
||||
# wraps the image entrypoint under its own PID-1 init (Fly Machines,
|
||||
# `docker run --init`, some schedulers), `/init` would abort with
|
||||
# `can only run as pid 1`; in that case the dispatcher falls back to
|
||||
# `stage2-hook.sh` + `main-wrapper.sh` directly so foreground commands still
|
||||
# work. See #38349.
|
||||
#
|
||||
# On the PID-1 path, s6-overlay's /init sets up the supervision tree, runs
|
||||
# /etc/cont-init.d/* (our stage2 hook), starts s6-rc services declared in
|
||||
# /etc/s6-overlay/s6-rc.d/, then exec's its remaining argv as the container's
|
||||
# "main program" with stdin/stdout/stderr inherited (this is what makes
|
||||
# interactive --tui work). When the main program exits, /init begins stage 3
|
||||
# shutdown and the container exits with the program's exit code. Replaces
|
||||
# tini — see Phase 2 of docs/plans/2026-05-07-s6-overlay-dynamic-subagent-gateways.md.
|
||||
#
|
||||
# We use the ENTRYPOINT+CMD split rather than CMD alone so the
|
||||
# wrapper is prepended to user-supplied args automatically:
|
||||
#
|
||||
# docker run <image> → /init main-wrapper.sh (CMD default)
|
||||
# docker run <image> chat -q "hi" → /init main-wrapper.sh chat -q hi
|
||||
# docker run <image> sleep infinity → /init main-wrapper.sh sleep infinity
|
||||
# docker run <image> --tui → /init main-wrapper.sh --tui
|
||||
# docker run <image> → entrypoint-dispatch.sh (CMD default)
|
||||
# docker run <image> chat -q "hi" → entrypoint-dispatch.sh chat -q hi
|
||||
# docker run <image> sleep infinity → entrypoint-dispatch.sh sleep infinity
|
||||
# docker run <image> --tui → entrypoint-dispatch.sh --tui
|
||||
#
|
||||
# main-wrapper.sh handles arg routing (bare-exec vs. hermes
|
||||
# subcommand vs. no-args), drops to the hermes user via s6-setuidgid,
|
||||
# and exec's the final program so its exit code becomes the container
|
||||
# exit code. Without the wrapper-as-ENTRYPOINT, leading-dash args
|
||||
# like `--version` would be intercepted by /init's POSIX shell.
|
||||
ENTRYPOINT [ "/init", "/opt/hermes/docker/main-wrapper.sh" ]
|
||||
# exit code. The dispatcher preserves that contract across both the
|
||||
# supervised PID-1 path and the non-PID-1 fallback path. Without the
|
||||
# wrapper-as-ENTRYPOINT, leading-dash args like `--version` would be
|
||||
# intercepted by /init's POSIX shell.
|
||||
ENTRYPOINT [ "/opt/hermes/docker/entrypoint-dispatch.sh" ]
|
||||
CMD [ ]
|
||||
|
||||
@@ -1,14 +0,0 @@
|
||||
graft skills
|
||||
graft optional-skills
|
||||
graft optional-mcps
|
||||
graft hermes_cli/web_dist
|
||||
graft locales
|
||||
# Bundled plugin manifests (plugin.yaml / plugin.yml). Without these the
|
||||
# PluginManager scan (hermes_cli/plugins.py) finds zero plugins on installs
|
||||
# built from the sdist (e.g. Homebrew, downstream packagers). package-data
|
||||
# below covers the wheel; this covers the sdist. See #34034 / #28149.
|
||||
recursive-include plugins plugin.yaml plugin.yml
|
||||
# Gateway assets include images plus YAML catalogs such as status_phrases.yaml.
|
||||
recursive-include gateway/assets *
|
||||
global-exclude __pycache__
|
||||
global-exclude *.py[cod]
|
||||
@@ -26,7 +26,7 @@ Use any model you want — [Nous Portal](https://portal.nousresearch.com), OpenR
|
||||
<tr><td><b>A closed learning loop</b></td><td>Agent-curated memory with periodic nudges. Autonomous skill creation after complex tasks. Skills self-improve during use. FTS5 session search with LLM summarization for cross-session recall. <a href="https://github.com/plastic-labs/honcho">Honcho</a> dialectic user modeling. Compatible with the <a href="https://agentskills.io">agentskills.io</a> open standard.</td></tr>
|
||||
<tr><td><b>Scheduled automations</b></td><td>Built-in cron scheduler with delivery to any platform. Daily reports, nightly backups, weekly audits — all in natural language, running unattended.</td></tr>
|
||||
<tr><td><b>Delegates and parallelizes</b></td><td>Spawn isolated subagents for parallel workstreams. Write Python scripts that call tools via RPC, collapsing multi-step pipelines into zero-context-cost turns.</td></tr>
|
||||
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Six terminal backends — local, Docker, SSH, Singularity, Modal, and Daytona. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
|
||||
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Seven terminal backends — local, Docker, SSH, Singularity, Modal, Daytona, and Vercel Sandbox. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
|
||||
<tr><td><b>Research-ready</b></td><td>Batch trajectory generation, trajectory compression for training the next generation of tool-calling models.</td></tr>
|
||||
</table>
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ Un informe útil incluye:
|
||||
- Una descripción concisa y evaluación de severidad.
|
||||
- El componente afectado, identificado por ruta de archivo y rango de líneas
|
||||
(ej. `path/to/file.py:120-145`).
|
||||
- Detalles del entorno (`hermes version`, SHA del commit, SO, versión de Python).
|
||||
- Detalles del entorno (`hermes --version`, SHA del commit, SO, versión de Python).
|
||||
- Una reproducción contra `main` o el último release.
|
||||
- Una declaración de qué límite de confianza del §2 se cruza.
|
||||
|
||||
@@ -173,9 +173,13 @@ modelo de autorización, pero las reglas a continuación se aplican uniformement
|
||||
|
||||
**Superficies en Hermes Agent:**
|
||||
|
||||
- **Adaptadores de plataforma del gateway.** Integraciones de mensajería en
|
||||
`gateway/platforms/` (Telegram, Discord, Slack, email, SMS, etc.)
|
||||
y adaptadores análogos incluidos como plugins.
|
||||
- **Adaptadores de plataforma del gateway.** La mayoría de las integraciones
|
||||
de mensajería se distribuyen como plugins empaquetados en
|
||||
`plugins/platforms/<name>/` (Telegram, Discord, Slack, email, SMS, etc.).
|
||||
Los tipos base compartidos y un conjunto menor de adaptadores
|
||||
legacy/directos viven en `gateway/platforms/` (`base.py`, Signal, servidor
|
||||
API, webhooks, …), con descubrimiento y carga diferida vía
|
||||
`gateway/platform_registry.py`.
|
||||
- **Superficies HTTP expuestas en red.** El adaptador del servidor API, el
|
||||
plugin del dashboard, los endpoints HTTP del plugin kanban, y cualquier
|
||||
otro plugin que vincule un socket de escucha.
|
||||
|
||||
@@ -16,7 +16,7 @@ A useful report includes:
|
||||
- A concise description and severity assessment.
|
||||
- The affected component, identified by file path and line range
|
||||
(e.g. `path/to/file.py:120-145`).
|
||||
- Environment details (`hermes version`, commit SHA, OS, Python
|
||||
- Environment details (`hermes --version`, commit SHA, OS, Python
|
||||
version).
|
||||
- A reproduction against `main` or the latest release.
|
||||
- A statement of which trust boundary in §2 is crossed.
|
||||
@@ -177,9 +177,12 @@ authorization model, but the rules below apply uniformly.
|
||||
|
||||
**Surfaces in Hermes Agent:**
|
||||
|
||||
- **Gateway platform adapters.** Messaging integrations in
|
||||
`gateway/platforms/` (Telegram, Discord, Slack, email, SMS, etc.)
|
||||
and analogous adapters shipped as plugins.
|
||||
- **Gateway platform adapters.** Most messaging integrations ship as
|
||||
bundled plugins under `plugins/platforms/<name>/` (Telegram, Discord,
|
||||
Slack, email, SMS, etc.). Shared base types and a smaller set of
|
||||
legacy/direct adapters live under `gateway/platforms/`
|
||||
(`base.py`, Signal, API server, webhooks, …), with discovery and
|
||||
deferred loading via `gateway/platform_registry.py`.
|
||||
- **Network-exposed HTTP surfaces.** The API server adapter, the
|
||||
dashboard plugin, the kanban plugin's HTTP endpoints, and any
|
||||
other plugin that binds a listening socket.
|
||||
|
||||
@@ -32,6 +32,7 @@ else:
|
||||
import argparse
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from hermes_constants import get_hermes_home
|
||||
@@ -79,9 +80,11 @@ class _BenignProbeMethodFilter(logging.Filter):
|
||||
|
||||
def _setup_logging() -> None:
|
||||
"""Route all logging to stderr so stdout stays clean for ACP stdio."""
|
||||
from agent.redact import RedactingFormatter
|
||||
|
||||
handler = logging.StreamHandler(sys.stderr)
|
||||
handler.setFormatter(
|
||||
logging.Formatter(
|
||||
RedactingFormatter(
|
||||
"%(asctime)s [%(levelname)s] %(name)s: %(message)s",
|
||||
datefmt="%Y-%m-%d %H:%M:%S",
|
||||
)
|
||||
@@ -190,7 +193,7 @@ def _run_setup_browser(assume_yes: bool = False) -> int:
|
||||
"""Bootstrap agent-browser + Chromium.
|
||||
|
||||
Routes through dep_ensure -> install.{sh,ps1} --ensure, sharing code
|
||||
with ``hermes postinstall`` and the runtime lazy installer.
|
||||
with the runtime lazy installer.
|
||||
|
||||
Returns 0 on success, 1 on failure.
|
||||
"""
|
||||
@@ -246,16 +249,24 @@ def main(argv: list[str] | None = None) -> None:
|
||||
import acp
|
||||
from .server import HermesACPAgent
|
||||
|
||||
# MCP tool discovery from config.yaml — run before asyncio.run() so
|
||||
# it's safe to use blocking waits. (ACP also registers per-session
|
||||
# MCP servers dynamically via asyncio.to_thread inside the event
|
||||
# loop; that path is unaffected.) Moved from model_tools.py module
|
||||
# scope to avoid freezing the gateway's loop on lazy import (#16856).
|
||||
try:
|
||||
from tools.mcp_tool import discover_mcp_tools
|
||||
discover_mcp_tools()
|
||||
except Exception:
|
||||
logger.debug("MCP tool discovery failed at ACP startup", exc_info=True)
|
||||
# MCP tool discovery from config.yaml — fire-and-forget in a
|
||||
# background daemon thread so the ACP server becomes responsive
|
||||
# immediately while MCP servers connect. Previously this blocked
|
||||
# asyncio.run() for 2-5 s. (ACP also registers per-session MCP
|
||||
# servers dynamically via asyncio.to_thread inside the event loop;
|
||||
# that path is unaffected.) Moved from model_tools.py module scope
|
||||
# to avoid freezing the gateway's loop on lazy import (#16856).
|
||||
# Metadata-only hosts can opt out of unrelated global MCP startup.
|
||||
if os.environ.get("HERMES_ACP_SKIP_CONFIGURED_MCP", "").strip() != "1":
|
||||
try:
|
||||
from hermes_cli.mcp_startup import start_background_mcp_discovery
|
||||
|
||||
start_background_mcp_discovery(
|
||||
logger=logger,
|
||||
thread_name="acp-mcp-discovery",
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("MCP tool discovery failed at ACP startup", exc_info=True)
|
||||
|
||||
agent = HermesACPAgent()
|
||||
try:
|
||||
|
||||
@@ -39,13 +39,19 @@ def _permission_option_supports_kind(kind: str) -> bool:
|
||||
|
||||
|
||||
def _build_permission_options(
|
||||
*, allow_permanent: bool, smart_denied: bool = False,
|
||||
*, allow_permanent: bool, allow_session: bool = True,
|
||||
smart_denied: bool = False,
|
||||
) -> list[PermissionOption]:
|
||||
"""Return ACP options that match Hermes approval semantics."""
|
||||
# A gate that re-asks every time (allow_session=False, e.g. protected
|
||||
# agent-instruction writes) collapses to the same two options as a
|
||||
# Smart DENY override — the editor must not offer a scope Hermes
|
||||
# discards, or every subsequent write re-prompts (#81887).
|
||||
once_only = smart_denied or not allow_session
|
||||
options = [PermissionOption(
|
||||
option_id="allow_once", kind="allow_once", name="Allow once",
|
||||
)]
|
||||
if not smart_denied:
|
||||
if not once_only:
|
||||
options.append(PermissionOption(
|
||||
option_id="allow_session",
|
||||
# ACP has no session-scoped kind, so use the closest persistent
|
||||
@@ -53,7 +59,7 @@ def _build_permission_options(
|
||||
kind="allow_always",
|
||||
name="Allow for session",
|
||||
))
|
||||
if allow_permanent and not smart_denied:
|
||||
if allow_permanent and not once_only:
|
||||
options.append(
|
||||
PermissionOption(
|
||||
option_id="allow_always",
|
||||
@@ -62,7 +68,7 @@ def _build_permission_options(
|
||||
),
|
||||
)
|
||||
options.append(PermissionOption(option_id="deny", kind="reject_once", name="Deny"))
|
||||
if not smart_denied and _permission_option_supports_kind("reject_always"):
|
||||
if not once_only and _permission_option_supports_kind("reject_always"):
|
||||
options.append(
|
||||
PermissionOption(
|
||||
option_id="deny_always",
|
||||
@@ -132,6 +138,7 @@ def make_approval_callback(
|
||||
description: str,
|
||||
*,
|
||||
allow_permanent: bool = True,
|
||||
allow_session: bool = True,
|
||||
smart_denied: bool = False,
|
||||
**_: object,
|
||||
) -> str:
|
||||
@@ -139,6 +146,7 @@ def make_approval_callback(
|
||||
|
||||
options = _build_permission_options(
|
||||
allow_permanent=allow_permanent,
|
||||
allow_session=allow_session,
|
||||
smart_denied=smart_denied,
|
||||
)
|
||||
|
||||
@@ -158,9 +166,16 @@ def make_approval_callback(
|
||||
|
||||
try:
|
||||
response = future.result(timeout=timeout)
|
||||
except (FutureTimeout, Exception) as exc:
|
||||
except FutureTimeout:
|
||||
future.cancel()
|
||||
logger.warning("Permission request timed out or failed: %s", exc)
|
||||
logger.warning("Permission request timed out after %ss", timeout)
|
||||
# Distinct from an explicit deny: the client never answered.
|
||||
# tools.approval callers report this as "timed out without user
|
||||
# response" instead of a user denial.
|
||||
return "timeout"
|
||||
except Exception as exc:
|
||||
future.cancel()
|
||||
logger.warning("Permission request failed: %s", exc)
|
||||
return "deny"
|
||||
|
||||
if response is None:
|
||||
|
||||
@@ -74,6 +74,11 @@ from acp_adapter.permissions import make_approval_callback
|
||||
from acp_adapter.provenance import session_provenance_meta
|
||||
from acp_adapter.session import SessionManager, SessionState, _expand_acp_enabled_toolsets
|
||||
from acp_adapter.tools import build_tool_complete, build_tool_start
|
||||
from agent.context_compressor import (
|
||||
COMPRESSED_SUMMARY_METADATA_KEY,
|
||||
ContextCompressor,
|
||||
)
|
||||
from agent.interrupt_compat import request_hard_interrupt
|
||||
from tools.approval import (
|
||||
reset_hermes_interactive_context,
|
||||
set_hermes_interactive_context,
|
||||
@@ -81,6 +86,142 @@ from tools.approval import (
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, str]]]]:
|
||||
"""Return ``(slug, label, [(model_id, description), ...])`` for named endpoints.
|
||||
|
||||
Covers both the v12 ``providers:`` mapping and the legacy
|
||||
``custom_providers:`` list. These endpoints never appear in canonical
|
||||
provider enumeration, so without this the ACP model selector hides every
|
||||
named endpoint that the TUI ``/model`` picker already renders (#47039
|
||||
implemented named-endpoint rows for the TUI surface only).
|
||||
|
||||
Model lists come from the entry's declared models (``default_model`` +
|
||||
``models``), refreshed from the endpoint's live ``/models`` listing when a
|
||||
credential is available and ``discover_models`` is not disabled. Declared
|
||||
models are kept even when live discovery fails — some OpenAI-compatible
|
||||
endpoints (e.g. Bedrock Mantle Responses) expose no ``/models`` route at
|
||||
all yet serve the declared models fine.
|
||||
|
||||
Slugs use the ``custom:<name>`` shape that ``parse_model_input`` and
|
||||
``resolve_runtime_provider`` already resolve, so encoded choice ids
|
||||
(``custom:<name>:<model>``) round-trip through ``set_session_model``
|
||||
unchanged.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.config import (
|
||||
get_compatible_custom_providers,
|
||||
is_provider_enabled,
|
||||
load_config,
|
||||
)
|
||||
from hermes_cli.model_switch import (
|
||||
_NativePickerModelList,
|
||||
_declared_model_ids,
|
||||
_entry_models_discovered,
|
||||
_fetch_picker_live_models,
|
||||
_models_config_is_allowlist,
|
||||
)
|
||||
from hermes_cli.models import should_use_ollama_native_catalog
|
||||
from hermes_cli.providers import custom_provider_slug
|
||||
except ImportError:
|
||||
return []
|
||||
|
||||
try:
|
||||
cfg = load_config()
|
||||
entries = get_compatible_custom_providers(cfg)
|
||||
except Exception:
|
||||
logger.debug("Could not load named custom providers", exc_info=True)
|
||||
return []
|
||||
|
||||
# ``get_compatible_custom_providers`` drops the ``enabled`` flag during
|
||||
# normalization, so collect explicitly disabled provider keys from the
|
||||
# raw config and skip their entries below.
|
||||
disabled_keys: set[str] = set()
|
||||
raw_providers = cfg.get("providers") if isinstance(cfg, dict) else None
|
||||
if isinstance(raw_providers, dict):
|
||||
for raw_key, raw_entry in raw_providers.items():
|
||||
if isinstance(raw_entry, dict) and not is_provider_enabled(raw_entry):
|
||||
disabled_keys.add(str(raw_key).strip().lower())
|
||||
|
||||
catalogs: list[tuple[str, str, list[tuple[str, str]]]] = []
|
||||
for entry in entries:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
provider_key = str(entry.get("provider_key", "") or "").strip()
|
||||
if provider_key.lower() in disabled_keys:
|
||||
continue
|
||||
name = str(entry.get("name", "") or "").strip()
|
||||
base_url = str(entry.get("base_url", "") or "").strip()
|
||||
if not name or not base_url:
|
||||
continue
|
||||
slug = custom_provider_slug(name, provider_key)
|
||||
|
||||
api_key = str(entry.get("api_key", "") or "").strip()
|
||||
if not api_key:
|
||||
key_env = str(
|
||||
entry.get("key_env") or entry.get("api_key_env") or ""
|
||||
).strip()
|
||||
api_key = os.environ.get(key_env, "").strip() if key_env else ""
|
||||
|
||||
declared: list[str] = []
|
||||
default_model = str(entry.get("model", "") or "").strip()
|
||||
if default_model:
|
||||
declared.append(default_model)
|
||||
models_cfg = entry.get("models")
|
||||
for mid in _declared_model_ids(models_cfg):
|
||||
if mid not in declared:
|
||||
declared.append(mid)
|
||||
|
||||
native_headers = entry.get("extra_headers") or None
|
||||
native_catalog_provider = (
|
||||
provider_key
|
||||
if provider_key.lower() in {"ollama", "custom:ollama"}
|
||||
else "custom"
|
||||
)
|
||||
is_native_ollama = should_use_ollama_native_catalog(
|
||||
native_catalog_provider, base_url, headers=native_headers
|
||||
)
|
||||
explicit_catalog = _models_config_is_allowlist(
|
||||
models_cfg, _entry_models_discovered(entry)
|
||||
)
|
||||
if not api_key and not declared and not is_native_ollama:
|
||||
# No credential to discover with and nothing declared:
|
||||
# not addressable from the selector.
|
||||
continue
|
||||
|
||||
model_ids = list(declared)
|
||||
discover = entry.get("discover_models", True)
|
||||
if isinstance(discover, str):
|
||||
discover = discover.lower() not in {"false", "no", "0"}
|
||||
native_catalog_provider = native_catalog_provider if is_native_ollama else "custom"
|
||||
live = None
|
||||
if discover and (api_key or is_native_ollama):
|
||||
try:
|
||||
live = _fetch_picker_live_models(
|
||||
api_key,
|
||||
base_url,
|
||||
native_catalog_provider,
|
||||
explicit_catalog,
|
||||
headers=native_headers,
|
||||
timeout=1.5,
|
||||
api_mode=entry.get("api_mode"),
|
||||
)
|
||||
except Exception:
|
||||
live = None
|
||||
if live is not None:
|
||||
if isinstance(live, _NativePickerModelList):
|
||||
model_ids = list(live)
|
||||
else:
|
||||
model_ids = declared + [m for m in live if m not in declared]
|
||||
|
||||
if not model_ids:
|
||||
if isinstance(live if "live" in locals() else None, _NativePickerModelList):
|
||||
catalogs.append((slug, name, []))
|
||||
continue
|
||||
catalogs.append((slug, name, [(mid, "") for mid in model_ids]))
|
||||
|
||||
return catalogs
|
||||
|
||||
try:
|
||||
from hermes_cli import __version__ as HERMES_VERSION
|
||||
except Exception:
|
||||
@@ -93,6 +234,13 @@ _executor = ThreadPoolExecutor(max_workers=4, thread_name_prefix="acp-agent")
|
||||
# does not expose a client-side limit, so this is a fixed cap that clients
|
||||
# paginate against using `cursor` / `next_cursor`.
|
||||
_LIST_SESSIONS_PAGE_SIZE = 50
|
||||
# Per-provider cap for the ACP model selector. ACP clients (Zed, Buzz) render
|
||||
# the whole `availableModels` array in one dropdown, so an unbounded
|
||||
# cross-provider catalog degrades the picker. Mirrors the cap the MoA picker
|
||||
# already uses (`hermes_cli/moa_cmd.py`). This bounds each provider's row, not
|
||||
# the total; aggregator providers stay intentionally uncapped inside the shared
|
||||
# inventory, and the current model is always kept via the fallback insert below.
|
||||
ACP_MAX_MODELS_PER_PROVIDER = 200
|
||||
_MAX_ACP_RESOURCE_BYTES = 512 * 1024
|
||||
_TEXT_RESOURCE_MIME_PREFIXES = ("text/",)
|
||||
_TEXT_RESOURCE_MIME_TYPES = {
|
||||
@@ -581,54 +729,240 @@ class HermesACPAgent(acp.Agent):
|
||||
return f"{raw_provider}:{raw_model}"
|
||||
|
||||
def _build_model_state(self, state: SessionState) -> SessionModelState | None:
|
||||
"""Return the ACP model selector payload for editors like Zed."""
|
||||
"""Return authenticated providers and their models for ACP clients.
|
||||
|
||||
The shared Hermes inventory is also used by ``hermes model``, the TUI,
|
||||
and the dashboard. Keeping ACP on that substrate prevents its selector
|
||||
from silently collapsing to the current provider's curated list.
|
||||
"""
|
||||
model = str(state.model or getattr(state.agent, "model", "") or "").strip()
|
||||
provider = getattr(state.agent, "provider", None) or detect_provider() or "openrouter"
|
||||
|
||||
try:
|
||||
from hermes_cli.models import curated_models_for_provider, normalize_provider, provider_label
|
||||
from hermes_cli.inventory import build_models_payload, load_picker_context
|
||||
from hermes_cli.models import normalize_provider, provider_label
|
||||
|
||||
normalized_provider = normalize_provider(provider)
|
||||
provider_name = provider_label(normalized_provider)
|
||||
context = load_picker_context().with_overrides(
|
||||
current_provider=normalized_provider,
|
||||
current_model=model,
|
||||
current_base_url=str(getattr(state.agent, "base_url", "") or ""),
|
||||
)
|
||||
payload = build_models_payload(
|
||||
context,
|
||||
explicit_only=True,
|
||||
include_unconfigured=False,
|
||||
picker_hints=False,
|
||||
canonical_order=True,
|
||||
pricing=False,
|
||||
capabilities=False,
|
||||
refresh=False,
|
||||
probe_custom_providers=False,
|
||||
probe_current_custom_provider=False,
|
||||
max_models=ACP_MAX_MODELS_PER_PROVIDER,
|
||||
)
|
||||
|
||||
available_models: list[ModelInfo] = []
|
||||
seen_ids: set[str] = set()
|
||||
current_choice_provider = str(provider or "").strip().lower()
|
||||
if current_choice_provider == "ollama":
|
||||
current_choice_provider = "custom:ollama"
|
||||
current_base_url = str(
|
||||
getattr(state.agent, "base_url", "") or ""
|
||||
).strip().rstrip("/").lower()
|
||||
|
||||
for model_id, description in curated_models_for_provider(normalized_provider):
|
||||
rendered_model = str(model_id or "").strip()
|
||||
if not rendered_model:
|
||||
def semantic_provider(provider_id: str) -> str:
|
||||
raw = str(provider_id or "").strip().lower()
|
||||
if raw in {"ollama", "custom:ollama"}:
|
||||
return "ollama"
|
||||
if raw.startswith("custom:"):
|
||||
return raw
|
||||
return normalize_provider(raw)
|
||||
|
||||
seen_semantic_ids: set[str] = set()
|
||||
native_empty_rows: set[str] = set()
|
||||
current_identity_resolved = current_choice_provider not in {"", "custom"}
|
||||
for row in payload.get("providers") or []:
|
||||
raw_row_provider = str(row.get("slug") or "").strip().lower()
|
||||
row_provider = normalize_provider(raw_row_provider)
|
||||
row_base_url = str(row.get("api_url") or "").strip().rstrip("/").lower()
|
||||
if row.get("native_catalog_empty"):
|
||||
native_empty_rows.add(raw_row_provider)
|
||||
if (
|
||||
not current_identity_resolved
|
||||
and raw_row_provider in {"ollama", "custom:ollama"}
|
||||
and current_base_url
|
||||
and row_base_url == current_base_url
|
||||
):
|
||||
current_choice_provider = "custom:ollama"
|
||||
current_identity_resolved = True
|
||||
if not row_provider:
|
||||
continue
|
||||
choice_id = self._encode_model_choice(normalized_provider, rendered_model)
|
||||
if choice_id in seen_ids:
|
||||
continue
|
||||
desc_parts = [f"Provider: {provider_name}"]
|
||||
if description:
|
||||
desc_parts.append(str(description).strip())
|
||||
if rendered_model == model:
|
||||
desc_parts.append("current")
|
||||
available_models.append(
|
||||
ModelInfo(
|
||||
model_id=choice_id,
|
||||
name=rendered_model,
|
||||
description=" • ".join(part for part in desc_parts if part),
|
||||
)
|
||||
provider_name = str(row.get("name") or "").strip() or provider_label(
|
||||
row_provider
|
||||
)
|
||||
seen_ids.add(choice_id)
|
||||
row_models = row.get("models")
|
||||
if not isinstance(row_models, (list, tuple)):
|
||||
continue
|
||||
for model_entry in row_models:
|
||||
if isinstance(model_entry, dict):
|
||||
rendered_model = str(
|
||||
model_entry.get("id")
|
||||
or model_entry.get("model")
|
||||
or model_entry.get("name")
|
||||
or ""
|
||||
).strip()
|
||||
else:
|
||||
rendered_model = str(model_entry or "").strip()
|
||||
if not rendered_model:
|
||||
continue
|
||||
encoded_provider = (
|
||||
"custom:ollama"
|
||||
if raw_row_provider == "ollama"
|
||||
else raw_row_provider
|
||||
if raw_row_provider == "custom:ollama"
|
||||
else raw_row_provider
|
||||
if raw_row_provider.startswith("custom:")
|
||||
else row_provider
|
||||
)
|
||||
choice_id = self._encode_model_choice(
|
||||
encoded_provider, rendered_model
|
||||
)
|
||||
semantic_id = f"{semantic_provider(encoded_provider)}:{rendered_model}"
|
||||
if choice_id in seen_ids or semantic_id in seen_semantic_ids:
|
||||
continue
|
||||
is_current = (
|
||||
semantic_provider(encoded_provider)
|
||||
== semantic_provider(current_choice_provider)
|
||||
and rendered_model == model
|
||||
)
|
||||
description = f"Provider: {provider_name}"
|
||||
if is_current:
|
||||
description += " • current"
|
||||
available_models.append(
|
||||
ModelInfo(
|
||||
model_id=choice_id,
|
||||
name=f"{provider_name} · {rendered_model}",
|
||||
description=description,
|
||||
)
|
||||
)
|
||||
seen_ids.add(choice_id)
|
||||
seen_semantic_ids.add(semantic_id)
|
||||
|
||||
current_model_id = self._encode_model_choice(normalized_provider, model)
|
||||
if current_model_id and current_model_id not in seen_ids:
|
||||
# Named user-defined endpoints (providers: / custom_providers:)
|
||||
# are invisible to canonical provider enumeration — append them
|
||||
# so editor clients can select them like the TUI /model picker.
|
||||
named_empty_authoritative: set[str] = set(native_empty_rows)
|
||||
for named_slug, named_label, named_catalog in _named_custom_provider_catalogs():
|
||||
if not named_catalog:
|
||||
named_empty_authoritative.add(str(named_slug).strip().lower())
|
||||
continue
|
||||
for named_model, named_desc in named_catalog:
|
||||
named_choice = self._encode_model_choice(named_slug, named_model)
|
||||
named_semantic_id = (
|
||||
f"{semantic_provider(named_slug)}:{named_model}"
|
||||
)
|
||||
if (
|
||||
not named_choice
|
||||
or named_choice in seen_ids
|
||||
or named_semantic_id in seen_semantic_ids
|
||||
):
|
||||
continue
|
||||
named_parts = [f"Provider: {named_label}"]
|
||||
if named_desc:
|
||||
named_parts.append(str(named_desc).strip())
|
||||
if named_slug == normalized_provider and named_model == model:
|
||||
named_parts.append("current")
|
||||
available_models.append(
|
||||
ModelInfo(
|
||||
model_id=named_choice,
|
||||
name=named_model,
|
||||
description=" • ".join(part for part in named_parts if part),
|
||||
)
|
||||
)
|
||||
seen_ids.add(named_choice)
|
||||
seen_semantic_ids.add(named_semantic_id)
|
||||
|
||||
def empty_catalog_applies(provider_id: str) -> bool:
|
||||
raw = str(provider_id or "").strip().lower()
|
||||
normalized = normalize_provider(raw)
|
||||
if normalized == "custom":
|
||||
return any(
|
||||
candidate == raw
|
||||
or f"custom:{candidate}" == raw
|
||||
or (raw == "custom" and candidate == "custom")
|
||||
for candidate in named_empty_authoritative
|
||||
)
|
||||
return any(
|
||||
candidate == raw
|
||||
or candidate == f"custom:{normalized}"
|
||||
or candidate == f"custom:{raw}"
|
||||
or normalize_provider(candidate) == normalized
|
||||
for candidate in named_empty_authoritative
|
||||
)
|
||||
|
||||
def choice_provider(model_id: str) -> str:
|
||||
parts = model_id.split(":")
|
||||
if parts[:1] == ["custom"] and len(parts) > 1:
|
||||
from hermes_cli.models import _configured_custom_provider_ids
|
||||
|
||||
lowered = model_id.lower()
|
||||
for candidate in sorted(
|
||||
(
|
||||
provider_id
|
||||
for provider_id in _configured_custom_provider_ids()
|
||||
if provider_id.startswith("custom:")
|
||||
),
|
||||
key=len,
|
||||
reverse=True,
|
||||
):
|
||||
if lowered.startswith(candidate + ":"):
|
||||
return candidate
|
||||
return "custom"
|
||||
return parts[0]
|
||||
|
||||
if named_empty_authoritative:
|
||||
available_models = [
|
||||
item
|
||||
for item in available_models
|
||||
if not empty_catalog_applies(choice_provider(item.model_id))
|
||||
]
|
||||
seen_ids = {item.model_id for item in available_models}
|
||||
|
||||
current_is_empty = empty_catalog_applies(current_choice_provider)
|
||||
if current_is_empty:
|
||||
available_models = [
|
||||
item
|
||||
for item in available_models
|
||||
if " • current" not in str(item.description or "")
|
||||
]
|
||||
seen_ids = {item.model_id for item in available_models}
|
||||
current_model_id = (
|
||||
"" if current_is_empty else self._encode_model_choice(current_choice_provider, model)
|
||||
)
|
||||
if (
|
||||
current_model_id
|
||||
and current_model_id not in seen_ids
|
||||
and not current_is_empty
|
||||
):
|
||||
provider_name = provider_label(normalized_provider)
|
||||
available_models.insert(
|
||||
0,
|
||||
ModelInfo(
|
||||
model_id=current_model_id,
|
||||
name=model,
|
||||
name=f"{provider_name} · {model}",
|
||||
description=f"Provider: {provider_name} • current",
|
||||
),
|
||||
)
|
||||
|
||||
if not available_models and current_is_empty:
|
||||
return SessionModelState(available_models=[], current_model_id="")
|
||||
if available_models:
|
||||
return SessionModelState(
|
||||
available_models=available_models,
|
||||
current_model_id=current_model_id or available_models[0].model_id,
|
||||
current_model_id=current_model_id
|
||||
if current_model_id or current_is_empty
|
||||
else available_models[0].model_id,
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("Could not build ACP model state", exc_info=True)
|
||||
@@ -860,6 +1194,102 @@ class HermesACPAgent(acp.Agent):
|
||||
exc_info=True,
|
||||
)
|
||||
|
||||
def _schedule_mcp_late_refresh(self, state: SessionState) -> None:
|
||||
"""Refresh the agent's tool snapshot when background MCP discovery lands late.
|
||||
|
||||
ACP entry.py starts MCP tool discovery in a background daemon thread so a
|
||||
slow/dead configured server can't block ``asyncio.run()``. ``_make_agent``
|
||||
briefly joins that thread (``wait_for_mcp_discovery``, bounded ~1.5s) so
|
||||
already-spawning fast servers land in the snapshot — but a server slower
|
||||
than the bound lands *after* the agent is built, leaving its tools absent
|
||||
for the whole session.
|
||||
|
||||
This schedules an off-critical-path daemon that waits for discovery to
|
||||
finish (bounded 30s), then rebuilds the snapshot via the shared
|
||||
``refresh_agent_mcp_tools`` helper — the same rebuild ``/reload-mcp``
|
||||
performs, but automatic. Mirrors the TUI late-refresh (PR #48403).
|
||||
|
||||
Cache safety: the rebuild only runs while the session is still
|
||||
pre-first-turn (no API call made yet → nothing cached to invalidate).
|
||||
Once the user has sent a message we leave the snapshot frozen rather
|
||||
than break the cached prompt prefix mid-conversation; servers that land
|
||||
later are picked up cache-safely by the between-turns prologue refresh
|
||||
(``agent/turn_context.py``) at the next turn boundary. The marginal
|
||||
value of this pre-first-turn daemon is therefore freshness in the
|
||||
window [session created → first message] — e.g. the "Available tools"
|
||||
listing a client may request before the first prompt.
|
||||
No-op when discovery already finished, when the join times out, when the
|
||||
registry was unchanged, or when the session was closed while waiting.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.mcp_startup import mcp_discovery_in_flight
|
||||
except Exception:
|
||||
return
|
||||
if not mcp_discovery_in_flight():
|
||||
return
|
||||
|
||||
import threading
|
||||
|
||||
agent = state.agent
|
||||
session_id = state.session_id
|
||||
|
||||
def _wait_then_refresh() -> None:
|
||||
try:
|
||||
from hermes_cli.mcp_startup import join_mcp_discovery
|
||||
|
||||
if not join_mcp_discovery(timeout=30.0):
|
||||
return
|
||||
|
||||
# Session may have been closed while we waited. In-memory-only
|
||||
# lookup on purpose: ``get_session()`` falls through to a DB
|
||||
# restore that builds a whole new AIAgent as a side effect just
|
||||
# to decide "no-op" here (the TUI equivalent also checks its
|
||||
# in-memory dict only).
|
||||
with self.session_manager._lock:
|
||||
current = self.session_manager._sessions.get(session_id)
|
||||
if current is None or current.agent is not agent:
|
||||
return
|
||||
|
||||
# Cache safety: never rebuild the tool list once the conversation
|
||||
# has started — that would invalidate the cached prompt prefix.
|
||||
# Serialized with turn start: ``prompt()`` flips ``is_running``
|
||||
# under ``runtime_lock`` before dispatching, so holding it here
|
||||
# (and bailing when a turn is already running) closes the window
|
||||
# where the guard passes but the first prompt starts before the
|
||||
# refresh publishes — which would swap ``tools=`` mid-turn and
|
||||
# break the just-created cache prefix.
|
||||
with current.runtime_lock:
|
||||
if current.is_running:
|
||||
return
|
||||
if (
|
||||
int(getattr(agent, "_user_turn_count", 0) or 0) > 0
|
||||
or int(getattr(agent, "_api_call_count", 0) or 0) > 0
|
||||
):
|
||||
return
|
||||
|
||||
from tools.mcp_tool import refresh_agent_mcp_tools
|
||||
|
||||
added = refresh_agent_mcp_tools(agent, quiet_mode=True)
|
||||
if added:
|
||||
logger.info(
|
||||
"Session %s: late MCP refresh added %d tools: %s",
|
||||
session_id,
|
||||
len(added),
|
||||
", ".join(sorted(added)),
|
||||
)
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"Session %s: late MCP refresh failed",
|
||||
session_id,
|
||||
exc_info=True,
|
||||
)
|
||||
|
||||
threading.Thread(
|
||||
target=_wait_then_refresh,
|
||||
name=f"acp-mcp-late-refresh-{session_id}",
|
||||
daemon=True,
|
||||
).start()
|
||||
|
||||
# ---- ACP lifecycle ------------------------------------------------------
|
||||
|
||||
async def initialize(
|
||||
@@ -969,11 +1399,49 @@ class HermesACPAgent(acp.Agent):
|
||||
return text
|
||||
return ""
|
||||
|
||||
@staticmethod
|
||||
def _history_summary_meta(message: dict[str, Any], text: str) -> dict[str, Any] | None:
|
||||
"""Build the ``_meta`` payload for a replayed compaction summary.
|
||||
|
||||
Compaction summaries are persisted as ordinary history messages —
|
||||
standalone handoffs under ``role="user"`` OR ``role="assistant"``
|
||||
(the compressor picks whichever role keeps alternation valid), and
|
||||
merge-into-tail messages where the summary is appended after the
|
||||
first preserved tail message's real content. Without a wire flag,
|
||||
ACP frontends render all of these as ordinary turns.
|
||||
|
||||
Two distinct keys under ``_meta.hermes`` (ACP's extensibility
|
||||
channel), so clients cannot accidentally hide real content:
|
||||
|
||||
* ``compactionSummary: true`` — the entire chunk is the handoff
|
||||
summary. Safe to restyle or collapse wholesale.
|
||||
* ``containsCompactionSummary: true`` — a merged-tail message: real
|
||||
preserved turn content followed by the summary. Clients may style
|
||||
it, but collapsing the whole chunk would hide the preserved
|
||||
content, hence the separate key.
|
||||
|
||||
Detection honors the in-process ``_compressed_summary`` flag and
|
||||
falls back to content classification, so it also works for a
|
||||
DB-reloaded session that lost the in-memory flag.
|
||||
"""
|
||||
kind = ContextCompressor.classify_summary_content(text)
|
||||
if kind is None and message.get(COMPRESSED_SUMMARY_METADATA_KEY):
|
||||
# Flagged in-process but content didn't classify (e.g. future
|
||||
# prefix drift): treat as a standalone summary — the flag is only
|
||||
# ever set on summary-bearing messages.
|
||||
kind = "standalone"
|
||||
if kind == "standalone":
|
||||
return {"hermes": {"compactionSummary": True}}
|
||||
if kind == "merged":
|
||||
return {"hermes": {"containsCompactionSummary": True}}
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _history_message_update(
|
||||
*,
|
||||
role: str,
|
||||
text: str,
|
||||
field_meta: dict[str, Any] | None = None,
|
||||
) -> UserMessageChunk | AgentMessageChunk | None:
|
||||
"""Build an ACP history replay update for a user/assistant message."""
|
||||
block = TextContentBlock(type="text", text=text)
|
||||
@@ -981,11 +1449,13 @@ class HermesACPAgent(acp.Agent):
|
||||
return UserMessageChunk(
|
||||
session_update="user_message_chunk",
|
||||
content=block,
|
||||
field_meta=field_meta,
|
||||
)
|
||||
if role == "assistant":
|
||||
return AgentMessageChunk(
|
||||
session_update="agent_message_chunk",
|
||||
content=block,
|
||||
field_meta=field_meta,
|
||||
)
|
||||
return None
|
||||
|
||||
@@ -1056,7 +1526,11 @@ class HermesACPAgent(acp.Agent):
|
||||
if role == "user":
|
||||
text = self._history_message_text(message)
|
||||
if text:
|
||||
update = self._history_message_update(role=role, text=text)
|
||||
update = self._history_message_update(
|
||||
role=role,
|
||||
text=text,
|
||||
field_meta=self._history_summary_meta(message, text),
|
||||
)
|
||||
if update is not None and not await _send(update):
|
||||
return
|
||||
continue
|
||||
@@ -1068,7 +1542,11 @@ class HermesACPAgent(acp.Agent):
|
||||
|
||||
text = self._history_message_text(message)
|
||||
if text:
|
||||
update = self._history_message_update(role=role, text=text)
|
||||
update = self._history_message_update(
|
||||
role=role,
|
||||
text=text,
|
||||
field_meta=self._history_summary_meta(message, text),
|
||||
)
|
||||
if update is not None and not await _send(update):
|
||||
return
|
||||
|
||||
@@ -1118,6 +1596,7 @@ class HermesACPAgent(acp.Agent):
|
||||
) -> NewSessionResponse:
|
||||
state = self.session_manager.create_session(cwd=cwd)
|
||||
await self._register_session_mcp_servers(state, mcp_servers)
|
||||
self._schedule_mcp_late_refresh(state)
|
||||
logger.info("New session %s (cwd=%s)", state.session_id, cwd)
|
||||
self._schedule_available_commands_update(state.session_id)
|
||||
self._schedule_usage_update(state)
|
||||
@@ -1142,6 +1621,7 @@ class HermesACPAgent(acp.Agent):
|
||||
logger.warning("load_session: session %s not found", session_id)
|
||||
return None
|
||||
await self._register_session_mcp_servers(state, mcp_servers)
|
||||
self._schedule_mcp_late_refresh(state)
|
||||
logger.info("Loaded session %s", session_id)
|
||||
# Per ACP spec, `session/load` must stream the prior conversation back
|
||||
# to the client via `session/update` notifications BEFORE responding,
|
||||
@@ -1189,6 +1669,7 @@ class HermesACPAgent(acp.Agent):
|
||||
logger.warning("resume_session: session %s not found, creating new", session_id)
|
||||
state = self.session_manager.create_session(cwd=cwd)
|
||||
await self._register_session_mcp_servers(state, mcp_servers)
|
||||
self._schedule_mcp_late_refresh(state)
|
||||
logger.info("Resumed session %s", state.session_id)
|
||||
# See `load_session` above for the spec rationale — replay must
|
||||
# complete before the response so clients receive the full transcript
|
||||
@@ -1218,12 +1699,19 @@ class HermesACPAgent(acp.Agent):
|
||||
with state.runtime_lock:
|
||||
if state.is_running and state.current_prompt_text:
|
||||
state.interrupted_prompt_text = state.current_prompt_text
|
||||
state.cancel_event.set()
|
||||
try:
|
||||
if getattr(state, "agent", None) and hasattr(state.agent, "interrupt"):
|
||||
state.agent.interrupt()
|
||||
except Exception:
|
||||
logger.debug("Failed to interrupt ACP session %s", session_id, exc_info=True)
|
||||
# Publish cancellation and hard-stop the agent before another
|
||||
# prompt can acquire this lock and mistake the turn for
|
||||
# redirectable work.
|
||||
state.cancel_event.set()
|
||||
try:
|
||||
if getattr(state, "agent", None):
|
||||
request_hard_interrupt(state.agent)
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"Failed to interrupt ACP session %s",
|
||||
session_id,
|
||||
exc_info=True,
|
||||
)
|
||||
logger.info("Cancelled session %s", session_id)
|
||||
|
||||
async def fork_session(
|
||||
@@ -1352,6 +1840,26 @@ class HermesACPAgent(acp.Agent):
|
||||
elif rewrite_idle:
|
||||
user_text = steer_text
|
||||
user_content = steer_text
|
||||
elif (
|
||||
text_only_prompt
|
||||
and isinstance(user_content, str)
|
||||
and not user_text.startswith("/")
|
||||
):
|
||||
# Some ACP clients implement "stop and send" as two protocol calls:
|
||||
# cancel the active prompt, then submit plain correction text. Keep
|
||||
# the cancelled request attached so deictic follow-ups ("not that
|
||||
# file") still have an explicit target.
|
||||
interrupted_prompt = ""
|
||||
with state.runtime_lock:
|
||||
if not state.is_running and state.interrupted_prompt_text:
|
||||
interrupted_prompt = state.interrupted_prompt_text
|
||||
state.interrupted_prompt_text = ""
|
||||
if interrupted_prompt:
|
||||
user_text = (
|
||||
f"{interrupted_prompt}\n\n"
|
||||
f"User correction/guidance after interrupt: {user_text}"
|
||||
)
|
||||
user_content = user_text
|
||||
|
||||
# Intercept slash commands — handle locally without calling the LLM.
|
||||
# Slash commands are text-only; if the client included images/resources,
|
||||
@@ -1366,23 +1874,54 @@ class HermesACPAgent(acp.Agent):
|
||||
await self._send_usage_update(state)
|
||||
return PromptResponse(stop_reason="end_turn")
|
||||
|
||||
# If Zed sends another regular prompt while the same ACP session is
|
||||
# still running, queue it instead of racing two AIAgent loops against
|
||||
# the same state.history. /steer and /queue are handled above and can
|
||||
# land immediately.
|
||||
# If the client sends another regular text prompt while this ACP session
|
||||
# is running, route it through the core active-turn redirect. Rich media
|
||||
# and older runtimes retain the proven next-turn queue fallback.
|
||||
redirected = False
|
||||
queued_depth: int | None = None
|
||||
with state.runtime_lock:
|
||||
if state.is_running:
|
||||
queued_text = user_text or "[Image attachment]"
|
||||
state.queued_prompts.append(queued_text)
|
||||
depth = len(state.queued_prompts)
|
||||
if self._conn:
|
||||
update = acp.update_agent_message_text(
|
||||
f"Queued for the next turn. ({depth} queued)"
|
||||
if (
|
||||
text_only_prompt
|
||||
and isinstance(user_content, str)
|
||||
and getattr(
|
||||
state.agent,
|
||||
"_supports_active_turn_redirect",
|
||||
False,
|
||||
)
|
||||
await self._conn.session_update(session_id, update)
|
||||
return PromptResponse(stop_reason="end_turn")
|
||||
state.is_running = True
|
||||
state.current_prompt_text = user_text or "[Image attachment]"
|
||||
is True
|
||||
and hasattr(state.agent, "redirect")
|
||||
):
|
||||
try:
|
||||
redirected = bool(state.agent.redirect(user_content))
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"ACP active-turn redirect failed for %s",
|
||||
session_id,
|
||||
exc_info=True,
|
||||
)
|
||||
if not redirected:
|
||||
queued_text = user_text or "[Image attachment]"
|
||||
state.queued_prompts.append(queued_text)
|
||||
queued_depth = len(state.queued_prompts)
|
||||
else:
|
||||
state.is_running = True
|
||||
state.current_prompt_text = user_text or "[Image attachment]"
|
||||
|
||||
if redirected:
|
||||
if self._conn:
|
||||
update = acp.update_agent_message_text(
|
||||
"Redirected the active turn with your correction."
|
||||
)
|
||||
await self._conn.session_update(session_id, update)
|
||||
return PromptResponse(stop_reason="end_turn")
|
||||
if queued_depth is not None:
|
||||
if self._conn:
|
||||
update = acp.update_agent_message_text(
|
||||
f"Queued for the next turn. ({queued_depth} queued)"
|
||||
)
|
||||
await self._conn.session_update(session_id, update)
|
||||
return PromptResponse(stop_reason="end_turn")
|
||||
|
||||
logger.info("Prompt on session %s: %s", session_id, user_text[:100])
|
||||
|
||||
@@ -1478,7 +2017,19 @@ class HermesACPAgent(acp.Agent):
|
||||
clear_session_vars,
|
||||
set_session_vars,
|
||||
)
|
||||
session_tokens = set_session_vars(session_key=session_id)
|
||||
# ``cwd`` pins the logical working directory for this context,
|
||||
# which is what the system prompt's "Current working directory"
|
||||
# line reports (agent/prompt_builder.py -> resolve_agent_cwd).
|
||||
# Without it the prompt advertises the global Hermes workspace
|
||||
# while the tools are rooted at the client's project, so the
|
||||
# model emits absolute paths under ~/.hermes/workspace and the
|
||||
# edit silently lands outside the editor's workspace.
|
||||
# cron_session="" explicitly marks this as a non-cron context,
|
||||
# masking any leaked process-global HERMES_CRON_SESSION (#37968).
|
||||
session_tokens = set_session_vars(
|
||||
session_key=session_id, session_id=session_id, cwd=state.cwd,
|
||||
cron_session="",
|
||||
)
|
||||
except Exception:
|
||||
session_tokens = None
|
||||
clear_session_vars = None # type: ignore[assignment]
|
||||
@@ -1509,6 +2060,17 @@ class HermesACPAgent(acp.Agent):
|
||||
# never leaks one session's id into the next session's tools.
|
||||
previous_session_id = os.environ.get("HERMES_SESSION_ID")
|
||||
os.environ["HERMES_SESSION_ID"] = session_id
|
||||
# Auto-titling fires inside the turn prologue now; give the agent
|
||||
# this session's notifier so a new title reaches the client as a
|
||||
# session-info update instead of waiting for the next one.
|
||||
def _notify_title_update(_title: str, _source: str) -> None:
|
||||
if conn:
|
||||
loop.call_soon_threadsafe(
|
||||
asyncio.create_task,
|
||||
self._send_session_info_update(session_id),
|
||||
)
|
||||
|
||||
agent._on_session_title = _notify_title_update
|
||||
try:
|
||||
result = agent.run_conversation(
|
||||
user_message=user_content,
|
||||
@@ -1606,43 +2168,6 @@ class HermesACPAgent(acp.Agent):
|
||||
suppress_interrupt_response = interrupted and final_response.startswith(
|
||||
INTERRUPT_WAITING_FOR_MODEL_PREFIX
|
||||
)
|
||||
if final_response and not suppress_interrupt_response:
|
||||
try:
|
||||
from agent.title_generator import maybe_auto_title
|
||||
|
||||
def _notify_title_update(_title: str) -> None:
|
||||
if conn:
|
||||
loop.call_soon_threadsafe(
|
||||
asyncio.create_task,
|
||||
self._send_session_info_update(session_id),
|
||||
)
|
||||
|
||||
# Snapshot the runtime identity; the validator lets the
|
||||
# background titler skip its LLM call if the session's model
|
||||
# changed before it fires (#19027).
|
||||
_title_model = getattr(state.agent, "model", None)
|
||||
_title_provider = getattr(state.agent, "provider", None)
|
||||
maybe_auto_title(
|
||||
self.session_manager._get_db(),
|
||||
session_id,
|
||||
user_text,
|
||||
final_response,
|
||||
state.history,
|
||||
main_runtime={
|
||||
"model": getattr(state.agent, "model", None),
|
||||
"provider": getattr(state.agent, "provider", None),
|
||||
"base_url": getattr(state.agent, "base_url", None),
|
||||
"api_key": getattr(state.agent, "api_key", None),
|
||||
"api_mode": getattr(state.agent, "api_mode", None),
|
||||
},
|
||||
runtime_validator=lambda: (
|
||||
getattr(state.agent, "model", None) == _title_model
|
||||
and getattr(state.agent, "provider", None) == _title_provider
|
||||
),
|
||||
title_callback=_notify_title_update,
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("Failed to auto-title ACP session %s", session_id, exc_info=True)
|
||||
if (
|
||||
final_response
|
||||
and conn
|
||||
@@ -1765,8 +2290,26 @@ class HermesACPAgent(acp.Agent):
|
||||
if handler is None:
|
||||
return None # not a known command — let the LLM handle it
|
||||
|
||||
try:
|
||||
# Slash handlers run on the event-loop thread, OUTSIDE the per-turn
|
||||
# contextvars.copy_context() that pins the session cwd for the agent
|
||||
# call. ``/compress`` and ``/model`` reach code that REBUILDS the
|
||||
# system prompt (agent._build_system_prompt -> resolve_agent_cwd), so
|
||||
# an unpinned handler bakes the Hermes install tree into the session's
|
||||
# cached prompt — persisted, and therefore poisoning every later turn
|
||||
# even though the turn itself is pinned. Pin inside a fresh context so
|
||||
# the write can't leak into other concurrent ACP sessions and needs no
|
||||
# teardown.
|
||||
def _dispatch() -> str | None:
|
||||
try:
|
||||
from agent.runtime_cwd import set_session_cwd
|
||||
|
||||
set_session_cwd(state.cwd)
|
||||
except Exception:
|
||||
logger.debug("Could not pin ACP session cwd for slash command", exc_info=True)
|
||||
return handler(args, state)
|
||||
|
||||
try:
|
||||
return contextvars.copy_context().run(_dispatch)
|
||||
except Exception as e:
|
||||
logger.error("Slash command /%s error: %s", cmd, e, exc_info=True)
|
||||
return f"Error executing /{cmd}: {e}"
|
||||
@@ -1826,8 +2369,8 @@ class HermesACPAgent(acp.Agent):
|
||||
return "No tools available."
|
||||
lines = [f"Available tools ({len(tools)}):"]
|
||||
for t in tools:
|
||||
name = t.get("function", {}).get("name", "?")
|
||||
desc = t.get("function", {}).get("description", "")
|
||||
name = (t.get("function") or {}).get("name", "?")
|
||||
desc = (t.get("function") or {}).get("description", "")
|
||||
# Truncate long descriptions
|
||||
if len(desc) > 80:
|
||||
desc = desc[:77] + "..."
|
||||
@@ -1911,7 +2454,10 @@ class HermesACPAgent(acp.Agent):
|
||||
lines.append(f"Compression threshold: ~{threshold_tokens:,} tokens")
|
||||
|
||||
if getattr(agent, "compression_enabled", True) is False:
|
||||
lines.append("Compression is disabled for this agent.")
|
||||
lines.append(
|
||||
"Auto-compaction is disabled (compression.enabled: false); "
|
||||
"/compress still compresses manually."
|
||||
)
|
||||
else:
|
||||
lines.append("Tip: run /compress to compress manually before the threshold.")
|
||||
|
||||
@@ -1938,8 +2484,9 @@ class HermesACPAgent(acp.Agent):
|
||||
return "Nothing to compress — conversation is empty."
|
||||
try:
|
||||
agent = state.agent
|
||||
if not getattr(agent, "compression_enabled", True):
|
||||
return "Context compression is disabled for this agent."
|
||||
# No compression_enabled gate: the flag disables *automatic*
|
||||
# compaction only; manual /compress must keep working (matches
|
||||
# the CLI /compress and gateway handlers).
|
||||
if not hasattr(agent, "_compress_context"):
|
||||
return "Context compression not available for this agent."
|
||||
|
||||
@@ -1964,6 +2511,7 @@ class HermesACPAgent(acp.Agent):
|
||||
getattr(agent, "_cached_system_prompt", "") or "",
|
||||
approx_tokens=approx_tokens,
|
||||
task_id=state.session_id,
|
||||
force=True,
|
||||
)
|
||||
finally:
|
||||
agent._session_db = original_session_db
|
||||
|
||||
@@ -56,7 +56,18 @@ def _normalize_cwd_for_compare(cwd: str | None) -> str:
|
||||
elif re.match(r"^/mnt/[A-Za-z]/", expanded):
|
||||
expanded = f"/mnt/{expanded[5].lower()}/{expanded[7:]}"
|
||||
|
||||
return os.path.normpath(expanded)
|
||||
# Resolve symlink aliases so equivalent spellings of the same directory
|
||||
# compare equal — macOS reports editor workspaces as ``/var/...`` while
|
||||
# sessions get stored under ``/private/var/...`` (and ``/tmp`` vs
|
||||
# ``/private/tmp``), which made ACP history filters silently drop a
|
||||
# workspace's own sessions. ``os.path.realpath`` is lexical for missing
|
||||
# paths (strict=False), so cwds that don't exist on this host — e.g.
|
||||
# WSL-translated Windows drives — keep the previous normpath behavior.
|
||||
# Ported from PrimeIntellect-ai/prime-agent#628.
|
||||
try:
|
||||
return os.path.realpath(expanded)
|
||||
except OSError:
|
||||
return os.path.normpath(expanded)
|
||||
|
||||
|
||||
def _build_session_title(title: Any, preview: Any, cwd: str | None) -> str:
|
||||
@@ -480,16 +491,17 @@ class SessionManager:
|
||||
# fresh agent with _session_db_created=False (so the check above
|
||||
# is False) yet leave the durable archived transcript in place.
|
||||
# A full-history replace would DELETE those archived rows just
|
||||
# like the owned-agent case. Guard against it: when archived
|
||||
# rows exist, replace ONLY the live (active=1) set and leave the
|
||||
# archived turns untouched; otherwise the destructive replace is
|
||||
# safe (fresh create/fork with no archived history to lose).
|
||||
try:
|
||||
has_archived = db.has_archived_messages(state.session_id)
|
||||
except Exception:
|
||||
has_archived = False
|
||||
# like the owned-agent case. Guard against it by replacing ONLY
|
||||
# the live (active=1) set unconditionally: on a fresh
|
||||
# create/fork every row is active=1, so active-only replace is
|
||||
# behaviorally identical to the full replace — and when archived
|
||||
# rows DO exist they survive. An existence probe here
|
||||
# (has_archived_messages) would fail OPEN into the destructive
|
||||
# replace on any DB error and can race a concurrent
|
||||
# archive_and_compact — the same probe failure mode #80216's
|
||||
# /retry fix (gateway/slash_commands.py) deliberately avoids.
|
||||
db.replace_messages(
|
||||
state.session_id, state.history, active_only=has_archived
|
||||
state.session_id, state.history, active_only=True
|
||||
)
|
||||
except Exception:
|
||||
logger.warning("Failed to persist ACP session %s", state.session_id, exc_info=True)
|
||||
@@ -648,6 +660,30 @@ class SessionManager:
|
||||
logger.debug("ACP session falling back to default provider resolution", exc_info=True)
|
||||
|
||||
_register_task_cwd(session_id, cwd)
|
||||
|
||||
# Bounded wait for background MCP discovery so already-spawning fast
|
||||
# servers land in the agent's tool snapshot. ACP entry.py fires
|
||||
# discovery in a background daemon thread (start_background_mcp_discovery);
|
||||
# the agent snapshots tools once at build (run_agent/agent_init) and
|
||||
# never re-reads the registry, so without this join a reachable-but-
|
||||
# slow configured server would be invisible for the whole session.
|
||||
# ``ensure_mcp_discovery_before_agent_build`` also (re)starts discovery
|
||||
# when the entry.py spawn never ran or exited with zero connected
|
||||
# servers (the retry-after-zero-connected allowance), making this
|
||||
# construction site self-sufficient. Bounded by
|
||||
# ``mcp_discovery_timeout`` (config.yaml, default ~1.5s) so a dead
|
||||
# server can't block — servers that miss the bound are picked up by
|
||||
# the automatic late-refresh (see HermesACPAgent._schedule_mcp_late_refresh).
|
||||
try:
|
||||
from hermes_cli.mcp_startup import ensure_mcp_discovery_before_agent_build
|
||||
|
||||
ensure_mcp_discovery_before_agent_build(
|
||||
logger=logger,
|
||||
thread_name="acp-mcp-discovery",
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("ACP: bounded MCP discovery wait failed", exc_info=True)
|
||||
|
||||
agent = AIAgent(**kwargs)
|
||||
# Codex app-server sessions are spawned lazily on the first turn. Stamp
|
||||
# the ACP workspace onto the agent so the Codex runtime starts from the
|
||||
|
||||
@@ -75,7 +75,8 @@ _POLISHED_TOOLS = {
|
||||
"feishu_doc_read", "feishu_drive_list_comments", "feishu_drive_list_comment_replies",
|
||||
"feishu_drive_reply_comment", "feishu_drive_add_comment",
|
||||
"kanban_create", "kanban_show", "kanban_comment", "kanban_complete",
|
||||
"kanban_block", "kanban_link", "kanban_heartbeat",
|
||||
"kanban_block", "kanban_request_review", "kanban_request_changes",
|
||||
"kanban_link", "kanban_heartbeat",
|
||||
"yb_query_group_info", "yb_query_group_members", "yb_search_sticker",
|
||||
"yb_send_dm", "yb_send_sticker",
|
||||
}
|
||||
|
||||
@@ -1,16 +0,0 @@
|
||||
{
|
||||
"id": "hermes-agent",
|
||||
"name": "Hermes Agent",
|
||||
"version": "0.19.0",
|
||||
"description": "Self-improving open-source AI agent by Nous Research with ACP editor integration, persistent memory, skills, and rich tool support.",
|
||||
"repository": "https://github.com/NousResearch/hermes-agent",
|
||||
"website": "https://hermes-agent.nousresearch.com/docs/user-guide/features/acp",
|
||||
"authors": ["Nous Research"],
|
||||
"license": "MIT",
|
||||
"distribution": {
|
||||
"uvx": {
|
||||
"package": "hermes-agent[acp]==0.19.0",
|
||||
"args": ["hermes-acp"]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,8 +0,0 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 16 16" width="16" height="16" fill="none">
|
||||
<path d="M8 1.5v13" stroke="currentColor" stroke-width="1.5" stroke-linecap="round"/>
|
||||
<path d="M8 3.25c-2.35-1.4-4.7-.95-6.25.35 1.85-.2 3.8.2 5.55 1.55" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
|
||||
<path d="M8 3.25c2.35-1.4 4.7-.95 6.25.35-1.85-.2-3.8.2-5.55 1.55" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
|
||||
<path d="M8 13.25c-2.3-1-3.05-2.65-1.35-4.15-2 .8-2.35 2.95-.35 4" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
|
||||
<path d="M8 13.25c2.3-1 3.05-2.65 1.35-4.15 2 .8 2.35 2.95.35 4" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
|
||||
<circle cx="8" cy="1.8" r="1.1" fill="currentColor"/>
|
||||
</svg>
|
||||
|
Before Width: | Height: | Size: 882 B |
@@ -701,6 +701,18 @@ def redeem_codex_reset_credit(
|
||||
remaining = max(0, available - 1)
|
||||
plural = "s" if remaining != 1 else ""
|
||||
if code == "reset":
|
||||
# The redeemed reset restores the account's quota upstream — lift any
|
||||
# persisted pool cooldowns so Hermes doesn't keep the credential
|
||||
# frozen behind the now-stale ``last_error_reset_at`` (issue #43747).
|
||||
try:
|
||||
from hermes_cli.auth import clear_codex_pool_quota_cooldowns
|
||||
|
||||
clear_codex_pool_quota_cooldowns()
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"Failed to clear Codex pool cooldowns after reset redemption",
|
||||
exc_info=True,
|
||||
)
|
||||
return CodexResetRedeemResult(
|
||||
status="reset",
|
||||
message=(
|
||||
|
||||
@@ -0,0 +1,287 @@
|
||||
"""OpenAI-shape bridge shared by Hermes' ACP clients.
|
||||
|
||||
An ACP agent (``copilot --acp``, and the ACP CLIs that reach Hermes as
|
||||
providers) speaks the Agent Client Protocol, which has no OpenAI-style
|
||||
``tools``/``tool_calls`` channel: a prompt is text, and a response is text plus
|
||||
the agent's *own* tool notifications. Hermes' agentic surface — ``memory``,
|
||||
``todo``, ``skill_manage`` and friends — is dispatched from OpenAI-shaped
|
||||
``tool_calls``, so on an ACP provider it can only work if the schemas travel
|
||||
*into* the prompt as text and the calls are parsed back *out* of the response
|
||||
text.
|
||||
|
||||
``agent/copilot_acp_client.py`` already carried a private copy of that bridge.
|
||||
This module is that code, lifted verbatim into one place so every ACP client
|
||||
shares it instead of re-deriving the wire contract:
|
||||
|
||||
* :func:`render_tool_bridge_sections` — prompt sections describing the
|
||||
forwarded tools and the ``<tool_call>{...}</tool_call>`` contract.
|
||||
* :func:`extract_tool_calls_from_text` — parse those blocks back into
|
||||
``ChatCompletionMessageToolCall`` objects and return the response text with
|
||||
the blocks stripped.
|
||||
* :func:`completion_to_stream_chunks` — re-shape a one-shot ACP response as
|
||||
OpenAI stream chunks for callers that asked for ``stream=True`` (an ACP turn
|
||||
is inherently one-shot from Hermes' perspective).
|
||||
|
||||
The one axis clients differ on is *which* tools they forward, so
|
||||
``render_tool_bridge_sections`` takes an optional allowlist. A CLI with no tools
|
||||
of its own (Copilot) forwards everything Hermes offers; a CLI that is an
|
||||
autonomous agent with its own read/edit/execute tools must forward only Hermes'
|
||||
agent-level tools, because re-offering the overlapping ones makes Hermes re-run
|
||||
work the agent already finished.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from types import SimpleNamespace
|
||||
from typing import Any, Iterable
|
||||
|
||||
from openai.types.chat.chat_completion_message_tool_call import (
|
||||
ChatCompletionMessageToolCall,
|
||||
Function,
|
||||
)
|
||||
|
||||
TOOL_CALL_BLOCK_RE = re.compile(r"<tool_call>\s*(\{.*?\})\s*</tool_call>", re.DOTALL)
|
||||
TOOL_CALL_JSON_RE = re.compile(
|
||||
r"\{\s*\"id\"\s*:\s*\"[^\"]+\"\s*,\s*\"type\"\s*:\s*\"function\"\s*,\s*\"function\"\s*:\s*\{.*?\}\s*\}",
|
||||
re.DOTALL,
|
||||
)
|
||||
|
||||
# The contract sentence shared by every ACP client: how to emit a call.
|
||||
TOOL_CALL_CONTRACT = (
|
||||
"Available tools (OpenAI function schema). "
|
||||
"When using a tool, emit ONLY <tool_call>{...}</tool_call> with one JSON object "
|
||||
"containing id/type/function{name,arguments}. arguments must be a JSON string."
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"TOOL_CALL_BLOCK_RE",
|
||||
"TOOL_CALL_JSON_RE",
|
||||
"TOOL_CALL_CONTRACT",
|
||||
"StreamChunks",
|
||||
"build_openai_tool_call",
|
||||
"tool_specs_from_openai_tools",
|
||||
"render_tool_bridge_sections",
|
||||
"extract_tool_calls_from_text",
|
||||
"completion_to_stream_chunks",
|
||||
]
|
||||
|
||||
|
||||
class StreamChunks(list):
|
||||
"""Stream chunks that can still carry response-level attributes.
|
||||
|
||||
Hermes reads provider-level extras off the object returned by
|
||||
``chat.completions.create`` (e.g. ``hermes_projected_messages``, consumed by
|
||||
``agent/provider_projection.py``). A plain list of chunks would silently drop
|
||||
them on the ``stream=True`` path, so ACP clients return this instead and copy
|
||||
the extras onto it.
|
||||
"""
|
||||
|
||||
|
||||
def completion_to_stream_chunks(completion: SimpleNamespace) -> StreamChunks:
|
||||
"""Convert a one-shot ACP response into OpenAI-style stream chunks.
|
||||
|
||||
Response-level attributes other than ``choices``/``usage``/``model`` are
|
||||
copied onto the returned object so nothing a caller reads off the completion
|
||||
is lost when it asked to stream.
|
||||
"""
|
||||
choice = completion.choices[0]
|
||||
message = choice.message
|
||||
tool_call_deltas = None
|
||||
if message.tool_calls:
|
||||
tool_call_deltas = []
|
||||
for index, tool_call in enumerate(message.tool_calls):
|
||||
tool_call_deltas.append(
|
||||
SimpleNamespace(
|
||||
index=index,
|
||||
id=getattr(tool_call, "id", None),
|
||||
type=getattr(tool_call, "type", "function"),
|
||||
function=SimpleNamespace(
|
||||
name=getattr(tool_call.function, "name", None),
|
||||
arguments=getattr(tool_call.function, "arguments", None),
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
delta = SimpleNamespace(
|
||||
role="assistant",
|
||||
content=message.content or None,
|
||||
tool_calls=tool_call_deltas,
|
||||
reasoning_content=getattr(message, "reasoning_content", None),
|
||||
reasoning=getattr(message, "reasoning", None),
|
||||
)
|
||||
data_chunk = SimpleNamespace(
|
||||
choices=[
|
||||
SimpleNamespace(
|
||||
index=0,
|
||||
delta=delta,
|
||||
finish_reason=choice.finish_reason,
|
||||
)
|
||||
],
|
||||
model=completion.model,
|
||||
usage=None,
|
||||
)
|
||||
usage_chunk = SimpleNamespace(
|
||||
choices=[],
|
||||
model=completion.model,
|
||||
usage=completion.usage,
|
||||
)
|
||||
chunks = StreamChunks([data_chunk, usage_chunk])
|
||||
for key, value in vars(completion).items():
|
||||
if key not in ("choices", "usage", "model"):
|
||||
setattr(chunks, key, value)
|
||||
return chunks
|
||||
|
||||
|
||||
def build_openai_tool_call(
|
||||
*,
|
||||
call_id: str,
|
||||
name: str,
|
||||
arguments: str,
|
||||
) -> ChatCompletionMessageToolCall:
|
||||
"""Build an OpenAI-compatible tool-call object for downstream handling."""
|
||||
return ChatCompletionMessageToolCall(
|
||||
id=call_id,
|
||||
call_id=call_id,
|
||||
response_item_id=None,
|
||||
type="function",
|
||||
function=Function(name=name, arguments=arguments),
|
||||
)
|
||||
|
||||
|
||||
def tool_specs_from_openai_tools(
|
||||
tools: list[dict[str, Any]] | None,
|
||||
*,
|
||||
allowlist: Iterable[str] | None = None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Flatten OpenAI ``tools`` into ``{name, description, parameters}`` specs.
|
||||
|
||||
Malformed entries are skipped. When ``allowlist`` is given, only tools whose
|
||||
name is in it survive — that is how a client forwards just Hermes'
|
||||
agent-level tools instead of the whole toolset.
|
||||
"""
|
||||
allowed = {str(n).strip() for n in allowlist} if allowlist is not None else None
|
||||
specs: list[dict[str, Any]] = []
|
||||
for t in tools or []:
|
||||
if not isinstance(t, dict):
|
||||
continue
|
||||
fn = t.get("function") or {}
|
||||
if not isinstance(fn, dict):
|
||||
continue
|
||||
name = fn.get("name")
|
||||
if not isinstance(name, str) or not name.strip():
|
||||
continue
|
||||
name = name.strip()
|
||||
if allowed is not None and name not in allowed:
|
||||
continue
|
||||
specs.append(
|
||||
{
|
||||
"name": name,
|
||||
"description": fn.get("description", ""),
|
||||
"parameters": fn.get("parameters", {}),
|
||||
}
|
||||
)
|
||||
return specs
|
||||
|
||||
|
||||
def render_tool_bridge_sections(
|
||||
tools: list[dict[str, Any]] | None,
|
||||
tool_choice: Any = None,
|
||||
*,
|
||||
allowlist: Iterable[str] | None = None,
|
||||
) -> list[str]:
|
||||
"""Prompt sections that carry the forwarded tool schemas + choice hint.
|
||||
|
||||
Returns an empty list when no tool survives filtering and no choice hint was
|
||||
requested, so callers can splice the result into their section list
|
||||
unconditionally.
|
||||
"""
|
||||
specs = tool_specs_from_openai_tools(tools, allowlist=allowlist)
|
||||
sections: list[str] = []
|
||||
if specs:
|
||||
sections.append(
|
||||
TOOL_CALL_CONTRACT + "\n" + json.dumps(specs, ensure_ascii=False)
|
||||
)
|
||||
if tool_choice is not None:
|
||||
sections.append(f"Tool choice hint: {json.dumps(tool_choice, ensure_ascii=False)}")
|
||||
return sections
|
||||
|
||||
|
||||
def extract_tool_calls_from_text(
|
||||
text: str,
|
||||
) -> tuple[list[ChatCompletionMessageToolCall], str]:
|
||||
"""Pull ``<tool_call>`` blocks out of an ACP response.
|
||||
|
||||
Returns ``(tool_calls, cleaned_text)`` where ``cleaned_text`` is the
|
||||
response with the consumed blocks removed, so the assistant message doesn't
|
||||
show raw JSON to the user.
|
||||
"""
|
||||
if not isinstance(text, str) or not text.strip():
|
||||
return [], ""
|
||||
|
||||
extracted: list[ChatCompletionMessageToolCall] = []
|
||||
consumed_spans: list[tuple[int, int]] = []
|
||||
|
||||
def _try_add_tool_call(raw_json: str) -> None:
|
||||
try:
|
||||
obj = json.loads(raw_json)
|
||||
except Exception:
|
||||
return
|
||||
if not isinstance(obj, dict):
|
||||
return
|
||||
fn = obj.get("function")
|
||||
if not isinstance(fn, dict):
|
||||
return
|
||||
fn_name = fn.get("name")
|
||||
if not isinstance(fn_name, str) or not fn_name.strip():
|
||||
return
|
||||
fn_args = fn.get("arguments", "{}")
|
||||
if not isinstance(fn_args, str):
|
||||
fn_args = json.dumps(fn_args, ensure_ascii=False)
|
||||
call_id = obj.get("id")
|
||||
if not isinstance(call_id, str) or not call_id.strip():
|
||||
call_id = f"acp_call_{len(extracted)+1}"
|
||||
|
||||
extracted.append(
|
||||
build_openai_tool_call(
|
||||
call_id=call_id,
|
||||
name=fn_name.strip(),
|
||||
arguments=fn_args,
|
||||
)
|
||||
)
|
||||
|
||||
for m in TOOL_CALL_BLOCK_RE.finditer(text):
|
||||
raw = m.group(1)
|
||||
_try_add_tool_call(raw)
|
||||
consumed_spans.append((m.start(), m.end()))
|
||||
|
||||
# Only try bare-JSON fallback when no XML blocks were found.
|
||||
if not extracted:
|
||||
for m in TOOL_CALL_JSON_RE.finditer(text):
|
||||
raw = m.group(0)
|
||||
_try_add_tool_call(raw)
|
||||
consumed_spans.append((m.start(), m.end()))
|
||||
|
||||
if not consumed_spans:
|
||||
return extracted, text.strip()
|
||||
|
||||
consumed_spans.sort()
|
||||
merged: list[tuple[int, int]] = []
|
||||
for start, end in consumed_spans:
|
||||
if not merged or start > merged[-1][1]:
|
||||
merged.append((start, end))
|
||||
else:
|
||||
merged[-1] = (merged[-1][0], max(merged[-1][1], end))
|
||||
|
||||
parts: list[str] = []
|
||||
cursor = 0
|
||||
for start, end in merged:
|
||||
if cursor < start:
|
||||
parts.append(text[cursor:start])
|
||||
cursor = max(cursor, end)
|
||||
if cursor < len(text):
|
||||
parts.append(text[cursor:])
|
||||
|
||||
cleaned = "\n".join(p.strip() for p in parts if p and p.strip()).strip()
|
||||
return extracted, cleaned
|
||||
@@ -23,7 +23,26 @@ from urllib.parse import urlparse
|
||||
|
||||
from hermes_constants import get_hermes_home
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
from utils import base_url_host_matches, normalize_proxy_env_vars
|
||||
from utils import base_url_host_matches, base_url_hostname, normalize_proxy_env_vars
|
||||
from agent.secret_scope import get_secret as _get_secret
|
||||
|
||||
try:
|
||||
import hermes_cli as _hermes_cli
|
||||
|
||||
_HERMES_VERSION = str(_hermes_cli.__version__)
|
||||
except Exception:
|
||||
_HERMES_VERSION = "0.0.0"
|
||||
|
||||
|
||||
def _getenv(name: str, default: str = "") -> str:
|
||||
"""Profile-scoped replacement for os.getenv on credential reads.
|
||||
|
||||
Routes through the secret scope (Workstream A): identical to os.getenv
|
||||
when multiplexing is off, scope-aware (and fail-closed on an unscoped
|
||||
read) when on. Mirrors the same wrapper in hermes_cli/runtime_provider.py.
|
||||
"""
|
||||
val = _get_secret(name, default)
|
||||
return val if val is not None else default
|
||||
|
||||
# NOTE: `import anthropic` is deliberately NOT at module top — the SDK pulls
|
||||
# ~220 ms of imports (anthropic.types, anthropic.lib.tools._beta_runner, etc.)
|
||||
@@ -113,6 +132,17 @@ _NO_XHIGH_CLAUDE_SUBSTRINGS = (
|
||||
"claude-sonnet-4-6", "claude-sonnet-4.6",
|
||||
)
|
||||
|
||||
# Adaptive Claude families that REJECT a thinking disable — thinking is
|
||||
# mandatory and ``thinking: {"type": "disabled"}`` answers HTTP 400. The Portal
|
||||
# catalog flags the same families with ``reasoning.mandatory``.
|
||||
#
|
||||
# Unlike the two lists above, the failure here is asymmetric: a missing entry
|
||||
# 400s the turn, while a spurious one only leaves thinking on. When in doubt,
|
||||
# add the family.
|
||||
_MANDATORY_THINKING_CLAUDE_SUBSTRINGS = (
|
||||
"claude-fable",
|
||||
)
|
||||
|
||||
|
||||
def _is_claude_model(model: str | None) -> bool:
|
||||
return "claude" in (model or "").lower()
|
||||
@@ -279,6 +309,32 @@ def _supports_xhigh_effort(model: str) -> bool:
|
||||
return not any(v in m for v in _NO_XHIGH_CLAUDE_SUBSTRINGS)
|
||||
|
||||
|
||||
def _accepts_thinking_disable(model: str) -> bool:
|
||||
"""Return True when *model* accepts an explicit thinking disable.
|
||||
|
||||
Adaptive Claude models default to thinking ON, so "thinking off" only
|
||||
takes effect if we actively send ``thinking: {"type": "disabled"}`` —
|
||||
omitting the parameter leaves the upstream default in place and the model
|
||||
thinks anyway. Reasoning-mandatory families reject the disable outright
|
||||
with an HTTP 400, so they keep the omit-everything behavior.
|
||||
|
||||
Legacy manual-thinking Claude models are excluded because they need no
|
||||
disable: thinking is opt-in there via ``budget_tokens``, so not sending
|
||||
the block already means off.
|
||||
|
||||
Scoped to Claude deliberately. Kimi/Moonshot endpoints also speak the
|
||||
adaptive contract, but their documented disable behavior is omission
|
||||
(#13848) and they are not part of this bug; sending them a new parameter
|
||||
on the strength of Claude's contract would be a guess.
|
||||
"""
|
||||
if not _is_claude_model(model):
|
||||
return False
|
||||
if not _supports_adaptive_thinking(model):
|
||||
return False
|
||||
m = model.lower()
|
||||
return not any(v in m for v in _MANDATORY_THINKING_CLAUDE_SUBSTRINGS)
|
||||
|
||||
|
||||
def _forbids_sampling_params(model: str) -> bool:
|
||||
"""Return True for models that 400 on any non-default temperature/top_p/top_k.
|
||||
|
||||
@@ -368,7 +424,7 @@ def _detect_claude_code_version() -> str:
|
||||
try:
|
||||
result = _sp.run(
|
||||
[cmd, "--version"],
|
||||
capture_output=True, text=True, timeout=5,
|
||||
capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5,
|
||||
)
|
||||
if result.returncode == 0 and result.stdout.strip():
|
||||
# Output is like "2.1.74 (Claude Code)" or just "2.1.74"
|
||||
@@ -455,6 +511,11 @@ def _is_kimi_coding_endpoint(base_url: str | None) -> bool:
|
||||
return normalized.rstrip("/").lower().startswith("https://api.kimi.com/coding")
|
||||
|
||||
|
||||
def _is_opencode_endpoint(base_url: str | None) -> bool:
|
||||
"""Return True for OpenCode's Zen/Go relay (opencode.ai)."""
|
||||
return base_url_host_matches(base_url or "", "opencode.ai")
|
||||
|
||||
|
||||
# Model-name prefixes that identify the Kimi / Moonshot family. Covers
|
||||
# - official slugs: ``kimi-k2.5``, ``kimi_thinking``, ``moonshot-v1-8k``
|
||||
# - common release lines: ``k1.5-...``, ``k2-thinking``, ``k25-...``, ``k2.5-...``,
|
||||
@@ -546,15 +607,49 @@ def _is_deepseek_anthropic_endpoint(base_url: str | None) -> bool:
|
||||
return "/anthropic" in normalized.rstrip("/").lower()
|
||||
|
||||
|
||||
def _is_nous_portal_endpoint(base_url: str | None) -> bool:
|
||||
"""Return True for Nous Portal's Anthropic Messages route.
|
||||
|
||||
Portal serves its ``anthropic/*`` catalog natively at
|
||||
``https://inference-api.nousresearch.com/v1/messages``. Portal-specific
|
||||
behaviours key off this: Bearer JWT auth, verbatim catalog model ids,
|
||||
and native thinking-signature replay.
|
||||
|
||||
Trusted hosts only:
|
||||
|
||||
1. Prod hostname ``inference-api.nousresearch.com``
|
||||
2. The operator-set ``NOUS_INFERENCE_BASE_URL`` hostname (staging/preview)
|
||||
|
||||
Lookalikes such as ``inference-api.nousresearch.com.attacker.test`` are
|
||||
rejected (hostname match, not substring).
|
||||
"""
|
||||
if base_url_host_matches(base_url or "", "inference-api.nousresearch.com"):
|
||||
return True
|
||||
try:
|
||||
from hermes_cli.auth import _nous_inference_env_override
|
||||
|
||||
override = _nous_inference_env_override()
|
||||
except Exception:
|
||||
return False
|
||||
if not override:
|
||||
return False
|
||||
# Exact host equality (not subdomain) so the env override can't broaden
|
||||
# into sibling hosts the operator did not set.
|
||||
override_host = base_url_hostname(override)
|
||||
return bool(override_host) and base_url_hostname(base_url or "") == override_host
|
||||
|
||||
|
||||
def _requires_bearer_auth(base_url: str | None) -> bool:
|
||||
"""Return True for Anthropic-compatible providers that require Bearer auth.
|
||||
|
||||
Some third-party /anthropic endpoints implement Anthropic's Messages API but
|
||||
require Authorization: Bearer instead of Anthropic's native x-api-key header.
|
||||
MiniMax's global and China Anthropic-compatible endpoints, Azure AI
|
||||
Foundry's Anthropic-style endpoint, and Palantir Foundry's LLM proxy
|
||||
follow this pattern.
|
||||
Foundry's Anthropic-style endpoint, Palantir Foundry's LLM proxy, and Nous
|
||||
Portal's Messages route follow this pattern.
|
||||
"""
|
||||
if _is_nous_portal_endpoint(base_url):
|
||||
return True
|
||||
normalized = _normalize_base_url_text(base_url)
|
||||
if not normalized:
|
||||
return False
|
||||
@@ -567,6 +662,10 @@ def _requires_bearer_auth(base_url: str | None) -> bool:
|
||||
# Hostname match (not substring) so e.g. evil.com/palantirfoundry
|
||||
# paths don't trigger Bearer auth.
|
||||
or base_url_host_matches(normalized, "palantirfoundry.com")
|
||||
# CommandCode's /provider/v1/messages endpoint uses Bearer auth,
|
||||
# not Anthropic's native x-api-key header. Hostname match for the
|
||||
# same reason as above.
|
||||
or base_url_host_matches(normalized, "api.commandcode.ai")
|
||||
)
|
||||
|
||||
|
||||
@@ -721,7 +820,11 @@ def _build_anthropic_client_with_bearer_hook(
|
||||
if common_betas:
|
||||
kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)}
|
||||
|
||||
return _anthropic_sdk.Anthropic(**kwargs)
|
||||
client = _anthropic_sdk.Anthropic(**kwargs)
|
||||
# Same env-inference trap as build_anthropic_client: auth_token-only
|
||||
# construction would otherwise also send ANTHROPIC_API_KEY as X-Api-Key.
|
||||
client.api_key = None
|
||||
return client
|
||||
|
||||
|
||||
def build_anthropic_client(
|
||||
@@ -807,12 +910,18 @@ def build_anthropic_client(
|
||||
)
|
||||
|
||||
if _is_kimi_coding_endpoint(base_url):
|
||||
# Kimi's /coding endpoint requires User-Agent: claude-code/0.1.0
|
||||
# to be recognized as a valid Coding Agent. Without it, returns 403.
|
||||
# Check this BEFORE _requires_bearer_auth since both match api.kimi.com/coding.
|
||||
# Kimi's /coding endpoint requires a non-empty User-Agent to be
|
||||
# recognized as a valid Coding Agent. Originally we sent
|
||||
# ``claude-code/0.1.0`` (the minimum that avoided a 403), but the Kimi
|
||||
# team asked us to identify ourselves properly so they can attribute
|
||||
# traffic correctly. Send the same attribution header set we send to
|
||||
# OpenRouter, Vercel AI Gateway, and Fireworks:
|
||||
# HTTP-Referer + X-Title + HermesAgent User-Agent.
|
||||
kwargs["api_key"] = api_key
|
||||
kwargs["default_headers"] = {
|
||||
"User-Agent": "claude-code/0.1.0",
|
||||
"HTTP-Referer": "https://hermes-agent.nousresearch.com",
|
||||
"X-Title": "Hermes Agent",
|
||||
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
|
||||
**( {"anthropic-beta": ",".join(common_betas)} if common_betas else {} )
|
||||
}
|
||||
elif _requires_bearer_auth(normalized_base_url):
|
||||
@@ -850,7 +959,28 @@ def build_anthropic_client(
|
||||
if common_betas:
|
||||
kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)}
|
||||
|
||||
return _anthropic_sdk.Anthropic(**kwargs)
|
||||
if _is_opencode_endpoint(base_url):
|
||||
# OpenCode identifies clients by request headers, like OpenRouter does.
|
||||
# The OpenAI-wire paths pick these up from profile.default_headers
|
||||
# (plugins/model-providers/opencode-zen), but the Anthropic Messages
|
||||
# route builds its client right here and never sees the profile. Merge
|
||||
# the same set on top of whatever auth branch ran above.
|
||||
headers = dict(kwargs.get("default_headers") or {})
|
||||
headers.setdefault("HTTP-Referer", "https://hermes-agent.nousresearch.com")
|
||||
headers.setdefault("X-Title", "Hermes Agent")
|
||||
headers.setdefault("User-Agent", f"HermesAgent/{_HERMES_VERSION}")
|
||||
kwargs["default_headers"] = headers
|
||||
|
||||
client = _anthropic_sdk.Anthropic(**kwargs)
|
||||
# Bearer-only construction leaves ``api_key`` unset, so the SDK fills it
|
||||
# from ``ANTHROPIC_API_KEY`` (Hermes loads that into the process env from
|
||||
# ``~/.hermes/.env``). The result is dual auth —
|
||||
# ``X-Api-Key: sk-ant-…`` *and* ``Authorization: Bearer <portal-jwt>`` —
|
||||
# on every Portal / MiniMax / OAuth Messages request. Clear the env-filled
|
||||
# key whenever we intentionally authenticated via auth_token alone.
|
||||
if "auth_token" in kwargs and "api_key" not in kwargs:
|
||||
client.api_key = None
|
||||
return client
|
||||
|
||||
|
||||
def build_anthropic_bedrock_client(region: str):
|
||||
@@ -914,7 +1044,7 @@ def _read_claude_code_credentials_from_keychain() -> Optional[Dict[str, Any]]:
|
||||
"-s", "Claude Code-credentials",
|
||||
"-w"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
timeout=5,
|
||||
stdin=subprocess.DEVNULL,
|
||||
)
|
||||
@@ -1275,7 +1405,7 @@ def _resolve_anthropic_pool_token() -> Optional[str]:
|
||||
# to auth.json or trigger a network refresh from a bare resolve. select()
|
||||
# is deliberately NOT used — it runs clear_expired=True, refresh=True,
|
||||
# which would violate this read-only contract.
|
||||
entries = pool._available_entries(clear_expired=False, refresh=False)
|
||||
entries, _pending = pool._available_entries(clear_expired=False, refresh=False)
|
||||
except Exception:
|
||||
logger.debug("Failed to read Anthropic credential_pool", exc_info=True)
|
||||
return None
|
||||
@@ -1301,47 +1431,55 @@ def resolve_anthropic_token() -> Optional[str]:
|
||||
Priority:
|
||||
1. ANTHROPIC_TOKEN env var (OAuth/setup token saved by Hermes)
|
||||
2. CLAUDE_CODE_OAUTH_TOKEN env var
|
||||
3. Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json)
|
||||
3. ANTHROPIC_API_KEY env var (explicit regular API key)
|
||||
4. Claude Code credentials (~/.claude.json or ~/.claude/.credentials.json)
|
||||
— with automatic refresh if expired and a refresh token is available
|
||||
4. Anthropic credential_pool OAuth entry (~/.hermes/auth.json)
|
||||
5. ANTHROPIC_API_KEY env var (regular API key, or legacy fallback)
|
||||
5. Anthropic credential_pool OAuth entry (~/.hermes/auth.json)
|
||||
|
||||
Returns the token string or None.
|
||||
"""
|
||||
creds = read_claude_code_credentials()
|
||||
creds: Optional[Dict[str, Any]] = None
|
||||
creds_loaded = False
|
||||
|
||||
def _read_creds() -> Optional[Dict[str, Any]]:
|
||||
nonlocal creds, creds_loaded
|
||||
if not creds_loaded:
|
||||
creds = read_claude_code_credentials()
|
||||
creds_loaded = True
|
||||
return creds
|
||||
|
||||
# 1. Hermes-managed OAuth/setup token env var
|
||||
token = os.getenv("ANTHROPIC_TOKEN", "").strip()
|
||||
token = _getenv("ANTHROPIC_TOKEN").strip()
|
||||
if token:
|
||||
preferred = _prefer_refreshable_claude_code_token(token, creds)
|
||||
preferred = _prefer_refreshable_claude_code_token(token, _read_creds())
|
||||
if preferred:
|
||||
return preferred
|
||||
return token
|
||||
|
||||
# 2. CLAUDE_CODE_OAUTH_TOKEN (used by Claude Code for setup-tokens)
|
||||
cc_token = os.getenv("CLAUDE_CODE_OAUTH_TOKEN", "").strip()
|
||||
cc_token = _getenv("CLAUDE_CODE_OAUTH_TOKEN").strip()
|
||||
if cc_token:
|
||||
preferred = _prefer_refreshable_claude_code_token(cc_token, creds)
|
||||
preferred = _prefer_refreshable_claude_code_token(cc_token, _read_creds())
|
||||
if preferred:
|
||||
return preferred
|
||||
return cc_token
|
||||
|
||||
# 3. Claude Code credential file
|
||||
resolved_claude_token = _resolve_claude_code_token_from_credentials(creds)
|
||||
# 3. Regular API key. An explicit user-configured key must not be shadowed
|
||||
# by auto-discovered Claude Code or credential-pool OAuth credentials.
|
||||
api_key = _getenv("ANTHROPIC_API_KEY").strip()
|
||||
if api_key:
|
||||
return api_key
|
||||
|
||||
# 4. Claude Code credential file
|
||||
resolved_claude_token = _resolve_claude_code_token_from_credentials(_read_creds())
|
||||
if resolved_claude_token:
|
||||
return resolved_claude_token
|
||||
|
||||
# 4. Hermes credential_pool OAuth entry.
|
||||
# 5. Hermes credential_pool OAuth entry.
|
||||
resolved_pool_token = _resolve_anthropic_pool_token()
|
||||
if resolved_pool_token:
|
||||
return resolved_pool_token
|
||||
|
||||
# 5. Regular API key, or a legacy OAuth token saved in ANTHROPIC_API_KEY.
|
||||
# This remains as a compatibility fallback for pre-migration Hermes configs.
|
||||
api_key = os.getenv("ANTHROPIC_API_KEY", "").strip()
|
||||
if api_key:
|
||||
return api_key
|
||||
|
||||
return None
|
||||
|
||||
|
||||
@@ -1381,7 +1519,7 @@ def run_oauth_setup_token() -> Optional[str]:
|
||||
|
||||
# Check env vars that may have been set
|
||||
for env_var in ("CLAUDE_CODE_OAUTH_TOKEN", "ANTHROPIC_TOKEN"):
|
||||
val = os.getenv(env_var, "").strip()
|
||||
val = _getenv(env_var).strip()
|
||||
if val:
|
||||
return val
|
||||
|
||||
@@ -1800,7 +1938,16 @@ def _to_plain_data(value: Any, *, _depth: int = 0, _path: Optional[set] = None)
|
||||
|
||||
if hasattr(value, "model_dump"):
|
||||
_path.add(obj_id)
|
||||
result = _to_plain_data(value.model_dump(), _depth=_depth + 1, _path=_path)
|
||||
try:
|
||||
# warnings=False: content blocks from the streaming accumulator
|
||||
# (ParsedTextBlock et al.) trip pydantic's serializer-mismatch
|
||||
# UserWarning against the generic Message union; the dump itself
|
||||
# is correct, and the warning leaks to the user's terminal.
|
||||
dumped = value.model_dump(warnings=False)
|
||||
except TypeError:
|
||||
# Duck-typed model_dump without pydantic's signature.
|
||||
dumped = value.model_dump()
|
||||
result = _to_plain_data(dumped, _depth=_depth + 1, _path=_path)
|
||||
_path.discard(obj_id)
|
||||
return result
|
||||
if isinstance(value, dict):
|
||||
@@ -1881,6 +2028,28 @@ def _content_parts_to_anthropic_blocks(parts: Any) -> List[Dict[str, Any]]:
|
||||
return out
|
||||
|
||||
|
||||
_EMPTY_TEXT_PLACEHOLDER = "(empty)"
|
||||
|
||||
|
||||
def _safe_text(text: Any) -> str:
|
||||
"""Return ``text`` if it's non-whitespace, else a non-whitespace placeholder.
|
||||
|
||||
The Anthropic Messages API rejects requests where a text content block is
|
||||
empty or whitespace-only (HTTP 400 "text content blocks must contain
|
||||
non-whitespace text"). When such a block gets stored in session history —
|
||||
e.g. produced by context compression — it is replayed verbatim on every
|
||||
subsequent turn, permanently wedging the session. Coercing to a
|
||||
non-whitespace placeholder is self-healing: the next API call recovers.
|
||||
|
||||
Mirrors ``bedrock_adapter._safe_text`` (#9486); ref #69512.
|
||||
"""
|
||||
if text is None:
|
||||
return _EMPTY_TEXT_PLACEHOLDER
|
||||
if not isinstance(text, str):
|
||||
text = str(text)
|
||||
return text if text.strip() else _EMPTY_TEXT_PLACEHOLDER
|
||||
|
||||
|
||||
def _sanitize_replay_block(b: Dict[str, Any]) -> Optional[Dict[str, Any]]:
|
||||
"""Strip output-only fields from a stored Anthropic content block so it is
|
||||
valid as REQUEST input on replay.
|
||||
@@ -1898,7 +2067,18 @@ def _sanitize_replay_block(b: Dict[str, Any]) -> Optional[Dict[str, Any]]:
|
||||
return None
|
||||
btype = b.get("type")
|
||||
if btype == "text":
|
||||
out: Dict[str, Any] = {"type": "text", "text": b.get("text", "")}
|
||||
text_val = b.get("text", "")
|
||||
# Bedrock and strict Anthropic-compatible endpoints reject text
|
||||
# blocks where "text" is empty or whitespace-only (#69512). Drop the
|
||||
# blank block (the caller relocates any cache_control it carried and
|
||||
# falls back to a non-whitespace placeholder when nothing survives)
|
||||
# rather than coercing in place — a coerced "(empty)" block would be
|
||||
# model-visible noise next to surviving thinking/tool_use blocks.
|
||||
# Type-safe: captured blocks can carry text=None from an invalid
|
||||
# upstream payload, which a bare .strip() would crash on.
|
||||
if not isinstance(text_val, str) or not text_val.strip():
|
||||
return None
|
||||
out: Dict[str, Any] = {"type": "text", "text": text_val}
|
||||
# citations is input-valid ONLY when it's a non-empty list; the SDK
|
||||
# emits citations=None on responses, which the input schema rejects.
|
||||
cits = b.get("citations")
|
||||
@@ -1986,9 +2166,17 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
|
||||
parsed_args = {}
|
||||
redacted_input_by_id[_sanitize_tool_id(tc.get("id", ""))] = parsed_args
|
||||
replayed: List[Dict[str, Any]] = []
|
||||
_relocated_replay_cache_control = None
|
||||
_dropped_blank_text = False
|
||||
for b in ordered_blocks:
|
||||
clean = _sanitize_replay_block(b)
|
||||
if clean is None:
|
||||
if isinstance(b, dict) and b.get("type") == "text":
|
||||
_dropped_blank_text = True
|
||||
if isinstance(b, dict) and isinstance(b.get("cache_control"), dict):
|
||||
# A dropped blank text block can still carry the cache
|
||||
# breakpoint marker -- relocate it rather than losing it.
|
||||
_relocated_replay_cache_control = b["cache_control"]
|
||||
continue
|
||||
if clean.get("type") == "tool_use":
|
||||
# Override raw (un-redacted) input with the redacted copy when
|
||||
@@ -1998,20 +2186,90 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
|
||||
if redacted is not None:
|
||||
clean["input"] = redacted
|
||||
replayed.append(clean)
|
||||
# When every text block was blank and nothing cacheable survived
|
||||
# (e.g. signed thinking + a blank text block, or a SOLE blank
|
||||
# cache-marked block), emit the non-whitespace placeholder so the
|
||||
# replayed message stays schema-valid (#69512) and a relocated cache
|
||||
# marker still has a carrier instead of being silently lost.
|
||||
_has_cacheable_replay = any(
|
||||
isinstance(b, dict) and b.get("type") in {"text", "tool_use"}
|
||||
for b in replayed
|
||||
)
|
||||
if not _has_cacheable_replay and (
|
||||
_dropped_blank_text or _relocated_replay_cache_control is not None
|
||||
):
|
||||
replayed.append({"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER})
|
||||
if replayed:
|
||||
if _relocated_replay_cache_control is not None:
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
replayed, _relocated_replay_cache_control
|
||||
)
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
replayed, m.get("cache_control")
|
||||
)
|
||||
# apply_anthropic_cache_control marks an assistant turn with
|
||||
# non-empty text by writing cache_control INTO ``content`` (see
|
||||
# _apply_cache_marker's list branch), not at the top level. This
|
||||
# branch rebuilds the message from ordered_blocks and never reads
|
||||
# ``content``, so that marker would be dropped -- and because
|
||||
# _can_carry_marker already counted this message as a carrier, the
|
||||
# breakpoint is burned rather than relocated. #56195 covered the
|
||||
# complementary shape (blank content -> top-level marker); this is
|
||||
# the interleaved thinking + preamble-text + tool_use shape.
|
||||
_inline_cc = None
|
||||
_msg_content = m.get("content")
|
||||
if isinstance(_msg_content, list):
|
||||
for _blk in _msg_content:
|
||||
if isinstance(_blk, dict) and isinstance(
|
||||
_blk.get("cache_control"), dict
|
||||
):
|
||||
_inline_cc = _blk["cache_control"]
|
||||
break
|
||||
if _inline_cc is not None:
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
replayed, _inline_cc
|
||||
)
|
||||
return {"role": "assistant", "content": replayed}
|
||||
|
||||
blocks = _extract_preserved_thinking_blocks(m)
|
||||
# Cache markers dropped along with a blank block are relocated onto the
|
||||
# last surviving cacheable block below (via
|
||||
# _apply_assistant_cache_control_to_last_cacheable_block), rather than
|
||||
# lost -- prompt_caching.py's _apply_cache_marker() sets cache_control
|
||||
# directly on content[-1] for list content, so if that last part happens
|
||||
# to be blank text, dropping it silently would lose the breakpoint.
|
||||
_relocated_cache_control = None
|
||||
if content:
|
||||
if isinstance(content, list):
|
||||
converted_content = _convert_content_to_anthropic(content)
|
||||
if isinstance(converted_content, list):
|
||||
blocks.extend(converted_content)
|
||||
# Bedrock and strict Anthropic-compatible endpoints reject
|
||||
# text blocks where "text" is empty or whitespace-only. The
|
||||
# ordered-replay path enforces the same invariant via
|
||||
# _sanitize_replay_block(). Type-safe against ANY invalid
|
||||
# "text" value from an upstream payload -- None, or a
|
||||
# truthy non-string like an int -- not just None: checking
|
||||
# isinstance() first (rather than `blk.get("text") or ""`)
|
||||
# means a non-string value is treated as blank/invalid
|
||||
# instead of reaching .strip() and raising AttributeError.
|
||||
for blk in converted_content:
|
||||
_blk_text = blk.get("text") if isinstance(blk, dict) else None
|
||||
if (
|
||||
isinstance(blk, dict)
|
||||
and blk.get("type") == "text"
|
||||
and (not isinstance(_blk_text, str) or not _blk_text.strip())
|
||||
):
|
||||
if isinstance(blk.get("cache_control"), dict):
|
||||
_relocated_cache_control = blk["cache_control"]
|
||||
continue
|
||||
blocks.append(blk)
|
||||
else:
|
||||
blocks.append({"type": "text", "text": str(content)})
|
||||
# Scalar (non-list) content: a whitespace-only string is the
|
||||
# same invalid-payload case as an empty list block -- drop it
|
||||
# rather than emitting a blank text block.
|
||||
text_str = str(content)
|
||||
if text_str.strip():
|
||||
blocks.append({"type": "text", "text": text_str})
|
||||
for tc in m.get("tool_calls", []):
|
||||
if not tc or not isinstance(tc, dict):
|
||||
continue
|
||||
@@ -2027,9 +2285,6 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"name": fn.get("name", ""),
|
||||
"input": parsed_args,
|
||||
})
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
blocks, m.get("cache_control")
|
||||
)
|
||||
# Kimi's /coding endpoint (Anthropic protocol) requires assistant
|
||||
# tool-call messages to carry reasoning_content when thinking is
|
||||
# enabled server-side. Preserve it as a thinking block so Kimi
|
||||
@@ -2055,10 +2310,26 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
|
||||
)
|
||||
if isinstance(reasoning_content, str) and not _already_has_thinking:
|
||||
blocks.insert(0, {"type": "thinking", "thinking": reasoning_content})
|
||||
# Anthropic rejects empty assistant content
|
||||
effective = blocks or content
|
||||
if not effective or effective == "":
|
||||
effective = [{"type": "text", "text": "(empty)"}]
|
||||
# Anthropic rejects empty assistant content. IMPORTANT: fall back only
|
||||
# to the placeholder, never to the raw `content` variable -- `content`
|
||||
# is the UNFILTERED original message content, and can itself be exactly
|
||||
# the blank/whitespace-only payload the filtering above just removed
|
||||
# (a sole blank text block, or scalar whitespace with no tool_calls).
|
||||
# `blocks or content` there would silently restore the invalid provider
|
||||
# payload this function exists to prevent (#69512).
|
||||
effective = blocks if blocks else [{"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}]
|
||||
# Applied here (after the empty-fallback resolution) rather than
|
||||
# earlier against `blocks` directly, so a cache_control relocated from
|
||||
# a dropped blank block that was the ONLY block still lands on the
|
||||
# (empty) placeholder instead of being silently lost when blocks was
|
||||
# empty at the point the marker would otherwise have been applied.
|
||||
if _relocated_cache_control is not None:
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
effective, _relocated_cache_control
|
||||
)
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
effective, m.get("cache_control")
|
||||
)
|
||||
return {"role": "assistant", "content": effective}
|
||||
|
||||
|
||||
@@ -2128,13 +2399,14 @@ def _convert_user_message(content: Any) -> Dict[str, Any]:
|
||||
"""Validate and convert a user message to anthropic format."""
|
||||
if isinstance(content, list):
|
||||
converted_blocks = _convert_content_to_anthropic(content)
|
||||
if not converted_blocks or all(
|
||||
(b.get("text") or "").strip() == ""
|
||||
for b in converted_blocks
|
||||
if isinstance(b, dict) and b.get("type") == "text"
|
||||
):
|
||||
converted_blocks = [{"type": "text", "text": "(empty message)"}]
|
||||
return {"role": "user", "content": converted_blocks}
|
||||
kept_blocks = _fix_blank_text_blocks_in_list(
|
||||
converted_blocks,
|
||||
placeholder_text="(empty message)",
|
||||
msg_index=-1,
|
||||
role="user",
|
||||
location="_convert_user_message",
|
||||
)
|
||||
return {"role": "user", "content": kept_blocks}
|
||||
else:
|
||||
if not content or (isinstance(content, str) and not content.strip()):
|
||||
content = "(empty message)"
|
||||
@@ -2292,10 +2564,22 @@ def _manage_thinking_signatures(
|
||||
replayed assistant tool-call messages. See hermes-agent#13848 (Kimi) and
|
||||
hermes-agent#16748 (DeepSeek).
|
||||
|
||||
Nous Portal's ``/v1/messages`` route is the exception among third-party
|
||||
hosts: it proxies Claude to Anthropic/Vertex/Bedrock and validates the
|
||||
same signed thinking blocks. Sticky ``session_id`` keeps a conversation
|
||||
on one upstream instance so those signatures stay warm — stripping them
|
||||
here would 400 the first tool-loop turn ("thinking must be passed back").
|
||||
Portal therefore takes the native Anthropic replay path below.
|
||||
|
||||
Mutates ``result`` in place.
|
||||
"""
|
||||
_THINKING_TYPES = frozenset(("thinking", "redacted_thinking"))
|
||||
_is_third_party = _is_third_party_anthropic_endpoint(base_url)
|
||||
# Portal speaks Anthropic's thinking contract end-to-end; do not treat it
|
||||
# as a signature-blind proxy even though the host is not anthropic.com.
|
||||
_is_third_party = (
|
||||
_is_third_party_anthropic_endpoint(base_url)
|
||||
and not _is_nous_portal_endpoint(base_url)
|
||||
)
|
||||
|
||||
last_assistant_idx = None
|
||||
for i in range(len(result) - 1, -1, -1):
|
||||
@@ -2425,9 +2709,114 @@ def _ensure_leading_user_turn(result: List[Dict[str, Any]]) -> None:
|
||||
Mirror the Bedrock Converse adapter, which unconditionally prepends a
|
||||
minimal user turn when the first message is not user
|
||||
(convert_messages_to_converse).
|
||||
|
||||
The inserted text block must be non-whitespace: Anthropic separately
|
||||
rejects any text content block whose text is empty or whitespace-only
|
||||
("text content blocks must contain non-whitespace text"), so a single
|
||||
space here traded the "leading assistant turn" 400 for that one (#69512
|
||||
class). Uses the same placeholder as every other synthesized filler
|
||||
block in this module for consistency.
|
||||
"""
|
||||
if result and result[0].get("role") != "user":
|
||||
result.insert(0, {"role": "user", "content": [{"type": "text", "text": " "}]})
|
||||
result.insert(
|
||||
0, {"role": "user", "content": [{"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}]}
|
||||
)
|
||||
|
||||
|
||||
def _fix_blank_text_blocks_in_list(
|
||||
blocks: List[Any],
|
||||
*,
|
||||
placeholder_text: str,
|
||||
msg_index: int,
|
||||
role: Any,
|
||||
location: str,
|
||||
) -> List[Any]:
|
||||
"""Drop blank/whitespace-only text blocks from ``blocks``, in place logic.
|
||||
|
||||
Non-text blocks (tool_use, tool_result, image, document, thinking, …)
|
||||
and the relative order of everything else are left untouched. A
|
||||
cache_control marker riding on a dropped block is relocated onto the
|
||||
last surviving text/tool_use block so a breakpoint is never silently
|
||||
lost. If nothing survives, a single non-blank placeholder text block
|
||||
takes the dropped blocks' place (carrying the relocated cache_control,
|
||||
if any) so the message never has empty content.
|
||||
|
||||
Returns a new list; does not mutate ``blocks``.
|
||||
"""
|
||||
kept: List[Any] = []
|
||||
relocated_cache_control = None
|
||||
for block_index, blk in enumerate(blocks):
|
||||
if (
|
||||
isinstance(blk, dict)
|
||||
and blk.get("type") == "text"
|
||||
and not (isinstance(blk.get("text"), str) and blk["text"].strip())
|
||||
):
|
||||
if isinstance(blk.get("cache_control"), dict):
|
||||
relocated_cache_control = blk["cache_control"]
|
||||
logger.warning(
|
||||
"Pre-call sanitizer: dropped blank text content block "
|
||||
"(message_index=%d role=%s location=%s block_index=%d "
|
||||
"block_type=text)",
|
||||
msg_index,
|
||||
role,
|
||||
location,
|
||||
block_index,
|
||||
)
|
||||
continue
|
||||
kept.append(blk)
|
||||
if not kept:
|
||||
placeholder: Dict[str, Any] = {"type": "text", "text": placeholder_text}
|
||||
if relocated_cache_control is not None:
|
||||
placeholder["cache_control"] = relocated_cache_control
|
||||
kept.append(placeholder)
|
||||
elif relocated_cache_control is not None:
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(kept, relocated_cache_control)
|
||||
return kept
|
||||
|
||||
|
||||
def _scrub_blank_text_blocks(result: List[Dict[str, Any]]) -> None:
|
||||
"""Final provider-boundary guard against blank Anthropic text blocks.
|
||||
|
||||
Anthropic rejects any text content block whose ``text`` is empty or
|
||||
whitespace-only with HTTP 400 ("text content blocks must contain
|
||||
non-whitespace text"). ``_convert_assistant_message``,
|
||||
``_convert_user_message`` and ``_ensure_leading_user_turn`` already
|
||||
avoid emitting these for the paths that build them, but this pass runs
|
||||
last — after every other transform in ``convert_messages_to_anthropic``
|
||||
— so a blank block from any current or future producer (including one
|
||||
nested inside a ``tool_result``'s own content list) never reaches the
|
||||
wire. Diagnostics are structural only: message index, role, content
|
||||
location, block index/type. Never logs message text, tool arguments,
|
||||
tokens, or credentials. Mutates ``result`` in place.
|
||||
"""
|
||||
for msg_index, msg in enumerate(result):
|
||||
if not isinstance(msg, dict):
|
||||
continue
|
||||
role = msg.get("role")
|
||||
content = msg.get("content")
|
||||
if not isinstance(content, list) or not content:
|
||||
continue
|
||||
placeholder_text = _EMPTY_TEXT_PLACEHOLDER if role == "assistant" else "(empty message)"
|
||||
new_content = _fix_blank_text_blocks_in_list(
|
||||
content,
|
||||
placeholder_text=placeholder_text,
|
||||
msg_index=msg_index,
|
||||
role=role,
|
||||
location="content",
|
||||
)
|
||||
for blk in new_content:
|
||||
if not isinstance(blk, dict) or blk.get("type") != "tool_result":
|
||||
continue
|
||||
inner = blk.get("content")
|
||||
if isinstance(inner, list) and inner:
|
||||
blk["content"] = _fix_blank_text_blocks_in_list(
|
||||
inner,
|
||||
placeholder_text="(no output)",
|
||||
msg_index=msg_index,
|
||||
role=role,
|
||||
location="tool_result",
|
||||
)
|
||||
msg["content"] = new_content
|
||||
|
||||
|
||||
def convert_messages_to_anthropic(
|
||||
@@ -2466,7 +2855,26 @@ def convert_messages_to_anthropic(
|
||||
p.get("cache_control") for p in content if isinstance(p, dict)
|
||||
)
|
||||
if has_cache:
|
||||
system = [p for p in content if isinstance(p, dict)]
|
||||
# Copy blocks before coercing so the caller's message
|
||||
# dicts are never mutated, then replace blank/whitespace
|
||||
# text with the shared non-whitespace placeholder —
|
||||
# Anthropic rejects a blank system text block with the
|
||||
# same HTTP 400 as message blocks ("text content blocks
|
||||
# must contain non-whitespace text"), and a blank block
|
||||
# carrying a cache_control breakpoint cannot simply be
|
||||
# dropped (#70909).
|
||||
system = []
|
||||
for p in content:
|
||||
if not isinstance(p, dict):
|
||||
continue
|
||||
if (
|
||||
p.get("type") == "text"
|
||||
and isinstance(p.get("text"), str)
|
||||
and not p["text"].strip()
|
||||
):
|
||||
p = dict(p)
|
||||
p["text"] = _EMPTY_TEXT_PLACEHOLDER
|
||||
system.append(p)
|
||||
else:
|
||||
system = "\n".join(
|
||||
p["text"] for p in content if p.get("type") == "text"
|
||||
@@ -2491,6 +2899,7 @@ def convert_messages_to_anthropic(
|
||||
_ensure_leading_user_turn(result)
|
||||
_manage_thinking_signatures(result, base_url, model)
|
||||
_evict_old_screenshots(result)
|
||||
_scrub_blank_text_blocks(result)
|
||||
|
||||
return system, result
|
||||
|
||||
@@ -2552,7 +2961,12 @@ def build_anthropic_kwargs(
|
||||
)
|
||||
anthropic_tools = convert_tools_to_anthropic(tools) if tools else []
|
||||
|
||||
model = normalize_model_name(model, preserve_dots=preserve_dots)
|
||||
# Nous Portal routes on its own catalog ids (``anthropic/claude-opus-4.8``);
|
||||
# normalizing to the bare Anthropic slug would make the model unresolvable
|
||||
# there. Skipping the call preserves the prefix AND the dots, so
|
||||
# ``preserve_dots`` stays irrelevant for Portal.
|
||||
if not _is_nous_portal_endpoint(base_url):
|
||||
model = normalize_model_name(model, preserve_dots=preserve_dots)
|
||||
# effective_max_tokens = output cap for this call (≠ total context window)
|
||||
# Use the resolver helper so non-positive values (negative ints,
|
||||
# fractional floats, NaN, non-numeric) fail locally with a clear error
|
||||
@@ -2676,7 +3090,15 @@ def build_anthropic_kwargs(
|
||||
# request "summarized" so the reasoning blocks stay populated — matching
|
||||
# 4.6 behavior and preserving the activity-feed UX during long tool runs.
|
||||
if reasoning_config and isinstance(reasoning_config, dict):
|
||||
if reasoning_config.get("enabled") is not False and "haiku" not in model.lower():
|
||||
if reasoning_config.get("enabled") is False:
|
||||
# "Thinking off". Adaptive models think by DEFAULT, so omitting the
|
||||
# parameter is not a disable — it silently leaves thinking on and
|
||||
# the user keeps paying for it. Send the disable explicitly.
|
||||
# Mandatory-thinking models reject it with a 400, so they keep the
|
||||
# omission: a silently-ignored disable beats a dead turn.
|
||||
if _accepts_thinking_disable(model):
|
||||
kwargs["thinking"] = {"type": "disabled"}
|
||||
elif "haiku" not in model.lower():
|
||||
effort = str(reasoning_config.get("effort", "medium")).lower()
|
||||
budget = THINKING_BUDGET.get(effort, 8000)
|
||||
if _supports_adaptive_thinking(model):
|
||||
@@ -2791,6 +3213,8 @@ def create_anthropic_message(
|
||||
*,
|
||||
log_prefix: str = "",
|
||||
prefer_stream: bool = True,
|
||||
on_stream_event=None,
|
||||
on_response=None,
|
||||
) -> Any:
|
||||
"""Create an Anthropic message, aggregating via stream when available.
|
||||
|
||||
@@ -2800,6 +3224,20 @@ def create_anthropic_message(
|
||||
crash on ``.content``. Prefer ``messages.stream().get_final_message()`` to
|
||||
match the main turn path, falling back to ``create()`` only for providers
|
||||
that explicitly do not support streaming, such as restricted Bedrock roles.
|
||||
|
||||
``on_stream_event``: optional callable invoked once per streamed event
|
||||
(best-effort, exceptions swallowed). Lets callers report forward progress
|
||||
to liveness watchdogs — e.g. the auxiliary compression path ticking its
|
||||
progress hook so a slow-but-generating summary model isn't treated as
|
||||
hung. Only fires on the streaming path; the ``create()`` fallback has no
|
||||
events to report.
|
||||
|
||||
``on_response``: optional callable invoked once with the underlying httpx
|
||||
response before the message is aggregated (best-effort, exceptions
|
||||
swallowed). Response *headers* carry out-of-band provider state that the
|
||||
parsed ``Message`` drops — Nous Portal's ``x-nous-credits-*`` balance family
|
||||
in particular. Only fires on the streaming path, which is the one the main
|
||||
turn loop takes.
|
||||
"""
|
||||
sanitize_anthropic_kwargs(api_kwargs, log_prefix=log_prefix)
|
||||
|
||||
@@ -2810,6 +3248,26 @@ def create_anthropic_message(
|
||||
stream_kwargs.pop("stream", None)
|
||||
try:
|
||||
with stream_fn(**stream_kwargs) as stream:
|
||||
if callable(on_response):
|
||||
try:
|
||||
on_response(getattr(stream, "response", None))
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"%son_response callback failed",
|
||||
log_prefix, exc_info=True,
|
||||
)
|
||||
if callable(on_stream_event):
|
||||
# Consume the event stream manually so each event can
|
||||
# tick the caller's progress callback; get_final_message
|
||||
# then returns the accumulated snapshot.
|
||||
for _event in stream:
|
||||
try:
|
||||
on_stream_event(_event)
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"%son_stream_event callback failed",
|
||||
log_prefix, exc_info=True,
|
||||
)
|
||||
return stream.get_final_message()
|
||||
except Exception as exc:
|
||||
if not _is_stream_unavailable_error(exc):
|
||||
|
||||
@@ -367,11 +367,27 @@ def describe_active_credential(config: Optional[EntraIdentityConfig] = None,
|
||||
info["tenant_id_env"] = os.environ["AZURE_TENANT_ID"].strip()
|
||||
|
||||
# Surface which env-var sources are present without minting yet.
|
||||
# Credential-bearing vars (AZURE_CLIENT_SECRET, AZURE_FEDERATED_TOKEN_FILE)
|
||||
# are read through the profile secret scope so a multiplexed profile's
|
||||
# diagnostics don't report another profile's env-bridged credentials;
|
||||
# unscoped CLI probes keep the legacy env read (Slack pattern).
|
||||
def _scoped_env(name: str) -> str:
|
||||
try:
|
||||
from agent.secret_scope import UnscopedSecretError, get_secret
|
||||
|
||||
try:
|
||||
return (get_secret(name) or "").strip()
|
||||
except UnscopedSecretError:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
return os.environ.get(name, "").strip()
|
||||
|
||||
env_sources = []
|
||||
if os.environ.get("AZURE_FEDERATED_TOKEN_FILE", "").strip():
|
||||
if _scoped_env("AZURE_FEDERATED_TOKEN_FILE"):
|
||||
env_sources.append("WorkloadIdentityCredential (AZURE_FEDERATED_TOKEN_FILE)")
|
||||
if (os.environ.get("AZURE_CLIENT_ID", "").strip()
|
||||
and os.environ.get("AZURE_CLIENT_SECRET", "").strip()
|
||||
and _scoped_env("AZURE_CLIENT_SECRET")
|
||||
and os.environ.get("AZURE_TENANT_ID", "").strip()):
|
||||
env_sources.append("EnvironmentCredential (client secret)")
|
||||
if os.environ.get("IDENTITY_ENDPOINT", "").strip() or os.environ.get("MSI_ENDPOINT", "").strip():
|
||||
|
||||
@@ -0,0 +1,204 @@
|
||||
"""Single owner for backend identity and failure-scoped skip decisions.
|
||||
|
||||
Every fallback / dedup / skip / quarantine decision in Hermes ultimately asks
|
||||
one question: **"is this candidate the same backend as the one that failed,
|
||||
along the axis that failure invalidated?"** Before this module, that
|
||||
question was re-implemented inline at six call sites across four subsystems,
|
||||
each comparing whatever string was locally convenient (provider label,
|
||||
provider+model, base_url+model, ...). Each incident fixed one site while the
|
||||
others kept the bug: #22548 (same-shim aliases), #70893 (xai-oauth vs xai —
|
||||
same host, distinct credential), #59561 (aux chain skipped sibling models),
|
||||
#72468 (aux main-model safety net, same bug three weeks later), #62984 /
|
||||
#54250 / #57584 (dedup ignoring base_url strands multi-endpoint pools).
|
||||
|
||||
The root insight: "provider" conflates three independent identity axes, and
|
||||
each failure class invalidates a different one:
|
||||
|
||||
* **credential surface** — auth 401 / payment 402 kill everything sharing the
|
||||
credential (every model, every host reached with that key/token).
|
||||
* **endpoint** — DNS failure / connection refused kill everything behind the
|
||||
URL, regardless of model or credential.
|
||||
* **model deployment** — timeout / overload / rate limit / model-incompatible
|
||||
kill ONE model's deployment. A sibling model behind the same URL is an
|
||||
independent deployment (real incident: aux ``glm-5.2`` hung and timed out
|
||||
while main ``macaron-v1-venti`` on the identical endpoint was serving
|
||||
448K-token turns).
|
||||
|
||||
Call sites should build :class:`BackendIdentity` values, classify the failure
|
||||
with :func:`classify_failure_scope`, and ask :func:`should_skip_candidate`.
|
||||
Do not re-implement any comparison inline — extend THIS module instead.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from typing import Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class FailureScope(Enum):
|
||||
"""Which identity axis a failure invalidates."""
|
||||
|
||||
#: Timeout, overload/429, connection blip, model-incompatible, invalid
|
||||
#: response: evidence against ONE model deployment only.
|
||||
MODEL = "model"
|
||||
#: Auth 401 / payment 402: evidence against the shared credential —
|
||||
#: every model reached with it is equally dead.
|
||||
CREDENTIAL = "credential"
|
||||
#: DNS / connection-refused / unreachable host: evidence against the
|
||||
#: endpoint — every model behind the URL is equally dead.
|
||||
ENDPOINT = "endpoint"
|
||||
|
||||
|
||||
#: Reason strings already used by auxiliary_client's except-chain, mapped to
|
||||
#: scopes. Unknown reasons default to MODEL — the least-invalidating scope —
|
||||
#: so an unrecognized failure never over-skips viable candidates.
|
||||
_REASON_SCOPES = {
|
||||
"auth error": FailureScope.CREDENTIAL,
|
||||
"payment error": FailureScope.CREDENTIAL,
|
||||
"rate limit": FailureScope.MODEL,
|
||||
"model incompatible with route": FailureScope.MODEL,
|
||||
"invalid provider response": FailureScope.MODEL,
|
||||
"connection error": FailureScope.MODEL,
|
||||
"timeout": FailureScope.MODEL,
|
||||
}
|
||||
|
||||
|
||||
def classify_failure_scope(reason: Optional[str]) -> FailureScope:
|
||||
"""Map a human-readable failure reason to the identity axis it kills."""
|
||||
return _REASON_SCOPES.get((reason or "").strip().lower(), FailureScope.MODEL)
|
||||
|
||||
|
||||
def _norm_provider(value: Optional[str]) -> str:
|
||||
return (value or "").strip().lower()
|
||||
|
||||
|
||||
def _norm_model(value: Optional[str]) -> str:
|
||||
return (value or "").strip().lower()
|
||||
|
||||
|
||||
def _norm_base_url(value: Optional[str]) -> str:
|
||||
return (value or "").strip().rstrip("/").lower()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BackendIdentity:
|
||||
"""Normalized identity of one (provider, model, endpoint) deployment.
|
||||
|
||||
Empty fields mean "unknown" — comparisons treat an unknown axis as
|
||||
non-distinguishing (it can neither prove sameness nor difference on its
|
||||
own; the remaining axes decide).
|
||||
"""
|
||||
|
||||
provider: str = ""
|
||||
model: str = ""
|
||||
base_url: str = ""
|
||||
|
||||
@classmethod
|
||||
def build(
|
||||
cls,
|
||||
provider: Optional[str] = None,
|
||||
model: Optional[str] = None,
|
||||
base_url: Optional[str] = None,
|
||||
) -> "BackendIdentity":
|
||||
return cls(
|
||||
provider=_norm_provider(provider),
|
||||
model=_norm_model(model),
|
||||
base_url=_norm_base_url(base_url),
|
||||
)
|
||||
|
||||
|
||||
def _both_first_class(a: BackendIdentity, b: BackendIdentity) -> bool:
|
||||
"""True when both providers are distinct registered first-class providers.
|
||||
|
||||
Two different registry providers have distinct credential surfaces even
|
||||
when they share an inference host (xai-oauth vs xai, openai-codex vs
|
||||
openai-api) — #70893. Custom/shim aliases are NOT in the registry, so
|
||||
two aliases pointing at one URL still count as the same backend (#22548).
|
||||
"""
|
||||
if not a.provider or not b.provider or a.provider == b.provider:
|
||||
return False
|
||||
try:
|
||||
from hermes_cli.auth import PROVIDER_REGISTRY
|
||||
|
||||
return a.provider in PROVIDER_REGISTRY and b.provider in PROVIDER_REGISTRY
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def same_credential_surface(a: BackendIdentity, b: BackendIdentity) -> bool:
|
||||
"""Do two identities share the credential a 401/402 just invalidated?
|
||||
|
||||
Conservative on purpose: an unprovable axis must answer "different"
|
||||
(try the candidate — worst case one wasted RTT) rather than "same"
|
||||
(skip — worst case stranded failover). Two distinct custom labels at
|
||||
one URL may carry different per-entry api_keys, so a shared URL alone
|
||||
never proves a shared credential; it is only used as a weak signal
|
||||
when a provider label is missing entirely.
|
||||
"""
|
||||
if a.provider and b.provider:
|
||||
# Same label = same configured credential. Different labels =
|
||||
# different credential config (first-class registry providers
|
||||
# explicitly so — #70893; custom entries can each carry their own
|
||||
# api_key, so sameness is unprovable and we must not skip).
|
||||
return a.provider == b.provider
|
||||
# Provider unknown on a side: same explicit URL is the best signal left.
|
||||
return bool(a.base_url and a.base_url == b.base_url)
|
||||
|
||||
|
||||
def same_endpoint(a: BackendIdentity, b: BackendIdentity) -> bool:
|
||||
"""Do two identities sit behind the endpoint that just went unreachable?"""
|
||||
if a.base_url and b.base_url:
|
||||
return a.base_url == b.base_url
|
||||
# An unknown base_url inherits the provider default → same provider
|
||||
# label implies the same default endpoint.
|
||||
return bool(a.provider and a.provider == b.provider)
|
||||
|
||||
|
||||
def same_deployment(a: BackendIdentity, b: BackendIdentity) -> bool:
|
||||
"""Are these the exact same model deployment (the thing a timeout kills)?
|
||||
|
||||
Provider+model must match; the base_url axis distinguishes only when BOTH
|
||||
sides carry an explicit URL (#62984: same provider+model on two different
|
||||
explicit URLs is two deployments — a pool). A side with an unknown URL
|
||||
inherits the provider default and cannot prove difference.
|
||||
"""
|
||||
if not (a.provider and b.provider and a.provider == b.provider):
|
||||
# Same-host different-label shims: same URL + same model IS the same
|
||||
# deployment even when the alias labels differ (#22548) — unless both
|
||||
# labels are first-class registry providers (#70893).
|
||||
if (
|
||||
a.base_url
|
||||
and a.base_url == b.base_url
|
||||
and a.model
|
||||
and a.model == b.model
|
||||
and not _both_first_class(a, b)
|
||||
):
|
||||
return True
|
||||
return False
|
||||
if not (a.model and b.model and a.model == b.model):
|
||||
return False
|
||||
if a.base_url and b.base_url and a.base_url != b.base_url:
|
||||
return False # distinct explicit endpoints — a pool, not a dup
|
||||
return True
|
||||
|
||||
|
||||
def should_skip_candidate(
|
||||
candidate: BackendIdentity,
|
||||
failed: BackendIdentity,
|
||||
scope: FailureScope = FailureScope.MODEL,
|
||||
) -> bool:
|
||||
"""THE skip predicate: would trying ``candidate`` just repeat the failure?
|
||||
|
||||
True when the candidate is the same backend as ``failed`` along the axis
|
||||
``scope`` says the failure invalidated. Every fallback/dedup/skip site
|
||||
must call this instead of comparing labels inline.
|
||||
"""
|
||||
if scope is FailureScope.CREDENTIAL:
|
||||
return same_credential_surface(candidate, failed)
|
||||
if scope is FailureScope.ENDPOINT:
|
||||
return same_endpoint(candidate, failed)
|
||||
return same_deployment(candidate, failed)
|
||||
@@ -33,6 +33,9 @@ import os
|
||||
import re
|
||||
from types import SimpleNamespace
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import httpx
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -57,6 +60,25 @@ except Exception:
|
||||
_bedrock_runtime_client_cache: Dict[str, Any] = {}
|
||||
_bedrock_control_client_cache: Dict[str, Any] = {}
|
||||
|
||||
# Bedrock-hosted OpenAI GPT-5.5 is not exposed through the native Converse
|
||||
# runtime. AWS serves it from the Bedrock Mantle OpenAI-compatible Responses
|
||||
# endpoint instead (https://bedrock-mantle.<region>.api.aws/openai/v1).
|
||||
# Keep the allowlist intentionally narrow so OpenAI GPT-OSS models that are
|
||||
# Converse-capable continue to use the native Bedrock path.
|
||||
BEDROCK_OPENAI_RESPONSES_MODEL_IDS: Tuple[str, ...] = (
|
||||
"openai.gpt-5.5",
|
||||
# GPT-5.6 family (GA on Bedrock 2026-07-13): Sol (frontier), Terra
|
||||
# (balanced), Luna (fast/affordable). All are Mantle-only — the model
|
||||
# cards list bedrock-runtime/Converse as unsupported.
|
||||
# https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html
|
||||
"openai.gpt-5.6-sol",
|
||||
"openai.gpt-5.6-terra",
|
||||
"openai.gpt-5.6-luna",
|
||||
)
|
||||
_BEDROCK_OPENAI_HOST_RE = re.compile(
|
||||
r"^bedrock-mantle\.([a-z0-9-]+)\.api\.aws$", re.IGNORECASE
|
||||
)
|
||||
|
||||
|
||||
_MIN_BOTO3_VERSION = (1, 34, 59)
|
||||
|
||||
@@ -133,6 +155,143 @@ def invalidate_runtime_client(region: str) -> bool:
|
||||
return existed
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Bedrock Mantle / OpenAI Responses support
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def is_openai_bedrock_model(model_id: str) -> bool:
|
||||
"""Return True for Bedrock-hosted OpenAI models that require Mantle.
|
||||
|
||||
Bedrock's GPT-OSS models are Converse-capable and intentionally do not
|
||||
match this helper. The allowlist tracks models served by the OpenAI
|
||||
Responses-compatible ``bedrock-mantle`` route.
|
||||
"""
|
||||
normalized = str(model_id or "").strip().lower()
|
||||
return normalized in {m.lower() for m in BEDROCK_OPENAI_RESPONSES_MODEL_IDS}
|
||||
|
||||
|
||||
def merge_bedrock_openai_model_ids(model_ids: List[str]) -> List[str]:
|
||||
"""Append Bedrock OpenAI Responses models to a discovered Bedrock list.
|
||||
|
||||
The Bedrock control plane's ListFoundationModels/ListInferenceProfiles
|
||||
discovery covers Converse models but does not enumerate Mantle-only
|
||||
OpenAI Responses models. The picker needs both surfaces under AWS Bedrock.
|
||||
"""
|
||||
merged = list(model_ids or [])
|
||||
seen = {str(m).lower() for m in merged}
|
||||
for model_id in BEDROCK_OPENAI_RESPONSES_MODEL_IDS:
|
||||
if model_id.lower() not in seen:
|
||||
merged.append(model_id)
|
||||
seen.add(model_id.lower())
|
||||
return merged
|
||||
|
||||
|
||||
def bedrock_openai_base_url(region: str) -> str:
|
||||
"""Return Bedrock Mantle's OpenAI-compatible base URL for *region*."""
|
||||
resolved = (region or "").strip() or resolve_bedrock_runtime_region()
|
||||
return f"https://bedrock-mantle.{resolved}.api.aws/openai/v1"
|
||||
|
||||
|
||||
def bedrock_openai_region_from_base_url(base_url: str) -> Optional[str]:
|
||||
"""Extract the AWS region from a Bedrock Mantle OpenAI base URL."""
|
||||
host = urlparse(str(base_url or "")).hostname or ""
|
||||
match = _BEDROCK_OPENAI_HOST_RE.match(host)
|
||||
return match.group(1) if match else None
|
||||
|
||||
|
||||
def is_bedrock_openai_base_url(base_url: str) -> bool:
|
||||
"""Return True for Bedrock Mantle OpenAI-compatible endpoints."""
|
||||
parsed = urlparse(str(base_url or ""))
|
||||
host = parsed.hostname or ""
|
||||
if not _BEDROCK_OPENAI_HOST_RE.match(host):
|
||||
return False
|
||||
# The OpenAI GPT-5.5 Bedrock route lives under /openai/v1. Accept a bare
|
||||
# host too so callers can normalize before appending the path.
|
||||
path = (parsed.path or "").rstrip("/").lower()
|
||||
return path in {"", "/openai", "/openai/v1"}
|
||||
|
||||
|
||||
def resolve_bedrock_bearer_token(env: Optional[Dict[str, str]] = None) -> str:
|
||||
"""Return AWS_BEARER_TOKEN_BEDROCK when Bedrock API-key auth is configured."""
|
||||
env = env if env is not None else os.environ
|
||||
return (env.get("AWS_BEARER_TOKEN_BEDROCK", "") or "").strip()
|
||||
|
||||
|
||||
class BedrockOpenAISigV4Auth(httpx.Auth):
|
||||
"""httpx auth hook that SigV4-signs Bedrock Mantle OpenAI requests."""
|
||||
|
||||
requires_request_body = True
|
||||
|
||||
def __init__(self, region: str, service: str = "bedrock"):
|
||||
self.region = (region or "").strip() or resolve_bedrock_runtime_region()
|
||||
self.service = service
|
||||
|
||||
def auth_flow(self, request): # pragma: no cover - exercised by live call
|
||||
import botocore.session
|
||||
from botocore.auth import SigV4Auth
|
||||
from botocore.awsrequest import AWSRequest
|
||||
|
||||
credentials = botocore.session.get_session().get_credentials()
|
||||
if credentials is None:
|
||||
raise RuntimeError(
|
||||
"No AWS credentials available for Bedrock OpenAI Responses. "
|
||||
"Configure AWS_ACCESS_KEY_ID/AWS_SECRET_ACCESS_KEY, AWS_PROFILE, "
|
||||
"SSO, or an instance/task role."
|
||||
)
|
||||
frozen = credentials.get_frozen_credentials()
|
||||
# Drop the OpenAI SDK's placeholder bearer header before signing; SigV4
|
||||
# must own Authorization. Keep all other SDK headers so AWS receives
|
||||
# content-type, accept, request IDs, etc.
|
||||
headers = {
|
||||
str(k): str(v)
|
||||
for k, v in request.headers.items()
|
||||
if str(k).lower() not in {"authorization", "x-amz-date", "x-amz-security-token"}
|
||||
}
|
||||
aws_request = AWSRequest(
|
||||
method=request.method,
|
||||
url=str(request.url),
|
||||
data=request.content or b"",
|
||||
headers=headers,
|
||||
)
|
||||
SigV4Auth(frozen, self.service, self.region).add_auth(aws_request)
|
||||
request.headers.update(dict(aws_request.headers.items()))
|
||||
yield request
|
||||
|
||||
|
||||
def build_bedrock_openai_http_client(region: str, *, timeout: Optional[float] = None):
|
||||
"""Build an httpx client that SigV4-signs Bedrock OpenAI requests."""
|
||||
import httpx
|
||||
|
||||
kwargs: Dict[str, Any] = {"auth": BedrockOpenAISigV4Auth(region)}
|
||||
if isinstance(timeout, (int, float)) and not isinstance(timeout, bool) and timeout > 0:
|
||||
kwargs["timeout"] = timeout
|
||||
return httpx.Client(**kwargs)
|
||||
|
||||
|
||||
def configure_bedrock_openai_client_kwargs(
|
||||
client_kwargs: Dict[str, Any],
|
||||
*,
|
||||
timeout: Optional[float] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Install SigV4 auth on OpenAI SDK kwargs for Bedrock Mantle.
|
||||
|
||||
``AWS_BEARER_TOKEN_BEDROCK``/explicit Bedrock API keys continue to use the
|
||||
SDK's normal bearer auth. The special ``aws-sdk`` placeholder means IAM
|
||||
credential-chain auth, so we attach a per-request SigV4 httpx client.
|
||||
"""
|
||||
base_url = str(client_kwargs.get("base_url") or "")
|
||||
if not is_bedrock_openai_base_url(base_url):
|
||||
return client_kwargs
|
||||
api_key = client_kwargs.get("api_key")
|
||||
if isinstance(api_key, str) and api_key.strip() and api_key not in {"aws-sdk", "no-key-required"}:
|
||||
return client_kwargs
|
||||
region = bedrock_openai_region_from_base_url(base_url) or resolve_bedrock_runtime_region()
|
||||
client_kwargs["api_key"] = "aws-sdk"
|
||||
client_kwargs["http_client"] = build_bedrock_openai_http_client(region, timeout=timeout)
|
||||
return client_kwargs
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stale-connection detection
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -384,6 +543,36 @@ def resolve_bedrock_region(env: Optional[Dict[str, str]] = None) -> str:
|
||||
return "us-east-1"
|
||||
|
||||
|
||||
def resolve_bedrock_runtime_region(config: Optional[Dict[str, Any]] = None) -> str:
|
||||
"""Resolve the Bedrock region with the same priority as the main runtime.
|
||||
|
||||
Priority (matches the runtime provider resolver in
|
||||
``hermes_cli/runtime_provider.py``):
|
||||
1. ``bedrock.region`` in config.yaml
|
||||
2. ``resolve_bedrock_region()`` (AWS_REGION / AWS_DEFAULT_REGION /
|
||||
botocore profile / us-east-1)
|
||||
|
||||
Callers that already hold a loaded config dict should pass it to avoid a
|
||||
disk read; when *config* is None the config is loaded read-only. Every
|
||||
non-runtime call site that constructs a Bedrock endpoint (auxiliary
|
||||
client resolution, model discovery for the picker) must use this helper —
|
||||
using bare ``resolve_bedrock_region()`` there lets auxiliary calls leave
|
||||
the primary runtime's configured region when ``bedrock.region`` and the
|
||||
ambient AWS env/profile disagree.
|
||||
"""
|
||||
if config is None:
|
||||
try:
|
||||
from hermes_cli.config import load_config_readonly
|
||||
config = load_config_readonly()
|
||||
except Exception:
|
||||
config = {}
|
||||
bedrock_cfg = (config or {}).get("bedrock") or {}
|
||||
cfg_region = str(bedrock_cfg.get("region") or "").strip()
|
||||
if cfg_region:
|
||||
return cfg_region
|
||||
return resolve_bedrock_region()
|
||||
|
||||
|
||||
def bedrock_model_ids_or_none() -> Optional[List[str]]:
|
||||
"""Live-discover Bedrock model IDs for the active region.
|
||||
|
||||
@@ -396,9 +585,9 @@ def bedrock_model_ids_or_none() -> Optional[List[str]]:
|
||||
``list_authenticated_providers`` section 2, and section 3.
|
||||
"""
|
||||
try:
|
||||
discovered = discover_bedrock_models(resolve_bedrock_region())
|
||||
discovered = discover_bedrock_models(resolve_bedrock_runtime_region())
|
||||
if discovered:
|
||||
return [m["id"] for m in discovered]
|
||||
return merge_bedrock_openai_model_ids([m["id"] for m in discovered])
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
@@ -433,6 +622,29 @@ def _model_supports_tool_use(model_id: str) -> bool:
|
||||
return not any(pattern in model_lower for pattern in _NON_TOOL_CALLING_PATTERNS)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Prompt-cache capability detection (Converse API cachePoint)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Claude on Bedrock already gets prompt caching through the AnthropicBedrock
|
||||
# SDK path (see is_anthropic_bedrock_model / runtime_provider.py's dual-path
|
||||
# routing) — it never reaches build_converse_kwargs unless bearer-token auth
|
||||
# forces the Converse path (#28156). This allowlist covers the Converse API
|
||||
# itself: sending an unsupported model a cachePoint block raises a
|
||||
# ValidationException, so — like _model_supports_tool_use but inverted —
|
||||
# unknown models default to NOT receiving cache markers until confirmed.
|
||||
# Ref: https://docs.aws.amazon.com/bedrock/latest/userguide/prompt-caching.html
|
||||
_CACHE_POINT_PATTERNS = [
|
||||
"anthropic.claude", # bearer-token fallback path
|
||||
"amazon.nova",
|
||||
]
|
||||
|
||||
|
||||
def _model_supports_prompt_cache(model_id: str) -> bool:
|
||||
"""Return True if the model accepts a Converse API cachePoint block."""
|
||||
model_lower = model_id.lower()
|
||||
return any(pattern in model_lower for pattern in _CACHE_POINT_PATTERNS)
|
||||
|
||||
|
||||
def is_anthropic_bedrock_model(model_id: str) -> bool:
|
||||
"""Return True if the model is an Anthropic Claude model on Bedrock.
|
||||
|
||||
@@ -764,14 +976,22 @@ def normalize_converse_response(response: Dict) -> SimpleNamespace:
|
||||
reasoning_content="\n\n".join(reasoning_parts) if reasoning_parts else None,
|
||||
)
|
||||
|
||||
# Build usage stats
|
||||
# Build usage stats. Converse's inputTokens excludes cache read/write
|
||||
# tokens (unlike OpenAI's prompt_tokens, which includes them) — restore
|
||||
# the OpenAI-style "total includes cache" convention here so downstream
|
||||
# normalize_usage() can subtract them back out consistently, and surface
|
||||
# the Anthropic-named fields it already falls back to for cache reads.
|
||||
usage_data = response.get("usage", {})
|
||||
input_tokens = usage_data.get("inputTokens", 0)
|
||||
cache_read_tokens = usage_data.get("cacheReadInputTokens", 0)
|
||||
cache_write_tokens = usage_data.get("cacheWriteInputTokens", 0)
|
||||
output_tokens = usage_data.get("outputTokens", 0)
|
||||
usage = SimpleNamespace(
|
||||
prompt_tokens=usage_data.get("inputTokens", 0),
|
||||
completion_tokens=usage_data.get("outputTokens", 0),
|
||||
total_tokens=(
|
||||
usage_data.get("inputTokens", 0) + usage_data.get("outputTokens", 0)
|
||||
),
|
||||
prompt_tokens=input_tokens + cache_read_tokens + cache_write_tokens,
|
||||
completion_tokens=output_tokens,
|
||||
total_tokens=input_tokens + cache_read_tokens + cache_write_tokens + output_tokens,
|
||||
cache_read_input_tokens=cache_read_tokens,
|
||||
cache_creation_input_tokens=cache_write_tokens,
|
||||
)
|
||||
|
||||
finish_reason = _converse_stop_reason_to_openai(stop_reason)
|
||||
@@ -936,6 +1156,8 @@ def stream_converse_with_callbacks(
|
||||
usage_data = {
|
||||
"inputTokens": meta_usage.get("inputTokens", 0),
|
||||
"outputTokens": meta_usage.get("outputTokens", 0),
|
||||
"cacheReadInputTokens": meta_usage.get("cacheReadInputTokens", 0),
|
||||
"cacheWriteInputTokens": meta_usage.get("cacheWriteInputTokens", 0),
|
||||
}
|
||||
|
||||
# Flush remaining text
|
||||
@@ -949,12 +1171,16 @@ def stream_converse_with_callbacks(
|
||||
reasoning_content="\n\n".join(reasoning_parts) if reasoning_parts else None,
|
||||
)
|
||||
|
||||
input_tokens = usage_data.get("inputTokens", 0)
|
||||
cache_read_tokens = usage_data.get("cacheReadInputTokens", 0)
|
||||
cache_write_tokens = usage_data.get("cacheWriteInputTokens", 0)
|
||||
output_tokens = usage_data.get("outputTokens", 0)
|
||||
usage = SimpleNamespace(
|
||||
prompt_tokens=usage_data.get("inputTokens", 0),
|
||||
completion_tokens=usage_data.get("outputTokens", 0),
|
||||
total_tokens=(
|
||||
usage_data.get("inputTokens", 0) + usage_data.get("outputTokens", 0)
|
||||
),
|
||||
prompt_tokens=input_tokens + cache_read_tokens + cache_write_tokens,
|
||||
completion_tokens=output_tokens,
|
||||
total_tokens=input_tokens + cache_read_tokens + cache_write_tokens + output_tokens,
|
||||
cache_read_input_tokens=cache_read_tokens,
|
||||
cache_creation_input_tokens=cache_write_tokens,
|
||||
)
|
||||
|
||||
finish_reason = _converse_stop_reason_to_openai(stop_reason)
|
||||
@@ -982,7 +1208,7 @@ def build_converse_kwargs(
|
||||
model: str,
|
||||
messages: List[Dict],
|
||||
tools: Optional[List[Dict]] = None,
|
||||
max_tokens: int = 4096,
|
||||
max_tokens: Optional[int] = 4096,
|
||||
temperature: Optional[float] = None,
|
||||
top_p: Optional[float] = None,
|
||||
stop_sequences: Optional[List[str]] = None,
|
||||
@@ -991,18 +1217,29 @@ def build_converse_kwargs(
|
||||
"""Build kwargs for ``bedrock-runtime.converse()`` or ``converse_stream()``.
|
||||
|
||||
Converts OpenAI-format inputs to Converse API parameters.
|
||||
|
||||
``max_tokens=None`` omits ``inferenceConfig.maxTokens`` entirely, in which
|
||||
case Bedrock defaults to the model's maximum allowed output — the Converse
|
||||
field is optional per the AWS API reference. The default stays 4096 so
|
||||
existing callers are unaffected; callers that want the model's full output
|
||||
budget (e.g. uncapped auxiliary vision calls) pass ``None`` explicitly.
|
||||
"""
|
||||
system_prompt, converse_messages = convert_messages_to_converse(messages)
|
||||
cache_enabled = _model_supports_prompt_cache(model)
|
||||
|
||||
inference_config: Dict[str, Any] = {}
|
||||
if max_tokens is not None:
|
||||
inference_config["maxTokens"] = max_tokens
|
||||
|
||||
kwargs: Dict[str, Any] = {
|
||||
"modelId": model,
|
||||
"messages": converse_messages,
|
||||
"inferenceConfig": {
|
||||
"maxTokens": max_tokens,
|
||||
},
|
||||
"inferenceConfig": inference_config,
|
||||
}
|
||||
|
||||
if system_prompt:
|
||||
if cache_enabled:
|
||||
system_prompt = system_prompt + [{"cachePoint": {"type": "default"}}]
|
||||
kwargs["system"] = system_prompt
|
||||
|
||||
from agent.anthropic_adapter import _forbids_sampling_params
|
||||
@@ -1026,6 +1263,8 @@ def build_converse_kwargs(
|
||||
# Strip tools for known non-tool-calling models and warn the user.
|
||||
# Ref: PR #7920 feedback from @ptlally, pattern from PR #4346.
|
||||
if _model_supports_tool_use(model):
|
||||
if cache_enabled:
|
||||
converse_tools = converse_tools + [{"cachePoint": {"type": "default"}}]
|
||||
kwargs["toolConfig"] = {"tools": converse_tools}
|
||||
else:
|
||||
logger.warning(
|
||||
@@ -1033,9 +1272,21 @@ def build_converse_kwargs(
|
||||
"The agent will operate in text-only mode.", model
|
||||
)
|
||||
|
||||
if cache_enabled and len(converse_messages) >= 2:
|
||||
# Checkpoint everything up to (not including) the newest turn, so the
|
||||
# marker survives unchanged across requests as only the tail grows —
|
||||
# mirroring the Anthropic system_and_3 strategy in prompt_caching.py.
|
||||
content = converse_messages[-2].get("content")
|
||||
if isinstance(content, list) and content:
|
||||
content.append({"cachePoint": {"type": "default"}})
|
||||
|
||||
if guardrail_config:
|
||||
kwargs["guardrailConfig"] = guardrail_config
|
||||
|
||||
if not kwargs["inferenceConfig"]:
|
||||
# inferenceConfig is optional on the wire; don't send an empty object.
|
||||
del kwargs["inferenceConfig"]
|
||||
|
||||
return kwargs
|
||||
|
||||
|
||||
@@ -1044,7 +1295,7 @@ def call_converse(
|
||||
model: str,
|
||||
messages: List[Dict],
|
||||
tools: Optional[List[Dict]] = None,
|
||||
max_tokens: int = 4096,
|
||||
max_tokens: Optional[int] = 4096,
|
||||
temperature: Optional[float] = None,
|
||||
top_p: Optional[float] = None,
|
||||
stop_sequences: Optional[List[str]] = None,
|
||||
@@ -1085,7 +1336,7 @@ def call_converse_stream(
|
||||
model: str,
|
||||
messages: List[Dict],
|
||||
tools: Optional[List[Dict]] = None,
|
||||
max_tokens: int = 4096,
|
||||
max_tokens: Optional[int] = 4096,
|
||||
temperature: Optional[float] = None,
|
||||
top_p: Optional[float] = None,
|
||||
stop_sequences: Optional[List[str]] = None,
|
||||
@@ -1388,6 +1639,12 @@ BEDROCK_CONTEXT_LENGTHS: Dict[str, int] = {
|
||||
"mistral.mistral-large": 128_000,
|
||||
# DeepSeek
|
||||
"deepseek.v3": 128_000,
|
||||
# OpenAI on Bedrock (Mantle/Responses route)
|
||||
# https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html
|
||||
"openai.gpt-5.5": 272_000,
|
||||
"openai.gpt-5.6-sol": 272_000,
|
||||
"openai.gpt-5.6-terra": 272_000,
|
||||
"openai.gpt-5.6-luna": 272_000,
|
||||
}
|
||||
|
||||
# Default for unknown Bedrock models
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
"""Provider-agnostic billing/credit recovery links.
|
||||
|
||||
Maps a billing-classified failure onto a recovery link + label. *Detection*
|
||||
is not done here — that is :mod:`agent.error_classifier`
|
||||
(``FailoverReason.billing``), the single source of truth for "credit wall vs.
|
||||
rate limit / auth / transport". The resulting :class:`BillingBlock` rides the
|
||||
turn result and the gateway ``message.complete`` event so every surface (CLI,
|
||||
TUI, desktop) renders one structured signal instead of re-parsing error text.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import asdict, dataclass
|
||||
from typing import Optional
|
||||
|
||||
from utils import base_url_host_matches
|
||||
|
||||
|
||||
@dataclass
|
||||
class BillingBlock:
|
||||
"""Structured billing-wall descriptor shared across every surface.
|
||||
|
||||
``is_nous`` is the routing bit: Nous has a first-class in-app billing surface
|
||||
(desktop Settings → Billing, TUI/CLI ``/topup``), so surfaces prefer that over
|
||||
``billing_url``; third-party providers have no in-app flow, so ``billing_url``
|
||||
is the deep link the user actually needs.
|
||||
"""
|
||||
|
||||
provider: str
|
||||
provider_label: str
|
||||
model: str
|
||||
billing_url: Optional[str]
|
||||
is_nous: bool
|
||||
message: str
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _Provider:
|
||||
label: str
|
||||
url: str
|
||||
slugs: tuple[str, ...]
|
||||
hosts: tuple[str, ...] = ()
|
||||
|
||||
|
||||
# Single source of truth: internal slug(s) + base_url host(s) → billing page.
|
||||
# Curated "add credits / manage billing" landing pages, not marketing homes.
|
||||
# Hosts back the OpenAI-compatible fallback where the slug is a generic bucket
|
||||
# (e.g. "openai_compatible") but base_url reveals the real upstream. An unknown
|
||||
# provider degrades to a readable label with no invented URL.
|
||||
_PROVIDERS: tuple[_Provider, ...] = (
|
||||
_Provider("OpenAI", "https://platform.openai.com/settings/organization/billing", ("openai",), ("api.openai.com",)),
|
||||
_Provider("Anthropic", "https://console.anthropic.com/settings/billing", ("anthropic",), ("api.anthropic.com",)),
|
||||
_Provider("OpenRouter", "https://openrouter.ai/settings/credits", ("openrouter",), ("openrouter.ai",)),
|
||||
_Provider("xAI", "https://console.x.ai/team/default/billing", ("xai", "xai-oauth"), ("api.x.ai",)),
|
||||
_Provider("DeepSeek", "https://platform.deepseek.com/top_up", ("deepseek",), ("api.deepseek.com",)),
|
||||
_Provider("Groq", "https://console.groq.com/settings/billing", ("groq",), ("api.groq.com",)),
|
||||
_Provider("Mistral", "https://console.mistral.ai/billing", ("mistral",), ("api.mistral.ai",)),
|
||||
_Provider("Together AI", "https://api.together.ai/settings/billing", ("together",), ("api.together.ai", "api.together.xyz")),
|
||||
_Provider("Fireworks AI", "https://fireworks.ai/account/billing", ("fireworks",), ("fireworks.ai",)),
|
||||
_Provider("Perplexity", "https://www.perplexity.ai/settings/api", ("perplexity",), ("perplexity.ai",)),
|
||||
_Provider("Google AI", "https://aistudio.google.com/app/billing", ("google", "gemini"), ("generativelanguage.googleapis.com",)),
|
||||
_Provider("Cohere", "https://dashboard.cohere.com/billing", ("cohere",)),
|
||||
_Provider("Moonshot AI", "https://platform.moonshot.ai/console/pay", ("moonshot",)),
|
||||
_Provider("NVIDIA", "https://build.nvidia.com/settings/billing", ("nvidia",)),
|
||||
)
|
||||
|
||||
_BY_SLUG: dict[str, _Provider] = {slug: p for p in _PROVIDERS for slug in p.slugs}
|
||||
|
||||
|
||||
def is_nous_inference_route(provider: str, base_url: str) -> bool:
|
||||
"""True when the failing route is the Nous-managed inference gateway."""
|
||||
if (provider or "").strip().lower() == "nous":
|
||||
return True
|
||||
return base_url_host_matches(str(base_url or ""), "inference-api.nousresearch.com")
|
||||
|
||||
|
||||
def _nous_billing_url() -> Optional[str]:
|
||||
"""Best-effort Nous portal billing URL (text-surface fallback; Nous prefers the in-app flow)."""
|
||||
try:
|
||||
from hermes_cli.nous_account import nous_portal_billing_url
|
||||
|
||||
return nous_portal_billing_url(None)
|
||||
except Exception:
|
||||
return "https://portal.nousresearch.com/billing"
|
||||
|
||||
|
||||
def _resolve_provider_link(slug: str, base_url: str) -> tuple[str, Optional[str]]:
|
||||
"""Resolve ``(label, url)``: exact slug → base_url host → readable-label fallback."""
|
||||
hit = _BY_SLUG.get(slug)
|
||||
if hit:
|
||||
return hit.label, hit.url
|
||||
|
||||
base = str(base_url or "")
|
||||
for p in _PROVIDERS:
|
||||
if any(base_url_host_matches(base, host) for host in p.hosts):
|
||||
return p.label, p.url
|
||||
|
||||
return slug.replace("_", " ").replace("-", " ").strip().title() or "your provider", None
|
||||
|
||||
|
||||
def build_billing_block(
|
||||
*,
|
||||
provider: str,
|
||||
base_url: str,
|
||||
model: str,
|
||||
message: str = "",
|
||||
) -> BillingBlock:
|
||||
"""Build the billing descriptor for a billing-classified failure.
|
||||
|
||||
``message`` is the guidance already assembled by the agent loop
|
||||
(:func:`agent.conversation_loop._billing_or_entitlement_message`), carried
|
||||
through unchanged so every surface shows identical copy.
|
||||
"""
|
||||
slug = (provider or "").strip().lower()
|
||||
model = (model or "").strip()
|
||||
|
||||
if is_nous_inference_route(slug, base_url):
|
||||
return BillingBlock(slug or "nous", "Nous Portal", model, _nous_billing_url(), True, message or "")
|
||||
|
||||
label, url = _resolve_provider_link(slug, base_url)
|
||||
return BillingBlock(slug, label, model, url, False, message or "")
|
||||
@@ -34,7 +34,7 @@ from __future__ import annotations
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -17,7 +17,7 @@ from __future__ import annotations
|
||||
import logging
|
||||
import os
|
||||
import uuid
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Optional
|
||||
|
||||
@@ -107,6 +107,22 @@ class CardInfo:
|
||||
return f"{self.masked} — {label}" if label else self.masked
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PaymentMethodInfo:
|
||||
"""The payment method on file. `kind` is "card", "link", or "unknown"
|
||||
— anything else is normalised to "unknown" at parse time, so consumers
|
||||
only ever see fields that belong to the kind they are looking at."""
|
||||
|
||||
kind: str
|
||||
brand: Optional[str] = None
|
||||
last4: Optional[str] = None
|
||||
wallet: Optional[str] = None
|
||||
email: Optional[str] = None
|
||||
resolved_via: Optional[str] = None
|
||||
#: What the server called it, when we did not recognise the kind.
|
||||
raw_kind: Optional[str] = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MonthlyCap:
|
||||
limit_usd: Optional[Decimal] = None
|
||||
@@ -150,6 +166,7 @@ class BillingState:
|
||||
min_usd: Optional[Decimal] = None
|
||||
max_usd: Optional[Decimal] = None
|
||||
card: Optional[CardInfo] = None
|
||||
payment_method: Optional[PaymentMethodInfo] = None
|
||||
monthly_cap: Optional[MonthlyCap] = None
|
||||
auto_reload: Optional[AutoReload] = None
|
||||
portal_url: Optional[str] = None
|
||||
@@ -201,6 +218,41 @@ def _parse_card(raw: Any) -> Optional[CardInfo]:
|
||||
return CardInfo(brand=brand, last4=last4, resolved_via=resolved_via)
|
||||
|
||||
|
||||
def _parse_payment_method(raw: Any) -> Optional[PaymentMethodInfo]:
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
kind = raw.get("kind")
|
||||
if not isinstance(kind, str):
|
||||
return None
|
||||
|
||||
def _optional_string(key: str) -> Optional[str]:
|
||||
value = raw.get(key)
|
||||
return value if isinstance(value, str) else None
|
||||
|
||||
resolved_via = _optional_string("resolvedVia")
|
||||
brand = _optional_string("brand")
|
||||
last4 = _optional_string("last4")
|
||||
# Settle the kind here, the way _parse_card settles a card, so nothing
|
||||
# downstream has to re-check which fields this kind is allowed to have.
|
||||
if kind == "card" and brand and last4:
|
||||
return PaymentMethodInfo(
|
||||
kind="card",
|
||||
brand=brand,
|
||||
last4=last4,
|
||||
wallet=_optional_string("wallet"),
|
||||
resolved_via=resolved_via,
|
||||
)
|
||||
if kind == "link":
|
||||
return PaymentMethodInfo(
|
||||
kind="link",
|
||||
email=_optional_string("email"),
|
||||
resolved_via=resolved_via,
|
||||
)
|
||||
return PaymentMethodInfo(
|
||||
kind="unknown", raw_kind=kind, resolved_via=resolved_via
|
||||
)
|
||||
|
||||
|
||||
def _parse_monthly_cap(raw: Any) -> Optional[MonthlyCap]:
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
@@ -274,6 +326,7 @@ def billing_state_from_payload(
|
||||
min_usd=parse_money(bounds.get("minUsd")),
|
||||
max_usd=parse_money(bounds.get("maxUsd")),
|
||||
card=_parse_card(payload.get("card")),
|
||||
payment_method=_parse_payment_method(payload.get("paymentMethod")),
|
||||
monthly_cap=_parse_monthly_cap(payload.get("monthlyCap")),
|
||||
auto_reload=_parse_auto_reload(payload.get("autoReload")),
|
||||
portal_url=portal_url,
|
||||
|
||||
@@ -26,6 +26,7 @@ Session metadata contract (preserved from the legacy ``CloudBrowserProvider``)::
|
||||
"session_name": str, # unique name for agent-browser --session
|
||||
"bb_session_id": str, # provider session ID (for close/cleanup)
|
||||
"cdp_url": str, # CDP websocket URL
|
||||
"expires_at": str, # optional provider-authoritative ISO timestamp
|
||||
"features": dict, # feature flags that were enabled
|
||||
"external_call_id": str, # optional, managed-gateway billing key
|
||||
}
|
||||
@@ -38,7 +39,7 @@ which provider is in use.
|
||||
from __future__ import annotations
|
||||
|
||||
import abc
|
||||
from typing import Any, Dict
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -96,6 +97,7 @@ class BrowserProvider(abc.ABC):
|
||||
"session_name": str, # unique name for agent-browser --session
|
||||
"bb_session_id": str, # provider session ID (for close/cleanup)
|
||||
"cdp_url": str, # CDP websocket URL
|
||||
"expires_at": str, # optional provider-authoritative ISO timestamp
|
||||
"features": dict, # feature flags that were enabled
|
||||
}
|
||||
|
||||
@@ -124,7 +126,7 @@ class BrowserProvider(abc.ABC):
|
||||
credentials, network errors, etc. — log and move on. Must not raise.
|
||||
"""
|
||||
|
||||
def get_setup_schema(self) -> Dict[str, Any]:
|
||||
def get_setup_schema(self) -> Optional[Dict[str, Any]]:
|
||||
"""Return provider metadata for the ``hermes tools`` picker.
|
||||
|
||||
Used by :mod:`hermes_cli.tools_config` to inject this provider as a
|
||||
|
||||
@@ -41,15 +41,19 @@ import threading
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from agent.browser_provider import BrowserProvider
|
||||
from hermes_constants import hermes_home_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
_providers: Dict[str, BrowserProvider] = {}
|
||||
_scoped_providers: Dict[str, Dict[str, BrowserProvider]] = {}
|
||||
_generation = 0
|
||||
_scoped_generations: Dict[str, int] = {}
|
||||
_lock = threading.Lock()
|
||||
|
||||
|
||||
def register_provider(provider: BrowserProvider) -> None:
|
||||
def register_provider(provider: BrowserProvider, *, scope: Optional[str] = None) -> None:
|
||||
"""Register a cloud browser provider.
|
||||
|
||||
Re-registration (same ``name``) overwrites the previous entry and logs
|
||||
@@ -61,12 +65,19 @@ def register_provider(provider: BrowserProvider) -> None:
|
||||
f"register_provider() expects a BrowserProvider instance, "
|
||||
f"got {type(provider).__name__}"
|
||||
)
|
||||
name = provider.name
|
||||
if not isinstance(name, str) or not name.strip():
|
||||
raw_name = provider.name
|
||||
if not isinstance(raw_name, str) or not raw_name.strip():
|
||||
raise ValueError("Browser provider .name must be a non-empty string")
|
||||
name = raw_name.strip()
|
||||
global _generation
|
||||
with _lock:
|
||||
existing = _providers.get(name)
|
||||
_providers[name] = provider
|
||||
target = _providers if scope is None else _scoped_providers.setdefault(scope, {})
|
||||
existing = target.get(name)
|
||||
target[name] = provider
|
||||
if scope is None:
|
||||
_generation += 1
|
||||
else:
|
||||
_scoped_generations[scope] = _scoped_generations.get(scope, 0) + 1
|
||||
if existing is not None:
|
||||
logger.debug(
|
||||
"Browser provider '%s' re-registered (was %r)",
|
||||
@@ -79,19 +90,64 @@ def register_provider(provider: BrowserProvider) -> None:
|
||||
)
|
||||
|
||||
|
||||
def list_providers() -> List[BrowserProvider]:
|
||||
def list_providers(*, scope: Optional[str] = None) -> List[BrowserProvider]:
|
||||
"""Return all registered providers, sorted by name."""
|
||||
with _lock:
|
||||
items = list(_providers.values())
|
||||
merged = dict(_providers)
|
||||
merged.update(_scoped_providers.get(scope or hermes_home_key(), {}))
|
||||
items = list(merged.values())
|
||||
return sorted(items, key=lambda p: p.name)
|
||||
|
||||
|
||||
def get_provider(name: str) -> Optional[BrowserProvider]:
|
||||
def get_provider(name: str, *, scope: Optional[str] = None) -> Optional[BrowserProvider]:
|
||||
"""Return the provider registered under *name*, or None."""
|
||||
if not isinstance(name, str):
|
||||
return None
|
||||
with _lock:
|
||||
return _providers.get(name.strip())
|
||||
key = name.strip()
|
||||
return _scoped_providers.get(scope or hermes_home_key(), {}).get(key) or _providers.get(key)
|
||||
|
||||
|
||||
def snapshot_registration(
|
||||
name: str, *, scope: Optional[str] = None
|
||||
) -> Optional[BrowserProvider]:
|
||||
with _lock:
|
||||
target = _providers if scope is None else _scoped_providers.get(scope, {})
|
||||
return target.get(name.strip())
|
||||
|
||||
|
||||
def registry_generation(*, scope: Optional[str] = None) -> tuple[int, int]:
|
||||
"""Return a cache fingerprint for the global base and one profile."""
|
||||
active_scope = scope or hermes_home_key()
|
||||
with _lock:
|
||||
return _generation, _scoped_generations.get(active_scope, 0)
|
||||
|
||||
|
||||
def restore_registration(
|
||||
name: str,
|
||||
current: BrowserProvider,
|
||||
previous: Optional[BrowserProvider],
|
||||
*,
|
||||
scope: Optional[str] = None,
|
||||
) -> bool:
|
||||
"""Restore a plugin registration only when *current* is still installed."""
|
||||
key = name.strip()
|
||||
global _generation
|
||||
with _lock:
|
||||
target = _providers if scope is None else _scoped_providers.setdefault(scope, {})
|
||||
if target.get(key) is not current:
|
||||
return False
|
||||
if previous is None:
|
||||
target.pop(key, None)
|
||||
else:
|
||||
target[key] = previous
|
||||
if scope is None:
|
||||
_generation += 1
|
||||
else:
|
||||
_scoped_generations[scope] = _scoped_generations.get(scope, 0) + 1
|
||||
if not target:
|
||||
_scoped_providers.pop(scope, None)
|
||||
return True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -145,6 +201,7 @@ def _resolve(configured: Optional[str]) -> Optional[BrowserProvider]:
|
||||
"""
|
||||
with _lock:
|
||||
snapshot = dict(_providers)
|
||||
snapshot.update(_scoped_providers.get(hermes_home_key(), {}))
|
||||
|
||||
def _is_available_safe(p: BrowserProvider) -> bool:
|
||||
"""Wrap ``is_available()`` so a buggy provider doesn't kill resolution."""
|
||||
@@ -188,5 +245,9 @@ def _resolve(configured: Optional[str]) -> Optional[BrowserProvider]:
|
||||
|
||||
def _reset_for_tests() -> None:
|
||||
"""Clear the registry. **Test-only.**"""
|
||||
global _generation
|
||||
with _lock:
|
||||
_providers.clear()
|
||||
_scoped_providers.clear()
|
||||
_scoped_generations.clear()
|
||||
_generation += 1
|
||||
|
||||
@@ -14,10 +14,12 @@ import hashlib
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import unicodedata
|
||||
import uuid
|
||||
from types import SimpleNamespace
|
||||
from typing import Any, Dict, List, Optional
|
||||
from typing import Any, Dict, List, NamedTuple, Optional
|
||||
|
||||
from agent.message_sanitization import deterministic_call_id
|
||||
from agent.prompt_builder import DEFAULT_AGENT_IDENTITY
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -72,6 +74,79 @@ _TOOL_CALL_LEAK_PATTERN = re.compile(
|
||||
)
|
||||
|
||||
|
||||
# The ChatGPT Codex backend reserves these Harmony wire tokens. If their
|
||||
# literal spellings are replayed anywhere in request text, the backend rejects
|
||||
# the request before inference with ``invalid_prompt: Request blocked.``.
|
||||
# Category-Cf handling covers persisted sessions from an earlier U+200B weak
|
||||
# defang; fullwidth bars survive format-character stripping while keeping the
|
||||
# inspected source legible.
|
||||
_HARMONY_CONTROL_TOKEN_RE = re.compile(
|
||||
r"<\|(start|end|channel|message|constrain|return|call)\|>"
|
||||
)
|
||||
_FULLWIDTH_PIPE = "\uff5c"
|
||||
|
||||
|
||||
def _neutralize_harmony_tokens(text: str) -> str:
|
||||
"""Keep Harmony source readable without emitting reserved wire tokens."""
|
||||
if not text or "<" not in text or "|" not in text:
|
||||
return text
|
||||
|
||||
replacement = rf"<{_FULLWIDTH_PIPE}\1{_FULLWIDTH_PIPE}>"
|
||||
if not any(unicodedata.category(char) == "Cf" for char in text):
|
||||
return _HARMONY_CONTROL_TOKEN_RE.sub(replacement, text)
|
||||
|
||||
# U+200B is confirmed to be stripped by the Codex backend before its
|
||||
# reserved-token check. Treat every Unicode format control equivalently so
|
||||
# moving the character elsewhere in the token (or swapping in another Cf)
|
||||
# cannot recreate the same visually hidden form.
|
||||
visible_chars: List[str] = []
|
||||
original_positions: List[int] = []
|
||||
for index, char in enumerate(text):
|
||||
if unicodedata.category(char) == "Cf":
|
||||
continue
|
||||
visible_chars.append(char)
|
||||
original_positions.append(index)
|
||||
|
||||
visible_text = "".join(visible_chars)
|
||||
matches = list(_HARMONY_CONTROL_TOKEN_RE.finditer(visible_text))
|
||||
if not matches:
|
||||
return text
|
||||
|
||||
result: List[str] = []
|
||||
original_cursor = 0
|
||||
for match in matches:
|
||||
original_start = original_positions[match.start()]
|
||||
original_end = original_positions[match.end() - 1] + 1
|
||||
result.append(text[original_cursor:original_start])
|
||||
result.append(f"<{_FULLWIDTH_PIPE}{match.group(1)}{_FULLWIDTH_PIPE}>")
|
||||
original_cursor = original_end
|
||||
result.append(text[original_cursor:])
|
||||
return "".join(result)
|
||||
|
||||
|
||||
def _neutralize_harmony_structure(value: Any) -> Any:
|
||||
"""Neutralize JSON-like values; normalize tuples and reject unsafe keys.
|
||||
|
||||
Rewriting an object key could desynchronize a tool schema from the executor
|
||||
contract, so a reserved token there is rejected explicitly instead.
|
||||
"""
|
||||
if isinstance(value, str):
|
||||
return _neutralize_harmony_tokens(value)
|
||||
if isinstance(value, (list, tuple)):
|
||||
return [_neutralize_harmony_structure(item) for item in value]
|
||||
if isinstance(value, dict):
|
||||
normalized = {}
|
||||
for key, item in value.items():
|
||||
if isinstance(key, str) and _neutralize_harmony_tokens(key) != key:
|
||||
raise ValueError(
|
||||
"Reserved Harmony tokens in a JSON object key cannot be "
|
||||
"neutralized without changing its contract."
|
||||
)
|
||||
normalized[key] = _neutralize_harmony_structure(item)
|
||||
return normalized
|
||||
return value
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Multimodal content helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -182,15 +257,92 @@ def _summarize_user_message_for_log(content: Any, *, sep: str = " ") -> str:
|
||||
def _deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
|
||||
"""Generate a deterministic call_id from tool call content.
|
||||
|
||||
Used as a fallback when the API doesn't provide a call_id.
|
||||
Thin wrapper over the single policy owner
|
||||
``agent.message_sanitization.deterministic_call_id`` (audit F4) — kept
|
||||
as a module-level name because run_agent and tests import it from here.
|
||||
Deterministic IDs prevent cache invalidation — random UUIDs would
|
||||
make every API call's prefix unique, breaking OpenAI's prompt cache.
|
||||
"""
|
||||
seed = f"{fn_name}:{arguments}:{index}"
|
||||
digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12]
|
||||
return deterministic_call_id(fn_name, arguments, index)
|
||||
|
||||
|
||||
def _clamp_responses_call_id(call_id: str) -> str:
|
||||
"""Keep a ``call_id`` within the Responses API's 64-char limit (#73492).
|
||||
|
||||
The codex app-server namespaces MCP tool call ids as
|
||||
``codex_mcp__<server>__<tool>_<codex_call_id>``; with an ``exec-<uuid>``
|
||||
component the built-in ``hermes-tools`` server already overflows 64 chars,
|
||||
and the Responses API rejects the whole payload with a non-retryable HTTP
|
||||
400 that then replays every turn — permanently bricking the session.
|
||||
|
||||
Sibling defect to #10788 (which clamped ``input[*].id``), applied here to
|
||||
``call_id``. The surrogate is a pure, deterministic function of the
|
||||
original, so the ``function_call`` and its matching ``function_call_output``
|
||||
— which carry the same original id — map to the same surrogate and stay
|
||||
paired without correlating the two items. Short ids pass through unchanged,
|
||||
preserving prompt-cache prefixes.
|
||||
"""
|
||||
if len(call_id) <= _MAX_RESPONSES_ITEM_ID_LENGTH:
|
||||
return call_id
|
||||
digest = hashlib.sha256(call_id.encode("utf-8", errors="replace")).hexdigest()[:32]
|
||||
return f"call_{digest}"
|
||||
|
||||
|
||||
# The Responses API enforces the same 64-char cap on function names as on
|
||||
# input item ids (_MAX_RESPONSES_ITEM_ID_LENGTH) — names over the cap are
|
||||
# rejected with the same non-retryable 400 as pattern violations.
|
||||
_VALID_RESPONSES_FN_NAME_RE = re.compile(r"[a-zA-Z0-9_-]{1,64}")
|
||||
|
||||
|
||||
def _sanitize_replayed_fn_name(name: str) -> str:
|
||||
"""Coerce a *replayed* function_call name to the Responses API contract.
|
||||
|
||||
The Responses API requires ``function_call.name`` to match
|
||||
``^[a-zA-Z0-9_-]+$`` and rejects the whole request with a non-retryable
|
||||
HTTP 400 otherwise (issue #31666). A name with invalid characters (dots,
|
||||
spaces, unicode — e.g. from an earlier model degeneration) stored in
|
||||
conversation history therefore bricks every subsequent turn of the
|
||||
session: the 400 replays forever until the user manually starts a new
|
||||
conversation.
|
||||
|
||||
Invalid characters are replaced with ``_`` (runs collapsed) rather than
|
||||
stripped, so an all-invalid name degrades to the ``"fn"`` placeholder
|
||||
instead of an empty string — an empty name would just trade one
|
||||
non-retryable 400 for a preflight ValueError. Valid names pass through
|
||||
unchanged, preserving prompt-cache prefixes.
|
||||
|
||||
Apply this ONLY to replayed function_call input items, never to live
|
||||
tool definitions: tool schema names must match the dispatch registry
|
||||
exactly. Pairing with function_call_output is by call_id, so renaming
|
||||
a replayed function_call is safe.
|
||||
"""
|
||||
if not isinstance(name, str):
|
||||
return "fn"
|
||||
if _VALID_RESPONSES_FN_NAME_RE.fullmatch(name):
|
||||
return name
|
||||
coerced = re.sub(r"[^A-Za-z0-9_-]", "_", name.strip())
|
||||
coerced = re.sub(r"_+", "_", coerced).strip("_")
|
||||
return coerced[:64] or "fn"
|
||||
|
||||
|
||||
def _canonical_call_id_from_fc(response_item_id: Any) -> Optional[str]:
|
||||
"""Map an ``fc_…`` response-item id to its canonical ``call_<suffix>``.
|
||||
|
||||
Both sides of a replayed pair — the assistant ``function_call`` and the
|
||||
tool ``function_call_output`` — must derive the SAME call_id from an
|
||||
fc_-only stored id, or an oversized pair clamps to two different
|
||||
surrogates and the API rejects the output as unmatched. Keep every
|
||||
caller on this single helper.
|
||||
"""
|
||||
if (
|
||||
isinstance(response_item_id, str)
|
||||
and response_item_id.startswith("fc_")
|
||||
and len(response_item_id) > len("fc_")
|
||||
):
|
||||
return f"call_{response_item_id[len('fc_'):]}"
|
||||
return None
|
||||
|
||||
|
||||
def _split_responses_tool_id(raw_id: Any) -> tuple[Optional[str], Optional[str]]:
|
||||
"""Split a stored tool id into (call_id, response_item_id)."""
|
||||
if not isinstance(raw_id, str):
|
||||
@@ -317,6 +469,7 @@ def _chat_messages_to_responses_input(
|
||||
is_github_responses: bool = False,
|
||||
replay_encrypted_reasoning: bool = True,
|
||||
current_issuer_kind: Optional[str] = None,
|
||||
native_compaction_eligible: bool = False,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Convert internal chat-style messages to Responses input items.
|
||||
|
||||
@@ -361,8 +514,32 @@ def _chat_messages_to_responses_input(
|
||||
``replay_encrypted_reasoning=False`` is the session-wide kill switch
|
||||
(drops ALL replay); ``current_issuer_kind`` is the per-item filter
|
||||
that runs only when replay is still enabled.
|
||||
|
||||
``native_compaction_eligible`` mirrors, for THIS request, the decision
|
||||
made by ``native_compaction.native_compaction_context_management`` — it
|
||||
is True only when that gate returned a payload, i.e. when the request
|
||||
actually carries ``context_management``. It controls two things that
|
||||
must never outlive the gate: replaying ``type: "compaction"`` checkpoint
|
||||
items, and restructuring the wire around them
|
||||
(``prune_pre_checkpoint_items``). Checkpoints are persisted in the
|
||||
``codex_reasoning_items`` sidecar and survive a mid-session model swap,
|
||||
a ``compression.enabled: false`` flip, the rejection kill switch and a
|
||||
resumed session; without this flag a single captured checkpoint would
|
||||
keep deleting every pre-checkpoint item from every later request, on a
|
||||
model that cannot decrypt the blob (#85914). Default False = pre-feature
|
||||
wire, which is also correct for every caller that never sends
|
||||
``context_management`` (auxiliary/compression client, ad-hoc
|
||||
``convert_messages``). Dropping the checkpoint costs nothing: Hermes'
|
||||
local history is never truncated by native compaction, so the full
|
||||
conversation is still on the wire.
|
||||
"""
|
||||
items: List[Dict[str, Any]] = []
|
||||
# Parallel to `items`: the raw chat message each converted item came
|
||||
# from. Pruning needs this to read a canonical summary carrier's
|
||||
# up-to-date, provenance-tagged content directly — the converted `item`
|
||||
# can be a lossy shape (stale exact-replay, or a typed
|
||||
# `function_call_output` wrapper) that no longer carries it (#90976).
|
||||
item_sources: List[Optional[Dict[str, Any]]] = []
|
||||
seen_item_ids: set = set()
|
||||
|
||||
for msg in messages:
|
||||
@@ -402,6 +579,20 @@ def _chat_messages_to_responses_input(
|
||||
item_id = ri.get("id")
|
||||
if item_id and item_id in seen_item_ids:
|
||||
continue
|
||||
# Native-compaction gate: a checkpoint is only
|
||||
# meaningful to the endpoint/model that minted it
|
||||
# AND only while this request still asks for
|
||||
# server-side compaction. Once the gate closes
|
||||
# (model swapped out of the gpt-5.6 family,
|
||||
# compression disabled, rejection kill switch),
|
||||
# the persisted checkpoint must not be replayed —
|
||||
# replaying it is what makes the wire restructure
|
||||
# below erase pre-checkpoint history forever.
|
||||
if (
|
||||
ri.get("type") == "compaction"
|
||||
and not native_compaction_eligible
|
||||
):
|
||||
continue
|
||||
# Cross-issuer guard: drop reasoning blocks that
|
||||
# were minted by a different Responses endpoint.
|
||||
# The current endpoint cannot decrypt foreign
|
||||
@@ -437,6 +628,7 @@ def _chat_messages_to_responses_input(
|
||||
if k not in ("id", "_issuer_kind")
|
||||
}
|
||||
items.append(replay_item)
|
||||
item_sources.append(msg)
|
||||
if item_id:
|
||||
seen_item_ids.add(item_id)
|
||||
has_codex_reasoning = True
|
||||
@@ -493,14 +685,17 @@ def _chat_messages_to_responses_input(
|
||||
if isinstance(phase, str) and phase.strip():
|
||||
replay_item["phase"] = phase.strip()
|
||||
items.append(replay_item)
|
||||
item_sources.append(msg)
|
||||
replayed_message_items += 1
|
||||
|
||||
if replayed_message_items > 0:
|
||||
pass
|
||||
elif content_parts:
|
||||
items.append({"role": "assistant", "content": content_parts})
|
||||
item_sources.append(msg)
|
||||
elif content_text.strip():
|
||||
items.append({"role": "assistant", "content": content_text})
|
||||
item_sources.append(msg)
|
||||
elif has_codex_reasoning:
|
||||
# The Responses API requires a following item after each
|
||||
# reasoning item (otherwise: missing_following_item error).
|
||||
@@ -508,6 +703,7 @@ def _chat_messages_to_responses_input(
|
||||
# content, emit an empty assistant message as the required
|
||||
# following item.
|
||||
items.append({"role": "assistant", "content": ""})
|
||||
item_sources.append(msg)
|
||||
|
||||
tool_calls = msg.get("tool_calls")
|
||||
if isinstance(tool_calls, list):
|
||||
@@ -526,13 +722,8 @@ def _chat_messages_to_responses_input(
|
||||
if not isinstance(call_id, str) or not call_id.strip():
|
||||
call_id = embedded_call_id
|
||||
if not isinstance(call_id, str) or not call_id.strip():
|
||||
if (
|
||||
isinstance(embedded_response_item_id, str)
|
||||
and embedded_response_item_id.startswith("fc_")
|
||||
and len(embedded_response_item_id) > len("fc_")
|
||||
):
|
||||
call_id = f"call_{embedded_response_item_id[len('fc_'):]}"
|
||||
else:
|
||||
call_id = _canonical_call_id_from_fc(embedded_response_item_id)
|
||||
if call_id is None:
|
||||
_raw_args = str(fn.get("arguments", "{}"))
|
||||
call_id = _deterministic_call_id(fn_name, _raw_args, len(items))
|
||||
call_id = call_id.strip()
|
||||
@@ -546,10 +737,11 @@ def _chat_messages_to_responses_input(
|
||||
|
||||
items.append({
|
||||
"type": "function_call",
|
||||
"call_id": call_id,
|
||||
"name": fn_name,
|
||||
"call_id": _clamp_responses_call_id(call_id),
|
||||
"name": _sanitize_replayed_fn_name(fn_name),
|
||||
"arguments": arguments,
|
||||
})
|
||||
item_sources.append(msg)
|
||||
continue
|
||||
|
||||
# Non-assistant (user) role: emit multimodal parts when present,
|
||||
@@ -558,13 +750,18 @@ def _chat_messages_to_responses_input(
|
||||
items.append({"role": role, "content": content_parts})
|
||||
else:
|
||||
items.append({"role": role, "content": content_text})
|
||||
item_sources.append(msg)
|
||||
continue
|
||||
|
||||
if role == "tool":
|
||||
raw_tool_call_id = msg.get("tool_call_id")
|
||||
call_id, _ = _split_responses_tool_id(raw_tool_call_id)
|
||||
call_id, tool_response_item_id = _split_responses_tool_id(raw_tool_call_id)
|
||||
if not isinstance(call_id, str) or not call_id.strip():
|
||||
if isinstance(raw_tool_call_id, str) and raw_tool_call_id.strip():
|
||||
# Legacy fc_-only stored ids: canonicalize to the same
|
||||
# ``call_<suffix>`` the assistant branch synthesizes above, so
|
||||
# a >64-char pair clamps to the SAME surrogate on both sides.
|
||||
call_id = _canonical_call_id_from_fc(tool_response_item_id)
|
||||
if call_id is None and isinstance(raw_tool_call_id, str) and raw_tool_call_id.strip():
|
||||
call_id = raw_tool_call_id.strip()
|
||||
if not isinstance(call_id, str) or not call_id.strip():
|
||||
continue
|
||||
@@ -589,11 +786,158 @@ def _chat_messages_to_responses_input(
|
||||
|
||||
items.append({
|
||||
"type": "function_call_output",
|
||||
"call_id": call_id,
|
||||
"call_id": _clamp_responses_call_id(call_id),
|
||||
"output": output_value,
|
||||
})
|
||||
item_sources.append(msg)
|
||||
|
||||
return items
|
||||
# Native server-side compaction: when a replayed checkpoint is present,
|
||||
# restructure the wire around it. The server renders nothing placed
|
||||
# before a compaction item (live-verified Aug 2026), so pre-checkpoint
|
||||
# history is dead upload weight and — worse — the user's plaintext asks,
|
||||
# and any local-compression summary already merged into that history,
|
||||
# silently vanish from the model's view. Keep the newest checkpoint
|
||||
# first, retain pre-checkpoint USER messages and compression-SUMMARY
|
||||
# messages (whole, never byte-sliced) verbatim within a token budget
|
||||
# each (Codex CLI parity for the user side), and leave the
|
||||
# post-checkpoint tail untouched. Gated on the CURRENT request's native
|
||||
# eligibility, not merely on the presence of a checkpoint: a persisted
|
||||
# checkpoint outlives the gate, and pruning for a request that carries no
|
||||
# ``context_management`` deletes history the server never compacted.
|
||||
#
|
||||
# ``item_sources`` (parallel to ``items``) carries the raw chat message
|
||||
# each converted item came from. A canonical summary carrier's content
|
||||
# can be lost or gone stale by the time it becomes a Responses item — a
|
||||
# merge-into-tail tool-result carrier becomes a typed
|
||||
# ``function_call_output`` (no ``content``/``role`` at all), and a
|
||||
# merge-into-tail assistant carrier can be shadowed by a stale exact
|
||||
# ``codex_message_items`` replay from before the merge rewrote its
|
||||
# content. Pruning reads the source message's own up-to-date,
|
||||
# provenance-tagged content directly instead of trying to recover it
|
||||
# from whatever shape the conversion produced (#90976).
|
||||
if not native_compaction_eligible:
|
||||
return items
|
||||
|
||||
from agent.native_compaction import prune_pre_checkpoint_items
|
||||
|
||||
return prune_pre_checkpoint_items(items, item_sources=item_sources)
|
||||
|
||||
|
||||
class ResponsesRouteFlags(NamedTuple):
|
||||
"""Which special Responses-API route an agent is talking to.
|
||||
|
||||
Single owner of the codex/xai/github route predicates. Every site that
|
||||
needs these flags (request kwargs build, preflight estimation, silent-
|
||||
reject hints) must call :func:`classify_responses_route` instead of
|
||||
re-implementing the string comparisons inline — inline copies drift
|
||||
(backend-identity class: #22548/#70893/#59561/#72468).
|
||||
"""
|
||||
|
||||
is_codex_backend: bool
|
||||
is_xai_responses: bool
|
||||
is_github_responses: bool
|
||||
|
||||
|
||||
def classify_responses_route(agent: Any) -> ResponsesRouteFlags:
|
||||
"""Classify the agent's Responses route from provider + base URL.
|
||||
|
||||
Host checks are exact-host-or-subdomain (``base_url_hostname``
|
||||
semantics), never substring matching — ``https://evil.com/models.github.ai``
|
||||
must not classify as a GitHub route.
|
||||
"""
|
||||
from utils import base_url_hostname
|
||||
|
||||
provider = getattr(agent, "provider", None)
|
||||
base_url = str(getattr(agent, "base_url", "") or "")
|
||||
hostname = str(getattr(agent, "_base_url_hostname", "") or "").lower()
|
||||
if not hostname:
|
||||
hostname = base_url_hostname(base_url)
|
||||
lower = str(getattr(agent, "_base_url_lower", "") or base_url).lower()
|
||||
|
||||
def _host_is(domain: str) -> bool:
|
||||
return hostname == domain or hostname.endswith("." + domain)
|
||||
|
||||
is_codex_backend = provider == "openai-codex" or (
|
||||
_host_is("chatgpt.com") and "/backend-api/codex" in lower
|
||||
)
|
||||
is_github_responses = _host_is("models.github.ai") or _host_is("githubcopilot.com")
|
||||
is_xai_responses = provider in {"xai", "xai-oauth"} or hostname == "api.x.ai"
|
||||
return ResponsesRouteFlags(
|
||||
is_codex_backend=is_codex_backend,
|
||||
is_xai_responses=is_xai_responses,
|
||||
is_github_responses=is_github_responses,
|
||||
)
|
||||
|
||||
|
||||
def estimate_native_responses_preflight_tokens(
|
||||
agent: Any,
|
||||
messages: List[Dict[str, Any]],
|
||||
*,
|
||||
system_prompt: str = "",
|
||||
tools: Optional[List[Dict[str, Any]]] = None,
|
||||
) -> Optional[int]:
|
||||
"""Estimate tokens for the checkpoint-pruned Responses payload.
|
||||
|
||||
Automatic preflight previously counted the full durable transcript.
|
||||
On a natively compacted Codex session that overstates the wire by
|
||||
several times and fires local compression against history the main
|
||||
request will never send (#96155).
|
||||
|
||||
Returns None when native compaction is not proven eligible for this
|
||||
request, or when conversion fails — the caller must then use the
|
||||
generic durable-transcript estimate (conservative).
|
||||
"""
|
||||
if getattr(agent, "api_mode", None) != "codex_responses":
|
||||
return None
|
||||
if not isinstance(messages, list):
|
||||
return None
|
||||
|
||||
is_codex_backend, is_xai_responses, is_github_responses = classify_responses_route(agent)
|
||||
|
||||
from agent.native_compaction import native_compaction_context_management
|
||||
|
||||
context_management = native_compaction_context_management(
|
||||
agent,
|
||||
is_codex_backend=is_codex_backend,
|
||||
is_xai_responses=is_xai_responses,
|
||||
is_github_responses=is_github_responses,
|
||||
)
|
||||
if not context_management:
|
||||
return None
|
||||
|
||||
try:
|
||||
items = _chat_messages_to_responses_input(
|
||||
messages,
|
||||
is_xai_responses=is_xai_responses,
|
||||
is_github_responses=is_github_responses,
|
||||
replay_encrypted_reasoning=bool(
|
||||
getattr(agent, "_codex_reasoning_replay_enabled", True)
|
||||
),
|
||||
current_issuer_kind=_classify_responses_issuer(
|
||||
is_xai_responses=is_xai_responses,
|
||||
is_github_responses=is_github_responses,
|
||||
is_codex_backend=is_codex_backend,
|
||||
base_url=getattr(agent, "base_url", None),
|
||||
),
|
||||
native_compaction_eligible=True,
|
||||
)
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"native Responses preflight conversion failed; falling back to generic estimate",
|
||||
exc_info=True,
|
||||
)
|
||||
return None
|
||||
|
||||
if not isinstance(items, list):
|
||||
return None
|
||||
|
||||
from agent.model_metadata import estimate_request_tokens_rough
|
||||
|
||||
return estimate_request_tokens_rough(
|
||||
items,
|
||||
system_prompt=system_prompt or "",
|
||||
tools=tools,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -604,10 +948,16 @@ def _preflight_codex_input_items(
|
||||
raw_items: Any,
|
||||
*,
|
||||
is_github_responses: bool = False,
|
||||
sanitize_harmony_tokens: bool = False,
|
||||
) -> List[Dict[str, Any]]:
|
||||
if not isinstance(raw_items, list):
|
||||
raise ValueError("Codex Responses input must be a list of input items.")
|
||||
|
||||
sanitize_text = (
|
||||
_neutralize_harmony_tokens
|
||||
if sanitize_harmony_tokens
|
||||
else lambda text: text
|
||||
)
|
||||
normalized: List[Dict[str, Any]] = []
|
||||
seen_ids: set = set()
|
||||
for idx, item in enumerate(raw_items):
|
||||
@@ -628,13 +978,13 @@ def _preflight_codex_input_items(
|
||||
arguments = json.dumps(arguments, ensure_ascii=False)
|
||||
elif not isinstance(arguments, str):
|
||||
arguments = str(arguments)
|
||||
arguments = arguments.strip() or "{}"
|
||||
arguments = sanitize_text(arguments.strip() or "{}")
|
||||
|
||||
normalized.append(
|
||||
{
|
||||
"type": "function_call",
|
||||
"call_id": call_id.strip(),
|
||||
"name": name.strip(),
|
||||
"name": _sanitize_replayed_fn_name(name),
|
||||
"arguments": arguments,
|
||||
}
|
||||
)
|
||||
@@ -662,7 +1012,7 @@ def _preflight_codex_input_items(
|
||||
if ptype == "input_text":
|
||||
text = part.get("text")
|
||||
if isinstance(text, str) and text:
|
||||
cleaned.append({"type": "input_text", "text": text})
|
||||
cleaned.append({"type": "input_text", "text": sanitize_text(text)})
|
||||
elif ptype == "input_image":
|
||||
url = part.get("image_url")
|
||||
if isinstance(url, str) and url:
|
||||
@@ -686,7 +1036,7 @@ def _preflight_codex_input_items(
|
||||
{
|
||||
"type": "function_call_output",
|
||||
"call_id": call_id.strip(),
|
||||
"output": output,
|
||||
"output": sanitize_text(output),
|
||||
}
|
||||
)
|
||||
continue
|
||||
@@ -699,19 +1049,37 @@ def _preflight_codex_input_items(
|
||||
if item_id in seen_ids:
|
||||
continue
|
||||
seen_ids.add(item_id)
|
||||
reasoning_item = {"type": "reasoning", "encrypted_content": encrypted}
|
||||
reasoning_item: Dict[str, Any] = {
|
||||
"type": "reasoning",
|
||||
"encrypted_content": encrypted,
|
||||
}
|
||||
# Do NOT include the "id" in the outgoing item — with
|
||||
# store=False (our default) the API tries to resolve the
|
||||
# id server-side and returns 404. The id is still used
|
||||
# above for local deduplication via seen_ids.
|
||||
summary = item.get("summary")
|
||||
if isinstance(summary, list):
|
||||
reasoning_item["summary"] = summary
|
||||
reasoning_item["summary"] = (
|
||||
_neutralize_harmony_structure(summary)
|
||||
if sanitize_harmony_tokens
|
||||
else summary
|
||||
)
|
||||
else:
|
||||
reasoning_item["summary"] = []
|
||||
normalized.append(reasoning_item)
|
||||
continue
|
||||
|
||||
if item_type == "compaction":
|
||||
# Replayed native server-side compaction checkpoint (gpt-5.6,
|
||||
# direct OpenAI/Codex routes). Opaque, issuer-sealed; forward
|
||||
# only the fields the API defines.
|
||||
encrypted = item.get("encrypted_content")
|
||||
if isinstance(encrypted, str) and encrypted:
|
||||
normalized.append(
|
||||
{"type": "compaction", "encrypted_content": encrypted}
|
||||
)
|
||||
continue
|
||||
|
||||
if item_type == "message":
|
||||
role = item.get("role")
|
||||
if role != "assistant":
|
||||
@@ -735,7 +1103,7 @@ def _preflight_codex_input_items(
|
||||
text = ""
|
||||
if not isinstance(text, str):
|
||||
text = str(text)
|
||||
normalized_content.append({"type": "output_text", "text": text})
|
||||
normalized_content.append({"type": "output_text", "text": sanitize_text(text)})
|
||||
if not normalized_content:
|
||||
raise ValueError(f"Codex Responses input[{idx}] message item must contain at least one text part.")
|
||||
normalized_item: Dict[str, Any] = {
|
||||
@@ -775,7 +1143,7 @@ def _preflight_codex_input_items(
|
||||
for part_idx, part in enumerate(content):
|
||||
if isinstance(part, str):
|
||||
if part:
|
||||
validated.append({"type": text_type, "text": part})
|
||||
validated.append({"type": text_type, "text": sanitize_text(part)})
|
||||
continue
|
||||
if not isinstance(part, dict):
|
||||
raise ValueError(
|
||||
@@ -786,7 +1154,7 @@ def _preflight_codex_input_items(
|
||||
text = part.get("text", "")
|
||||
if not isinstance(text, str):
|
||||
text = str(text or "")
|
||||
validated.append({"type": text_type, "text": text})
|
||||
validated.append({"type": text_type, "text": sanitize_text(text)})
|
||||
elif ptype in {"input_image", "image_url"}:
|
||||
image_ref = part.get("image_url", "")
|
||||
detail = part.get("detail")
|
||||
@@ -810,7 +1178,7 @@ def _preflight_codex_input_items(
|
||||
if not isinstance(content, str):
|
||||
content = str(content)
|
||||
|
||||
normalized.append({"role": role, "content": content})
|
||||
normalized.append({"role": role, "content": sanitize_text(content)})
|
||||
continue
|
||||
|
||||
raise ValueError(
|
||||
@@ -825,6 +1193,7 @@ def _preflight_codex_api_kwargs(
|
||||
*,
|
||||
allow_stream: bool = False,
|
||||
is_github_responses: bool = False,
|
||||
sanitize_harmony_tokens: bool = False,
|
||||
) -> Dict[str, Any]:
|
||||
if not isinstance(api_kwargs, dict):
|
||||
raise ValueError("Codex Responses request must be a dict.")
|
||||
@@ -845,10 +1214,13 @@ def _preflight_codex_api_kwargs(
|
||||
if not isinstance(instructions, str):
|
||||
instructions = str(instructions)
|
||||
instructions = instructions.strip() or DEFAULT_AGENT_IDENTITY
|
||||
if sanitize_harmony_tokens:
|
||||
instructions = _neutralize_harmony_tokens(instructions)
|
||||
|
||||
normalized_input = _preflight_codex_input_items(
|
||||
api_kwargs.get("input"),
|
||||
is_github_responses=is_github_responses,
|
||||
sanitize_harmony_tokens=sanitize_harmony_tokens,
|
||||
)
|
||||
|
||||
tools = api_kwargs.get("tools")
|
||||
@@ -905,6 +1277,9 @@ def _preflight_codex_api_kwargs(
|
||||
}
|
||||
)
|
||||
|
||||
if sanitize_harmony_tokens and normalized_tools is not None:
|
||||
normalized_tools = _neutralize_harmony_structure(normalized_tools)
|
||||
|
||||
store = api_kwargs.get("store", False)
|
||||
if store is not False:
|
||||
raise ValueError("Codex Responses contract requires 'store' to be false.")
|
||||
@@ -912,7 +1287,8 @@ def _preflight_codex_api_kwargs(
|
||||
allowed_keys = {
|
||||
"model", "instructions", "input", "tools", "store",
|
||||
"reasoning", "include", "max_output_tokens", "temperature",
|
||||
"tool_choice", "parallel_tool_calls", "prompt_cache_key", "service_tier",
|
||||
"tool_choice", "parallel_tool_calls", "prompt_cache_key",
|
||||
"prompt_cache_retention", "service_tier", "context_management",
|
||||
"extra_headers", "extra_body", "timeout",
|
||||
}
|
||||
normalized: Dict[str, Any] = {
|
||||
@@ -950,12 +1326,24 @@ def _preflight_codex_api_kwargs(
|
||||
if isinstance(temperature, (int, float)):
|
||||
normalized["temperature"] = float(temperature)
|
||||
|
||||
# Pass through tool_choice, parallel_tool_calls, prompt_cache_key
|
||||
for passthrough_key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key"):
|
||||
# Pass through cache routing/retention and tool-dispatch hints.
|
||||
for passthrough_key in (
|
||||
"tool_choice",
|
||||
"parallel_tool_calls",
|
||||
"prompt_cache_key",
|
||||
"prompt_cache_retention",
|
||||
):
|
||||
val = api_kwargs.get(passthrough_key)
|
||||
if val is not None:
|
||||
normalized[passthrough_key] = val
|
||||
|
||||
# Native server-side compaction directive (gpt-5.6 on direct OpenAI /
|
||||
# Codex routes — eligibility already resolved upstream in
|
||||
# agent/native_compaction.py; the preflight only preserves the shape).
|
||||
context_management = api_kwargs.get("context_management")
|
||||
if isinstance(context_management, list) and context_management:
|
||||
normalized["context_management"] = context_management
|
||||
|
||||
extra_headers = api_kwargs.get("extra_headers")
|
||||
if extra_headers is not None:
|
||||
if not isinstance(extra_headers, dict):
|
||||
@@ -1293,6 +1681,23 @@ def _normalize_codex_response(
|
||||
raw_summary.append({"type": "summary_text", "text": text})
|
||||
raw_item["summary"] = raw_summary
|
||||
reasoning_items_raw.append(raw_item)
|
||||
elif item_type == "compaction":
|
||||
# Native server-side compaction checkpoint (gpt-5.6 on direct
|
||||
# OpenAI/Codex routes). The encrypted blob stands in for the
|
||||
# pruned older context on subsequent requests. It rides the
|
||||
# codex_reasoning_items sidecar so it inherits persistence
|
||||
# (state.db), session replay, the cross-issuer guard, and the
|
||||
# invalid-encrypted-content kill switch without new state.
|
||||
encrypted = getattr(item, "encrypted_content", None)
|
||||
if isinstance(encrypted, str) and encrypted:
|
||||
raw_item = {"type": "compaction", "encrypted_content": encrypted}
|
||||
if issuer_kind:
|
||||
raw_item["_issuer_kind"] = issuer_kind
|
||||
reasoning_items_raw.append(raw_item)
|
||||
logger.info(
|
||||
"Native Responses compaction item captured (%d chars encrypted).",
|
||||
len(encrypted),
|
||||
)
|
||||
elif item_type == "function_call":
|
||||
if item_status in {"queued", "in_progress", "incomplete"}:
|
||||
continue
|
||||
|
||||
@@ -28,6 +28,65 @@ from agent.stream_single_writer import claim_stream_writer, stream_writer_is_cur
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _codex_request_failure_details(error: BaseException) -> tuple[int | None, str]:
|
||||
"""Return the serialized request size and exception class chain.
|
||||
|
||||
OpenAI connection exceptions retain the final ``httpx.Request``. Reading
|
||||
its already-buffered content gives us the exact byte count handed to the
|
||||
transport without logging any request content. The class-only chain keeps
|
||||
the underlying transport failure visible without exposing URLs or payloads
|
||||
from exception messages.
|
||||
"""
|
||||
request_body_bytes: int | None = None
|
||||
exception_classes: list[str] = []
|
||||
current: BaseException | None = error
|
||||
seen: set[int] = set()
|
||||
|
||||
while current is not None and id(current) not in seen and len(seen) < 8:
|
||||
seen.add(id(current))
|
||||
exception_classes.append(type(current).__name__)
|
||||
|
||||
if request_body_bytes is None:
|
||||
try:
|
||||
request = getattr(current, "request", None)
|
||||
except Exception:
|
||||
request = None
|
||||
if request is not None:
|
||||
try:
|
||||
content = request.content
|
||||
except Exception:
|
||||
content = None
|
||||
if isinstance(content, str):
|
||||
request_body_bytes = len(content.encode("utf-8"))
|
||||
elif isinstance(content, (bytes, bytearray, memoryview)):
|
||||
request_body_bytes = len(content)
|
||||
|
||||
cause = current.__cause__
|
||||
if cause is None and not current.__suppress_context__:
|
||||
cause = current.__context__
|
||||
current = cause
|
||||
|
||||
return request_body_bytes, " <- ".join(exception_classes)
|
||||
|
||||
|
||||
def _log_codex_request_failure(
|
||||
agent: Any,
|
||||
error: BaseException,
|
||||
*,
|
||||
stream_opened: bool,
|
||||
) -> None:
|
||||
request_body_bytes, exception_chain = _codex_request_failure_details(error)
|
||||
logger.warning(
|
||||
"Codex Responses request failed: "
|
||||
"serialized_request_body_bytes=%s stream_opened=%s "
|
||||
"exception_chain=%s model=%s",
|
||||
request_body_bytes if request_body_bytes is not None else "unknown",
|
||||
str(stream_opened).lower(),
|
||||
exception_chain,
|
||||
getattr(agent, "model", "unknown"),
|
||||
)
|
||||
|
||||
|
||||
def _coerce_usage_int(value: Any) -> int:
|
||||
if isinstance(value, bool):
|
||||
return 0
|
||||
@@ -74,7 +133,10 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
|
||||
try:
|
||||
if not agent._session_db_created:
|
||||
agent._ensure_db_session()
|
||||
agent._session_db.update_token_counts(
|
||||
# Enqueued for the SessionDB background writer — keeps the
|
||||
# per-call accounting write off the turn thread (see
|
||||
# conversation_loop's queue_token_counts call).
|
||||
agent._session_db.queue_token_counts(
|
||||
agent.session_id,
|
||||
model=agent.model,
|
||||
billing_provider=agent.provider,
|
||||
@@ -154,7 +216,8 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
|
||||
try:
|
||||
if not agent._session_db_created:
|
||||
agent._ensure_db_session()
|
||||
agent._session_db.update_token_counts(
|
||||
# Enqueued for the SessionDB background writer (see above).
|
||||
agent._session_db.queue_token_counts(
|
||||
agent.session_id,
|
||||
input_tokens=canonical_usage.input_tokens,
|
||||
output_tokens=canonical_usage.output_tokens,
|
||||
@@ -628,6 +691,20 @@ def run_codex_app_server_turn(
|
||||
Called from run_conversation() when agent.api_mode == "codex_app_server".
|
||||
Returns the same dict shape as the chat_completions path.
|
||||
"""
|
||||
# Defense in depth for compression.checkpoint_required: agent init
|
||||
# already refuses this combination, but api_mode is a plain attribute a
|
||||
# future code path could mutate on a live agent. Fail closed before the
|
||||
# codex agent can compact its thread — once run_turn() executes, a
|
||||
# codex-owned compaction may already have happened with no pre-compress
|
||||
# checkpoint. Explicit-True check matches the compress_context() gate.
|
||||
if getattr(agent, "compression_checkpoint_required", False) is True:
|
||||
from agent.conversation_compression import _checkpoint_blocked
|
||||
|
||||
raise _checkpoint_blocked(
|
||||
"codex_app_server owns the authoritative thread and compacts it "
|
||||
"without a truthful pre-compaction transcript boundary"
|
||||
)
|
||||
|
||||
from agent.transports.codex_app_server_session import (
|
||||
CodexAppServerSession,
|
||||
_ServerRequestRouting,
|
||||
@@ -702,6 +779,16 @@ def run_codex_app_server_turn(
|
||||
except Exception:
|
||||
pass
|
||||
agent._codex_session = None
|
||||
_user_interrupted = bool(
|
||||
getattr(agent, "_interrupt_requested", False)
|
||||
)
|
||||
_interrupt_message = (
|
||||
getattr(agent, "_interrupt_message", None)
|
||||
if _user_interrupted
|
||||
else None
|
||||
)
|
||||
if _user_interrupted:
|
||||
agent.clear_interrupt()
|
||||
return {
|
||||
"final_response": (
|
||||
f"Codex app-server turn failed: {exc}. "
|
||||
@@ -711,9 +798,27 @@ def run_codex_app_server_turn(
|
||||
"api_calls": 0,
|
||||
"completed": False,
|
||||
"partial": True,
|
||||
"interrupted": _user_interrupted,
|
||||
**(
|
||||
{"interrupt_message": _interrupt_message}
|
||||
if _interrupt_message
|
||||
else {}
|
||||
),
|
||||
"error": str(exc),
|
||||
}
|
||||
|
||||
# This runtime bypasses the normal conversation-loop finalizer. Mirror its
|
||||
# interrupt handoff/cleanup so a hard stop cannot poison the next turn and a
|
||||
# message-bearing compatibility interrupt can still be replayed by callers.
|
||||
_user_interrupted = bool(
|
||||
turn.interrupted and getattr(agent, "_interrupt_requested", False)
|
||||
)
|
||||
_interrupt_message = (
|
||||
getattr(agent, "_interrupt_message", None) if _user_interrupted else None
|
||||
)
|
||||
if _user_interrupted:
|
||||
agent.clear_interrupt()
|
||||
|
||||
# If the turn signalled the underlying client is wedged (deadline
|
||||
# blown, post-tool watchdog tripped, OAuth refresh died, subprocess
|
||||
# exited), retire the session so the next turn respawns codex
|
||||
@@ -734,7 +839,10 @@ def run_codex_app_server_turn(
|
||||
# standard {role, content, tool_calls, tool_call_id} entries, which
|
||||
# is exactly what curator.py / sessions DB expect.
|
||||
if turn.projected_messages:
|
||||
messages.extend(turn.projected_messages)
|
||||
from agent.message_metadata import append_message
|
||||
|
||||
for projected_message in turn.projected_messages:
|
||||
append_message(messages, projected_message)
|
||||
|
||||
# Persist the newly-projected assistant/tool messages ourselves.
|
||||
# This path is an early return that bypasses conversation_loop, whose
|
||||
@@ -750,12 +858,27 @@ def run_codex_app_server_turn(
|
||||
# the already-flushed user turn). See gateway/run.py agent_persisted.
|
||||
if getattr(agent, "_session_db", None) is not None:
|
||||
try:
|
||||
agent._flush_messages_to_session_db(messages)
|
||||
_codex_flush_ok = agent._flush_messages_to_session_db(messages)
|
||||
except Exception:
|
||||
logger.debug(
|
||||
_codex_flush_ok = False
|
||||
logger.warning(
|
||||
"codex app-server projected-message flush failed",
|
||||
exc_info=True,
|
||||
)
|
||||
if _codex_flush_ok is False:
|
||||
# Unlike the chat-completions loop (which fails closed BEFORE
|
||||
# projection — see conversation_loop session_persistence_failed),
|
||||
# codex output has already streamed to the user by the time this
|
||||
# flush runs, so there is nothing left to withhold. We cannot
|
||||
# flip agent_persisted=False either: the gateway fallback write
|
||||
# would re-INSERT the already-flushed user turn (#860/#42039).
|
||||
# Surface the durability gap loudly instead of a silent debug.
|
||||
logger.warning(
|
||||
"codex app-server turn was delivered but could NOT be "
|
||||
"persisted to the session DB (session=%s) — this turn "
|
||||
"will be missing after restart/resume",
|
||||
getattr(agent, "session_id", None),
|
||||
)
|
||||
|
||||
|
||||
# Counter ticks for the agent-improvement loop.
|
||||
@@ -819,6 +942,12 @@ def run_codex_app_server_turn(
|
||||
"api_calls": api_calls,
|
||||
"completed": not turn.interrupted and turn.error is None,
|
||||
"partial": turn.interrupted or turn.error is not None,
|
||||
"interrupted": _user_interrupted,
|
||||
**(
|
||||
{"interrupt_message": _interrupt_message}
|
||||
if _interrupt_message
|
||||
else {}
|
||||
),
|
||||
"error": turn.error,
|
||||
# The codex app-server runtime IS an early-return path that bypasses
|
||||
# conversation_loop, but we flush the projected assistant/tool messages
|
||||
@@ -970,17 +1099,41 @@ def _consume_codex_event_stream(
|
||||
* ``interrupt_check()`` — returns True to break the loop early.
|
||||
"""
|
||||
collected_output_items: List[Any] = []
|
||||
# output_index of each collected_output_items entry, appended in lockstep
|
||||
# so settled pending calls can be merged back in stream order.
|
||||
collected_output_indexes: List[Any] = []
|
||||
collected_output_sequences: List[int] = []
|
||||
collected_text_deltas: List[str] = []
|
||||
has_tool_calls = False
|
||||
# Function calls announced via output_item.added but not yet confirmed by
|
||||
# output_item.done, keyed by item id. Some OpenAI-compatible backends omit
|
||||
# per-item done events on a successful completion (upstream evidence:
|
||||
# anomalyco/opencode#37159); these are settled from accumulated stream
|
||||
# state at the terminal event so the tool call executes instead of being
|
||||
# silently dropped.
|
||||
pending_function_calls: Dict[str, Dict[str, Any]] = {}
|
||||
# First-observed (sequence, output_index) per announced item id, so items
|
||||
# confirmed later via output_item.done keep their announced stream
|
||||
# position when merged with settled pending calls.
|
||||
announced_output_order: Dict[str, tuple] = {}
|
||||
first_delta_fired = False
|
||||
active_message_phase: str | None = None
|
||||
commentary_text_deltas: List[str] = []
|
||||
# Last reasoning summary_index seen. The Responses stream delimits summary
|
||||
# parts by this index and gives each part no separator of its own, so a
|
||||
# change of index is where the blank line belongs.
|
||||
active_summary_index: Any = None
|
||||
terminal_status: str = "completed"
|
||||
terminal_usage: Any = None
|
||||
terminal_response_id: str = None
|
||||
terminal_incomplete_details: Any = None
|
||||
terminal_error: Any = None
|
||||
saw_terminal = False
|
||||
# Settlement of pending calls requires an actually observed successful
|
||||
# terminal frame. ``terminal_status`` defaults to "completed", so it
|
||||
# cannot distinguish a real response.completed from EOF/interruption.
|
||||
saw_response_completed = False
|
||||
next_output_sequence = 0
|
||||
|
||||
for event in event_iter:
|
||||
if on_event is not None:
|
||||
@@ -1023,8 +1176,32 @@ def _consume_codex_event_stream(
|
||||
commentary_text_deltas = []
|
||||
else:
|
||||
active_message_phase = None
|
||||
# First-observed ordering metadata for EVERY announced item (not
|
||||
# just function calls): when this item later lands via
|
||||
# output_item.done, the done path must reuse the announced
|
||||
# sequence/index instead of allocating a fresh tail position, or
|
||||
# a mixed announced/pending stream without output_index values
|
||||
# reorders the calls (review P1 on PR #92767).
|
||||
item_id = str(_item_field(item, "id", ""))
|
||||
if item_id and item_id not in announced_output_order:
|
||||
announced_output_order[item_id] = (
|
||||
next_output_sequence,
|
||||
_event_field(event, "output_index", None),
|
||||
)
|
||||
next_output_sequence += 1
|
||||
if "function_call" in str(item_type):
|
||||
has_tool_calls = True
|
||||
if item_id:
|
||||
announced_sequence, announced_index = announced_output_order[item_id]
|
||||
# Seed from the announced item's own arguments when the
|
||||
# backend attaches them up front, and remember the stream
|
||||
# position so a settled call keeps its place in the output.
|
||||
pending_function_calls[item_id] = {
|
||||
"item": item,
|
||||
"arguments": str(_item_field(item, "arguments", "") or ""),
|
||||
"output_index": announced_index,
|
||||
"sequence": announced_sequence,
|
||||
}
|
||||
continue
|
||||
|
||||
if "output_text.delta" in event_type or event_type == "response.output_text.delta":
|
||||
@@ -1063,11 +1240,42 @@ def _consume_codex_event_stream(
|
||||
|
||||
if "function_call" in event_type:
|
||||
has_tool_calls = True
|
||||
# fall through — function_call items still get added on output_item.done
|
||||
# Accumulate streamed argument deltas for calls announced via
|
||||
# output_item.added, so a stream that completes without per-item
|
||||
# done events can still be settled from accumulated state.
|
||||
if "delta" in event_type:
|
||||
delta_args = _event_field(event, "delta", "")
|
||||
pending = pending_function_calls.get(str(_event_field(event, "item_id", "")))
|
||||
if pending is not None and delta_args:
|
||||
pending["arguments"] += delta_args
|
||||
continue
|
||||
if event_type.endswith("function_call_arguments.done"):
|
||||
done_args = _event_field(event, "arguments", None)
|
||||
pending = pending_function_calls.get(str(_event_field(event, "item_id", "")))
|
||||
if pending is not None and done_args is not None:
|
||||
# Per-item arguments.done is authoritative for the
|
||||
# accumulated string when the item itself never lands.
|
||||
# An explicit empty string (zero-argument call) counts as
|
||||
# authoritative; only a missing field leaves the streamed
|
||||
# deltas in place.
|
||||
pending["arguments"] = str(done_args)
|
||||
continue
|
||||
# other function_call frames fall through — function_call items still get added on output_item.done
|
||||
|
||||
if "reasoning" in event_type and "delta" in event_type:
|
||||
reasoning_text = _event_field(event, "delta", "")
|
||||
if reasoning_text and on_reasoning_delta is not None:
|
||||
# Summary parts stream one after another with no separator of
|
||||
# their own; summary_index is the boundary the wire gives us.
|
||||
summary_index = _event_field(event, "summary_index")
|
||||
if (
|
||||
summary_index is not None
|
||||
and active_summary_index is not None
|
||||
and summary_index != active_summary_index
|
||||
):
|
||||
reasoning_text = f"\n\n{reasoning_text}"
|
||||
if summary_index is not None:
|
||||
active_summary_index = summary_index
|
||||
try:
|
||||
on_reasoning_delta(reasoning_text)
|
||||
except Exception:
|
||||
@@ -1078,6 +1286,26 @@ def _consume_codex_event_stream(
|
||||
done_item = _event_field(event, "item")
|
||||
if done_item is not None:
|
||||
collected_output_items.append(done_item)
|
||||
# Reuse the first-observed position when this item was
|
||||
# announced earlier via output_item.added; a fresh tail
|
||||
# sequence is allocated only for genuinely unannounced items.
|
||||
# The .done event's own output_index wins when present, with
|
||||
# the announced index as its fallback.
|
||||
done_id = str(_item_field(done_item, "id", ""))
|
||||
announced_sequence, announced_index = announced_output_order.get(
|
||||
done_id, (None, None)
|
||||
)
|
||||
done_index = _event_field(event, "output_index", None)
|
||||
if done_index is None:
|
||||
done_index = announced_index
|
||||
if announced_sequence is None:
|
||||
announced_sequence = next_output_sequence
|
||||
next_output_sequence += 1
|
||||
collected_output_indexes.append(done_index)
|
||||
collected_output_sequences.append(announced_sequence)
|
||||
# Confirmed by the authoritative per-item done event; remove
|
||||
# from pending so it is not settled twice.
|
||||
pending_function_calls.pop(done_id, None)
|
||||
done_phase = _item_field(done_item, "phase", None)
|
||||
done_phase = done_phase.strip().lower() if isinstance(done_phase, str) else None
|
||||
if done_phase == "commentary" and on_commentary_message is not None:
|
||||
@@ -1126,6 +1354,7 @@ def _consume_codex_event_stream(
|
||||
if terminal_error is None and isinstance(resp_obj, dict):
|
||||
terminal_error = resp_obj.get("error")
|
||||
if event_type == "response.completed":
|
||||
saw_response_completed = True
|
||||
terminal_status = terminal_status or "completed"
|
||||
elif event_type == "response.incomplete":
|
||||
terminal_status = terminal_status or "incomplete"
|
||||
@@ -1150,6 +1379,56 @@ def _consume_codex_event_stream(
|
||||
else:
|
||||
output = []
|
||||
|
||||
# Settle function calls that were announced via output_item.added and
|
||||
# streamed argument deltas but never confirmed by output_item.done: some
|
||||
# OpenAI-compatible backends omit per-item done events on a successful
|
||||
# completion (anomalyco/opencode#37159). Done items stay authoritative;
|
||||
# this only fills the gap so the call executes instead of vanishing.
|
||||
if pending_function_calls and saw_response_completed:
|
||||
# Assemble settled calls and .done items in output_index order instead
|
||||
# of appending at the tail: a pending call that streamed before a later
|
||||
# .done item must keep its position, or dependent side effects invert.
|
||||
indexed = [
|
||||
(index, sequence, position, item)
|
||||
for position, (index, sequence, item) in enumerate(
|
||||
zip(
|
||||
collected_output_indexes,
|
||||
collected_output_sequences,
|
||||
collected_output_items,
|
||||
)
|
||||
)
|
||||
]
|
||||
for position, pending in enumerate(pending_function_calls.values(), start=len(indexed)):
|
||||
item = pending["item"]
|
||||
# Canonicalize empty/whitespace arguments so zero-delta calls stay
|
||||
# executable; malformed non-empty JSON passes through untouched and
|
||||
# stays rejected by downstream argument parsing.
|
||||
arguments = (pending["arguments"] or "").strip() or "{}"
|
||||
indexed.append((pending.get("output_index"), pending["sequence"], position, SimpleNamespace(
|
||||
type="function_call",
|
||||
id=_item_field(item, "id", None),
|
||||
call_id=_item_field(item, "call_id", None),
|
||||
name=_item_field(item, "name", None),
|
||||
arguments=arguments,
|
||||
status="completed",
|
||||
)))
|
||||
|
||||
# output_index is optional in compatible Responses streams. A partial
|
||||
# ordering (sorting indexed entries while interleaving unindexed ones)
|
||||
# is not well-defined and can produce contradictory comparisons. Keep
|
||||
# the observed wire order whenever any index is missing; use the
|
||||
# protocol ordering only when every entry provides an index.
|
||||
if all(entry[0] is not None for entry in indexed):
|
||||
try:
|
||||
indexed.sort(key=lambda entry: entry[0])
|
||||
except TypeError:
|
||||
# Preserve wire order if a backend sends non-comparable index
|
||||
# values instead of integers.
|
||||
pass
|
||||
else:
|
||||
indexed.sort(key=lambda entry: entry[1])
|
||||
output = [entry[3] for entry in indexed]
|
||||
|
||||
# If the stream ended without any terminal event AND produced no usable
|
||||
# content (no items, no text deltas), surface that as a RuntimeError so
|
||||
# callers can distinguish "stream truncated mid-flight / provider rejected
|
||||
@@ -1176,6 +1455,134 @@ def _consume_codex_event_stream(
|
||||
return final
|
||||
|
||||
|
||||
def _sanitize_consumer_codex_request(
|
||||
agent: Any,
|
||||
request: dict[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
"""Drop fields the ChatGPT OAuth Codex endpoint does not accept.
|
||||
|
||||
This guard intentionally lives at the final wire boundary, after Relay or
|
||||
other request middleware has had a chance to transform the request. The
|
||||
normal transport builder already omits ``prompt_cache_retention`` for this
|
||||
endpoint, but a late mutation must not be allowed to turn a valid tool
|
||||
follow-up into a non-retryable HTTP 400.
|
||||
|
||||
Explicit ``request_overrides`` are subject to the same endpoint contract:
|
||||
unsupported retention is dropped with a warning instead of being sent and
|
||||
rejected by the provider. The check covers both the top-level kwarg and a
|
||||
nested ``extra_body`` entry — the OpenAI SDK merges ``extra_body`` into
|
||||
the outgoing JSON body, so either shape reaches the endpoint.
|
||||
"""
|
||||
sanitized = dict(request)
|
||||
# Resolved defensively on purpose: run_codex_stream is also driven with
|
||||
# lightweight stand-in agents that carry only the attributes a given path
|
||||
# needs (see tests/agent/test_codex_request_transport_diagnostics.py), so a
|
||||
# bare agent._is_codex_backend() here would raise AttributeError on them.
|
||||
backend_predicate = getattr(agent, "_is_codex_backend", None)
|
||||
is_consumer_codex = (
|
||||
bool(backend_predicate()) if callable(backend_predicate) else False
|
||||
)
|
||||
if not is_consumer_codex:
|
||||
return sanitized
|
||||
dropped_from: list[str] = []
|
||||
if "prompt_cache_retention" in sanitized:
|
||||
sanitized.pop("prompt_cache_retention")
|
||||
dropped_from.append("top-level")
|
||||
# The OpenAI SDK merges ``extra_body`` into the outgoing JSON body, so a
|
||||
# nested ``extra_body.prompt_cache_retention`` reaches the endpoint just
|
||||
# like the top-level field would. Copy before editing — the caller's
|
||||
# mapping must not be mutated — and drop the mapping when it empties.
|
||||
extra_body = sanitized.get("extra_body")
|
||||
if isinstance(extra_body, dict) and "prompt_cache_retention" in extra_body:
|
||||
extra_body = dict(extra_body)
|
||||
extra_body.pop("prompt_cache_retention")
|
||||
if extra_body:
|
||||
sanitized["extra_body"] = extra_body
|
||||
else:
|
||||
sanitized.pop("extra_body")
|
||||
dropped_from.append("extra_body")
|
||||
if dropped_from:
|
||||
logger.warning(
|
||||
"Dropped unsupported prompt_cache_retention at consumer Codex "
|
||||
"wire boundary (model=%s, via %s).",
|
||||
sanitized.get("model", getattr(agent, "model", "unknown")),
|
||||
", ".join(dropped_from),
|
||||
)
|
||||
return sanitized
|
||||
|
||||
|
||||
# Bulk request fields that carry the conversation payload. Everything else in
|
||||
# the request is scalar configuration the SDK transform handles in microseconds.
|
||||
_SDK_TRANSFORM_BYPASS_FIELDS = ("input", "tools")
|
||||
|
||||
|
||||
def _is_plain_json_data(value: Any) -> bool:
|
||||
"""True when ``value`` is composed purely of JSON wire types.
|
||||
|
||||
The SDK's request transform exists to convert typed params (TypedDict
|
||||
key aliases, pydantic models, ``PropertyInfo`` formats) into wire
|
||||
format. Hermes assembles Codex payloads from JSON round-trips, so they
|
||||
are already wire format — but that is only provable when every node is
|
||||
a plain JSON type. Anything else must keep the typed SDK path.
|
||||
"""
|
||||
if value is None or isinstance(value, (str, int, float, bool)):
|
||||
return True
|
||||
if isinstance(value, dict):
|
||||
return all(
|
||||
isinstance(key, str) and _is_plain_json_data(item)
|
||||
for key, item in value.items()
|
||||
)
|
||||
if isinstance(value, list):
|
||||
return all(_is_plain_json_data(item) for item in value)
|
||||
return False
|
||||
|
||||
|
||||
def _bypass_sdk_request_transform(stream_kwargs: dict) -> dict:
|
||||
"""Route bulk payload fields around the SDK's ``maybe_transform`` (#93650).
|
||||
|
||||
``responses.create`` re-walks the entire request body against the
|
||||
``ResponseCreateParams`` union graph before any byte leaves the process.
|
||||
That walk runs with the GIL held, and #93650 documents it wedging for
|
||||
12+ hours on a ~1.4 MB conversation — starving every other thread,
|
||||
including the TTFB/stale watchdogs whose job is to rescue this exact
|
||||
call. Because the hang is client-side and pre-network, no socket kill
|
||||
can unblock it.
|
||||
|
||||
The SDK merges ``extra_body`` into the JSON body *after* the transform
|
||||
(``_base_client._build_request``), so moving the already-wire-format
|
||||
bulk fields there skips the walk entirely and produces a byte-identical
|
||||
request. Fields containing anything that is not plain JSON data (e.g.
|
||||
pydantic models, generators) stay on the typed path, which still needs
|
||||
the transform. Set HERMES_CODEX_SDK_TRANSFORM=1 to restore the pre-fix
|
||||
behavior.
|
||||
"""
|
||||
if os.environ.get("HERMES_CODEX_SDK_TRANSFORM", "").strip().lower() in {
|
||||
"1", "true", "yes", "on"
|
||||
}:
|
||||
return stream_kwargs
|
||||
|
||||
moved = {
|
||||
field: stream_kwargs[field]
|
||||
for field in _SDK_TRANSFORM_BYPASS_FIELDS
|
||||
if isinstance(stream_kwargs.get(field), (dict, list))
|
||||
and _is_plain_json_data(stream_kwargs[field])
|
||||
}
|
||||
if not moved:
|
||||
return stream_kwargs
|
||||
|
||||
bypassed = {
|
||||
key: value for key, value in stream_kwargs.items() if key not in moved
|
||||
}
|
||||
extra_body = bypassed.get("extra_body")
|
||||
merged = dict(extra_body) if isinstance(extra_body, dict) else {}
|
||||
for field, value in moved.items():
|
||||
# An explicit caller-provided extra_body entry keeps precedence,
|
||||
# matching what the SDK's post-transform merge would have done.
|
||||
merged.setdefault(field, value)
|
||||
bypassed["extra_body"] = merged
|
||||
return bypassed
|
||||
|
||||
|
||||
def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta=None):
|
||||
"""Execute one streaming Responses API request and return the final response.
|
||||
|
||||
@@ -1186,6 +1593,9 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
|
||||
the terminal event's ``output`` field.
|
||||
"""
|
||||
import httpx as _httpx
|
||||
from openai import APIConnectionError as _APIConnectionError
|
||||
|
||||
from agent import relay_llm
|
||||
|
||||
active_client = client or agent._ensure_primary_openai_client(reason="codex_stream_direct")
|
||||
max_stream_retries = 1
|
||||
@@ -1211,48 +1621,104 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
|
||||
if agent._interrupt_requested:
|
||||
raise InterruptedError("Agent interrupted before Codex stream retry")
|
||||
|
||||
stream_kwargs = dict(api_kwargs)
|
||||
stream_kwargs["stream"] = True
|
||||
intercepted_events = []
|
||||
writer_token = {"value": None}
|
||||
|
||||
try:
|
||||
event_stream = active_client.responses.create(**stream_kwargs)
|
||||
except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc:
|
||||
if attempt < max_stream_retries:
|
||||
logger.debug(
|
||||
"Codex Responses stream connect failed (attempt %s/%s); retrying. %s error=%s",
|
||||
attempt + 1, max_stream_retries + 1,
|
||||
agent._client_log_context(), exc,
|
||||
)
|
||||
continue
|
||||
raise
|
||||
def _open_codex_stream(next_api_kwargs: dict[str, Any]):
|
||||
stream_kwargs = _sanitize_consumer_codex_request(
|
||||
agent,
|
||||
next_api_kwargs,
|
||||
)
|
||||
stream_kwargs["stream"] = True
|
||||
stream_kwargs = _bypass_sdk_request_transform(stream_kwargs)
|
||||
return active_client.responses.create(**stream_kwargs)
|
||||
|
||||
# Claim the delta sink for THIS attempt (#65991) — parity with the
|
||||
# chat_completions/anthropic/bedrock paths. If a prior attempt's
|
||||
# stream is somehow still alive, this claim supersedes it so its
|
||||
# late deltas are fenced out of the turn; conversely, a newer
|
||||
# attempt supersedes us and the interrupt_check below stops our
|
||||
# consumption immediately.
|
||||
_writer_token = claim_stream_writer(agent)
|
||||
def _codex_stream_created(_raw_stream: Any) -> None:
|
||||
# Claim the delta sink for THIS physical attempt. A newer attempt
|
||||
# supersedes this token and fences late deltas out of the turn.
|
||||
writer_token["value"] = claim_stream_writer(agent)
|
||||
|
||||
def _interrupt_or_superseded(_tok=_writer_token) -> bool:
|
||||
if agent._interrupt_requested:
|
||||
return True
|
||||
if not stream_writer_is_current(agent, _tok):
|
||||
logger.warning(
|
||||
"Codex streaming attempt superseded by a newer stream; "
|
||||
"stopping consumption to preserve the single-writer "
|
||||
"invariant (model=%s).",
|
||||
api_kwargs.get("model", "unknown"),
|
||||
)
|
||||
def _accept_codex_chunk(_chunk: Any) -> bool:
|
||||
token = writer_token["value"]
|
||||
if token is None or stream_writer_is_current(agent, token):
|
||||
return True
|
||||
logger.warning(
|
||||
"Codex streaming attempt superseded by a newer stream; "
|
||||
"stopping consumption to preserve the single-writer "
|
||||
"invariant (model=%s).",
|
||||
api_kwargs.get("model", "unknown"),
|
||||
)
|
||||
return False
|
||||
|
||||
try:
|
||||
# Compatibility: some mocks/providers return a concrete response
|
||||
# instead of an iterable. Pass it straight through.
|
||||
if hasattr(event_stream, "output") and not hasattr(event_stream, "__iter__"):
|
||||
return event_stream
|
||||
def _finalize_codex_stream() -> Any:
|
||||
return _consume_codex_event_stream(
|
||||
list(intercepted_events),
|
||||
model=api_kwargs.get("model"),
|
||||
)
|
||||
|
||||
try:
|
||||
event_stream = relay_llm.stream(
|
||||
dict(api_kwargs),
|
||||
_open_codex_stream,
|
||||
session_id=str(getattr(agent, "session_id", "") or ""),
|
||||
name=str(getattr(agent, "provider", "") or "codex"),
|
||||
model_name=str(api_kwargs.get("model") or ""),
|
||||
finalizer=_finalize_codex_stream,
|
||||
on_stream_created=_codex_stream_created,
|
||||
on_chunk=intercepted_events.append,
|
||||
chunk_adapter=lambda chunk: chunk,
|
||||
accept_chunk=_accept_codex_chunk,
|
||||
completed_response_predicate=lambda response: bool(
|
||||
hasattr(response, "output") and not hasattr(response, "__iter__")
|
||||
),
|
||||
metadata={
|
||||
"api_mode": "codex_responses",
|
||||
"api_request_id": getattr(agent, "_current_api_request_id", None),
|
||||
"call_role": (
|
||||
"delegated"
|
||||
if getattr(agent, "is_subagent", False)
|
||||
else "fallback"
|
||||
if int(getattr(agent, "_fallback_index", 0) or 0) > 0
|
||||
else "primary"
|
||||
),
|
||||
"retry_count": attempt,
|
||||
},
|
||||
defer_logical_completion=True,
|
||||
)
|
||||
except (
|
||||
_httpx.RemoteProtocolError,
|
||||
_httpx.ReadTimeout,
|
||||
_httpx.ConnectError,
|
||||
ConnectionError,
|
||||
) as exc:
|
||||
if attempt < max_stream_retries:
|
||||
logger.debug(
|
||||
"Codex Responses stream connect failed (attempt %s/%s); "
|
||||
"retrying. %s error=%s",
|
||||
attempt + 1,
|
||||
max_stream_retries + 1,
|
||||
agent._client_log_context(),
|
||||
exc,
|
||||
)
|
||||
continue
|
||||
_log_codex_request_failure(
|
||||
agent,
|
||||
exc,
|
||||
stream_opened=writer_token["value"] is not None,
|
||||
)
|
||||
raise
|
||||
except _APIConnectionError as exc:
|
||||
_log_codex_request_failure(
|
||||
agent,
|
||||
exc,
|
||||
stream_opened=writer_token["value"] is not None,
|
||||
)
|
||||
raise
|
||||
|
||||
def _interrupt_or_superseded() -> bool:
|
||||
return bool(agent._interrupt_requested)
|
||||
|
||||
try:
|
||||
try:
|
||||
final = _consume_codex_event_stream(
|
||||
event_stream,
|
||||
@@ -1280,7 +1746,60 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
|
||||
agent._client_log_context(), exc,
|
||||
)
|
||||
continue
|
||||
_log_codex_request_failure(
|
||||
agent,
|
||||
exc,
|
||||
stream_opened=writer_token["value"] is not None,
|
||||
)
|
||||
raise
|
||||
except RuntimeError:
|
||||
if event_stream.final_response is not None:
|
||||
return event_stream.final_response
|
||||
raise
|
||||
except _APIConnectionError as exc:
|
||||
_log_codex_request_failure(
|
||||
agent,
|
||||
exc,
|
||||
stream_opened=writer_token["value"] is not None,
|
||||
)
|
||||
raise
|
||||
|
||||
# A terminal response has already been assembled at this point
|
||||
# (``final`` is built), so a transport error while draining the
|
||||
# rest of the iterator — done only to let Relay run its response
|
||||
# finalizer — must NOT discard it or trigger a new physical
|
||||
# request. Record it as a non-fatal finalization warning and
|
||||
# still return the already-completed, already-billed response.
|
||||
if not agent._interrupt_requested:
|
||||
try:
|
||||
for _ignored in event_stream:
|
||||
pass
|
||||
except (
|
||||
_httpx.RemoteProtocolError,
|
||||
_httpx.ReadTimeout,
|
||||
_httpx.ConnectError,
|
||||
ConnectionError,
|
||||
) as exc:
|
||||
logger.warning(
|
||||
"Codex Responses stream transport finalization failed "
|
||||
"after a terminal response was already received; "
|
||||
"returning the completed response instead of "
|
||||
"retrying. %s error=%s",
|
||||
agent._client_log_context(), exc,
|
||||
)
|
||||
except _APIConnectionError as exc:
|
||||
_log_codex_request_failure(
|
||||
agent,
|
||||
exc,
|
||||
stream_opened=writer_token["value"] is not None,
|
||||
)
|
||||
logger.warning(
|
||||
"Codex Responses stream transport finalization failed "
|
||||
"after a terminal response was already received; "
|
||||
"returning the completed response instead of "
|
||||
"retrying. %s error=%s",
|
||||
agent._client_log_context(), exc,
|
||||
)
|
||||
|
||||
if final.status in {"incomplete", "failed"}:
|
||||
logger.warning(
|
||||
@@ -1298,7 +1817,20 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
|
||||
try:
|
||||
close_fn()
|
||||
except Exception:
|
||||
pass
|
||||
# A failed close can leave this response's connection
|
||||
# checked out of the httpx pool while the caller's finally
|
||||
# reports a reuse-reason close (e.g. interrupt_check broke
|
||||
# the event loop with collected output) — caching the
|
||||
# client with the leaked connection. Poison the slot so
|
||||
# that close really closes the pool (owner-thread abort;
|
||||
# mirrors the chat-streaming interrupt-break handling).
|
||||
# ``client is None`` means the shared primary client,
|
||||
# which is never reuse-cached and must not have its
|
||||
# sockets force-shut here.
|
||||
if client is not None:
|
||||
agent._abort_request_openai_client(
|
||||
active_client, reason="codex_stream_close_failed"
|
||||
)
|
||||
|
||||
|
||||
def run_codex_create_stream_fallback(agent, api_kwargs: dict, client: Any = None):
|
||||
|
||||
@@ -337,9 +337,9 @@ def _coding_mode(config: Optional[dict[str, Any]]) -> str:
|
||||
"""Return the normalized ``agent.coding_context`` mode (auto/focus/on/off)."""
|
||||
if config is None:
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
from hermes_cli.config import load_config_readonly
|
||||
|
||||
config = load_config()
|
||||
config = load_config_readonly()
|
||||
except Exception:
|
||||
config = {}
|
||||
raw = ((config or {}).get("agent", {}) or {}).get("coding_context", "auto")
|
||||
@@ -520,30 +520,61 @@ class RuntimeMode:
|
||||
return None
|
||||
return [self.profile.toolset, *_enabled_mcp_servers(config)]
|
||||
|
||||
def system_blocks(self) -> list[str]:
|
||||
"""Stable system-prompt blocks for this posture (brief + workspace).
|
||||
def system_prompt_parts(
|
||||
self, valid_tool_names=None
|
||||
) -> tuple[list[str], list[str], list[str]]:
|
||||
"""Return prefix, workspace, and trailing posture blocks separately.
|
||||
|
||||
The operating brief carries a model-family edit-format nudge appended
|
||||
to it (one cached string, not a separate block) so the model is steered
|
||||
toward the `patch` mode it handles best — see ``_edit_format_line``.
|
||||
|
||||
``valid_tool_names`` (when provided) tailors the brief to the session's
|
||||
toolset: the ``todo`` tracking sentence is dropped when the todo tool
|
||||
isn't loaded (e.g. Blank Slate), so the brief never references a tool
|
||||
the model can't call. The toolset is fixed at session construction,
|
||||
so the rendered brief is deterministic per session — cache-safe.
|
||||
|
||||
The three lists preserve the historical flat prompt order: the brief,
|
||||
the live workspace snapshot, then configured operator instructions.
|
||||
Prompt assembly can therefore put a cache boundary before the snapshot
|
||||
without changing the persisted system-prompt bytes.
|
||||
"""
|
||||
if not self.is_coding:
|
||||
return []
|
||||
blocks: list[str] = []
|
||||
return [], [], []
|
||||
prefix: list[str] = []
|
||||
workspace_parts: list[str] = []
|
||||
trailing: list[str] = []
|
||||
if self.profile.guidance:
|
||||
brief = self.profile.guidance
|
||||
if valid_tool_names is not None and "todo" not in valid_tool_names:
|
||||
brief = brief.replace(
|
||||
"- Track multi-step work with `todo`. Reference code as "
|
||||
"`path:line` instead of pasting whole files.",
|
||||
"- Reference code as `path:line` instead of pasting "
|
||||
"whole files.",
|
||||
)
|
||||
edit_line = _edit_format_line(self.model)
|
||||
if edit_line:
|
||||
brief = f"{brief}\n{edit_line}"
|
||||
blocks.append(brief)
|
||||
prefix.append(brief)
|
||||
workspace = build_coding_workspace_block(self.cwd)
|
||||
if workspace:
|
||||
blocks.append(workspace)
|
||||
workspace_parts.append(workspace)
|
||||
# Operator instructions ride their own block so the brief (block 0) stays
|
||||
# byte-stable and cache-keyed independently of user config.
|
||||
if self.instructions:
|
||||
blocks.append(f"Operator instructions (from config):\n{self.instructions}")
|
||||
return blocks
|
||||
trailing.append(f"Operator instructions (from config):\n{self.instructions}")
|
||||
return prefix, workspace_parts, trailing
|
||||
|
||||
def system_blocks(self) -> list[str]:
|
||||
"""Return posture blocks in their historical display order.
|
||||
|
||||
``system_prompt_parts`` is the cache-aware API. This compatibility
|
||||
helper retains the public flat list for callers outside prompt assembly.
|
||||
"""
|
||||
prefix, workspace, trailing = self.system_prompt_parts()
|
||||
return [*prefix, *workspace, *trailing]
|
||||
|
||||
def compact_skill_categories(self) -> frozenset[str]:
|
||||
"""Skill categories to demote to names-only in the prompt's skill index.
|
||||
@@ -644,6 +675,20 @@ def coding_system_blocks(
|
||||
).system_blocks()
|
||||
|
||||
|
||||
def coding_system_prompt_parts(
|
||||
*,
|
||||
platform: Optional[str] = None,
|
||||
cwd: Optional[str | Path] = None,
|
||||
config: Optional[dict[str, Any]] = None,
|
||||
model: Optional[str] = None,
|
||||
valid_tool_names=None,
|
||||
) -> tuple[list[str], list[str], list[str]]:
|
||||
"""Return coding prefix, workspace snapshot, and trailing guidance."""
|
||||
return resolve_runtime_mode(
|
||||
platform=platform, cwd=cwd, config=config, model=model
|
||||
).system_prompt_parts(valid_tool_names=valid_tool_names)
|
||||
|
||||
|
||||
def coding_compact_skill_categories(
|
||||
*,
|
||||
platform: Optional[str] = None,
|
||||
|
||||
@@ -0,0 +1,190 @@
|
||||
"""Mint a provider API key by running a command (``key_cmd``).
|
||||
|
||||
Static API keys are the exception at enterprise gateways: SSO/OIDC brokers,
|
||||
cloud IAM, and internal auth proxies all issue SHORT-LIVED bearers instead.
|
||||
A key copied into ``.env`` (``key_env``) is stale within the hour, so every
|
||||
request after that 401s and the user has to restart the session.
|
||||
|
||||
``key_cmd`` names a command that PRINTS a token, so the credential is derived
|
||||
rather than stored::
|
||||
|
||||
providers:
|
||||
my-gateway:
|
||||
base_url: https://gateway.internal.example.com/v1
|
||||
api_mode: chat_completions
|
||||
key_cmd: my-auth-cli print-token --profile prod
|
||||
|
||||
This is the established pattern for agent tooling — Claude Code's
|
||||
``apiKeyHelper``, the ``gcloud auth print-access-token`` / ``aws ecr
|
||||
get-login-password`` idiom, and vendor helpers such as ``databricks auth
|
||||
token`` all expose exactly this contract. Hermes already accepts a callable
|
||||
API key on both wire clients (the Entra ID / Azure identity path) and invokes
|
||||
it per request, so nothing downstream changes: the token is simply always
|
||||
fresh. It is cached until shortly before expiry, so the command runs about
|
||||
once per token lifetime rather than once per request.
|
||||
|
||||
Output contract: print ONLY the token on stdout, either bare or as JSON with
|
||||
an ``access_token`` field (``expires_in`` is honoured when present) — the
|
||||
shape OAuth 2.0 token endpoints and the helpers above already emit.
|
||||
|
||||
Precedence: an explicit ``--api-key`` still wins (the one-off recovery escape
|
||||
hatch); otherwise ``key_cmd`` is preferred over a static ``api_key`` /
|
||||
``key_env`` on the same entry.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import subprocess
|
||||
import threading
|
||||
import time
|
||||
from typing import Callable, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Treat a cached token as spent slightly before its stated expiry, so a request
|
||||
# can't be signed with a token that dies in flight. 60s matches the leeway used
|
||||
# by comparable OAuth token caches.
|
||||
_TOKEN_REFRESH_LEEWAY_SECONDS = 60.0
|
||||
# A token helper reads a local credential cache and should answer in
|
||||
# milliseconds; anything approaching this budget is hung, not slow.
|
||||
_MINT_TIMEOUT_SECONDS = 15
|
||||
# When a helper advertises NO expiry, the token cannot be cached for the life
|
||||
# of the process: nothing in the request path re-mints on 401 (the SDK retries
|
||||
# 429/5xx only), so an expired no-TTL token would 401 every request until
|
||||
# restart. Re-mint on a bounded window instead — the helper answers from a
|
||||
# local credential cache in milliseconds, so a periodic re-run is cheap, and a
|
||||
# helper that wants a longer cache can simply advertise its real expiry.
|
||||
_NO_TTL_REFRESH_SECONDS = 900.0
|
||||
|
||||
|
||||
class CommandTokenError(RuntimeError):
|
||||
"""A ``key_cmd`` failed to produce a usable token."""
|
||||
|
||||
|
||||
def _mint(command: str, label: str) -> tuple[str, Optional[float]]:
|
||||
"""Run *command*, returning ``(token, ttl_seconds_or_None)``."""
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
command,
|
||||
shell=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=_MINT_TIMEOUT_SECONDS,
|
||||
)
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
raise CommandTokenError(
|
||||
f"key_cmd for provider {label!r} timed out after "
|
||||
f"{_MINT_TIMEOUT_SECONDS}s"
|
||||
) from exc
|
||||
except OSError as exc:
|
||||
raise CommandTokenError(
|
||||
f"key_cmd for provider {label!r} could not be executed: {exc}"
|
||||
) from exc
|
||||
|
||||
if completed.returncode != 0:
|
||||
# NEVER include stdout/stderr: a partially-successful auth helper can
|
||||
# print a token or refresh secret there. The command STRING is also
|
||||
# withheld — a key_cmd can legitimately embed a secret
|
||||
# (`print-token --client-secret=…`), so echoing it back would leak the
|
||||
# very credential this module exists to protect. Name the provider so
|
||||
# the user knows which config entry to run by hand.
|
||||
raise CommandTokenError(
|
||||
f"key_cmd for provider {label!r} exited {completed.returncode}. "
|
||||
f"Run that provider's key_cmd manually to see why "
|
||||
f"(e.g. `databricks auth login` if its OAuth session expired)."
|
||||
)
|
||||
|
||||
stdout = completed.stdout or ""
|
||||
if not stdout.strip():
|
||||
raise CommandTokenError(f"key_cmd for provider {label!r} produced no output")
|
||||
|
||||
# JSON payload — the shape `databricks auth token --output json` prints.
|
||||
# Token extraction mirrors databricks/ucode's get_databricks_token:
|
||||
# json.loads(result.stdout or "{}").get("access_token", "")
|
||||
if stdout.lstrip().startswith("{"):
|
||||
try:
|
||||
payload = json.loads(stdout)
|
||||
except json.JSONDecodeError:
|
||||
payload = None
|
||||
if isinstance(payload, dict):
|
||||
token = str(payload.get("access_token") or "").strip()
|
||||
if not token:
|
||||
raise CommandTokenError(
|
||||
f"key_cmd for provider {label!r} returned JSON without an "
|
||||
"'access_token' field"
|
||||
)
|
||||
ttl = payload.get("expires_in")
|
||||
if isinstance(ttl, (int, float)) and ttl > 0:
|
||||
return token, float(ttl)
|
||||
# A relative lifetime is the OAuth 2.0 field, but CLI token helpers
|
||||
# commonly print an absolute ISO 8601 deadline instead. Treating
|
||||
# that as "no TTL advertised" caches the token for the life of the
|
||||
# process, so every request 401s once the deadline passes.
|
||||
# Imported lazily: hermes_cli.auth imports from agent.* at module
|
||||
# level, so a top-level import here would risk a cycle.
|
||||
from hermes_cli.auth import _parse_iso_timestamp
|
||||
|
||||
for field in ("expiry", "expiresOn"):
|
||||
deadline = _parse_iso_timestamp(payload.get(field))
|
||||
if deadline is not None:
|
||||
remaining = deadline - time.time()
|
||||
if remaining > 0:
|
||||
return token, remaining
|
||||
return token, None
|
||||
|
||||
# Bare token. The contract every comparable helper documents is "stdout
|
||||
# carries the token and nothing else" — extra output would be consumed as
|
||||
# part of the credential. Strip surrounding whitespace and take the rest
|
||||
# verbatim; do NOT silently keep one line of several, which converts a
|
||||
# misconfigured helper (banner, warning, two tokens) into a corrupt-key 401
|
||||
# that is far harder to diagnose than an explicit refusal.
|
||||
token = stdout.strip()
|
||||
if "\n" in token:
|
||||
raise CommandTokenError(
|
||||
f"key_cmd for provider {label!r} printed multiple lines; it must "
|
||||
"print only the token (or JSON with an 'access_token' field)"
|
||||
)
|
||||
return token, None
|
||||
|
||||
|
||||
class CommandTokenSource:
|
||||
"""Callable returning a bearer token, cached until shortly before expiry."""
|
||||
|
||||
def __init__(self, command: str, label: str = "custom") -> None:
|
||||
self._command = command
|
||||
self._label = label or "custom"
|
||||
self._lock = threading.Lock()
|
||||
self._token = ""
|
||||
self._expires_at: float = 0.0
|
||||
|
||||
def __call__(self) -> str:
|
||||
with self._lock:
|
||||
if self._token and time.monotonic() < self._expires_at:
|
||||
return self._token
|
||||
token, ttl = _mint(self._command, self._label)
|
||||
self._token = token
|
||||
self._expires_at = (
|
||||
time.monotonic() + max(ttl - _TOKEN_REFRESH_LEEWAY_SECONDS, 5.0)
|
||||
if ttl
|
||||
# No advertised TTL: bounded cache (see _NO_TTL_REFRESH_SECONDS)
|
||||
# — there is no 401-driven re-mint hook to fall back on.
|
||||
else time.monotonic() + _NO_TTL_REFRESH_SECONDS
|
||||
)
|
||||
logger.debug(
|
||||
"Minted key_cmd token for provider %s (ttl=%s)",
|
||||
self._label, f"{int(ttl)}s" if ttl else "unknown",
|
||||
)
|
||||
return token
|
||||
|
||||
|
||||
def build_command_token_provider(
|
||||
key_cmd: str,
|
||||
provider_label: str = "custom",
|
||||
) -> Optional[Callable[[], str]]:
|
||||
"""A per-request token provider for *key_cmd*, or ``None`` when unset."""
|
||||
command = str(key_cmd or "").strip()
|
||||
if not command:
|
||||
return None
|
||||
return CommandTokenSource(command, provider_label)
|
||||
@@ -0,0 +1,47 @@
|
||||
"""Client-facing projection helpers for model-only compaction carriers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from agent.context_compressor import (
|
||||
ContextCompressor,
|
||||
is_compaction_summary_message,
|
||||
)
|
||||
|
||||
|
||||
_COMPACTION_INTERNAL_FIELDS = (
|
||||
"tool_calls",
|
||||
"finish_reason",
|
||||
"reasoning",
|
||||
"reasoning_content",
|
||||
"reasoning_details",
|
||||
"codex_reasoning_items",
|
||||
"codex_message_items",
|
||||
)
|
||||
|
||||
|
||||
def project_compaction_message_for_display(
|
||||
message: Dict[str, Any],
|
||||
) -> Optional[Dict[str, Any]]:
|
||||
"""Return authentic transcript content, or ``None`` for a pure handoff.
|
||||
|
||||
Model-facing recovery history retains the complete carrier. Display
|
||||
projections instead remove the handoff, inherited tool state, and internal
|
||||
reasoning while preserving any real prior-tail content or live user ask
|
||||
embedded in the carrier.
|
||||
"""
|
||||
if not isinstance(message, dict):
|
||||
return None
|
||||
if not is_compaction_summary_message(message):
|
||||
return message.copy()
|
||||
|
||||
projected = ContextCompressor._strip_context_summary_handoff_message(message)
|
||||
if projected is None:
|
||||
return None
|
||||
|
||||
projected = projected.copy()
|
||||
for key in _COMPACTION_INTERNAL_FIELDS:
|
||||
projected.pop(key, None)
|
||||
projected.pop("display_kind", None)
|
||||
return projected
|
||||
@@ -154,3 +154,207 @@ def compute_session_context_breakdown(
|
||||
"estimated_total": estimated_total,
|
||||
"model": getattr(agent, "model", "") or "",
|
||||
}
|
||||
|
||||
|
||||
# ── /context rendering (CLI + gateway) ──────────────────────────────────────
|
||||
#
|
||||
# Pure text renderers over the payload above. The CLI shows a glyph block-grid
|
||||
# plus a category table; the gateway uses the same table without the grid
|
||||
# (proportional monospace is not guaranteed on messaging platforms).
|
||||
|
||||
_CATEGORY_GLYPHS = {
|
||||
"system_prompt": "■",
|
||||
"tool_definitions": "▣",
|
||||
"rules": "▩",
|
||||
"skills": "▤",
|
||||
"mcp": "▥",
|
||||
"subagent_definitions": "▦",
|
||||
"memory": "▧",
|
||||
"conversation": "▨",
|
||||
}
|
||||
_FREE_GLYPH = "·"
|
||||
_GRID_COLUMNS = 20
|
||||
_GRID_ROWS = 5 # 100 cells → 1 cell per percent of the context window
|
||||
|
||||
# Human-readable tables cap the expanded listings; nothing is dropped from
|
||||
# the underlying data.
|
||||
_DETAILS_TABLE_LIMIT = 15
|
||||
|
||||
|
||||
def _bytes_to_tokens(size: Optional[int]) -> Optional[int]:
|
||||
if size is None:
|
||||
return None
|
||||
return (int(size) + 3) // 4
|
||||
|
||||
|
||||
def compute_context_details(agent: Any) -> Dict[str, Any]:
|
||||
"""Expanded per-skill / per-toolset cost listing for ``/context all``.
|
||||
|
||||
Reuses the ``hermes prompt-size`` attribution mechanism (PR #66656):
|
||||
per-skill index-line bytes parsed from the live ``<available_skills>``
|
||||
block, and per-toolset schema bytes attributed via the tool registry's
|
||||
canonical tool→toolset map. Byte figures are converted to the same
|
||||
chars/4 token heuristic the categories above use.
|
||||
"""
|
||||
from hermes_cli.prompt_size import (
|
||||
_compute_skills_breakdown,
|
||||
_compute_toolsets_breakdown,
|
||||
)
|
||||
from agent.system_prompt import build_system_prompt_parts
|
||||
|
||||
parts = build_system_prompt_parts(agent)
|
||||
stable = parts.get("stable", "") or ""
|
||||
skills_match = _SKILLS_BLOCK_RE.search(stable)
|
||||
skills_block = skills_match.group(0) if skills_match else ""
|
||||
|
||||
skills: List[Dict[str, Any]] = []
|
||||
if skills_block:
|
||||
for entry in _compute_skills_breakdown(skills_block):
|
||||
skills.append({
|
||||
"name": entry.get("name", ""),
|
||||
"index_tokens": _bytes_to_tokens(entry.get("index_line_bytes")) or 0,
|
||||
"skill_md_tokens": _bytes_to_tokens(entry.get("skill_md_bytes")),
|
||||
})
|
||||
|
||||
toolsets: List[Dict[str, Any]] = []
|
||||
tools = list(getattr(agent, "tools", None) or [])
|
||||
if tools:
|
||||
for group in _compute_toolsets_breakdown(tools):
|
||||
toolsets.append({
|
||||
"toolset": group.get("toolset", ""),
|
||||
"tool_count": int(group.get("tool_count", 0) or 0),
|
||||
"schema_tokens": _bytes_to_tokens(group.get("json_bytes")) or 0,
|
||||
})
|
||||
|
||||
return {"skills": skills, "toolsets": toolsets}
|
||||
|
||||
|
||||
def render_context_grid(payload: Dict[str, Any]) -> List[str]:
|
||||
"""Render the payload as a Claude Code-style glyph block grid.
|
||||
|
||||
100 cells (5×20), each one percent of the model context window. Categories
|
||||
fill in declaration order; the remainder renders as free space.
|
||||
"""
|
||||
context_max = int(payload.get("context_max") or 0)
|
||||
categories = payload.get("categories") or []
|
||||
total_cells = _GRID_COLUMNS * _GRID_ROWS
|
||||
|
||||
cells: List[str] = []
|
||||
if context_max > 0:
|
||||
for cat in categories:
|
||||
tokens = int(cat.get("tokens") or 0)
|
||||
n = round(tokens / context_max * total_cells)
|
||||
if tokens > 0 and n == 0:
|
||||
n = 1 # never render a nonzero category as invisible
|
||||
glyph = _CATEGORY_GLYPHS.get(str(cat.get("id") or ""), "▪")
|
||||
cells.extend([glyph] * n)
|
||||
cells = cells[:total_cells]
|
||||
cells.extend([_FREE_GLYPH] * (total_cells - len(cells)))
|
||||
|
||||
return [
|
||||
" ".join(cells[row * _GRID_COLUMNS:(row + 1) * _GRID_COLUMNS])
|
||||
for row in range(_GRID_ROWS)
|
||||
]
|
||||
|
||||
|
||||
def render_context_category_lines(payload: Dict[str, Any]) -> List[str]:
|
||||
"""Render the 'Estimated usage by category' table as plain-text lines."""
|
||||
categories = payload.get("categories") or []
|
||||
context_max = int(payload.get("context_max") or 0)
|
||||
estimated_total = int(payload.get("estimated_total") or 0)
|
||||
denom = context_max or estimated_total
|
||||
|
||||
lines = ["Estimated usage by category"]
|
||||
if not categories:
|
||||
lines.append(" (no data yet — send a message first)")
|
||||
return lines
|
||||
|
||||
width = max(len(str(cat.get("label") or "")) for cat in categories)
|
||||
width = max(width, len("Free space"))
|
||||
for cat in categories:
|
||||
tokens = int(cat.get("tokens") or 0)
|
||||
glyph = _CATEGORY_GLYPHS.get(str(cat.get("id") or ""), "▪")
|
||||
pct = tokens / denom * 100 if denom else 0.0
|
||||
label = str(cat.get("label") or cat.get("id") or "")
|
||||
lines.append(f"{glyph} {label:<{width}} {tokens:>9,} tokens {pct:>5.1f}%")
|
||||
if context_max > 0:
|
||||
free = max(0, context_max - estimated_total)
|
||||
pct = free / context_max * 100
|
||||
lines.append(f"{_FREE_GLYPH} {'Free space':<{width}} {free:>9,} tokens {pct:>5.1f}%")
|
||||
return lines
|
||||
|
||||
|
||||
def render_context_details_lines(details: Dict[str, Any]) -> List[str]:
|
||||
"""Render the expanded ``/context all`` per-skill / per-toolset tables."""
|
||||
lines: List[str] = []
|
||||
|
||||
toolsets = details.get("toolsets") or []
|
||||
if toolsets:
|
||||
lines.append("Toolsets by schema cost (largest first)")
|
||||
for group in toolsets[:_DETAILS_TABLE_LIMIT]:
|
||||
lines.append(
|
||||
f" {group['toolset']:<24} {group['tool_count']:>3} tools"
|
||||
f" {group['schema_tokens']:>8,} tokens"
|
||||
)
|
||||
remaining = len(toolsets) - _DETAILS_TABLE_LIMIT
|
||||
if remaining > 0:
|
||||
lines.append(f" … and {remaining} more")
|
||||
|
||||
skills = details.get("skills") or []
|
||||
if skills:
|
||||
if lines:
|
||||
lines.append("")
|
||||
lines.append("Skills by cost (index = always-on; SKILL.md = cost when loaded)")
|
||||
for entry in skills[:_DETAILS_TABLE_LIMIT]:
|
||||
name = str(entry.get("name") or "")
|
||||
if len(name) > 28:
|
||||
name = name[:27] + "…"
|
||||
md = entry.get("skill_md_tokens")
|
||||
md_str = f"{md:>8,}" if md is not None else f"{'n/a':>8}"
|
||||
lines.append(
|
||||
f" {name:<28} index {entry['index_tokens']:>6,}"
|
||||
f" SKILL.md {md_str} tokens"
|
||||
)
|
||||
remaining = len(skills) - _DETAILS_TABLE_LIMIT
|
||||
if remaining > 0:
|
||||
lines.append(f" … and {remaining} more")
|
||||
|
||||
return lines
|
||||
|
||||
|
||||
def render_context_breakdown_lines(
|
||||
payload: Dict[str, Any],
|
||||
*,
|
||||
details: Optional[Dict[str, Any]] = None,
|
||||
grid: bool = True,
|
||||
) -> List[str]:
|
||||
"""Render the full /context view as plain-text lines.
|
||||
|
||||
``grid=True`` (CLI) prepends the glyph block grid; the gateway passes
|
||||
``grid=False`` and keeps its own gauge. ``details`` (from
|
||||
:func:`compute_context_details`) appends the expanded listings.
|
||||
"""
|
||||
lines: List[str] = []
|
||||
if grid:
|
||||
lines.extend(render_context_grid(payload))
|
||||
lines.append("")
|
||||
lines.extend(render_context_category_lines(payload))
|
||||
|
||||
context_max = int(payload.get("context_max") or 0)
|
||||
context_used = int(payload.get("context_used") or 0)
|
||||
if context_max > 0:
|
||||
pct = int(payload.get("context_percent") or 0)
|
||||
lines.append("")
|
||||
lines.append(
|
||||
f"Context window: {context_used:,} / {context_max:,} tokens ({pct}%)"
|
||||
)
|
||||
|
||||
if details is not None:
|
||||
detail_lines = render_context_details_lines(details)
|
||||
if detail_lines:
|
||||
lines.append("")
|
||||
lines.extend(detail_lines)
|
||||
else:
|
||||
lines.append("")
|
||||
lines.append("Use /context all for per-skill and per-toolset costs.")
|
||||
return lines
|
||||
|
||||
@@ -53,6 +53,39 @@ def sanitize_memory_context(memory_context: str) -> str:
|
||||
)
|
||||
|
||||
|
||||
def automatic_compaction_status_message(
|
||||
engine: Any,
|
||||
*,
|
||||
phase: str,
|
||||
default_message: str,
|
||||
**context: Any,
|
||||
) -> str | None:
|
||||
"""Resolve host-visible status for an automatic compaction event.
|
||||
|
||||
Engines can suppress routine automatic status with
|
||||
``emit_automatic_compaction_status = False`` or customize it by defining
|
||||
``get_automatic_compaction_status_message(...)``. Empty strings and
|
||||
``None`` mean "do not emit a lifecycle status".
|
||||
"""
|
||||
if not getattr(engine, "emit_automatic_compaction_status", True):
|
||||
return None
|
||||
|
||||
formatter = getattr(engine, "get_automatic_compaction_status_message", None)
|
||||
if callable(formatter):
|
||||
message = formatter(
|
||||
phase=phase,
|
||||
default_message=default_message,
|
||||
**context,
|
||||
)
|
||||
else:
|
||||
message = default_message
|
||||
|
||||
if message is None:
|
||||
return None
|
||||
message = str(message).strip()
|
||||
return message or None
|
||||
|
||||
|
||||
class ContextEngine(ABC):
|
||||
"""Base class all context engines must implement."""
|
||||
|
||||
@@ -89,6 +122,12 @@ class ContextEngine(ABC):
|
||||
protect_first_n: int = 3
|
||||
protect_last_n: int = 6
|
||||
|
||||
# User-visible lifecycle status for automatic host-triggered compaction.
|
||||
# Alternative engines that treat compaction as routine background
|
||||
# maintenance can set this false to keep successful automatic passes silent;
|
||||
# warnings, errors, and explicit manual commands should still surface.
|
||||
emit_automatic_compaction_status: bool = True
|
||||
|
||||
# -- Core interface ----------------------------------------------------
|
||||
|
||||
@abstractmethod
|
||||
@@ -107,6 +146,19 @@ class ContextEngine(ABC):
|
||||
def should_compress(self, prompt_tokens: int = None) -> bool:
|
||||
"""Return True if compaction should fire this turn."""
|
||||
|
||||
def should_compress_info(self, prompt_tokens: int = None) -> "tuple[bool, str | None]":
|
||||
"""Return ``(should_compress, reason)``.
|
||||
|
||||
The base implementation is backward-compatible: engines that only
|
||||
implement ``should_compress`` get ``(should_compress(prompt_tokens),
|
||||
None)``. Concrete engines with richer block reasons (e.g. a
|
||||
summary-LLM cooldown or an anti-thrashing guard) override this to
|
||||
surface a human-readable reason so callers can warn the user instead
|
||||
of silently skipping compression. Added for the silent-overflow
|
||||
warning fix (#62625) so plugin engines don't raise AttributeError.
|
||||
"""
|
||||
return self.should_compress(prompt_tokens), None
|
||||
|
||||
@abstractmethod
|
||||
def compress(
|
||||
self,
|
||||
@@ -137,6 +189,144 @@ class ContextEngine(ABC):
|
||||
host filters unsupported optional arguments by signature.
|
||||
"""
|
||||
|
||||
# -- Optional: proactive tool-result prune -----------------------------
|
||||
|
||||
def prune_tool_results_only(
|
||||
self,
|
||||
messages: List[Dict[str, Any]],
|
||||
current_tokens: int | None = None,
|
||||
) -> tuple[List[Dict[str, Any]], int]:
|
||||
"""Deterministically trim old tool-result payloads without an LLM call.
|
||||
|
||||
Runs on a low, cost-oriented trigger independent of ``should_compress``
|
||||
so large-window engines can reclaim re-sent tool output long before full
|
||||
compaction would fire. Returns ``(messages, n_pruned)``.
|
||||
|
||||
Default is a safe no-op: the list is returned unchanged with ``0``
|
||||
pruned. Engines that don't implement a cheap prune — and any engine that
|
||||
predates this hook — inherit this default, so the agent loop's
|
||||
post-tool-call prune path never raises ``AttributeError`` on them. The
|
||||
built-in ContextCompressor overrides this with the real implementation.
|
||||
"""
|
||||
return messages, 0
|
||||
|
||||
# -- Optional: per-turn context selection (distinct from compression) --
|
||||
|
||||
def select_context(
|
||||
self,
|
||||
request_messages: List[Dict[str, Any]],
|
||||
*,
|
||||
conversation_messages: List[Dict[str, Any]] = None,
|
||||
incoming_message: Dict[str, Any] = None,
|
||||
budget_tokens: int = 0,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Optionally choose/replace the context for THIS request, pre-generation.
|
||||
|
||||
Called every turn after the request message list is assembled and
|
||||
before it is dispatched to the provider — independent of
|
||||
``should_compress()``. This lets an engine *select* which context
|
||||
enters the prompt (retrieval, topic routing, role/branch switching)
|
||||
rather than *shrink* context that is already there. The two verbs are
|
||||
orthogonal:
|
||||
|
||||
- ``compress()`` : context is too long -> make it shorter.
|
||||
- ``select_context()``: this turn belongs to a different context
|
||||
-> use that one instead.
|
||||
|
||||
Without this hook, engines that need per-turn access to the message
|
||||
list have to force ``should_compress()`` to return ``True`` so that
|
||||
``compress()`` is invoked every turn purely as a callback — which
|
||||
conflates selection with compression and degrades behaviour when the
|
||||
engine's backend is unavailable. ``select_context()`` removes the need
|
||||
for that workaround.
|
||||
|
||||
The returned list is request-only: it replaces the messages sent to
|
||||
the provider for this single call and MUST NOT be treated as persisted
|
||||
transcript state. The conversation history in the session DB is left
|
||||
untouched, so nothing leaks across turns. Return ``None`` to leave the
|
||||
request unchanged.
|
||||
|
||||
Unlike the ``pre_llm_call`` plugin hook (which appends to the user
|
||||
message and intentionally never rewrites the list, to preserve the
|
||||
cache prefix), ``select_context()`` may *replace* the message list.
|
||||
|
||||
Ordering / cache contract: the host runs this hook **before** prompt
|
||||
cache-control and **before** every request sanitizer (orphaned-tool
|
||||
cleanup, thinking-only/role normalization, whitespace/JSON
|
||||
normalization). So (a) whatever the hook returns still passes through
|
||||
the same validation as any request — a malformed replacement cannot
|
||||
reach the provider — and (b) prompt-cache stability (an AGENTS.md
|
||||
invariant) is preserved: the default no-op leaves the request
|
||||
byte-identical, so cache behaviour is unchanged for the built-in
|
||||
compressor and any non-implementing engine. An engine that *does*
|
||||
replace the list changes its own cache prefix by definition; that is
|
||||
the engine's concern, and cache-control breakpoints are re-derived on
|
||||
the selected list. The hook is evaluated per provider request (so it
|
||||
re-runs on retries within a turn), consistent with "select the context
|
||||
for THIS request".
|
||||
|
||||
Args:
|
||||
request_messages: The assembled request message list (system
|
||||
prompt + history + any ephemeral prefill), in OpenAI format.
|
||||
conversation_messages: The unmodified persisted conversation
|
||||
history, for reference only (do not mutate).
|
||||
incoming_message: The current turn's user message, if available.
|
||||
budget_tokens: The active model's context length, or 0 if unknown.
|
||||
|
||||
Default returns ``None`` (no-op) — zero impact on the built-in
|
||||
compressor or any existing engine.
|
||||
"""
|
||||
return None
|
||||
|
||||
def on_turn_complete(
|
||||
self,
|
||||
messages: List[Dict[str, Any]],
|
||||
usage: Dict[str, Any] = None,
|
||||
**kwargs: Any,
|
||||
) -> None:
|
||||
"""Observe a finished user turn (post-turn ingestion / observation).
|
||||
|
||||
Called from the standard turn-finalization path once the assistant/tool
|
||||
loop completes, with the finalized in-memory transcript snapshot. This
|
||||
is the complement to ``select_context()``: selection happens *before*
|
||||
the request, while observation happens *after* the turn. It lets an
|
||||
engine ingest, index, summarize, or update routing / topic / session
|
||||
state from what actually happened — so the next ``select_context()``
|
||||
can act on it.
|
||||
|
||||
Coverage: this fires from the normal finalization seam. Some abnormal
|
||||
early-return paths in the loop (e.g. a content-policy block or a
|
||||
provider terminal failure) persist and return without routing through
|
||||
finalization, and therefore do not currently emit this hook. Treat it
|
||||
as a best-effort post-turn observation for completed turns, not a
|
||||
guaranteed callback for every possible early exit; unifying all
|
||||
terminal paths behind one finalization seam is a separate follow-up.
|
||||
|
||||
Together the two hooks remove the need to abuse ``should_compress()`` /
|
||||
``compress()`` as a generic per-turn callback just to observe history,
|
||||
and they cover the case where a turn finishes and there may be no next
|
||||
request from which to infer the previous turn.
|
||||
|
||||
``messages`` is a shallow copy and should be treated as read-only:
|
||||
return values are ignored and this hook must not rely on transcript
|
||||
mutation for persistence. ``kwargs`` may include ``turn_id``,
|
||||
``task_id``, ``api_call_count``, ``interrupted``, ``failed``, and
|
||||
``turn_exit_reason``.
|
||||
|
||||
``usage`` carries the completed turn's canonical token usage (the same
|
||||
dict shape passed to ``update_from_response`` — ``prompt_tokens`` /
|
||||
``completion_tokens`` / ``total_tokens`` plus the canonical
|
||||
``input_tokens`` / ``output_tokens`` / ``cache_read_tokens`` /
|
||||
``cache_write_tokens`` / ``reasoning_tokens`` buckets) so an engine can
|
||||
weigh how large/expensive the selected context actually was when
|
||||
deciding the next ``select_context()``. It is ``None`` on finalized
|
||||
turns that never reached a provider response (e.g. interrupt); engines
|
||||
must treat it as optional.
|
||||
|
||||
Default is a no-op.
|
||||
"""
|
||||
return None
|
||||
|
||||
# -- Optional: pre-flight check ----------------------------------------
|
||||
|
||||
def should_compress_preflight(self, messages: List[Dict[str, Any]]) -> bool:
|
||||
@@ -156,6 +346,27 @@ class ContextEngine(ABC):
|
||||
"""
|
||||
return False
|
||||
|
||||
def get_automatic_compaction_status_message(
|
||||
self,
|
||||
*,
|
||||
phase: str,
|
||||
default_message: str,
|
||||
**context: Any,
|
||||
) -> str | None:
|
||||
"""Return user-visible status for automatic host-triggered compaction.
|
||||
|
||||
Return ``None`` to suppress successful automatic lifecycle status for
|
||||
this compaction event. ``phase`` identifies the host call site (for
|
||||
example ``"preflight"`` or ``"compress"``). ``context`` contains
|
||||
best-effort fields such as ``approx_tokens`` and ``threshold_tokens``.
|
||||
|
||||
This hook does not control warning/error messages or explicit manual
|
||||
commands such as ``/compress``.
|
||||
"""
|
||||
if not self.emit_automatic_compaction_status:
|
||||
return None
|
||||
return default_message
|
||||
|
||||
# -- Optional: manual /compress preflight ------------------------------
|
||||
|
||||
def has_content_to_compress(self, messages: List[Dict[str, Any]]) -> bool:
|
||||
|
||||
@@ -13,12 +13,82 @@ from typing import Awaitable, Callable
|
||||
|
||||
from agent.model_metadata import estimate_tokens_rough
|
||||
from hermes_cli._subprocess_compat import IS_WINDOWS, windows_hide_flags
|
||||
from hermes_cli.sizefmt import format_bytes
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Plugin context-reference provider API (Issue #26193)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
BUILTIN_PREFIXES = frozenset({"diff", "staged", "file", "folder", "git", "url"})
|
||||
|
||||
_context_reference_providers: dict[str, "ContextReferenceProvider"] = {}
|
||||
|
||||
|
||||
class ContextCompletionItem:
|
||||
"""A single autocomplete result from a context reference provider."""
|
||||
|
||||
__slots__ = ("text", "display", "meta")
|
||||
|
||||
def __init__(self, text: str, display: str = "", meta: str = "") -> None:
|
||||
self.text = text
|
||||
self.display = display or text
|
||||
self.meta = meta
|
||||
|
||||
|
||||
class ContextReferenceProvider(ABC):
|
||||
"""Base class for plugin-registered @-prefix context reference providers.
|
||||
|
||||
Plugins subclass this and register via
|
||||
``PluginContext.register_context_reference()``.
|
||||
"""
|
||||
|
||||
prefix: str = "" # e.g. "issue", "channel", "doc"
|
||||
description: str = "" # shown in autocomplete meta column
|
||||
|
||||
@abstractmethod
|
||||
async def autocomplete(self, query: str, *, limit: int = 10) -> list[ContextCompletionItem]:
|
||||
"""Return autocomplete items for the given query string."""
|
||||
...
|
||||
|
||||
@abstractmethod
|
||||
async def expand(self, target: str) -> str | None:
|
||||
"""Expand *target* to prompt content. Return ``None`` to skip."""
|
||||
...
|
||||
|
||||
|
||||
def register_context_reference_provider(provider: ContextReferenceProvider) -> None:
|
||||
"""Register a plugin context reference provider."""
|
||||
if not isinstance(provider, ContextReferenceProvider):
|
||||
raise TypeError("provider must be a ContextReferenceProvider instance")
|
||||
prefix = provider.prefix.lower().strip()
|
||||
if not prefix:
|
||||
raise ValueError("prefix must be a non-empty string")
|
||||
if prefix in BUILTIN_PREFIXES:
|
||||
raise ValueError(f"prefix '{prefix}' is reserved for built-in references")
|
||||
if prefix in _context_reference_providers:
|
||||
raise ValueError(f"prefix '{prefix}' is already registered")
|
||||
_context_reference_providers[prefix] = provider
|
||||
|
||||
|
||||
def get_context_reference_providers() -> dict[str, ContextReferenceProvider]:
|
||||
"""Return a snapshot of all registered plugin providers."""
|
||||
return dict(_context_reference_providers)
|
||||
|
||||
|
||||
_QUOTED_REFERENCE_VALUE = r'(?:`[^`\n]+`|"[^"\n]+"|\'[^\'\n]+\')'
|
||||
REFERENCE_PATTERN = re.compile(
|
||||
rf"(?<![\w/])@(?:(?P<simple>diff|staged)\b|(?P<kind>file|folder|git|url):(?P<value>{_QUOTED_REFERENCE_VALUE}(?::\d+(?:-\d+)?)?|\S+))"
|
||||
)
|
||||
# Plugin fallback pattern – catches any @<word>:<value> not handled by the
|
||||
# built-in regex so that plugin-registered prefixes can be resolved.
|
||||
_PLUGIN_REFERENCE_PATTERN = re.compile(
|
||||
rf"(?<![\w/])@(?P<kind>[a-zA-Z][a-zA-Z0-9_-]*):(?P<value>{_QUOTED_REFERENCE_VALUE}(?::\d+(?:-\d+)?)?|\S+)"
|
||||
)
|
||||
|
||||
TRAILING_PUNCTUATION = ",.;!?"
|
||||
_NEEDS_QUOTING = re.compile(r"""[\s()\[\]{}<>"'`]""")
|
||||
_SENSITIVE_HOME_DIRS = (".ssh", ".aws", ".gnupg", ".kube", ".docker", ".azure", ".config/gh")
|
||||
_SENSITIVE_HERMES_DIRS = (Path("skills") / ".hub",)
|
||||
_SENSITIVE_HOME_FILES = (
|
||||
@@ -60,6 +130,21 @@ class ContextReferenceResult:
|
||||
blocked: bool = False
|
||||
|
||||
|
||||
def format_reference_value(value: str) -> str:
|
||||
"""Quote a reference value so ``REFERENCE_PATTERN`` reads it back whole.
|
||||
|
||||
The unquoted alternative in the pattern is ``\\S+``, so a path containing a
|
||||
space parses as a truncated ref with the tail left behind as loose text.
|
||||
Mirrors ``formatRefValue`` in the desktop's directive-text.tsx.
|
||||
"""
|
||||
if not _NEEDS_QUOTING.search(value):
|
||||
return value
|
||||
for quote in ("`", '"', "'"):
|
||||
if quote not in value:
|
||||
return f"{quote}{value}{quote}"
|
||||
return value
|
||||
|
||||
|
||||
def parse_context_references(message: str) -> list[ContextReference]:
|
||||
refs: list[ContextReference] = []
|
||||
if not message:
|
||||
@@ -100,6 +185,27 @@ def parse_context_references(message: str) -> list[ContextReference]:
|
||||
)
|
||||
)
|
||||
|
||||
# Second pass: resolve plugin-registered prefixes the built-in pattern missed
|
||||
if _context_reference_providers:
|
||||
for match in _PLUGIN_REFERENCE_PATTERN.finditer(message):
|
||||
kind = match.group("kind")
|
||||
if kind in BUILTIN_PREFIXES:
|
||||
continue
|
||||
# Skip if already captured by the built-in pattern
|
||||
if any(r.kind == kind and r.start == match.start() for r in refs):
|
||||
continue
|
||||
if kind in _context_reference_providers:
|
||||
value = _strip_trailing_punctuation(match.group("value") or "")
|
||||
refs.append(
|
||||
ContextReference(
|
||||
raw=match.group(0),
|
||||
kind=kind,
|
||||
target=_strip_reference_wrappers(value),
|
||||
start=match.start(),
|
||||
end=match.end(),
|
||||
)
|
||||
)
|
||||
|
||||
return refs
|
||||
|
||||
|
||||
@@ -197,8 +303,12 @@ async def preprocess_context_references_async(
|
||||
f"@ context injection warning: {injected_tokens} tokens exceeds the 25% soft limit ({soft_limit})."
|
||||
)
|
||||
|
||||
stripped = _remove_reference_tokens(message, refs)
|
||||
final = stripped
|
||||
# Leave the `@file:`/`@folder:` tokens where the user typed them. The token
|
||||
# IS the reference, not scaffolding around it: clients render each one as an
|
||||
# inline chip, so stripping them left a sentence with a hole in it ("review
|
||||
# and ship") and made the desktop re-derive the refs from the attached block
|
||||
# to show them as a detached list above the prose.
|
||||
final = message
|
||||
if warnings:
|
||||
final = f"{final}\n\n--- Context Warnings ---\n" + "\n".join(f"- {warning}" for warning in warnings)
|
||||
if blocks:
|
||||
@@ -242,6 +352,16 @@ async def _expand_reference(
|
||||
except Exception as exc:
|
||||
return f"{ref.raw}: {exc}", None
|
||||
|
||||
# Plugin-provided context references
|
||||
provider = _context_reference_providers.get(ref.kind)
|
||||
if provider is not None:
|
||||
try:
|
||||
plugin_content = await provider.expand(ref.target)
|
||||
if plugin_content is not None:
|
||||
return None, f"📌 {ref.raw} ({estimate_tokens_rough(plugin_content)} tokens)\n{plugin_content}"
|
||||
except Exception as exc:
|
||||
return f"{ref.raw}: plugin expansion error: {exc}", None
|
||||
|
||||
return f"{ref.raw}: unsupported reference type", None
|
||||
|
||||
|
||||
@@ -308,7 +428,7 @@ def _expand_git_reference(
|
||||
["git", *args],
|
||||
cwd=cwd,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
timeout=30,
|
||||
stdin=subprocess.DEVNULL,
|
||||
**_popen_kwargs,
|
||||
@@ -457,19 +577,6 @@ def _parse_file_reference_value(value: str) -> tuple[str, int | None, int | None
|
||||
return _strip_reference_wrappers(value), None, None
|
||||
|
||||
|
||||
def _remove_reference_tokens(message: str, refs: list[ContextReference]) -> str:
|
||||
pieces: list[str] = []
|
||||
cursor = 0
|
||||
for ref in refs:
|
||||
pieces.append(message[cursor:ref.start])
|
||||
cursor = ref.end
|
||||
pieces.append(message[cursor:])
|
||||
text = "".join(pieces)
|
||||
text = re.sub(r"\s{2,}", " ", text)
|
||||
text = re.sub(r"\s+([,.;:!?])", r"\1", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
def _is_binary_file(path: Path) -> bool:
|
||||
mime, _ = mimetypes.guess_type(path.name)
|
||||
if mime and not mime.startswith("text/") and not any(
|
||||
@@ -534,7 +641,7 @@ def _rg_files(path: Path, cwd: Path, limit: int) -> list[Path] | None:
|
||||
["rg", "--files", str(path.relative_to(cwd))],
|
||||
cwd=cwd,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
timeout=10,
|
||||
stdin=subprocess.DEVNULL,
|
||||
**_popen_kwargs,
|
||||
@@ -547,25 +654,40 @@ def _rg_files(path: Path, cwd: Path, limit: int) -> list[Path] | None:
|
||||
return files[:limit]
|
||||
|
||||
|
||||
def _human_bytes(n: int) -> str:
|
||||
size = float(n)
|
||||
for unit in ("B", "KB", "MB", "GB"):
|
||||
if size < 1024 or unit == "GB":
|
||||
return f"{int(size)} {unit}" if unit == "B" else f"{size:.1f} {unit}"
|
||||
size /= 1024
|
||||
return f"{size:.1f} GB"
|
||||
def _agent_visible_path(path: Path) -> str:
|
||||
"""Map a host path to the path the agent's tools can read in the active backend.
|
||||
|
||||
Under a container backend (docker) the gateway host path dangles inside the
|
||||
sandbox — the container has its own filesystem and the host path is not
|
||||
mounted. Files staged into an auto-mounted cache dir (``images/``,
|
||||
``attachments/``, ...) are translated to their in-container path via the
|
||||
existing ``tools.credential_files`` machinery (#76577). Falls back to the
|
||||
host path when the backend is local or translation is unavailable.
|
||||
"""
|
||||
try:
|
||||
# Desktop/in-process gateways may not have bridged ``terminal.*``
|
||||
# config into ``TERMINAL_ENV`` at startup; run the idempotent bridge so
|
||||
# the credential_files translation gate sees the active backend.
|
||||
from tools.terminal_tool import _ensure_terminal_env_bridged
|
||||
|
||||
_ensure_terminal_env_bridged()
|
||||
from tools.credential_files import to_agent_visible_cache_path
|
||||
|
||||
return to_agent_visible_cache_path(str(path))
|
||||
except Exception:
|
||||
return str(path)
|
||||
|
||||
|
||||
def _binary_reference_block(ref: ContextReference, path: Path) -> str:
|
||||
mime, _ = mimetypes.guess_type(path.name)
|
||||
mime = mime or "application/octet-stream"
|
||||
try:
|
||||
size = _human_bytes(path.stat().st_size)
|
||||
size = format_bytes(path.stat().st_size)
|
||||
except OSError:
|
||||
size = "unknown size"
|
||||
return (
|
||||
f"📎 {ref.raw} ({mime}, {size}) — binary file, not inlined as text. "
|
||||
f"It is available on disk at `{path}`. Use your tools to work with it "
|
||||
f"It is available on disk at `{_agent_visible_path(path)}`. Use your tools to work with it "
|
||||
f"(read or convert it, extract its text, or view/render it as needed); "
|
||||
f"do not tell the user the file type is unsupported."
|
||||
)
|
||||
|
||||
@@ -21,21 +21,18 @@ from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
from openai.types.chat.chat_completion_message_tool_call import (
|
||||
ChatCompletionMessageToolCall,
|
||||
Function,
|
||||
from agent.acp_openai_bridge import (
|
||||
completion_to_stream_chunks as _completion_to_stream_chunks,
|
||||
extract_tool_calls_from_text as _extract_tool_calls_from_text,
|
||||
render_tool_bridge_sections as _render_tool_bridge_sections,
|
||||
)
|
||||
|
||||
from agent.file_safety import get_read_block_error, get_write_denied_error
|
||||
from agent.file_safety import get_read_block_error, get_write_denied_error, is_write_approval_required
|
||||
from agent.redact import redact_sensitive_text
|
||||
from tools.environments.local import hermes_subprocess_env
|
||||
|
||||
ACP_MARKER_BASE_URL = "acp://copilot"
|
||||
_DEFAULT_TIMEOUT_SECONDS = 900.0
|
||||
|
||||
_TOOL_CALL_BLOCK_RE = re.compile(r"<tool_call>\s*(\{.*?\})\s*</tool_call>", re.DOTALL)
|
||||
_TOOL_CALL_JSON_RE = re.compile(r"\{\s*\"id\"\s*:\s*\"[^\"]+\"\s*,\s*\"type\"\s*:\s*\"function\"\s*,\s*\"function\"\s*:\s*\{.*?\}\s*\}", re.DOTALL)
|
||||
|
||||
# Stderr fingerprint of the deprecated `gh copilot` CLI extension
|
||||
# (https://github.blog/changelog/2025-09-25-upcoming-deprecation-of-gh-copilot-cli-extension).
|
||||
# We require BOTH the literal product name ("gh-copilot") AND a deprecation
|
||||
@@ -74,6 +71,60 @@ def _resolve_args() -> list[str]:
|
||||
return shlex.split(raw)
|
||||
|
||||
|
||||
# Probe verdicts cached per binary path so repeated prompts against a
|
||||
# CLI that supports --acp pay the ~50ms --help cost exactly once per
|
||||
# process. Only definitive verdicts (True/False) are cached; an
|
||||
# inconclusive probe (binary missing, --help crashed or timed out) is
|
||||
# not cached so a CLI installed mid-session is picked up.
|
||||
_ACP_PROBE_CACHE: dict[str, bool] = {}
|
||||
|
||||
|
||||
def _acp_supported(command: str, args: list[str]) -> bool | None:
|
||||
"""Tri-state probe: does ``command`` accept the ACP args we'd pass?
|
||||
|
||||
Different CLI versions support different transports. The GitHub
|
||||
Copilot CLI (`@github/copilot`, late 2025+) ships with ``--acp``;
|
||||
older releases (and Claude Code v2.x as of Aug 2026) do not.
|
||||
Spawning a CLI that doesn't recognize the flag silently exits
|
||||
with code 1 and ``error: unknown option '--acp'`` on stderr,
|
||||
after which every delegate_task call hangs the parent for
|
||||
``child_timeout_seconds`` (default 600s) waiting for stdout
|
||||
that never arrives.
|
||||
|
||||
Returns:
|
||||
- ``True`` — help text advertises ``--acp``; safe to spawn.
|
||||
- ``False`` — help ran cleanly but ``--acp`` is absent; spawning
|
||||
would hang, so the caller should fast-fail with a clear error.
|
||||
- ``None`` — inconclusive (binary missing, --help failed or
|
||||
timed out). The caller must fall through to the normal spawn
|
||||
path, which surfaces the existing "Could not start Copilot ACP
|
||||
command" error with full context.
|
||||
|
||||
Only probes when ``--acp`` is actually among ``args``: a custom
|
||||
HERMES_COPILOT_ACP_ARGS transport is the operator's business.
|
||||
"""
|
||||
if "--acp" not in args:
|
||||
return True
|
||||
cached = _ACP_PROBE_CACHE.get(command)
|
||||
if cached is not None:
|
||||
return cached
|
||||
try:
|
||||
probe = subprocess.run(
|
||||
[command, "--help"],
|
||||
capture_output=True, text=True, timeout=5,
|
||||
)
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired, OSError):
|
||||
return None
|
||||
if probe.returncode != 0:
|
||||
# --help itself failed; can't tell anything about --acp.
|
||||
return None
|
||||
# Match ``--acp`` as a flag in the help text; tolerate spacing and
|
||||
# variants like ``[--acp]``.
|
||||
verdict = bool(re.search(r"(?:^|[\s\[])--acp(?:[\s=\],]|$)", probe.stdout, re.MULTILINE))
|
||||
_ACP_PROBE_CACHE[command] = verdict
|
||||
return verdict
|
||||
|
||||
|
||||
def _resolve_home_dir() -> str:
|
||||
"""Return a stable HOME for child ACP processes."""
|
||||
home = os.environ.get("HOME", "").strip()
|
||||
@@ -149,34 +200,9 @@ def _format_messages_as_prompt(
|
||||
if model:
|
||||
sections.append(f"Hermes requested model hint: {model}")
|
||||
|
||||
if isinstance(tools, list) and tools:
|
||||
tool_specs: list[dict[str, Any]] = []
|
||||
for t in tools:
|
||||
if not isinstance(t, dict):
|
||||
continue
|
||||
fn = t.get("function") or {}
|
||||
if not isinstance(fn, dict):
|
||||
continue
|
||||
name = fn.get("name")
|
||||
if not isinstance(name, str) or not name.strip():
|
||||
continue
|
||||
tool_specs.append(
|
||||
{
|
||||
"name": name.strip(),
|
||||
"description": fn.get("description", ""),
|
||||
"parameters": fn.get("parameters", {}),
|
||||
}
|
||||
)
|
||||
if tool_specs:
|
||||
sections.append(
|
||||
"Available tools (OpenAI function schema). "
|
||||
"When using a tool, emit ONLY <tool_call>{...}</tool_call> with one JSON object "
|
||||
"containing id/type/function{name,arguments}. arguments must be a JSON string.\n"
|
||||
+ json.dumps(tool_specs, ensure_ascii=False)
|
||||
)
|
||||
|
||||
if tool_choice is not None:
|
||||
sections.append(f"Tool choice hint: {json.dumps(tool_choice, ensure_ascii=False)}")
|
||||
# Copilot has no tools of its own that would collide with Hermes', so it
|
||||
# forwards the whole toolset (no allowlist).
|
||||
sections.extend(_render_tool_bridge_sections(tools, tool_choice))
|
||||
|
||||
transcript: list[str] = []
|
||||
for message in messages:
|
||||
@@ -233,140 +259,6 @@ def _render_message_content(content: Any) -> str:
|
||||
return str(content).strip()
|
||||
|
||||
|
||||
def _build_openai_tool_call(
|
||||
*,
|
||||
call_id: str,
|
||||
name: str,
|
||||
arguments: str,
|
||||
) -> ChatCompletionMessageToolCall:
|
||||
"""Build an OpenAI-compatible tool-call object for downstream handling."""
|
||||
return ChatCompletionMessageToolCall(
|
||||
id=call_id,
|
||||
call_id=call_id,
|
||||
response_item_id=None,
|
||||
type="function",
|
||||
function=Function(name=name, arguments=arguments),
|
||||
)
|
||||
|
||||
|
||||
def _completion_to_stream_chunks(completion: SimpleNamespace) -> list[SimpleNamespace]:
|
||||
"""Convert a one-shot ACP response into OpenAI-style stream chunks."""
|
||||
choice = completion.choices[0]
|
||||
message = choice.message
|
||||
tool_call_deltas = None
|
||||
if message.tool_calls:
|
||||
tool_call_deltas = []
|
||||
for index, tool_call in enumerate(message.tool_calls):
|
||||
tool_call_deltas.append(
|
||||
SimpleNamespace(
|
||||
index=index,
|
||||
id=getattr(tool_call, "id", None),
|
||||
type=getattr(tool_call, "type", "function"),
|
||||
function=SimpleNamespace(
|
||||
name=getattr(tool_call.function, "name", None),
|
||||
arguments=getattr(tool_call.function, "arguments", None),
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
delta = SimpleNamespace(
|
||||
role="assistant",
|
||||
content=message.content or None,
|
||||
tool_calls=tool_call_deltas,
|
||||
reasoning_content=message.reasoning_content,
|
||||
reasoning=message.reasoning,
|
||||
)
|
||||
data_chunk = SimpleNamespace(
|
||||
choices=[
|
||||
SimpleNamespace(
|
||||
index=0,
|
||||
delta=delta,
|
||||
finish_reason=choice.finish_reason,
|
||||
)
|
||||
],
|
||||
model=completion.model,
|
||||
usage=None,
|
||||
)
|
||||
usage_chunk = SimpleNamespace(
|
||||
choices=[],
|
||||
model=completion.model,
|
||||
usage=completion.usage,
|
||||
)
|
||||
return [data_chunk, usage_chunk]
|
||||
|
||||
|
||||
def _extract_tool_calls_from_text(text: str) -> tuple[list[ChatCompletionMessageToolCall], str]:
|
||||
if not isinstance(text, str) or not text.strip():
|
||||
return [], ""
|
||||
|
||||
extracted: list[ChatCompletionMessageToolCall] = []
|
||||
consumed_spans: list[tuple[int, int]] = []
|
||||
|
||||
def _try_add_tool_call(raw_json: str) -> None:
|
||||
try:
|
||||
obj = json.loads(raw_json)
|
||||
except Exception:
|
||||
return
|
||||
if not isinstance(obj, dict):
|
||||
return
|
||||
fn = obj.get("function")
|
||||
if not isinstance(fn, dict):
|
||||
return
|
||||
fn_name = fn.get("name")
|
||||
if not isinstance(fn_name, str) or not fn_name.strip():
|
||||
return
|
||||
fn_args = fn.get("arguments", "{}")
|
||||
if not isinstance(fn_args, str):
|
||||
fn_args = json.dumps(fn_args, ensure_ascii=False)
|
||||
call_id = obj.get("id")
|
||||
if not isinstance(call_id, str) or not call_id.strip():
|
||||
call_id = f"acp_call_{len(extracted)+1}"
|
||||
|
||||
extracted.append(
|
||||
_build_openai_tool_call(
|
||||
call_id=call_id,
|
||||
name=fn_name.strip(),
|
||||
arguments=fn_args,
|
||||
)
|
||||
)
|
||||
|
||||
for m in _TOOL_CALL_BLOCK_RE.finditer(text):
|
||||
raw = m.group(1)
|
||||
_try_add_tool_call(raw)
|
||||
consumed_spans.append((m.start(), m.end()))
|
||||
|
||||
# Only try bare-JSON fallback when no XML blocks were found.
|
||||
if not extracted:
|
||||
for m in _TOOL_CALL_JSON_RE.finditer(text):
|
||||
raw = m.group(0)
|
||||
_try_add_tool_call(raw)
|
||||
consumed_spans.append((m.start(), m.end()))
|
||||
|
||||
if not consumed_spans:
|
||||
return extracted, text.strip()
|
||||
|
||||
consumed_spans.sort()
|
||||
merged: list[tuple[int, int]] = []
|
||||
for start, end in consumed_spans:
|
||||
if not merged or start > merged[-1][1]:
|
||||
merged.append((start, end))
|
||||
else:
|
||||
merged[-1] = (merged[-1][0], max(merged[-1][1], end))
|
||||
|
||||
parts: list[str] = []
|
||||
cursor = 0
|
||||
for start, end in merged:
|
||||
if cursor < start:
|
||||
parts.append(text[cursor:start])
|
||||
cursor = max(cursor, end)
|
||||
if cursor < len(text):
|
||||
parts.append(text[cursor:])
|
||||
|
||||
cleaned = "\n".join(p.strip() for p in parts if p and p.strip()).strip()
|
||||
return extracted, cleaned
|
||||
|
||||
|
||||
|
||||
def _ensure_path_within_cwd(path_text: str, cwd: str) -> Path:
|
||||
candidate = Path(path_text)
|
||||
if not candidate.is_absolute():
|
||||
@@ -502,16 +394,43 @@ class CopilotACPClient:
|
||||
return completion
|
||||
|
||||
def _run_prompt(self, prompt_text: str, *, timeout_seconds: float) -> tuple[str, str]:
|
||||
# Fast-fail when the CLI doesn't support the ACP args we'd pass.
|
||||
# Without this guard, a CLI like Claude Code v2.x exits with
|
||||
# ``error: unknown option '--acp'`` immediately, then the parent
|
||||
# ACP loop waits the full ``child_timeout_seconds`` (default 600s)
|
||||
# for stdout that never arrives. The probe costs ~50ms and turns
|
||||
# a 600s silent hang into a 280ms clear error.
|
||||
# ``None`` (inconclusive probe — e.g. binary missing) falls
|
||||
# through to the spawn below, which raises the established
|
||||
# "Could not start Copilot ACP command" error.
|
||||
if _acp_supported(self._acp_command, self._acp_args) is False:
|
||||
preview = " ".join(self._acp_args[:3]) if self._acp_args else "(none)"
|
||||
raise RuntimeError(
|
||||
f"ACP transport not supported by '{self._acp_command}': "
|
||||
f"`{preview}` is rejected as an unknown option. "
|
||||
f"This usually means the CLI is an older release (e.g. "
|
||||
f"Claude Code v2.x) or a different tool than expected. "
|
||||
f"Either install a CLI that ships with --acp support "
|
||||
f"(e.g. `@github/copilot` late 2025+), or set "
|
||||
f"HERMES_COPILOT_ACP_COMMAND / HERMES_COPILOT_ACP_ARGS "
|
||||
f"to a working pair."
|
||||
)
|
||||
|
||||
try:
|
||||
# Hide the console the CLI child would otherwise flash on Windows
|
||||
# (#56747). Hide-only — stdio pipes stay intact for the ACP wire.
|
||||
from hermes_cli._subprocess_compat import windows_hide_flags
|
||||
|
||||
proc = subprocess.Popen(
|
||||
[self._acp_command] + self._acp_args,
|
||||
stdin=subprocess.PIPE,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
text=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
bufsize=1,
|
||||
cwd=self._acp_cwd,
|
||||
env=_build_subprocess_env(),
|
||||
creationflags=windows_hide_flags(),
|
||||
)
|
||||
except FileNotFoundError as exc:
|
||||
raise RuntimeError(
|
||||
@@ -703,7 +622,7 @@ class CopilotACPClient:
|
||||
if block_error:
|
||||
raise PermissionError(block_error)
|
||||
try:
|
||||
content = path.read_text()
|
||||
content = path.read_text(encoding="utf-8")
|
||||
except FileNotFoundError:
|
||||
content = ""
|
||||
line = params.get("line")
|
||||
@@ -730,8 +649,16 @@ class CopilotACPClient:
|
||||
denied = get_write_denied_error(str(path))
|
||||
if denied:
|
||||
raise PermissionError(denied)
|
||||
# Approval-gated paths (e.g. ~/.ssh/config) are not hard-denied
|
||||
# for interactive tools, but the ACP shim has no human channel
|
||||
# to confirm the write — fail closed here.
|
||||
if is_write_approval_required(str(path)):
|
||||
raise PermissionError(
|
||||
f"Write denied: '{path}' requires interactive approval "
|
||||
"and cannot be written through the ACP file bridge."
|
||||
)
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(str(params.get("content") or ""))
|
||||
path.write_text(str(params.get("content") or ""), encoding="utf-8")
|
||||
response = {
|
||||
"jsonrpc": "2.0",
|
||||
"id": message_id,
|
||||
|
||||
@@ -162,9 +162,17 @@ def _remove_env_source(provider: str, removed) -> RemovalResult:
|
||||
try:
|
||||
env_path = get_env_path()
|
||||
if env_path.exists():
|
||||
# Read the .env as UTF-8 with BOM tolerance, matching the
|
||||
# canonical reader in hermes_cli/config.py. read_text() with no
|
||||
# encoding falls back to the system locale (cp1252/GBK on Windows)
|
||||
# and never strips a BOM, so a Notepad-edited .env (BOM + non-ASCII
|
||||
# values) would make the first line fail the startswith() check —
|
||||
# misreporting a .env-backed var as a shell export.
|
||||
env_in_dotenv = any(
|
||||
line.strip().startswith(f"{env_var}=")
|
||||
for line in env_path.read_text(errors="replace").splitlines()
|
||||
for line in env_path.read_text(
|
||||
encoding="utf-8-sig", errors="replace"
|
||||
).splitlines()
|
||||
)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
@@ -170,6 +170,27 @@ CREDITS_USAGE_BANDS: tuple[tuple[float, str, int], ...] = (
|
||||
)
|
||||
CREDITS_USAGE_KEY = "credits.usage" # single key for the escalating usage notice
|
||||
|
||||
# Minimum subscription balance that counts as "grant not yet spent" for the
|
||||
# grant_spent crossing gate (see evaluate_credits_notices). 1¢: portal-seeded
|
||||
# states derive micros from float dollars and can carry sub-cent residue where
|
||||
# the inference headers report exactly 0 — without this floor such a seed
|
||||
# opens the gate and the first header re-creates the at-open nag.
|
||||
GRANT_UNSPENT_MIN_MICROS = 10_000
|
||||
|
||||
|
||||
def new_credits_latch() -> dict:
|
||||
"""Fresh notice latch in the shape :func:`evaluate_credits_notices` expects.
|
||||
|
||||
The policy owns this schema — every producer (agent build, lazy re-init,
|
||||
tests) must build the latch through here so a new gate key lands everywhere
|
||||
at once instead of drifting across hand-rolled literals."""
|
||||
return {
|
||||
"active": set(),
|
||||
"seen_below_90": False,
|
||||
"usage_band": None,
|
||||
"seen_grant_unspent": False,
|
||||
}
|
||||
|
||||
|
||||
# ── AgentNotice (out-of-band notice payload; driver-agnostic) ────────────────
|
||||
|
||||
@@ -205,12 +226,15 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool:
|
||||
1. The ``:free`` suffix — the canonical Nous free SKU marker (e.g.
|
||||
``nvidia/nemotron-3-ultra:free``). Free by construction on the API side
|
||||
(spend is forced to 0 for ``:free`` ids).
|
||||
2. A peek into the in-process pricing cache in ``hermes_cli.models``
|
||||
2. The ``stealth/`` prefix — Nous stealth-preview SKUs (e.g.
|
||||
``stealth/ox-alpha``) are free-tier but carry no ``:free`` suffix. Spend
|
||||
is forced to zero server-side, so these are also free by construction.
|
||||
3. A peek into the in-process pricing cache in ``hermes_cli.models``
|
||||
(populated when the model picker fetched ``/v1/models`` pricing for
|
||||
*base_url*). PEEK ONLY — a cache miss never triggers a fetch. This is
|
||||
CLI/TUI-session best-effort: gateway sessions never run the picker's
|
||||
pricing fetch, so suppression there rests entirely on the ``:free``
|
||||
suffix (which all Nous free SKUs carry).
|
||||
suffix and ``stealth/`` prefix.
|
||||
|
||||
Fail-open to False (the depleted notice still shows) on any error: wrongly
|
||||
showing the warning is recoverable noise; wrongly hiding it on a paid model
|
||||
@@ -220,6 +244,11 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool:
|
||||
return False
|
||||
if model.endswith(":free"):
|
||||
return True
|
||||
# Stealth-preview SKUs are free-tier but carry no ``:free`` suffix (see
|
||||
# docstring point 2). Naming-convention trust: if a PAID model ever shipped
|
||||
# under ``stealth/`` this would wrongly suppress the banner on it.
|
||||
if model.startswith("stealth/"):
|
||||
return True
|
||||
if not base_url:
|
||||
return False
|
||||
try:
|
||||
@@ -250,7 +279,8 @@ def evaluate_credits_notices(
|
||||
) -> tuple[list[AgentNotice], list[str]]:
|
||||
"""Reconcile credits notices against the latch. Mutates ``latch`` IN PLACE.
|
||||
|
||||
latch = {"active": set[str], "seen_below_90": bool, "usage_band": Optional[int]}.
|
||||
latch = {"active": set[str], "seen_below_90": bool, "usage_band": Optional[int],
|
||||
"seen_grant_unspent": bool}.
|
||||
|
||||
``model_is_free``: True when the session's active model is a Nous free-tier
|
||||
model (see :func:`is_free_tier_model`). Suppresses the ``credits.depleted``
|
||||
@@ -277,6 +307,18 @@ def evaluate_credits_notices(
|
||||
if uf is not None and uf < _lowest_band:
|
||||
latch["seen_below_90"] = True # gate opened: usage-band notices may now fire
|
||||
|
||||
# Grant-spent crossing gate: grant_spent may fire only after this session
|
||||
# has OBSERVED the grant meaningfully unspent (≥1¢ left — see
|
||||
# GRANT_UNSPENT_MIN_MICROS). Opening at grant-spent is a steady STATE, not
|
||||
# an event — /usage carries it; only a live in-session crossing announces.
|
||||
# Unlike seen_below_90, seeds must NOT prime this gate.
|
||||
if (
|
||||
uf is not None
|
||||
and uf < 1.0
|
||||
and state.subscription_micros >= GRANT_UNSPENT_MIN_MICROS
|
||||
):
|
||||
latch["seen_grant_unspent"] = True
|
||||
|
||||
active = latch["active"]
|
||||
|
||||
# ── Conditions ───────────────────────────────────────────────────────────
|
||||
@@ -316,12 +358,21 @@ def evaluate_credits_notices(
|
||||
active.discard(CREDITS_USAGE_KEY)
|
||||
if target_band is not None:
|
||||
# Belt-and-suspenders: a producer could set subscription_limit_micros
|
||||
# without subscription_limit_usd. Render "$? cap" rather than "$None cap".
|
||||
# without subscription_limit_usd. Render "$?" rather than "$None".
|
||||
_cap_usd = state.subscription_limit_usd or "?"
|
||||
_level = current_band[1] # type: ignore[index] (current_band set when target_band set)
|
||||
# Report absolute dollars used, not a bare "N% used": the percentage is
|
||||
# only meaningful against a Nous subscription cap (no cap → never fires),
|
||||
# so dollars are clearer and don't imply a universal %. Used = cap −
|
||||
# remaining (micros, money-safe), clamped to [0, cap]. Re-emits on band
|
||||
# change (50 → 75 → 90), not every turn — a snapshot, not a live ticker.
|
||||
_lim = state.subscription_limit_micros or 0
|
||||
_used_micros = max(0, min(_lim, _lim - state.subscription_micros))
|
||||
_used_usd = f"{_used_micros / 1_000_000:.2f}" if _lim else "?"
|
||||
_glyph = "⚠" if _level == "warn" else "•"
|
||||
to_show.append(
|
||||
AgentNotice(
|
||||
text=f"{'⚠' if _level == 'warn' else '•'} Credits {target_band}% used · ${_cap_usd} cap",
|
||||
text=f"{_glyph} You've used ${_used_usd} of your ${_cap_usd} cap",
|
||||
level=_level,
|
||||
kind=CREDITS_NOTICE_KIND,
|
||||
key=CREDITS_USAGE_KEY,
|
||||
@@ -332,7 +383,17 @@ def evaluate_credits_notices(
|
||||
latch["usage_band"] = target_band
|
||||
|
||||
# ── grant_spent ──────────────────────────────────────────────────────────
|
||||
if grant_cond and "credits.grant_spent" not in active:
|
||||
# The crossing gate guards only the SHOW and is CONSUMED by it — one
|
||||
# announcement per crossing. A header flicker (uf → None → back to 1.0)
|
||||
# clears the sticky line via grant_cond but cannot re-announce; only a
|
||||
# renewal that re-opens the gate (a fresh ≥1¢ observation) arms the next
|
||||
# announcement. .get(): default closed for any hand-built latch missing
|
||||
# the key, so a first observation can never fire this notice.
|
||||
if (
|
||||
grant_cond
|
||||
and "credits.grant_spent" not in active
|
||||
and latch.get("seen_grant_unspent", False)
|
||||
):
|
||||
to_show.append(
|
||||
AgentNotice(
|
||||
text=f"• Grant spent · ${state.purchased_usd} top-up left",
|
||||
@@ -343,6 +404,7 @@ def evaluate_credits_notices(
|
||||
)
|
||||
)
|
||||
active.add("credits.grant_spent")
|
||||
latch["seen_grant_unspent"] = False
|
||||
elif "credits.grant_spent" in active and not grant_cond:
|
||||
to_clear.append("credits.grant_spent")
|
||||
active.discard("credits.grant_spent")
|
||||
@@ -618,7 +680,8 @@ _DEV_FIXTURES: dict[str, dict] = {
|
||||
subscription_limit_micros=20_000_000, subscription_limit_usd="20.00",
|
||||
denominator_kind="subscription_cap", paid_access=True,
|
||||
),
|
||||
"grant_exhausted": dict( # used_fraction == 1.0 + purchased>0 → credits.grant_spent
|
||||
"grant_exhausted": dict( # uf == 1.0 + purchased>0 → SILENT at open (crossing-gated);
|
||||
# flip healthy → grant_exhausted via the fixture-file path to see credits.grant_spent
|
||||
remaining_micros=12_340_000, remaining_usd="12.34",
|
||||
subscription_micros=0, subscription_usd="0.00",
|
||||
subscription_limit_micros=20_000_000, subscription_limit_usd="20.00",
|
||||
@@ -732,6 +795,9 @@ def _hydrate_seed_state(agent, state) -> None:
|
||||
agent._credits_session_start_micros = state.remaining_micros
|
||||
_latch = getattr(agent, "_credits_latch", None)
|
||||
if isinstance(_latch, dict) and state.used_fraction is not None:
|
||||
# Prime ONLY seen_below_90 (open-high band warnings are wanted at open).
|
||||
# Never prime seen_grant_unspent here: a seed observing grant-spent is a
|
||||
# steady state, and priming it would revive the every-session nag.
|
||||
_latch["seen_below_90"] = True
|
||||
emit = getattr(agent, "_emit_credits_notices", None)
|
||||
if callable(emit):
|
||||
|
||||
@@ -138,8 +138,8 @@ def is_paused() -> bool:
|
||||
def _load_config() -> Dict[str, Any]:
|
||||
"""Read curator.* config from ~/.hermes/config.yaml. Tolerates missing file."""
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
cfg = load_config()
|
||||
from hermes_cli.config import load_config_readonly
|
||||
cfg = load_config_readonly()
|
||||
except Exception as e:
|
||||
logger.debug("Failed to load config for curator: %s", e)
|
||||
return {}
|
||||
@@ -325,7 +325,7 @@ def apply_automatic_transitions(now: Optional[datetime] = None) -> Dict[str, int
|
||||
|
||||
counts = {"marked_stale": 0, "archived": 0, "reactivated": 0, "checked": 0, "seeded": 0}
|
||||
|
||||
for row in _u.agent_created_report():
|
||||
for row in _u.curated_report():
|
||||
counts["checked"] += 1
|
||||
name = row["name"]
|
||||
if row.get("pinned"):
|
||||
@@ -369,7 +369,22 @@ def apply_automatic_transitions(now: Optional[datetime] = None) -> Dict[str, int
|
||||
continue
|
||||
|
||||
if anchor <= archive_cutoff and current != _u.STATE_ARCHIVED:
|
||||
ok, _msg = _u.archive_skill(name)
|
||||
# Tag the ledger entry with the curator actor: this archive is an
|
||||
# autonomous curator transition, not a foreground agent/user call.
|
||||
try:
|
||||
from tools.skill_ledger import reset_ledger_actor, set_ledger_actor
|
||||
_tok = set_ledger_actor("curator")
|
||||
except Exception:
|
||||
_tok = None
|
||||
reset_ledger_actor = None # type: ignore[assignment]
|
||||
try:
|
||||
ok, _msg = _u.archive_skill(name)
|
||||
finally:
|
||||
if _tok is not None and reset_ledger_actor is not None:
|
||||
try:
|
||||
reset_ledger_actor(_tok)
|
||||
except Exception:
|
||||
pass
|
||||
if ok:
|
||||
counts["archived"] += 1
|
||||
elif anchor <= stale_cutoff and current == _u.STATE_ACTIVE:
|
||||
@@ -422,7 +437,9 @@ CURATOR_REVIEW_PROMPT = (
|
||||
"INSTRUCTIONS AND EXPERIENTIAL KNOWLEDGE. A collection of hundreds of "
|
||||
"narrow skills where each one captures one session's specific bug is "
|
||||
"a FAILURE of the library — not a feature. An agent searching skills "
|
||||
"matches on descriptions, not on exact names; one broad umbrella "
|
||||
"matches on descriptions, not on exact names (note: long descriptions "
|
||||
"are truncated to 57 chars in the system prompt skill index — keep the "
|
||||
"trigger class in that window). One broad umbrella "
|
||||
"skill with labeled subsections beats five narrow siblings for "
|
||||
"discoverability, not the other way around.\n\n"
|
||||
"The right target shape is CLASS-LEVEL skills with rich SKILL.md "
|
||||
@@ -522,6 +539,13 @@ CURATOR_REVIEW_PROMPT = (
|
||||
"merges.\n\n"
|
||||
"Your toolset:\n"
|
||||
" - skills_list, skill_view — read the current landscape\n"
|
||||
" READ BEFORE WRITE — enforced, not advisory. Before skill_manage "
|
||||
"action=patch, action=edit, action=write_file on a file that already "
|
||||
"exists, or action=remove_file, call skill_view on that SAME target in "
|
||||
"this review turn — skill_view(name) for SKILL.md, "
|
||||
"skill_view(name, file_path=...) for a supporting file — and build the "
|
||||
"write from the content it just returned. A write without that read is "
|
||||
"REFUSED and nothing is saved.\n"
|
||||
" - skill_manage action=patch — add sections to the umbrella\n"
|
||||
" - skill_manage action=create — create a new umbrella SKILL.md\n"
|
||||
" - skill_manage action=write_file — add a references/, templates/, "
|
||||
@@ -900,7 +924,6 @@ def _reconcile_classification(
|
||||
Every removed skill is placed in exactly one bucket.
|
||||
"""
|
||||
heur_cons = {e["name"]: e for e in heuristic.get("consolidated", [])}
|
||||
heur_pruned = {e["name"] for e in heuristic.get("pruned", [])}
|
||||
|
||||
model_cons = {e["from"]: e for e in model_block.get("consolidations", [])}
|
||||
model_pruned = {e["name"]: e for e in model_block.get("prunings", [])}
|
||||
@@ -1470,15 +1493,16 @@ def _render_report_markdown(p: Dict[str, Any]) -> str:
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _render_candidate_list() -> str:
|
||||
"""Human/agent-readable list of agent-created skills with usage stats."""
|
||||
rows = skill_usage.agent_created_report()
|
||||
"""Human/agent-readable list of curator-managed skills with usage stats."""
|
||||
rows = skill_usage.curated_report()
|
||||
if not rows:
|
||||
return "No agent-created skills to review."
|
||||
return "No curator-managed skills to review."
|
||||
cron_referenced = _cron_referenced_skills()
|
||||
lines = [f"Agent-created skills ({len(rows)}):\n"]
|
||||
lines = [f"Curator-managed skills ({len(rows)}):\n"]
|
||||
for r in rows:
|
||||
lines.append(
|
||||
f"- {r['name']} "
|
||||
f"provenance={r.get('provenance', 'agent')} "
|
||||
f"state={r['state']} "
|
||||
f"pinned={'yes' if r.get('pinned') else 'no'} "
|
||||
f"cron={'yes' if r['name'] in cron_referenced else 'no'} "
|
||||
@@ -1531,7 +1555,7 @@ def run_curator_review(
|
||||
if dry_run:
|
||||
# Count candidates without mutating state.
|
||||
try:
|
||||
report = skill_usage.agent_created_report()
|
||||
report = skill_usage.curated_report()
|
||||
counts = {
|
||||
"checked": len(report),
|
||||
"marked_stale": 0,
|
||||
@@ -1584,7 +1608,7 @@ def run_curator_review(
|
||||
nonlocal auto_summary
|
||||
# Snapshot skill state BEFORE the LLM pass so the report can diff.
|
||||
try:
|
||||
before_report = skill_usage.agent_created_report()
|
||||
before_report = skill_usage.curated_report()
|
||||
except Exception:
|
||||
before_report = []
|
||||
before_names = {r.get("name") for r in before_report if isinstance(r, dict)}
|
||||
@@ -1610,7 +1634,7 @@ def run_curator_review(
|
||||
state2["last_run_duration_seconds"] = elapsed
|
||||
state2["last_run_summary"] = final_summary
|
||||
try:
|
||||
after_report = skill_usage.agent_created_report()
|
||||
after_report = skill_usage.curated_report()
|
||||
except Exception:
|
||||
after_report = []
|
||||
try:
|
||||
@@ -1697,7 +1721,7 @@ def run_curator_review(
|
||||
try:
|
||||
rename_lines = _build_rename_summary(
|
||||
before_names=before_names,
|
||||
after_report=skill_usage.agent_created_report(),
|
||||
after_report=skill_usage.curated_report(),
|
||||
tool_calls=llm_meta.get("tool_calls", []) or [],
|
||||
model_final=llm_meta.get("final", "") or "",
|
||||
)
|
||||
@@ -1715,7 +1739,7 @@ def run_curator_review(
|
||||
# reporting bug never breaks the curator itself. Report path is
|
||||
# recorded in state so `hermes curator status` can point at it.
|
||||
try:
|
||||
after_report = skill_usage.agent_created_report()
|
||||
after_report = skill_usage.curated_report()
|
||||
except Exception:
|
||||
after_report = []
|
||||
try:
|
||||
@@ -1873,9 +1897,9 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
|
||||
_acp_args = None
|
||||
_model_name = ""
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
from hermes_cli.config import load_config_readonly
|
||||
from hermes_cli.runtime_provider import resolve_runtime_provider
|
||||
_cfg = load_config()
|
||||
_cfg = load_config_readonly()
|
||||
_binding = _resolve_review_runtime(_cfg)
|
||||
_provider, _model_name = _binding.provider, _binding.model
|
||||
_rp = resolve_runtime_provider(
|
||||
@@ -1921,6 +1945,7 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
|
||||
credential_pool=_credential_pool,
|
||||
request_overrides=_request_overrides,
|
||||
**_agent_kwargs,
|
||||
enabled_toolsets=["skills", "terminal"],
|
||||
# Umbrella-building over a large skill collection is worth a
|
||||
# high iteration ceiling — the pass typically takes 50-100
|
||||
# API calls against hundreds of candidate skills. The
|
||||
|
||||
@@ -50,6 +50,7 @@ from typing import Any, Dict, List, Optional, Set, Tuple
|
||||
|
||||
from hermes_constants import get_hermes_home
|
||||
from agent.skill_utils import is_excluded_skill_path
|
||||
from hermes_cli.sizefmt import format_bytes
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -147,8 +148,8 @@ def _utc_id(now: Optional[datetime] = None) -> str:
|
||||
|
||||
def _load_config() -> Dict[str, Any]:
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
cfg = load_config()
|
||||
from hermes_cli.config import load_config_readonly
|
||||
cfg = load_config_readonly()
|
||||
except Exception as e:
|
||||
logger.debug("Failed to load config for curator backup: %s", e)
|
||||
return {}
|
||||
@@ -541,6 +542,33 @@ def _restore_cron_skill_links(snapshot_dir: Path) -> Dict[str, Any]:
|
||||
|
||||
|
||||
|
||||
def _unstage(moved: List[Tuple[Path, Path]]) -> List[str]:
|
||||
"""Move staged entries back to their original paths.
|
||||
|
||||
``shutil.move`` moves *into* an existing destination directory rather than
|
||||
replacing it, so a partially-completed extract leaves debris that would
|
||||
otherwise bury the user's real skill one level deeper
|
||||
(``skills/foo/foo/``) while the tree still looks populated. Clear whatever
|
||||
the failed extract created at each original path first. The staged copy is
|
||||
authoritative, and the pre-rollback safety snapshot is the undo handle for
|
||||
the extract's own output.
|
||||
|
||||
Returns the names that could not be restored, so the caller can report an
|
||||
incomplete recovery instead of claiming the state was restored.
|
||||
"""
|
||||
failed: List[str] = []
|
||||
for orig, dest in moved:
|
||||
try:
|
||||
if orig.is_dir() and not orig.is_symlink():
|
||||
shutil.rmtree(orig)
|
||||
elif orig.exists() or orig.is_symlink():
|
||||
orig.unlink()
|
||||
shutil.move(str(dest), str(orig))
|
||||
except OSError:
|
||||
failed.append(orig.name)
|
||||
return failed
|
||||
|
||||
|
||||
def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]]:
|
||||
"""Restore ``~/.hermes/skills/`` from a snapshot.
|
||||
|
||||
@@ -582,12 +610,19 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]
|
||||
# Protect the target from this snapshot's prune step: at the steady
|
||||
# keep limit, pruning the oldest snapshot would otherwise delete the
|
||||
# very snapshot we are about to extract from.
|
||||
snapshot_skills(
|
||||
safety_snapshot = snapshot_skills(
|
||||
reason=f"pre-rollback to {target.name}",
|
||||
protect_ids={target.name},
|
||||
)
|
||||
except Exception as e:
|
||||
return (False, f"pre-rollback safety snapshot failed: {e}", None)
|
||||
if safety_snapshot is None:
|
||||
return (
|
||||
False,
|
||||
"pre-rollback safety snapshot failed; backups may be disabled "
|
||||
"or unavailable, and current skills were not changed",
|
||||
None,
|
||||
)
|
||||
|
||||
# Additionally move current entries into an internal staging dir so
|
||||
# the extract happens into an empty skills tree (predictable result).
|
||||
@@ -609,11 +644,7 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]
|
||||
moved.append((entry, dest))
|
||||
except OSError as e:
|
||||
# Best-effort rollback of the move
|
||||
for orig, dest in moved:
|
||||
try:
|
||||
shutil.move(str(dest), str(orig))
|
||||
except OSError:
|
||||
pass
|
||||
_unstage(moved)
|
||||
try:
|
||||
shutil.rmtree(staged, ignore_errors=True)
|
||||
except OSError:
|
||||
@@ -638,12 +669,30 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]
|
||||
# Python < 3.12 — no filter kwarg
|
||||
tf.extractall(str(skills))
|
||||
except (OSError, tarfile.TarError) as e:
|
||||
# Best-effort recover: move staged contents back
|
||||
for orig, dest in moved:
|
||||
# Best-effort recover. A partial extract can leave entries the
|
||||
# original tree never had, so drop those first, otherwise the
|
||||
# "restored" tree is the user's skills plus a slice of the snapshot.
|
||||
staged_names = {orig.name for orig, _ in moved}
|
||||
for entry in list(skills.iterdir()):
|
||||
if entry.name in _EXCLUDE_TOP_LEVEL or entry.name in staged_names:
|
||||
continue
|
||||
try:
|
||||
shutil.move(str(dest), str(orig))
|
||||
if entry.is_dir() and not entry.is_symlink():
|
||||
shutil.rmtree(entry)
|
||||
else:
|
||||
entry.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
unrestored = _unstage(moved)
|
||||
if unrestored:
|
||||
# Do not claim a clean restore we did not achieve, and keep the
|
||||
# staging dir so the entries can be recovered by hand.
|
||||
return (
|
||||
False,
|
||||
f"snapshot extract failed: {e} - could not restore "
|
||||
f"{', '.join(sorted(unrestored))}; staged copies kept at {staged}",
|
||||
None,
|
||||
)
|
||||
try:
|
||||
shutil.rmtree(staged, ignore_errors=True)
|
||||
except OSError:
|
||||
@@ -692,13 +741,6 @@ def rollback(backup_id: Optional[str] = None) -> Tuple[bool, str, Optional[Path]
|
||||
# Human-readable summary for CLI
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def format_size(n: int) -> str:
|
||||
for unit in ("B", "KB", "MB", "GB"):
|
||||
if n < 1024 or unit == "GB":
|
||||
return f"{n:.1f} {unit}" if unit != "B" else f"{n} B"
|
||||
n /= 1024
|
||||
return f"{n:.1f} GB"
|
||||
|
||||
|
||||
def summarize_backups() -> str:
|
||||
rows = list_backups()
|
||||
@@ -711,6 +753,6 @@ def summarize_backups() -> str:
|
||||
f"{r.get('id','?'):<24} "
|
||||
f"{(r.get('reason','?') or '?')[:40]:<40} "
|
||||
f"{r.get('skill_files', 0):>6} "
|
||||
f"{format_size(int(r.get('archive_bytes', 0))):>8}"
|
||||
f"{format_bytes(int(r.get('archive_bytes', 0))):>8}"
|
||||
)
|
||||
return "\n".join(lines)
|
||||
|
||||