Merge upstream/main into linux-keychain-auto-detect

Resolves conflicts from upstream's DEFAULT_CONFIG extraction into
hermes_cli/config_defaults.py (password_store default moved there) and
the test-pruning waves (dropped the pruned pre-existing launch-option
tests; kept the new password-store tests).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Houston Searcy
2026-07-31 11:31:47 -04:00
4985 changed files with 475630 additions and 369004 deletions
-3
View File
@@ -97,9 +97,6 @@ packaging/
plans/
.plans/
# ACP registry manifest (icon + agent.json) — not consumed at runtime
acp_registry/
# Repo-level dotfiles that are git-only or dev-tooling config
.env.example
.envrc
+12 -2
View File
@@ -7,7 +7,7 @@ description: >-
inputs:
github-token:
description: Token for the GitHub API (gh CLI). Pass secrets.AUTOFIX_BOT_PAT from the calling workflow.
description: Token for the GitHub API (gh CLI). Pass steps.app-token.outputs.token from the calling workflow.
required: false
default: ${{ github.token }}
@@ -39,6 +39,9 @@ outputs:
ci_review:
description: Require CI-sensitive file review label.
value: ${{ steps.classify.outputs.ci_review }}
ci_review_files:
description: JSON list of CI-sensitive files changed by the pull request.
value: ${{ steps.classify.outputs.ci_review_files }}
runs:
using: composite
@@ -72,12 +75,19 @@ runs:
# Retried: a rate-limit blip or eventual-consistency 404 on a
# freshly-pushed HEAD would otherwise silently fall open (all lanes
# run — safe, but wasteful and it masks the API failure).
#
# `.files[]?` (null-safe): with --paginate, a PR more than 100
# commits ahead of its merge-base paginates the compare, and pages
# after the first carry `files: null` — bare `.files[]` makes jq
# die with "cannot iterate over: null", which fails every retry
# and forces the fail-open path (seen on stacked PRs). The full
# file list (up to the API's 300-file cap) is on page one.
CHANGED=""
for i in 1 2 3; do
if CHANGED="$(gh api \
--paginate \
"repos/${REPO}/compare/${BASE_SHA}...${HEAD_SHA}" \
--jq '.files[].filename')"; then
--jq '.files[]?.filename')"; then
break
fi
if [ "$i" = 3 ]; then
+69
View File
@@ -0,0 +1,69 @@
name: Get GitHub App Token
description: >-
Mint a short-lived (1-hour) installation access token from the repo's
GitHub App, replacing the long-lived AUTOFIX_BOT_PAT. App tokens get
5,000 req/hr per installation (vs 1,000 for the default GITHUB_TOKEN)
and are scoped to the App's installation permissions, not a user account.
Callers must source App credentials from a protected, main-only environment.
Never pass an App private key to a pull_request job, a local action, or a
reusable workflow resolved from an untrusted PR ref. The fallback keeps a
trusted caller functional when its protected environment is misconfigured.
Composite actions cannot access contexts directly, so callers pass the
public vars.APP_CLIENT_ID and protected secrets.APP_PRIVATE_KEY as inputs.
When the private key is empty, the fallback fires.
inputs:
client-id:
description: GitHub App Client ID. Pass vars.APP_CLIENT_ID from the calling workflow.
required: false
default: ''
private-key:
description: GitHub App private key PEM. Pass secrets.APP_PRIVATE_KEY from the calling workflow.
required: false
default: ''
owner:
description: GitHub App installation owner. Empty scopes the token to the current repository.
required: false
default: ''
repositories:
description: Comma- or newline-separated repositories to scope within the installation owner.
required: false
default: ''
outputs:
token:
description: A GitHub App installation access token (1-hour TTL), or GITHUB_TOKEN on forks.
value: ${{ steps.app-token.outputs.token || steps.fallback.outputs.token }}
runs:
using: composite
steps:
- name: Check if App credentials exist
id: check
shell: bash
env:
CLIENT_ID: ${{ inputs.client-id }}
run: |
if [ -n "$CLIENT_ID" ]; then
echo "has_app=true" >> "$GITHUB_OUTPUT"
else
echo "has_app=false" >> "$GITHUB_OUTPUT"
fi
- name: Create GitHub App token
id: app-token
if: steps.check.outputs.has_app == 'true'
uses: actions/create-github-app-token@bcd2ba49218906704ab6c1aa796996da409d3eb1 # v3.2.0
with:
client-id: ${{ inputs.client-id }}
private-key: ${{ inputs.private-key }}
owner: ${{ inputs.owner }}
repositories: ${{ inputs.repositories }}
- name: Fall back to GITHUB_TOKEN
id: fallback
if: steps.check.outputs.has_app != 'true'
shell: bash
run: echo "token=${{ github.token }}" >> "$GITHUB_OUTPUT"
+156 -29
View File
@@ -9,6 +9,10 @@ name: CI
# definitions, matrices, and concurrency settings. They no longer have
# ``push:`` / ``pull_request:`` triggers of their own — everything flows
# through this file.
#
# SECURITY: this workflow runs PR-controlled actions, workflows, and code.
# Do not add ``secrets: inherit`` or GitHub App credentials here. Trusted
# main-only automation uses protected environments in its own workflows.
on:
pull_request:
@@ -17,7 +21,7 @@ on:
permissions:
contents: read
pull-requests: write # needed by lint (PR comment) + supply-chain (PR comment)
pull-requests: write # needed by lint (PR comment) + supply-chain review_status
actions: read # needed by osv-scanner (SARIF upload)
security-events: write # needed by osv-scanner (SARIF upload)
packages: write # needed by docker build
@@ -46,6 +50,7 @@ jobs:
docker_meta: ${{ steps.classify.outputs.docker_meta }}
mcp_catalog: ${{ steps.classify.outputs.mcp_catalog }}
ci_review: ${{ steps.classify.outputs.ci_review }}
ci_review_files: ${{ steps.classify.outputs.ci_review_files }}
event_name: ${{ github.event_name }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
@@ -53,9 +58,7 @@ jobs:
id: classify
uses: ./.github/actions/detect-changes
with:
# Forks get no repo secrets (AUTOFIX_BOT_PAT is empty); fall back to
# the built-in read-only token so classification still works there.
github-token: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
github-token: ${{ github.token }}
# ─────────────────────────────────────────────────────────────────────
# Lane-gated sub-workflows. Each runs in parallel after detect finishes.
@@ -68,89 +71,177 @@ jobs:
uses: ./.github/workflows/tests.yml
with:
slice_count: 8
secrets: inherit
lint:
name: Python lints
needs: detect
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.ci_review == 'true'
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/lint.yml
with:
event_name: ${{ needs.detect.outputs.event_name }}
ci_review: ${{ needs.detect.outputs.ci_review == 'true' }}
secrets: inherit
js-tests:
name: JS & TS checks
needs: detect
if: needs.detect.outputs.frontend == 'true'
uses: ./.github/workflows/js-tests.yml
secrets: inherit
e2e-desktop:
name: Desktop E2E
needs: detect
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true'
uses: ./.github/workflows/e2e-desktop.yml
docs-site:
name: Docs Site
needs: detect
if: needs.detect.outputs.site == 'true'
uses: ./.github/workflows/docs-site-checks.yml
secrets: inherit
history-check:
name: Deny unrelated histories
needs: detect
if: needs.detect.outputs.event_name == 'pull_request'
uses: ./.github/workflows/history-check.yml
secrets: inherit
contributor-check:
name: Check contributors
needs: detect
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/contributor-check.yml
secrets: inherit
uv-lockfile:
name: Check uv.lock
needs: detect
uses: ./.github/workflows/uv-lockfile-check.yml
secrets: inherit
infographic-check:
name: Check no committed infographics
needs: detect
uses: ./.github/workflows/infographic-check.yml
lockfile-diff:
name: package-lock.json diff
needs: detect
if: needs.detect.outputs.event_name == 'pull_request' && needs.detect.outputs.npm_lock == 'true'
uses: ./.github/workflows/lockfile-diff.yml
secrets: inherit
docker-lint:
name: Lint Docker scripts
needs: detect
if: needs.detect.outputs.docker_meta == 'true'
uses: ./.github/workflows/docker-lint.yml
secrets: inherit
docker:
name: Build&Test Docker image
needs: detect
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true' || needs.detect.outputs.docker_meta == 'true'
# Trusted main pushes run docker.yml directly so its container-publish
# environment secrets never cross this reusable-workflow call. PR runs
# remain build/test-only and secret-free.
if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true' || needs.detect.outputs.docker_meta == 'true')
uses: ./.github/workflows/docker.yml
secrets: inherit
supply-chain:
name: Supply-chain scan
needs: detect
if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.scan == 'true' || needs.detect.outputs.deps == 'true' || needs.detect.outputs.mcp_catalog == 'true')
if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.scan == 'true' || needs.detect.outputs.deps == 'true')
uses: ./.github/workflows/supply-chain-audit.yml
with:
event_name: ${{ needs.detect.outputs.event_name }}
scan: ${{ needs.detect.outputs.scan == 'true' }}
deps: ${{ needs.detect.outputs.deps == 'true' }}
review-labels:
name: Review label gate
needs: [detect, supply-chain]
if: always() && needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.ci_review == 'true' || needs.detect.outputs.mcp_catalog == 'true' || needs.supply-chain.outputs.critical_findings == 'true')
uses: ./.github/workflows/review-labels.yml
with:
ci_review: ${{ needs.detect.outputs.ci_review == 'true' }}
ci_review_files: ${{ needs.detect.outputs.ci_review_files }}
mcp_catalog: ${{ needs.detect.outputs.mcp_catalog == 'true' }}
secrets: inherit
supply_chain: ${{ needs.supply-chain.outputs.critical_findings == 'true' }}
osv-scanner:
name: OSV scan
uses: ./.github/workflows/osv-scanner.yml
secrets: inherit
# ─────────────────────────────────────────────────────────────────────
# Live-updating PR review comment.
#
# A single ``comment-live`` job polls the GitHub Actions API every 15s
# for job statuses in this run, re-assembles the review comment from
# whatever results are available, and upserts it via the
# ``<!-- hermes-ci-review-bot -->`` marker.
#
# When the visible job set goes quiet, the poller waits 10 seconds and polls
# once more so downstream jobs created by an aggregate gate get included.
# ─────────────────────────────────────────────────────────────────────
comment-live:
name: CI review comment (live)
needs: [detect, review-labels, lockfile-diff, supply-chain, osv-scanner, uv-lockfile, history-check, contributor-check, e2e-desktop]
if: always() && github.event_name == 'pull_request' && github.event.pull_request.head.repo.fork != true
runs-on: ubuntu-latest
timeout-minutes: 40
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
ref: ${{ github.event.repository.default_branch }}
persist-credentials: false
- name: Run live comment poller
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
GITHUB_REPOSITORY: ${{ github.repository }}
GITHUB_RUN_ID: ${{ github.run_id }}
PR_NUMBER: ${{ github.event.pull_request.number }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
# Commit info for the review comment header.
COMMIT_SHA: ${{ github.event.pull_request.head.sha }}
COMMIT_MESSAGE: ${{ github.event.pull_request.head.commit.message }}
COMMIT_URL: ${{ github.server_url }}/${{ github.repository }}/pull/${{ github.event.pull_request.number }}/commits/${{ github.event.pull_request.head.sha }}
# Structured review statuses from workflow_call jobs.
# Each job outputs a JSON array of {source, results: [...]} objects
# that the assembler renders directly — no hardcoded job-name
# matching. We merge all available outputs into one array.
REVIEW_STATUSES: ${{ toJSON(needs.*.outputs.review_status) }}
run: |
set -uo pipefail
# REVIEW_STATUSES is a JSON array of strings (some may be empty
# when a job was skipped). Parse each string and merge into one
# flat array for the assembler.
python3 - <<'PYEOF'
import json, os, sys
raw = os.environ.get("REVIEW_STATUSES", "")
merged = []
if raw:
try:
arr = json.loads(raw)
except (json.JSONDecodeError, TypeError):
arr = []
for item in arr:
if not item:
continue
try:
statuses = json.loads(item)
except (json.JSONDecodeError, TypeError):
continue
if isinstance(statuses, list):
merged.extend(statuses)
# Write merged array to a temp file the poller reads.
with open("/tmp/review_statuses.json", "w") as f:
json.dump(merged, f)
print(f"Merged {len(merged)} review status entries")
PYEOF
python3 scripts/ci/live_comment.py \
--interval 15 \
--timeout 2100 \
--review-statuses-file /tmp/review_statuses.json
# ─────────────────────────────────────────────────────────────────────
# Gate: runs after everything. ``if: always()`` ensures it reports a
@@ -158,13 +249,18 @@ jobs:
# results cause it to fail; ``skipped`` is treated as success.
#
# Branch protection should require ONLY this check.
#
# Outputs ``needs-json`` — a compact ``{job_name: result}`` dict — so
# the live comment poller can list failed jobs in the PR comment.
# ─────────────────────────────────────────────────────────────────────
all-checks-pass:
name: All required checks pass
needs:
- detect
- tests
- lint
- js-tests
- e2e-desktop
- docs-site
- history-check
- contributor-check
@@ -172,20 +268,30 @@ jobs:
- lockfile-diff
- docker-lint
- supply-chain
- review-labels
- osv-scanner
# comment-live is a polling job — it doesn't block the gate.
# we don't require docker to pass rn because it's so slow lol
# - docker
if: always()
runs-on: ubuntu-latest
timeout-minutes: 10
outputs:
needs-json: ${{ steps.evaluate.outputs.needs-json }}
steps:
- name: Evaluate job results
id: evaluate
env:
NEEDS: ${{ toJSON(needs) }}
run: |
echo "$NEEDS" | python3 -c "
import json, sys
needs = json.load(sys.stdin)
# Emit compact {job_name: result} for the comment assembler.
compact = {name: info['result'] for name, info in needs.items()}
print(f'needs-json={json.dumps(compact)}')
with open('$GITHUB_OUTPUT', 'a') as f:
f.write(f'needs-json={json.dumps(compact)}\n')
failed = [name for name, info in needs.items() if info['result'] == 'failure']
for name, info in sorted(needs.items()):
result = info['result']
@@ -202,6 +308,9 @@ jobs:
# cache them on main (as a baseline), and on PRs generate an HTML diff
# report with a gantt chart + per-step breakdown. The report is uploaded
# as an artifact and a markdown summary is written to $GITHUB_STEP_SUMMARY.
#
# The live comment poller can read the standalone review-status artifact
# after the HTML report is uploaded, so its link points straight at that report.
# ─────────────────────────────────────────────────────────────────────
ci-timings:
name: CI timing report
@@ -226,10 +335,7 @@ jobs:
- name: Collect timings and generate report
env:
# Forks get no repo secrets (AUTOFIX_BOT_PAT is empty); fall back to
# the built-in read-only token so the timings API read still works
# there instead of hard-failing this advisory job on every fork PR.
GITHUB_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
GITHUB_TOKEN: ${{ github.token }}
run: |
python3 scripts/ci/timings_report.py \
--baseline ci-timings-baseline.json \
@@ -241,19 +347,40 @@ jobs:
# Advisory report — artifact-service blips must not fail the job.
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
id: ci-timings-artifact
id: ci-timings-html
with:
name: ci-timings-report
path: ci-timings-report.html
retention-days: 14
archive: false
- name: Build linked review status
if: hashFiles('ci-timings.json') != ''
env:
CI_TIMINGS_REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url }}
run: |
python3 scripts/ci/timings_report.py \
--from-json ci-timings.json \
--baseline ci-timings-baseline.json \
--review-status-out review-status.json \
--review-status-only
- name: Upload review status
if: hashFiles('review-status.json') != ''
continue-on-error: true
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: ci-timings-review-status
path: review-status.json
retention-days: 14
- name: Output summary
env:
REPORT_URL: ${{ steps.ci-timings-artifact.outputs.artifact-url}}
REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url}}
run: |
echo "# CI Timing report" >> "$GITHUB_STEP_SUMMARY"
echo "[View the full interactive report]($REPORT_URL)" >> "$GITHUB_STEP_SUMMARY"
{
echo "# CI Timing report"
echo "[View the full interactive report]($REPORT_URL)"
} >> "$GITHUB_STEP_SUMMARY"
cat ci-timings-summary.md >> "$GITHUB_STEP_SUMMARY"
- name: Save baseline cache (main only)
+18
View File
@@ -2,6 +2,10 @@ name: Contributor Attribution Check
on:
workflow_call:
outputs:
review_status:
description: "JSON array of review status objects"
value: ${{ jobs.check-attribution.outputs.review_status }}
permissions:
contents: read
@@ -10,12 +14,15 @@ jobs:
check-attribution:
runs-on: ubuntu-latest
timeout-minutes: 10
outputs:
review_status: ${{ steps.check-emails.outputs.review_status }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
fetch-depth: 0 # Full history needed for git log
- name: Check for unmapped contributor emails
id: check-emails
run: |
# Get the merge base between this PR and main
MERGE_BASE=$(git merge-base origin/main HEAD)
@@ -25,6 +32,7 @@ jobs:
if [ -z "$NEW_EMAILS" ]; then
echo "No new commits to check."
echo "review_status=[]" >> "$GITHUB_OUTPUT"
exit 0
fi
@@ -67,6 +75,16 @@ jobs:
echo ""
echo "To find the GitHub username for an email:"
echo " gh api 'search/users?q=EMAIL+in:email' --jq '.items[0].login'"
# Emit review_status for unmapped emails
DETAIL=$(echo -e "$MISSING" | sed '/^$/d; s/^ //')
HOW_TO_FIX=$'Add mappings to scripts/release.py AUTHOR_MAP:\n```\n"<email>": "<github-username>",\n```\nTo find the GitHub username for an email:\n```\ngh api \'search/users?q=EMAIL+in:email\' --jq \'.items[0].login\'\n```\n'
REVIEW_STATUS=$(jq -nc \
--arg detail "$DETAIL" \
--arg how_to_fix "$HOW_TO_FIX" \
'[{"source":"contributor attribution","results":[{"kind":"action_required","title":"Unmapped contributor email(s)","summary":"New contributor email(s) are not in AUTHOR_MAP.","detail":$detail,"how_to_fix":$how_to_fix}]}]')
echo "review_status=$REVIEW_STATUS" >> "$GITHUB_OUTPUT"
exit 1
else
echo "✅ All contributor emails are mapped."
+9 -2
View File
@@ -56,6 +56,13 @@ jobs:
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Get GitHub App token
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ vars.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
@@ -73,8 +80,8 @@ jobs:
- name: Prepare skills index (unified multi-source catalog)
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
GITHUB_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
GH_TOKEN: ${{ steps.app-token.outputs.token }}
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
SKILLS_INDEX_RUN_ID: ${{ github.event.inputs.skills_index_run_id || '' }}
REBUILD_SKILLS_INDEX: ${{ github.event.inputs.rebuild_skills_index || 'false' }}
run: |
+113 -45
View File
@@ -1,8 +1,15 @@
name: Docker Build, Test, and Publish
on:
# Trusted main pushes run this workflow directly so environment-scoped
# Docker Hub secrets are resolved by the top-level workflow, never across
# a reusable-workflow boundary.
push:
branches: [main]
release:
types: [published]
# CI calls this only for untrusted PR build/test coverage. Those runs never
# reach the protected publish or merge jobs below.
workflow_call:
permissions:
@@ -20,7 +27,9 @@ env:
IMAGE_NAME: nousresearch/hermes-agent
jobs:
# Build, test, and optionally push the image for each architecture.
# Build and test the image for each architecture. This job runs PR code,
# so it must remain secret-free. Publishing happens in the separate,
# protected publish job after these tests pass.
build:
if: github.repository == 'NousResearch/hermes-agent'
strategy:
@@ -44,7 +53,19 @@ jobs:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Retry once on transient Docker Hub / buildkit pull failures
# (connection reset, auth token timeout, rate limiting). The action
# generates a unique builder name per invocation so the retry doesn't
# collide with the failed first attempt. A genuine persistent failure
# still fails the job — only the first attempt has continue-on-error.
# Refs: docker/setup-buildx-action#510
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
# Build once, load into the local daemon for testing. Cached
@@ -62,49 +83,6 @@ jobs:
cache-from: ${{ matrix.cache-from }}
cache-to: ${{ (github.event_name != 'pull_request') && matrix.cache-to || '' }}
- name: Log in to Docker Hub
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
username: ${{ secrets.DOCKERHUB_USERNAME }}
password: ${{ secrets.DOCKERHUB_TOKEN }}
# Push by digest only (no tag). The merge job assembles the
# tagged manifest list. `push-by-digest=true` is docker's recommended
# pattern for multi-runner multi-platform builds.
- name: Push ${{ matrix.arch }} by digest
id: push
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
with:
context: .
file: Dockerfile
platforms: ${{ matrix.platform }}
labels: |
org.opencontainers.image.revision=${{ github.sha }}
build-args: |
HERMES_GIT_SHA=${{ github.sha }}
outputs: type=image,name=${{ env.IMAGE_NAME }},push-by-digest=true,name-canonical=true,push=true
cache-from: ${{ matrix.cache-from }}
cache-to: ${{ matrix.cache-to }}
# Write the digest to a file and upload it as an artifact so the
# merge job can stitch both per-arch digests into a manifest list.
- name: Export digest
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
run: |
mkdir -p /tmp/digests
digest="${{ steps.push.outputs.digest }}"
touch "/tmp/digests/${digest#sha256:}"
- name: Upload digest artifact
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: digest-${{ matrix.arch }}
path: /tmp/digests/*
if-no-files-found: error
retention-days: 1
# Run the docker-integration test suite against the freshly-built
# image already loaded into the local daemon (`:test`).
@@ -122,6 +100,11 @@ jobs:
# ---------------------------------------------------------------------
- name: Install uv (for docker tests)
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
# raw.githubusercontent.com every job; transient fetch failures
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
version: "0.9.28"
- name: Set up Python 3.11 (for docker tests)
run: uv python install 3.11
@@ -147,6 +130,82 @@ jobs:
run: |
scripts/run_tests.sh tests/docker/ --file-timeout 600
# ---------------------------------------------------------------------------
# Rebuild and push each architecture only after the unprivileged build/test
# matrix passes. This job is the sole Docker Hub credential boundary.
# ---------------------------------------------------------------------------
publish:
if: github.repository == 'NousResearch/hermes-agent' && (github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release')
needs: [build]
environment: container-publish
strategy:
fail-fast: false
matrix:
include:
- arch: amd64
runner: ubuntu-latest
platform: linux/amd64
cache-from: type=gha,scope=docker-amd64
cache-to: type=gha,mode=max,scope=docker-amd64
- arch: arm64
runner: ubuntu-24.04-arm
platform: linux/arm64
cache-from: type=gha,scope=docker-arm64
cache-to: type=gha,mode=max,scope=docker-arm64
runs-on: ${{ matrix.runner }}
timeout-minutes: 30
steps:
- name: Checkout trusted source
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Retry once on transient Docker Hub / buildkit pull failures.
# See build job for rationale; same pattern.
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Log in to Docker Hub
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
username: ${{ secrets.DOCKERHUB_USERNAME }}
password: ${{ secrets.DOCKERHUB_TOKEN }}
# Push by digest only (no tag). The merge job assembles the tagged
# manifest list after both architecture publishers complete.
- name: Push ${{ matrix.arch }} by digest
id: push
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
with:
context: .
file: Dockerfile
platforms: ${{ matrix.platform }}
labels: |
org.opencontainers.image.revision=${{ github.sha }}
build-args: |
HERMES_GIT_SHA=${{ github.sha }}
outputs: type=image,name=${{ env.IMAGE_NAME }},push-by-digest=true,name-canonical=true,push=true
cache-from: ${{ matrix.cache-from }}
cache-to: ${{ matrix.cache-to }}
- name: Export digest
run: |
mkdir -p /tmp/digests
digest="${{ steps.push.outputs.digest }}"
touch "/tmp/digests/${digest#sha256:}"
- name: Upload digest artifact
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: digest-${{ matrix.arch }}
path: /tmp/digests/*
if-no-files-found: error
retention-days: 1
# ---------------------------------------------------------------------------
# Stitch both per-arch digests into a single tagged multi-arch manifest.
# This is a registry-side operation — no building, no layer re-push —
@@ -158,8 +217,9 @@ jobs:
merge:
if: github.repository == 'NousResearch/hermes-agent' && (github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release')
runs-on: ubuntu-latest
needs: [build]
needs: [publish]
timeout-minutes: 10
environment: container-publish
steps:
- name: Download digests
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
@@ -168,7 +228,15 @@ jobs:
pattern: digest-*
merge-multiple: true
# Retry once on transient Docker Hub / buildkit pull failures.
# See build job for rationale; same pattern.
- name: Set up Docker Buildx
id: buildx
continue-on-error: true
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Set up Docker Buildx (retry)
if: steps.buildx.outcome == 'failure'
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
- name: Log in to Docker Hub
+260
View File
@@ -0,0 +1,260 @@
name: E2E Desktop
on:
workflow_call:
outputs:
review_status:
description: Screenshot and visual-diff status for the CI review comment.
value: ${{ jobs.e2e.outputs.review_status }}
permissions:
contents: read
concurrency:
group: e2e-desktop-${{ github.ref }}
cancel-in-progress: true
jobs:
e2e:
name: Playwright E2E (Linux)
runs-on: ubuntu-latest
timeout-minutes: 20
outputs:
review_status: ${{ steps.review-status.outputs.review_status }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# ── System deps for Electron on headless Ubuntu ───────────────────
# Electron needs GTK, NSS,atk, etc. even under xvfb. Playwright's
# install-deps covers browsers; for Electron we install the apt
# packages directly.
- name: Install system dependencies for Electron
run: |
sudo apt-get update -qq
sudo apt-get install -y -qq \
xvfb \
libgtk-3-0 libnotify4 libnss3 libxss1 libxtst6 \
xdg-utils libatspi2.0-0 libdrm2 libgbm1 libasound2t64
# ── Node ───────────────────────────────────────────────────────────
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 22
cache: npm
# Full npm ci (not --ignore-scripts): electron's postinstall
# downloads the binary we launch, and node-pty's native build is
# needed for the terminal pane.
- uses: ./.github/actions/retry
with:
command: npm ci
# ── Python (for the hermes serve backend) ──────────────────────────
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pin the uv version: unpinned, setup-uv resolves "latest" by
# fetching a manifest from raw.githubusercontent.com on EVERY job —
# a transient fetch failure fails the whole job (2026-07-28 slice-5
# incident). Pinned, the binary downloads directly; no manifest hop.
version: "0.9.28"
enable-cache: true
cache-dependency-glob: |
pyproject.toml
uv.lock
- name: Set up Python 3.11
run: uv python install 3.11
- name: Install Python dependencies
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev
# ── Build desktop app ─────────────────────────────────────────────
# The Playwright step below runs `npm run build` before testing so
# dist/ is always fresh — no separate build step needed here.
# ── Restore visual baseline screenshots from main ──────────────────
# Baselines are generated on main (via --update-snapshots) and cached.
# On PRs, we restore them so toHaveScreenshot has something to compare
# against. The cache key is keyed on the desktop source files so a
# UI change naturally invalidates it — but we fall back to the main
# cache to avoid cold starts on unrelated PRs.
- name: Restore visual baseline screenshots
id: restore-baselines
uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4
with:
path: apps/desktop/e2e/*-snapshots
key: visual-baselines-${{ github.ref_name }}
restore-keys: |
visual-baselines-main
# ── Run Playwright E2E under xvfb ─────────────────────────────────
# xvfb runs at a fixed 1280x1024 screen so the 1220x800 Electron
# window always has a consistent viewport for screenshot comparison.
# On main, we run with --update-snapshots to generate baselines.
# `npm run test:e2e` builds dist/ as a pretest hook so the renderer
# is always fresh — no separate build step needed.
- name: Run Playwright E2E tests
working-directory: apps/desktop
run: |
if [ "${{ github.ref_name }}" = "main" ]; then
echo "On main — generating/updating baseline screenshots"
npm run build && xvfb-run -a --server-args="-screen 0 1280x1024x24" \
npx playwright test --reporter=list --update-snapshots
else
echo "On PR — comparing against cached baselines"
npm run build && xvfb-run -a --server-args="-screen 0 1280x1024x24" \
npx playwright test --reporter=list
fi
env:
CI: "true"
# Ensure no real API keys leak into the test env.
OPENROUTER_API_KEY: ""
OPENAI_API_KEY: ""
NOUS_API_KEY: ""
# ── Save updated baselines to cache (main only) ───────────────────
- name: Save updated baselines to cache
if: github.ref_name == 'main' && always()
uses: actions/cache/save@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4
with:
path: apps/desktop/e2e/*-snapshots
key: visual-baselines-main
# ── Upload Playwright report (HTML + traces) ──────────────────────
- name: Upload Playwright report
id: upload-report
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: playwright-report-${{ github.sha }}
path: apps/desktop/playwright-report
retention-days: 14
overwrite: true
# ── Upload test results (screenshots, traces, diffs) ───────────────
- name: Upload test results
id: upload-results
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: playwright-test-results-${{ github.sha }}
path: apps/desktop/test-results
retention-days: 14
overwrite: true
# ── Upload just the visual diffs (small, fast to review) ──────────
- name: Upload visual diffs
id: upload-diffs
if: always() && github.ref_name != 'main'
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: visual-diffs-${{ github.sha }}
path: |
apps/desktop/test-results/**/*-diff.png
apps/desktop/test-results/**/*-actual.png
apps/desktop/test-results/**/*-expected.png
retention-days: 14
overwrite: true
if-no-files-found: ignore
- name: Build screenshot review status
id: review-status
if: always()
working-directory: apps/desktop
env:
RESULTS_URL: ${{ steps.upload-results.outputs.artifact-url }}
run: |
python3 ../../scripts/ci/e2e_screenshot_status.py \
--results-dir test-results \
--manifest-output /tmp/e2e-screenshot-manifest.json \
--evidence-dir /tmp/e2e-evidence \
--artifact-url "$RESULTS_URL" \
--output /tmp/e2e-review-status.json
{
echo 'review_status<<__E2E_REVIEW_STATUS__'
cat /tmp/e2e-review-status.json
echo '__E2E_REVIEW_STATUS__'
} >> "$GITHUB_OUTPUT"
# The trusted workflow_run publisher consumes only this flat, bounded
# artifact. It turns selected images into GitHub attachment URLs; it
# never checks out or runs this PR's code.
- name: Upload inline E2E evidence
if: always() && github.ref_name != 'main'
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: e2e-evidence-${{ github.sha }}
path: /tmp/e2e-evidence
retention-days: 14
overwrite: true
if-no-files-found: error
# ── Generate step summary with visual diff info ───────────────────
# Parse the JSON report + scan for diff images, then post a summary
# to the GitHub Actions step output so reviewers can see what changed
# without downloading artifacts. Runs AFTER uploads so it can link
# the artifact download URLs from their step outputs.
- name: Generate visual diff summary
if: always()
working-directory: apps/desktop
env:
REPORT_URL: ${{ steps.upload-report.outputs.artifact-url }}
RESULTS_URL: ${{ steps.upload-results.outputs.artifact-url }}
DIFFS_URL: ${{ steps.upload-diffs.outputs.artifact-url }}
run: |
{
echo "## Desktop E2E — Visual Diff Report"
echo ""
# Count diff images (playwright writes *-diff.png on mismatch)
DIFF_COUNT=$(find test-results -name '*-diff.png' 2>/dev/null | wc -l)
ACTUAL_COUNT=$(find test-results -name '*-actual.png' 2>/dev/null | wc -l)
if [ "$DIFF_COUNT" -eq 0 ]; then
echo "✅ All $ACTUAL_COUNT screenshot(s) matched their baselines (or no baselines existed yet)."
else
echo "📸 **$DIFF_COUNT of $ACTUAL_COUNT screenshot(s) differ from baseline:**"
echo ""
echo "| Test | Diff | Actual | Expected |"
echo "|------|------|--------|----------|"
# List each diff image with a link to the artifact
for diff in $(find test-results -name '*-diff.png' 2>/dev/null | sort); do
base=${diff%-diff.png}
test_name=$(basename "$base")
echo "| $test_name | [diff]($diff) | [actual](${base}-actual.png) | [expected](${base}-expected.png) |"
done
fi
echo ""
echo "📥 **Artifacts:**"
echo ""
if [ -n "$RESULTS_URL" ]; then
echo "- [playwright-test-results]($RESULTS_URL) — all screenshots (actual + expected + diff) + traces"
fi
if [ -n "$REPORT_URL" ]; then
echo "- [playwright-report]($REPORT_URL) — interactive HTML report"
fi
if [ -n "$DIFFS_URL" ]; then
echo "- [visual-diffs]($DIFFS_URL) — just the diffed screenshots (small, fast to review)"
fi
echo ""
echo "**To update baselines:** merge to main (baselines auto-update on main runs) or run \`npx playwright test --update-snapshots\` locally."
# Also parse the JSON report for pass/fail counts
if [ -f playwright-report/results.json ]; then
echo ""
echo "### Test Results"
echo ""
node -e "
const r = require('./playwright-report/results.json');
const stats = r.stats || {};
console.log('| Status | Count |');
console.log('|--------|-------|');
console.log('| ✅ Passed | ' + (stats.expected || 0) + ' |');
console.log('| ❌ Failed | ' + (stats.unexpected || 0) + ' |');
console.log('| ⏭️ Skipped | ' + (stats.skipped || 0) + ' |');
console.log('| 🔄 Flaky | ' + (stats.flaky || 0) + ' |');
" 2>/dev/null || true
fi
} >> "$GITHUB_STEP_SUMMARY"
+11 -1
View File
@@ -15,6 +15,10 @@ name: History Check
on:
workflow_call:
outputs:
review_status:
description: "JSON array of review_status objects for the synthesizer."
value: ${{ jobs.check-common-ancestor.outputs.review_status }}
permissions:
contents: read
@@ -23,18 +27,23 @@ jobs:
check-common-ancestor:
runs-on: ubuntu-latest
timeout-minutes: 10
outputs:
review_status: ${{ steps.merge-base-check.outputs.review_status }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
fetch-depth: 0 # full history both sides for merge-base
- name: Reject PRs with no common ancestor on main
- id: merge-base-check
name: Reject PRs with no common ancestor on main
run: |
# `git merge-base` exits non-zero AND prints nothing when the two
# commits share no ancestor. We check both conditions explicitly
# so the failure message is clear regardless of which signal fires
# first.
if ! BASE=$(git merge-base origin/main HEAD 2>/dev/null) || [ -z "$BASE" ]; then
STATUS='[{"source":"unrelated histories","results":[{"kind":"action_required","title":"Unrelated histories","summary":"This PR has no common ancestor with main.","detail":"","how_to_fix":"Rebase your changes onto current main:\n```\ngit fetch origin main\ngit checkout -b fix-branch origin/main\n# re-apply your changes (cherry-pick, copy files, etc.)\ngit push -f origin fix-branch\n```\n"}]}]'
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
echo ""
echo "::error::This PR has no common ancestor with main."
echo ""
@@ -56,3 +65,4 @@ jobs:
exit 1
fi
echo "::notice::Common ancestor with main: $BASE"
echo "review_status=[]" >> "$GITHUB_OUTPUT"
+78
View File
@@ -0,0 +1,78 @@
name: Infographic Check
# Rejects PRs that commit PR-infographic images into the repo.
#
# PR infographics are rendered to an image-provider URL (fal.media) and
# embedded in the PR *description*. The PR body is the archive; the binary
# never belongs in git history.
#
# This has now leaked twice. PR #48261 removed the first batch, PR #54564
# removed a second batch and added `infographic/` to `.gitignore` — but
# `.gitignore` only stops *accidental* `git add`. It does nothing against
# `git add -f`, and it does nothing for a path that does not literally match
# the ignore pattern. Nine more PNGs (~14MB) were committed in the four
# weeks AFTER that rule landed, plus PR #70552 caught an `infograficos/`
# spelling that sidestepped the pattern entirely.
#
# A passive ignore rule cannot enforce a policy. This check can.
on:
workflow_call:
outputs:
review_status:
description: "JSON array of review_status objects for the synthesizer."
value: ${{ jobs.check-no-committed-infographics.outputs.review_status }}
permissions:
contents: read
jobs:
check-no-committed-infographics:
runs-on: ubuntu-latest
timeout-minutes: 10
outputs:
review_status: ${{ steps.infographic-check.outputs.review_status }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- id: infographic-check
name: Reject committed PR-infographic images
run: |
# Match on the IMAGE, not on a directory name. Keying this to
# `infographic/` is what let `infograficos/` through in #70552 —
# any localized or typo'd directory would sidestep it again.
# Instead: find tracked raster images whose path contains an
# infographic-ish segment, in any spelling, at any depth.
#
# `docs/assets` and `website/` legitimately hold product imagery
# and are excluded; those are referenced from shipped docs pages.
OFFENDERS=$(git ls-files -z \
| tr '\0' '\n' \
| grep -iE '(^|/)(infograph|infograf)[^/]*/' \
| grep -iE '\.(png|jpe?g|webp|gif)$' \
|| true)
if [ -n "$OFFENDERS" ]; then
COUNT=$(printf '%s\n' "$OFFENDERS" | wc -l | tr -d ' ')
STATUS='[{"source":"committed infographics","results":[{"kind":"action_required","title":"PR infographic committed to the repo","summary":"Infographic images belong in the PR description, never in git.","detail":"","how_to_fix":"Untrack the image and reference the provider URL from the PR body instead:\n```\ngit rm --cached <path-to-image>\n```\nThen put it in the PR description:\n```\n## Infographic\n\n![slug](https://<provider-url>)\n```\n"}]}]'
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
echo ""
echo "::error::${COUNT} PR-infographic image(s) are tracked in git."
echo ""
printf '%s\n' "$OFFENDERS" | sed 's/^/ /'
echo ""
echo "PR infographics are rendered to an image-provider URL and"
echo "embedded in the PR DESCRIPTION. The PR body is the archive —"
echo "the binary never enters git history."
echo ""
echo "This rule has been re-established twice already (#48261,"
echo "#54564) and leaked both times, because .gitignore cannot stop"
echo "'git add -f' or a differently-spelled directory (#70552)."
echo ""
echo "To fix:"
echo " git rm --cached <path> # keeps your local copy"
echo " # then embed the provider URL in the PR description"
exit 1
fi
echo "::notice::No committed PR-infographic images."
echo "review_status=[]" >> "$GITHUB_OUTPUT"
+11 -3
View File
@@ -7,7 +7,7 @@ name: auto-fix lint issues & formatting
# auto-corrected on merge so PRs aren't blocked by them. The PR-time eslint
# check in typecheck.yml fails only when un-fixable errors remain.
#
# NOTE: AUTOFIX_BOT_PAT pushes DO trigger further workflow runs (unlike
# NOTE: App token pushes DO trigger further workflow runs (unlike
# secrets.GITHUB_TOKEN). The concurrency group (ts-autofix-${{ github.ref }})
# with cancel-in-progress: true prevents an infinite loop — a re-triggered
# run cancels the in-flight one, and since the second run finds no new fixes
@@ -122,12 +122,20 @@ jobs:
if: needs.generate-patch.outputs.has-fixes == 'true'
runs-on: ubuntu-latest
timeout-minutes: 15
environment: trusted-automation
permissions:
contents: write # needed to push to bot/js-autofix
pull-requests: write # needed for PR creation + auto-merge
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Get GitHub App token
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ vars.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- name: Download patch
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
with:
@@ -170,7 +178,7 @@ jobs:
- name: Create/update PR and enable auto-merge
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
GH_TOKEN: ${{ steps.app-token.outputs.token }}
BOT_BRANCH: bot/js-autofix
run: |
set -euo pipefail
@@ -193,7 +201,7 @@ jobs:
- name: Wait for merge, auto-close on failure or stale
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
GH_TOKEN: ${{ steps.app-token.outputs.token }}
START_SHA: ${{ github.sha }}
run: |
set -euo pipefail
+29 -11
View File
@@ -10,7 +10,7 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 20
outputs:
packages: ${{ steps.set-matrix.outputs.packages }}
checks: ${{ steps.set-matrix.outputs.checks }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
@@ -22,21 +22,40 @@ jobs:
command: npm ci --ignore-scripts
- id: set-matrix
run: |
PACKAGES=$(npm query .workspace | jq -c '[.[].location]')
if [ "$PACKAGES" = "[]" ] || [ -z "$PACKAGES" ]; then
echo "::error::Workspace discovery produced an empty package list — refusing to emit a zero-length matrix (would skip all JS/TS checks silently)."
exit 1
fi
echo "packages=$PACKAGES" >> "$GITHUB_OUTPUT"
node -e '
const { execSync } = require("child_process");
const pkgs = JSON.parse(execSync("npm query .workspace", { encoding: "utf-8" }));
if (pkgs.length === 0) {
console.error("::error::Workspace discovery produced an empty package list — refusing to emit a zero-length matrix (would skip all JS/TS checks silently).");
process.exit(1);
}
const checks = [];
for (const pkg of pkgs) {
const scripts = pkg.scripts || {};
const subs = Object.keys(scripts).filter(s => /^check:.+$/.test(s));
if (subs.length > 0) {
for (const script of subs) {
checks.push({ package: pkg.location, script });
}
} else if (scripts.check) {
checks.push({ package: pkg.location, script: "check" });
}
}
if (checks.length === 0) {
console.error("::error::No check scripts found in any workspace package.");
process.exit(1);
}
process.stdout.write("checks=" + JSON.stringify(checks) + "\n");
' >> "$GITHUB_OUTPUT"
check:
name: Typecheck & Test
name: ${{ matrix.package }} / ${{ matrix.script }}
needs: workspaces
runs-on: ubuntu-latest
timeout-minutes: 20
strategy:
matrix:
package: ${{ fromJson(needs.workspaces.outputs.packages) }}
include: ${{ fromJson(needs.workspaces.outputs.checks) }}
fail-fast: false # report all failures, not just the first one
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
@@ -47,5 +66,4 @@ jobs:
- uses: ./.github/actions/retry
with:
command: npm ci
- run: npm run --prefix ${{ matrix.package }} check
- run: npm run --prefix ${{ matrix.package }} fix
- run: npm run --prefix ${{ matrix.package }} ${{ matrix.script }}
+81
View File
@@ -0,0 +1,81 @@
name: Label rerun
# When the ``ci-reviewed`` label is added to a PR, rerun all failed jobs in
# the latest CI run. This re-evaluates ``review-labels`` (which now sees the
# label) and GitHub automatically reruns dependent jobs (``comment-live``,
# ``all-checks-pass``) — so the review comment gets updated too.
#
# If the CI run is still in progress when the label is added, we wait for it
# to finish before rerunning (``gh run rerun`` only works on completed runs).
# The wait can be long (20+ min for a full CI run), but it's better than
# silently failing and leaving the reviewer stuck.
on:
pull_request:
types: [labeled]
permissions:
actions: write
pull-requests: read
concurrency:
group: label-rerun-${{ github.event.pull_request.number }}
cancel-in-progress: true
jobs:
rerun-review-labels:
name: Rerun review-labels job
if: github.event.label.name == 'ci-reviewed'
runs-on: ubuntu-latest
timeout-minutes: 40
steps:
- name: Wait for CI run to finish, then rerun failed jobs
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
REPO: ${{ github.repository }}
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
run: |
set -uo pipefail
# Find the latest CI run for this PR's head SHA.
RUN_ID=$(gh run list \
--repo "$REPO" \
--commit "$HEAD_SHA" \
--workflow ci.yml \
--limit 1 \
--json databaseId,status \
--jq '.[0] | "\(.databaseId) \(.status)"' 2>/dev/null || true)
if [ -z "$RUN_ID" ]; then
echo "No CI run found for this PR — nothing to rerun."
exit 0
fi
# Split "RUN_ID STATUS" into two vars.
RUN_ID="${RUN_ID%% *}"
STATUS="${RUN_ID##* }"
echo "Latest CI run: $RUN_ID (status: $STATUS)"
# If the run is still in progress, wait for it to finish.
# gh run rerun only works on completed runs — if we try while it's
# running, GitHub rejects with "cannot be rerun; This workflow is
# already running".
if [ "$STATUS" != "completed" ]; then
echo "Run is $STATUS — waiting for completion (this may take a while)..."
# gh run watch --exit-status exits non-zero if the run fails,
# which is expected (the label gate fails). Don't let that kill
# the workflow — we WANT to rerun failed jobs.
timeout 2100 gh run watch "$RUN_ID" --repo "$REPO" --interval 15 || true
# Verify it's actually completed now.
STATUS=$(gh run view "$RUN_ID" --repo "$REPO" --json status --jq '.status' 2>/dev/null || echo "unknown")
if [ "$STATUS" != "completed" ]; then
echo "Run is still $STATUS after wait — giving up."
exit 0
fi
fi
echo "Run completed. Rerunning all failed jobs..."
gh run rerun "$RUN_ID" --repo "$REPO" --failed || true
echo "Done. GitHub will rerun review-labels and all dependent jobs."
+16 -120
View File
@@ -2,11 +2,14 @@ name: Lint (ruff + ty)
# Two things here:
# 1. Advisory diff — ruff + ty diagnostics as a diff vs the target branch.
# Posts a Markdown summary and a PR comment. Exit zero always.
# Writes a Markdown summary to the run page. Exit zero always.
# 2. Blocking ``ruff check .`` — enforces the explicit rules in
# ``[tool.ruff.lint.select]`` (currently PLW1514). Failure blocks merge.
# Separate job so the advisory diff still runs and posts even when
# enforcement fails.
# Separate job so the advisory diff still runs even when enforcement
# fails.
#
# CI-sensitive file review was previously here as a ``ci-review`` job but
# has moved to ``review-labels.yml`` so it can be rerun independently.
on:
workflow_call:
@@ -15,14 +18,9 @@ on:
description: The event name from the calling orchestrator (pull_request or push).
type: string
required: true
ci_review:
description: Whether CI-sensitive files (eslint config, workflows, actions) changed and require a review label.
type: boolean
default: false
permissions:
contents: read
pull-requests: write # needed to post/update PR comments
concurrency:
group: lint-${{ github.ref }}
@@ -42,6 +40,11 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
# raw.githubusercontent.com every job; transient fetch failures
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
version: "0.9.28"
- name: Install ruff + ty
uses: ./.github/actions/retry
@@ -131,6 +134,11 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
# raw.githubusercontent.com every job; transient fetch failures
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
version: "0.9.28"
- name: Install ruff
uses: ./.github/actions/retry
@@ -162,115 +170,3 @@ jobs:
- name: Run footgun checker
run: python scripts/check-windows-footguns.py --all
ci-review:
# Require explicit maintainer review when CI-sensitive files change:
# eslint config, workflow YAMLs, or composite actions. These files
# influence what code the js-autofix job executes and pushes to
# main, so a malicious PR could inject arbitrary code via a custom eslint
# rule's `fix` function. The label gate ensures a human reviews before
# merge. Mirrors the mcp-catalog-reviewed pattern in supply-chain-audit.yml.
name: CI-sensitive file review
if: inputs.event_name == 'pull_request' && inputs.ci_review
runs-on: ubuntu-latest
timeout-minutes: 2
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Require ci-reviewed label
id: label-check
env:
# Read-only label lookup. Use the built-in GITHUB_TOKEN (present and
# read-only on forks) so the gate works on fork PRs; fall back to it
# when AUTOFIX_BOT_PAT is empty. `|| true` degrades an API blip to
# "label absent" rather than hard-failing the step.
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
run: |
set -euo pipefail
PR="${{ github.event.pull_request.number }}"
LABELS=$(gh pr view "$PR" --json labels --jq '.labels[].name' || true)
if echo "$LABELS" | grep -Fxq 'ci-reviewed'; then
echo "reviewed=true" >> "$GITHUB_OUTPUT"
echo "ci-reviewed label present."
exit 0
fi
echo "reviewed=false" >> "$GITHUB_OUTPUT"
# On failure: find the bot's previous comment and edit it, or create
# a new one if none exists. Using an HTML comment marker so we can
# locate it reliably across runs without parsing the body text.
# Skipped on fork PRs — GITHUB_TOKEN is read-only there, so the API
# call would fail. The label gate still holds via the step below.
- name: Post or update review warning
if: steps.label-check.outputs.reviewed != 'true' && github.event.pull_request.head.repo.fork != true
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
run: |
set -euo pipefail
PR="${{ github.event.pull_request.number }}"
MARKER="<!-- ci-review-bot -->"
BODY="${MARKER}
## ⚠️ CI-sensitive file review required
This PR changes CI-sensitive files (eslint config, workflow YAMLs,
or composite actions). These files influence what code the
js-autofix job executes and pushes to main.
A maintainer should verify:
- no new eslint rules with custom \`fix\` functions that write outside linted paths,
- no workflow changes that widen permissions or remove guards,
- no composite action changes that alter what gets executed.
After review, add the \`ci-reviewed\` label and re-run this check."
# Find an existing comment with our marker.
COMMENT_ID=$(gh api \
"repos/${{ github.repository }}/issues/${PR}/comments" \
--paginate --jq ".[] | select(.body | contains(\"${MARKER}\")) | .id" \
| head -1 || true)
if [ -n "$COMMENT_ID" ]; then
gh api --method PATCH \
"repos/${{ github.repository }}/issues/comments/${COMMENT_ID}" \
-f body="$BODY"
else
gh pr comment "$PR" --body "$BODY"
fi
# Fail the job when the label is missing — always runs (including
# fork PRs) so the security gate holds even when the comment step
# was skipped above.
- name: Fail on missing label
if: steps.label-check.outputs.reviewed != 'true'
run: |
echo "::error::CI-sensitive changes require the ci-reviewed label."
exit 1
# On success: if a previous warning comment exists, edit it to show
# the review passed so the PR doesn't have a stale ⚠️ sitting around.
# Skipped on fork PRs — no comment was ever posted to update.
- name: Update previous warning to passed
if: steps.label-check.outputs.reviewed == 'true' && github.event.pull_request.head.repo.fork != true
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
run: |
set -euo pipefail
PR="${{ github.event.pull_request.number }}"
MARKER="<!-- ci-review-bot -->"
# Find an existing comment with our marker.
COMMENT_ID=$(gh api \
"repos/${{ github.repository }}/issues/${PR}/comments" \
--paginate --jq ".[] | select(.body | contains(\"${MARKER}\")) | .id" \
| head -1 || true)
if [ -n "$COMMENT_ID" ]; then
BODY="${MARKER}
## ✅ CI-sensitive file review passed
The \`ci-reviewed\` label is present on this PR."
gh api --method PATCH \
"repos/${{ github.repository }}/issues/comments/${COMMENT_ID}" \
-f body="$BODY"
fi
+39 -42
View File
@@ -7,22 +7,25 @@ name: Lockfile diff
# the ``packages`` map at the merge base and at HEAD and set-diffs the
# {install path: version} maps instead.
#
# The comment is upserted: the script embeds a hidden HTML marker and the
# workflow PATCHes the existing comment when one is found, so a PR gets
# exactly one lockfile-diff comment that tracks the latest push instead
# of a stack of stale ones. When a later push reverts all lockfile
# changes, the comment is updated to say so (deleting it would be more
# surprising than telling the reviewer it's resolved).
# The semantic diff is exposed as a workflow_call output ``review_status``
# (a JSON array in the unified status format) and an artifact
# (``lockfile-diff`` containing the markdown fragment) for the step
# summary.
#
# Never blocking — this is review signal, not enforcement. Exit is 0 even
# when commenting fails (fork PRs get a read-only GITHUB_TOKEN).
# Never blocking — this is review signal, not enforcement.
on:
workflow_call:
outputs:
changed:
description: Whether package-lock.json changed relative to the target branch.
value: ${{ jobs.diff.outputs.changed }}
review_status:
description: JSON array of review status objects for the unified PR comment.
value: ${{ jobs.diff.outputs.review_status }}
permissions:
contents: read
pull-requests: write # post/update the diff comment
concurrency:
group: lockfile-diff-${{ github.event.pull_request.number || github.ref }}
@@ -33,6 +36,9 @@ jobs:
name: package-lock.json semantic diff
runs-on: ubuntu-latest
timeout-minutes: 5
outputs:
changed: ${{ steps.diff.outputs.changed }}
review_status: ${{ steps.emit-status.outputs.review_status }}
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
@@ -54,45 +60,36 @@ jobs:
--output /tmp/lockfile-diff.md
if [ -s /tmp/lockfile-diff.md ]; then
echo "changed=true" >> "$GITHUB_OUTPUT"
cat /tmp/lockfile-diff.md >> "$GITHUB_STEP_SUMMARY"
{
echo "## package-lock.json semantic diff"
echo ""
cat /tmp/lockfile-diff.md
} >> "$GITHUB_STEP_SUMMARY"
else
echo "changed=false" >> "$GITHUB_OUTPUT"
: > /tmp/lockfile-diff.md
fi
- name: Post or update PR comment
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
REPO: ${{ github.repository }}
PR: ${{ github.event.pull_request.number }}
CHANGED: ${{ steps.diff.outputs.changed }}
- name: Emit review_status
id: emit-status
run: |
set -euo pipefail
MARKER='<!-- hermes-lockfile-diff -->'
CHANGED="${{ steps.diff.outputs.changed }}"
STATUS="[]"
# Find our previous comment (paginated — busy PRs exceed one page).
EXISTING=$(gh api --paginate "repos/${REPO}/issues/${PR}/comments" \
--jq ".[] | select(.body | startswith(\"$MARKER\")) | .id" \
| head -1 || true)
if [ "$CHANGED" != "true" ]; then
if [ -n "$EXISTING" ]; then
# A previous push changed the lockfile but the latest one
# doesn't — update the comment rather than leave stale info.
printf '%s\n✅ package-lock.json changes from an earlier push have been reverted — locked versions now match the target branch.\n' "$MARKER" > /tmp/lockfile-diff.md
else
echo "No lockfile changes and no existing comment — nothing to do."
exit 0
fi
fi
if [ -n "$EXISTING" ]; then
echo "Updating existing comment ${EXISTING}"
gh api --method PATCH "repos/${REPO}/issues/comments/${EXISTING}" \
-F body=@/tmp/lockfile-diff.md > /dev/null \
|| echo "::warning::Could not update PR comment (expected for fork PRs — GITHUB_TOKEN is read-only)"
if [ "$CHANGED" = "true" ]; then
CONTENT=$(cat /tmp/lockfile-diff.md | python3 -c "import sys,json; print(json.dumps(sys.stdin.read()))")
STATUS="[{\"source\":\"lockfile-diff\",\"results\":[{\"kind\":\"action_required\",\"title\":\"package-lock.json\",\"summary\":\"Locked npm dependency versions changed.\",\"detail\":${CONTENT},\"how_to_fix\":\"Add the \`ci-reviewed\` label after verifying the version changes are expected.\"}]}"
else
echo "Creating new comment"
gh api "repos/${REPO}/issues/${PR}/comments" \
-F body=@/tmp/lockfile-diff.md > /dev/null \
|| echo "::warning::Could not post PR comment (expected for fork PRs — GITHUB_TOKEN is read-only)"
STATUS="[]"
fi
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
- name: Upload diff artifact
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: lockfile-diff
path: /tmp/lockfile-diff.md
retention-days: 1
overwrite: true
+82 -56
View File
@@ -14,14 +14,14 @@ name: OSV-Scanner
# code patterns in PR diffs) by covering the orthogonal "currently-pinned
# dep became known-vulnerable" case.
#
# Steps below are inlined from Google's officially-recommended reusable
# workflow (google/osv-scanner-action/.github/workflows/osv-scanner-reusable.yml),
# rather than called via `uses:` so we can set a `timeout-minutes` in the
# degenerate case where this job hangs.
# Uses Google's officially-recommended reusable workflow, pinned by SHA.
# Findings land in the repo's Security tab (Code Scanning > OSV-Scanner).
# fail-on-vuln is disabled so the job does not block merges on pre-existing
# vulnerabilities in pinned deps that we may need to patch deliberately.
#
# The reusable workflow can't emit custom outputs, so a wrapper job
# downloads the SARIF result and summarizes the vulnerability count into
# a review_status for the unified PR comment.
on:
workflow_call:
@@ -40,62 +40,88 @@ permissions:
jobs:
scan:
name: Scan lockfiles
uses: google/osv-scanner-action/.github/workflows/osv-scanner-reusable.yml@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8
with:
# Scan explicit lockfiles rather than recursing, so we only look at
# the three sources of truth and skip vendored / test / worktree dirs.
scan-args: |-
--lockfile=uv.lock
--lockfile=package-lock.json
--lockfile=website/package-lock.json
# The upstream reusable workflow uploads this exact file under its
# fixed artifact name, which the wrapper downloads below.
results-file-name: osv-results.sarif
fail-on-vuln: false
emit-status:
name: Emit review status
runs-on: ubuntu-latest
needs: scan
if: always()
outputs:
review_status: ${{ steps.emit.outputs.review_status }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: 'Run scanner'
uses: google/osv-scanner-action/osv-scanner-action@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8
with:
# Scan explicit lockfiles rather than recursing, so we only look at
# the three sources of truth and skip vendored / test / worktree dirs.
scan-args: |-
--output=results.json
--format=json
--lockfile=uv.lock
--lockfile=package-lock.json
--lockfile=website/package-lock.json
continue-on-error: true
- name: 'Run osv-scanner-reporter'
uses: google/osv-scanner-action/osv-reporter-action@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8
with:
scan-args: |-
--output=results.sarif
--new=results.json
--gh-annotations=false
--fail-on-vuln=false
# Upload the results as artifacts (optional). Commenting out will disable uploads of run results in SARIF
# format to the repository Actions tab.
- name: 'Upload artifact'
id: 'upload_artifact'
if: ${{ !cancelled() }}
uses: actions/upload-artifact@bbbca2ddaa5d8feaa63e36b76fdaad77386f024f # v7.0.0
- name: Download SARIF result
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
with:
name: OSV Scanner SARIF file
path: results.sarif
retention-days: 5
path: /tmp/osv-results
continue-on-error: true
# Upload the results to GitHub's code scanning dashboard.
- name: 'Upload to code-scanning'
if: ${{ !cancelled() }}
uses: github/codeql-action/upload-sarif@cdefb33c0f6224e58673d9004f47f7cb3e328b89 # v4.31.10
with:
sarif_file: results.sarif
- name: 'Print Code Scanning URL'
if: ${{ !cancelled() }}
- name: Emit review_status
id: emit
run: |
echo "View the OSV-Scanner results in the 'Security' tab, using the following link:"
echo "${{ github.server_url }}/${{ github.repository }}/security/code-scanning?query=is%3Aopen+branch%3A${GITHUB_REF_NAME}+tool%3Aosv-scanner"
env:
GITHUB_REF_NAME: ${{ github.ref_name }}
set -euo pipefail
STATUS="[]"
- name: 'Error troubleshooter'
if: ${{ always() && steps.upload_artifact.outcome == 'failure' }}
run: |
echo "::error::Artifact upload failed. This is most likely caused by a error during scanning earlier in the workflow."
exit 1
if [ -f /tmp/osv-results/osv-results.sarif ]; then
# Count vulnerabilities from the SARIF file
VULN_COUNT=$(python3 -c "
import json, sys
try:
with open('/tmp/osv-results/osv-results.sarif') as f:
data = json.load(f)
count = 0
vulns = []
for run in data.get('runs', []):
for result in run.get('results', []):
count += 1
rule_id = result.get('ruleId', 'unknown')
message = result.get('message', {}).get('text', '')
loc = result.get('locations', [{}])[0].get('physicalLocation', {}).get('artifactLocation', {}).get('uri', '')
vulns.append(f'- {rule_id} in {loc}: {message}')
print(count)
if vulns:
print('\n'.join(vulns[:20]), file=sys.stderr)
except Exception:
print(0)
")
VULN_DETAIL=""
if [ "$VULN_COUNT" -gt 0 ] 2>/dev/null; then
VULN_PLURAL=$([ "$VULN_COUNT" -eq 1 ] && echo "y" || echo "ies")
VULN_DETAIL=$(python3 -c "
import json, sys
try:
with open('/tmp/osv-results/osv-results.sarif') as f:
data = json.load(f)
vulns = []
for run in data.get('runs', []):
for result in run.get('results', []):
rule_id = result.get('ruleId', 'unknown')
loc = result.get('locations', [{}])[0].get('physicalLocation', {}).get('artifactLocation', {}).get('uri', '')
vulns.append(f'- {rule_id} in {loc}')
print(json.dumps('\n'.join(vulns[:20])))
except Exception:
print(json.dumps(''))
")
STATUS="[{\"source\":\"osv scan\",\"results\":[{\"kind\":\"warning\",\"title\":\"OSV vulnerability scan\",\"summary\":\"${VULN_COUNT} known vulnerabilit${VULN_PLURAL} found in pinned dependencies.\",\"detail\":${VULN_DETAIL},\"how_to_fix\":\"Review the findings in the [Security tab](../../security/code-scanning). Update the affected dependencies if a patched version is available.\"}]}]"
else
STATUS="[]"
fi
fi
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
@@ -0,0 +1,71 @@
name: Publish E2E evidence
# This runs only from the default branch after CI completes. It intentionally
# checks out main, never the PR ref, and treats the downloaded artifact as
# untrusted input before uploading validated GitHub attachments.
on:
workflow_run:
workflows: [CI]
types: [completed]
permissions:
actions: read
contents: read
pull-requests: write
concurrency:
group: publish-e2e-evidence-${{ github.event.workflow_run.id }}
cancel-in-progress: false
jobs:
publish:
name: Publish inline E2E evidence
if: github.event.workflow_run.event == 'pull_request'
runs-on: ubuntu-latest
timeout-minutes: 10
environment: gh-image
steps:
- name: Check out trusted publisher
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
ref: ${{ github.event.repository.default_branch }}
persist-credentials: false
# v1.2.0 resolves to 44f4b93ecbbe22de6c45fa2f62f519aee564ca8c.
- name: Install gh-image
env:
GH_TOKEN: ${{ github.token }}
run: gh extension install drogers0/gh-image --pin v1.2.0
- name: Download and attach evidence
env:
GH_TOKEN: ${{ github.token }}
GITHUB_TOKEN: ${{ github.token }}
GH_SESSION_TOKEN: ${{ secrets.GH_IMAGE_SESSION_TOKEN }}
SOURCE_REPO: ${{ github.repository }}
SOURCE_RUN_ID: ${{ github.event.workflow_run.id }}
run: |
set -euo pipefail
PR_NUMBER=$(gh api "repos/$SOURCE_REPO/actions/runs/$SOURCE_RUN_ID" --jq '.pull_requests[0].number // empty')
if [ -z "$PR_NUMBER" ]; then
echo "No pull request is associated with CI run $SOURCE_RUN_ID."
exit 0
fi
ARTIFACT_NAME=$(gh api "repos/$SOURCE_REPO/actions/runs/$SOURCE_RUN_ID/artifacts" \
--jq '.artifacts[] | select(.expired == false and (.name | startswith("e2e-evidence-"))) | .name' \
| python3 -c 'import sys; print(next(iter(sys.stdin), "").strip())')
if [ -z "$ARTIFACT_NAME" ]; then
echo "No E2E evidence artifact was produced for CI run $SOURCE_RUN_ID."
exit 0
fi
EVIDENCE_DIR="$RUNNER_TEMP/e2e-evidence"
mkdir -p "$EVIDENCE_DIR"
gh run download "$SOURCE_RUN_ID" --repo "$SOURCE_REPO" --name "$ARTIFACT_NAME" --dir "$EVIDENCE_DIR"
python3 scripts/ci/publish_e2e_evidence.py \
--evidence-dir "$EVIDENCE_DIR" \
--source-repo "$SOURCE_REPO" \
--pr-number "$PR_NUMBER"
+109
View File
@@ -0,0 +1,109 @@
name: Review labels
# Require explicit maintainer review when CI-sensitive files or the MCP
# catalog change. Previously this was split across two jobs in two
# workflows: ``ci-review`` in lint.yml (gated on ``ci_review``) and
# ``mcp-catalog-review`` in supply-chain-audit.yml (gated on
# ``mcp_catalog``). Both checked for their own label.
#
# Now consolidated: a single ``ci-reviewed`` label covers both. The
# comment sections tell the reviewer exactly what to verify per area,
# so one label is enough — the human reads the comment, not the label
# name.
#
# Outputs:
# ci_reviewed — "true" / "false" / "" (empty when neither lane ran)
# review_status — JSON array of status objects consumed by the review
# comment assembler. See scripts/ci/emit_review_status.py.
on:
workflow_call:
inputs:
ci_review:
description: Whether CI-sensitive files (eslint config, workflows, actions) changed.
type: boolean
default: false
ci_review_files:
description: JSON list of CI-sensitive files changed by the pull request.
type: string
default: '[]'
mcp_catalog:
description: Whether the MCP catalog / installer changed.
type: boolean
default: false
supply_chain:
description: Whether the critical supply-chain scan found a risk requiring review.
type: boolean
default: false
outputs:
ci_reviewed:
description: Whether the ci-reviewed label is present. Empty when neither input was true.
value: ${{ jobs.check.outputs.ci_reviewed }}
review_status:
description: JSON array of status objects for the review comment assembler.
value: ${{ jobs.check.outputs.review_status }}
permissions:
contents: read
pull-requests: read # read PR labels
jobs:
check:
name: Review label gate
if: inputs.ci_review || inputs.mcp_catalog || inputs.supply_chain
runs-on: ubuntu-latest
timeout-minutes: 2
outputs:
ci_reviewed: ${{ steps.label-check.outputs.ci_reviewed }}
review_status: ${{ steps.build-status.outputs.review_status }}
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Check ci-reviewed label
id: label-check
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
REPO: ${{ github.repository }}
run: |
set -euo pipefail
PR="${{ github.event.pull_request.number }}"
LABELS=$(gh pr view "$PR" --repo "$REPO" --json labels --jq '.labels[].name' || true)
if echo "$LABELS" | grep -Fxq 'ci-reviewed'; then
echo "ci-reviewed label present."
echo "ci_reviewed=true" >> "$GITHUB_OUTPUT"
else
echo "ci-reviewed label missing."
echo "ci_reviewed=false" >> "$GITHUB_OUTPUT"
fi
- name: Build review_status JSON
id: build-status
env:
CI_REVIEW: ${{ inputs.ci_review }}
CI_REVIEW_FILES: ${{ inputs.ci_review_files }}
MCP_CATALOG: ${{ inputs.mcp_catalog }}
SUPPLY_CHAIN: ${{ inputs.supply_chain }}
LABEL_PRESENT: ${{ steps.label-check.outputs.ci_reviewed }}
REPO_URL: ${{ github.server_url }}/${{ github.repository }}
BASE_SHA: ${{ github.event.pull_request.base.sha }}
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
run: |
set -euo pipefail
args=()
if [ "$CI_REVIEW" = "true" ]; then args+=(--ci-review); fi
args+=(--ci-review-files "$CI_REVIEW_FILES")
if [ "$MCP_CATALOG" = "true" ]; then args+=(--mcp-catalog); fi
if [ "$SUPPLY_CHAIN" = "true" ]; then args+=(--supply-chain); fi
if [ "$LABEL_PRESENT" = "true" ]; then args+=(--label-present); fi
python3 scripts/ci/emit_review_status.py "${args[@]}" \
--repo-url "$REPO_URL" --base-sha "$BASE_SHA" --head-sha "$HEAD_SHA" \
--output "$GITHUB_OUTPUT"
- name: Fail on missing label
if: steps.label-check.outputs.ci_reviewed != 'true'
run: |
echo "::error::CI-sensitive changes require the ci-reviewed label. Add the label and re-run this check."
exit 1
+10 -1
View File
@@ -21,6 +21,7 @@ jobs:
if: github.repository == 'NousResearch/hermes-agent'
runs-on: ubuntu-latest
timeout-minutes: 10
environment: trusted-automation
steps:
- name: Probe live index
id: probe
@@ -108,10 +109,18 @@ jobs:
echo "Summary: ${{ steps.probe.outputs.summary }}"
fi
- name: Get GitHub App token
if: steps.probe.outputs.status != 'ok'
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ vars.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- name: Open issue on degraded / failed probe
if: steps.probe.outputs.status != 'ok'
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
GH_TOKEN: ${{ steps.app-token.outputs.token }}
STATUS: ${{ steps.probe.outputs.status }}
DETAIL: ${{ steps.probe.outputs.detail }}
run: |
+17 -2
View File
@@ -21,9 +21,17 @@ jobs:
if: github.repository == 'NousResearch/hermes-agent'
runs-on: ubuntu-latest
timeout-minutes: 15
environment: trusted-automation
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Get GitHub App token
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ vars.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
with:
python-version: "3.11"
@@ -35,7 +43,7 @@ jobs:
- name: Build skills index
env:
GITHUB_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
run: python scripts/build_skills_index.py
- name: Upload index artifact
@@ -53,8 +61,15 @@ jobs:
if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest
timeout-minutes: 15
environment: trusted-automation
steps:
- name: Get GitHub App token
id: app-token
uses: ./.github/actions/get-app-token
with:
client-id: ${{ vars.APP_CLIENT_ID }}
private-key: ${{ secrets.APP_PRIVATE_KEY }}
- name: Trigger Deploy Site workflow
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
GH_TOKEN: ${{ steps.app-token.outputs.token }}
run: gh workflow run deploy-site.yml --repo ${{ github.repository }} -f skills_index_run_id=${{ github.run_id }}
+101 -79
View File
@@ -10,9 +10,18 @@ name: Supply Chain Audit
# advisory-only workflow instead.
#
# Path-gating is handled centrally by the ``ci.yml`` orchestrator's
# ``detect`` job. The orchestrator passes ``scan`` / ``deps`` /
# ``mcp_catalog`` booleans as inputs; this workflow's jobs gate on those
# inputs instead of re-computing the diff.
# ``detect`` job. The orchestrator passes ``scan`` / ``deps`` booleans as
# inputs; this workflow's jobs gate on those inputs instead of re-computing
# the diff. MCP catalog review was previously here but has moved to
# ``review-labels.yml`` so it can be rerun independently.
#
# Outputs:
# review_status — JSON array of status objects consumed by the review
# comment assembler (scripts/ci/assemble_review_comment.py).
# critical_findings — "true" when the narrow critical-pattern scan found
# something. The review-label gate consumes this and
# owns the action-required result, so adding
# ``ci-reviewed`` can heal the run on rerun.
on:
workflow_call:
@@ -29,10 +38,13 @@ on:
description: Whether pyproject.toml changed.
type: boolean
required: true
mcp_catalog:
description: Whether the MCP catalog / installer changed.
type: boolean
required: true
outputs:
review_status:
description: JSON array of review status objects for the review comment assembler.
value: ${{ jobs.aggregate.outputs.review_status }}
critical_findings:
description: Whether the critical-pattern scan found a risk requiring maintainer review.
value: ${{ jobs.aggregate.outputs.critical_findings }}
permissions:
pull-requests: write
@@ -44,6 +56,9 @@ jobs:
if: inputs.scan
runs-on: ubuntu-latest
timeout-minutes: 15
outputs:
review_status: ${{ steps.emit-status.outputs.review_status }}
critical_findings: ${{ steps.scan.outputs.found }}
steps:
- name: Checkout
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
@@ -53,7 +68,8 @@ jobs:
- name: Scan diff for critical patterns
id: scan
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
GH_TOKEN: ${{ github.token }}
CI_REVIEWED: ${{ contains(github.event.pull_request.labels.*.name, 'ci-reviewed') }}
run: |
set -euo pipefail
@@ -61,7 +77,7 @@ jobs:
HEAD="${{ github.event.pull_request.head.sha }}"
# Added lines only, excluding lockfiles.
# Three-dot diff (base...head) diffs from the merge base to HEAD,
# Three-point diff (base...head) diffs from the merge base to HEAD,
# so only changes introduced by this PR are included — not changes
# that landed on main after the PR branched off.
DIFF=$(git diff "$BASE"..."$HEAD" -- . ':!uv.lock' ':!*.lock' ':!package-lock.json' ':!yarn.lock' || true)
@@ -71,7 +87,7 @@ jobs:
# --- .pth files (auto-execute on Python startup) ---
# The exact mechanism used in the litellm supply chain attack:
# https://github.com/BerriAI/litellm/issues/24512
PTH_FILES=$(git diff --name-only "$BASE"..."$HEAD" | grep '\.pth$' || true)
PTH_FILES=$(git diff --diff-filter=d --name-only "$BASE"..."$HEAD" | grep '\.pth$' || true)
if [ -n "$PTH_FILES" ]; then
FINDINGS="${FINDINGS}
### 🚨 CRITICAL: .pth file added or modified
@@ -119,8 +135,11 @@ jobs:
# auto-loaded by the interpreter via site.py. Any nested file with the
# same name (e.g. hermes_cli/setup.py — the CLI setup wizard) is unrelated
# and produced false positives that trained reviewers to ignore the scanner.
SETUP_HITS=$(git diff --name-only "$BASE"..."$HEAD" | grep -E '^(setup\.py|setup\.cfg|sitecustomize\.py|usercustomize\.py|__init__\.pth)$' || true)
if [ -n "$SETUP_HITS" ]; then
SETUP_HITS=$(git diff --diff-filter=d --name-only "$BASE"..."$HEAD" | grep -E '^(setup\.py|setup\.cfg|sitecustomize\.py|usercustomize\.py|__init__\.pth)$' || true)
# A maintainer-applied ci-reviewed label records the manual review
# required for intentional changes to an install hook. The scanner
# still blocks every unreviewed addition or modification.
if [ -n "$SETUP_HITS" ] && [ "$CI_REVIEWED" != "true" ]; then
FINDINGS="${FINDINGS}
### 🚨 CRITICAL: Install-hook file added or modified
These files can execute code during package installation or interpreter startup.
@@ -139,33 +158,32 @@ jobs:
echo "found=false" >> "$GITHUB_OUTPUT"
fi
- name: Post critical finding comment
if: steps.scan.outputs.found == 'true'
- name: Emit review_status
id: emit-status
if: always()
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
FOUND: ${{ steps.scan.outputs.found }}
run: |
BODY="## 🚨 CRITICAL Supply Chain Risk Detected
python3 - <<'PYEOF'
import json, os
This PR contains a pattern that has been used in real supply chain attacks. A maintainer must review the flagged code carefully before merging.
# The review-label gate renders and blocks critical findings. Keep
# this scan a fact-finder so adding ci-reviewed can rerun the gate
# without requiring the scanner itself to fail again.
status = []
$(cat /tmp/findings.md)
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as f:
f.write(f"review_status={json.dumps(status)}\n")
PYEOF
---
*Scanner only fires on high-signal indicators: .pth files, base64+exec/eval combos, subprocess with encoded commands, or install-hook files. Low-signal warnings were removed intentionally — if you're seeing this comment, the finding is worth inspecting.*"
gh pr comment "${{ github.event.pull_request.number }}" --body "$BODY" || echo "::warning::Could not post PR comment (expected for fork PRs — GITHUB_TOKEN is read-only)"
- name: Fail on critical findings
if: steps.scan.outputs.found == 'true'
run: |
echo "::error::CRITICAL supply chain risk patterns detected in this PR. See the PR comment for details."
exit 1
dep-bounds:
name: Check PyPI dependency upper bounds
if: inputs.deps
runs-on: ubuntu-latest
timeout-minutes: 15
outputs:
review_status: ${{ steps.emit-status.outputs.review_status }}
steps:
- name: Checkout
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
@@ -188,7 +206,7 @@ jobs:
exit 0
fi
# Match PyPI dep specs that have >= but no < ceiling.
# Match PyPI dep specs that have >= and no < ceiling.
# Pattern: "package>=version" without a following ",<" bound.
# Excludes git+ URLs (which use commit SHAs) and comments.
UNBOUNDED=$(echo "$ADDED" | grep -oE '"[a-zA-Z0-9_-]+(\[[^\]]*\])?>=[ 0-9.]+"' | grep -v ',<' || true)
@@ -200,26 +218,36 @@ jobs:
echo "found=false" >> "$GITHUB_OUTPUT"
fi
- name: Post unbounded dep warning
if: steps.bounds.outputs.found == 'true'
- name: Emit review_status
id: emit-status
if: always()
env:
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
FOUND: ${{ steps.bounds.outputs.found }}
run: |
BODY="## ⚠️ Unbounded PyPI Dependency Detected
python3 - <<'PYEOF'
import json, os
This PR adds PyPI dependencies without a \`<next_major\` upper bound. Per our [supply chain policy](../blob/main/CONTRIBUTING.md#dependency-pinning-policy-supply-chain-hardening), all PyPI deps must be pinned as \`>=floor,<next_major\`.
found = os.environ.get("FOUND", "") == "true"
**Unbounded specs found:**
\`\`\`
$(cat /tmp/unbounded.txt)
\`\`\`
if found:
with open("/tmp/unbounded.txt", encoding="utf-8") as f:
detail = f.read()
status = [{
"source": "supply chain",
"results": [{
"kind": "action_required",
"title": "Unbounded PyPI dependencies",
"summary": "This PR adds PyPI dependencies without upper bounds.",
"detail": detail,
"how_to_fix": 'Add a `<next_major` upper bound, e.g. `"package>=1.2.0,<2"`. See CONTRIBUTING.md dependency pinning policy.'
}]
}]
else:
status = []
**Fix:** Add an upper bound, e.g. \`"package>=1.2.0,<2"\`
---
*See PR #2810 and CONTRIBUTING.md for the full policy rationale.*"
gh pr comment "${{ github.event.pull_request.number }}" --body "$BODY" || echo "::warning::Could not post PR comment (expected for fork PRs)"
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as f:
f.write(f"review_status={json.dumps(status)}\n")
PYEOF
- name: Fail on unbounded deps
if: steps.bounds.outputs.found == 'true'
@@ -227,45 +255,39 @@ jobs:
echo "::error::PyPI dependencies without upper bounds detected. Add <next_major ceiling per CONTRIBUTING.md policy."
exit 1
mcp-catalog-review:
name: MCP catalog security review
if: inputs.mcp_catalog
aggregate:
name: Aggregate review statuses
needs: [scan, dep-bounds]
if: always()
runs-on: ubuntu-latest
timeout-minutes: 15
outputs:
review_status: ${{ steps.merge.outputs.review_status }}
critical_findings: ${{ steps.merge.outputs.critical_findings }}
steps:
- name: Checkout
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
fetch-depth: 0
- name: Require explicit MCP catalog review label
- name: Merge review statuses
id: merge
env:
# Read-only label lookup. Use the built-in GITHUB_TOKEN (present and
# read-only on forks) so the gate works on fork PRs; fall back to it
# when AUTOFIX_BOT_PAT is empty. `|| true` degrades an API blip to
# "label absent" rather than hard-failing the step.
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
SCAN_STATUS: ${{ needs.scan.outputs.review_status }}
DEP_STATUS: ${{ needs.dep-bounds.outputs.review_status }}
CRITICAL_FINDINGS: ${{ needs.scan.outputs.critical_findings }}
run: |
set -euo pipefail
PR="${{ github.event.pull_request.number }}"
LABELS=$(gh pr view "$PR" --json labels --jq '.labels[].name' || true)
if echo "$LABELS" | grep -Fxq 'mcp-catalog-reviewed'; then
echo "MCP catalog review label present."
exit 0
fi
python3 - <<'PYEOF'
import json, os
BODY="## ⚠️ MCP catalog security review required
merged = []
for key in ("SCAN_STATUS", "DEP_STATUS"):
raw = os.environ.get(key, "")
if not raw:
continue
try:
data = json.loads(raw)
except (json.JSONDecodeError, TypeError):
continue
if isinstance(data, list):
merged.extend(data)
This PR changes the bundled MCP catalog or MCP catalog installer code. MCP entries can define local commands that users later install into \`mcp_servers\`, so this needs explicit maintainer review before merge.
A maintainer should verify:
- any new/changed \`optional-mcps/**/manifest.yaml\` command and args are expected,
- stdio transports do not use shell+egress/exfiltration payloads,
- git install refs are pinned and bootstrap commands are minimal,
- requested env vars/secrets match the upstream MCP's documented needs.
After review, add the \`mcp-catalog-reviewed\` label and re-run this check."
gh pr comment "$PR" --body "$BODY" || echo "::warning::Could not post PR comment (expected for fork PRs)"
echo "::error::MCP catalog changes require the mcp-catalog-reviewed label."
exit 1
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as f:
f.write(f"review_status={json.dumps(merged)}\n")
f.write("critical_findings=" + os.environ.get("CRITICAL_FINDINGS", "false") + "\n")
PYEOF
+26 -7
View File
@@ -74,6 +74,11 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pin the uv version: unpinned, setup-uv resolves "latest" by
# fetching a manifest from raw.githubusercontent.com on EVERY job —
# a transient fetch failure fails the whole job (2026-07-28 slice-5
# incident). Pinned, the binary downloads directly; no manifest hop.
version: "0.9.28"
# Persist uv's download/wheel cache (~/.cache/uv) across runs.
# Keyed on the dependency manifests, so the cache is reused until
# pyproject.toml or uv.lock changes. `uv sync` still runs every
@@ -92,9 +97,18 @@ jobs:
# fails if the lock is out of sync with pyproject.toml), giving a
# reproducible env. It also creates .venv itself, so no separate
# `uv venv` step is needed.
#
# The trailing extras beyond all/dev are the lazy-install features
# (tools/lazy_deps.py) that tests exercise for real: provider.anthropic,
# stt/tts.mistral, image.fal, terminal.modal, terminal.daytona,
# memory.hindsight, search.parallel. The hermetic test env forbids
# mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
# tests/conftest.py), so the SDKs those tests need must be in the
# venv up front — resolved from uv.lock like everything else, which
# also honors the exact supply-chain pins these extras carry.
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
# Optimized for CI: prunes pre-built wheels that are cheap to
@@ -188,6 +202,11 @@ jobs:
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pin the uv version: unpinned, setup-uv resolves "latest" by
# fetching a manifest from raw.githubusercontent.com on EVERY job —
# a transient fetch failure fails the whole job (2026-07-28 slice-5
# incident). Pinned, the binary downloads directly; no manifest hop.
version: "0.9.28"
# Persist uv's download/wheel cache (~/.cache/uv) across runs.
# Keyed on the dependency manifests, so the cache is reused until
# pyproject.toml or uv.lock changes. `uv sync` still runs every
@@ -206,20 +225,20 @@ jobs:
# fails if the lock is out of sync with pyproject.toml), giving a
# reproducible env. It also creates .venv itself, so no separate
# `uv venv` step is needed.
#
# Same extras as the test job's sync above: the hermetic test env
# forbids mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
# tests/conftest.py), so lazy-install SDKs exercised by tests must be
# in the venv up front.
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra all --extra dev
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
- name: Minimize uv cache
# Optimized for CI: prunes pre-built wheels that are cheap to
# re-download, keeping the persisted cache small and fast to restore.
run: uv cache prune --ci
- name: Packaged-wheel i18n smoke test
run: |
source .venv/bin/activate
python -m pytest -m integration tests/test_wheel_locales_e2e.py -v
- name: Run e2e tests
run: |
source .venv/bin/activate
-181
View File
@@ -1,181 +0,0 @@
name: Publish to PyPI
# Triggered by CalVer tag pushes from scripts/release.py (e.g. v2026.5.15)
# Can also be triggered manually from the Actions tab as an escape hatch.
on:
push:
tags:
- "v20*" # CalVer tags: v2026.5.15, v2026.5.15.2, etc.
workflow_dispatch:
inputs:
confirm_tag:
description: "Tag to publish (e.g. v2026.5.15). Must already exist."
required: true
type: string
# Restrict default token to read-only; each job escalates as needed.
permissions:
contents: read
# Prevent overlapping publishes (e.g. two same-day tags pushed quickly).
concurrency:
group: pypi-publish
cancel-in-progress: false
jobs:
build:
name: Build distribution 📦
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
# On workflow_dispatch, check out the confirmed tag.
ref: ${{ inputs.confirm_tag || github.ref }}
fetch-tags: true
- name: Validate tag exists
if: github.event_name == 'workflow_dispatch'
run: |
if ! git tag -l "${{ inputs.confirm_tag }}" | grep -q .; then
echo "::error::Tag '${{ inputs.confirm_tag }}' does not exist in the repo"
exit 1
fi
- name: Set up Python
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
with:
python-version: "3.13"
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
- name: Set up Node.js
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: "22"
- name: Build web dashboard
uses: ./.github/actions/retry
with:
command: npm ci
working-directory: web
- name: Compile web dashboard
run: npm run build
working-directory: web
- name: Build TUI bundle
uses: ./.github/actions/retry
with:
command: npm ci
working-directory: ui-tui
- name: Compile TUI bundle
run: npm run build
working-directory: ui-tui
- name: Bundle TUI into hermes_cli
run: |
mkdir -p hermes_cli/tui_dist
cp ui-tui/dist/entry.js hermes_cli/tui_dist/entry.js
- name: Verify frontend assets exist
run: |
test -f hermes_cli/web_dist/index.html || { echo "ERROR: web_dist not built"; exit 1; }
test -f hermes_cli/tui_dist/entry.js || { echo "ERROR: tui_dist not built"; exit 1; }
- name: Bundle install scripts into wheel
run: |
mkdir -p hermes_cli/scripts
cp scripts/install.sh hermes_cli/scripts/install.sh
cp scripts/install.ps1 hermes_cli/scripts/install.ps1
- name: Build wheel and sdist
run: uv build --sdist --wheel
- name: Upload distribution artifacts
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: python-package-distributions
path: dist/
publish:
name: Publish to PyPI
needs: build
runs-on: ubuntu-latest
timeout-minutes: 30
environment:
name: pypi
url: https://pypi.org/p/hermes-agent
permissions:
id-token: write # OIDC trusted publishing
steps:
- name: Download distribution artifacts
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
with:
name: python-package-distributions
path: dist/
- name: Publish to PyPI
uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # v1.14.0
with:
skip-existing: true
sign:
name: Sign and attach to GitHub Release
# Only runs on tag pushes — release.py creates the GitHub Release,
# and workflow_dispatch won't have a matching release to attach to.
if: startsWith(github.ref, 'refs/tags/')
needs: publish
runs-on: ubuntu-latest
timeout-minutes: 30
permissions:
contents: write # attach assets to the existing release
id-token: write # sigstore signing
steps:
- name: Download distribution artifacts
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
with:
name: python-package-distributions
path: dist/
- name: Wait for GitHub Release to exist
env:
GITHUB_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
# release.py creates the GitHub Release after pushing the tag,
# but this workflow starts from the tag push — wait for it.
run: |
for i in $(seq 1 30); do
if gh release view "$GITHUB_REF_NAME" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then
echo "Release $GITHUB_REF_NAME found"
exit 0
fi
echo "Waiting for release... ($i/30)"
sleep 10
done
echo "::warning::Release $GITHUB_REF_NAME not found after 5 minutes — skipping signature upload"
echo "skip_sign=true" >> "$GITHUB_ENV"
- name: Sign with Sigstore
if: env.skip_sign != 'true'
uses: sigstore/gh-action-sigstore-python@04cffa1d795717b140764e8b640de88853c92acc # v3.3.0
with:
inputs: >-
./dist/*.tar.gz
./dist/*.whl
- name: Attach signed artifacts to GitHub Release
if: env.skip_sign != 'true'
env:
GITHUB_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
# release.py already created the GitHub Release — just upload
# the Sigstore signatures alongside the existing assets.
run: >-
gh release upload
"$GITHUB_REF_NAME" dist/*.sigstore.json
--repo "$GITHUB_REPOSITORY"
--clobber
+16
View File
@@ -45,6 +45,10 @@ name: uv.lock check
on:
workflow_call:
outputs:
review_status:
description: "JSON review status for the review-status aggregator"
value: ${{ jobs.check.outputs.review_status }}
permissions:
contents: read
@@ -58,12 +62,19 @@ jobs:
name: uv lock --check
runs-on: ubuntu-latest
timeout-minutes: 5
outputs:
review_status: ${{ steps.verify.outputs.review_status }}
steps:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Install uv
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
with:
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
# raw.githubusercontent.com every job; transient fetch failures
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
version: "0.9.28"
# `uv lock --check` re-resolves the project from pyproject.toml and
# compares the result to uv.lock, exiting non-zero if they disagree.
@@ -73,6 +84,7 @@ jobs:
# of this file) — failures often mean "your branch is behind main,
# rebase and regenerate uv.lock."
- name: Verify uv.lock is up-to-date
id: verify
run: |
# uv lock --check re-resolves against PyPI (network). Retry so a
# registry blip doesn't read as "lockfile stale". A genuinely stale
@@ -117,5 +129,9 @@ jobs:
on `main` post-merge.
EOF
echo "::error title=uv.lock out of sync::Run \`uv lock\` locally and commit the result. If on a PR, sync with main first."
review_status='[{"source":"uv.lock check","results":[{"kind":"action_required","title":"uv.lock out of sync","summary":"uv.lock is out of sync with pyproject.toml.","how_to_fix":"Run `uv lock` locally and commit the result. If on a PR, sync with main first:\n```\ngit fetch origin main\ngit rebase origin/main\nuv lock\ngit add uv.lock\ngit commit -m \"chore: refresh uv.lock\"\n```\n"}]}]'
echo "review_status=${review_status}" >> "$GITHUB_OUTPUT"
exit 1
fi
review_status='[]'
echo "review_status=${review_status}" >> "$GITHUB_OUTPUT"
+37 -1
View File
@@ -1,9 +1,13 @@
.DS_Store
/venv/
/venv.old/
/venv.stale.runtime-*/
/.hermes-runtime/
/_pycache/
*.pyc*
__pycache__/
act/
.act-sandbox-agent.*
.venv/
.venv
.vscode/
@@ -42,7 +46,10 @@ run_datagen_sonnet.sh
source-data/*
run_datagen_megascience_glm4-6.sh
data/*
node_modules/
# No trailing slash: also matches node_modules SYMLINKS (worktrees often
# symlink node_modules to the main checkout; the dir-only pattern let one
# slip into a commit and break `npm ci` on CI with ENOTDIR).
node_modules
browser-use/
agent-browser/
# Private keys
@@ -54,6 +61,10 @@ __pycache__/
hermes_agent.egg-info/
wandb/
testlogs
playwright-report/
test-results/
# Playwright visual regression baselines — cached from main in CI, not committed
*-snapshots/
# CLI config (may contain sensitive SSH paths)
cli-config.yaml
@@ -66,6 +77,8 @@ environments/benchmarks/evals/
# Web UI build output
hermes_cli/web_dist/
# Cross-process web UI build lock (flock target, always empty)
.web_ui_build.lock
apps/desktop/build/
apps/desktop/dist/
@@ -139,6 +152,16 @@ docs/superpowers/*
.update-incomplete
.update-incomplete.lock
# Checkout fingerprint the __pycache__ tree was last validated against
# (launch-time stale-bytecode sweep). Runtime state, never a code change.
.bytecode-fingerprint
.bytecode-fingerprint.tmp
# Installer-written method stamp in the managed checkout root (scripts/install.sh).
# Runtime metadata only — never a code change. Ignore so `git status` stays clean
# and `hermes update`'s untracked autostash does not treat it as a local edit (#66189 / #54855).
/.install_method
# Tool Search live-test harness output — non-deterministic model transcripts,
# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo.
scripts/out/
@@ -157,4 +180,17 @@ apps/desktop/demo/
# image-provider (fal.media) URL — they are NEVER committed to the repo. The
# PR body is the archive. See the hermes-agent-dev skill's
# pr-infographic-workflow reference (storage rule + lapse #8 / #COMMIT-1).
#
# Spelling variants are listed because a single `infographic/` pattern was
# sidestepped by an `infograficos/` directory (#70552). .gitignore is only
# the first line of defence and cannot stop `git add -f` at all — the
# infographic-check CI job is what actually enforces this.
infographic/
infographics/
infograficos/
infografico/
native/fts5_cjk/*.so
# Runtime marker written by hermes update when a lazy dependency refresh is
# interrupted; consumed by launch-time recovery. Never commit it (was tracked
# by accident via 3a69e34702, removed in the #72002 salvage).
.lazy-refresh-incomplete
+146
View File
@@ -0,0 +1,146 @@
# Message reactions (desktop tapbacks)
Two-way emoji reactions on individual messages in the desktop transcript: the
user reacts to any message, the agent reacts to a user message, and both sides
read the other's reactions as conversational signal.
## What already exists
Hermes already models reactions on the **platform** side — the desktop is the
only surface without them.
| Surface | Reaction support | Where |
|---|---|---|
| Agent → platform message | `send_message(action="react"/"unreact")` | `tools/send_message_tool.py:266` `_handle_react()` |
| Photon / iMessage | tapbacks in + out, routed only for messages we sent | `plugins/platforms/photon/adapter.py:1240-1283` |
| Telegram | `setMessageReaction`, config-gated | `plugins/platforms/telegram/adapter.py:9669+` |
| Slack / Matrix / Feishu / Discord | inbound reaction events → hooks | `gateway/run.py:4688` `_handle_reaction_event()` → `HookRegistry.emit("reaction:added")` |
| Adapter contract | `add_reaction()` / `remove_reaction()` coroutines, `set_reaction_handler()` | `gateway/platforms/base.py:3330` |
| Core "affection" detector | regex on user text → `vibe`, drives CLI pet / TUI heart / desktop hearts | `agent/reactions.py`, `agent/turn_context.py:592-604` |
Two things follow from that table:
1. **The agent-facing verb already exists.** `send_message(action="react")` is
the established shape. A desktop reaction should extend that tool, not add a
new core tool — every new tool ships on every API call (AGENTS.md footprint
ladder).
2. **The inbound convention already exists.** Photon turns a tapback into a
normal message event with `reply_to_message_id` + `reply_to_is_own_message`,
and the gateway prefixes `[Replying to your previous message: "…"]`
(`gateway/run.py:13125-13132`). Desktop reactions should read the same way to
the model.
Nothing exists on the desktop side: `grep -ri reaction` across `apps/desktop`
finds only the pet-overlay hearts.
## Prior art
**iOS Tapback** ([Apple](https://support.apple.com/guide/iphone/react-with-tapbacks-iph018d3c336/ios)):
double-tap or touch-and-hold a message → floating pill above the bubble with
heart / thumbs-up / thumbs-down / haha / ‼️ / ❓, swipe left for suggested emoji
and stickers, or tap the emoji button for the full keyboard. **One tapback per
message per person** — tapping the same one again removes it, tapping a
different one replaces it. Multiple people's tapbacks stack on the badge.
**Platform data models** converge on the same shape:
| Platform | Model | Add / remove |
|---|---|---|
| Slack | `{name, count, users[]}` | [`reactions.add`](https://docs.slack.dev/reference/methods/reactions.add) / `reactions.remove`, emits `reaction_added` |
| Discord | `{emoji, count, me}` on the message object | `PUT`/`DELETE .../reactions/{emoji}/@me` |
| Telegram | `reaction: [{type:"emoji", emoji:"👍"}]` — replaces the whole set | `setMessageReaction`, `is_big` for the big animation |
Telegram's "set the whole array" is the closest match to iOS semantics and the
simplest thing to persist.
**assistant-ui has no reaction primitive.** `@assistant-ui/react` 0.14.24 (MIT,
vendored at `apps/desktop/node_modules`): zero hits for "reaction" in `core/src`,
`react/src`, `dist/`, or the 2.2 MB `llms-full.txt` docs dump. What exists is a
hard-coded binary `FeedbackAdapter` (`"positive" | "negative"`,
`core/src/adapters/feedback.ts`) that throws when unconfigured and only writes
back onto assistant messages. Not usable for emoji, not usable on user messages.
**But `metadata.custom` is the supported extension channel** and this repo
already uses it: `ThreadUserMessage`/`ThreadAssistantMessage`/`ThreadSystemMessage`
all carry `metadata.custom: Record<string, unknown>` (`core/src/types/message.ts:319-366`),
and `chat-runtime.ts:397` already ships `custom: { attachmentRefs }` through it.
**Emoji picker survey** (npm week of 2026-07-22, sizes measured from the
published ESM entry):
| Library | License | Weekly DL | gzip | Headless | Latest |
|---|---|---|---|---|---|
| **frimousse** | MIT | 573k | **8.5 kB** | ✅ fully unstyled, composable parts | 0.3.0 · 2025-07-15 |
| emoji-picker-react | MIT | 1.31M | 87 kB | ❌ own CSS-in-JS (flairup) | 4.19.1 · 2026-04-27 |
| emoji-mart | MIT | 2.22M | ~120 kB w/ data | ❌ Preact + shadow styling | 5.6.0 · **2024-04-25**, 217 open issues |
| emoji-picker-element | Apache-2.0 | 183k | — | ❌ Web Component / Shadow DOM | 1.29.1 · 2026-03-01 |
No picker is currently a dependency (only `emoji-regex`, transitive). Already
paid for and reusable: `radix-ui` (Popover), `motion`, `@tanstack/react-virtual`,
Tailwind v4.
## Recommendation
**Hand-roll the tapback pill; add frimousse only behind the "+".** Six fixed
emoji in a pill is ~40 lines of JSX against existing tokens — pulling 87 kB of
`emoji-picker-react` to render six buttons, plus a CSS engine that fights
`DESIGN.md`, is backwards. frimousse is headless, dependency-free, 10× smaller,
and exposes `emojibaseUrl` so the data can be bundled as a Vite asset instead of
hitting jsDelivr (Electron must work offline).
### Data model
One reaction per author per message, Telegram-style whole-set replacement:
```ts
type MessageReaction = { emoji: string; author: 'user' | 'agent'; at: number }
```
Persisted in the existing `messages.display_metadata` JSON column
(`hermes_state_common.py:215`) — no new table. It already survives insert,
compaction, and every read projection, and
`set_latest_matching_message_display_kind()` (`hermes_state.py:5292`) is the
precedent for stamping metadata onto an already-persisted row.
### Model context
Reactions must reach the model **without breaking prompt caching**. The
`api_messages` build loop strips `display_metadata` from every outgoing copy
(`agent/conversation_loop.py:1443-1446`) precisely so display state never
becomes a provider field. Two candidate paths:
| Path | Cache impact | Notes |
|---|---|---|
| Rewrite the reacted-to message's content to carry the annotation | **Breaks the cached prefix** — mutates past context | Rejected. AGENTS.md: prompt caching is sacred. |
| Deliver the reaction as the *next* turn's leading annotation, mirroring photon | Prefix untouched; only the new turn carries it | Matches `[Replying to your previous message: "…"]` (`gateway/run.py:13125`), which the agent already understands |
The second is the same trick the platform adapters already use, so the model
sees a familiar shape and no existing conversation is rewritten.
### Attach points
| Concern | File | Lines |
|---|---|---|
| Assistant hover bar | `apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx` | 134–175 |
| User hover cluster | `apps/desktop/src/components/assistant-ui/thread/user-message.tsx` | 296–336 |
| Callback threading (ref caveat 79–99) | `apps/desktop/src/components/assistant-ui/thread/index.tsx` | 109–133 |
| `metadata.custom` → runtime | `apps/desktop/src/lib/chat-runtime.ts` | 384–432 |
| RPC client ↔ server pattern | `sidebar/session-actions-menu.tsx:62-89` ↔ `tui_gateway/server.py:8322` | — |
| Persistence | `hermes_state_common.py:192-216`, `hermes_state.py:5292-5324` | — |
| Prompt injection / strip | `agent/conversation_loop.py` | 1430–1529 |
### Known gaps to solve first
- **No durable message id crosses the gateway RPC path.** `_history_to_messages()`
(`tui_gateway/server.py:6545`) builds `{"role", "text"}` and drops the id. The
REST path carries `messages.id` incidentally via `SELECT *` but TS
`SessionMessage` (`types/hermes.ts:513-533`) doesn't declare it. Renderer ids
are ephemeral and change shape between rehydrated (`<ts>-<i>-<role>`), live
(`assistant-<ms>`), and optimistic (`user-<ms>-<rand>`) messages. A reaction
needs a stable key — this is the first thing to fix.
- **WeakMap identity cache** in `apps/desktop/src/app/chat/runtime-repository.ts:26-66`
keys normalized `ThreadMessage` by `ChatMessage` identity. A reaction change
must produce a **new** `ChatMessage` object or the UI renders stale.
- **Rewind rewrites rows** (`replace_messages`), so anything keyed by row id
needs cascade handling — an argument for keeping reactions in
`display_metadata` on the row itself rather than a side table.
+6 -4
View File
@@ -325,7 +325,7 @@ class AIAgent:
provider: str = None,
api_mode: str = None, # "chat_completions" | "codex_responses" | ...
model: str = "", # empty → resolved from config/provider later
max_iterations: int = 90, # tool-calling iterations (shared with subagents)
max_iterations: int = 500, # tool-calling iterations (shared with subagents)
enabled_toolsets: list = None,
disabled_toolsets: list = None,
quiet_mode: bool = False,
@@ -998,7 +998,8 @@ Two shapes:
Roles:
- `role="leaf"` (default) — focused worker. Cannot call `delegate_task`,
`clarify`, `memory`, `send_message`, `execute_code`.
`clarify`, `memory`, `send_message`, `cronjob`. Retains `execute_code`
(programmatic tool calling).
- `role="orchestrator"` — retains `delegate_task` so it can spawn its
own workers. Gated by `delegation.orchestrator_enabled` (default true)
and bounded by `delegation.max_spawn_depth` (default 2).
@@ -1283,14 +1284,15 @@ def profile_env(tmp_path, monkeypatch):
### Python
**ALWAYS use `scripts/run_tests.sh`** — do not call `pytest` directly. The script enforces
hermetic environment parity with CI (unset credential vars, TZ=UTC, LANG=C.UTF-8,
`-n auto` xdist workers, in-tree subprocess-isolation plugin). Direct `pytest`
per-file subprocess isolation via `scripts/run_tests_parallel.py` — no xdist,
worker count auto-scaled from CPU count). Direct `pytest`
on a 16+ core developer machine with API keys set diverges from CI in ways
that have caused multiple "works locally, fails in CI" incidents (and the reverse).
```bash
scripts/run_tests.sh # full suite, CI-parity
scripts/run_tests.sh tests/gateway/ # one directory
scripts/run_tests.sh tests/agent/test_foo.py::test_x # one test
scripts/run_tests.sh tests/agent/test_foo.py -k test_x # one test (file + -k; the runner is file-granular)
scripts/run_tests.sh -v --tb=long # pass-through pytest flags
```
+3 -2
View File
@@ -201,7 +201,8 @@ ln -sf "$(pwd)/venv/bin/hermes" ~/.local/bin/hermes
### Run tests
```bash
# Preferred — matches CI (hermetic env, 4 xdist workers); see AGENTS.md
# Preferred — matches CI (hermetic `env -i`, per-file subprocess isolation
# via run_tests_parallel.py, worker count auto-scaled); see AGENTS.md
scripts/run_tests.sh
# Alternative (activate the venv first). The wrapper is still recommended
@@ -848,7 +849,7 @@ that touches the OS, assume *any* platform can hit your code path.
Tests that use POSIX-only syscalls need a skip marker. Common ones:
- Symlinks → `@pytest.mark.skipif(sys.platform == "win32", ...)`
- `0o600` file modes → `@pytest.mark.skipif(sys.platform.startswith("win"), ...)`
- `signal.SIGALRM` → Unix-only (see `tests/conftest.py::_enforce_test_timeout`)
- `signal.SIGALRM` → Unix-only (per-test timeouts no longer use it directly; see the win32 timeout-method shim in `tests/conftest.py::pytest_configure`)
- `os.setsid` / `os.fork` → Unix-only
- Live Winsock / Windows-specific regression tests →
`@pytest.mark.skipif(sys.platform != "win32", reason="Windows-specific regression")`
+81 -2
View File
@@ -1,3 +1,45 @@
# Debian 13 still ships SQLite 3.46.1, which contains the upstream WAL-reset
# corruption bug. Build a pinned shared library for the runtime image instead
# of relying on a distro backport that trixie does not currently provide.
# See #70480 and https://sqlite.org/wal.html#walresetbug.
FROM debian:13.4 AS sqlite_build
ARG SQLITE_AUTOCONF_VERSION=3530400
ARG SQLITE_SHA256=0e9483900e92cd5de8fd48d16bf9200145a61f7fd5be542a5ac81d8a9516eb9c
RUN apt-get -o Acquire::Retries=3 update && \
apt-get -o Acquire::Retries=3 install -y --no-install-recommends \
build-essential ca-certificates curl && \
rm -rf /var/lib/apt/lists/* && \
(curl -fsSL --retry 1 --retry-all-errors --connect-timeout 15 --max-time 60 \
-o /tmp/sqlite.tar.gz \
"https://sqlite.org/2026/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}.tar.gz" || \
curl -fsSL --retry 3 --retry-all-errors --connect-timeout 15 --max-time 120 \
-o /tmp/sqlite.tar.gz \
"https://sources.buildroot.net/sqlite/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}.tar.gz") && \
printf '%s %s\n' "${SQLITE_SHA256}" /tmp/sqlite.tar.gz > /tmp/sqlite.sha256 && \
sha256sum -c /tmp/sqlite.sha256 && \
tar -xzf /tmp/sqlite.tar.gz -C /tmp && \
cd "/tmp/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}" && \
CFLAGS="-O2 \
-DSQLITE_ENABLE_FTS3 \
-DSQLITE_ENABLE_FTS3_PARENTHESIS \
-DSQLITE_ENABLE_FTS4 \
-DSQLITE_ENABLE_FTS5 \
-DSQLITE_ENABLE_RTREE \
-DSQLITE_ENABLE_GEOPOLY \
-DSQLITE_ENABLE_COLUMN_METADATA \
-DSQLITE_ENABLE_UNLOCK_NOTIFY \
-DSQLITE_ENABLE_DBSTAT_VTAB \
-DSQLITE_ENABLE_DBPAGE_VTAB \
-DSQLITE_ENABLE_MATH_FUNCTIONS \
-DSQLITE_ENABLE_PREUPDATE_HOOK \
-DSQLITE_ENABLE_SESSION \
-DSQLITE_SECURE_DELETE \
-DSQLITE_THREADSAFE=1 \
-DSQLITE_MAX_VARIABLE_NUMBER=250000" \
./configure --prefix=/opt/sqlite-fixed --disable-static && \
make -j"$(nproc)" && \
make install
FROM ghcr.io/astral-sh/uv:0.11.6-python3.13-trixie@sha256:b3c543b6c4f23a5f2df22866bd7857e5d304b67a564f4feab6ac22044dde719b AS uv_source
# Node 22 LTS source stage. Debian trixie's bundled nodejs is pinned to 20.x
# which reached EOL in April 2026 — we copy node + npm + corepack from the
@@ -31,6 +73,23 @@ RUN apt-get -o Acquire::Retries=3 update && \
ca-certificates curl iputils-ping python3 python-is-python3 ripgrep ffmpeg gcc g++ make cmake python3-dev python3-venv libffi-dev libolm-dev procps git openssh-client docker-cli xz-utils && \
rm -rf /var/lib/apt/lists/*
# Prefer the fixed SQLite over Debian's vulnerable libsqlite3.so.0. Keep the
# public library name stable so both the system interpreter and the uv-created
# venv resolve the replacement without changing Python import paths.
COPY --from=sqlite_build /opt/sqlite-fixed/lib/libsqlite3.so.3.53.4 /usr/local/lib/
RUN ln -sf libsqlite3.so.3.53.4 /usr/local/lib/libsqlite3.so.0 && \
ln -sf libsqlite3.so.3.53.4 /usr/local/lib/libsqlite3.so && \
printf '/usr/local/lib\n' > /etc/ld.so.conf.d/000-sqlite-fixed.conf && \
ldconfig && \
python3 -c "import sqlite3, sys; \
v = sqlite3.sqlite_version_info; \
sys.exit(f'linked SQLite {sqlite3.sqlite_version} still has the WAL-reset bug') if v < (3, 51, 3) else None; \
db = sqlite3.connect(':memory:'); \
db.execute(\"CREATE VIRTUAL TABLE docs USING fts5(content, tokenize='trigram')\"); \
db.execute(\"INSERT INTO docs VALUES ('hermes')\"); \
sys.exit('SQLite FTS5 trigram self-test failed') if db.execute(\"SELECT count(*) FROM docs WHERE docs MATCH 'erm'\").fetchone()[0] != 1 else None; \
db.close()"
# ---------- s6-overlay install ----------
# s6-overlay provides supervision for the main hermes process, the dashboard,
# and per-profile gateways. /init becomes PID 1 below — see ENTRYPOINT.
@@ -141,6 +200,22 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
done && \
npm cache clean --force
# ---------- Photon iMessage sidecar deps (baked, NS-606) ----------
# The photon plugin's Node sidecar needs its own node_modules
# (spectrum-ts). The install tree is immutable at runtime, so a lazy
# `npm ci` on first connect would hit EROFS — bake the deps here instead
# (deterministic installs, NS-559). The patch script is copied alongside
# the manifests because package.json's postinstall runs it, which also
# means the spectrum-ts patch is applied at build time. Layer-cached:
# only re-runs when the sidecar manifests/patch change.
COPY plugins/platforms/photon/sidecar/package.json \
plugins/platforms/photon/sidecar/package-lock.json \
plugins/platforms/photon/sidecar/patch-spectrum-mixed-attachments.mjs \
plugins/platforms/photon/sidecar/
RUN cd plugins/platforms/photon/sidecar && \
npm ci --no-audit --fetch-retries=5 && \
npm cache clean --force
# ---------- Layer-cached Python dependency install ----------
# Copy only pyproject.toml + uv.lock so the Python dep resolve + wheel
# download + native-extension compile layer is cached unless those inputs
@@ -152,7 +227,7 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
# frontend stats the readme path during dep resolution, so we `touch` an
# empty placeholder — the real README is restored by `COPY . .` below.
#
# `uv sync --frozen --no-install-project --extra all --extra messaging`
# `uv sync --frozen --no-install-project --extra all --extra messaging --extra otlp`
# installs the deps reachable through the composite `[all]` extra
# (handpicked set intended for the production image — excludes `[dev]`),
# plus gateway messaging adapters that should work in the published image
@@ -165,6 +240,10 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
# so Docker users can use these providers without requiring runtime
# lazy-install access to PyPI (often blocked in containerized envs).
#
# The [otlp] extra contains the SDK/exporter imported by Hermes when Gateway
# Health export is enabled. Collector and observability-backend dependencies
# remain external and are not part of the Hermes production image.
#
# The hindsight memory provider's client (hindsight-client) is baked in
# for the same reason: it lazy-installs into /opt/hermes/.venv at first
# use, which lives inside the (immutable) image layer rather than the
@@ -182,7 +261,7 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
# The editable link is created after the source copy below.
COPY pyproject.toml uv.lock ./
RUN touch ./README.md
RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra anthropic --extra bedrock --extra azure-identity --extra hindsight --extra matrix
RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra otlp --extra anthropic --extra bedrock --extra azure-identity --extra hindsight --extra matrix
# ---------- Frontend build (cached independently from Python source) ----------
# Copy only the frontend source trees first so that Python-only changes don't
-13
View File
@@ -1,13 +0,0 @@
graft skills
graft optional-skills
graft optional-mcps
graft locales
# Bundled plugin manifests (plugin.yaml / plugin.yml). Without these the
# PluginManager scan (hermes_cli/plugins.py) finds zero plugins on installs
# built from the sdist (e.g. Homebrew, downstream packagers). package-data
# below covers the wheel; this covers the sdist. See #34034 / #28149.
recursive-include plugins plugin.yaml plugin.yml
# Gateway assets include images plus YAML catalogs such as status_phrases.yaml.
recursive-include gateway/assets *
global-exclude __pycache__
global-exclude *.py[cod]
+1 -1
View File
@@ -26,7 +26,7 @@ Use any model you want — [Nous Portal](https://portal.nousresearch.com), OpenR
<tr><td><b>A closed learning loop</b></td><td>Agent-curated memory with periodic nudges. Autonomous skill creation after complex tasks. Skills self-improve during use. FTS5 session search with LLM summarization for cross-session recall. <a href="https://github.com/plastic-labs/honcho">Honcho</a> dialectic user modeling. Compatible with the <a href="https://agentskills.io">agentskills.io</a> open standard.</td></tr>
<tr><td><b>Scheduled automations</b></td><td>Built-in cron scheduler with delivery to any platform. Daily reports, nightly backups, weekly audits — all in natural language, running unattended.</td></tr>
<tr><td><b>Delegates and parallelizes</b></td><td>Spawn isolated subagents for parallel workstreams. Write Python scripts that call tools via RPC, collapsing multi-step pipelines into zero-context-cost turns.</td></tr>
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Six terminal backends — local, Docker, SSH, Singularity, Modal, and Daytona. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Seven terminal backends — local, Docker, SSH, Singularity, Modal, Daytona, and Vercel Sandbox. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
<tr><td><b>Research-ready</b></td><td>Batch trajectory generation, trajectory compression for training the next generation of tool-calling models.</td></tr>
</table>
+7 -3
View File
@@ -173,9 +173,13 @@ modelo de autorización, pero las reglas a continuación se aplican uniformement
**Superficies en Hermes Agent:**
- **Adaptadores de plataforma del gateway.** Integraciones de mensajería en
`gateway/platforms/` (Telegram, Discord, Slack, email, SMS, etc.)
y adaptadores análogos incluidos como plugins.
- **Adaptadores de plataforma del gateway.** La mayoría de las integraciones
de mensajería se distribuyen como plugins empaquetados en
`plugins/platforms/<name>/` (Telegram, Discord, Slack, email, SMS, etc.).
Los tipos base compartidos y un conjunto menor de adaptadores
legacy/directos viven en `gateway/platforms/` (`base.py`, Signal, servidor
API, webhooks, …), con descubrimiento y carga diferida vía
`gateway/platform_registry.py`.
- **Superficies HTTP expuestas en red.** El adaptador del servidor API, el
plugin del dashboard, los endpoints HTTP del plugin kanban, y cualquier
otro plugin que vincule un socket de escucha.
+6 -3
View File
@@ -177,9 +177,12 @@ authorization model, but the rules below apply uniformly.
**Surfaces in Hermes Agent:**
- **Gateway platform adapters.** Messaging integrations in
`gateway/platforms/` (Telegram, Discord, Slack, email, SMS, etc.)
and analogous adapters shipped as plugins.
- **Gateway platform adapters.** Most messaging integrations ship as
bundled plugins under `plugins/platforms/<name>/` (Telegram, Discord,
Slack, email, SMS, etc.). Shared base types and a smaller set of
legacy/direct adapters live under `gateway/platforms/`
(`base.py`, Signal, API server, webhooks, …), with discovery and
deferred loading via `gateway/platform_registry.py`.
- **Network-exposed HTTP surfaces.** The API server adapter, the
dashboard plugin, the kanban plugin's HTTP endpoints, and any
other plugin that binds a listening socket.
+9 -6
View File
@@ -32,6 +32,7 @@ else:
import argparse
import asyncio
import logging
import os
import sys
from pathlib import Path
from hermes_constants import get_hermes_home
@@ -190,7 +191,7 @@ def _run_setup_browser(assume_yes: bool = False) -> int:
"""Bootstrap agent-browser + Chromium.
Routes through dep_ensure -> install.{sh,ps1} --ensure, sharing code
with ``hermes postinstall`` and the runtime lazy installer.
with the runtime lazy installer.
Returns 0 on success, 1 on failure.
"""
@@ -251,11 +252,13 @@ def main(argv: list[str] | None = None) -> None:
# MCP servers dynamically via asyncio.to_thread inside the event
# loop; that path is unaffected.) Moved from model_tools.py module
# scope to avoid freezing the gateway's loop on lazy import (#16856).
try:
from tools.mcp_tool import discover_mcp_tools
discover_mcp_tools()
except Exception:
logger.debug("MCP tool discovery failed at ACP startup", exc_info=True)
# Metadata-only hosts can opt out of unrelated global MCP startup.
if os.environ.get("HERMES_ACP_SKIP_CONFIGURED_MCP", "").strip() != "1":
try:
from tools.mcp_tool import discover_mcp_tools
discover_mcp_tools()
except Exception:
logger.debug("MCP tool discovery failed at ACP startup", exc_info=True)
agent = HermesACPAgent()
try:
+373 -58
View File
@@ -74,6 +74,10 @@ from acp_adapter.permissions import make_approval_callback
from acp_adapter.provenance import session_provenance_meta
from acp_adapter.session import SessionManager, SessionState, _expand_acp_enabled_toolsets
from acp_adapter.tools import build_tool_complete, build_tool_start
from agent.context_compressor import (
COMPRESSED_SUMMARY_METADATA_KEY,
ContextCompressor,
)
from tools.approval import (
reset_hermes_interactive_context,
set_hermes_interactive_context,
@@ -81,6 +85,110 @@ from tools.approval import (
logger = logging.getLogger(__name__)
def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, str]]]]:
"""Return ``(slug, label, [(model_id, description), ...])`` for named endpoints.
Covers both the v12 ``providers:`` mapping and the legacy
``custom_providers:`` list. These endpoints never appear in canonical
provider enumeration, so without this the ACP model selector hides every
named endpoint that the TUI ``/model`` picker already renders (#47039
implemented named-endpoint rows for the TUI surface only).
Model lists come from the entry's declared models (``default_model`` +
``models``), refreshed from the endpoint's live ``/models`` listing when a
credential is available and ``discover_models`` is not disabled. Declared
models are kept even when live discovery fails — some OpenAI-compatible
endpoints (e.g. Bedrock Mantle Responses) expose no ``/models`` route at
all yet serve the declared models fine.
Slugs use the ``custom:<name>`` shape that ``parse_model_input`` and
``resolve_runtime_provider`` already resolve, so encoded choice ids
(``custom:<name>:<model>``) round-trip through ``set_session_model``
unchanged.
"""
try:
from hermes_cli.config import (
get_compatible_custom_providers,
is_provider_enabled,
load_config,
)
from hermes_cli.models import fetch_api_models
from hermes_cli.providers import custom_provider_slug
except ImportError:
return []
try:
cfg = load_config()
entries = get_compatible_custom_providers(cfg)
except Exception:
logger.debug("Could not load named custom providers", exc_info=True)
return []
# ``get_compatible_custom_providers`` drops the ``enabled`` flag during
# normalization, so collect explicitly disabled provider keys from the
# raw config and skip their entries below.
disabled_keys: set[str] = set()
raw_providers = cfg.get("providers") if isinstance(cfg, dict) else None
if isinstance(raw_providers, dict):
for raw_key, raw_entry in raw_providers.items():
if isinstance(raw_entry, dict) and not is_provider_enabled(raw_entry):
disabled_keys.add(str(raw_key).strip().lower())
catalogs: list[tuple[str, str, list[tuple[str, str]]]] = []
for entry in entries:
if not isinstance(entry, dict):
continue
provider_key = str(entry.get("provider_key", "") or "").strip()
if provider_key.lower() in disabled_keys:
continue
name = str(entry.get("name", "") or "").strip()
base_url = str(entry.get("base_url", "") or "").strip()
if not name or not base_url:
continue
slug = custom_provider_slug(name, provider_key)
api_key = str(entry.get("api_key", "") or "").strip()
if not api_key:
key_env = str(entry.get("key_env", "") or "").strip()
api_key = os.environ.get(key_env, "").strip() if key_env else ""
declared: list[str] = []
default_model = str(entry.get("model", "") or "").strip()
if default_model:
declared.append(default_model)
models_cfg = entry.get("models")
if isinstance(models_cfg, dict):
for mid in models_cfg:
mid = str(mid or "").strip()
if mid and mid not in declared:
declared.append(mid)
if not api_key and not declared:
# No credential to discover with and nothing declared:
# not addressable from the selector.
continue
model_ids = list(declared)
discover = entry.get("discover_models", True)
if isinstance(discover, str):
discover = discover.lower() not in {"false", "no", "0"}
if discover and api_key:
try:
live = fetch_api_models(
api_key, base_url, api_mode=entry.get("api_mode")
)
except Exception:
live = None
if live:
model_ids = declared + [m for m in live if m not in declared]
if not model_ids:
continue
catalogs.append((slug, name, [(mid, "") for mid in model_ids]))
return catalogs
try:
from hermes_cli import __version__ as HERMES_VERSION
except Exception:
@@ -93,6 +201,13 @@ _executor = ThreadPoolExecutor(max_workers=4, thread_name_prefix="acp-agent")
# does not expose a client-side limit, so this is a fixed cap that clients
# paginate against using `cursor` / `next_cursor`.
_LIST_SESSIONS_PAGE_SIZE = 50
# Per-provider cap for the ACP model selector. ACP clients (Zed, Buzz) render
# the whole `availableModels` array in one dropdown, so an unbounded
# cross-provider catalog degrades the picker. Mirrors the cap the MoA picker
# already uses (`hermes_cli/moa_cmd.py`). This bounds each provider's row, not
# the total; aggregator providers stay intentionally uncapped inside the shared
# inventory, and the current model is always kept via the fallback insert below.
ACP_MAX_MODELS_PER_PROVIDER = 200
_MAX_ACP_RESOURCE_BYTES = 512 * 1024
_TEXT_RESOURCE_MIME_PREFIXES = ("text/",)
_TEXT_RESOURCE_MIME_TYPES = {
@@ -456,7 +571,7 @@ class HermesACPAgent(acp.Agent):
"tools": "List available tools",
"context": "Show conversation context info",
"reset": "Clear conversation history",
"compact": "Compress conversation context",
"compress": "Compress conversation context",
"steer": "Inject guidance into the currently running agent turn",
"queue": "Queue a prompt to run after the current turn finishes",
"version": "Show Hermes version",
@@ -485,7 +600,7 @@ class HermesACPAgent(acp.Agent):
"description": "Clear conversation history",
},
{
"name": "compact",
"name": "compress",
"description": "Compress conversation context",
},
{
@@ -581,46 +696,108 @@ class HermesACPAgent(acp.Agent):
return f"{raw_provider}:{raw_model}"
def _build_model_state(self, state: SessionState) -> SessionModelState | None:
"""Return the ACP model selector payload for editors like Zed."""
"""Return authenticated providers and their models for ACP clients.
The shared Hermes inventory is also used by ``hermes model``, the TUI,
and the dashboard. Keeping ACP on that substrate prevents its selector
from silently collapsing to the current provider's curated list.
"""
model = str(state.model or getattr(state.agent, "model", "") or "").strip()
provider = getattr(state.agent, "provider", None) or detect_provider() or "openrouter"
try:
from hermes_cli.models import curated_models_for_provider, normalize_provider, provider_label
from hermes_cli.inventory import build_models_payload, load_picker_context
from hermes_cli.models import normalize_provider, provider_label
normalized_provider = normalize_provider(provider)
provider_name = provider_label(normalized_provider)
context = load_picker_context().with_overrides(
current_provider=normalized_provider,
current_model=model,
current_base_url=str(getattr(state.agent, "base_url", "") or ""),
)
payload = build_models_payload(
context,
explicit_only=True,
include_unconfigured=False,
picker_hints=False,
canonical_order=True,
pricing=False,
capabilities=False,
refresh=False,
probe_custom_providers=False,
probe_current_custom_provider=False,
max_models=ACP_MAX_MODELS_PER_PROVIDER,
)
available_models: list[ModelInfo] = []
seen_ids: set[str] = set()
for model_id, description in curated_models_for_provider(normalized_provider):
rendered_model = str(model_id or "").strip()
if not rendered_model:
for row in payload.get("providers") or []:
row_provider = normalize_provider(str(row.get("slug") or "").strip())
if not row_provider:
continue
choice_id = self._encode_model_choice(normalized_provider, rendered_model)
if choice_id in seen_ids:
continue
desc_parts = [f"Provider: {provider_name}"]
if description:
desc_parts.append(str(description).strip())
if rendered_model == model:
desc_parts.append("current")
available_models.append(
ModelInfo(
model_id=choice_id,
name=rendered_model,
description=" • ".join(part for part in desc_parts if part),
)
provider_name = str(row.get("name") or "").strip() or provider_label(
row_provider
)
seen_ids.add(choice_id)
for model_entry in row.get("models") or []:
if isinstance(model_entry, dict):
rendered_model = str(
model_entry.get("id")
or model_entry.get("model")
or model_entry.get("name")
or ""
).strip()
else:
rendered_model = str(model_entry or "").strip()
if not rendered_model:
continue
choice_id = self._encode_model_choice(row_provider, rendered_model)
if choice_id in seen_ids:
continue
is_current = (
row_provider == normalized_provider and rendered_model == model
)
description = f"Provider: {provider_name}"
if is_current:
description += " • current"
available_models.append(
ModelInfo(
model_id=choice_id,
name=f"{provider_name} · {rendered_model}",
description=description,
)
)
seen_ids.add(choice_id)
# Named user-defined endpoints (providers: / custom_providers:)
# are invisible to canonical provider enumeration — append them
# so editor clients can select them like the TUI /model picker.
for named_slug, named_label, named_catalog in _named_custom_provider_catalogs():
for named_model, named_desc in named_catalog:
named_choice = self._encode_model_choice(named_slug, named_model)
if not named_choice or named_choice in seen_ids:
continue
named_parts = [f"Provider: {named_label}"]
if named_desc:
named_parts.append(str(named_desc).strip())
if named_slug == normalized_provider and named_model == model:
named_parts.append("current")
available_models.append(
ModelInfo(
model_id=named_choice,
name=named_model,
description=" • ".join(part for part in named_parts if part),
)
)
seen_ids.add(named_choice)
current_model_id = self._encode_model_choice(normalized_provider, model)
if current_model_id and current_model_id not in seen_ids:
provider_name = provider_label(normalized_provider)
available_models.insert(
0,
ModelInfo(
model_id=current_model_id,
name=model,
name=f"{provider_name} · {model}",
description=f"Provider: {provider_name} • current",
),
)
@@ -969,11 +1146,49 @@ class HermesACPAgent(acp.Agent):
return text
return ""
@staticmethod
def _history_summary_meta(message: dict[str, Any], text: str) -> dict[str, Any] | None:
"""Build the ``_meta`` payload for a replayed compaction summary.
Compaction summaries are persisted as ordinary history messages —
standalone handoffs under ``role="user"`` OR ``role="assistant"``
(the compressor picks whichever role keeps alternation valid), and
merge-into-tail messages where the summary is appended after the
first preserved tail message's real content. Without a wire flag,
ACP frontends render all of these as ordinary turns.
Two distinct keys under ``_meta.hermes`` (ACP's extensibility
channel), so clients cannot accidentally hide real content:
* ``compactionSummary: true`` — the entire chunk is the handoff
summary. Safe to restyle or collapse wholesale.
* ``containsCompactionSummary: true`` — a merged-tail message: real
preserved turn content followed by the summary. Clients may style
it, but collapsing the whole chunk would hide the preserved
content, hence the separate key.
Detection honors the in-process ``_compressed_summary`` flag and
falls back to content classification, so it also works for a
DB-reloaded session that lost the in-memory flag.
"""
kind = ContextCompressor.classify_summary_content(text)
if kind is None and message.get(COMPRESSED_SUMMARY_METADATA_KEY):
# Flagged in-process but content didn't classify (e.g. future
# prefix drift): treat as a standalone summary — the flag is only
# ever set on summary-bearing messages.
kind = "standalone"
if kind == "standalone":
return {"hermes": {"compactionSummary": True}}
if kind == "merged":
return {"hermes": {"containsCompactionSummary": True}}
return None
@staticmethod
def _history_message_update(
*,
role: str,
text: str,
field_meta: dict[str, Any] | None = None,
) -> UserMessageChunk | AgentMessageChunk | None:
"""Build an ACP history replay update for a user/assistant message."""
block = TextContentBlock(type="text", text=text)
@@ -981,11 +1196,13 @@ class HermesACPAgent(acp.Agent):
return UserMessageChunk(
session_update="user_message_chunk",
content=block,
field_meta=field_meta,
)
if role == "assistant":
return AgentMessageChunk(
session_update="agent_message_chunk",
content=block,
field_meta=field_meta,
)
return None
@@ -1056,7 +1273,11 @@ class HermesACPAgent(acp.Agent):
if role == "user":
text = self._history_message_text(message)
if text:
update = self._history_message_update(role=role, text=text)
update = self._history_message_update(
role=role,
text=text,
field_meta=self._history_summary_meta(message, text),
)
if update is not None and not await _send(update):
return
continue
@@ -1068,7 +1289,11 @@ class HermesACPAgent(acp.Agent):
text = self._history_message_text(message)
if text:
update = self._history_message_update(role=role, text=text)
update = self._history_message_update(
role=role,
text=text,
field_meta=self._history_summary_meta(message, text),
)
if update is not None and not await _send(update):
return
@@ -1218,12 +1443,19 @@ class HermesACPAgent(acp.Agent):
with state.runtime_lock:
if state.is_running and state.current_prompt_text:
state.interrupted_prompt_text = state.current_prompt_text
state.cancel_event.set()
try:
if getattr(state, "agent", None) and hasattr(state.agent, "interrupt"):
state.agent.interrupt()
except Exception:
logger.debug("Failed to interrupt ACP session %s", session_id, exc_info=True)
# Publish cancellation and hard-stop the agent before another
# prompt can acquire this lock and mistake the turn for
# redirectable work.
state.cancel_event.set()
try:
if getattr(state, "agent", None) and hasattr(state.agent, "interrupt"):
state.agent.interrupt()
except Exception:
logger.debug(
"Failed to interrupt ACP session %s",
session_id,
exc_info=True,
)
logger.info("Cancelled session %s", session_id)
async def fork_session(
@@ -1352,6 +1584,26 @@ class HermesACPAgent(acp.Agent):
elif rewrite_idle:
user_text = steer_text
user_content = steer_text
elif (
text_only_prompt
and isinstance(user_content, str)
and not user_text.startswith("/")
):
# Some ACP clients implement "stop and send" as two protocol calls:
# cancel the active prompt, then submit plain correction text. Keep
# the cancelled request attached so deictic follow-ups ("not that
# file") still have an explicit target.
interrupted_prompt = ""
with state.runtime_lock:
if not state.is_running and state.interrupted_prompt_text:
interrupted_prompt = state.interrupted_prompt_text
state.interrupted_prompt_text = ""
if interrupted_prompt:
user_text = (
f"{interrupted_prompt}\n\n"
f"User correction/guidance after interrupt: {user_text}"
)
user_content = user_text
# Intercept slash commands — handle locally without calling the LLM.
# Slash commands are text-only; if the client included images/resources,
@@ -1366,23 +1618,54 @@ class HermesACPAgent(acp.Agent):
await self._send_usage_update(state)
return PromptResponse(stop_reason="end_turn")
# If Zed sends another regular prompt while the same ACP session is
# still running, queue it instead of racing two AIAgent loops against
# the same state.history. /steer and /queue are handled above and can
# land immediately.
# If the client sends another regular text prompt while this ACP session
# is running, route it through the core active-turn redirect. Rich media
# and older runtimes retain the proven next-turn queue fallback.
redirected = False
queued_depth: int | None = None
with state.runtime_lock:
if state.is_running:
queued_text = user_text or "[Image attachment]"
state.queued_prompts.append(queued_text)
depth = len(state.queued_prompts)
if self._conn:
update = acp.update_agent_message_text(
f"Queued for the next turn. ({depth} queued)"
if (
text_only_prompt
and isinstance(user_content, str)
and getattr(
state.agent,
"_supports_active_turn_redirect",
False,
)
await self._conn.session_update(session_id, update)
return PromptResponse(stop_reason="end_turn")
state.is_running = True
state.current_prompt_text = user_text or "[Image attachment]"
is True
and hasattr(state.agent, "redirect")
):
try:
redirected = bool(state.agent.redirect(user_content))
except Exception:
logger.debug(
"ACP active-turn redirect failed for %s",
session_id,
exc_info=True,
)
if not redirected:
queued_text = user_text or "[Image attachment]"
state.queued_prompts.append(queued_text)
queued_depth = len(state.queued_prompts)
else:
state.is_running = True
state.current_prompt_text = user_text or "[Image attachment]"
if redirected:
if self._conn:
update = acp.update_agent_message_text(
"Redirected the active turn with your correction."
)
await self._conn.session_update(session_id, update)
return PromptResponse(stop_reason="end_turn")
if queued_depth is not None:
if self._conn:
update = acp.update_agent_message_text(
f"Queued for the next turn. ({queued_depth} queued)"
)
await self._conn.session_update(session_id, update)
return PromptResponse(stop_reason="end_turn")
logger.info("Prompt on session %s: %s", session_id, user_text[:100])
@@ -1478,7 +1761,16 @@ class HermesACPAgent(acp.Agent):
clear_session_vars,
set_session_vars,
)
session_tokens = set_session_vars(session_key=session_id)
# ``cwd`` pins the logical working directory for this context,
# which is what the system prompt's "Current working directory"
# line reports (agent/prompt_builder.py -> resolve_agent_cwd).
# Without it the prompt advertises the global Hermes workspace
# while the tools are rooted at the client's project, so the
# model emits absolute paths under ~/.hermes/workspace and the
# edit silently lands outside the editor's workspace.
session_tokens = set_session_vars(
session_key=session_id, cwd=state.cwd,
)
except Exception:
session_tokens = None
clear_session_vars = None # type: ignore[assignment]
@@ -1756,7 +2048,7 @@ class HermesACPAgent(acp.Agent):
"tools": self._cmd_tools,
"context": self._cmd_context,
"reset": self._cmd_reset,
"compact": self._cmd_compact,
"compress": self._cmd_compress,
"steer": self._cmd_steer,
"queue": self._cmd_queue,
"version": self._cmd_version,
@@ -1765,8 +2057,26 @@ class HermesACPAgent(acp.Agent):
if handler is None:
return None # not a known command — let the LLM handle it
try:
# Slash handlers run on the event-loop thread, OUTSIDE the per-turn
# contextvars.copy_context() that pins the session cwd for the agent
# call. ``/compress`` and ``/model`` reach code that REBUILDS the
# system prompt (agent._build_system_prompt -> resolve_agent_cwd), so
# an unpinned handler bakes the Hermes install tree into the session's
# cached prompt — persisted, and therefore poisoning every later turn
# even though the turn itself is pinned. Pin inside a fresh context so
# the write can't leak into other concurrent ACP sessions and needs no
# teardown.
def _dispatch() -> str | None:
try:
from agent.runtime_cwd import set_session_cwd
set_session_cwd(state.cwd)
except Exception:
logger.debug("Could not pin ACP session cwd for slash command", exc_info=True)
return handler(args, state)
try:
return contextvars.copy_context().run(_dispatch)
except Exception as e:
logger.error("Slash command /%s error: %s", cmd, e, exc_info=True)
return f"Error executing /{cmd}: {e}"
@@ -1826,8 +2136,8 @@ class HermesACPAgent(acp.Agent):
return "No tools available."
lines = [f"Available tools ({len(tools)}):"]
for t in tools:
name = t.get("function", {}).get("name", "?")
desc = t.get("function", {}).get("description", "")
name = (t.get("function") or {}).get("name", "?")
desc = (t.get("function") or {}).get("description", "")
# Truncate long descriptions
if len(desc) > 80:
desc = desc[:77] + "..."
@@ -1898,7 +2208,7 @@ class HermesACPAgent(acp.Agent):
lines.append(
f"Compression: due now (threshold ~{threshold_tokens:,}"
+ (f", {threshold_pct:.0f}%" if threshold_pct else "")
+ "). Run /compact."
+ "). Run /compress."
)
else:
lines.append(
@@ -1911,9 +2221,12 @@ class HermesACPAgent(acp.Agent):
lines.append(f"Compression threshold: ~{threshold_tokens:,} tokens")
if getattr(agent, "compression_enabled", True) is False:
lines.append("Compression is disabled for this agent.")
lines.append(
"Auto-compaction is disabled (compression.enabled: false); "
"/compress still compresses manually."
)
else:
lines.append("Tip: run /compact to compress manually before the threshold.")
lines.append("Tip: run /compress to compress manually before the threshold.")
return "\n".join(lines)
@@ -1933,13 +2246,14 @@ class HermesACPAgent(acp.Agent):
return "Conversation history cleared. Agent session state reset failed; see logs."
return "Conversation history cleared."
def _cmd_compact(self, args: str, state: SessionState) -> str:
def _cmd_compress(self, args: str, state: SessionState) -> str:
if not state.history:
return "Nothing to compress — conversation is empty."
try:
agent = state.agent
if not getattr(agent, "compression_enabled", True):
return "Context compression is disabled for this agent."
# No compression_enabled gate: the flag disables *automatic*
# compaction only; manual /compress must keep working (matches
# the CLI /compress and gateway handlers).
if not hasattr(agent, "_compress_context"):
return "Context compression not available for this agent."
@@ -1964,6 +2278,7 @@ class HermesACPAgent(acp.Agent):
getattr(agent, "_cached_system_prompt", "") or "",
approx_tokens=approx_tokens,
task_id=state.session_id,
force=True,
)
finally:
agent._session_db = original_session_db
-16
View File
@@ -1,16 +0,0 @@
{
"id": "hermes-agent",
"name": "Hermes Agent",
"version": "0.18.2",
"description": "Self-improving open-source AI agent by Nous Research with ACP editor integration, persistent memory, skills, and rich tool support.",
"repository": "https://github.com/NousResearch/hermes-agent",
"website": "https://hermes-agent.nousresearch.com/docs/user-guide/features/acp",
"authors": ["Nous Research"],
"license": "MIT",
"distribution": {
"uvx": {
"package": "hermes-agent[acp]==0.18.2",
"args": ["hermes-acp"]
}
}
}
-8
View File
@@ -1,8 +0,0 @@
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 16 16" width="16" height="16" fill="none">
<path d="M8 1.5v13" stroke="currentColor" stroke-width="1.5" stroke-linecap="round"/>
<path d="M8 3.25c-2.35-1.4-4.7-.95-6.25.35 1.85-.2 3.8.2 5.55 1.55" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
<path d="M8 3.25c2.35-1.4 4.7-.95 6.25.35-1.85-.2-3.8.2-5.55 1.55" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
<path d="M8 13.25c-2.3-1-3.05-2.65-1.35-4.15-2 .8-2.35 2.95-.35 4" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
<path d="M8 13.25c2.3-1 3.05-2.65 1.35-4.15 2 .8 2.35 2.95.35 4" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
<circle cx="8" cy="1.8" r="1.1" fill="currentColor"/>
</svg>

Before

Width:  |  Height:  |  Size: 882 B

+12
View File
@@ -701,6 +701,18 @@ def redeem_codex_reset_credit(
remaining = max(0, available - 1)
plural = "s" if remaining != 1 else ""
if code == "reset":
# The redeemed reset restores the account's quota upstream — lift any
# persisted pool cooldowns so Hermes doesn't keep the credential
# frozen behind the now-stale ``last_error_reset_at`` (issue #43747).
try:
from hermes_cli.auth import clear_codex_pool_quota_cooldowns
clear_codex_pool_quota_cooldowns()
except Exception:
logger.debug(
"Failed to clear Codex pool cooldowns after reset redemption",
exc_info=True,
)
return CodexResetRedeemResult(
status="reset",
message=(
+617 -74
View File
@@ -28,7 +28,7 @@ import time
import uuid
from datetime import datetime
from typing import Any, Callable, Dict, List, Optional
from urllib.parse import urlparse, parse_qs, urlunparse
from urllib.parse import parse_qs, urlparse, urlunparse
from agent.context_compressor import ContextCompressor
from agent.iteration_budget import IterationBudget
@@ -48,6 +48,7 @@ from agent.tool_guardrails import (
ToolGuardrailDecision,
)
from hermes_cli.config import cfg_get
from hermes_cli.route_identity import normalize_route_base_url
from hermes_cli.timeouts import get_provider_request_timeout
from hermes_constants import get_hermes_home
from utils import base_url_host_matches, is_truthy_value
@@ -68,18 +69,188 @@ def _ra():
return run_agent
def _build_codex_gpt5_autoraise_notice(autoraise: Dict[str, Any]) -> str:
def _moa_reference_output_allowed(agent: Any) -> bool:
"""Keep MoA display events off only the machine-readable ``-Q`` surface."""
return not (
getattr(agent, "platform", None) == "cli"
and getattr(agent, "tool_progress_mode", "all") == "off"
)
def _relay_moa_reference_event(agent: Any, event: str, **kwargs: Any) -> None:
"""Relay MoA display events while preserving the ``-Q`` stdout contract."""
if not _moa_reference_output_allowed(agent):
return
cb = getattr(agent, "tool_progress_callback", None)
if cb is None:
return
try:
if event == "moa.reference":
cb(
"moa.reference",
str(kwargs.get("label") or ""),
str(kwargs.get("text") or ""),
None,
moa_index=kwargs.get("index"),
moa_count=kwargs.get("count"),
)
elif event == "moa.aggregating":
cb(
"moa.aggregating",
str(kwargs.get("aggregator") or ""),
None,
None,
moa_ref_count=kwargs.get("ref_count"),
)
except Exception:
pass
def _normalize_route_base_url(base_url: Any) -> str:
"""Canonicalize an endpoint URL for model-route identity comparisons."""
return normalize_route_base_url(base_url)
def _provider_default_routes(provider: str) -> set[str]:
"""Return known exact default routes for a canonical provider id."""
routes: set[str] = set()
try:
from hermes_cli.providers import HERMES_OVERLAYS, get_provider
overlay = HERMES_OVERLAYS.get(provider)
provider_def = get_provider(provider, allow_network=False)
for value in (
getattr(overlay, "base_url_override", ""),
getattr(provider_def, "base_url", ""),
):
route = _normalize_route_base_url(value)
if route:
routes.add(route)
except Exception:
pass
try:
from providers import get_provider_profile
profile = get_provider_profile(provider)
route = _normalize_route_base_url(
getattr(profile, "base_url", "")
)
if route:
routes.add(route)
except Exception:
pass
try:
from hermes_cli.auth import PROVIDER_REGISTRY
from hermes_cli.models import normalize_provider as normalize_model_provider
from hermes_cli.providers import normalize_provider as normalize_registry_provider
for provider_id, config in PROVIDER_REGISTRY.items():
canonical_id = normalize_registry_provider(
normalize_model_provider(provider_id)
)
if canonical_id != provider:
continue
route = _normalize_route_base_url(
getattr(config, "inference_base_url", "")
)
if route:
routes.add(route)
except Exception:
pass
if provider == "gemini":
routes.update(
f"{route.rstrip('/')}/openai"
for route in list(routes)
)
return routes
def _context_route_mismatch(
configured_base_url: Any,
active_base_url: Any,
configured_provider: Any,
active_provider: Any,
*,
already_normalized: bool = False,
) -> bool:
"""Return whether a context pin's configured route differs from runtime."""
if already_normalized:
configured_route = str(configured_base_url or "")
active_route = str(active_base_url or "")
else:
configured_route = _normalize_route_base_url(configured_base_url)
active_route = _normalize_route_base_url(active_base_url)
if configured_route:
return configured_route != active_route
configured_provider = str(configured_provider or "").strip()
active_provider = str(active_provider or "").strip()
if not configured_provider:
return False
try:
from hermes_cli.models import normalize_provider as normalize_model_provider
configured_provider = normalize_model_provider(configured_provider)
active_provider = normalize_model_provider(active_provider)
except Exception:
configured_provider = configured_provider.lower()
active_provider = active_provider.lower()
try:
from hermes_cli.providers import normalize_provider as normalize_registry_provider
configured_provider = normalize_registry_provider(configured_provider)
active_provider = normalize_registry_provider(active_provider)
except Exception:
pass
if active_route:
configured_routes = _provider_default_routes(configured_provider)
return not configured_routes or active_route not in configured_routes
return bool(
configured_provider
and active_provider
and configured_provider != active_provider
)
def _normalize_custom_provider_name(value: Any) -> str:
"""Mirror runtime normalization for a requested custom-provider identity."""
return str(value or "").strip().lower().replace(" ", "-")
def _custom_provider_runtime_ids(value: Any) -> set[str]:
"""Return raw/menu identities that runtime accepts for a configured name."""
normalized = _normalize_custom_provider_name(value)
if not normalized:
return set()
return {normalized, f"custom:{normalized}"}
def _build_codex_gpt5_autoraise_notice(
autoraise: Dict[str, Any], context_length: Optional[int] = None
) -> str:
"""Build the one-time notice shown when Codex gpt-5.x raises compaction.
``autoraise`` is ``{"model": <slug>, "from": <old_ratio>, "to": <new_ratio>}``.
The same text is printed inline for CLI users and replayed via
``context_length`` is the live-resolved window from the context compressor
(Codex's /models catalog is authoritative and can change server-side, e.g.
the gpt-5.6 family's 272K → 372K → 272K shifts in July 2026), so the banner
reports what this session actually got rather than a hardcoded cap. The
same text is printed inline for CLI users and replayed via
``status_callback`` for gateway users, so it must be self-contained and
include the exact opt-back-out command.
"""
model = str(autoraise.get("model") or "gpt-5.4/5.5").strip().lower().rsplit("/", 1)[-1]
# gpt-5.3-codex-spark has a native 128K window; the gpt-5.4/5.5/5.6 family
# is capped at 272K by the Codex OAuth backend.
cap = "128K" if model.startswith("gpt-5.3-codex-spark") else "272K"
if isinstance(context_length, int) and context_length > 0:
cap = f"{round(context_length / 1000)}K"
else:
# Static fallback when the resolved window isn't available:
# gpt-5.3-codex-spark has a native 128K window; the gpt-5.4/5.5/5.6
# family is capped at 272K by the Codex OAuth backend.
cap = "128K" if model.startswith("gpt-5.3-codex-spark") else "272K"
from_pct = int(round(autoraise["from"] * 100))
to_pct = int(round(autoraise["to"] * 100))
return (
@@ -285,7 +456,6 @@ def init_agent(
args: list[str] | None = None,
model: str = "",
max_iterations: int = 90, # Default tool-calling iterations (shared with subagents)
tool_delay: float = 1.0,
enabled_toolsets: List[str] = None,
disabled_toolsets: List[str] = None,
save_trajectories: bool = False,
@@ -346,6 +516,7 @@ def init_agent(
checkpoint_max_total_size_mb: int = 500,
checkpoint_max_file_size_mb: int = 10,
pass_session_id: bool = False,
requested_provider: str = None,
):
"""
Initialize the AI Agent.
@@ -354,10 +525,10 @@ def init_agent(
base_url (str): Base URL for the model API (optional)
api_key (str): API key for authentication (optional, uses env var if not provided)
provider (str): Provider identifier (optional; used for telemetry/routing hints)
requested_provider (str): Original provider identity before runtime canonicalization
api_mode (str): API mode override: "chat_completions" or "codex_responses"
model (str): Model name to use (default: "anthropic/claude-opus-4.6")
max_iterations (int): Maximum number of tool calling iterations (default: 90)
tool_delay (float): Delay between tool calls in seconds (default: 1.0)
enabled_toolsets (List[str]): Only enable tools from these toolsets (optional)
disabled_toolsets (List[str]): Disable tools from these toolsets (optional)
save_trajectories (bool): Whether to save conversation trajectories to JSONL files (default: False)
@@ -403,7 +574,6 @@ def init_agent(
# Shared iteration budget — parent creates, children inherit.
# Consumed by every LLM turn across parent + all subagents.
agent.iteration_budget = iteration_budget or IterationBudget(max_iterations)
agent.tool_delay = tool_delay
agent.save_trajectories = save_trajectories
agent.verbose_logging = verbose_logging
agent.quiet_mode = quiet_mode
@@ -434,6 +604,11 @@ def init_agent(
agent.base_url = base_url or ""
provider_name = provider.strip().lower() if isinstance(provider, str) and provider.strip() else None
agent.provider = provider_name or ""
agent.requested_provider = (
requested_provider.strip().lower()
if isinstance(requested_provider, str) and requested_provider.strip()
else agent.provider
)
agent._credential_pool = credential_pool
agent.acp_command = acp_command or command
agent.acp_args = list(acp_args or args or [])
@@ -467,6 +642,13 @@ def init_agent(
# AWS Bedrock — auto-detect from provider name or base URL
# (bedrock-runtime.<region>.amazonaws.com).
agent.api_mode = "bedrock_converse"
elif agent.provider in {"nous", "nous-portal", "nousresearch"}:
# Portal is dual-wire: anthropic/* → Messages, everything else →
# chat_completions. Callers that already pass api_mode win above;
# this covers direct AIAgent construction without a resolved runtime.
from hermes_cli.providers import nous_api_mode
agent.api_mode = nous_api_mode(agent.model)
else:
agent.api_mode = "chat_completions"
@@ -586,6 +768,8 @@ def init_agent(
agent._execution_thread_id: int | None = None # Set at run_conversation() start
agent._interrupt_thread_signal_pending = False
agent._client_lock = threading.RLock()
agent._model_request_active = threading.Event()
agent._supports_active_turn_redirect = True
# /steer mechanism — inject a user note into the next tool result
# without interrupting the agent. Unlike interrupt(), steer() does
@@ -597,6 +781,13 @@ def init_agent(
agent._pending_steer: Optional[str] = None
agent._pending_steer_lock = threading.Lock()
# Active-turn redirect mechanism. A regular follow-up sent while the model
# is generating is different from a hard /stop: preserve the valid turn
# prefix, cancel only the in-flight model request, and rebuild its tail with
# the correction. The loop drains this slot at a role-safe boundary.
agent._pending_redirect: Optional[str] = None
agent._pending_redirect_lock = threading.Lock()
# Concurrent-tool worker thread tracking. `_execute_tool_calls_concurrent`
# runs each tool on its own ThreadPoolExecutor worker — those worker
# threads have tids distinct from `_execution_thread_id`, so
@@ -636,9 +827,10 @@ def init_agent(
# Anthropic prompt caching: auto-enabled for Claude models on native
# Anthropic, OpenRouter, and third-party gateways that speak the
# Anthropic protocol (``api_mode == 'anthropic_messages'``). Reduces
# input costs by ~75% on multi-turn conversations. Uses system_and_3
# strategy (4 breakpoints). See ``_anthropic_prompt_cache_policy``
# for the layout-vs-transport decision.
# input costs by ~75% on multi-turn conversations. Uses four breakpoints:
# the static system prefix, full system prompt, and last two messages
# (falling back to system-and-3 when no static prefix is available). See
# ``_anthropic_prompt_cache_policy`` for the layout-vs-transport decision.
agent._use_prompt_caching, agent._use_native_cache_layout = (
agent._anthropic_prompt_cache_policy()
)
@@ -648,7 +840,7 @@ def init_agent(
# sessions with >5-minute pauses between turns (#14971).
agent._cache_ttl = "5m"
try:
from hermes_cli.config import load_config as _load_pc_cfg
from hermes_cli.config import load_config_readonly as _load_pc_cfg
_pc_cfg = _load_pc_cfg().get("prompt_caching", {}) or {}
_ttl = _pc_cfg.get("cache_ttl", "5m")
@@ -692,8 +884,10 @@ def init_agent(
# report cumulative micros spent. Surfaced behind HERMES_DEV_CREDITS.
agent._credits_state = None
agent._credits_session_start_micros = None
# Threshold-notice latch (L4): active sticky-notice keys + the warn90 crossing gate.
agent._credits_latch = {"active": set(), "seen_below_90": False, "usage_band": None}
# Threshold-notice latch (L4): active sticky-notice keys + the crossing gates.
from agent.credits_tracker import new_credits_latch
agent._credits_latch = new_credits_latch()
# OpenRouter response cache hit counter — incremented when
# X-OpenRouter-Cache-Status: HIT is seen in streaming response headers.
@@ -869,49 +1063,20 @@ def init_agent(
elif isinstance(effective_key, str) and len(effective_key) > 12:
print(f"🔑 Using token: {effective_key[:8]}...{effective_key[-4:]}")
elif agent.provider == "moa":
from agent.moa_loop import MoAClient
from agent.moa_loop import build_moa_facade
agent.api_mode = "chat_completions"
# Route reference-model outputs to the agent's tool_progress_callback so
# build_moa_facade wires the reference relay that routes
# reference-model outputs to the agent's tool_progress_callback so
# every surface that already consumes it (CLI spinner/scrollback, TUI,
# desktop, gateway) can show each reference's answer as a labelled block
# before the aggregator acts. The facade emits "moa.reference" and
# "moa.aggregating" events; we forward them through the same callback
# the tool lifecycle uses. Best-effort and cache-safe — these are
# display-only events, they never touch the message history.
def _moa_reference_relay(event: str, **kwargs: Any) -> None:
cb = getattr(agent, "tool_progress_callback", None)
if cb is None:
return
try:
if event == "moa.reference":
label = str(kwargs.get("label") or "")
text = str(kwargs.get("text") or "")
idx = kwargs.get("index")
count = kwargs.get("count")
cb(
"moa.reference",
label,
text,
None,
moa_index=idx,
moa_count=count,
)
elif event == "moa.aggregating":
cb(
"moa.aggregating",
str(kwargs.get("aggregator") or ""),
None,
None,
moa_ref_count=kwargs.get("ref_count"),
)
except Exception:
pass
agent.client = MoAClient(
agent.model or "default",
reference_callback=_moa_reference_relay,
)
# desktop, gateway) can show each reference's answer as a labelled
# block before the aggregator acts. The facade emits "moa.reference",
# "moa.progress", "moa.phase", and "moa.aggregating" events, forwarded
# through the same callback the tool lifecycle uses. Best-effort and
# cache-safe — display-only events, they never touch the message
# history. The factory is shared with the fallback-restore/recovery
# paths so a restored facade keeps emitting these events (#53802).
agent.client = build_moa_facade(agent, agent.model)
agent._client_kwargs = {}
agent.api_key = api_key or "moa-virtual-provider"
agent.base_url = "moa://local"
@@ -925,7 +1090,7 @@ def init_agent(
# Guardrail config — read from config.yaml at init time.
agent._bedrock_guardrail_config = None
try:
from hermes_cli.config import load_config as _load_br_cfg
from hermes_cli.config import load_config_readonly as _load_br_cfg
_gr = _load_br_cfg().get("bedrock", {}).get("guardrail", {})
if _gr.get("guardrail_identifier") and _gr.get("guardrail_version"):
agent._bedrock_guardrail_config = {
@@ -990,10 +1155,14 @@ def init_agent(
elif base_url_host_matches(effective_base, "chatgpt.com"):
from agent.auxiliary_client import _codex_cloudflare_headers
client_kwargs["default_headers"] = _codex_cloudflare_headers(api_key)
elif base_url_host_matches(effective_base, "x.ai"):
from tools.xai_http import hermes_xai_default_headers
client_kwargs["default_headers"] = hermes_xai_default_headers()
elif "default_headers" not in client_kwargs:
# Fall back to profile.default_headers for providers that
# declare custom headers (e.g. Kimi User-Agent on non-kimi.com
# endpoints).
# declare custom headers (e.g. Vercel AI Gateway attribution,
# Kimi User-Agent on non-kimi.com endpoints).
try:
from providers import get_provider_profile as _gpf
_ph = _gpf(agent.provider)
@@ -1177,6 +1346,13 @@ def init_agent(
print("⚠️ Warning: API key appears invalid or missing")
except Exception as e:
raise RuntimeError(f"Failed to initialize OpenAI client: {e}")
# Keep a stable identity for the pool entry that supplied this runtime.
# OAuth refreshes can replace the runtime token before a failed request is
# recovered, so the mutable API-key value alone cannot reliably attribute
# the failure to its source entry.
from agent.agent_runtime_helpers import sync_credential_pool_entry_id
sync_credential_pool_entry_id(agent)
# Provider fallback chain — ordered list of backup providers tried
# when the primary is exhausted (rate-limit, overload, connection
@@ -1301,7 +1477,7 @@ def init_agent(
# reads the JSON files directly. See run_agent._save_session_log.
agent._session_json_enabled = False
try:
from hermes_cli.config import load_config as _load_sess_cfg
from hermes_cli.config import load_config_readonly as _load_sess_cfg
_sess_cfg = (_load_sess_cfg().get("sessions") or {})
agent._session_json_enabled = bool(_sess_cfg.get("write_json_snapshots", False))
except Exception:
@@ -1323,6 +1499,9 @@ def init_agent(
# Cached system prompt -- built once per session, only rebuilt on compression
agent._cached_system_prompt: Optional[str] = None
# Cross-session-stable prefix of the cached prompt. It remains separate
# from the persisted string and is used only to place an early cache marker.
agent._cached_system_prompt_static: Optional[str] = None
# Filesystem checkpoint manager (transparent — not a tool)
from tools.checkpoint_manager import CheckpointManager
@@ -1369,7 +1548,7 @@ def init_agent(
# Load config once for memory, skills, and compression sections
try:
from hermes_cli.config import load_config as _load_agent_config
from hermes_cli.config import load_config_readonly as _load_agent_config
_agent_cfg = _load_agent_config()
except Exception:
_agent_cfg = {}
@@ -1427,7 +1606,14 @@ def init_agent(
agent._memory_nudge_interval = 10
agent._turns_since_memory = 0
agent._iters_since_skill = 0
if not skip_memory:
# A flush/background agent may pass skip_memory=True to avoid spinning up an
# external memory *provider*, but if the caller also explicitly enables the
# "memory" toolset it still needs the built-in file-backed store — otherwise
# the memory tool dispatches with store=None and every call fails (#65429).
# So the built-in store is created unless memory is globally disabled, while
# the external-provider block below stays gated on skip_memory.
_memory_toolset_requested = "memory" in (agent.enabled_toolsets or [])
if not skip_memory or _memory_toolset_requested:
try:
mem_config = _agent_cfg.get("memory", {})
agent._memory_enabled = mem_config.get("memory_enabled", False)
@@ -1647,6 +1833,89 @@ def init_agent(
compression_enabled = str(_compression_cfg.get("enabled", True)).lower() in {"true", "1", "yes"}
compression_target_ratio = float(_compression_cfg.get("target_ratio", 0.20))
compression_protect_last = int(_compression_cfg.get("protect_last_n", 20))
# Minimum REAL (actionable) user messages guaranteed to survive in the
# uncompressed tail (compression.min_tail_user_messages). Default 1
# preserves current behavior exactly — the existing single-user tail
# anchor. Values > 1 extend the guarantee to the last N actionable
# user turns. Booleans rejected (bool subclasses int), non-int-like
# values fall back to 1, floor at 1.
_raw_min_tail_users = _compression_cfg.get("min_tail_user_messages", 1)
if isinstance(_raw_min_tail_users, bool):
compression_min_tail_users = 1
elif isinstance(_raw_min_tail_users, int):
compression_min_tail_users = _raw_min_tail_users
elif isinstance(_raw_min_tail_users, float):
compression_min_tail_users = (
int(_raw_min_tail_users) if _raw_min_tail_users.is_integer() else 1
)
else:
try:
compression_min_tail_users = int(str(_raw_min_tail_users).strip())
except (TypeError, ValueError):
compression_min_tail_users = 1
if compression_min_tail_users < 1:
compression_min_tail_users = 1
# Cap on compression retry rounds before a turn gives up with "max
# compression attempts reached" (compression.max_attempts). Hardcoding 3
# strands sessions that legitimately need more rounds — e.g. a restart
# history reload whose incompressible tool schemas keep the request
# estimate above the threshold even though the messages compress fine
# (the #62605 failure class). Default 3 preserves current behavior, so
# an unset key is behavior-neutral; validated >= 1, hard-capped at 10,
# and any non-int-like value falls back to 3. Booleans are rejected
# (bool subclasses int, so int(True) would silently become 1) and
# fractional floats are rejected rather than truncated — "4.7 attempts"
# is a config mistake, not a request for 4.
_raw_max_attempts = _compression_cfg.get("max_attempts", 3)
if isinstance(_raw_max_attempts, bool):
compression_max_attempts = 3
elif isinstance(_raw_max_attempts, int):
compression_max_attempts = _raw_max_attempts
elif isinstance(_raw_max_attempts, float):
compression_max_attempts = (
int(_raw_max_attempts) if _raw_max_attempts.is_integer() else 3
)
else:
try:
compression_max_attempts = int(str(_raw_max_attempts).strip())
except (TypeError, ValueError):
compression_max_attempts = 3
if compression_max_attempts < 1:
compression_max_attempts = 3
compression_max_attempts = min(compression_max_attempts, 10)
def _parse_prune_int(raw, default):
# Same parser semantics as compression.max_attempts above: reject
# booleans (bool subclasses int — YAML `true` would coerce to 1),
# reject fractional floats rather than truncating them, accept
# integral floats and numeric strings, fall back to the default on
# anything else.
if isinstance(raw, bool):
return default
if isinstance(raw, int):
return raw
if isinstance(raw, float):
return int(raw) if raw.is_integer() else default
try:
return int(str(raw).strip())
except (TypeError, ValueError):
return default
# Opt-in proactive tool-result prune trigger (0 = disabled — the
# default, so an unset key is behavior-neutral). Negative values are
# treated as disabled rather than erroring.
compression_proactive_prune_tokens = max(
0, _parse_prune_int(_compression_cfg.get("proactive_prune_tokens", 0), 0)
)
compression_proactive_prune_min_chars = _parse_prune_int(
_compression_cfg.get("proactive_prune_min_result_chars", 8000), 8000
)
compression_proactive_prune_min_reclaim = max(
0,
_parse_prune_int(
_compression_cfg.get("proactive_prune_min_reclaim_tokens", 4096), 4096
),
)
# protect_first_n is the number of non-system messages to protect at
# the head, in addition to the system prompt (which is always
# implicitly protected by the compressor). Floor at 0 — a value of
@@ -1659,13 +1928,65 @@ def init_agent(
compression_abort_on_summary_failure = str(
_compression_cfg.get("abort_on_summary_failure", False)
).lower() in {"true", "1", "yes"}
# Per-model threshold overrides: keys are substring-matched against the
# model name (longest match wins). Empty dict = use the global threshold
# for all models (backward compatible).
_raw_model_thresholds = _compression_cfg.get("model_thresholds", {})
if isinstance(_raw_model_thresholds, dict):
compression_model_thresholds = {
str(k): float(v) for k, v in _raw_model_thresholds.items()
if isinstance(v, (int, float)) and not isinstance(v, bool)
}
else:
compression_model_thresholds = {}
# Absolute token cap: when set, compression triggers at the lower of
# the ratio-based threshold and this absolute count. Clamped to the
# model's context length at apply-time so a cap above the window is
# a no-op (ratio-based threshold wins).
compression_threshold_tokens = _compression_cfg.get("threshold_tokens")
if compression_threshold_tokens is not None:
try:
compression_threshold_tokens = int(compression_threshold_tokens)
if compression_threshold_tokens <= 0:
compression_threshold_tokens = None
except (TypeError, ValueError):
compression_threshold_tokens = None
# In-place compaction: when True, compress_context() rewrites the message
# list + rebuilds the system prompt WITHOUT rotating the session id (no
# parent_session_id chain, no `name #N` renumber). See #38763 and
# agent/conversation_compression.py. Consumed by compress_context(), not the
# compressor, so it rides on the agent.
# Default True must match DEFAULT_CONFIG["compression"]["in_place"]
# (#38763). default=False here previously flipped agents into rotation
# mode whenever the merged config omitted the key (partial configs,
# load_config failure → {}), re-arming the pre-lease drift abort.
compression_in_place = is_truthy_value(
_compression_cfg.get("in_place"), default=False
_compression_cfg.get("in_place"), default=True
)
# Opt-in (default False): a micro-compaction pass rewrites already-sent
# history every turn, which breaks the provider prompt-cache prefix on a
# per-turn cadence rather than at an episodic boundary. That is the cost
# `proactive_prune_min_reclaim_tokens` exists to amortize, so the feature
# stays off until an operator opts in and accepts the tradeoff.
compression_micro_compact = is_truthy_value(
_compression_cfg.get("micro_compact"), default=False
)
# How often a pass runs, in completed turns. Each pass rewrites
# already-sent history and costs one prompt-cache break, so this is the
# dial for how often that cost is paid: 1 = every turn (most aggressive
# reclaim), 5 = one break per five turns. Clamped to >= 1.
compression_micro_compact_every_n_turns = max(
1,
_parse_prune_int(_compression_cfg.get("micro_compact_every_n_turns", 1), 1),
)
# Rolling-summary defrag threshold, in tokens. Lived on the compressor as
# a hardcoded attribute with no path from config until now.
compression_micro_compact_defrag_tokens = max(
1,
_parse_prune_int(
_compression_cfg.get("micro_compact_defrag_threshold_tokens", 2000),
2000,
),
)
codex_app_server_auto_compaction = str(
_compression_cfg.get("codex_app_server_auto", "native") or "native"
@@ -1677,6 +1998,12 @@ def init_agent(
codex_app_server_auto_compaction,
)
codex_app_server_auto_compaction = "native"
# Opt-in idle compaction: compact a session up front when it resumes after
# this many seconds of inactivity (0 = disabled). Time-based, so it
# complements the size-based threshold above. Consumed by build_turn_context().
compression_idle_compact_after_seconds = max(
0, int(_compression_cfg.get("idle_compact_after_seconds", 0))
)
# Read optional explicit context_length override for the auxiliary
# compression model. Custom endpoints often cannot report this via
@@ -1747,8 +2074,9 @@ def init_agent(
)
_config_context_length = None
# Resolve custom_providers list once for reuse below (startup
# context-length override and plugin context-engine init).
# Resolve custom_providers once before route-scoping a global context pin:
# a named custom provider may keep its base URL only in this list rather
# than repeating it under ``model``.
try:
from hermes_cli.config import get_compatible_custom_providers
_custom_providers = get_compatible_custom_providers(_agent_cfg)
@@ -1757,6 +2085,163 @@ def init_agent(
if not isinstance(_custom_providers, list):
_custom_providers = []
# ``model.context_length`` describes the configured default model. A
# process launched directly with ``--model`` / ``-m`` has already replaced
# ``agent.model`` before this initializer loads config, so carrying the
# default model's explicit window into that different runtime is stale. The
# live switch/fallback paths already clear this override; keep direct-start
# overrides consistent with them and let provider metadata resolve the
# active model's window instead.
if _config_context_length is not None and isinstance(_model_cfg, dict):
_configured_default_model = str(_model_cfg.get("default") or "").strip()
_configured_default_runtime_model = _configured_default_model
_active_runtime_model = agent.model
if _configured_default_model:
try:
from hermes_cli.model_normalize import normalize_model_for_provider
_configured_default_runtime_model = normalize_model_for_provider(
_configured_default_model, agent.provider
)
_active_runtime_model = normalize_model_for_provider(
agent.model, agent.provider
)
except Exception:
pass
_configured_provider = str(_model_cfg.get("provider") or "").strip()
_configured_base_url = _normalize_route_base_url(
_model_cfg.get("base_url")
)
_configured_provider_norm = _normalize_custom_provider_name(
_configured_provider
)
_custom_provider_candidate = bool(_configured_provider_norm)
_runtime_first_provider_ids = {
"auto",
"moa",
"vertex",
"google-vertex",
"vertex-ai",
"gcp-vertex",
"vertexai",
}
if _configured_provider_norm in _runtime_first_provider_ids:
_custom_provider_candidate = False
elif (
_custom_provider_candidate
and _configured_provider_norm != "custom"
and not _configured_provider_norm.startswith("custom:")
):
try:
from hermes_cli.auth import resolve_provider as resolve_auth_provider
_resolved_auth_provider = resolve_auth_provider(
_configured_provider_norm
)
_custom_provider_candidate = (
str(_resolved_auth_provider or "").strip().lower()
!= _configured_provider_norm
)
except Exception:
pass
if not _configured_base_url and _custom_provider_candidate:
_configured_custom_provider = _normalize_custom_provider_name(
_configured_provider
)
_user_providers = _agent_cfg.get("providers")
_disabled_custom_provider_ids: set[str] = set()
if isinstance(_user_providers, dict):
from hermes_cli.config import is_provider_enabled
for _provider_key, _provider_entry in _user_providers.items():
if not isinstance(_provider_entry, dict):
continue
_entry_name = str(
_provider_entry.get("name") or ""
).strip()
_entry_provider_ids = _custom_provider_runtime_ids(
_provider_key
) | _custom_provider_runtime_ids(_entry_name)
if not is_provider_enabled(_provider_entry):
_disabled_custom_provider_ids.update(
provider_id
for provider_id in _entry_provider_ids
if provider_id
)
continue
if _configured_custom_provider not in _entry_provider_ids:
continue
_configured_base_url = _normalize_route_base_url(
_provider_entry.get("api")
or _provider_entry.get("url")
or _provider_entry.get("base_url")
)
if _configured_base_url:
break
if not _configured_base_url:
for _provider_entry in _custom_providers:
if not isinstance(_provider_entry, dict):
continue
_entry_name = str(
_provider_entry.get("name") or ""
).strip()
_entry_provider_key = str(
_provider_entry.get("provider_key") or ""
).strip().lower()
_entry_provider_ids = _custom_provider_runtime_ids(
_entry_name
) | _custom_provider_runtime_ids(_entry_provider_key)
if (
_entry_provider_key
and _custom_provider_runtime_ids(_entry_provider_key)
& _disabled_custom_provider_ids
):
continue
if _configured_custom_provider not in _entry_provider_ids:
continue
_configured_base_url = _normalize_route_base_url(
_provider_entry.get("base_url")
)
if _configured_base_url:
break
_active_route_url = str(agent.base_url or "")
_requested_route_url = str(base_url or "")
if "?" in _requested_route_url.split("#", 1)[0]:
try:
_requested_parts = urlparse(_requested_route_url)
_requested_without_query = urlunparse(
_requested_parts._replace(query="")
)
if _normalize_route_base_url(
_requested_without_query
) == _normalize_route_base_url(_active_route_url):
_active_route_url = _requested_route_url
except (TypeError, ValueError):
pass
_active_base_url = _normalize_route_base_url(_active_route_url)
_route_mismatch = _context_route_mismatch(
_configured_base_url,
_active_base_url,
_configured_provider,
agent.provider,
already_normalized=True,
)
_model_mismatch = bool(
_configured_default_runtime_model
and _configured_default_runtime_model != _active_runtime_model
)
if _model_mismatch or _route_mismatch:
_ra().logger.debug(
"Ignoring model.context_length=%s for startup runtime %s at %s "
"(configured default is %s at %s)",
_config_context_length,
agent.model,
_active_base_url or agent.provider,
_configured_default_model,
_configured_base_url or _model_cfg.get("provider"),
)
_config_context_length = None
# Store for reuse by _check_compression_model_feasibility (auxiliary
# compression model context-length detection needs the same list).
agent._custom_providers = _custom_providers
@@ -1779,11 +2264,11 @@ def init_agent(
# Surface a clear warning if the user set a context_length but it
# wasn't a valid positive int — the helper silently skips those.
if _config_context_length is None:
_target = agent.base_url.rstrip("/") if agent.base_url else ""
_target = _normalize_route_base_url(agent.base_url)
for _cp_entry in _custom_providers:
if not isinstance(_cp_entry, dict):
continue
_cp_url = (_cp_entry.get("base_url") or "").rstrip("/")
_cp_url = _normalize_route_base_url(_cp_entry.get("base_url"))
if _target and _cp_url == _target:
_cp_models = _cp_entry.get("models", {})
if isinstance(_cp_models, dict):
@@ -1815,7 +2300,18 @@ def init_agent(
# AFTER the custom_providers branch so per-model overrides aren't lost.
agent._config_context_length = _config_context_length
agent._ensure_lmstudio_runtime_loaded(_config_context_length)
_lmstudio_runtime_context_length = agent._ensure_lmstudio_runtime_loaded(
_config_context_length
)
if agent._lmstudio_load_was_unverified(_lmstudio_runtime_context_length):
_ra().logger.warning(
"LM Studio model activation was rejected or completed without a "
"verifiable active context length; falling back to configured context"
)
_effective_context_length = agent._effective_lmstudio_context_length(
_config_context_length,
_lmstudio_runtime_context_length,
)
@@ -1892,10 +2388,20 @@ def init_agent(
agent.model,
base_url=agent.base_url,
api_key=getattr(agent, "api_key", ""),
config_context_length=_config_context_length,
config_context_length=_effective_context_length,
provider=agent.provider,
custom_providers=_custom_providers,
)
# Per-model threshold overrides are part of the explicit
# context-engine contract: assign them BEFORE the initial
# update_model() call so the first resolution (which derives
# threshold_percent/threshold_tokens for the initial model) already
# sees the overrides. Assigning after update_model() left the initial
# model on the engine's global threshold until the first /model
# switch. Engines that override update_model() own their own policy
# and may ignore the attribute.
if compression_model_thresholds:
agent.context_compressor.model_thresholds = compression_model_thresholds
agent.context_compressor.update_model(
model=agent.model,
context_length=_plugin_ctx_len,
@@ -1917,11 +2423,17 @@ def init_agent(
quiet_mode=agent.quiet_mode,
base_url=agent.base_url,
api_key=getattr(agent, "api_key", ""),
config_context_length=_config_context_length,
config_context_length=_effective_context_length,
provider=agent.provider,
api_mode=agent.api_mode,
abort_on_summary_failure=compression_abort_on_summary_failure,
max_tokens=agent.max_tokens,
model_thresholds=compression_model_thresholds,
threshold_tokens_cap=compression_threshold_tokens,
proactive_prune_tokens=compression_proactive_prune_tokens,
proactive_prune_min_result_chars=compression_proactive_prune_min_chars,
proactive_prune_min_reclaim_tokens=compression_proactive_prune_min_reclaim,
min_tail_user_messages=compression_min_tail_users,
)
_bind_session_state = getattr(agent.context_compressor, "bind_session_state", None)
if callable(_bind_session_state):
@@ -1931,12 +2443,32 @@ def init_agent(
pass
agent.compression_enabled = compression_enabled
agent.compression_in_place = compression_in_place
# Apply micro-compaction settings to the compressor (feature is opt-in)
_cc = getattr(agent, "context_compressor", None)
if _cc is not None and hasattr(_cc, "_micro_compact_enabled"):
_cc._micro_compact_enabled = compression_micro_compact
if _cc is not None and hasattr(_cc, "_micro_compact_every_n_turns"):
_cc._micro_compact_every_n_turns = compression_micro_compact_every_n_turns
if _cc is not None and hasattr(_cc, "_micro_compact_defrag_threshold_tokens"):
_cc._micro_compact_defrag_threshold_tokens = (
compression_micro_compact_defrag_tokens
)
agent.codex_app_server_auto_compaction = codex_app_server_auto_compaction
agent.max_compression_attempts = compression_max_attempts
agent.compression_idle_compact_after_seconds = (
compression_idle_compact_after_seconds
)
# Reject models whose context window is below the minimum required
# for reliable tool-calling workflows (64K tokens).
_ctx = getattr(agent.context_compressor, "context_length", 0)
if _ctx and _ctx < MINIMUM_CONTEXT_LENGTH:
_allow_lmstudio_explicit_below_floor = (
str(getattr(agent, "provider", "") or "").strip().lower() == "lmstudio"
and isinstance(agent._config_context_length, int)
and not isinstance(agent._config_context_length, bool)
and agent._config_context_length > 0
)
if _ctx and _ctx < MINIMUM_CONTEXT_LENGTH and not _allow_lmstudio_explicit_below_floor:
raise ValueError(
f"Model {agent.model} has a context window of {_ctx:,} tokens, "
f"which is below the minimum {MINIMUM_CONTEXT_LENGTH:,} required "
@@ -2116,7 +2648,7 @@ def init_agent(
# autoraised model) updates the marker state and re-notifies once. The
# config display gate (compression.codex_gpt55_autoraise_notice) still
# suppresses the banner entirely without disabling the threshold autoraise.
_autoraise = getattr(agent, "_compression_threshold_autoraised", None)
_autoraise = getattr(agent, "_compression_threshold_autoraised", None) or {}
_show_autoraise_notice = (
bool(_autoraise)
and compression_enabled
@@ -2132,14 +2664,21 @@ def init_agent(
_active_threshold_pct = getattr(
agent.context_compressor, "threshold_percent", compression_threshold
)
print(f"📊 Context limit: {agent.context_compressor.context_length:,} tokens (compress at {int(_active_threshold_pct*100)}% = {agent.context_compressor.threshold_tokens:,})")
_cap_note = ""
_cap = getattr(agent.context_compressor, "threshold_tokens_cap", None)
if _cap and _cap > 0:
_cap_note = f" (capped at {_cap:,} tokens)"
print(f"📊 Context limit: {agent.context_compressor.context_length:,} tokens (compress at {int(_active_threshold_pct*100)}% = {agent.context_compressor.threshold_tokens:,}{_cap_note})")
else:
print(f"📊 Context limit: {agent.context_compressor.context_length:,} tokens (auto-compression disabled)")
# Notice with the exact opt-back-out command. Printed inline at startup
# for CLI users; gateway users get the same text replayed via
# _compression_warning on turn 1 (set below).
if _show_autoraise_notice:
print(_build_codex_gpt5_autoraise_notice(_autoraise))
print(_build_codex_gpt5_autoraise_notice(
_autoraise,
context_length=getattr(agent.context_compressor, "context_length", None),
))
# Check immediately so CLI users see the warning at startup.
# Gateway status_callback is not yet wired, so any warning is stored
@@ -2149,7 +2688,10 @@ def init_agent(
# above only reaches the CLI, so stash the same text here to be replayed
# through status_callback on the first turn (Telegram/Discord/Slack/etc.).
if _show_autoraise_notice:
agent._compression_warning = _build_codex_gpt5_autoraise_notice(_autoraise)
agent._compression_warning = _build_codex_gpt5_autoraise_notice(
_autoraise,
context_length=getattr(agent.context_compressor, "context_length", None),
)
# Mark shown so repeated inits in this profile (e.g. every gateway message)
# stay silent. Recorded once, whether the notice went to the CLI print or
@@ -2172,6 +2714,7 @@ def init_agent(
agent._primary_runtime = {
"model": agent.model,
"provider": agent.provider,
"requested_provider": agent.requested_provider,
"base_url": agent.base_url,
"api_mode": agent.api_mode,
"api_key": getattr(agent, "api_key", ""),
File diff suppressed because it is too large Load Diff
+297 -43
View File
@@ -23,7 +23,7 @@ from urllib.parse import urlparse
from hermes_constants import get_hermes_home
from typing import Any, Dict, List, Optional, Tuple
from utils import base_url_host_matches, normalize_proxy_env_vars
from utils import base_url_host_matches, base_url_hostname, normalize_proxy_env_vars
# NOTE: `import anthropic` is deliberately NOT at module top — the SDK pulls
# ~220 ms of imports (anthropic.types, anthropic.lib.tools._beta_runner, etc.)
@@ -127,6 +127,8 @@ _FAST_MODE_SUPPORTED_SUBSTRINGS = ("opus-4-6", "opus-4.6")
_ANTHROPIC_OUTPUT_LIMITS = {
# Mythos-class named models (claude-fable-5, …) — 1M context, reasoning
"claude-fable": 128_000,
# Claude Sonnet 5
"claude-sonnet-5": 128_000,
# Claude 4.8
"claude-opus-4-8": 128_000,
# Claude 4.7
@@ -247,7 +249,13 @@ def _supports_adaptive_thinking(model: str) -> bool:
only returns False for the explicit legacy list of older Claude families
that require manual budget-based thinking. Non-Claude Anthropic-Messages
models (minimax, qwen3, …) return False so they keep the manual path.
Kimi / Moonshot models are the exception: their Anthropic-compatible
endpoints implement the adaptive contract (``thinking.type="adaptive"``
+ ``output_config.effort``, including ``xhigh`` and ``display``).
"""
if _model_name_is_kimi_family(model):
return True
if not _is_claude_model(model):
return False
m = model.lower()
@@ -360,7 +368,7 @@ def _detect_claude_code_version() -> str:
try:
result = _sp.run(
[cmd, "--version"],
capture_output=True, text=True, timeout=5,
capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5,
)
if result.returncode == 0 and result.stdout.strip():
# Output is like "2.1.74 (Claude Code)" or just "2.1.74"
@@ -449,7 +457,8 @@ def _is_kimi_coding_endpoint(base_url: str | None) -> bool:
# Model-name prefixes that identify the Kimi / Moonshot family. Covers
# - official slugs: ``kimi-k2.5``, ``kimi_thinking``, ``moonshot-v1-8k``
# - common release lines: ``k1.5-...``, ``k2-thinking``, ``k25-...``, ``k2.5-...``
# - common release lines: ``k1.5-...``, ``k2-thinking``, ``k25-...``, ``k2.5-...``,
# and the bare Coding Plan slug ``k3`` (plus ``k3.x``/``k3-...`` variants)
# Matched case-insensitively against the post-``normalize_model_name`` form,
# so a caller's ``provider/vendor/model`` slug is handled the same as a
# bare name.
@@ -459,8 +468,14 @@ _KIMI_FAMILY_MODEL_PREFIXES = (
"k1.", "k1-",
"k2.", "k2-",
"k25", "k2.5",
"k3.", "k3-",
)
# Bare release slugs with no separator suffix (Kimi Coding Plan serves K3
# as the exact slug ``k3``). Kept exact-match so unrelated model names that
# merely start with the same characters don't get misclassified.
_KIMI_FAMILY_EXACT_SLUGS = frozenset({"k3"})
def _model_name_is_kimi_family(model: str | None) -> bool:
if not isinstance(model, str):
@@ -471,6 +486,8 @@ def _model_name_is_kimi_family(model: str | None) -> bool:
# Strip vendor prefix (e.g. ``moonshotai/kimi-k2.5`` → ``kimi-k2.5``)
if "/" in m:
m = m.rsplit("/", 1)[-1]
if m in _KIMI_FAMILY_EXACT_SLUGS:
return True
return m.startswith(_KIMI_FAMILY_MODEL_PREFIXES)
@@ -529,15 +546,49 @@ def _is_deepseek_anthropic_endpoint(base_url: str | None) -> bool:
return "/anthropic" in normalized.rstrip("/").lower()
def _is_nous_portal_endpoint(base_url: str | None) -> bool:
"""Return True for Nous Portal's Anthropic Messages route.
Portal serves its ``anthropic/*`` catalog natively at
``https://inference-api.nousresearch.com/v1/messages``. Portal-specific
behaviours key off this: Bearer JWT auth, verbatim catalog model ids,
and native thinking-signature replay.
Trusted hosts only:
1. Prod hostname ``inference-api.nousresearch.com``
2. The operator-set ``NOUS_INFERENCE_BASE_URL`` hostname (staging/preview)
Lookalikes such as ``inference-api.nousresearch.com.attacker.test`` are
rejected (hostname match, not substring).
"""
if base_url_host_matches(base_url or "", "inference-api.nousresearch.com"):
return True
try:
from hermes_cli.auth import _nous_inference_env_override
override = _nous_inference_env_override()
except Exception:
return False
if not override:
return False
# Exact host equality (not subdomain) so the env override can't broaden
# into sibling hosts the operator did not set.
override_host = base_url_hostname(override)
return bool(override_host) and base_url_hostname(base_url or "") == override_host
def _requires_bearer_auth(base_url: str | None) -> bool:
"""Return True for Anthropic-compatible providers that require Bearer auth.
Some third-party /anthropic endpoints implement Anthropic's Messages API but
require Authorization: Bearer instead of Anthropic's native x-api-key header.
MiniMax's global and China Anthropic-compatible endpoints, Azure AI
Foundry's Anthropic-style endpoint, and Palantir Foundry's LLM proxy
follow this pattern.
Foundry's Anthropic-style endpoint, Palantir Foundry's LLM proxy, and Nous
Portal's Messages route follow this pattern.
"""
if _is_nous_portal_endpoint(base_url):
return True
normalized = _normalize_base_url_text(base_url)
if not normalized:
return False
@@ -704,7 +755,11 @@ def _build_anthropic_client_with_bearer_hook(
if common_betas:
kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)}
return _anthropic_sdk.Anthropic(**kwargs)
client = _anthropic_sdk.Anthropic(**kwargs)
# Same env-inference trap as build_anthropic_client: auth_token-only
# construction would otherwise also send ANTHROPIC_API_KEY as X-Api-Key.
client.api_key = None
return client
def build_anthropic_client(
@@ -833,7 +888,16 @@ def build_anthropic_client(
if common_betas:
kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)}
return _anthropic_sdk.Anthropic(**kwargs)
client = _anthropic_sdk.Anthropic(**kwargs)
# Bearer-only construction leaves ``api_key`` unset, so the SDK fills it
# from ``ANTHROPIC_API_KEY`` (Hermes loads that into the process env from
# ``~/.hermes/.env``). The result is dual auth —
# ``X-Api-Key: sk-ant-…`` *and* ``Authorization: Bearer <portal-jwt>`` —
# on every Portal / MiniMax / OAuth Messages request. Clear the env-filled
# key whenever we intentionally authenticated via auth_token alone.
if "auth_token" in kwargs and "api_key" not in kwargs:
client.api_key = None
return client
def build_anthropic_bedrock_client(region: str):
@@ -897,7 +961,7 @@ def _read_claude_code_credentials_from_keychain() -> Optional[Dict[str, Any]]:
"-s", "Claude Code-credentials",
"-w"],
capture_output=True,
text=True,
text=True, encoding='utf-8', errors='replace',
timeout=5,
stdin=subprocess.DEVNULL,
)
@@ -1574,7 +1638,10 @@ def _is_bedrock_model_id(model: str) -> bool:
"""
lower = model.lower()
# Regional inference-profile prefixes
if any(lower.startswith(p) for p in ("global.", "us.", "eu.", "ap.", "jp.")):
if any(lower.startswith(p) for p in (
"global.", "us.", "eu.", "apac.", "ap.", "au.", "jp.",
"ca.", "sa.", "me.", "af.",
)):
return True
# Bare Bedrock model IDs: provider.model-family
if lower.startswith("anthropic."):
@@ -1861,6 +1928,28 @@ def _content_parts_to_anthropic_blocks(parts: Any) -> List[Dict[str, Any]]:
return out
_EMPTY_TEXT_PLACEHOLDER = "(empty)"
def _safe_text(text: Any) -> str:
"""Return ``text`` if it's non-whitespace, else a non-whitespace placeholder.
The Anthropic Messages API rejects requests where a text content block is
empty or whitespace-only (HTTP 400 "text content blocks must contain
non-whitespace text"). When such a block gets stored in session history —
e.g. produced by context compression — it is replayed verbatim on every
subsequent turn, permanently wedging the session. Coercing to a
non-whitespace placeholder is self-healing: the next API call recovers.
Mirrors ``bedrock_adapter._safe_text`` (#9486); ref #69512.
"""
if text is None:
return _EMPTY_TEXT_PLACEHOLDER
if not isinstance(text, str):
text = str(text)
return text if text.strip() else _EMPTY_TEXT_PLACEHOLDER
def _sanitize_replay_block(b: Dict[str, Any]) -> Optional[Dict[str, Any]]:
"""Strip output-only fields from a stored Anthropic content block so it is
valid as REQUEST input on replay.
@@ -1878,7 +1967,18 @@ def _sanitize_replay_block(b: Dict[str, Any]) -> Optional[Dict[str, Any]]:
return None
btype = b.get("type")
if btype == "text":
out: Dict[str, Any] = {"type": "text", "text": b.get("text", "")}
text_val = b.get("text", "")
# Bedrock and strict Anthropic-compatible endpoints reject text
# blocks where "text" is empty or whitespace-only (#69512). Drop the
# blank block (the caller relocates any cache_control it carried and
# falls back to a non-whitespace placeholder when nothing survives)
# rather than coercing in place — a coerced "(empty)" block would be
# model-visible noise next to surviving thinking/tool_use blocks.
# Type-safe: captured blocks can carry text=None from an invalid
# upstream payload, which a bare .strip() would crash on.
if not isinstance(text_val, str) or not text_val.strip():
return None
out: Dict[str, Any] = {"type": "text", "text": text_val}
# citations is input-valid ONLY when it's a non-empty list; the SDK
# emits citations=None on responses, which the input schema rejects.
cits = b.get("citations")
@@ -1966,9 +2066,17 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
parsed_args = {}
redacted_input_by_id[_sanitize_tool_id(tc.get("id", ""))] = parsed_args
replayed: List[Dict[str, Any]] = []
_relocated_replay_cache_control = None
_dropped_blank_text = False
for b in ordered_blocks:
clean = _sanitize_replay_block(b)
if clean is None:
if isinstance(b, dict) and b.get("type") == "text":
_dropped_blank_text = True
if isinstance(b, dict) and isinstance(b.get("cache_control"), dict):
# A dropped blank text block can still carry the cache
# breakpoint marker -- relocate it rather than losing it.
_relocated_replay_cache_control = b["cache_control"]
continue
if clean.get("type") == "tool_use":
# Override raw (un-redacted) input with the redacted copy when
@@ -1978,20 +2086,90 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
if redacted is not None:
clean["input"] = redacted
replayed.append(clean)
# When every text block was blank and nothing cacheable survived
# (e.g. signed thinking + a blank text block, or a SOLE blank
# cache-marked block), emit the non-whitespace placeholder so the
# replayed message stays schema-valid (#69512) and a relocated cache
# marker still has a carrier instead of being silently lost.
_has_cacheable_replay = any(
isinstance(b, dict) and b.get("type") in {"text", "tool_use"}
for b in replayed
)
if not _has_cacheable_replay and (
_dropped_blank_text or _relocated_replay_cache_control is not None
):
replayed.append({"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER})
if replayed:
if _relocated_replay_cache_control is not None:
_apply_assistant_cache_control_to_last_cacheable_block(
replayed, _relocated_replay_cache_control
)
_apply_assistant_cache_control_to_last_cacheable_block(
replayed, m.get("cache_control")
)
# apply_anthropic_cache_control marks an assistant turn with
# non-empty text by writing cache_control INTO ``content`` (see
# _apply_cache_marker's list branch), not at the top level. This
# branch rebuilds the message from ordered_blocks and never reads
# ``content``, so that marker would be dropped -- and because
# _can_carry_marker already counted this message as a carrier, the
# breakpoint is burned rather than relocated. #56195 covered the
# complementary shape (blank content -> top-level marker); this is
# the interleaved thinking + preamble-text + tool_use shape.
_inline_cc = None
_msg_content = m.get("content")
if isinstance(_msg_content, list):
for _blk in _msg_content:
if isinstance(_blk, dict) and isinstance(
_blk.get("cache_control"), dict
):
_inline_cc = _blk["cache_control"]
break
if _inline_cc is not None:
_apply_assistant_cache_control_to_last_cacheable_block(
replayed, _inline_cc
)
return {"role": "assistant", "content": replayed}
blocks = _extract_preserved_thinking_blocks(m)
# Cache markers dropped along with a blank block are relocated onto the
# last surviving cacheable block below (via
# _apply_assistant_cache_control_to_last_cacheable_block), rather than
# lost -- prompt_caching.py's _apply_cache_marker() sets cache_control
# directly on content[-1] for list content, so if that last part happens
# to be blank text, dropping it silently would lose the breakpoint.
_relocated_cache_control = None
if content:
if isinstance(content, list):
converted_content = _convert_content_to_anthropic(content)
if isinstance(converted_content, list):
blocks.extend(converted_content)
# Bedrock and strict Anthropic-compatible endpoints reject
# text blocks where "text" is empty or whitespace-only. The
# ordered-replay path enforces the same invariant via
# _sanitize_replay_block(). Type-safe against ANY invalid
# "text" value from an upstream payload -- None, or a
# truthy non-string like an int -- not just None: checking
# isinstance() first (rather than `blk.get("text") or ""`)
# means a non-string value is treated as blank/invalid
# instead of reaching .strip() and raising AttributeError.
for blk in converted_content:
_blk_text = blk.get("text") if isinstance(blk, dict) else None
if (
isinstance(blk, dict)
and blk.get("type") == "text"
and (not isinstance(_blk_text, str) or not _blk_text.strip())
):
if isinstance(blk.get("cache_control"), dict):
_relocated_cache_control = blk["cache_control"]
continue
blocks.append(blk)
else:
blocks.append({"type": "text", "text": str(content)})
# Scalar (non-list) content: a whitespace-only string is the
# same invalid-payload case as an empty list block -- drop it
# rather than emitting a blank text block.
text_str = str(content)
if text_str.strip():
blocks.append({"type": "text", "text": text_str})
for tc in m.get("tool_calls", []):
if not tc or not isinstance(tc, dict):
continue
@@ -2007,9 +2185,6 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
"name": fn.get("name", ""),
"input": parsed_args,
})
_apply_assistant_cache_control_to_last_cacheable_block(
blocks, m.get("cache_control")
)
# Kimi's /coding endpoint (Anthropic protocol) requires assistant
# tool-call messages to carry reasoning_content when thinking is
# enabled server-side. Preserve it as a thinking block so Kimi
@@ -2035,10 +2210,26 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
)
if isinstance(reasoning_content, str) and not _already_has_thinking:
blocks.insert(0, {"type": "thinking", "thinking": reasoning_content})
# Anthropic rejects empty assistant content
effective = blocks or content
if not effective or effective == "":
effective = [{"type": "text", "text": "(empty)"}]
# Anthropic rejects empty assistant content. IMPORTANT: fall back only
# to the placeholder, never to the raw `content` variable -- `content`
# is the UNFILTERED original message content, and can itself be exactly
# the blank/whitespace-only payload the filtering above just removed
# (a sole blank text block, or scalar whitespace with no tool_calls).
# `blocks or content` there would silently restore the invalid provider
# payload this function exists to prevent (#69512).
effective = blocks if blocks else [{"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}]
# Applied here (after the empty-fallback resolution) rather than
# earlier against `blocks` directly, so a cache_control relocated from
# a dropped blank block that was the ONLY block still lands on the
# (empty) placeholder instead of being silently lost when blocks was
# empty at the point the marker would otherwise have been applied.
if _relocated_cache_control is not None:
_apply_assistant_cache_control_to_last_cacheable_block(
effective, _relocated_cache_control
)
_apply_assistant_cache_control_to_last_cacheable_block(
effective, m.get("cache_control")
)
return {"role": "assistant", "content": effective}
@@ -2272,16 +2463,21 @@ def _manage_thinking_signatures(
replayed assistant tool-call messages. See hermes-agent#13848 (Kimi) and
hermes-agent#16748 (DeepSeek).
Nous Portal's ``/v1/messages`` route is the exception among third-party
hosts: it proxies Claude to Anthropic/Vertex/Bedrock and validates the
same signed thinking blocks. Sticky ``session_id`` keeps a conversation
on one upstream instance so those signatures stay warm — stripping them
here would 400 the first tool-loop turn ("thinking must be passed back").
Portal therefore takes the native Anthropic replay path below.
Mutates ``result`` in place.
"""
_THINKING_TYPES = frozenset(("thinking", "redacted_thinking"))
_is_third_party = _is_third_party_anthropic_endpoint(base_url)
# Kimi / DeepSeek share a contract: strip signed Anthropic blocks
# (neither upstream can validate Anthropic signatures), preserve unsigned
# ones synthesised from reasoning_content. See #13848, #16748.
_preserve_unsigned_thinking = (
_is_kimi_family_endpoint(base_url, model)
or _is_deepseek_anthropic_endpoint(base_url)
# Portal speaks Anthropic's thinking contract end-to-end; do not treat it
# as a signature-blind proxy even though the host is not anthropic.com.
_is_third_party = (
_is_third_party_anthropic_endpoint(base_url)
and not _is_nous_portal_endpoint(base_url)
)
last_assistant_idx = None
@@ -2294,8 +2490,12 @@ def _manage_thinking_signatures(
if m.get("role") != "assistant" or not isinstance(m.get("content"), list):
continue
if _preserve_unsigned_thinking:
# Kimi / DeepSeek: strip signed, preserve unsigned.
if _is_kimi_family_endpoint(base_url, model):
# Kimi does not enforce thinking signatures — replay as-is
# (shared cleanup below still strips cache markers + the internal flag).
pass
elif _is_deepseek_anthropic_endpoint(base_url):
# DeepSeek: strip signed, preserve unsigned.
new_content = []
for b in m["content"]:
if not isinstance(b, dict) or b.get("type") not in _THINKING_TYPES:
@@ -2395,6 +2595,24 @@ def _evict_old_screenshots(result: List[Dict[str, Any]]) -> None:
]
def _ensure_leading_user_turn(result: List[Dict[str, Any]]) -> None:
"""Anthropic requires messages[0] to have role=user.
After a second context compaction on the auto path the summary can be
emitted as role=assistant with nothing in front of it (the system prompt
lives outside messages[] or is extracted into the separate ``system``
param), so messages[0] ends up assistant and the Messages API rejects
the request with HTTP 400 — often masked by a misleading
"tool_use ids were found without tool_result blocks" error (#52160).
Mirror the Bedrock Converse adapter, which unconditionally prepends a
minimal user turn when the first message is not user
(convert_messages_to_converse).
"""
if result and result[0].get("role") != "user":
result.insert(0, {"role": "user", "content": [{"type": "text", "text": " "}]})
def convert_messages_to_anthropic(
messages: List[Dict],
base_url: str | None = None,
@@ -2453,6 +2671,7 @@ def convert_messages_to_anthropic(
_strip_orphaned_tool_blocks(result)
result = _merge_consecutive_roles(result)
_ensure_leading_user_turn(result)
_manage_thinking_signatures(result, base_url, model)
_evict_old_screenshots(result)
@@ -2516,7 +2735,12 @@ def build_anthropic_kwargs(
)
anthropic_tools = convert_tools_to_anthropic(tools) if tools else []
model = normalize_model_name(model, preserve_dots=preserve_dots)
# Nous Portal routes on its own catalog ids (``anthropic/claude-opus-4.8``);
# normalizing to the bare Anthropic slug would make the model unresolvable
# there. Skipping the call preserves the prefix AND the dots, so
# ``preserve_dots`` stays irrelevant for Portal.
if not _is_nous_portal_endpoint(base_url):
model = normalize_model_name(model, preserve_dots=preserve_dots)
# effective_max_tokens = output cap for this call (≠ total context window)
# Use the resolver helper so non-positive values (negative ints,
# fractional floats, NaN, non-numeric) fail locally with a clear error
@@ -2627,25 +2851,19 @@ def build_anthropic_kwargs(
# MiniMax Anthropic-compat endpoints support thinking (manual mode only,
# not adaptive). Haiku does NOT support extended thinking — skip entirely.
#
# Kimi's /coding endpoint speaks the Anthropic Messages protocol but has
# its own thinking semantics: when ``thinking.enabled`` is sent, Kimi
# validates the message history and requires every prior assistant
# tool-call message to carry OpenAI-style ``reasoning_content``. The
# Anthropic path never populates that field, and
# ``convert_messages_to_anthropic`` strips all Anthropic thinking blocks
# on third-party endpoints — so the request fails with HTTP 400
# "thinking is enabled but reasoning_content is missing in assistant
# tool call message at index N". Kimi's reasoning is driven server-side
# on the /coding route, so skip Anthropic's thinking parameter entirely
# for that host. (Kimi on chat_completions enables thinking via
# extra_body in the ChatCompletionsTransport — see #13503.)
# Kimi / Moonshot models also use adaptive thinking: their
# Anthropic-compatible endpoints (api.moonshot.cn/anthropic,
# api.kimi.com/coding) accept ``thinking.type="adaptive"`` +
# ``output_config.effort``, and the replay-validation 400s that
# originally motivated dropping the parameter (#13848) no longer
# occur. (Kimi on chat_completions enables thinking via extra_body
# in the ChatCompletionsTransport — see #13503.)
#
# On 4.7+ the `thinking.display` field defaults to "omitted", which
# silently hides reasoning text that Hermes surfaces in its CLI. We
# request "summarized" so the reasoning blocks stay populated — matching
# 4.6 behavior and preserving the activity-feed UX during long tool runs.
_is_kimi_coding = _is_kimi_family_endpoint(base_url, model)
if reasoning_config and isinstance(reasoning_config, dict) and not _is_kimi_coding:
if reasoning_config and isinstance(reasoning_config, dict):
if reasoning_config.get("enabled") is not False and "haiku" not in model.lower():
effort = str(reasoning_config.get("effort", "medium")).lower()
budget = THINKING_BUDGET.get(effort, 8000)
@@ -2761,6 +2979,8 @@ def create_anthropic_message(
*,
log_prefix: str = "",
prefer_stream: bool = True,
on_stream_event=None,
on_response=None,
) -> Any:
"""Create an Anthropic message, aggregating via stream when available.
@@ -2770,6 +2990,20 @@ def create_anthropic_message(
crash on ``.content``. Prefer ``messages.stream().get_final_message()`` to
match the main turn path, falling back to ``create()`` only for providers
that explicitly do not support streaming, such as restricted Bedrock roles.
``on_stream_event``: optional callable invoked once per streamed event
(best-effort, exceptions swallowed). Lets callers report forward progress
to liveness watchdogs — e.g. the auxiliary compression path ticking its
progress hook so a slow-but-generating summary model isn't treated as
hung. Only fires on the streaming path; the ``create()`` fallback has no
events to report.
``on_response``: optional callable invoked once with the underlying httpx
response before the message is aggregated (best-effort, exceptions
swallowed). Response *headers* carry out-of-band provider state that the
parsed ``Message`` drops — Nous Portal's ``x-nous-credits-*`` balance family
in particular. Only fires on the streaming path, which is the one the main
turn loop takes.
"""
sanitize_anthropic_kwargs(api_kwargs, log_prefix=log_prefix)
@@ -2780,6 +3014,26 @@ def create_anthropic_message(
stream_kwargs.pop("stream", None)
try:
with stream_fn(**stream_kwargs) as stream:
if callable(on_response):
try:
on_response(getattr(stream, "response", None))
except Exception:
logger.debug(
"%son_response callback failed",
log_prefix, exc_info=True,
)
if callable(on_stream_event):
# Consume the event stream manually so each event can
# tick the caller's progress callback; get_final_message
# then returns the accumulated snapshot.
for _event in stream:
try:
on_stream_event(_event)
except Exception:
logger.debug(
"%son_stream_event callback failed",
log_prefix, exc_info=True,
)
return stream.get_final_message()
except Exception as exc:
if not _is_stream_unavailable_error(exc):
+16
View File
@@ -66,3 +66,19 @@ def safe_schedule_threadsafe(
coro.close()
log.log(log_level, "%s: %s", log_message, exc)
return None
def consume_detached_task_result(task: "asyncio.Future[Any]") -> None:
"""Retrieve a detached task's result without surfacing cancellation.
Used as an ``add_done_callback`` on tasks that were cancelled and
detached (e.g. an adapter close path that swallows ``CancelledError``
past its teardown deadline). Observing ``task.exception()`` prevents
"exception was never retrieved" noise on the event loop; cancellation
and any terminal error are deliberately swallowed — the task's owner
already gave up on it.
"""
try:
task.exception()
except (asyncio.CancelledError, Exception):
pass
+1314 -119
View File
File diff suppressed because it is too large Load Diff
+204
View File
@@ -0,0 +1,204 @@
"""Single owner for backend identity and failure-scoped skip decisions.
Every fallback / dedup / skip / quarantine decision in Hermes ultimately asks
one question: **"is this candidate the same backend as the one that failed,
along the axis that failure invalidated?"** Before this module, that
question was re-implemented inline at six call sites across four subsystems,
each comparing whatever string was locally convenient (provider label,
provider+model, base_url+model, ...). Each incident fixed one site while the
others kept the bug: #22548 (same-shim aliases), #70893 (xai-oauth vs xai —
same host, distinct credential), #59561 (aux chain skipped sibling models),
#72468 (aux main-model safety net, same bug three weeks later), #62984 /
#54250 / #57584 (dedup ignoring base_url strands multi-endpoint pools).
The root insight: "provider" conflates three independent identity axes, and
each failure class invalidates a different one:
* **credential surface** — auth 401 / payment 402 kill everything sharing the
credential (every model, every host reached with that key/token).
* **endpoint** — DNS failure / connection refused kill everything behind the
URL, regardless of model or credential.
* **model deployment** — timeout / overload / rate limit / model-incompatible
kill ONE model's deployment. A sibling model behind the same URL is an
independent deployment (real incident: aux ``glm-5.2`` hung and timed out
while main ``macaron-v1-venti`` on the identical endpoint was serving
448K-token turns).
Call sites should build :class:`BackendIdentity` values, classify the failure
with :func:`classify_failure_scope`, and ask :func:`should_skip_candidate`.
Do not re-implement any comparison inline — extend THIS module instead.
"""
from __future__ import annotations
import logging
from dataclasses import dataclass
from enum import Enum
from typing import Optional
logger = logging.getLogger(__name__)
class FailureScope(Enum):
"""Which identity axis a failure invalidates."""
#: Timeout, overload/429, connection blip, model-incompatible, invalid
#: response: evidence against ONE model deployment only.
MODEL = "model"
#: Auth 401 / payment 402: evidence against the shared credential —
#: every model reached with it is equally dead.
CREDENTIAL = "credential"
#: DNS / connection-refused / unreachable host: evidence against the
#: endpoint — every model behind the URL is equally dead.
ENDPOINT = "endpoint"
#: Reason strings already used by auxiliary_client's except-chain, mapped to
#: scopes. Unknown reasons default to MODEL — the least-invalidating scope —
#: so an unrecognized failure never over-skips viable candidates.
_REASON_SCOPES = {
"auth error": FailureScope.CREDENTIAL,
"payment error": FailureScope.CREDENTIAL,
"rate limit": FailureScope.MODEL,
"model incompatible with route": FailureScope.MODEL,
"invalid provider response": FailureScope.MODEL,
"connection error": FailureScope.MODEL,
"timeout": FailureScope.MODEL,
}
def classify_failure_scope(reason: Optional[str]) -> FailureScope:
"""Map a human-readable failure reason to the identity axis it kills."""
return _REASON_SCOPES.get((reason or "").strip().lower(), FailureScope.MODEL)
def _norm_provider(value: Optional[str]) -> str:
return (value or "").strip().lower()
def _norm_model(value: Optional[str]) -> str:
return (value or "").strip().lower()
def _norm_base_url(value: Optional[str]) -> str:
return (value or "").strip().rstrip("/").lower()
@dataclass(frozen=True)
class BackendIdentity:
"""Normalized identity of one (provider, model, endpoint) deployment.
Empty fields mean "unknown" — comparisons treat an unknown axis as
non-distinguishing (it can neither prove sameness nor difference on its
own; the remaining axes decide).
"""
provider: str = ""
model: str = ""
base_url: str = ""
@classmethod
def build(
cls,
provider: Optional[str] = None,
model: Optional[str] = None,
base_url: Optional[str] = None,
) -> "BackendIdentity":
return cls(
provider=_norm_provider(provider),
model=_norm_model(model),
base_url=_norm_base_url(base_url),
)
def _both_first_class(a: BackendIdentity, b: BackendIdentity) -> bool:
"""True when both providers are distinct registered first-class providers.
Two different registry providers have distinct credential surfaces even
when they share an inference host (xai-oauth vs xai, openai-codex vs
openai-api) — #70893. Custom/shim aliases are NOT in the registry, so
two aliases pointing at one URL still count as the same backend (#22548).
"""
if not a.provider or not b.provider or a.provider == b.provider:
return False
try:
from hermes_cli.auth import PROVIDER_REGISTRY
return a.provider in PROVIDER_REGISTRY and b.provider in PROVIDER_REGISTRY
except Exception:
return False
def same_credential_surface(a: BackendIdentity, b: BackendIdentity) -> bool:
"""Do two identities share the credential a 401/402 just invalidated?
Conservative on purpose: an unprovable axis must answer "different"
(try the candidate — worst case one wasted RTT) rather than "same"
(skip — worst case stranded failover). Two distinct custom labels at
one URL may carry different per-entry api_keys, so a shared URL alone
never proves a shared credential; it is only used as a weak signal
when a provider label is missing entirely.
"""
if a.provider and b.provider:
# Same label = same configured credential. Different labels =
# different credential config (first-class registry providers
# explicitly so — #70893; custom entries can each carry their own
# api_key, so sameness is unprovable and we must not skip).
return a.provider == b.provider
# Provider unknown on a side: same explicit URL is the best signal left.
return bool(a.base_url and a.base_url == b.base_url)
def same_endpoint(a: BackendIdentity, b: BackendIdentity) -> bool:
"""Do two identities sit behind the endpoint that just went unreachable?"""
if a.base_url and b.base_url:
return a.base_url == b.base_url
# An unknown base_url inherits the provider default → same provider
# label implies the same default endpoint.
return bool(a.provider and a.provider == b.provider)
def same_deployment(a: BackendIdentity, b: BackendIdentity) -> bool:
"""Are these the exact same model deployment (the thing a timeout kills)?
Provider+model must match; the base_url axis distinguishes only when BOTH
sides carry an explicit URL (#62984: same provider+model on two different
explicit URLs is two deployments — a pool). A side with an unknown URL
inherits the provider default and cannot prove difference.
"""
if not (a.provider and b.provider and a.provider == b.provider):
# Same-host different-label shims: same URL + same model IS the same
# deployment even when the alias labels differ (#22548) — unless both
# labels are first-class registry providers (#70893).
if (
a.base_url
and a.base_url == b.base_url
and a.model
and a.model == b.model
and not _both_first_class(a, b)
):
return True
return False
if not (a.model and b.model and a.model == b.model):
return False
if a.base_url and b.base_url and a.base_url != b.base_url:
return False # distinct explicit endpoints — a pool, not a dup
return True
def should_skip_candidate(
candidate: BackendIdentity,
failed: BackendIdentity,
scope: FailureScope = FailureScope.MODEL,
) -> bool:
"""THE skip predicate: would trying ``candidate`` just repeat the failure?
True when the candidate is the same backend as ``failed`` along the axis
``scope`` says the failure invalidated. Every fallback/dedup/skip site
must call this instead of comparing labels inline.
"""
if scope is FailureScope.CREDENTIAL:
return same_credential_surface(candidate, failed)
if scope is FailureScope.ENDPOINT:
return same_endpoint(candidate, failed)
return same_deployment(candidate, failed)
+30 -12
View File
@@ -70,8 +70,8 @@ def _resolve_review_runtime(agent: Any) -> Dict[str, Any]:
"routed": False,
}
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception:
return parent
aux = cfg.get("auxiliary", {}) if isinstance(cfg.get("auxiliary"), dict) else {}
@@ -209,7 +209,10 @@ _SKILL_REVIEW_PROMPT = (
"conversation for skills the user loaded via /skill-name or you "
"read via skill_view. If any of them covers the territory of the "
"new learning, PATCH that one first. It is the skill that was in "
"play, so it's the right one to extend.\n"
"play, so it's the right one to extend — but only if it is "
"curator-managed. Bundled, hub, pinned, and user-owned skills are "
"off-limits to you no matter how relevant (see Protected skills "
"below); for those, fall through to the next option.\n"
" 2. UPDATE AN EXISTING UMBRELLA (via skills_list + skill_view). "
"If no loaded skill fits but an existing class-level skill does, "
"patch it. Add a subsection, a pitfall, or broaden a trigger.\n"
@@ -251,10 +254,18 @@ _SKILL_REVIEW_PROMPT = (
"Protected skills (DO NOT edit these):\n"
" • Bundled skills (shipped with Hermes, e.g. 'hermes-agent').\n"
" • Hub-installed skills (installed via 'hermes skills install').\n"
"Pinned skills (marked via 'hermes curator pin') CAN be improved — "
"pin only blocks deletion/archive/consolidation by the curator, not "
"content updates. Patch them when a pitfall or missing step turns up, "
"same as any other agent-created skill.\n"
" • Skills in skills.external_dirs (externally owned).\n"
" • PINNED skills (marked via 'hermes curator pin'). You are an "
"autonomous no-user-present actor, so pin blocks your writes too — "
"content updates included. Only the user, in a foreground session, "
"can change a pinned skill.\n"
" • USER-OWNED skills — anything not curator-managed. A skill the "
"user hand-wrote, installed by URL, or asked a foreground agent to "
"create is theirs, not yours; your writes to it WILL be refused. "
"This includes skills that were loaded or consulted this session: "
"being in play does not make one yours to edit. If such a skill is "
"wrong or outdated, say so in your reply and recommend "
"'hermes curator adopt <name>' — do not try to patch it.\n"
"If the only skills that need updating are protected, say\n"
"'Nothing to save.' and stop.\n\n"
"Do NOT capture (these become persistent self-imposed constraints "
@@ -309,7 +320,9 @@ _COMBINED_REVIEW_PROMPT = (
" 1. UPDATE A CURRENTLY-LOADED SKILL. Check what skills were "
"loaded via /skill-name or skill_view in the conversation. If one "
"of them covers the learning, PATCH it first. It was in play; "
"it's the right place.\n"
"it's the right place — provided it is curator-managed. Protected "
"and user-owned skills are off-limits however relevant; fall "
"through when one of those is the best fit.\n"
" 2. UPDATE AN EXISTING UMBRELLA (skills_list + skill_view to "
"find the right one). Patch it.\n"
" 3. ADD A SUPPORT FILE under an existing umbrella via "
@@ -337,10 +350,15 @@ _COMBINED_REVIEW_PROMPT = (
"Protected skills (DO NOT edit these):\n"
" • Bundled skills (shipped with Hermes, e.g. 'hermes-agent').\n"
" • Hub-installed skills (installed via 'hermes skills install').\n"
"Pinned skills (marked via 'hermes curator pin') CAN be improved — "
"pin only blocks deletion/archive/consolidation by the curator, not "
"content updates. Patch them when a pitfall or missing step turns up, "
"same as any other agent-created skill.\n"
" • Skills in skills.external_dirs (externally owned).\n"
" • PINNED skills (marked via 'hermes curator pin'). Pin blocks "
"autonomous writes entirely — content updates included — because no "
"user is present to consent. Only a foreground session can change one.\n"
" • USER-OWNED skills — anything not curator-managed (hand-written, "
"URL-installed, or created by a foreground agent at the user's "
"request). Your writes to these WILL be refused, including to skills "
"loaded or consulted this session. If one is wrong, say so in your "
"reply and recommend 'hermes curator adopt <name>' instead.\n"
"If the only skills that need updating are protected, say\n"
"'Nothing to save.' and stop.\n\n"
"Do NOT capture as skills (these become persistent self-imposed "
+131
View File
@@ -0,0 +1,131 @@
"""System-battery read-out for the CLI/TUI status bar.
Reads the host battery through ``psutil`` (already a Hermes dependency) and
exposes a compact, colour-coded label. Everything degrades to "unavailable"
when there is no battery (desktops, servers, VMs) or when the read fails, so
callers can render the result unconditionally and simply show nothing.
The status bar repaints often (every keystroke and on a ~1s idle refresh), so
:func:`read_battery` memoises the last reading for a few seconds instead of
hitting ``psutil`` on every frame.
"""
from __future__ import annotations
import time
from dataclasses import dataclass
from typing import Optional
@dataclass(frozen=True)
class BatteryStatus:
"""A single battery reading.
``available`` is False on machines without a battery (or when the read
failed). ``percent`` is clamped to 0-100. ``plugged`` is True when on AC
power, False on battery, and None when the platform can't tell.
"""
available: bool
percent: Optional[int] = None
plugged: Optional[bool] = None
@property
def charging(self) -> bool:
return bool(self.plugged)
UNAVAILABLE = BatteryStatus(available=False)
# Colour buckets, mirroring the status-bar context styles but inverted (a full
# battery is "good", an empty one is "critical").
CATEGORY_GOOD = "good"
CATEGORY_WARN = "warn"
CATEGORY_BAD = "bad"
CATEGORY_CRITICAL = "critical"
CATEGORY_DIM = "dim"
_CACHE_TTL_SECONDS = 8.0
_cache: Optional[tuple[float, BatteryStatus]] = None
def _read_battery_uncached() -> BatteryStatus:
try:
import psutil
except Exception:
return UNAVAILABLE
# ``sensors_battery`` is missing on some platforms/builds of psutil.
reader = getattr(psutil, "sensors_battery", None)
if reader is None:
return UNAVAILABLE
try:
batt = reader()
except Exception:
return UNAVAILABLE
if batt is None:
return UNAVAILABLE
percent: Optional[int] = None
raw_percent = getattr(batt, "percent", None)
if raw_percent is not None:
try:
percent = max(0, min(100, int(round(float(raw_percent)))))
except (TypeError, ValueError):
percent = None
plugged = getattr(batt, "power_plugged", None)
if plugged is not None:
plugged = bool(plugged)
return BatteryStatus(available=True, percent=percent, plugged=plugged)
def read_battery(use_cache: bool = True) -> BatteryStatus:
"""Return the current battery status (cached for a few seconds)."""
global _cache
if use_cache and _cache is not None:
ts, cached = _cache
if time.monotonic() - ts < _CACHE_TTL_SECONDS:
return cached
status = _read_battery_uncached()
_cache = (time.monotonic(), status)
return status
def clear_cache() -> None:
"""Drop the memoised reading (used by tests)."""
global _cache
_cache = None
def battery_category(status: BatteryStatus) -> str:
"""Bucket a reading into a colour category: good/warn/bad/critical/dim."""
if not status.available or status.percent is None:
return CATEGORY_DIM
# On AC power the level isn't a concern — always read as healthy.
if status.charging:
return CATEGORY_GOOD
pct = status.percent
if pct <= 10:
return CATEGORY_CRITICAL
if pct <= 20:
return CATEGORY_BAD
if pct <= 50:
return CATEGORY_WARN
return CATEGORY_GOOD
def battery_glyph(status: BatteryStatus) -> str:
"""Return the leading glyph: a bolt while charging, else a battery."""
return "\u26a1" if status.charging else "\U0001f50b" # ⚡ / 🔋
def format_battery(status: BatteryStatus) -> str:
"""Return a compact label like ``🔋 82%`` / ``⚡ 82%`` (empty if N/A)."""
if not status.available or status.percent is None:
return ""
return f"{battery_glyph(status)} {status.percent}%"
+255 -33
View File
@@ -433,6 +433,29 @@ def _model_supports_tool_use(model_id: str) -> bool:
return not any(pattern in model_lower for pattern in _NON_TOOL_CALLING_PATTERNS)
# ---------------------------------------------------------------------------
# Prompt-cache capability detection (Converse API cachePoint)
# ---------------------------------------------------------------------------
# Claude on Bedrock already gets prompt caching through the AnthropicBedrock
# SDK path (see is_anthropic_bedrock_model / runtime_provider.py's dual-path
# routing) — it never reaches build_converse_kwargs unless bearer-token auth
# forces the Converse path (#28156). This allowlist covers the Converse API
# itself: sending an unsupported model a cachePoint block raises a
# ValidationException, so — like _model_supports_tool_use but inverted —
# unknown models default to NOT receiving cache markers until confirmed.
# Ref: https://docs.aws.amazon.com/bedrock/latest/userguide/prompt-caching.html
_CACHE_POINT_PATTERNS = [
"anthropic.claude", # bearer-token fallback path
"amazon.nova",
]
def _model_supports_prompt_cache(model_id: str) -> bool:
"""Return True if the model accepts a Converse API cachePoint block."""
model_lower = model_id.lower()
return any(pattern in model_lower for pattern in _CACHE_POINT_PATTERNS)
def is_anthropic_bedrock_model(model_id: str) -> bool:
"""Return True if the model is an Anthropic Claude model on Bedrock.
@@ -448,7 +471,10 @@ def is_anthropic_bedrock_model(model_id: str) -> bool:
"""
model_lower = model_id.lower()
# Strip regional prefix if present
for prefix in ("us.", "global.", "eu.", "ap.", "jp."):
for prefix in (
"global.", "us.", "eu.", "apac.", "ap.", "au.", "jp.",
"ca.", "sa.", "me.", "af.",
):
if model_lower.startswith(prefix):
model_lower = model_lower[len(prefix):]
break
@@ -490,6 +516,26 @@ def convert_tools_to_converse(tools: List[Dict]) -> List[Dict]:
return result
# Bedrock's Converse API rejects any text content block whose text is empty
# OR whitespace-only (ValidationException: "text content blocks must contain
# non-whitespace text"). A lone space is whitespace and is rejected too — the
# placeholder MUST itself be non-whitespace. Ref: issue #9486.
_EMPTY_TEXT_PLACEHOLDER = "(empty)"
def _safe_text(text) -> str:
"""Return ``text`` if it's non-whitespace, else a non-whitespace placeholder.
Handles None, empty string, and whitespace-only string (spaces, tabs,
newlines) — all of which Bedrock's Converse API rejects as text content.
"""
if text is None:
return _EMPTY_TEXT_PLACEHOLDER
if not isinstance(text, str):
text = str(text)
return text if text.strip() else _EMPTY_TEXT_PLACEHOLDER
def _convert_content_to_converse(content) -> List[Dict]:
"""Convert OpenAI message content (string or list) to Converse content blocks.
@@ -497,26 +543,27 @@ def _convert_content_to_converse(content) -> List[Dict]:
- Plain text strings → [{"text": "..."}]
- Content arrays with text/image_url parts → mixed text/image blocks
Filters out empty text blocks — Bedrock's Converse API rejects messages
where a text content block has an empty ``text`` field (ValidationException:
"text content blocks must be non-empty"). Ref: issue #9486.
Replaces empty/whitespace-only text blocks with a non-whitespace
placeholder — Bedrock's Converse API rejects messages where a text
content block is empty or whitespace-only (ValidationException:
"text content blocks must contain non-whitespace text"). Ref: issue #9486.
"""
if content is None:
return [{"text": " "}]
return [{"text": _safe_text(content)}]
if isinstance(content, str):
return [{"text": content}] if content.strip() else [{"text": " "}]
return [{"text": _safe_text(content)}]
if isinstance(content, list):
blocks = []
for part in content:
if isinstance(part, str):
blocks.append({"text": part})
blocks.append({"text": _safe_text(part)})
continue
if not isinstance(part, dict):
continue
part_type = part.get("type", "")
if part_type == "text":
text = part.get("text", "")
blocks.append({"text": text if text else " "})
blocks.append({"text": _safe_text(text)})
elif part_type == "image_url":
image_url = part.get("image_url", {})
url = image_url.get("url", "") if isinstance(image_url, dict) else ""
@@ -547,8 +594,8 @@ def _convert_content_to_converse(content) -> List[Dict]:
# Remote URL — Converse doesn't support URLs directly,
# include as text reference for the model.
blocks.append({"text": f"[Image: {url}]"})
return blocks if blocks else [{"text": " "}]
return [{"text": str(content)}]
return blocks if blocks else [{"text": _EMPTY_TEXT_PLACEHOLDER}]
return [{"text": _safe_text(content)}]
def convert_messages_to_converse(
@@ -578,14 +625,18 @@ def convert_messages_to_converse(
content = msg.get("content")
if role == "system":
# System messages become the system prompt
# System messages become the system prompt. Blank/whitespace-only
# parts are dropped entirely (not placeholder-filled) since a
# system prompt made up of only placeholder text is meaningless.
if isinstance(content, str) and content.strip():
system_blocks.append({"text": content})
elif isinstance(content, list):
for part in content:
if isinstance(part, dict) and part.get("type") == "text":
system_blocks.append({"text": part.get("text", "")})
elif isinstance(part, str):
text = part.get("text", "")
if isinstance(text, str) and text.strip():
system_blocks.append({"text": text})
elif isinstance(part, str) and part.strip():
system_blocks.append({"text": part})
continue
@@ -596,7 +647,7 @@ def convert_messages_to_converse(
tool_result_block = {
"toolResult": {
"toolUseId": tool_call_id,
"content": [{"text": result_content}],
"content": [{"text": _safe_text(result_content)}],
}
}
# In Converse, tool results go in a "user" role message
@@ -635,7 +686,7 @@ def convert_messages_to_converse(
})
if not content_blocks:
content_blocks = [{"text": " "}]
content_blocks = [{"text": _EMPTY_TEXT_PLACEHOLDER}]
# Merge with previous assistant message if needed (strict alternation)
if converse_msgs and converse_msgs[-1]["role"] == "assistant":
@@ -661,11 +712,11 @@ def convert_messages_to_converse(
# Converse requires the first message to be from the user
if converse_msgs and converse_msgs[0]["role"] != "user":
converse_msgs.insert(0, {"role": "user", "content": [{"text": " "}]})
converse_msgs.insert(0, {"role": "user", "content": [{"text": _EMPTY_TEXT_PLACEHOLDER}]})
# Converse requires the last message to be from the user
if converse_msgs and converse_msgs[-1]["role"] != "user":
converse_msgs.append({"role": "user", "content": [{"text": " "}]})
converse_msgs.append({"role": "user", "content": [{"text": _EMPTY_TEXT_PLACEHOLDER}]})
return (system_blocks if system_blocks else None, converse_msgs)
@@ -736,14 +787,22 @@ def normalize_converse_response(response: Dict) -> SimpleNamespace:
reasoning_content="\n\n".join(reasoning_parts) if reasoning_parts else None,
)
# Build usage stats
# Build usage stats. Converse's inputTokens excludes cache read/write
# tokens (unlike OpenAI's prompt_tokens, which includes them) — restore
# the OpenAI-style "total includes cache" convention here so downstream
# normalize_usage() can subtract them back out consistently, and surface
# the Anthropic-named fields it already falls back to for cache reads.
usage_data = response.get("usage", {})
input_tokens = usage_data.get("inputTokens", 0)
cache_read_tokens = usage_data.get("cacheReadInputTokens", 0)
cache_write_tokens = usage_data.get("cacheWriteInputTokens", 0)
output_tokens = usage_data.get("outputTokens", 0)
usage = SimpleNamespace(
prompt_tokens=usage_data.get("inputTokens", 0),
completion_tokens=usage_data.get("outputTokens", 0),
total_tokens=(
usage_data.get("inputTokens", 0) + usage_data.get("outputTokens", 0)
),
prompt_tokens=input_tokens + cache_read_tokens + cache_write_tokens,
completion_tokens=output_tokens,
total_tokens=input_tokens + cache_read_tokens + cache_write_tokens + output_tokens,
cache_read_input_tokens=cache_read_tokens,
cache_creation_input_tokens=cache_write_tokens,
)
finish_reason = _converse_stop_reason_to_openai(stop_reason)
@@ -789,6 +848,7 @@ def stream_converse_with_callbacks(
on_tool_start=None,
on_reasoning_delta=None,
on_interrupt_check=None,
on_event=None,
) -> SimpleNamespace:
"""Process a Bedrock ConverseStream event stream with real-time callbacks.
@@ -808,6 +868,12 @@ def stream_converse_with_callbacks(
on supported models (Claude 4.6+).
on_interrupt_check: Called on each event. Should return True if the
agent has been interrupted and streaming should stop.
on_event: Called once at the top of the loop body for EVERY yielded
Bedrock event (text/tool-input/reasoning/metadata deltas alike),
before any branching. Provides a wire-level liveness signal so an
external watchdog can distinguish "still receiving events" from
"stream wedged with no data". Errors raised by the callback are
swallowed so a liveness hook can never abort the stream.
Returns:
An OpenAI-compatible SimpleNamespace response, identical in shape to
@@ -823,6 +889,15 @@ def stream_converse_with_callbacks(
usage_data: Dict[str, int] = {}
for event in event_stream.get("stream", []):
# Wire-level liveness signal: fire on EVERY yielded event (text, tool
# input, reasoning, metadata) before branching so an external watchdog
# can tell a still-flowing stream from a wedged one. Best-effort — a
# liveness callback must never be able to abort the stream.
if on_event is not None:
try:
on_event()
except Exception:
pass
# Check for interrupt
if on_interrupt_check and on_interrupt_check():
break
@@ -892,6 +967,8 @@ def stream_converse_with_callbacks(
usage_data = {
"inputTokens": meta_usage.get("inputTokens", 0),
"outputTokens": meta_usage.get("outputTokens", 0),
"cacheReadInputTokens": meta_usage.get("cacheReadInputTokens", 0),
"cacheWriteInputTokens": meta_usage.get("cacheWriteInputTokens", 0),
}
# Flush remaining text
@@ -905,12 +982,16 @@ def stream_converse_with_callbacks(
reasoning_content="\n\n".join(reasoning_parts) if reasoning_parts else None,
)
input_tokens = usage_data.get("inputTokens", 0)
cache_read_tokens = usage_data.get("cacheReadInputTokens", 0)
cache_write_tokens = usage_data.get("cacheWriteInputTokens", 0)
output_tokens = usage_data.get("outputTokens", 0)
usage = SimpleNamespace(
prompt_tokens=usage_data.get("inputTokens", 0),
completion_tokens=usage_data.get("outputTokens", 0),
total_tokens=(
usage_data.get("inputTokens", 0) + usage_data.get("outputTokens", 0)
),
prompt_tokens=input_tokens + cache_read_tokens + cache_write_tokens,
completion_tokens=output_tokens,
total_tokens=input_tokens + cache_read_tokens + cache_write_tokens + output_tokens,
cache_read_input_tokens=cache_read_tokens,
cache_creation_input_tokens=cache_write_tokens,
)
finish_reason = _converse_stop_reason_to_openai(stop_reason)
@@ -949,6 +1030,7 @@ def build_converse_kwargs(
Converts OpenAI-format inputs to Converse API parameters.
"""
system_prompt, converse_messages = convert_messages_to_converse(messages)
cache_enabled = _model_supports_prompt_cache(model)
kwargs: Dict[str, Any] = {
"modelId": model,
@@ -959,6 +1041,8 @@ def build_converse_kwargs(
}
if system_prompt:
if cache_enabled:
system_prompt = system_prompt + [{"cachePoint": {"type": "default"}}]
kwargs["system"] = system_prompt
from agent.anthropic_adapter import _forbids_sampling_params
@@ -982,6 +1066,8 @@ def build_converse_kwargs(
# Strip tools for known non-tool-calling models and warn the user.
# Ref: PR #7920 feedback from @ptlally, pattern from PR #4346.
if _model_supports_tool_use(model):
if cache_enabled:
converse_tools = converse_tools + [{"cachePoint": {"type": "default"}}]
kwargs["toolConfig"] = {"tools": converse_tools}
else:
logger.warning(
@@ -989,6 +1075,14 @@ def build_converse_kwargs(
"The agent will operate in text-only mode.", model
)
if cache_enabled and len(converse_messages) >= 2:
# Checkpoint everything up to (not including) the newest turn, so the
# marker survives unchanged across requests as only the tail grows —
# mirroring the Anthropic system_and_3 strategy in prompt_caching.py.
content = converse_messages[-2].get("content")
if isinstance(content, list) and content:
content.append({"cachePoint": {"type": "default"}})
if guardrail_config:
kwargs["guardrailConfig"] = guardrail_config
@@ -1305,9 +1399,24 @@ def classify_bedrock_error(error_message: str) -> str:
# detection is unavailable.
BEDROCK_CONTEXT_LENGTHS: Dict[str, int] = {
# Anthropic Claude models on Bedrock
"anthropic.claude-opus-4-6": 200_000,
"anthropic.claude-sonnet-4-6": 200_000,
# Anthropic Claude models on Bedrock.
# Context windows per Anthropic's official models comparison
# (https://platform.claude.com/docs/en/about-claude/models/overview).
# Fable / Sonnet 5 / Opus 4.8 / 4.7 / 4.6 / Sonnet 4.6 have 1M generally
# available (no beta header required as of April 2026). Sonnet 4.5 and
# Sonnet 4 had their `context-1m-2025-08-07` beta retired on
# April 30, 2026, so they are standard 200K; Haiku 4.5 is 200K.
# These 1M entries must match agent/model_metadata.py
# DEFAULT_CONTEXT_LENGTHS or the agent compresses context prematurely.
# Keys are matched by longest-substring, so the versioned 4-6/4-7/4-8
# entries win over the generic "anthropic.claude-opus-4" fallback.
"anthropic.claude-fable-5": 1_000_000,
"anthropic.claude-fable": 1_000_000,
"anthropic.claude-sonnet-5": 1_000_000,
"anthropic.claude-opus-4-8": 1_000_000,
"anthropic.claude-opus-4-7": 1_000_000,
"anthropic.claude-opus-4-6": 1_000_000,
"anthropic.claude-sonnet-4-6": 1_000_000,
"anthropic.claude-sonnet-4-5": 200_000,
"anthropic.claude-haiku-4-5": 200_000,
"anthropic.claude-opus-4": 200_000,
@@ -1334,9 +1443,22 @@ BEDROCK_CONTEXT_LENGTHS: Dict[str, int] = {
# Default for unknown Bedrock models
BEDROCK_DEFAULT_CONTEXT_LENGTH = 128_000
# Probe tiers (in tokens). We send a request padded just past each tier and
# read the real window from Bedrock's length-validation error. Two reasons
# this is tiered rather than one giant request:
# 1. A wildly oversized payload (e.g. 5M tokens) makes Bedrock return an
# opaque InternalServerException after retries instead of a clean
# ValidationException — so we must stay within a sane overage.
# 2. Stepping up lets us discover larger windows (2M+) without over-padding
# smaller ones.
# Each tier value is the *padding target*; the error reports the true maximum,
# which is what we actually return.
_BEDROCK_PROBE_TIERS = (1_300_000, 2_200_000)
_WORDS_PER_TOKEN = 0.9 # conservative: ensures the padded prompt clears the tier
def get_bedrock_context_length(model_id: str) -> int:
"""Look up the context window size for a Bedrock model.
def _static_bedrock_context_length(model_id: str) -> int:
"""Longest-substring-match lookup against the static fallback table.
Uses substring matching so versioned IDs like
``anthropic.claude-sonnet-4-6-20250514-v1:0`` resolve correctly.
@@ -1349,3 +1471,103 @@ def get_bedrock_context_length(model_id: str) -> int:
best_key = key
best_val = val
return best_val
def probe_bedrock_context_length(model_id: str, region: str) -> Optional[int]:
"""Discover a Bedrock model's real context window by provoking a length error.
Bedrock does not expose the context window via any metadata API
(``get-foundation-model`` omits it, ``Converse`` metrics omit it,
``CountTokens`` is unsupported on several models). The only authoritative
source is the ``ValidationException`` raised when a prompt exceeds the
window:
"The model returned the following errors: prompt is too long:
1300032 tokens > 1000000 maximum"
Length validation happens *before* inference, so an oversized request is
rejected immediately and cheaply — no tokens are generated and no input is
actually processed. We pad a request just past each tier in
``_BEDROCK_PROBE_TIERS`` and parse the reported ``maximum``. Tiers exist
because (a) a *wildly* oversized payload makes Bedrock fail with an opaque
InternalServerException instead of a clean length error, and (b) stepping
up discovers larger windows without over-padding smaller ones.
Returns the detected window, or ``None`` if the probe could not run
(missing credentials, network error, or no parseable limit) so the caller
can fall back to the static table.
"""
try:
from agent.model_metadata import parse_context_limit_from_error
except ImportError: # pragma: no cover — same package
return None
try:
client = _get_bedrock_runtime_client(region)
except Exception as exc: # boto3 missing / credential resolution failure
logger.debug("Bedrock context probe skipped for %s: %s", model_id, exc)
return None
last_error = ""
for tier_tokens in _BEDROCK_PROBE_TIERS:
pad_words = int(tier_tokens / _WORDS_PER_TOKEN)
oversized = "data " * pad_words
try:
client.converse(
modelId=model_id,
messages=[{"role": "user", "content": [{"text": oversized}]}],
inferenceConfig={"maxTokens": 8},
)
# Accepted a prompt this large → the window is at least this tier.
# Returning the tier as a lower bound is safe and avoids inventing
# a number we can't confirm.
logger.debug(
"Bedrock context probe for %s accepted ~%s-token prompt; "
"window is at least that", model_id, f"{tier_tokens:,}",
)
return tier_tokens
except Exception as exc:
msg = str(exc)
last_error = msg
limit = parse_context_limit_from_error(msg)
if limit and limit >= 1024:
logger.info(
"Probed Bedrock context window for %s: %s tokens",
model_id, f"{limit:,}",
)
return limit
# No parseable limit at this tier (opaque server error, auth,
# throttle). Try the next, smaller-overage strategy is N/A here —
# tiers ascend — so just continue; if all fail we return None.
continue
logger.debug(
"Bedrock context probe for %s returned no parseable limit: %s",
model_id, last_error[:200],
)
return None
def get_bedrock_context_length(model_id: str, region: str = "", probe: bool = True) -> int:
"""Resolve the context window for a Bedrock model.
Resolution order:
1. Live probe against Bedrock (authoritative; cached by the caller).
2. Static fallback table (longest-substring match).
3. Conservative default.
The static table is intentionally a *fallback*, not the primary source:
AWS ships new model versions (opus-4-7, opus-4-8, ...) faster than the
table can track, and a stale entry silently caps the window (e.g. a
1M-token Opus pinned to 200K via an ``opus-4`` substring match). The
probe asks Bedrock directly so every model — current or future — gets its
real window with no table maintenance.
``probe=False`` (or an empty ``region``) skips the network call and uses
the static table only — used by pure-offline/display code paths.
"""
if probe and region:
probed = probe_bedrock_context_length(model_id, region)
if probed:
return probed
return _static_bedrock_context_length(model_id)
+124
View File
@@ -0,0 +1,124 @@
"""Provider-agnostic billing/credit recovery links.
Maps a billing-classified failure onto a recovery link + label. *Detection*
is not done here — that is :mod:`agent.error_classifier`
(``FailoverReason.billing``), the single source of truth for "credit wall vs.
rate limit / auth / transport". The resulting :class:`BillingBlock` rides the
turn result and the gateway ``message.complete`` event so every surface (CLI,
TUI, desktop) renders one structured signal instead of re-parsing error text.
"""
from __future__ import annotations
from dataclasses import asdict, dataclass
from typing import Optional
from utils import base_url_host_matches
@dataclass
class BillingBlock:
"""Structured billing-wall descriptor shared across every surface.
``is_nous`` is the routing bit: Nous has a first-class in-app billing surface
(desktop Settings → Billing, TUI/CLI ``/topup``), so surfaces prefer that over
``billing_url``; third-party providers have no in-app flow, so ``billing_url``
is the deep link the user actually needs.
"""
provider: str
provider_label: str
model: str
billing_url: Optional[str]
is_nous: bool
message: str
def to_dict(self) -> dict:
return asdict(self)
@dataclass(frozen=True)
class _Provider:
label: str
url: str
slugs: tuple[str, ...]
hosts: tuple[str, ...] = ()
# Single source of truth: internal slug(s) + base_url host(s) → billing page.
# Curated "add credits / manage billing" landing pages, not marketing homes.
# Hosts back the OpenAI-compatible fallback where the slug is a generic bucket
# (e.g. "openai_compatible") but base_url reveals the real upstream. An unknown
# provider degrades to a readable label with no invented URL.
_PROVIDERS: tuple[_Provider, ...] = (
_Provider("OpenAI", "https://platform.openai.com/settings/organization/billing", ("openai",), ("api.openai.com",)),
_Provider("Anthropic", "https://console.anthropic.com/settings/billing", ("anthropic",), ("api.anthropic.com",)),
_Provider("OpenRouter", "https://openrouter.ai/settings/credits", ("openrouter",), ("openrouter.ai",)),
_Provider("xAI", "https://console.x.ai/team/default/billing", ("xai", "xai-oauth"), ("api.x.ai",)),
_Provider("DeepSeek", "https://platform.deepseek.com/top_up", ("deepseek",), ("api.deepseek.com",)),
_Provider("Groq", "https://console.groq.com/settings/billing", ("groq",), ("api.groq.com",)),
_Provider("Mistral", "https://console.mistral.ai/billing", ("mistral",), ("api.mistral.ai",)),
_Provider("Together AI", "https://api.together.ai/settings/billing", ("together",), ("api.together.ai", "api.together.xyz")),
_Provider("Fireworks AI", "https://fireworks.ai/account/billing", ("fireworks",), ("fireworks.ai",)),
_Provider("Perplexity", "https://www.perplexity.ai/settings/api", ("perplexity",), ("perplexity.ai",)),
_Provider("Google AI", "https://aistudio.google.com/app/billing", ("google", "gemini"), ("generativelanguage.googleapis.com",)),
_Provider("Cohere", "https://dashboard.cohere.com/billing", ("cohere",)),
_Provider("Moonshot AI", "https://platform.moonshot.ai/console/pay", ("moonshot",)),
_Provider("NVIDIA", "https://build.nvidia.com/settings/billing", ("nvidia",)),
)
_BY_SLUG: dict[str, _Provider] = {slug: p for p in _PROVIDERS for slug in p.slugs}
def is_nous_inference_route(provider: str, base_url: str) -> bool:
"""True when the failing route is the Nous-managed inference gateway."""
if (provider or "").strip().lower() == "nous":
return True
return base_url_host_matches(str(base_url or ""), "inference-api.nousresearch.com")
def _nous_billing_url() -> Optional[str]:
"""Best-effort Nous portal billing URL (text-surface fallback; Nous prefers the in-app flow)."""
try:
from hermes_cli.nous_account import nous_portal_billing_url
return nous_portal_billing_url(None)
except Exception:
return "https://portal.nousresearch.com/billing"
def _resolve_provider_link(slug: str, base_url: str) -> tuple[str, Optional[str]]:
"""Resolve ``(label, url)``: exact slug → base_url host → readable-label fallback."""
hit = _BY_SLUG.get(slug)
if hit:
return hit.label, hit.url
base = str(base_url or "")
for p in _PROVIDERS:
if any(base_url_host_matches(base, host) for host in p.hosts):
return p.label, p.url
return slug.replace("_", " ").replace("-", " ").strip().title() or "your provider", None
def build_billing_block(
*,
provider: str,
base_url: str,
model: str,
message: str = "",
) -> BillingBlock:
"""Build the billing descriptor for a billing-classified failure.
``message`` is the guidance already assembled by the agent loop
(:func:`agent.conversation_loop._billing_or_entitlement_message`), carried
through unchanged so every surface shows identical copy.
"""
slug = (provider or "").strip().lower()
model = (model or "").strip()
if is_nous_inference_route(slug, base_url):
return BillingBlock(slug or "nous", "Nous Portal", model, _nous_billing_url(), True, message or "")
label, url = _resolve_provider_link(slug, base_url)
return BillingBlock(slug, label, model, url, False, message or "")
+1 -1
View File
@@ -34,7 +34,7 @@ from __future__ import annotations
import logging
import math
import os
from dataclasses import dataclass, field
from dataclasses import dataclass
from typing import Any, Optional
logger = logging.getLogger(__name__)
+55 -2
View File
@@ -1,4 +1,4 @@
"""Surface-agnostic core for the Phase 2b terminal-billing screens.
"""Surface-agnostic core for the Phase 2b Remote Spending screens.
One fetch/parse per concern, consumed identically by the CLI handler
(``cli.py::_show_billing``), the TUI JSON-RPC methods
@@ -17,7 +17,7 @@ from __future__ import annotations
import logging
import os
import uuid
from dataclasses import dataclass, field
from dataclasses import dataclass
from decimal import Decimal, InvalidOperation
from typing import Any, Optional
@@ -107,6 +107,22 @@ class CardInfo:
return f"{self.masked} — {label}" if label else self.masked
@dataclass(frozen=True)
class PaymentMethodInfo:
"""The payment method on file. `kind` is "card", "link", or "unknown"
— anything else is normalised to "unknown" at parse time, so consumers
only ever see fields that belong to the kind they are looking at."""
kind: str
brand: Optional[str] = None
last4: Optional[str] = None
wallet: Optional[str] = None
email: Optional[str] = None
resolved_via: Optional[str] = None
#: What the server called it, when we did not recognise the kind.
raw_kind: Optional[str] = None
@dataclass(frozen=True)
class MonthlyCap:
limit_usd: Optional[Decimal] = None
@@ -150,6 +166,7 @@ class BillingState:
min_usd: Optional[Decimal] = None
max_usd: Optional[Decimal] = None
card: Optional[CardInfo] = None
payment_method: Optional[PaymentMethodInfo] = None
monthly_cap: Optional[MonthlyCap] = None
auto_reload: Optional[AutoReload] = None
portal_url: Optional[str] = None
@@ -201,6 +218,41 @@ def _parse_card(raw: Any) -> Optional[CardInfo]:
return CardInfo(brand=brand, last4=last4, resolved_via=resolved_via)
def _parse_payment_method(raw: Any) -> Optional[PaymentMethodInfo]:
if not isinstance(raw, dict):
return None
kind = raw.get("kind")
if not isinstance(kind, str):
return None
def _optional_string(key: str) -> Optional[str]:
value = raw.get(key)
return value if isinstance(value, str) else None
resolved_via = _optional_string("resolvedVia")
brand = _optional_string("brand")
last4 = _optional_string("last4")
# Settle the kind here, the way _parse_card settles a card, so nothing
# downstream has to re-check which fields this kind is allowed to have.
if kind == "card" and brand and last4:
return PaymentMethodInfo(
kind="card",
brand=brand,
last4=last4,
wallet=_optional_string("wallet"),
resolved_via=resolved_via,
)
if kind == "link":
return PaymentMethodInfo(
kind="link",
email=_optional_string("email"),
resolved_via=resolved_via,
)
return PaymentMethodInfo(
kind="unknown", raw_kind=kind, resolved_via=resolved_via
)
def _parse_monthly_cap(raw: Any) -> Optional[MonthlyCap]:
if not isinstance(raw, dict):
return None
@@ -274,6 +326,7 @@ def billing_state_from_payload(
min_usd=parse_money(bounds.get("minUsd")),
max_usd=parse_money(bounds.get("maxUsd")),
card=_parse_card(payload.get("card")),
payment_method=_parse_payment_method(payload.get("paymentMethod")),
monthly_cap=_parse_monthly_cap(payload.get("monthlyCap")),
auto_reload=_parse_auto_reload(payload.get("autoReload")),
portal_url=portal_url,
File diff suppressed because it is too large Load Diff
+37 -8
View File
@@ -18,6 +18,7 @@ import uuid
from types import SimpleNamespace
from typing import Any, Dict, List, Optional
from agent.message_sanitization import deterministic_call_id
from agent.prompt_builder import DEFAULT_AGENT_IDENTITY
logger = logging.getLogger(__name__)
@@ -182,12 +183,34 @@ def _summarize_user_message_for_log(content: Any, *, sep: str = " ") -> str:
def _deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
"""Generate a deterministic call_id from tool call content.
Used as a fallback when the API doesn't provide a call_id.
Thin wrapper over the single policy owner
``agent.message_sanitization.deterministic_call_id`` (audit F4) — kept
as a module-level name because run_agent and tests import it from here.
Deterministic IDs prevent cache invalidation — random UUIDs would
make every API call's prefix unique, breaking OpenAI's prompt cache.
"""
seed = f"{fn_name}:{arguments}:{index}"
digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12]
return deterministic_call_id(fn_name, arguments, index)
def _clamp_responses_call_id(call_id: str) -> str:
"""Keep a ``call_id`` within the Responses API's 64-char limit (#73492).
The codex app-server namespaces MCP tool call ids as
``codex_mcp__<server>__<tool>_<codex_call_id>``; with an ``exec-<uuid>``
component the built-in ``hermes-tools`` server already overflows 64 chars,
and the Responses API rejects the whole payload with a non-retryable HTTP
400 that then replays every turn — permanently bricking the session.
Sibling defect to #10788 (which clamped ``input[*].id``), applied here to
``call_id``. The surrogate is a pure, deterministic function of the
original, so the ``function_call`` and its matching ``function_call_output``
— which carry the same original id — map to the same surrogate and stay
paired without correlating the two items. Short ids pass through unchanged,
preserving prompt-cache prefixes.
"""
if len(call_id) <= _MAX_RESPONSES_ITEM_ID_LENGTH:
return call_id
digest = hashlib.sha256(call_id.encode("utf-8", errors="replace")).hexdigest()[:32]
return f"call_{digest}"
@@ -546,7 +569,7 @@ def _chat_messages_to_responses_input(
items.append({
"type": "function_call",
"call_id": call_id,
"call_id": _clamp_responses_call_id(call_id),
"name": fn_name,
"arguments": arguments,
})
@@ -589,7 +612,7 @@ def _chat_messages_to_responses_input(
items.append({
"type": "function_call_output",
"call_id": call_id,
"call_id": _clamp_responses_call_id(call_id),
"output": output_value,
})
@@ -912,7 +935,8 @@ def _preflight_codex_api_kwargs(
allowed_keys = {
"model", "instructions", "input", "tools", "store",
"reasoning", "include", "max_output_tokens", "temperature",
"tool_choice", "parallel_tool_calls", "prompt_cache_key", "service_tier",
"tool_choice", "parallel_tool_calls", "prompt_cache_key",
"prompt_cache_retention", "service_tier",
"extra_headers", "extra_body", "timeout",
}
normalized: Dict[str, Any] = {
@@ -950,8 +974,13 @@ def _preflight_codex_api_kwargs(
if isinstance(temperature, (int, float)):
normalized["temperature"] = float(temperature)
# Pass through tool_choice, parallel_tool_calls, prompt_cache_key
for passthrough_key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key"):
# Pass through cache routing/retention and tool-dispatch hints.
for passthrough_key in (
"tool_choice",
"parallel_tool_calls",
"prompt_cache_key",
"prompt_cache_retention",
):
val = api_kwargs.get(passthrough_key)
if val is not None:
normalized[passthrough_key] = val
+155 -38
View File
@@ -18,7 +18,6 @@ from __future__ import annotations
import json
import logging
import os
import time
from types import SimpleNamespace
from typing import Any, Callable, Dict, List
@@ -74,7 +73,10 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
try:
if not agent._session_db_created:
agent._ensure_db_session()
agent._session_db.update_token_counts(
# Enqueued for the SessionDB background writer — keeps the
# per-call accounting write off the turn thread (see
# conversation_loop's queue_token_counts call).
agent._session_db.queue_token_counts(
agent.session_id,
model=agent.model,
billing_provider=agent.provider,
@@ -154,7 +156,8 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
try:
if not agent._session_db_created:
agent._ensure_db_session()
agent._session_db.update_token_counts(
# Enqueued for the SessionDB background writer (see above).
agent._session_db.queue_token_counts(
agent.session_id,
input_tokens=canonical_usage.input_tokens,
output_tokens=canonical_usage.output_tokens,
@@ -702,6 +705,16 @@ def run_codex_app_server_turn(
except Exception:
pass
agent._codex_session = None
_user_interrupted = bool(
getattr(agent, "_interrupt_requested", False)
)
_interrupt_message = (
getattr(agent, "_interrupt_message", None)
if _user_interrupted
else None
)
if _user_interrupted:
agent.clear_interrupt()
return {
"final_response": (
f"Codex app-server turn failed: {exc}. "
@@ -711,9 +724,27 @@ def run_codex_app_server_turn(
"api_calls": 0,
"completed": False,
"partial": True,
"interrupted": _user_interrupted,
**(
{"interrupt_message": _interrupt_message}
if _interrupt_message
else {}
),
"error": str(exc),
}
# This runtime bypasses the normal conversation-loop finalizer. Mirror its
# interrupt handoff/cleanup so a hard stop cannot poison the next turn and a
# message-bearing compatibility interrupt can still be replayed by callers.
_user_interrupted = bool(
turn.interrupted and getattr(agent, "_interrupt_requested", False)
)
_interrupt_message = (
getattr(agent, "_interrupt_message", None) if _user_interrupted else None
)
if _user_interrupted:
agent.clear_interrupt()
# If the turn signalled the underlying client is wedged (deadline
# blown, post-tool watchdog tripped, OAuth refresh died, subprocess
# exited), retire the session so the next turn respawns codex
@@ -750,12 +781,27 @@ def run_codex_app_server_turn(
# the already-flushed user turn). See gateway/run.py agent_persisted.
if getattr(agent, "_session_db", None) is not None:
try:
agent._flush_messages_to_session_db(messages)
_codex_flush_ok = agent._flush_messages_to_session_db(messages)
except Exception:
logger.debug(
_codex_flush_ok = False
logger.warning(
"codex app-server projected-message flush failed",
exc_info=True,
)
if _codex_flush_ok is False:
# Unlike the chat-completions loop (which fails closed BEFORE
# projection — see conversation_loop session_persistence_failed),
# codex output has already streamed to the user by the time this
# flush runs, so there is nothing left to withhold. We cannot
# flip agent_persisted=False either: the gateway fallback write
# would re-INSERT the already-flushed user turn (#860/#42039).
# Surface the durability gap loudly instead of a silent debug.
logger.warning(
"codex app-server turn was delivered but could NOT be "
"persisted to the session DB (session=%s) — this turn "
"will be missing after restart/resume",
getattr(agent, "session_id", None),
)
# Counter ticks for the agent-improvement loop.
@@ -819,6 +865,12 @@ def run_codex_app_server_turn(
"api_calls": api_calls,
"completed": not turn.interrupted and turn.error is None,
"partial": turn.interrupted or turn.error is not None,
"interrupted": _user_interrupted,
**(
{"interrupt_message": _interrupt_message}
if _interrupt_message
else {}
),
"error": turn.error,
# The codex app-server runtime IS an early-return path that bypasses
# conversation_loop, but we flush the projected assistant/tool messages
@@ -1187,6 +1239,8 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
"""
import httpx as _httpx
from agent import relay_llm
active_client = client or agent._ensure_primary_openai_client(reason="codex_stream_direct")
max_stream_retries = 1
# Accumulate streamed text so callers / compat shims can read it.
@@ -1211,48 +1265,88 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
if agent._interrupt_requested:
raise InterruptedError("Agent interrupted before Codex stream retry")
stream_kwargs = dict(api_kwargs)
stream_kwargs["stream"] = True
intercepted_events = []
writer_token = {"value": None}
def _open_codex_stream(next_api_kwargs: dict[str, Any]):
stream_kwargs = dict(next_api_kwargs)
stream_kwargs["stream"] = True
return active_client.responses.create(**stream_kwargs)
def _codex_stream_created(_raw_stream: Any) -> None:
# Claim the delta sink for THIS physical attempt. A newer attempt
# supersedes this token and fences late deltas out of the turn.
writer_token["value"] = claim_stream_writer(agent)
def _accept_codex_chunk(_chunk: Any) -> bool:
token = writer_token["value"]
if token is None or stream_writer_is_current(agent, token):
return True
logger.warning(
"Codex streaming attempt superseded by a newer stream; "
"stopping consumption to preserve the single-writer "
"invariant (model=%s).",
api_kwargs.get("model", "unknown"),
)
return False
def _finalize_codex_stream() -> Any:
return _consume_codex_event_stream(
list(intercepted_events),
model=api_kwargs.get("model"),
)
try:
event_stream = active_client.responses.create(**stream_kwargs)
except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc:
event_stream = relay_llm.stream(
dict(api_kwargs),
_open_codex_stream,
session_id=str(getattr(agent, "session_id", "") or ""),
name=str(getattr(agent, "provider", "") or "codex"),
model_name=str(api_kwargs.get("model") or ""),
finalizer=_finalize_codex_stream,
on_stream_created=_codex_stream_created,
on_chunk=intercepted_events.append,
chunk_adapter=lambda chunk: chunk,
accept_chunk=_accept_codex_chunk,
completed_response_predicate=lambda response: bool(
hasattr(response, "output") and not hasattr(response, "__iter__")
),
metadata={
"api_mode": "codex_responses",
"api_request_id": getattr(agent, "_current_api_request_id", None),
"call_role": (
"delegated"
if getattr(agent, "is_subagent", False)
else "fallback"
if int(getattr(agent, "_fallback_index", 0) or 0) > 0
else "primary"
),
"retry_count": attempt,
},
defer_logical_completion=True,
)
except (
_httpx.RemoteProtocolError,
_httpx.ReadTimeout,
_httpx.ConnectError,
ConnectionError,
) as exc:
if attempt < max_stream_retries:
logger.debug(
"Codex Responses stream connect failed (attempt %s/%s); retrying. %s error=%s",
attempt + 1, max_stream_retries + 1,
agent._client_log_context(), exc,
"Codex Responses stream connect failed (attempt %s/%s); "
"retrying. %s error=%s",
attempt + 1,
max_stream_retries + 1,
agent._client_log_context(),
exc,
)
continue
raise
# Claim the delta sink for THIS attempt (#65991) — parity with the
# chat_completions/anthropic/bedrock paths. If a prior attempt's
# stream is somehow still alive, this claim supersedes it so its
# late deltas are fenced out of the turn; conversely, a newer
# attempt supersedes us and the interrupt_check below stops our
# consumption immediately.
_writer_token = claim_stream_writer(agent)
def _interrupt_or_superseded(_tok=_writer_token) -> bool:
if agent._interrupt_requested:
return True
if not stream_writer_is_current(agent, _tok):
logger.warning(
"Codex streaming attempt superseded by a newer stream; "
"stopping consumption to preserve the single-writer "
"invariant (model=%s).",
api_kwargs.get("model", "unknown"),
)
return True
return False
def _interrupt_or_superseded() -> bool:
return bool(agent._interrupt_requested)
try:
# Compatibility: some mocks/providers return a concrete response
# instead of an iterable. Pass it straight through.
if hasattr(event_stream, "output") and not hasattr(event_stream, "__iter__"):
return event_stream
try:
final = _consume_codex_event_stream(
event_stream,
@@ -1271,6 +1365,12 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
on_event=_on_event,
interrupt_check=_interrupt_or_superseded,
)
# The terminal SSE frame is contractually last. Request the
# end-of-stream marker so Relay can run its response finalizer
# and close the physical attempt scope before Hermes returns.
if not agent._interrupt_requested:
for _ignored in event_stream:
pass
except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc:
if attempt < max_stream_retries:
logger.debug(
@@ -1281,6 +1381,10 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
)
continue
raise
except RuntimeError:
if event_stream.final_response is not None:
return event_stream.final_response
raise
if final.status in {"incomplete", "failed"}:
logger.warning(
@@ -1298,7 +1402,20 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
try:
close_fn()
except Exception:
pass
# A failed close can leave this response's connection
# checked out of the httpx pool while the caller's finally
# reports a reuse-reason close (e.g. interrupt_check broke
# the event loop with collected output) — caching the
# client with the leaked connection. Poison the slot so
# that close really closes the pool (owner-thread abort;
# mirrors the chat-streaming interrupt-break handling).
# ``client is None`` means the shared primary client,
# which is never reuse-cached and must not have its
# sockets force-shut here.
if client is not None:
agent._abort_request_openai_client(
active_client, reason="codex_stream_close_failed"
)
def run_codex_create_stream_fallback(agent, api_kwargs: dict, client: Any = None):
+48 -24
View File
@@ -55,13 +55,12 @@ import json
import logging
import os
import re
import subprocess
import tempfile
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Optional
from hermes_cli._subprocess_compat import IS_WINDOWS, windows_hide_flags
from hermes_cli._subprocess_compat import bounded_git_probe
logger = logging.getLogger("hermes.coding_context")
@@ -338,9 +337,9 @@ def _coding_mode(config: Optional[dict[str, Any]]) -> str:
"""Return the normalized ``agent.coding_context`` mode (auto/focus/on/off)."""
if config is None:
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
config = load_config()
config = load_config_readonly()
except Exception:
config = {}
raw = ((config or {}).get("agent", {}) or {}).get("coding_context", "auto")
@@ -521,30 +520,46 @@ class RuntimeMode:
return None
return [self.profile.toolset, *_enabled_mcp_servers(config)]
def system_blocks(self) -> list[str]:
"""Stable system-prompt blocks for this posture (brief + workspace).
def system_prompt_parts(self) -> tuple[list[str], list[str], list[str]]:
"""Return prefix, workspace, and trailing posture blocks separately.
The operating brief carries a model-family edit-format nudge appended
to it (one cached string, not a separate block) so the model is steered
toward the `patch` mode it handles best — see ``_edit_format_line``.
The three lists preserve the historical flat prompt order: the brief,
the live workspace snapshot, then configured operator instructions.
Prompt assembly can therefore put a cache boundary before the snapshot
without changing the persisted system-prompt bytes.
"""
if not self.is_coding:
return []
blocks: list[str] = []
return [], [], []
prefix: list[str] = []
workspace_parts: list[str] = []
trailing: list[str] = []
if self.profile.guidance:
brief = self.profile.guidance
edit_line = _edit_format_line(self.model)
if edit_line:
brief = f"{brief}\n{edit_line}"
blocks.append(brief)
prefix.append(brief)
workspace = build_coding_workspace_block(self.cwd)
if workspace:
blocks.append(workspace)
workspace_parts.append(workspace)
# Operator instructions ride their own block so the brief (block 0) stays
# byte-stable and cache-keyed independently of user config.
if self.instructions:
blocks.append(f"Operator instructions (from config):\n{self.instructions}")
return blocks
trailing.append(f"Operator instructions (from config):\n{self.instructions}")
return prefix, workspace_parts, trailing
def system_blocks(self) -> list[str]:
"""Return posture blocks in their historical display order.
``system_prompt_parts`` is the cache-aware API. This compatibility
helper retains the public flat list for callers outside prompt assembly.
"""
prefix, workspace, trailing = self.system_prompt_parts()
return [*prefix, *workspace, *trailing]
def compact_skill_categories(self) -> frozenset[str]:
"""Skill categories to demote to names-only in the prompt's skill index.
@@ -645,6 +660,19 @@ def coding_system_blocks(
).system_blocks()
def coding_system_prompt_parts(
*,
platform: Optional[str] = None,
cwd: Optional[str | Path] = None,
config: Optional[dict[str, Any]] = None,
model: Optional[str] = None,
) -> tuple[list[str], list[str], list[str]]:
"""Return coding prefix, workspace snapshot, and trailing guidance."""
return resolve_runtime_mode(
platform=platform, cwd=cwd, config=config, model=model
).system_prompt_parts()
def coding_compact_skill_categories(
*,
platform: Optional[str] = None,
@@ -689,18 +717,14 @@ def _enabled_mcp_servers(config: Optional[dict[str, Any]]) -> list[str]:
def _git(cwd: Path, *args: str) -> str:
_popen_kwargs = {"creationflags": windows_hide_flags()} if IS_WINDOWS else {}
try:
out = subprocess.run(
["git", "-C", str(cwd), *args],
capture_output=True,
text=True,
timeout=_GIT_TIMEOUT,
**_popen_kwargs,
)
except (OSError, subprocess.SubprocessError):
return ""
return out.stdout.strip() if out.returncode == 0 else ""
"""``git -C <cwd> <args>`` → stripped stdout, or ``""`` on any failure.
Uses the shared :func:`bounded_git_probe` so the post-kill cleanup is bounded
on Windows — a plain ``subprocess.run(timeout=...)`` here deadlocked the agent
turn inside ``build_coding_workspace_block`` when a killed git left a suspended
descendant holding the pipe handles (issue #66037).
"""
return bounded_git_probe(["git", "-C", str(cwd), *args], timeout=_GIT_TIMEOUT)
def _parse_status(porcelain: str) -> tuple[dict[str, str], dict[str, int]]:
+204
View File
@@ -154,3 +154,207 @@ def compute_session_context_breakdown(
"estimated_total": estimated_total,
"model": getattr(agent, "model", "") or "",
}
# ── /context rendering (CLI + gateway) ──────────────────────────────────────
#
# Pure text renderers over the payload above. The CLI shows a glyph block-grid
# plus a category table; the gateway uses the same table without the grid
# (proportional monospace is not guaranteed on messaging platforms).
_CATEGORY_GLYPHS = {
"system_prompt": "■",
"tool_definitions": "▣",
"rules": "▩",
"skills": "▤",
"mcp": "▥",
"subagent_definitions": "▦",
"memory": "▧",
"conversation": "▨",
}
_FREE_GLYPH = "·"
_GRID_COLUMNS = 20
_GRID_ROWS = 5 # 100 cells → 1 cell per percent of the context window
# Human-readable tables cap the expanded listings; nothing is dropped from
# the underlying data.
_DETAILS_TABLE_LIMIT = 15
def _bytes_to_tokens(size: Optional[int]) -> Optional[int]:
if size is None:
return None
return (int(size) + 3) // 4
def compute_context_details(agent: Any) -> Dict[str, Any]:
"""Expanded per-skill / per-toolset cost listing for ``/context all``.
Reuses the ``hermes prompt-size`` attribution mechanism (PR #66656):
per-skill index-line bytes parsed from the live ``<available_skills>``
block, and per-toolset schema bytes attributed via the tool registry's
canonical tool→toolset map. Byte figures are converted to the same
chars/4 token heuristic the categories above use.
"""
from hermes_cli.prompt_size import (
_compute_skills_breakdown,
_compute_toolsets_breakdown,
)
from agent.system_prompt import build_system_prompt_parts
parts = build_system_prompt_parts(agent)
stable = parts.get("stable", "") or ""
skills_match = _SKILLS_BLOCK_RE.search(stable)
skills_block = skills_match.group(0) if skills_match else ""
skills: List[Dict[str, Any]] = []
if skills_block:
for entry in _compute_skills_breakdown(skills_block):
skills.append({
"name": entry.get("name", ""),
"index_tokens": _bytes_to_tokens(entry.get("index_line_bytes")) or 0,
"skill_md_tokens": _bytes_to_tokens(entry.get("skill_md_bytes")),
})
toolsets: List[Dict[str, Any]] = []
tools = list(getattr(agent, "tools", None) or [])
if tools:
for group in _compute_toolsets_breakdown(tools):
toolsets.append({
"toolset": group.get("toolset", ""),
"tool_count": int(group.get("tool_count", 0) or 0),
"schema_tokens": _bytes_to_tokens(group.get("json_bytes")) or 0,
})
return {"skills": skills, "toolsets": toolsets}
def render_context_grid(payload: Dict[str, Any]) -> List[str]:
"""Render the payload as a Claude Code-style glyph block grid.
100 cells (5×20), each one percent of the model context window. Categories
fill in declaration order; the remainder renders as free space.
"""
context_max = int(payload.get("context_max") or 0)
categories = payload.get("categories") or []
total_cells = _GRID_COLUMNS * _GRID_ROWS
cells: List[str] = []
if context_max > 0:
for cat in categories:
tokens = int(cat.get("tokens") or 0)
n = round(tokens / context_max * total_cells)
if tokens > 0 and n == 0:
n = 1 # never render a nonzero category as invisible
glyph = _CATEGORY_GLYPHS.get(str(cat.get("id") or ""), "▪")
cells.extend([glyph] * n)
cells = cells[:total_cells]
cells.extend([_FREE_GLYPH] * (total_cells - len(cells)))
return [
" ".join(cells[row * _GRID_COLUMNS:(row + 1) * _GRID_COLUMNS])
for row in range(_GRID_ROWS)
]
def render_context_category_lines(payload: Dict[str, Any]) -> List[str]:
"""Render the 'Estimated usage by category' table as plain-text lines."""
categories = payload.get("categories") or []
context_max = int(payload.get("context_max") or 0)
estimated_total = int(payload.get("estimated_total") or 0)
denom = context_max or estimated_total
lines = ["Estimated usage by category"]
if not categories:
lines.append(" (no data yet — send a message first)")
return lines
width = max(len(str(cat.get("label") or "")) for cat in categories)
width = max(width, len("Free space"))
for cat in categories:
tokens = int(cat.get("tokens") or 0)
glyph = _CATEGORY_GLYPHS.get(str(cat.get("id") or ""), "▪")
pct = tokens / denom * 100 if denom else 0.0
label = str(cat.get("label") or cat.get("id") or "")
lines.append(f"{glyph} {label:<{width}} {tokens:>9,} tokens {pct:>5.1f}%")
if context_max > 0:
free = max(0, context_max - estimated_total)
pct = free / context_max * 100
lines.append(f"{_FREE_GLYPH} {'Free space':<{width}} {free:>9,} tokens {pct:>5.1f}%")
return lines
def render_context_details_lines(details: Dict[str, Any]) -> List[str]:
"""Render the expanded ``/context all`` per-skill / per-toolset tables."""
lines: List[str] = []
toolsets = details.get("toolsets") or []
if toolsets:
lines.append("Toolsets by schema cost (largest first)")
for group in toolsets[:_DETAILS_TABLE_LIMIT]:
lines.append(
f" {group['toolset']:<24} {group['tool_count']:>3} tools"
f" {group['schema_tokens']:>8,} tokens"
)
remaining = len(toolsets) - _DETAILS_TABLE_LIMIT
if remaining > 0:
lines.append(f" … and {remaining} more")
skills = details.get("skills") or []
if skills:
if lines:
lines.append("")
lines.append("Skills by cost (index = always-on; SKILL.md = cost when loaded)")
for entry in skills[:_DETAILS_TABLE_LIMIT]:
name = str(entry.get("name") or "")
if len(name) > 28:
name = name[:27] + "…"
md = entry.get("skill_md_tokens")
md_str = f"{md:>8,}" if md is not None else f"{'n/a':>8}"
lines.append(
f" {name:<28} index {entry['index_tokens']:>6,}"
f" SKILL.md {md_str} tokens"
)
remaining = len(skills) - _DETAILS_TABLE_LIMIT
if remaining > 0:
lines.append(f" … and {remaining} more")
return lines
def render_context_breakdown_lines(
payload: Dict[str, Any],
*,
details: Optional[Dict[str, Any]] = None,
grid: bool = True,
) -> List[str]:
"""Render the full /context view as plain-text lines.
``grid=True`` (CLI) prepends the glyph block grid; the gateway passes
``grid=False`` and keeps its own gauge. ``details`` (from
:func:`compute_context_details`) appends the expanded listings.
"""
lines: List[str] = []
if grid:
lines.extend(render_context_grid(payload))
lines.append("")
lines.extend(render_context_category_lines(payload))
context_max = int(payload.get("context_max") or 0)
context_used = int(payload.get("context_used") or 0)
if context_max > 0:
pct = int(payload.get("context_percent") or 0)
lines.append("")
lines.append(
f"Context window: {context_used:,} / {context_max:,} tokens ({pct}%)"
)
if details is not None:
detail_lines = render_context_details_lines(details)
if detail_lines:
lines.append("")
lines.extend(detail_lines)
else:
lines.append("")
lines.append("Use /context all for per-skill and per-toolset costs.")
return lines
+3246 -275
View File
File diff suppressed because it is too large Load Diff
+261 -3
View File
@@ -26,7 +26,64 @@ Lifecycle:
"""
from abc import ABC, abstractmethod
from typing import Any, Dict, List
from typing import Any, Dict, List, Optional
from agent.redact import redact_sensitive_text
MEMORY_CONTEXT_MAX_CHARS = 6_000
_MEMORY_CONTEXT_HEAD_CHARS = 4_000
_MEMORY_CONTEXT_TAIL_CHARS = 1_500
_MEMORY_CONTEXT_TRUNCATION_MARKER = "\n...[memory provider context truncated]...\n"
def sanitize_memory_context(memory_context: str) -> str:
"""Prepare provider context for a context-engine/LLM egress boundary."""
sanitized = redact_sensitive_text(
memory_context.strip(),
force=True,
redact_url_credentials=True,
)
if len(sanitized) <= MEMORY_CONTEXT_MAX_CHARS:
return sanitized
return (
sanitized[:_MEMORY_CONTEXT_HEAD_CHARS]
+ _MEMORY_CONTEXT_TRUNCATION_MARKER
+ sanitized[-_MEMORY_CONTEXT_TAIL_CHARS:]
)
def automatic_compaction_status_message(
engine: Any,
*,
phase: str,
default_message: str,
**context: Any,
) -> str | None:
"""Resolve host-visible status for an automatic compaction event.
Engines can suppress routine automatic status with
``emit_automatic_compaction_status = False`` or customize it by defining
``get_automatic_compaction_status_message(...)``. Empty strings and
``None`` mean "do not emit a lifecycle status".
"""
if not getattr(engine, "emit_automatic_compaction_status", True):
return None
formatter = getattr(engine, "get_automatic_compaction_status_message", None)
if callable(formatter):
message = formatter(
phase=phase,
default_message=default_message,
**context,
)
else:
message = default_message
if message is None:
return None
message = str(message).strip()
return message or None
class ContextEngine(ABC):
@@ -65,6 +122,12 @@ class ContextEngine(ABC):
protect_first_n: int = 3
protect_last_n: int = 6
# User-visible lifecycle status for automatic host-triggered compaction.
# Alternative engines that treat compaction as routine background
# maintenance can set this false to keep successful automatic passes silent;
# warnings, errors, and explicit manual commands should still surface.
emit_automatic_compaction_status: bool = True
# -- Core interface ----------------------------------------------------
@abstractmethod
@@ -83,12 +146,27 @@ class ContextEngine(ABC):
def should_compress(self, prompt_tokens: int = None) -> bool:
"""Return True if compaction should fire this turn."""
def should_compress_info(self, prompt_tokens: int = None) -> "tuple[bool, str | None]":
"""Return ``(should_compress, reason)``.
The base implementation is backward-compatible: engines that only
implement ``should_compress`` get ``(should_compress(prompt_tokens),
None)``. Concrete engines with richer block reasons (e.g. a
summary-LLM cooldown or an anti-thrashing guard) override this to
surface a human-readable reason so callers can warn the user instead
of silently skipping compression. Added for the silent-overflow
warning fix (#62625) so plugin engines don't raise AttributeError.
"""
return self.should_compress(prompt_tokens), None
@abstractmethod
def compress(
self,
messages: List[Dict[str, Any]],
current_tokens: int = None,
focus_topic: str = None,
current_tokens: Optional[int] = None,
focus_topic: Optional[str] = None,
force: bool = False,
memory_context: str = "",
) -> List[Dict[str, Any]]:
"""Compact the message list and return the new message list.
@@ -103,8 +181,152 @@ class ContextEngine(ABC):
Engines that support guided compression should prioritise
preserving information related to this topic. Engines that
don't support it may simply ignore this argument.
force: Whether a user-requested compression should bypass an
engine-owned cooldown. Engines without cooldowns may ignore it.
memory_context: Text returned by memory providers immediately before
compaction. Summarizing engines should include non-empty text in
their handoff prompt. Older engines may omit this parameter; the
host filters unsupported optional arguments by signature.
"""
# -- Optional: proactive tool-result prune -----------------------------
def prune_tool_results_only(
self,
messages: List[Dict[str, Any]],
current_tokens: int | None = None,
) -> tuple[List[Dict[str, Any]], int]:
"""Deterministically trim old tool-result payloads without an LLM call.
Runs on a low, cost-oriented trigger independent of ``should_compress``
so large-window engines can reclaim re-sent tool output long before full
compaction would fire. Returns ``(messages, n_pruned)``.
Default is a safe no-op: the list is returned unchanged with ``0``
pruned. Engines that don't implement a cheap prune — and any engine that
predates this hook — inherit this default, so the agent loop's
post-tool-call prune path never raises ``AttributeError`` on them. The
built-in ContextCompressor overrides this with the real implementation.
"""
return messages, 0
# -- Optional: per-turn context selection (distinct from compression) --
def select_context(
self,
request_messages: List[Dict[str, Any]],
*,
conversation_messages: List[Dict[str, Any]] = None,
incoming_message: Dict[str, Any] = None,
budget_tokens: int = 0,
) -> List[Dict[str, Any]]:
"""Optionally choose/replace the context for THIS request, pre-generation.
Called every turn after the request message list is assembled and
before it is dispatched to the provider — independent of
``should_compress()``. This lets an engine *select* which context
enters the prompt (retrieval, topic routing, role/branch switching)
rather than *shrink* context that is already there. The two verbs are
orthogonal:
- ``compress()`` : context is too long -> make it shorter.
- ``select_context()``: this turn belongs to a different context
-> use that one instead.
Without this hook, engines that need per-turn access to the message
list have to force ``should_compress()`` to return ``True`` so that
``compress()`` is invoked every turn purely as a callback — which
conflates selection with compression and degrades behaviour when the
engine's backend is unavailable. ``select_context()`` removes the need
for that workaround.
The returned list is request-only: it replaces the messages sent to
the provider for this single call and MUST NOT be treated as persisted
transcript state. The conversation history in the session DB is left
untouched, so nothing leaks across turns. Return ``None`` to leave the
request unchanged.
Unlike the ``pre_llm_call`` plugin hook (which appends to the user
message and intentionally never rewrites the list, to preserve the
cache prefix), ``select_context()`` may *replace* the message list.
Ordering / cache contract: the host runs this hook **before** prompt
cache-control and **before** every request sanitizer (orphaned-tool
cleanup, thinking-only/role normalization, whitespace/JSON
normalization). So (a) whatever the hook returns still passes through
the same validation as any request — a malformed replacement cannot
reach the provider — and (b) prompt-cache stability (an AGENTS.md
invariant) is preserved: the default no-op leaves the request
byte-identical, so cache behaviour is unchanged for the built-in
compressor and any non-implementing engine. An engine that *does*
replace the list changes its own cache prefix by definition; that is
the engine's concern, and cache-control breakpoints are re-derived on
the selected list. The hook is evaluated per provider request (so it
re-runs on retries within a turn), consistent with "select the context
for THIS request".
Args:
request_messages: The assembled request message list (system
prompt + history + any ephemeral prefill), in OpenAI format.
conversation_messages: The unmodified persisted conversation
history, for reference only (do not mutate).
incoming_message: The current turn's user message, if available.
budget_tokens: The active model's context length, or 0 if unknown.
Default returns ``None`` (no-op) — zero impact on the built-in
compressor or any existing engine.
"""
return None
def on_turn_complete(
self,
messages: List[Dict[str, Any]],
usage: Dict[str, Any] = None,
**kwargs: Any,
) -> None:
"""Observe a finished user turn (post-turn ingestion / observation).
Called from the standard turn-finalization path once the assistant/tool
loop completes, with the finalized in-memory transcript snapshot. This
is the complement to ``select_context()``: selection happens *before*
the request, while observation happens *after* the turn. It lets an
engine ingest, index, summarize, or update routing / topic / session
state from what actually happened — so the next ``select_context()``
can act on it.
Coverage: this fires from the normal finalization seam. Some abnormal
early-return paths in the loop (e.g. a content-policy block or a
provider terminal failure) persist and return without routing through
finalization, and therefore do not currently emit this hook. Treat it
as a best-effort post-turn observation for completed turns, not a
guaranteed callback for every possible early exit; unifying all
terminal paths behind one finalization seam is a separate follow-up.
Together the two hooks remove the need to abuse ``should_compress()`` /
``compress()`` as a generic per-turn callback just to observe history,
and they cover the case where a turn finishes and there may be no next
request from which to infer the previous turn.
``messages`` is a shallow copy and should be treated as read-only:
return values are ignored and this hook must not rely on transcript
mutation for persistence. ``kwargs`` may include ``turn_id``,
``task_id``, ``api_call_count``, ``interrupted``, ``failed``, and
``turn_exit_reason``.
``usage`` carries the completed turn's canonical token usage (the same
dict shape passed to ``update_from_response`` — ``prompt_tokens`` /
``completion_tokens`` / ``total_tokens`` plus the canonical
``input_tokens`` / ``output_tokens`` / ``cache_read_tokens`` /
``cache_write_tokens`` / ``reasoning_tokens`` buckets) so an engine can
weigh how large/expensive the selected context actually was when
deciding the next ``select_context()``. It is ``None`` on finalized
turns that never reached a provider response (e.g. interrupt); engines
must treat it as optional.
Default is a no-op.
"""
return None
# -- Optional: pre-flight check ----------------------------------------
def should_compress_preflight(self, messages: List[Dict[str, Any]]) -> bool:
@@ -124,6 +346,27 @@ class ContextEngine(ABC):
"""
return False
def get_automatic_compaction_status_message(
self,
*,
phase: str,
default_message: str,
**context: Any,
) -> str | None:
"""Return user-visible status for automatic host-triggered compaction.
Return ``None`` to suppress successful automatic lifecycle status for
this compaction event. ``phase`` identifies the host call site (for
example ``"preflight"`` or ``"compress"``). ``context`` contains
best-effort fields such as ``approx_tokens`` and ``threshold_tokens``.
This hook does not control warning/error messages or explicit manual
commands such as ``/compress``.
"""
if not self.emit_automatic_compaction_status:
return None
return default_message
# -- Optional: manual /compress preflight ------------------------------
def has_content_to_compress(self, messages: List[Dict[str, Any]]) -> bool:
@@ -228,4 +471,19 @@ class ContextEngine(ABC):
(e.g. recalculate DAG budgets, switch summary models).
"""
self.context_length = context_length
# Apply per-model threshold overrides if set (longest substring match).
# Falls back to _config_threshold_percent (the raw config value) when
# no override matches. Plugin engines that override update_model() can
# call resolve_model_threshold() for the same logic.
from agent.context_compressor import resolve_model_threshold
if not hasattr(self, "_config_threshold_percent"):
# Snapshot the pre-override percent ONCE so repeated model
# switches fall back to the engine's configured value, not the
# previous model's override.
self._config_threshold_percent = self.threshold_percent
self._base_threshold_percent = resolve_model_threshold(
model, getattr(self, "model_thresholds", {}),
self._config_threshold_percent,
)
self.threshold_percent = self._base_threshold_percent
self.threshold_tokens = int(context_length * self.threshold_percent)
+24 -17
View File
@@ -19,6 +19,7 @@ REFERENCE_PATTERN = re.compile(
rf"(?<![\w/])@(?:(?P<simple>diff|staged)\b|(?P<kind>file|folder|git|url):(?P<value>{_QUOTED_REFERENCE_VALUE}(?::\d+(?:-\d+)?)?|\S+))"
)
TRAILING_PUNCTUATION = ",.;!?"
_NEEDS_QUOTING = re.compile(r"""[\s()\[\]{}<>"'`]""")
_SENSITIVE_HOME_DIRS = (".ssh", ".aws", ".gnupg", ".kube", ".docker", ".azure", ".config/gh")
_SENSITIVE_HERMES_DIRS = (Path("skills") / ".hub",)
_SENSITIVE_HOME_FILES = (
@@ -60,6 +61,21 @@ class ContextReferenceResult:
blocked: bool = False
def format_reference_value(value: str) -> str:
"""Quote a reference value so ``REFERENCE_PATTERN`` reads it back whole.
The unquoted alternative in the pattern is ``\\S+``, so a path containing a
space parses as a truncated ref with the tail left behind as loose text.
Mirrors ``formatRefValue`` in the desktop's directive-text.tsx.
"""
if not _NEEDS_QUOTING.search(value):
return value
for quote in ("`", '"', "'"):
if quote not in value:
return f"{quote}{value}{quote}"
return value
def parse_context_references(message: str) -> list[ContextReference]:
refs: list[ContextReference] = []
if not message:
@@ -197,8 +213,12 @@ async def preprocess_context_references_async(
f"@ context injection warning: {injected_tokens} tokens exceeds the 25% soft limit ({soft_limit})."
)
stripped = _remove_reference_tokens(message, refs)
final = stripped
# Leave the `@file:`/`@folder:` tokens where the user typed them. The token
# IS the reference, not scaffolding around it: clients render each one as an
# inline chip, so stripping them left a sentence with a hole in it ("review
# and ship") and made the desktop re-derive the refs from the attached block
# to show them as a detached list above the prose.
final = message
if warnings:
final = f"{final}\n\n--- Context Warnings ---\n" + "\n".join(f"- {warning}" for warning in warnings)
if blocks:
@@ -308,7 +328,7 @@ def _expand_git_reference(
["git", *args],
cwd=cwd,
capture_output=True,
text=True,
text=True, encoding='utf-8', errors='replace',
timeout=30,
stdin=subprocess.DEVNULL,
**_popen_kwargs,
@@ -457,19 +477,6 @@ def _parse_file_reference_value(value: str) -> tuple[str, int | None, int | None
return _strip_reference_wrappers(value), None, None
def _remove_reference_tokens(message: str, refs: list[ContextReference]) -> str:
pieces: list[str] = []
cursor = 0
for ref in refs:
pieces.append(message[cursor:ref.start])
cursor = ref.end
pieces.append(message[cursor:])
text = "".join(pieces)
text = re.sub(r"\s{2,}", " ", text)
text = re.sub(r"\s+([,.;:!?])", r"\1", text)
return text.strip()
def _is_binary_file(path: Path) -> bool:
mime, _ = mimetypes.guess_type(path.name)
if mime and not mime.startswith("text/") and not any(
@@ -534,7 +541,7 @@ def _rg_files(path: Path, cwd: Path, limit: int) -> list[Path] | None:
["rg", "--files", str(path.relative_to(cwd))],
cwd=cwd,
capture_output=True,
text=True,
text=True, encoding='utf-8', errors='replace',
timeout=10,
stdin=subprocess.DEVNULL,
**_popen_kwargs,
File diff suppressed because it is too large Load Diff
+1512 -151
View File
File diff suppressed because it is too large Load Diff
+8 -3
View File
@@ -503,15 +503,20 @@ class CopilotACPClient:
def _run_prompt(self, prompt_text: str, *, timeout_seconds: float) -> tuple[str, str]:
try:
# Hide the console the CLI child would otherwise flash on Windows
# (#56747). Hide-only — stdio pipes stay intact for the ACP wire.
from hermes_cli._subprocess_compat import windows_hide_flags
proc = subprocess.Popen(
[self._acp_command] + self._acp_args,
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
text=True, encoding='utf-8', errors='replace',
bufsize=1,
cwd=self._acp_cwd,
env=_build_subprocess_env(),
creationflags=windows_hide_flags(),
)
except FileNotFoundError as exc:
raise RuntimeError(
@@ -703,7 +708,7 @@ class CopilotACPClient:
if block_error:
raise PermissionError(block_error)
try:
content = path.read_text()
content = path.read_text(encoding="utf-8")
except FileNotFoundError:
content = ""
line = params.get("line")
@@ -731,7 +736,7 @@ class CopilotACPClient:
if denied:
raise PermissionError(denied)
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(str(params.get("content") or ""))
path.write_text(str(params.get("content") or ""), encoding="utf-8")
response = {
"jsonrpc": "2.0",
"id": message_id,
+290 -85
View File
@@ -594,22 +594,64 @@ class CredentialPool:
# Re-armed to None on every successful selection so a recover→re-exhaust
# transition logs promptly instead of being swallowed by a stale window.
self._last_no_entries_log_at: Optional[float] = None
# #70401: consecutive mark_exhausted_and_rotate() calls whose supplied
# credential identity matched no pool entry (OAuth wrappers whose
# runtime key rotates, entries pruned by another process, ...). These
# rotations mark nothing exhausted, so without a cap the pool can
# never converge to "no available entries" and the caller's 401 retry
# loop runs unbounded and non-interruptible. Reset whenever a real
# entry is identified or an escape path returns None.
self._unmatched_rotation_streak: int = 0
def has_credentials(self) -> bool:
return bool(self._entries)
with self._lock:
return bool(self._entries)
def has_available(self) -> bool:
"""True if at least one entry is not currently in exhaustion cooldown."""
return bool(self._available_entries())
# ``_available_entries`` is not read-only: it prunes aged-out DEAD
# manual entries (rebinding ``self._entries``) and persists. It must
# run under ``self._lock`` like every other caller (``select`` etc.),
# otherwise a status probe here can race a concurrent ``select`` /
# rotation and tear ``self._entries`` or double-write auth.json.
with self._lock:
return bool(self._available_entries())
def entries(self) -> List[PooledCredential]:
return list(self._entries)
with self._lock:
return list(self._entries)
def current(self) -> Optional[PooledCredential]:
def _current_unlocked(self) -> Optional[PooledCredential]:
if not self._current_id:
return None
return next((entry for entry in self._entries if entry.id == self._current_id), None)
def current(self) -> Optional[PooledCredential]:
with self._lock:
return self._current_unlocked()
def entry_id_for_api_key(self, api_key_hint: Any = None) -> Optional[str]:
"""Return the stable id for the runtime credential in use.
Prefer the current selection when it still supplies ``api_key_hint``.
If the cursor was cleared, fall back to an unambiguous key match.
"""
with self._lock:
current = self._current_unlocked()
if current is not None and (
api_key_hint is None
or current.runtime_api_key == api_key_hint
):
return current.id
if api_key_hint is None:
return None
matches = [
entry
for entry in self._entries
if entry.runtime_api_key == api_key_hint
]
return matches[0].id if len(matches) == 1 else None
def _replace_entry(self, old: PooledCredential, new: PooledCredential) -> None:
"""Swap an entry in-place by id, preserving sort order."""
for idx, entry in enumerate(self._entries):
@@ -652,6 +694,8 @@ class CredentialPool:
entry: PooledCredential,
status_code: Optional[int],
error_context: Optional[Dict[str, Any]] = None,
*,
persist: bool = True,
) -> PooledCredential:
normalized_error = _normalize_error_context(error_context)
# Permanent OAuth failures (token_invalidated, token_revoked, etc.)
@@ -675,7 +719,8 @@ class CredentialPool:
last_error_reset_at=normalized_error.get("reset_at"),
)
self._replace_entry(entry, updated)
self._persist()
if persist:
self._persist()
return updated
def _sync_anthropic_entry_from_credentials_file(self, entry: PooledCredential) -> PooledCredential:
@@ -1484,6 +1529,43 @@ class CredentialPool:
self._sync_device_code_entry_to_auth_store(updated)
return updated
def _codex_quota_restored_upstream(self, entry: PooledCredential) -> bool:
"""Live-check whether an exhausted Codex entry's quota reset early.
A Codex 429 persists a ``last_error_reset_at`` that can be days in
the future (weekly windows), but the upstream window can reopen
before then — the user redeems a banked rate-limit reset via the
Codex CLI / ChatGPT UI, upgrades their plan, or OpenAI resets the
window. Without this check the pool keeps the credential frozen
until the stale timestamp elapses even though the account is
usable (issue #43747).
Only fires for openai-codex entries frozen by a 429/quota-shaped
error. The underlying probe is throttled per token (5 min) so this
is safe on the hot selection path.
"""
if self.provider != "openai-codex" or entry.last_status != STATUS_EXHAUSTED:
return False
if not auth_mod._is_codex_rate_limit_shaped(
entry.last_error_code,
entry.last_error_reason,
entry.last_error_message,
):
return False
token = entry.access_token or ""
if not token:
return False
try:
return bool(
auth_mod._probe_codex_quota_restored(
token,
base_url=entry.base_url,
)
)
except Exception:
logger.debug("Codex quota-restored probe failed", exc_info=True)
return False
def _entry_needs_refresh(self, entry: PooledCredential) -> bool:
if entry.auth_type != AUTH_TYPE_OAUTH:
return False
@@ -1510,7 +1592,13 @@ class CredentialPool:
def select(self) -> Optional[PooledCredential]:
with self._lock:
return self._select_unlocked()
entry = self._select_unlocked()
if entry is not None:
# A normal (non-recovery) selection starts a fresh episode —
# don't let a leftover unmatched-rotation streak from an old
# failure trip the #70401 bound early next time.
self._unmatched_rotation_streak = 0
return entry
def _available_entries(self, *, clear_expired: bool = False, refresh: bool = False) -> List[PooledCredential]:
"""Return entries not currently in exhaustion cooldown.
@@ -1605,7 +1693,18 @@ class CredentialPool:
if entry.last_status == STATUS_EXHAUSTED:
exhausted_until = _exhausted_until(entry)
if exhausted_until is not None and now < exhausted_until:
continue
# Codex quota windows can reopen EARLY: the user redeems a
# banked rate-limit reset (Codex CLI / ChatGPT UI), upgrades
# their plan, or OpenAI resets the window. The persisted
# ``last_error_reset_at`` can then be days in the future
# while the account is already usable again — a throttled
# live probe of the Codex usage endpoint detects that and
# lifts the stale cooldown (issue #43747).
if not (
clear_expired
and self._codex_quota_restored_upstream(entry)
):
continue
if clear_expired:
cleared = replace(
entry,
@@ -1678,18 +1777,21 @@ class CredentialPool:
self._entries = [replace(candidate, priority=idx) for idx, candidate in enumerate(rotated)]
self._persist()
self._current_id = entry.id
return self.current() or entry
return self._current_unlocked() or entry
entry = available[0]
self._current_id = entry.id
return entry
def peek(self) -> Optional[PooledCredential]:
current = self.current()
if current is not None:
return current
available = self._available_entries()
return available[0] if available else None
# Single lock acquisition for the whole read; call the unlocked
# helpers so we don't re-enter the non-reentrant ``self._lock``.
with self._lock:
current = self._current_unlocked()
if current is not None:
return current
available = self._available_entries()
return available[0] if available else None
def mark_exhausted_and_rotate(
self,
@@ -1697,10 +1799,17 @@ class CredentialPool:
status_code: Optional[int],
error_context: Optional[Dict[str, Any]] = None,
api_key_hint: Optional[str] = None,
credential_id: Optional[str] = None,
) -> Optional[PooledCredential]:
with self._lock:
entry = None
if api_key_hint:
identity_supplied = bool(credential_id or api_key_hint)
if credential_id:
entry = next(
(e for e in self._entries if e.id == credential_id),
None,
)
if entry is None and api_key_hint:
# Prefer the specific entry whose API key matches the one that
# actually failed. When this pool was freshly loaded from disk
# (another process already rotated), current() is None and
@@ -1709,12 +1818,89 @@ class CredentialPool:
(e for e in self._entries if e.runtime_api_key == api_key_hint),
None,
)
if entry is None and identity_supplied:
# The failed credential is identifiable but matches no entry
# (rotated away, or a wrapper whose runtime key differs).
# Falling through to current()/_select_unlocked() would mark an
# innocent healthy key exhausted for the full cooldown TTL.
#
# #70401: this branch must still be BOUNDED. With OAuth-token
# auth the upstream 401's key hint never matches any entry's
# ``runtime_api_key``, so every retry lands here, nothing is
# ever marked exhausted, and the pool can never reach the
# "no available entries" state — the caller retries the same
# dead token forever (~6/sec, starving the event loop so chat
# interrupts are never processed). The single-entry case
# below already escapes; multi-entry pools could still
# ping-pong A→B→A indefinitely without marking anything.
# Cap consecutive no-mark rotations at one full lap of the
# available entries: past that, every candidate has been
# handed back at least once without recovery, so stop
# guessing and surface the error (no cooldown is written for
# anybody — healthy keys stay available for the next turn).
self._unmatched_rotation_streak += 1
available_count = len(self._available_entries())
if self._unmatched_rotation_streak > max(available_count, 1):
logger.warning(
"credential pool: failed credential identity matched no "
"%s entry for %d consecutive rotations (pool size %d) — "
"surfacing the error instead of rotating again",
self.provider,
self._unmatched_rotation_streak,
available_count,
)
self._unmatched_rotation_streak = 0
self._current_id = None
return None
logger.info(
"credential pool: failed credential identity matched no %s "
"entry; rotating without marking any credential exhausted",
self.provider,
)
self._current_id = None
next_entry = self._select_unlocked()
if next_entry is not None and len(self._available_entries()) == 1:
# A single-entry pool cannot rotate. Returning its only
# entry reports a successful recovery without changing
# the credential, so the caller retries the same 401
# indefinitely. Let fallback/error propagation proceed.
self._unmatched_rotation_streak = 0
self._current_id = None
return None
return next_entry
# A real entry was identified — any prior unmatched-rotation
# streak is stale (this mark WILL advance pool state).
self._unmatched_rotation_streak = 0
if entry is None:
entry = self.current() or self._select_unlocked()
entry = self._current_unlocked() or self._select_unlocked()
if entry is None:
return None
_label = entry.label or entry.id[:8]
self._mark_exhausted(entry, status_code, error_context)
# A 402/429/401 is an API-key–level failure: the account is out of
# balance, rate-limited, or its key is rejected. The same key can
# back more than one pool entry (e.g. an explicit pool entry plus a
# ``model_config`` entry auto-seeded from ``model.api_key`` — both
# carry the identical ``runtime_api_key``). Marking only the first
# match leaves the sibling entries OK, so ``_select_unlocked()``
# keeps handing back the same depleted key and rotation never
# converges — the caller ``continue``s forever until the client
# disconnects (a ~2.5min hang with no error surfaced to the user).
# Mark every entry sharing the failed key so the pool can reach the
# "no available entries" state and let the error propagate.
failed_runtime_key = getattr(entry, "runtime_api_key", None)
if identity_supplied and failed_runtime_key:
siblings_marked = False
for sibling in self._entries:
if sibling.id == entry.id:
continue
if sibling.runtime_api_key == failed_runtime_key:
self._mark_exhausted(
sibling, status_code, error_context, persist=False
)
siblings_marked = True
if siblings_marked:
self._persist()
# Re-read the updated entry to log the correct terminal state.
updated_entry = next(
(e for e in self._entries if e.id == entry.id), entry,
@@ -1782,9 +1968,11 @@ class CredentialPool:
return self._try_refresh_current_unlocked()
def try_refresh_matching(
self, api_key_hint: Optional[str] = None
self,
api_key_hint: Optional[str] = None,
credential_id: Optional[str] = None,
) -> Optional[PooledCredential]:
"""Force-refresh the entry that supplied ``api_key_hint``.
"""Force-refresh the entry that supplied the failed request.
Direct provider integrations may reload the pool after a request has
already failed, so they cannot rely on ``current_id`` identifying the
@@ -1794,24 +1982,36 @@ class CredentialPool:
"""
with self._lock:
entry = None
if api_key_hint:
if credential_id:
entry = next(
(
candidate
for candidate in self._entries
if candidate.runtime_api_key == api_key_hint
if candidate.id == credential_id
),
None,
)
else:
entry = self.current() or self._select_unlocked(refresh=False)
if entry is None:
if api_key_hint:
entry = next(
(
candidate
for candidate in self._entries
if candidate.runtime_api_key == api_key_hint
),
None,
)
else:
entry = self._current_unlocked() or self._select_unlocked(
refresh=False
)
if entry is None:
return None
self._current_id = entry.id
return self._try_refresh_current_unlocked()
def _try_refresh_current_unlocked(self) -> Optional[PooledCredential]:
entry = self.current()
entry = self._current_unlocked()
if entry is None:
return None
refreshed = self._refresh_entry(entry, force=True)
@@ -1820,76 +2020,80 @@ class CredentialPool:
return refreshed
def reset_statuses(self) -> int:
count = 0
new_entries = []
for entry in self._entries:
if entry.last_status or entry.last_status_at or entry.last_error_code:
new_entries.append(
replace(
entry,
last_status=None,
last_status_at=None,
last_error_code=None,
last_error_reason=None,
last_error_message=None,
last_error_reset_at=None,
with self._lock:
count = 0
new_entries = []
for entry in self._entries:
if entry.last_status or entry.last_status_at or entry.last_error_code:
new_entries.append(
replace(
entry,
last_status=None,
last_status_at=None,
last_error_code=None,
last_error_reason=None,
last_error_message=None,
last_error_reset_at=None,
)
)
)
count += 1
else:
new_entries.append(entry)
if count:
self._entries = new_entries
self._persist()
return count
count += 1
else:
new_entries.append(entry)
if count:
self._entries = new_entries
self._persist()
return count
def remove_index(self, index: int) -> Optional[PooledCredential]:
if index < 1 or index > len(self._entries):
return None
removed = self._entries.pop(index - 1)
self._entries = [
replace(entry, priority=new_priority)
for new_priority, entry in enumerate(self._entries)
]
write_credential_pool(
self.provider,
[entry.to_dict() for entry in self._entries],
removed_ids=[removed.id],
)
if self._current_id == removed.id:
self._current_id = None
return removed
with self._lock:
if index < 1 or index > len(self._entries):
return None
removed = self._entries.pop(index - 1)
self._entries = [
replace(entry, priority=new_priority)
for new_priority, entry in enumerate(self._entries)
]
write_credential_pool(
self.provider,
[entry.to_dict() for entry in self._entries],
removed_ids=[removed.id],
)
if self._current_id == removed.id:
self._current_id = None
return removed
def resolve_target(self, target: Any) -> Tuple[Optional[int], Optional[PooledCredential], Optional[str]]:
raw = str(target or "").strip()
if not raw:
return None, None, "No credential target provided."
for idx, entry in enumerate(self._entries, start=1):
if entry.id == raw:
return idx, entry, None
with self._lock:
for idx, entry in enumerate(self._entries, start=1):
if entry.id == raw:
return idx, entry, None
label_matches = [
(idx, entry)
for idx, entry in enumerate(self._entries, start=1)
if entry.label.strip().lower() == raw.lower()
]
if len(label_matches) == 1:
return label_matches[0][0], label_matches[0][1], None
if len(label_matches) > 1:
return None, None, f'Ambiguous credential label "{raw}". Use the numeric index or entry id instead.'
if raw.isdigit():
index = int(raw)
if 1 <= index <= len(self._entries):
return index, self._entries[index - 1], None
return None, None, f"No credential #{index}."
return None, None, f'No credential matching "{raw}".'
label_matches = [
(idx, entry)
for idx, entry in enumerate(self._entries, start=1)
if entry.label.strip().lower() == raw.lower()
]
if len(label_matches) == 1:
return label_matches[0][0], label_matches[0][1], None
if len(label_matches) > 1:
return None, None, f'Ambiguous credential label "{raw}". Use the numeric index or entry id instead.'
if raw.isdigit():
index = int(raw)
if 1 <= index <= len(self._entries):
return index, self._entries[index - 1], None
return None, None, f"No credential #{index}."
return None, None, f'No credential matching "{raw}".'
def add_entry(self, entry: PooledCredential) -> PooledCredential:
entry = replace(entry, priority=_next_priority(self._entries))
self._entries.append(entry)
self._persist()
return entry
with self._lock:
entry = replace(entry, priority=_next_priority(self._entries))
self._entries.append(entry)
self._persist()
return entry
def _upsert_entry(entries: List[PooledCredential], provider: str, source: str, payload: Dict[str, Any]) -> bool:
@@ -2309,9 +2513,10 @@ def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool
def _get_env_prefer_dotenv(key: str) -> str:
env_file = load_env()
raw = env_file.get(key, "").strip()
env_val = os.environ.get(key, "").strip()
scoped_value = (_get_secret(key, "") or "").strip()
# If .env contains an unresolved op:// reference, prefer the
# already-resolved value from os.environ (set by
# already-resolved value supplied by the active secret scope (or by
# os.environ in legacy single-profile mode), set by
# load_hermes_dotenv() -> apply_onepassword_secrets()). The raw
# "op://Vault/Item/field" string would otherwise win and every
# provider auth attempt would receive a URL instead of a key. This
@@ -2319,9 +2524,9 @@ def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool
# references straight into .env rather than the secrets.onepassword
# config block. For every non-op:// value the original
# .env-takes-precedence behaviour is preserved unchanged.
if raw.startswith("op://") and env_val:
return env_val
return raw or _get_secret(key, "") or env_val
if raw.startswith("op://") and scoped_value:
return scoped_value
return raw or scoped_value
# Honour user suppression — `hermes auth remove <provider> <N>` for an
# env-seeded credential marks the env:<VAR> source as suppressed so it
+1 -1
View File
@@ -164,7 +164,7 @@ def _remove_env_source(provider: str, removed) -> RemovalResult:
if env_path.exists():
env_in_dotenv = any(
line.strip().startswith(f"{env_var}=")
for line in env_path.read_text(errors="replace").splitlines()
for line in env_path.read_text(errors="replace", encoding="utf-8").splitlines()
)
except OSError:
pass
+63 -5
View File
@@ -170,6 +170,27 @@ CREDITS_USAGE_BANDS: tuple[tuple[float, str, int], ...] = (
)
CREDITS_USAGE_KEY = "credits.usage" # single key for the escalating usage notice
# Minimum subscription balance that counts as "grant not yet spent" for the
# grant_spent crossing gate (see evaluate_credits_notices). 1¢: portal-seeded
# states derive micros from float dollars and can carry sub-cent residue where
# the inference headers report exactly 0 — without this floor such a seed
# opens the gate and the first header re-creates the at-open nag.
GRANT_UNSPENT_MIN_MICROS = 10_000
def new_credits_latch() -> dict:
"""Fresh notice latch in the shape :func:`evaluate_credits_notices` expects.
The policy owns this schema — every producer (agent build, lazy re-init,
tests) must build the latch through here so a new gate key lands everywhere
at once instead of drifting across hand-rolled literals."""
return {
"active": set(),
"seen_below_90": False,
"usage_band": None,
"seen_grant_unspent": False,
}
# ── AgentNotice (out-of-band notice payload; driver-agnostic) ────────────────
@@ -250,7 +271,8 @@ def evaluate_credits_notices(
) -> tuple[list[AgentNotice], list[str]]:
"""Reconcile credits notices against the latch. Mutates ``latch`` IN PLACE.
latch = {"active": set[str], "seen_below_90": bool, "usage_band": Optional[int]}.
latch = {"active": set[str], "seen_below_90": bool, "usage_band": Optional[int],
"seen_grant_unspent": bool}.
``model_is_free``: True when the session's active model is a Nous free-tier
model (see :func:`is_free_tier_model`). Suppresses the ``credits.depleted``
@@ -277,6 +299,18 @@ def evaluate_credits_notices(
if uf is not None and uf < _lowest_band:
latch["seen_below_90"] = True # gate opened: usage-band notices may now fire
# Grant-spent crossing gate: grant_spent may fire only after this session
# has OBSERVED the grant meaningfully unspent (≥1¢ left — see
# GRANT_UNSPENT_MIN_MICROS). Opening at grant-spent is a steady STATE, not
# an event — /usage carries it; only a live in-session crossing announces.
# Unlike seen_below_90, seeds must NOT prime this gate.
if (
uf is not None
and uf < 1.0
and state.subscription_micros >= GRANT_UNSPENT_MIN_MICROS
):
latch["seen_grant_unspent"] = True
active = latch["active"]
# ── Conditions ───────────────────────────────────────────────────────────
@@ -316,12 +350,21 @@ def evaluate_credits_notices(
active.discard(CREDITS_USAGE_KEY)
if target_band is not None:
# Belt-and-suspenders: a producer could set subscription_limit_micros
# without subscription_limit_usd. Render "$? cap" rather than "$None cap".
# without subscription_limit_usd. Render "$?" rather than "$None".
_cap_usd = state.subscription_limit_usd or "?"
_level = current_band[1] # type: ignore[index] (current_band set when target_band set)
# Report absolute dollars used, not a bare "N% used": the percentage is
# only meaningful against a Nous subscription cap (no cap → never fires),
# so dollars are clearer and don't imply a universal %. Used = cap −
# remaining (micros, money-safe), clamped to [0, cap]. Re-emits on band
# change (50 → 75 → 90), not every turn — a snapshot, not a live ticker.
_lim = state.subscription_limit_micros or 0
_used_micros = max(0, min(_lim, _lim - state.subscription_micros))
_used_usd = f"{_used_micros / 1_000_000:.2f}" if _lim else "?"
_glyph = "⚠" if _level == "warn" else "•"
to_show.append(
AgentNotice(
text=f"{'⚠' if _level == 'warn' else '•'} Credits {target_band}% used · ${_cap_usd} cap",
text=f"{_glyph} You've used ${_used_usd} of your ${_cap_usd} cap",
level=_level,
kind=CREDITS_NOTICE_KIND,
key=CREDITS_USAGE_KEY,
@@ -332,7 +375,17 @@ def evaluate_credits_notices(
latch["usage_band"] = target_band
# ── grant_spent ──────────────────────────────────────────────────────────
if grant_cond and "credits.grant_spent" not in active:
# The crossing gate guards only the SHOW and is CONSUMED by it — one
# announcement per crossing. A header flicker (uf → None → back to 1.0)
# clears the sticky line via grant_cond but cannot re-announce; only a
# renewal that re-opens the gate (a fresh ≥1¢ observation) arms the next
# announcement. .get(): default closed for any hand-built latch missing
# the key, so a first observation can never fire this notice.
if (
grant_cond
and "credits.grant_spent" not in active
and latch.get("seen_grant_unspent", False)
):
to_show.append(
AgentNotice(
text=f"• Grant spent · ${state.purchased_usd} top-up left",
@@ -343,6 +396,7 @@ def evaluate_credits_notices(
)
)
active.add("credits.grant_spent")
latch["seen_grant_unspent"] = False
elif "credits.grant_spent" in active and not grant_cond:
to_clear.append("credits.grant_spent")
active.discard("credits.grant_spent")
@@ -618,7 +672,8 @@ _DEV_FIXTURES: dict[str, dict] = {
subscription_limit_micros=20_000_000, subscription_limit_usd="20.00",
denominator_kind="subscription_cap", paid_access=True,
),
"grant_exhausted": dict( # used_fraction == 1.0 + purchased>0 → credits.grant_spent
"grant_exhausted": dict( # uf == 1.0 + purchased>0 → SILENT at open (crossing-gated);
# flip healthy → grant_exhausted via the fixture-file path to see credits.grant_spent
remaining_micros=12_340_000, remaining_usd="12.34",
subscription_micros=0, subscription_usd="0.00",
subscription_limit_micros=20_000_000, subscription_limit_usd="20.00",
@@ -732,6 +787,9 @@ def _hydrate_seed_state(agent, state) -> None:
agent._credits_session_start_micros = state.remaining_micros
_latch = getattr(agent, "_credits_latch", None)
if isinstance(_latch, dict) and state.used_fraction is not None:
# Prime ONLY seen_below_90 (open-high band warnings are wanted at open).
# Never prime seen_grant_unspent here: a seed observing grant-spent is a
# steady state, and priming it would revive the every-session nag.
_latch["seen_below_90"] = True
emit = getattr(agent, "_emit_credits_notices", None)
if callable(emit):
+18 -16
View File
@@ -138,8 +138,8 @@ def is_paused() -> bool:
def _load_config() -> Dict[str, Any]:
"""Read curator.* config from ~/.hermes/config.yaml. Tolerates missing file."""
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e:
logger.debug("Failed to load config for curator: %s", e)
return {}
@@ -325,7 +325,7 @@ def apply_automatic_transitions(now: Optional[datetime] = None) -> Dict[str, int
counts = {"marked_stale": 0, "archived": 0, "reactivated": 0, "checked": 0, "seeded": 0}
for row in _u.agent_created_report():
for row in _u.curated_report():
counts["checked"] += 1
name = row["name"]
if row.get("pinned"):
@@ -422,7 +422,9 @@ CURATOR_REVIEW_PROMPT = (
"INSTRUCTIONS AND EXPERIENTIAL KNOWLEDGE. A collection of hundreds of "
"narrow skills where each one captures one session's specific bug is "
"a FAILURE of the library — not a feature. An agent searching skills "
"matches on descriptions, not on exact names; one broad umbrella "
"matches on descriptions, not on exact names (note: long descriptions "
"are truncated to 57 chars in the system prompt skill index — keep the "
"trigger class in that window). One broad umbrella "
"skill with labeled subsections beats five narrow siblings for "
"discoverability, not the other way around.\n\n"
"The right target shape is CLASS-LEVEL skills with rich SKILL.md "
@@ -900,7 +902,6 @@ def _reconcile_classification(
Every removed skill is placed in exactly one bucket.
"""
heur_cons = {e["name"]: e for e in heuristic.get("consolidated", [])}
heur_pruned = {e["name"] for e in heuristic.get("pruned", [])}
model_cons = {e["from"]: e for e in model_block.get("consolidations", [])}
model_pruned = {e["name"]: e for e in model_block.get("prunings", [])}
@@ -1470,15 +1471,16 @@ def _render_report_markdown(p: Dict[str, Any]) -> str:
# ---------------------------------------------------------------------------
def _render_candidate_list() -> str:
"""Human/agent-readable list of agent-created skills with usage stats."""
rows = skill_usage.agent_created_report()
"""Human/agent-readable list of curator-managed skills with usage stats."""
rows = skill_usage.curated_report()
if not rows:
return "No agent-created skills to review."
return "No curator-managed skills to review."
cron_referenced = _cron_referenced_skills()
lines = [f"Agent-created skills ({len(rows)}):\n"]
lines = [f"Curator-managed skills ({len(rows)}):\n"]
for r in rows:
lines.append(
f"- {r['name']} "
f"provenance={r.get('provenance', 'agent')} "
f"state={r['state']} "
f"pinned={'yes' if r.get('pinned') else 'no'} "
f"cron={'yes' if r['name'] in cron_referenced else 'no'} "
@@ -1531,7 +1533,7 @@ def run_curator_review(
if dry_run:
# Count candidates without mutating state.
try:
report = skill_usage.agent_created_report()
report = skill_usage.curated_report()
counts = {
"checked": len(report),
"marked_stale": 0,
@@ -1584,7 +1586,7 @@ def run_curator_review(
nonlocal auto_summary
# Snapshot skill state BEFORE the LLM pass so the report can diff.
try:
before_report = skill_usage.agent_created_report()
before_report = skill_usage.curated_report()
except Exception:
before_report = []
before_names = {r.get("name") for r in before_report if isinstance(r, dict)}
@@ -1610,7 +1612,7 @@ def run_curator_review(
state2["last_run_duration_seconds"] = elapsed
state2["last_run_summary"] = final_summary
try:
after_report = skill_usage.agent_created_report()
after_report = skill_usage.curated_report()
except Exception:
after_report = []
try:
@@ -1697,7 +1699,7 @@ def run_curator_review(
try:
rename_lines = _build_rename_summary(
before_names=before_names,
after_report=skill_usage.agent_created_report(),
after_report=skill_usage.curated_report(),
tool_calls=llm_meta.get("tool_calls", []) or [],
model_final=llm_meta.get("final", "") or "",
)
@@ -1715,7 +1717,7 @@ def run_curator_review(
# reporting bug never breaks the curator itself. Report path is
# recorded in state so `hermes curator status` can point at it.
try:
after_report = skill_usage.agent_created_report()
after_report = skill_usage.curated_report()
except Exception:
after_report = []
try:
@@ -1873,9 +1875,9 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
_acp_args = None
_model_name = ""
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
from hermes_cli.runtime_provider import resolve_runtime_provider
_cfg = load_config()
_cfg = load_config_readonly()
_binding = _resolve_review_runtime(_cfg)
_provider, _model_name = _binding.provider, _binding.model
_rp = resolve_runtime_provider(
+2 -2
View File
@@ -147,8 +147,8 @@ def _utc_id(now: Optional[datetime] = None) -> str:
def _load_config() -> Dict[str, Any]:
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e:
logger.debug("Failed to load config for curator backup: %s", e)
return {}
+85
View File
@@ -0,0 +1,85 @@
"""Context-local state for delegate_task child execution.
The parent Hermes process may itself be a Kanban dispatcher worker with
HERMES_KANBAN_* variables in process env. delegate_task children run inside the
same Python process, but they are not dispatcher-owned Kanban workers. This
module lets code paths that resolve tool schemas or spawn subprocesses fail
closed for delegated children without mutating global os.environ for the parent.
"""
from __future__ import annotations
from contextlib import contextmanager
from contextvars import ContextVar
from typing import Iterator, Mapping, MutableMapping
_DELEGATED_CHILD_CONTEXT: ContextVar[bool] = ContextVar(
"hermes_delegated_child_context",
default=False,
)
DELEGATED_CHILD_ENV_MARKER = "HERMES_DELEGATED_CHILD_CONTEXT"
KANBAN_ENV_KEYS: tuple[str, ...] = (
"HERMES_KANBAN_TASK",
"HERMES_KANBAN_RUN_ID",
"HERMES_KANBAN_WORKSPACE",
"HERMES_KANBAN_WORKSPACES_ROOT",
"HERMES_KANBAN_CLAIM_LOCK",
"HERMES_KANBAN_BOARD",
"HERMES_KANBAN_DB",
)
@contextmanager
def delegated_child_context() -> Iterator[None]:
"""Mark the current execution context as a delegate_task child."""
token = _DELEGATED_CHILD_CONTEXT.set(True)
try:
yield
finally:
_DELEGATED_CHILD_CONTEXT.reset(token)
def is_delegated_child_context() -> bool:
"""Return True while code is running for a delegate_task child."""
return bool(_DELEGATED_CHILD_CONTEXT.get())
def is_delegated_child_process_context() -> bool:
"""Return True in this process or a subprocess spawned by a child."""
import os
return bool(_DELEGATED_CHILD_CONTEXT.get()) or bool(
os.environ.get(DELEGATED_CHILD_ENV_MARKER)
)
def scrub_kanban_env(env: Mapping[str, str] | MutableMapping[str, str]) -> dict[str, str]:
"""Return *env* with dispatcher-only Kanban variables removed."""
cleaned = dict(env)
for key in KANBAN_ENV_KEYS:
cleaned.pop(key, None)
cleaned[DELEGATED_CHILD_ENV_MARKER] = "1"
return cleaned
def delegated_child_subprocess_env(
env: Mapping[str, str] | MutableMapping[str, str] | None = None,
) -> dict[str, str] | None:
"""Return an env override only when delegated-child lineage must cross fork.
Most subprocess call sites historically used ``env=None`` to inherit the
process environment. In a ``delegate_task`` child, inheriting as-is leaks
parent dispatcher ``HERMES_KANBAN_*`` vars while losing the ContextVar in
the new process. This helper preserves normal ``env=None`` semantics for
non-delegated calls, and only materializes a scrubbed env when the lineage
marker must be propagated across a child-process boundary.
"""
if not is_delegated_child_process_context():
return None if env is None else dict(env)
if env is None:
import os
env = os.environ
return scrub_kanban_env(env)
+46
View File
@@ -645,6 +645,52 @@ def verb_drops_preview(tool_name: str) -> bool:
return tool_name in _TOOL_VERBS_NO_PREVIEW
def build_status_phrase(tool_name: str, args: dict | None, max_len: int = 49) -> str | None:
"""Build a short present-tense status phrase for platform status surfaces.
Used by text-rendering "typing" indicators (Slack's
``assistant.threads.setStatus`` line) to show what the agent is doing
right now: ``is running scripts/run_tests.sh…`` instead of a static
``is thinking...``. The phrase is phrased to follow the bot's display
name ("Hermes is running …"), so it starts lowercase with "is".
Pass ``args=None`` for a verb-only phrase (``is running…``) — used when
``display.live_status`` is ``verb`` to keep argument previews out of
shared channels.
Returns None for the ``_thinking`` pseudo-tool and when friendly labels
are disabled (callers fall back to their static default). ``max_len``
caps the total phrase length; Slack truncates its status line around 50
characters, so the default stays just under that.
"""
if not tool_name or tool_name == "_thinking":
return None
if not _friendly_tool_labels:
return None
verb = _TOOL_VERBS.get(tool_name)
if verb:
head = f"is {verb[0].lower()}{verb[1:]}"
else:
# Custom / plugin / MCP tools: generic but still informative.
head = f"is using {tool_name}"
phrase = head
if args and verb and tool_name not in _TOOL_VERBS_NO_PREVIEW:
preview = build_tool_preview(tool_name, args, max_len=None)
if preview:
# Previews can contain newlines (terminal commands); keep the
# status to the first line.
preview = preview.splitlines()[0].strip()
phrase = f"{head}{tool_verb_connector(tool_name)}{preview}"
if len(phrase) > max_len - 1:
phrase = phrase[: max_len - 2].rstrip() + "…"
else:
phrase = phrase + "…"
return phrase
def build_tool_label(tool_name: str, args: dict, max_len: int | None = None) -> str | None:
"""Build a human-phrased status label for a tool call.
+147 -3
View File
@@ -159,6 +159,14 @@ _RATE_LIMIT_PATTERNS = [
"throttlingexception",
"too many concurrent requests",
"servicequotaexceededexception",
# Generic throttle prefix — Bedrock (and some proxies) surface throttling
# as "Throttling error: Too many tokens, please wait before trying
# again." Without this entry the message falls through to the
# context-overflow list (which contains "too many tokens") and the retry
# loop compresses a healthy session instead of backing off. Matched
# BEFORE _CONTEXT_OVERFLOW_PATTERNS in the message-only path, so the
# throttle wins. (port of anomalyco/opencode#37848's exclusion guard)
"throttling",
]
# Patterns that indicate provider-side overload, NOT a per-credential rate
@@ -212,6 +220,12 @@ _PAYLOAD_TOO_LARGE_PATTERNS = [
"request entity too large",
"payload too large",
"error code: 413",
# Anthropic's structured 413 error type. Normally arrives with an HTTP
# 413 status (handled by the status path), but aggregators/proxies can
# re-wrap it into a plain message with no status attribute — route it to
# the same compression recovery. (port of anomalyco/opencode#37848)
"request_too_large",
"request exceeds the maximum size",
]
# Image-size patterns. Matched against 400 bodies (not 413) because most
@@ -269,6 +283,11 @@ _CONTEXT_OVERFLOW_PATTERNS = [
"context window",
"prompt is too long",
"prompt exceeds max length",
# NOTE: bare "max_tokens" is load-bearing — the output-cap-retry path keys
# off it (e.g. "max_tokens: 65536 > context_window: 200000 ..."). Do NOT
# remove it. Provider empty-response advisories also contain "very low
# max_tokens", but those are intercepted by _EMPTY_PROVIDER_RESPONSE_PATTERNS
# BEFORE this list is consulted, so they never mis-route into compression.
"max_tokens",
"maximum number of tokens",
# vLLM / local inference server patterns
@@ -293,6 +312,10 @@ _CONTEXT_OVERFLOW_PATTERNS = [
"max input token",
"input token",
"exceeds the maximum number of input tokens",
# Together/Fireworks-style: "Input length 131393 exceeds the maximum
# allowed input length of 131040 tokens." No other pattern in this list
# matches that wording. (port of anomalyco/opencode#37848)
"maximum allowed input length",
]
# Model not found patterns
@@ -316,6 +339,30 @@ _MODEL_NOT_FOUND_PATTERNS = [
"no endpoints found that support tool use",
]
# Malformed-message-array 400s. Deterministic request-shape rejections that
# describe the *transcript* being invalid, not a parameter. The canonical
# case: a stream dies mid-response and Hermes persists a content-less
# assistant stub; on the next turn the Anthropic message schema (and the
# litellm/Bedrock proxies in front of it) reject the whole request with
# "all messages must have non-empty content except for the optional final
# assistant message" / errorCode INVALID_REQUEST_BODY
# These are NOT context overflow — the input may be tiny — but a large
# session used to mis-route them into the compression loop via the generic
# "400 + large session" heuristic below, ending in "Cannot compress further"
# every retry (the input is unchanged, so compression cannot help). Match
# the message-shape signals explicitly and fail fast as a format_error so the
# loop stops looping. The empty-stub creation is the root cause (fixed in
# chat_completion_helpers); this pattern stops the misclassification symptom
# for transcripts that already contain a poisoned stub.
_INVALID_MESSAGE_BODY_PATTERNS = [
"must have non-empty content",
"messages must have non-empty",
"invalid_request_body",
"text content blocks must be non-empty",
"content field is required",
"messages: at least one message is required",
]
# Request-validation patterns — the request is malformed and will fail
# identically on every retry. Some OpenAI-compatible gateways (notably
# codex.nekos.me) return these as 5xx instead of the standard 4xx, which
@@ -408,6 +455,7 @@ _CONTENT_POLICY_BLOCKED_PATTERNS = [
_AUTH_PATTERNS = [
"invalid api key",
"invalid_api_key",
"gateway_auth_failed",
"authentication",
"unauthorized",
"forbidden",
@@ -426,6 +474,19 @@ _THINKING_SIG_PATTERNS = [
# the exception type is generic (e.g. RuntimeError from a local shim that
# wraps a subprocess timeout). Checked before the type-based transport
# heuristics so custom-provider "timed out" errors don't fall through to
# Provider empty-response advisories (OpenRouter / nano-gpt / similar).
# Checked before context-overflow matching because the advisory text often
# mentions "max_tokens" as a possible cause, which historically sat in
# _CONTEXT_OVERFLOW_PATTERNS and sent healthy sessions into a compression
# death spiral ending in "Cannot compress further".
_EMPTY_PROVIDER_RESPONSE_PATTERNS = [
"returned an empty response",
"empty response despite retries",
"provider returned an empty response",
"model returning empty responses",
"empty response stream",
]
# the unknown bucket and get misreported as empty responses.
_TIMEOUT_MESSAGE_PATTERNS = [
"timed out",
@@ -1077,6 +1138,14 @@ def _classify_by_status(
# remaining explicit context-overflow signal routes into the
# compression-and-retry path (mirroring _classify_400) instead of
# blind server_error retries that exhaust and drop the turn.
# Empty-response advisories that mention "max_tokens" must not enter
# that compression path.
if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS):
return result_fn(
FailoverReason.server_error,
retryable=True,
should_compress=False,
)
if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS):
return result_fn(
FailoverReason.context_overflow,
@@ -1090,6 +1159,12 @@ def _classify_by_status(
# Cloudflare/Tailscale hop relabeling the status). Route explicit
# overflow bodies into compression; otherwise treat as transient
# overload and retry.
if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS):
return result_fn(
FailoverReason.server_error,
retryable=True,
should_compress=False,
)
if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS):
return result_fn(
FailoverReason.context_overflow,
@@ -1215,8 +1290,8 @@ def _classify_400(
# returns:
# "Unsupported parameter: 'max_tokens' is not supported with this model.
# Use 'max_completion_tokens' instead."
# That string contains the literal substring "max_tokens", which is one of
# the _CONTEXT_OVERFLOW_PATTERNS — so without this guard the 400 is
# That string contains the literal substring "max_tokens", which historically
# sat in _CONTEXT_OVERFLOW_PATTERNS — so without this guard the 400 is
# misclassified as context_overflow, routed into the compression loop,
# re-sent with the same bad parameter, and ends in "Cannot compress
# further". These errors are deterministic (every retry gets the identical
@@ -1238,6 +1313,44 @@ def _classify_400(
should_fallback=True,
)
# Malformed message array (empty-content assistant stub, etc.). Must be
# checked BEFORE context_overflow: the input can be tiny, so the generic
# "400 + large session" heuristic would otherwise mis-route it into the
# compression loop and thrash until "Cannot compress further" on every
# retry (the request is unchanged, so compression cannot fix it). This is
# a deterministic request-shape rejection — fail fast as a non-retryable
# format_error and fall back. Checked against the message text AND the
# structured error code, since proxies (litellm/Bedrock) surface the
# signal in errorCode=INVALID_REQUEST_BODY.
if (
any(p in error_msg for p in _INVALID_MESSAGE_BODY_PATTERNS)
or error_code_lower == "invalid_request_body"
):
logger.warning(
"Malformed message array 400 (invalid request body) classified as "
"format_error, NOT context overflow — failing fast + falling back "
"instead of entering the compression loop. This usually means an "
"empty-content assistant stub is in the transcript; num_messages=%s "
"approx_tokens=%s. error=%.200s",
num_messages, approx_tokens, error_msg,
)
return result_fn(
FailoverReason.format_error,
retryable=False,
should_fallback=True,
)
# Empty-provider-response advisories must not enter compression. They
# often mention "max_tokens" as a possible cause and used to match the
# bare overflow pattern, then thrash compress until "Cannot compress
# further" on an otherwise healthy session (custom endpoints / nano-gpt).
if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS):
return result_fn(
FailoverReason.server_error,
retryable=True,
should_compress=False,
)
# Context overflow from 400
if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS):
return result_fn(
@@ -1287,6 +1400,18 @@ def _classify_400(
# Responses API (and some providers) use flat body: {"message": "..."}
if not err_body_msg:
err_body_msg = str(body.get("message") or "").strip().lower()
# litellm / Bedrock proxies use a custom shape: {"errorMessage": "...",
# "errorCode": "...", "errorArgs": {"reason": "..."}}. Without these
# keys err_body_msg stays "" and a long, descriptive rejection is
# wrongly treated as a "generic" (bare) error below, which — on a
# large session — mis-routes into the compression loop. Recognize
# them so the is_generic heuristic sees the real message length.
if not err_body_msg:
err_body_msg = str(body.get("errorMessage") or "").strip().lower()
if not err_body_msg:
_args = body.get("errorArgs")
if isinstance(_args, dict):
err_body_msg = str(_args.get("reason") or "").strip().lower()
is_generic = len(err_body_msg) < 30 or err_body_msg in {"error", ""}
# Absolute token/message-count thresholds are only a proxy for smaller
# context windows. Large-context sessions can have many messages while
@@ -1441,6 +1566,15 @@ def _classify_by_message(
should_fallback=True,
)
# Empty-provider-response advisories (often mention "max_tokens") must
# retry without compression — see the matching 400-path guard above.
if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS):
return result_fn(
FailoverReason.server_error,
retryable=True,
should_compress=False,
)
# Context overflow patterns
if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS):
return result_fn(
@@ -1576,7 +1710,7 @@ def _extract_error_code(body: dict) -> str:
return nested_code
# Top-level code
code = body.get("code") or body.get("error_code") or ""
code = body.get("code") or body.get("error_code") or body.get("errorCode") or ""
if isinstance(code, (str, int)):
text = str(code).strip()
if text and text != "400":
@@ -1596,6 +1730,16 @@ def _extract_message(error: Exception, body: dict) -> str:
msg = body.get("message", "")
if isinstance(msg, str) and msg.strip():
return msg.strip()[:500]
# litellm / Bedrock proxy shape: {"errorMessage": "...",
# "errorArgs": {"reason": "..."}}.
msg = body.get("errorMessage", "")
if isinstance(msg, str) and msg.strip():
return msg.strip()[:500]
args = body.get("errorArgs")
if isinstance(args, dict):
reason = args.get("reason", "")
if isinstance(reason, str) and reason.strip():
return reason.strip()[:500]
# Fallback to str(error)
return str(error)[:500]
+3
View File
@@ -46,6 +46,9 @@ def build_write_denied_paths(home: str) -> set[str]:
# Top-level Anthropic PKCE credential store remains sensitive even
# when a profile is active; default/non-profile sessions still read it.
str(hermes_root / ".anthropic_oauth.json"),
# Bitwarden Secrets Manager encrypted disk cache.
str(hermes_home / "cache" / "bws_cache.enc.json"),
str(hermes_root / "cache" / "bws_cache.enc.json"),
os.path.join(home, ".netrc"),
os.path.join(home, ".pgpass"),
os.path.join(home, ".npmrc"),
+58 -8
View File
@@ -73,7 +73,7 @@ def probe_gemini_tier(
api_key: str,
base_url: str = DEFAULT_GEMINI_BASE_URL,
*,
model: str = "gemini-2.5-flash",
model: str = "gemini-3.6-flash",
timeout: float = 10.0,
) -> str:
"""Probe a Google AI Studio API key and return its tier.
@@ -154,8 +154,8 @@ def is_free_tier_quota_error(error_message: str) -> bool:
_FREE_TIER_GUIDANCE = (
"\n\nYour Google API key is on the free tier (<= 250 requests/day for "
"gemini-2.5-flash). Hermes typically makes 3-10 API calls per user turn, "
"\n\nYour Google API key is on the free tier (a few hundred requests/day "
"for Gemini Flash models). Hermes typically makes 3-10 API calls per user turn, "
"so the free tier is exhausted in a handful of messages and cannot sustain "
"an agent session. Enable billing on your Google Cloud project and "
"regenerate the key in a billing-enabled project: "
@@ -163,6 +163,42 @@ _FREE_TIER_GUIDANCE = (
)
def is_standard_key_auth_error(
status: int, error_message: str, reason: str = ""
) -> bool:
"""Return True when a Gemini 401 indicates Google rejected the key TYPE.
Google began rejecting unrestricted legacy "Standard" Google Cloud API
keys on the Gemini API on June 19, 2026, and ALL Standard keys stop
working in September 2026. The rejection surfaces as a misleading 401
telling the user to supply an OAuth 2 access token ("Request had invalid
authentication credentials. Expected OAuth 2 access token, login cookie
or other valid authentication credential."), optionally carrying
``google.rpc.ErrorInfo`` reason ``ACCESS_TOKEN_TYPE_UNSUPPORTED``.
Scoped narrowly so a plain bad key (reason ``API_KEY_INVALID``,
"API key not valid") keeps its existing message.
"""
if status != 401:
return False
if reason == "ACCESS_TOKEN_TYPE_UNSUPPORTED":
return True
return "expected oauth 2 access token" in (error_message or "").lower()
_STANDARD_KEY_GUIDANCE = (
"\n\nGoogle Gemini rejected this API key's type — you do NOT need OAuth. "
"Google began rejecting legacy 'Standard' Google Cloud keys for the "
"Gemini API on June 19, 2026, and all Standard keys stop working in "
"September 2026. Open https://aistudio.google.com/api-keys, check the "
"key's type and status, and create a replacement Gemini API key (or, as "
"a temporary bridge, restrict the Standard key to "
"generativelanguage.googleapis.com). Then update GEMINI_API_KEY / "
"GOOGLE_API_KEY in ~/.hermes/.env and restart your session. "
"Details: https://ai.google.dev/gemini-api/docs/api-key"
)
class GeminiAPIError(Exception):
"""Error shape compatible with Hermes retry/error classification."""
@@ -270,8 +306,12 @@ def _translate_tool_call_to_gemini(tool_call: Dict[str, Any]) -> Dict[str, Any]:
}
}
thought_signature = _tool_call_extra_signature(tool_call)
if thought_signature:
part["thoughtSignature"] = thought_signature
# Fallback sentinel for cross-provider tool_calls (e.g. fallback from
# xAI/Anthropic to Gemini, where the original tool_call carries no
# Gemini thoughtSignature). Mirrors gemini_cloudcode_adapter.py:106.
# Without this, Gemini 3 thinking models reject replayed history with
# 400 INVALID_ARGUMENT on the missing thoughtSignature.
part["thoughtSignature"] = thought_signature or "skip_thought_signature_validator"
return part
@@ -281,9 +321,13 @@ def _translate_tool_result_to_gemini(
) -> Dict[str, Any]:
tool_name_by_call_id = tool_name_by_call_id or {}
tool_call_id = str(message.get("tool_call_id") or "")
# A tool result can carry the unwrapped internal tool name (for example,
# an MCP tool invoked through the `tool_call` bridge). Gemini requires
# functionResponse.name to echo the matching functionCall.name, so the
# call-id mapping must take precedence over the internal result name.
name = str(
message.get("name")
or tool_name_by_call_id.get(tool_call_id)
tool_name_by_call_id.get(tool_call_id)
or message.get("name")
or tool_call_id
or "tool"
)
@@ -820,6 +864,12 @@ def gemini_http_error(
if status == 429 and is_free_tier_quota_error(err_message or body_text):
message = message + _FREE_TIER_GUIDANCE
# Legacy "Standard" Google Cloud key rejection (June 19, 2026 onward) ->
# Google's raw 401 misleadingly tells the user to use OAuth. Append the
# actual fix (mint a new Gemini API key in AI Studio).
if is_standard_key_auth_error(status, err_message or body_text, reason):
message = message + _STANDARD_KEY_GUIDANCE
return GeminiAPIError(
message,
code=code,
@@ -930,7 +980,7 @@ class GeminiNativeClient:
def _create_chat_completion(
self,
*,
model: str = "gemini-2.5-flash",
model: str = "gemini-3.6-flash",
messages: Optional[List[Dict[str, Any]]] = None,
stream: bool = False,
tools: Any = None,
+23 -6
View File
@@ -2,6 +2,7 @@
from __future__ import annotations
import math
from typing import Any, Dict
# Gemini's ``FunctionDeclaration.parameters`` field accepts the ``Schema``
@@ -76,15 +77,31 @@ def sanitize_gemini_schema(schema: Any) -> Dict[str, Any]:
# Gemini's Schema validator requires every ``enum`` entry to be a string,
# even when the parent ``type`` is ``integer`` / ``number`` / ``boolean``.
# OpenAI / OpenRouter / Anthropic accept typed enums (e.g. Discord's
# ``auto_archive_duration: {type: integer, enum: [60, 1440, 4320, 10080]}``),
# so we only drop the ``enum`` when it would collide with Gemini's rule.
# Keeping ``type: integer`` plus the human-readable description gives the
# model enough guidance; the tool handler still validates the value.
# Preserve those constraints by stringifying scalar values while keeping
# the declared type intact; Gemini uses the strings as schema metadata and
# still emits typed tool arguments at runtime.
enum_val = cleaned.get("enum")
type_val = cleaned.get("type")
if isinstance(enum_val, list) and type_val in {"integer", "number", "boolean"}:
if any(not isinstance(item, str) for item in enum_val):
stringified = []
for item in enum_val:
if isinstance(item, str):
value = item
elif isinstance(item, bool):
value = "true" if item else "false"
elif (
isinstance(item, (int, float))
and not isinstance(item, bool)
and math.isfinite(item)
):
value = str(item)
else:
continue
if value not in stringified:
stringified.append(value)
if stringified:
cleaned["enum"] = stringified
else:
cleaned.pop("enum", None)
# Gemini validates ``required`` strictly against the same node's
+9 -29
View File
@@ -25,14 +25,14 @@ Language resolution order:
3. ``display.language`` from config.yaml
4. ``"en"`` (baseline)
Supported languages: en, zh, ja, de, es, fr, tr, uk. Unknown values fall back to en.
Supported languages: en, zh, zh-hant, ja, de, es, fr, tr, uk, af, ko, it, ga,
pt, ru, hu, ar. Unknown values fall back to en.
"""
from __future__ import annotations
import logging
import os
import sysconfig
import threading
from functools import lru_cache
from pathlib import Path
@@ -42,7 +42,7 @@ logger = logging.getLogger(__name__)
SUPPORTED_LANGUAGES: tuple[str, ...] = (
"en", "zh", "zh-hant", "ja", "de", "es", "fr", "tr", "uk",
"af", "ko", "it", "ga", "pt", "ru", "hu",
"af", "ko", "it", "ga", "pt", "ru", "hu", "ar",
)
DEFAULT_LANGUAGE = "en"
@@ -79,6 +79,9 @@ _LANGUAGE_ALIASES: dict[str, str] = {
"russian": "ru", "русский": "ru", "ru-ru": "ru",
# Hungarian
"hungarian": "hu", "magyar": "hu", "hu-hu": "hu",
# Arabic — bare "arabic"/endonym plus the common regional BCP-47 tags.
"arabic": "ar", "العربية": "ar",
"ar-sa": "ar", "ar-eg": "ar", "ar-ae": "ar", "ar-ma": "ar", "ar-dz": "ar",
}
_catalog_cache: dict[str, dict[str, str]] = {}
@@ -92,12 +95,8 @@ def _locales_dir() -> Path:
1. ``HERMES_BUNDLED_LOCALES`` env var -- set by the Nix wrapper (or any
sealed-packaging system) to point at the installed catalog directory.
2. ``<repo-root>/locales`` -- source checkouts and ``pip install -e .``,
2. ``<repo-root>/locales`` -- source checkouts and editable installs,
where the working tree sits next to ``agent/``.
3. ``<sysconfig data|purelib|platlib>/locales`` -- pip wheel installs.
setuptools ``data-files`` extracts ``locales/*.yaml`` under the
interpreter's ``data`` scheme; the other schemes are checked as a
safety net for nonstandard layouts.
Falling through to the source-style path (even when missing) keeps
``_load_catalog`` error messages informative -- it logs the path it
@@ -116,25 +115,6 @@ def _locales_dir() -> Path:
# agent/i18n.py -> agent/ -> repo root (source checkout, editable install)
source_dir = Path(__file__).resolve().parent.parent / "locales"
if source_dir.is_dir():
return source_dir
# pip wheel install: data-files lands under the interpreter data scheme.
# ``data`` (== sys.prefix in a venv) is where setuptools data-files extract
# and is checked first. ``purelib``/``platlib`` (site-packages) are a safety
# net for nonstandard layouts. NOTE: this does NOT cover ``pip install
# --user`` (user scheme, ~/.local/locales) or ``pip install --target`` --
# both are out of scope; see the plan header.
for scheme in ("data", "purelib", "platlib"):
raw = sysconfig.get_path(scheme)
if not raw:
continue
candidate = Path(raw) / "locales"
if candidate.is_dir():
return candidate
# Last resort: return the source-style path so _load_catalog's catalog-missing
# log (logger.debug "i18n catalog missing for %s at %s") stays informative.
return source_dir
@@ -217,8 +197,8 @@ def _config_language_cached() -> str | None:
(e.g. after the setup wizard).
"""
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
lang = (cfg.get("display") or {}).get("language")
if lang:
return _normalize_lang(lang)
+2 -2
View File
@@ -91,9 +91,9 @@ def get_active_provider() -> Optional[ImageGenProvider]:
"""
configured: Optional[str] = None
try:
from hermes_cli.config import load_config
from hermes_cli.config import load_config_readonly
cfg = load_config()
cfg = load_config_readonly()
section = cfg.get("image_gen") if isinstance(cfg, dict) else None
if isinstance(section, dict):
raw = section.get("provider")
+89 -35
View File
@@ -181,6 +181,8 @@ def _supports_vision_override(
cfg: Optional[Dict[str, Any]],
provider: str,
model: str,
*,
requested_provider: str = "",
) -> Optional[bool]:
"""Resolve user-declared vision capability from config.yaml.
@@ -188,9 +190,14 @@ def _supports_vision_override(
1. ``model.supports_vision`` (top-level shortcut for the active model)
2. ``providers.<provider>.models.<model>.supports_vision``
(named custom providers — ``provider`` may be the runtime-resolved
value ``"custom"`` and/or the user-declared name under
``model.provider``; both are tried. For ``custom:<name>`` syntax,
the stripped ``<name>`` is also tried as a provider key.)
value ``"custom"``, the runtime's originally requested provider,
and/or the user-declared name under ``model.provider``; all are
tried. For ``custom:<name>`` syntax, the stripped ``<name>`` is also
tried as a provider key.)
2b. ``custom_providers`` (legacy list form) ``.models.<model>``
Under (2) and (2b), the per-model capability key may be written as
either ``supports_vision`` or the shorter ``vision`` alias; both work.
Returns None when no override is set, so the caller falls through to
models.dev. Returns False explicitly only when the user wrote a
@@ -210,23 +217,30 @@ def _supports_vision_override(
# get rewritten to provider="custom" at runtime
# (hermes_cli/runtime_provider.py:_resolve_named_custom_runtime), so the
# config still holds the user-declared name under model.provider. Try
# both as candidate provider keys, plus the stripped suffix from
# "custom:<name>" (where <name> is the key under providers:).
# both as candidate provider keys. Either identity may use the
# "custom:<name>" form while providers: is keyed by bare <name>.
config_provider = str(model_cfg.get("provider") or "").strip()
# Extract the stripped name from "custom:<name>" if present
stripped_suffix = ""
if config_provider.startswith("custom:"):
stripped_suffix = config_provider[len("custom:"):]
provider_candidates: List[str] = []
for candidate in (requested_provider, provider, config_provider):
if not candidate:
continue
provider_candidates.append(candidate)
if candidate.startswith("custom:"):
stripped_candidate = candidate[len("custom:"):]
if stripped_candidate:
provider_candidates.append(stripped_candidate)
providers_raw = cfg.get("providers")
providers_cfg: Dict[str, Any] = providers_raw if isinstance(providers_raw, dict) else {}
for p in dict.fromkeys(filter(None, (provider, config_provider, stripped_suffix))):
for p in dict.fromkeys(provider_candidates):
entry_raw = providers_cfg.get(p)
entry: Dict[str, Any] = entry_raw if isinstance(entry_raw, dict) else {}
models_raw = entry.get("models")
models_cfg: Dict[str, Any] = models_raw if isinstance(models_raw, dict) else {}
per_model_raw = models_cfg.get(model)
per_model: Dict[str, Any] = per_model_raw if isinstance(per_model_raw, dict) else {}
coerced = _coerce_capability_bool(per_model.get("supports_vision"))
coerced = _coerce_capability_bool(
per_model.get("supports_vision", per_model.get("vision"))
)
if coerced is not None:
return coerced
@@ -235,28 +249,26 @@ def _supports_vision_override(
# may appear as the raw name or "custom:<name>" at runtime).
custom_providers = cfg.get("custom_providers")
if isinstance(custom_providers, list):
# Build candidate names: the provider value and the config provider
# value, both raw and with "custom:" prefix stripped/added.
candidate_names: set = set()
for p in filter(None, (provider, config_provider)):
candidate_names.add(p)
if p.startswith("custom:"):
candidate_names.add(p[len("custom:"):])
else:
candidate_names.add(f"custom:{p}")
for entry_raw in custom_providers:
if not isinstance(entry_raw, dict):
continue
entry_name = str(entry_raw.get("name") or "").strip()
if entry_name not in candidate_names:
continue
models_raw = entry_raw.get("models")
models_cfg = models_raw if isinstance(models_raw, dict) else {}
per_model_raw = models_cfg.get(model)
per_model = per_model_raw if isinstance(per_model_raw, dict) else {}
coerced = _coerce_capability_bool(per_model.get("supports_vision"))
if coerced is not None:
return coerced
# Candidate priority matters when the CLI-selected provider differs
# from model.provider. Walk identities first, then config entries, so
# list order cannot let the persisted default shadow the live route.
for candidate in dict.fromkeys(provider_candidates):
candidate_name = candidate.strip().lower()
for entry_raw in custom_providers:
if not isinstance(entry_raw, dict):
continue
entry_name = str(entry_raw.get("name") or "").strip().lower()
if entry_name != candidate_name:
continue
models_raw = entry_raw.get("models")
models_cfg = models_raw if isinstance(models_raw, dict) else {}
per_model_raw = models_cfg.get(model)
per_model = per_model_raw if isinstance(per_model_raw, dict) else {}
coerced = _coerce_capability_bool(
per_model.get("supports_vision", per_model.get("vision"))
)
if coerced is not None:
return coerced
return None
@@ -376,6 +388,8 @@ def _lookup_supports_vision(
provider: str,
model: str,
cfg: Optional[Dict[str, Any]] = None,
*,
requested_provider: str = "",
) -> Optional[bool]:
"""Return True/False if we can resolve caps, None if unknown.
@@ -383,7 +397,34 @@ def _lookup_supports_vision(
(so custom/local models declared as vision-capable don't fall through to
text routing in ``auto`` mode), then falls back to models.dev.
"""
override = _supports_vision_override(cfg, provider, model)
# Named custom providers are canonicalized to ``provider="custom"`` by
# runtime resolution. The original CLI/config name is carried in the
# context-local main runtime so capability lookup can still select the
# exact custom_providers entry. Require an exact provider+model match:
# background/auxiliary lookups must never borrow another turn's identity.
if not requested_provider:
try:
from agent.auxiliary_client import _runtime_main_value
runtime_provider = str(
_runtime_main_value("provider") or ""
).strip().lower()
runtime_model = str(_runtime_main_value("model") or "").strip()
lookup_provider = str(provider or "").strip().lower()
lookup_model = str(model or "").strip()
if runtime_provider == lookup_provider and runtime_model == lookup_model:
requested_provider = str(
_runtime_main_value("requested_provider") or ""
).strip()
except Exception:
pass
override = _supports_vision_override(
cfg,
provider,
model,
requested_provider=requested_provider,
)
if override is not None:
return override
if not provider or not model:
@@ -421,6 +462,8 @@ def decide_image_input_mode(
provider: str,
model: str,
cfg: Optional[Dict[str, Any]],
*,
requested_provider: str = "",
) -> str:
"""Return ``"native"`` or ``"text"`` for the given turn.
@@ -428,6 +471,7 @@ def decide_image_input_mode(
provider: active inference provider ID (e.g. ``"anthropic"``, ``"openrouter"``).
model: active model slug as it would be sent to the provider.
cfg: loaded config.yaml dict, or None. When None, behaves as auto.
requested_provider: provider identity before runtime canonicalization.
"""
mode_cfg = "auto"
if isinstance(cfg, dict):
@@ -444,7 +488,17 @@ def decide_image_input_mode(
# explicit auxiliary.vision config acts as a *fallback* for text-only
# main models — it should not preempt native vision on a model that
# can natively inspect the pixels (issue #29135).
supports = _lookup_supports_vision(provider, model, cfg)
if requested_provider:
supports = _lookup_supports_vision(
provider,
model,
cfg,
requested_provider=requested_provider,
)
else:
# Keep the long-standing three-argument call contract for callers and
# tests that replace the capability lookup hook.
supports = _lookup_supports_vision(provider, model, cfg)
if supports is True:
return "native"
if _explicit_aux_vision_override(cfg):
+7
View File
@@ -113,6 +113,13 @@ class InsightsEngine:
"""
cutoff = time.time() - (days * 86400)
# Token/cost totals may still sit on the SessionDB's async
# accounting queue; drain so the report reflects exact counters.
# (self.db may be a raw sqlite3 connection in tests — guard.)
flush = getattr(self.db, "flush_token_counts", None)
if callable(flush):
flush()
# Gather raw data
sessions = self._get_sessions(cutoff, source)
tool_usage = self._get_tool_usage(cutoff, source)
+2 -2
View File
@@ -2,7 +2,7 @@
Extracted from ``run_agent.py``. Each ``AIAgent`` instance (parent or
subagent) holds an :class:`IterationBudget`; the parent's cap comes from
``max_iterations`` (default 90), each subagent's cap comes from
``max_iterations`` (default 500), each subagent's cap comes from
``delegation.max_iterations`` (default 50).
``run_agent`` re-exports ``IterationBudget`` so existing
@@ -18,7 +18,7 @@ class IterationBudget:
"""Thread-safe iteration counter for an agent.
Each agent (parent or subagent) gets its own ``IterationBudget``.
The parent's budget is capped at ``max_iterations`` (default 90).
The parent's budget is capped at ``max_iterations`` (default 500).
Each subagent gets an independent budget capped at
``delegation.max_iterations`` (default 50) — this means total
iterations across parent + subagents can exceed the parent's cap.
-1
View File
@@ -403,7 +403,6 @@ def _category_counts(payload: dict[str, Any]) -> list[tuple[str, int]]:
def category_color_map(payload: dict[str, Any]) -> dict[str, str]:
"""Deterministic, evenly-spread hue per skill category (theme-independent)."""
clusters = _category_counts(payload)
n = max(1, len(clusters))
# Golden-angle hue spacing so adjacent categories never collide in color.
return {cat: rgb_to_hex(_hsl_to_rgb((i * 137.508) % 360, 0.55, 0.62)) for i, (cat, _c) in enumerate(clusters)}
+1 -1
View File
@@ -55,7 +55,7 @@ def register_subparser(subparsers: argparse._SubParsersAction) -> None:
help="Even attempt servers marked manual-install (best effort)",
)
sub_restart = sub.add_parser(
sub.add_parser(
"restart",
help="Tear down running LSP clients (next edit re-spawns)",
)
+149 -71
View File
@@ -18,9 +18,15 @@ into it via :func:`agent.lsp.manager.LSPService.touch_file`.
Implementation notes:
- Push diagnostics are stored per-URI in :attr:`_push_diagnostics` from
``textDocument/publishDiagnostics`` notifications. Pull diagnostics
go in :attr:`_pull_diagnostics`. The merged view dedupes by content.
- All per-document state lives in one :class:`_DocState` keyed by
absolute path. Freshness is tracked with **document versions**,
not timestamps: every didChange bumps ``version``, and each stored
push/pull result is tagged with the version it describes. A
result is fresh iff its tag >= the version being waited on, so a
didChange implicitly invalidates everything older — no clearing,
no clock comparisons, no race windows. This is what prevents
"ghost diagnostics": a slow server's leftovers from the previous
edit can never masquerade as a verdict on the current content.
- Whole-document sync. Even when the server advertises incremental
sync, we send a single ``contentChanges`` entry replacing the
@@ -45,10 +51,13 @@ import asyncio
import logging
import os
import sys
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Awaitable, Callable, Dict, List, Optional, Set
from urllib.parse import quote, unquote
from hermes_cli._subprocess_compat import windows_hide_flags
from agent.lsp.protocol import (
ERROR_CONTENT_MODIFIED,
ERROR_METHOD_NOT_FOUND,
@@ -124,6 +133,40 @@ def _end_position(text: str) -> Dict[str, int]:
return {"line": last_line, "character": last_col}
@dataclass
class _DocState:
"""Everything the client tracks for one open document.
``version`` is the LSP document version we last sent (didOpen=0,
each didChange +1). It doubles as the freshness token: stored
push/pull results are tagged with the version they describe
(``push_version`` / ``pull_version``), and a result is *fresh*
iff its tag has caught up to ``version``. Bumping the version on
didChange therefore invalidates all older results implicitly —
no store-clearing, no timestamps.
``push_version``/``pull_version`` start at -1 = "no data yet".
Servers that echo a document version in publishDiagnostics get
exact tagging; those that don't are credited with the current
version at receipt time (a push observed after we sent the
change describes the changed content or newer).
"""
version: int = 0
text: str = ""
push: List[Dict[str, Any]] = field(default_factory=list)
pull: List[Dict[str, Any]] = field(default_factory=list)
push_version: int = -1
pull_version: int = -1
seed_seen: bool = False
def fresh_push(self, version: Optional[int] = None) -> bool:
return self.push_version >= (self.version if version is None else version)
def fresh_pull(self, version: Optional[int] = None) -> bool:
return self.pull_version >= (self.version if version is None else version)
class LSPClient:
"""Async LSP client tied to one server process and one workspace root.
@@ -186,18 +229,10 @@ class LSPClient:
# is silently dropped by default.
}
# Tracked file state — required for didChange version bumps.
self._files: Dict[str, Dict[str, Any]] = {}
# Diagnostic stores, keyed by file path (NOT URI).
self._push_diagnostics: Dict[str, List[Dict[str, Any]]] = {}
self._pull_diagnostics: Dict[str, List[Dict[str, Any]]] = {}
# Per-path "last published" time so wait-for-fresh logic works.
self._published: Dict[str, float] = {}
# Per-path version of the latest push (matches our didChange
# version when the server respects it).
self._published_version: Dict[str, int] = {}
# First-push seen flag, for typescript-style seed-on-first-push.
self._first_push_seen: Set[str] = set()
# Per-document state (version, text, diagnostic stores, and
# their freshness tags), keyed by absolute file path (NOT URI).
# See _DocState for the version-based freshness model.
self._docs: Dict[str, _DocState] = {}
# Capability registrations — only diagnostic ones are tracked.
self._diagnostic_registrations: Dict[str, Dict[str, Any]] = {}
@@ -261,6 +296,12 @@ class LSPClient:
cmd = self._command
if sys.platform == "win32":
cmd = self._win_wrap_cmd(cmd)
# Suppress the cmd.exe console window that would otherwise flash
# every time we launch a ``.cmd``-wrapped language server
# (e.g. pyright-langserver.CMD) from a console-less host such as
# a VS Code/Zed extension running the ACP adapter.
# windows_hide_flags() is CREATE_NO_WINDOW on Windows, 0 on POSIX.
creationflags = windows_hide_flags()
try:
# start_new_session=True detaches the LSP server into its own
@@ -279,6 +320,7 @@ class LSPClient:
env=env,
cwd=self._cwd,
start_new_session=True,
creationflags=creationflags,
)
except FileNotFoundError as e:
raise LSPProtocolError(
@@ -647,25 +689,25 @@ class LSPClient:
if not isinstance(diagnostics, list):
diagnostics = []
version = params.get("version")
loop_time = asyncio.get_event_loop().time()
if self._seed_first_push and path not in self._first_push_seen:
# First push: seed without firing the event so a waiter
# doesn't resolve on the very first push (which arrives
# before the user-triggered didChange could've produced
# fresh diagnostics).
self._first_push_seen.add(path)
self._push_diagnostics[path] = diagnostics
self._published[path] = loop_time
if isinstance(version, int):
self._published_version[path] = version
doc = self._docs.setdefault(path, _DocState(version=-1))
if self._seed_first_push and not doc.seed_seen:
# First push: seed the store WITHOUT a freshness tag. It
# arrives before the user-triggered didChange could've
# produced fresh diagnostics, so it must never satisfy a
# waiter — it's baseline data only.
doc.seed_seen = True
doc.push = diagnostics
return
self._push_diagnostics[path] = diagnostics
self._published[path] = loop_time
if isinstance(version, int):
self._published_version[path] = version
self._first_push_seen.add(path)
doc.seed_seen = True
doc.push = diagnostics
# Tag with the echoed document version when the server provides
# one; otherwise credit the current version — a push observed
# after we sent the change describes the changed content (or
# newer). Note doc.version is -1 for never-opened paths
# (e.g. relatedDocuments spillover), keeping them unfresh.
doc.push_version = version if isinstance(version, int) else doc.version
# Bump the monotonic push counter and wake every waiter. We
# keep the Event sticky-set so any wait already in progress
# resolves; waiters re-check their predicate after waking and
@@ -694,16 +736,16 @@ class LSPClient:
raise LSPProtocolError(f"cannot read {abs_path}: {e}") from e
uri = file_uri(abs_path)
existing = self._files.get(abs_path)
doc = self._docs.get(abs_path)
if existing is not None:
if doc is not None and doc.version >= 0:
# Re-open: bump version, fire didChangeWatchedFiles + didChange.
await self._send_notification(
"workspace/didChangeWatchedFiles",
{"changes": [{"uri": uri, "type": 2}]}, # 2 = CHANGED
)
new_version = existing["version"] + 1
old_text = existing["text"]
new_version = doc.version + 1
old_text = doc.text
content_changes: List[Dict[str, Any]]
if self._sync_kind == 2:
content_changes = [
@@ -724,7 +766,11 @@ class LSPClient:
"contentChanges": content_changes,
},
)
self._files[abs_path] = {"version": new_version, "text": text}
# Bumping the version is the whole invalidation story:
# every stored result tagged with an older version is now
# stale by definition (see _DocState).
doc.version = new_version
doc.text = text
return new_version
# First open: didChangeWatchedFiles CREATED + didOpen.
@@ -732,12 +778,9 @@ class LSPClient:
"workspace/didChangeWatchedFiles",
{"changes": [{"uri": uri, "type": 1}]}, # 1 = CREATED
)
# Clear any stale push/pull entries — fresh open should start
# from scratch.
self._push_diagnostics.pop(abs_path, None)
self._pull_diagnostics.pop(abs_path, None)
self._published.pop(abs_path, None)
self._published_version.pop(abs_path, None)
# Fresh doc state — anything stashed under this path by a
# pre-open push (relatedDocuments spillover etc.) is discarded.
self._docs[abs_path] = _DocState(version=0, text=text)
await self._send_notification(
"textDocument/didOpen",
{
@@ -749,7 +792,6 @@ class LSPClient:
}
},
)
self._files[abs_path] = {"version": 0, "text": text}
return 0
async def save_file(self, path: str) -> None:
@@ -769,12 +811,19 @@ class LSPClient:
async def _pull_document_diagnostics(self, path: str) -> None:
"""Send ``textDocument/diagnostic`` for one file.
Stores results into :attr:`_pull_diagnostics`. Silently
no-ops on errors (server may not support the pull endpoint).
Stores results into the doc's pull store, tagged with the
document version captured at request send time. If a didChange
races past the in-flight request, the version bump makes the
stored result stale automatically — no explicit invalidation.
Silently no-ops on errors (server may not support the pull
endpoint).
"""
abs_path = os.path.abspath(path)
doc = self._docs.get(abs_path)
sent_version = doc.version if doc else -1
try:
params: Dict[str, Any] = {
"textDocument": {"uri": file_uri(os.path.abspath(path))}
"textDocument": {"uri": file_uri(abs_path)}
}
result = await self._send_request_with_retry(
"textDocument/diagnostic",
@@ -788,7 +837,9 @@ class LSPClient:
return
items = result.get("items")
if isinstance(items, list):
self._pull_diagnostics[os.path.abspath(path)] = items
doc = self._docs.setdefault(abs_path, _DocState(version=-1))
doc.pull = items
doc.pull_version = sent_version
related = result.get("relatedDocuments")
if isinstance(related, dict):
for uri, sub in related.items():
@@ -796,7 +847,11 @@ class LSPClient:
continue
sub_items = sub.get("items")
if isinstance(sub_items, list):
self._pull_diagnostics[uri_to_path(uri)] = sub_items
rel = self._docs.setdefault(uri_to_path(uri), _DocState(version=-1))
rel.pull = sub_items
# Same send-anchored tagging: fresh only if that
# doc hasn't changed since the request went out.
rel.pull_version = rel.version
async def wait_for_diagnostics(
self,
@@ -804,22 +859,36 @@ class LSPClient:
version: int,
*,
mode: str = "document",
) -> None:
timeout: Optional[float] = None,
) -> bool:
"""Wait for the server to publish diagnostics for ``path`` at ``version``.
``mode`` is ``"document"`` (5s budget, document pulls) or
``"full"`` (10s budget, also workspace pulls). Best-effort —
returns silently on timeout. Does NOT throw if the server
doesn't support pull diagnostics; we still get the push side.
``"full"`` (10s budget, also workspace pulls). ``timeout``
overrides the mode's default budget when provided — this is
how the user's ``lsp.wait_timeout`` config reaches the wait
loop (slow servers like tsserver on big projects need more
than the 5s default).
Returns ``True`` when *fresh* diagnostics arrived (a push at
or after our didChange, or a pull answered after it) and
``False`` on timeout. Callers must treat ``False`` as "no
data", NOT as "no errors" — the diagnostic stores may still
hold stale entries from the previous edit at that point.
Best-effort — never throws if the server doesn't support pull
diagnostics; we still get the push side.
"""
budget = DIAGNOSTICS_FULL_WAIT if mode == "full" else DIAGNOSTICS_DOCUMENT_WAIT
if timeout is not None and timeout > 0:
budget = timeout
else:
budget = DIAGNOSTICS_FULL_WAIT if mode == "full" else DIAGNOSTICS_DOCUMENT_WAIT
deadline = asyncio.get_event_loop().time() + budget
abs_path = os.path.abspath(path)
while True:
remaining = deadline - asyncio.get_event_loop().time()
if remaining <= 0:
return
return False
# Concurrent: document pull + push wait.
pull_task = asyncio.create_task(self._pull_document_diagnostics(abs_path))
@@ -838,26 +907,24 @@ class LSPClient:
pass
# If we got a fresh push for our version, we're done.
current_v = self._published_version.get(abs_path)
if abs_path in self._published and (
current_v is None or current_v >= version
):
return
doc = self._docs.get(abs_path)
if doc and doc.fresh_push(version):
return True
# Pull may have populated _pull_diagnostics — that's also
# success.
if abs_path in self._pull_diagnostics:
return
# Pull may have answered for the current version — that's
# also success.
if doc and doc.fresh_pull(version):
return True
# Loop until budget runs out.
async def _wait_for_fresh_push(self, path: str, version: int, timeout: float) -> None:
"""Wait until a publishDiagnostics arrives for ``path`` at ``version``+."""
"""Wait until a fresh publishDiagnostics arrives for ``path`` at ``version``+."""
deadline = asyncio.get_event_loop().time() + timeout
baseline = self._push_counter
while True:
current_v = self._published_version.get(path)
if path in self._published and (current_v is None or current_v >= version):
doc = self._docs.get(path)
if doc and doc.fresh_push(version):
# Debounce — wait a tick in case more diagnostics arrive
# immediately after. TS often emits in pairs. We
# snapshot the counter so we wake on a *new* push, not
@@ -888,17 +955,28 @@ class LSPClient:
except asyncio.TimeoutError:
continue
def diagnostics_for(self, path: str) -> List[Dict[str, Any]]:
def diagnostics_for(self, path: str, *, fresh_only: bool = False) -> List[Dict[str, Any]]:
"""Return current merged + deduped diagnostics for one file.
Diagnostics from push and pull stores are concatenated and
deduplicated by ``(severity, code, message, range)`` content
key. Empty list if the server hasn't published anything.
With ``fresh_only=True``, a store only contributes when its
version tag has caught up to the document's current version —
stale leftovers from the previous edit cycle are excluded.
This is what report paths should use: after an edit, "stale
errors" and "no errors" must not be conflated.
"""
abs_path = os.path.abspath(path)
push = self._push_diagnostics.get(abs_path) or []
pull = self._pull_diagnostics.get(abs_path) or []
return _dedupe(push, pull)
doc = self._docs.get(os.path.abspath(path))
if doc is None:
return []
if fresh_only:
return _dedupe(
doc.push if doc.fresh_push() else [],
doc.pull if doc.fresh_pull() else [],
)
return _dedupe(doc.push, doc.pull)
def _dedupe(*lists: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
+21 -1
View File
@@ -40,7 +40,7 @@ from __future__ import annotations
import logging
import os
import threading
from typing import Tuple
from typing import List, Tuple
# Dedicated logger name so the documented grep recipe survives a
# ``logging.getLogger(__name__)`` rename of any internal module.
@@ -188,6 +188,25 @@ def log_spawn_failed(server_id: str, workspace_root: str, exc: BaseException) ->
)
def log_reaped(keys: List[Tuple[str, str]], idle_timeout: float) -> None:
"""Idle clients were shut down by the reaper. INFO — one line per
sweep so users can correlate memory drops with LSP activity.
Also clears the ``log_active`` announce cache for the reaped keys so
a later respawn re-announces at INFO instead of logging a misleading
DEBUG "reused client".
"""
with _announce_lock:
for key in keys:
_announced_active.discard(key)
summary = ", ".join(f"{sid} ({root})" for sid, root in keys)
_emit(
"reaper",
logging.INFO,
f"reaped {len(keys)} idle client(s) after {idle_timeout:.0f}s: {summary}",
)
def reset_announce_caches() -> None:
"""Test-only: clear the dedup caches. Production code never calls this."""
with _announce_lock:
@@ -209,5 +228,6 @@ __all__ = [
"log_timeout",
"log_server_error",
"log_spawn_failed",
"log_reaped",
"reset_announce_caches",
]
+9 -7
View File
@@ -30,11 +30,12 @@ import logging
import os
import shutil
import subprocess
import sys
import threading
from pathlib import Path
from typing import Any, Dict, Optional
from hermes_cli._subprocess_compat import windows_hide_flags
logger = logging.getLogger("agent.lsp.install")
# Package-name → install-strategy hint registry. Each entry is a
@@ -122,10 +123,9 @@ def _is_windows() -> bool:
def hermes_lsp_bin_dir() -> Path:
"""Return the Hermes-owned bin staging dir for LSP servers."""
home = os.environ.get("HERMES_HOME")
if home is None:
home = os.path.join(os.path.expanduser("~"), ".hermes")
p = Path(home) / "lsp" / "bin"
from hermes_constants import get_hermes_home
p = get_hermes_home() / "lsp" / "bin"
p.mkdir(parents=True, exist_ok=True)
return p
@@ -265,9 +265,10 @@ def _install_npm(
[npm, "install", "--prefix", str(staging), "--silent", "--no-fund", "--no-audit", *install_targets],
check=False,
capture_output=True,
text=True,
text=True, encoding="utf-8", errors="replace",
timeout=300,
stdin=subprocess.DEVNULL,
creationflags=windows_hide_flags(),
)
if proc.returncode != 0:
logger.warning(
@@ -313,10 +314,11 @@ def _install_go(pkg: str, bin_name: str) -> Optional[str]:
[go, "install", pkg],
check=False,
capture_output=True,
text=True,
text=True, encoding="utf-8", errors="replace",
timeout=600,
env=env,
stdin=subprocess.DEVNULL,
creationflags=windows_hide_flags(),
)
if proc.returncode != 0:
logger.warning(
+120 -15
View File
@@ -59,6 +59,7 @@ from agent.lsp.workspace import (
logger = logging.getLogger("agent.lsp.manager")
DEFAULT_IDLE_TIMEOUT = 600 # seconds; servers idle for >10min get reaped
MIN_IDLE_TIMEOUT = 30 # floor for config values; must exceed any per-op wait budget
class _BackgroundLoop:
@@ -176,6 +177,7 @@ class LSPService:
self._spawning: Dict[Tuple[str, str], asyncio.Future] = {}
self._last_used: Dict[Tuple[str, str], float] = {}
self._state_lock = threading.Lock()
self._idle_reaper_task: Optional[asyncio.Task] = None
# Delta baseline: file path → snapshot of diagnostics taken
# immediately before a write. ``get_diagnostics_sync`` filters
@@ -183,6 +185,9 @@ class LSPService:
# introduced by the current edit.
self._delta_baseline: Dict[str, List[Dict[str, Any]]] = {}
if self._enabled and self._idle_timeout > 0:
self._loop.run(self._start_idle_reaper(), timeout=2.0)
@classmethod
def create_from_config(cls) -> Optional["LSPService"]:
"""Build a service from ``hermes_cli.config`` settings.
@@ -191,8 +196,8 @@ class LSPService:
itself returns ``is_active()`` False when LSP is disabled.
"""
try:
from hermes_cli.config import load_config
cfg = load_config()
from hermes_cli.config import load_config_readonly
cfg = load_config_readonly()
except Exception as e: # noqa: BLE001
logger.debug("LSP config load failed: %s", e)
return None
@@ -205,6 +210,16 @@ class LSPService:
wait_mode = lsp_cfg.get("wait_mode", "document")
wait_timeout = float(lsp_cfg.get("wait_timeout", DIAGNOSTICS_DOCUMENT_WAIT))
install_strategy = lsp_cfg.get("install_strategy", "auto")
try:
idle_timeout = float(lsp_cfg.get("idle_timeout", DEFAULT_IDLE_TIMEOUT))
except (TypeError, ValueError):
idle_timeout = DEFAULT_IDLE_TIMEOUT
if 0 < idle_timeout < MIN_IDLE_TIMEOUT:
# A timeout below the per-operation wait budget could reap a
# client mid-flight; the resulting outer timeout would then
# mark the (server, workspace) pair broken for the process
# lifetime. Clamp to a safe floor (0 still disables).
idle_timeout = MIN_IDLE_TIMEOUT
servers_cfg = lsp_cfg.get("servers") or {}
disabled = []
binary_overrides: Dict[str, List[str]] = {}
@@ -235,6 +250,7 @@ class LSPService:
env_overrides=env_overrides,
init_overrides=init_overrides,
disabled_servers=disabled,
idle_timeout=idle_timeout,
)
# ------------------------------------------------------------------
@@ -292,7 +308,10 @@ class LSPService:
if not self.enabled_for(file_path):
return
try:
diags = self._loop.run(self._snapshot_async(file_path), timeout=8.0)
# Outer join budget must exceed the inner wait budget or a
# slow-but-alive server gets falsely marked broken.
t = max(8.0, self._wait_timeout + 3.0)
diags = self._loop.run(self._snapshot_async(file_path), timeout=t)
self._delta_baseline[os.path.abspath(file_path)] = diags or []
except Exception as e: # noqa: BLE001
logger.debug("baseline snapshot failed for %s: %s", file_path, e)
@@ -341,7 +360,7 @@ class LSPService:
try:
t = timeout if timeout is not None else self._wait_timeout + 2.0
diags = self._loop.run(self._open_and_wait_async(file_path), timeout=t) or []
diags = self._loop.run(self._open_and_wait_async(file_path), timeout=t)
except asyncio.TimeoutError as e:
eventlog.log_timeout(server_id, file_path)
logger.debug("LSP diagnostics timeout for %s: %s", file_path, e)
@@ -353,6 +372,17 @@ class LSPService:
self._mark_broken_for_file(file_path, e)
return []
if diags is None:
# The server is alive but never produced diagnostics for the
# post-edit content within the wait budget (common for
# tsserver on large projects). Report "no data" rather than
# whatever stale state is in the stores — surfacing the
# previous edit's errors as if they were current is the
# ghost-diagnostics bug. The server is NOT marked broken:
# slow is not dead, and the next edit may well succeed.
eventlog.log_timeout(server_id, file_path, kind="fresh diagnostics")
return []
abs_path = os.path.abspath(file_path)
if delta:
baseline = self._delta_baseline.get(abs_path) or []
@@ -420,6 +450,7 @@ class LSPService:
# ``_clients`` with a half-initialized state.
with self._state_lock:
client = self._clients.pop(key, None)
self._last_used.pop(key, None)
if client is not None:
try:
# Fire-and-forget shutdown — give it a second to cleanup,
@@ -452,26 +483,43 @@ class LSPService:
return []
try:
version = await client.open_file(file_path, language_id=language_id_for(file_path))
await client.wait_for_diagnostics(file_path, version, mode=self._wait_mode)
fresh = await client.wait_for_diagnostics(file_path, version, mode=self._wait_mode)
except Exception as e: # noqa: BLE001
logger.debug("snapshot open/wait failed: %s", e)
return []
self._last_used[(client.server_id, client.workspace_root)] = time.time()
return list(client.diagnostics_for(file_path))
self._touch(client)
if not fresh:
# No fresh data for the pre-edit content — an empty baseline
# is safe: worst case the delta filter removes less, never
# more. Never seed the baseline from stale stores.
return []
return list(client.diagnostics_for(file_path, fresh_only=True))
async def _open_and_wait_async(self, file_path: str) -> List[Dict[str, Any]]:
async def _open_and_wait_async(self, file_path: str) -> Optional[List[Dict[str, Any]]]:
"""Open + wait for FRESH diagnostics.
Returns the fresh diagnostic list, or ``None`` when the server
never produced post-change data within the wait budget. The
distinction matters: ``[]`` means "server checked the new
content, it's clean", ``None`` means "no verdict" — the caller
must not substitute stale data for either.
"""
client = await self._get_or_spawn(file_path)
if client is None:
return []
return None
try:
version = await client.open_file(file_path, language_id=language_id_for(file_path))
await client.save_file(file_path)
await client.wait_for_diagnostics(file_path, version, mode=self._wait_mode)
fresh = await client.wait_for_diagnostics(
file_path, version, mode=self._wait_mode, timeout=self._wait_timeout
)
except Exception as e: # noqa: BLE001
logger.debug("open/wait failed for %s: %s", file_path, e)
return []
self._last_used[(client.server_id, client.workspace_root)] = time.time()
return list(client.diagnostics_for(file_path))
return None
self._touch(client)
if not fresh:
return None
return list(client.diagnostics_for(file_path, fresh_only=True))
async def _current_diags_async(self, file_path: str) -> List[Dict[str, Any]]:
ws, gated = resolve_workspace_for_file(file_path)
@@ -482,7 +530,7 @@ class LSPService:
client = self._clients.get((srv.server_id, ws))
if client is None:
return []
return list(client.diagnostics_for(file_path))
return list(client.diagnostics_for(file_path, fresh_only=True))
async def _get_or_spawn(self, file_path: str) -> Optional[LSPClient]:
srv = find_server_for_file(file_path)
@@ -508,6 +556,7 @@ class LSPService:
with self._state_lock:
client = self._clients.get(key)
if client is not None and client.is_running:
self._last_used[key] = time.time()
eventlog.log_active(srv.server_id, per_server_root)
return client
spawning = self._spawning.get(key)
@@ -558,7 +607,7 @@ class LSPService:
return None
with self._state_lock:
self._clients[key] = client
self._last_used[key] = time.time()
self._last_used[key] = time.time()
eventlog.log_active(srv.server_id, per_server_root)
spawn_future.set_result(client)
return client
@@ -566,7 +615,63 @@ class LSPService:
with self._state_lock:
self._spawning.pop(key, None)
async def _start_idle_reaper(self) -> None:
self._idle_reaper_task = asyncio.create_task(self._idle_reaper_loop())
def _touch(self, client: LSPClient) -> None:
"""Refresh the last-used timestamp for a client we just used.
Guarded on membership so a reaped-mid-operation client can't
resurrect an orphan ``_last_used`` entry after the reaper popped
the key. All writers and the reaper run on the background loop
thread; the lock keeps this consistent with the reader anyway.
"""
key = (client.server_id, client.workspace_root)
with self._state_lock:
if key in self._clients:
self._last_used[key] = time.time()
async def _idle_reaper_loop(self) -> None:
interval = min(60.0, self._idle_timeout)
while True:
await asyncio.sleep(interval)
try:
await self._reap_idle_once()
except asyncio.CancelledError:
raise
except Exception as e: # noqa: BLE001
# A transient sweep error must not kill the reaper —
# otherwise one bad shutdown permanently re-opens the
# unbounded-accumulation leak this loop exists to fix.
logger.debug("LSP idle reaper sweep error: %s", e)
async def _reap_idle_once(self) -> None:
cutoff = time.time() - self._idle_timeout
with self._state_lock:
idle_keys = [
key
for key in self._clients
if self._last_used.get(key, 0) < cutoff
]
clients = [self._clients.pop(key) for key in idle_keys]
for key in idle_keys:
self._last_used.pop(key, None)
if clients:
eventlog.log_reaped(
[(c.server_id, c.workspace_root) for c in clients],
self._idle_timeout,
)
await asyncio.gather(
*(client.shutdown() for client in clients),
return_exceptions=True,
)
async def _shutdown_async(self) -> None:
reaper = self._idle_reaper_task
self._idle_reaper_task = None
if reaper is not None:
reaper.cancel()
await asyncio.gather(reaper, return_exceptions=True)
with self._state_lock:
clients = list(self._clients.values())
self._clients.clear()
+6 -6
View File
@@ -710,9 +710,9 @@ def _find_pses_bundle(ctx: ServerContext) -> Optional[str]:
env_path = os.environ.get("PSES_BUNDLE_PATH")
if env_path:
candidates.append(env_path)
home = os.environ.get("HERMES_HOME") or os.path.join(
os.path.expanduser("~"), ".hermes"
)
from hermes_constants import get_hermes_home
home = str(get_hermes_home())
candidates.append(os.path.join(home, "lsp", "PowerShellEditorServices"))
for cand in candidates:
@@ -796,9 +796,9 @@ def _spawn_powershell_es(root: str, ctx: ServerContext) -> Optional[SpawnSpec]:
def hermes_lsp_session_dir() -> str:
"""Return (and create) the dir for PSES session/log scratch files."""
home = os.environ.get("HERMES_HOME") or os.path.join(
os.path.expanduser("~"), ".hermes"
)
from hermes_constants import get_hermes_home
home = str(get_hermes_home())
d = os.path.join(home, "lsp", "pses")
os.makedirs(d, exist_ok=True)
return d
+30
View File
@@ -7,6 +7,36 @@ from typing import Any, Sequence
from agent.redact import redact_sensitive_text
def describe_compression_lock_skip(lock_signal: Any) -> str:
"""User-facing text for a manual /compress skipped by the compression lock.
``lock_signal`` is ``agent._compression_skipped_due_to_lock`` (or the
``holder`` carried by the TUI's ``CompressionLockHeld``): a descriptive
holder string when another compressor CONFIRMED holds the lock, or
``True``/``None`` when acquisition failed without a confirmed holder
(``hermes_state.try_acquire_compression_lock`` catches ``sqlite3.Error``
internally and returns ``False``, so a failed acquire is NOT proof that
another compression is running). The two cases must be worded
differently: claiming "already in progress" on an unconfirmed failure
misdirects the user when the real problem is a broken lock subsystem.
"""
holder = (
lock_signal
if isinstance(lock_signal, str) and lock_signal.strip()
else None
)
if holder:
return (
f"⏳ Compression already in progress for this session "
f"(holder: {holder}). Please wait for it to finish."
)
return (
"⏳ Compression skipped: could not acquire this session's "
"compression lock. Another compression may still be running, or "
"the lock check failed — try again shortly."
)
def summarize_manual_compression(
before_messages: Sequence[dict[str, Any]],
after_messages: Sequence[dict[str, Any]],
+14 -4
View File
@@ -80,8 +80,17 @@ def normalize_tool_schema(schema: Any) -> Optional[Dict[str, Any]]:
return schema
def memory_provider_tools_enabled(enabled_toolsets: Optional[List[str]]) -> bool:
def memory_provider_tools_enabled(
enabled_toolsets: Optional[List[str]],
disabled_toolsets: Optional[List[str]] = None,
*,
memory_tool_present: bool = False,
) -> bool:
"""Return whether external memory-provider tools should be exposed."""
if disabled_toolsets and "memory" in disabled_toolsets:
return False
if memory_tool_present:
return True
if enabled_toolsets is None:
return True
if not enabled_toolsets:
@@ -110,9 +119,10 @@ def inject_memory_provider_tools(agent: Any) -> int:
for tool in tools
if isinstance(tool, dict)
}
if (
"memory" not in existing_tool_names
and not memory_provider_tools_enabled(getattr(agent, "enabled_toolsets", None))
if not memory_provider_tools_enabled(
getattr(agent, "enabled_toolsets", None),
getattr(agent, "disabled_toolsets", None),
memory_tool_present="memory" in existing_tool_names,
):
return 0
+375
View File
@@ -14,6 +14,7 @@ re-exports from ``run_agent`` remain in place so existing imports
from __future__ import annotations
import hashlib
import json
import logging
import re
@@ -474,4 +475,378 @@ __all__ = [
"_sanitize_tools_non_ascii",
"_strip_images_from_messages",
"_sanitize_structure_non_ascii",
# call_id policy owners (F4 consolidation)
"deterministic_call_id",
"coalesce_tool_call_id",
"uniquify_tool_call_ids",
# reasoning_content policy owners (F4 consolidation)
"reasoning_echo_family",
"matches_reasoning_echo_family",
"needs_reasoning_echo",
"apply_reasoning_content_policy",
"reapply_reasoning_echo",
]
# ---------------------------------------------------------------------------
# call_id policy — single owner (audit F4, incident chain I4)
# ---------------------------------------------------------------------------
#
# Three forked policy sites converged here:
# * agent/codex_responses_adapter.py `_deterministic_call_id` — hash
# synthesis when a provider omits call_id (fa3ab2ffd0 → e45f2b39e2).
# * run_agent.AIAgent._get_tool_call_id_static — `call_id or id`
# coalescing for dicts and SDK objects.
# * run_agent.AIAgent._uniquify_tool_call_ids — duplicate-id repair with
# deterministic `_d<n>` suffixes (#58327 loss class).
#
# NOT consolidated (different scheme on purpose):
# agent/transports/codex_event_projector._deterministic_call_id maps codex
# app-server ITEM ids (`codex_<type>_<item_id>`), not chat tool-call
# content; merging the two would change ids and invalidate prompt caches.
#
# HARD INVARIANT: everything here must stay deterministic (never uuid4) and
# byte-identical for existing inputs — these ids feed prompt-cache prefixes.
def deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
"""Generate a deterministic call_id from tool call content.
Used as a fallback when the API doesn't provide a call_id.
Deterministic IDs prevent cache invalidation — random UUIDs would
make every API call's prefix unique, breaking OpenAI's prompt cache.
"""
seed = f"{fn_name}:{arguments}:{index}"
digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12]
return f"call_{digest}"
def coalesce_tool_call_id(tc: Any) -> str:
"""Extract the effective call ID from a tool_call entry (dict or object).
Single owner for the ``call_id or id`` coalescing rule: Codex Responses
tool calls carry ``call_id`` (authoritative pairing key), Chat
Completions ones carry ``id`` only. Returns ``""`` when neither is set.
"""
if isinstance(tc, dict):
return (tc.get("call_id", "") or tc.get("id", "") or "").strip()
return (getattr(tc, "call_id", "") or getattr(tc, "id", "") or "").strip()
def uniquify_tool_call_ids(tool_calls: list) -> list:
"""Ensure every tool call in a single assistant turn has a distinct id.
Some models/providers reuse one call id across different calls in a
single batch (observed with native Kimi Responses replays, Ollama-
compatible endpoints, and degraded models at long context; same bug
class as openclaw/openclaw#110518 / #110956). Duplicate ids are lossy
downstream: the pre-API sanitizer keeps only the first call/result
pair per id (#58327), so the later call's result silently vanishes
from every replayed payload, and strict providers (Anthropic
tool_use, DeepSeek) reject duplicate ids outright.
The first occurrence keeps its id; later collisions get a
deterministic ``<id>_d<n>`` suffix — never a random UUID, which would
break prompt-cache prefix stability across replays. Mutates the
entries in place (SDK models / SimpleNamespace / dicts) and returns
the same list. Blank/missing ids are left for the deterministic
fallback in ``build_assistant_message``.
"""
seen: set = set()
for tc in tool_calls or []:
# Same coalescing rule as ``coalesce_tool_call_id`` but tolerant of
# non-string ids (degraded models can emit ints/None here).
if isinstance(tc, dict):
raw = tc.get("call_id") or tc.get("id") or ""
else:
raw = getattr(tc, "call_id", None) or getattr(tc, "id", None) or ""
raw = raw.strip() if isinstance(raw, str) else ""
if not raw:
continue
# Composite Responses ids ("call_x|fc_y") collide on the call
# half — that's the pairing key providers enforce per turn.
cid = raw.split("|", 1)[0]
if not cid:
continue
if cid not in seen:
seen.add(cid)
continue
n = 2
new_id = f"{cid}_d{n}"
while new_id in seen:
n += 1
new_id = f"{cid}_d{n}"
seen.add(new_id)
def _renamed(value):
# Preserve a composite id's response-item half so the
# provider's real fc_/item id survives the rename.
if isinstance(value, str) and "|" in value:
return f"{new_id}|{value.split('|', 1)[1]}"
return new_id
try:
if isinstance(tc, dict):
if tc.get("id"):
tc["id"] = _renamed(tc["id"])
else:
tc["id"] = new_id
if tc.get("call_id"):
tc["call_id"] = new_id
else:
tc.id = _renamed(getattr(tc, "id", None))
if getattr(tc, "call_id", None):
tc.call_id = new_id
except Exception:
logger.warning(
"Could not uniquify duplicate tool call id %s", cid
)
continue
_fn = tc.get("function") if isinstance(tc, dict) else getattr(tc, "function", None)
_fn_name = (_fn.get("name") if isinstance(_fn, dict) else getattr(_fn, "name", None)) or "?"
logger.warning(
"Model reused tool call id %s within one turn; renamed the "
"duplicate to %s (tool=%s) to keep call/result pairing "
"lossless.", cid, new_id, _fn_name,
)
return tool_calls
# ---------------------------------------------------------------------------
# reasoning_content policy — single owner (audit F4)
# ---------------------------------------------------------------------------
#
# The strip-vs-repad decision was previously forked across the wire files in
# separate incident commits (2b3a4f0af8 strip for strict providers,
# b5495db701 re-pad for require-side, 94b3131be7/9a9f8a6d99 kimi pad). The
# POLICY — which provider direction gets which treatment — lives here as one
# rule table + apply functions; adapters keep only SYNTAX mapping (e.g.
# anthropic_adapter turning reasoning_content into a thinking block).
#
# Direction table:
# require-side (echo-back enforced; replays 400 without the field):
# kimi — provider kimi-coding/kimi-coding-cn, or host api.kimi.com /
# moonshot.ai / moonshot.cn. Host-driven on purpose:
# aggregators re-exporting kimi models reject the echo.
# deepseek — provider "deepseek", model contains "deepseek", or host
# api.deepseek.com (#15250; V4 rejects empty-string pads,
# hence the " " single-space pad, #17341).
# mimo — provider "xiaomi", model contains "mimo", or host
# *.xiaomimimo.com.
# strict side (field rejected with 400/422 "Extra inputs are not
# permitted"): everyone else — Mistral, Cerebras, Groq, SambaNova, …
# (#45655). Strip the key entirely, even a single-space pad.
_REASONING_ECHO_RULES: tuple = (
# (family, exact providers (raw), exact providers (lowered),
# model substrings (lowered), base_url hosts)
("kimi", frozenset({"kimi-coding", "kimi-coding-cn"}), frozenset(), (),
("api.kimi.com", "moonshot.ai", "moonshot.cn")),
("deepseek", frozenset(), frozenset({"deepseek"}), ("deepseek",),
("api.deepseek.com",)),
("mimo", frozenset(), frozenset({"xiaomi"}), ("mimo",),
("api.xiaomimimo.com", "xiaomimimo.com")),
)
def _family_rule(family: str) -> tuple:
for rule in _REASONING_ECHO_RULES:
if rule[0] == family:
return rule
raise KeyError(family)
def matches_reasoning_echo_family(
family: str, provider: Any, model: Any, base_url: Any
) -> bool:
"""True when (provider, model, base_url) matches one echo-back family.
Families can overlap (e.g. a deepseek-named model pointed at a kimi
host); this membership test is independent per family so per-family
predicates keep their original semantics.
"""
from utils import base_url_host_matches
_, raw_providers, lowered_providers, model_subs, hosts = _family_rule(family)
provider_lower = (provider or "").lower()
model_lower = (model or "").lower()
if provider in raw_providers or provider_lower in lowered_providers:
return True
if any(sub in model_lower for sub in model_subs):
return True
return any(base_url_host_matches(base_url, host) for host in hosts)
def reasoning_echo_family(provider: Any, model: Any, base_url: Any) -> "str | None":
"""Classify the provider direction for the reasoning_content echo policy.
Returns ``"kimi"``, ``"deepseek"``, or ``"mimo"`` (first match in table
order) when the target endpoint enforces reasoning_content echo-back on
assistant turns, else ``None`` (strict/indifferent side — the field must
be stripped).
"""
for rule in _REASONING_ECHO_RULES:
if matches_reasoning_echo_family(rule[0], provider, model, base_url):
return rule[0]
return None
def needs_reasoning_echo(provider: Any, model: Any, base_url: Any) -> bool:
"""True when the endpoint requires reasoning_content echo-back."""
return reasoning_echo_family(provider, model, base_url) is not None
def apply_reasoning_content_policy(
source_msg: dict, api_msg: dict, needs_thinking_pad: bool
) -> None:
"""Copy provider-facing reasoning fields onto an API replay message.
``needs_thinking_pad`` is the require-side flag (see
``needs_reasoning_echo`` / the agent's cached
``_needs_thinking_reasoning_pad``). Mutates ``api_msg`` in place.
"""
if source_msg.get("role") != "assistant":
return
# 1. Explicit reasoning_content already set.
#
# When the active provider enforces the thinking-mode echo-back
# (DeepSeek / Kimi / MiMo), preserve it verbatim — that includes their
# own space-placeholder written at creation time and any valid reasoning
# from the same provider. Sessions persisted BEFORE #17341 have
# empty-string placeholders pinned at creation time; DeepSeek V4 Pro
# rejects those with HTTP 400, so upgrade "" → " " on replay.
#
# When the active provider does NOT enforce echo-back, strip the field
# entirely. Strict OpenAI-compatible providers (Mistral, Cerebras, Groq,
# SambaNova, …) reject ANY reasoning_content key in input messages with
# HTTP 400/422 ("Extra inputs are not permitted"), even an empty string
# or a single-space pad. This is the cross-provider fallback case: a
# reasoning primary (DeepSeek/Kimi/MiMo) pads history with " ", then a
# fallback to a strict provider replays that pad and 422s. Stripping
# here covers the rebuild path; ``reapply_reasoning_echo`` covers the
# already-built api_messages path. Refs #45655.
existing = source_msg.get("reasoning_content")
if isinstance(existing, str):
if not needs_thinking_pad:
api_msg.pop("reasoning_content", None)
elif existing == "":
api_msg["reasoning_content"] = " "
else:
api_msg["reasoning_content"] = existing
return
# 2. Cross-provider poisoned history (#15748): on DeepSeek/Kimi,
# if the source turn has tool_calls AND a 'reasoning' field but no
# 'reasoning_content' key, the 'reasoning' text was written by a
# prior provider (e.g. MiniMax) — DeepSeek's own _build_assistant_message
# pins reasoning_content at creation time for tool-call turns, so the
# shape (reasoning set, reasoning_content absent, tool_calls present)
# is unreachable from same-provider DeepSeek history after this fix.
# Inject a single space to satisfy the API without leaking another
# provider's chain of thought to DeepSeek/Kimi. Space (not "")
# because DeepSeek V4 Pro rejects empty-string reasoning_content
# in thinking mode (refs #17341).
normalized_reasoning = source_msg.get("reasoning")
if (
needs_thinking_pad
and source_msg.get("tool_calls")
and isinstance(normalized_reasoning, str)
and normalized_reasoning
):
api_msg["reasoning_content"] = " "
return
# 3. Healthy session: promote 'reasoning' field to 'reasoning_content'
# for providers that use the internal 'reasoning' key.
# This must happen before the unconditional empty-string fallback so
# genuine reasoning content is not overwritten (#15812 regression in
# PR #15478). Only promote for providers that enforce echo-back —
# strict providers reject the field (refs #45655).
if isinstance(normalized_reasoning, str) and normalized_reasoning:
if needs_thinking_pad:
api_msg["reasoning_content"] = normalized_reasoning
else:
api_msg.pop("reasoning_content", None)
return
# 4. DeepSeek / Kimi thinking mode: all assistant messages need
# reasoning_content. Inject a single space to satisfy the provider's
# requirement when no explicit reasoning content is present. Covers
# both tool-call turns (already-poisoned history with no reasoning
# at all) and plain text turns. Space (not "") because DeepSeek V4
# Pro tightened validation and rejects empty string with HTTP 400
# ("The reasoning content in the thinking mode must be passed back
# to the API"). Refs #17341.
if needs_thinking_pad:
api_msg["reasoning_content"] = " "
return
# 5. reasoning_content was present but not a string (e.g. None after
# context compaction). Don't pass null to the API.
api_msg.pop("reasoning_content", None)
def reapply_reasoning_echo(api_messages: list, needs_thinking_pad: bool) -> int:
"""Re-pad (or strip) assistant turns' reasoning_content for the active provider.
``api_messages`` is built once, before the retry loop, while the *primary*
provider is active. A mid-conversation fallback can then switch providers,
so the reasoning fields baked into ``api_messages`` are shaped for the
*prior* provider and must be reconciled against the *current* one:
* Switching TO a require-side provider (DeepSeek / Kimi / MiMo thinking
mode): assistant turns built when the prior provider did NOT need the
echo-back go out without ``reasoning_content`` and the new provider
rejects them with HTTP 400 ("The reasoning_content in the thinking mode
must be passed back"). Re-apply the pad.
* Switching TO a strict provider that rejects the field (Mistral,
Cerebras, Groq, SambaNova, …): assistant turns built under a reasoning
primary carry a ``reasoning_content`` pad (often a single space ``" "``),
and the strict provider rejects it with HTTP 400/422 ("Extra inputs are
not permitted"). Strip the field. This is the exact cross-provider
fallback bug from #45655 — a DeepSeek primary pads history with ``" "``,
the request falls back to Mistral, and Mistral 422s on the stale pad.
Calling this immediately before building the request kwargs reconciles the
fields against the *current* provider. It is idempotent and safe to call
every iteration; it covers every fallback path.
Returns the number of assistant turns whose reasoning_content was added or
removed.
"""
changed = 0
for api_msg in api_messages:
if api_msg.get("role") != "assistant":
continue
if needs_thinking_pad:
if api_msg.get("reasoning_content"):
continue
apply_reasoning_content_policy(api_msg, api_msg, needs_thinking_pad)
if api_msg.get("reasoning_content"):
changed += 1
else:
# Strict provider — strip any stale reasoning_content pad left
# over from a reasoning primary so the fallback request doesn't
# 400/422 on it.
if "reasoning_content" in api_msg:
api_msg.pop("reasoning_content", None)
changed += 1
return changed
# ---------------------------------------------------------------------------
# Image / multimodal parts — evaluated, NOT consolidated (verdict: syntax)
# ---------------------------------------------------------------------------
#
# The per-adapter image handling is format-specific SYNTAX, not shared policy:
# * anthropic_adapter (~1817): data-URL → Anthropic `source: {type: base64}`
# block mapping — Anthropic wire shape only.
# * codex_responses_adapter (~113/165/812): chat `image_url` parts →
# Responses `input_image` items and image counting for log summaries —
# Responses wire shape only.
# * transports/chat_completions: pass-through (native format).
# The one genuinely shared image POLICY — removing images when a server
# rejects them while preserving tool_call_id pairing — already has a single
# owner here: ``_strip_images_from_messages`` above.
+1231 -230
View File
File diff suppressed because it is too large Load Diff
+500 -86
View File
@@ -4,6 +4,8 @@ Pure utility functions with no AIAgent dependency. Used by ContextCompressor
and run_agent.py for pre-flight context checks.
"""
import base64
import hashlib
import ipaddress
import json
import logging
@@ -11,18 +13,40 @@ import os
import re
import time
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
from typing import Any, Dict, List, Optional, Tuple, TYPE_CHECKING
from urllib.parse import urlparse
import requests
import yaml
if TYPE_CHECKING: # pragma: no cover — runtime import is lazy (see below)
import requests
from utils import atomic_json_write, base_url_host_matches, base_url_hostname
from hermes_constants import OPENROUTER_MODELS_URL
logger = logging.getLogger(__name__)
# ``requests`` (with urllib3) costs ~27 ms of the `import cli` waterfall and
# is only used inside the fetch functions below. It's resolved lazily:
# ``_ensure_requests()`` populates the module global on the runtime path, and
# the PEP 562 ``__getattr__`` covers external attribute access — notably
# ``patch("agent.model_metadata.requests.get")`` in tests, which resolves the
# attribute at patch time.
def _ensure_requests():
if "requests" not in globals():
import requests as _requests
globals()["requests"] = _requests
return globals()["requests"]
def __getattr__(name: str):
if name == "requests":
return _ensure_requests()
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
def _resolve_requests_verify() -> bool | str:
"""Resolve SSL verify setting for `requests` calls from env vars.
@@ -48,7 +72,7 @@ def _resolve_requests_verify() -> bool | str:
_PROVIDER_PREFIXES: frozenset[str] = frozenset({
"openrouter", "nous", "openai-codex", "copilot", "copilot-acp",
"gemini", "ollama-cloud", "zai", "kimi-coding", "kimi-coding-cn", "stepfun", "minimax", "minimax-oauth", "minimax-cn", "anthropic", "deepseek", "deepinfra",
"opencode-zen", "opencode-go", "kilocode", "alibaba", "novita",
"opencode-zen", "opencode-go", "ai-gateway", "kilocode", "alibaba", "novita",
"qwen-oauth",
"xiaomi",
"arcee",
@@ -60,7 +84,7 @@ _PROVIDER_PREFIXES: frozenset[str] = frozenset({
"glm", "z-ai", "z.ai", "zhipu", "github", "github-copilot",
"github-models", "kimi", "moonshot", "kimi-cn", "moonshot-cn", "claude", "deep-seek", "deep-infra",
"ollama",
"stepfun", "opencode", "zen", "go", "kilo", "dashscope", "aliyun", "qwen",
"stepfun", "opencode", "zen", "go", "vercel", "kilo", "dashscope", "aliyun", "qwen",
"mimo", "xiaomi-mimo",
"tencent", "tokenhub", "tencent-cloud", "tencentmaas",
"arcee-ai", "arceeai",
@@ -121,6 +145,67 @@ _ENDPOINT_MODEL_CACHE_TTL = 300
_ENDPOINT_PROBE_TTL_SECONDS = 3600.0
_endpoint_probe_path_cache: Dict[str, tuple] = {}
# ── Disk L2 for local-endpoint probe results ────────────────────────────────
# The in-process caches above die with the process, so every CLI cold start
# with a local model re-paid the probe waterfall in AIAgent.__init__:
# detect_local_server_type (up to 4 HTTP GETs, ≤2 s each on a hung server)
# + /api/show (≤3 s). A short-TTL disk cache makes back-to-back CLI
# invocations hit disk instead of the network. Only SUCCESSFUL probes are
# persisted (a down server must not pin a negative verdict), and the TTL is
# short enough that swapping the server on a port (stop Ollama, start
# LM Studio) is picked up within minutes — strictly fresher than the 1 h
# in-process TTL that already accepts that staleness.
_LOCAL_PROBE_DISK_TTL_SECONDS = 300.0
def _local_probe_disk_cache_path() -> Path:
from hermes_constants import get_hermes_home
return get_hermes_home() / "cache" / "local_endpoint_probes.json"
def _load_local_probe_disk_cache() -> Dict[str, Any]:
try:
with _local_probe_disk_cache_path().open("r", encoding="utf-8") as f:
data = json.load(f)
return data if isinstance(data, dict) else {}
except Exception:
return {}
def _local_probe_disk_get(kind: str, key: str) -> Optional[Any]:
"""Return a fresh cached value for ``kind:key``, else None."""
entry = _load_local_probe_disk_cache().get(f"{kind}:{key}")
if not isinstance(entry, dict):
return None
try:
if (time.time() - float(entry["ts"])) >= _LOCAL_PROBE_DISK_TTL_SECONDS:
return None
return entry["value"]
except Exception:
return None
def _local_probe_disk_put(kind: str, key: str, value: Any) -> None:
"""Persist a successful probe result. Best-effort; prunes stale entries."""
try:
now = time.time()
data = _load_local_probe_disk_cache()
data = {
k: v
for k, v in data.items()
if isinstance(v, dict)
and (now - float(v.get("ts", 0))) < _LOCAL_PROBE_DISK_TTL_SECONDS
}
data[f"{kind}:{key}"] = {"value": value, "ts": now}
atomic_json_write(
_local_probe_disk_cache_path(),
data,
indent=0,
separators=(",", ":"),
)
except Exception as e:
logger.debug("Failed to save local probe disk cache: %s", e)
def _get_model_metadata_cache_path() -> Path:
"""Return path to the OpenRouter model metadata disk cache."""
@@ -213,6 +298,8 @@ DEFAULT_CONTEXT_LENGTHS = {
# OpenRouter-prefixed models resolve via OpenRouter live API or models.dev.
"claude-fable-5": 1000000,
"claude-fable": 1000000,
"claude-opus-5": 1000000,
"claude-sonnet-5": 1000000,
"claude-opus-4-8": 1000000,
"claude-opus-4.8": 1000000,
"claude-opus-4-7": 1000000,
@@ -275,8 +362,10 @@ DEFAULT_CONTEXT_LENGTHS = {
# Qwen — specific model families before the catch-all.
# Official docs: https://help.aliyun.com/zh/model-studio/developer-reference/
"qwen3.6-plus": 1048576, # 1M context (DashScope/Alibaba & OpenRouter)
"qwen3.7-plus": 1048576, # 1M context (DashScope/Alibaba)
"qwen3-coder-plus": 1000000, # 1M context
"qwen3-coder": 262144, # 256K context
"qwen3-max": 262144, # 256K context (qwen3-max-2026-01-23 snapshot, Coding Plan)
"qwen": 131072,
# MiniMax — M3 is 1M context (max output 512K); M2.x series is 204,800.
# Keys use substring matching (longest-first), so "minimax-m3" wins over
@@ -316,7 +405,12 @@ DEFAULT_CONTEXT_LENGTHS = {
"grok-3": 131072, # grok-3, grok-3-mini, grok-3-fast, grok-3-mini-fast
"grok-2": 131072, # grok-2, grok-2-1212, grok-2-latest
"grok": 131072, # catch-all (grok-beta, unknown grok-*)
# Kimi
# Kimi — K3 ships with a 1 Mi context window (1,048,576; verified against
# models.dev and OpenRouter live metadata, matching the endpoint-scoped
# override in _endpoint_scoped_context_length). Longest-key-first substring
# matching ensures "kimi-k3" resolves to 1M while older/unknown Kimi models
# still hit the generic 256K fallback.
"kimi-k3": 1_048_576,
"kimi": 262144,
# Upstage Solar — api.upstage.ai/v1/models does not return context_length,
# so these fallbacks keep token budgeting / compression from probing down
@@ -540,7 +634,13 @@ def _is_known_provider_base_url(base_url: str) -> bool:
def _endpoint_scoped_context_length(model: str, base_url: str) -> Optional[int]:
"""Return metadata confirmed only for one provider endpoint."""
"""Return metadata confirmed only for the Kimi Coding endpoint.
Kimi Coding serves K3 under the bare slug ``k3``, but users may also
configure or select the public-facing aliases ``kimi-k3`` and
``kimi-k3-cot``. Only canonical ``https://api.kimi.com/coding`` endpoints
(legacy Moonshot keys do not serve K3) get the 1 Mi context window.
"""
normalized = _normalize_base_url(base_url)
try:
parsed = urlparse(normalized)
@@ -556,7 +656,7 @@ def _endpoint_scoped_context_length(model: str, base_url: str) -> Optional[int]:
and parsed.path.rstrip("/") in {"/coding", "/coding/v1"}
and not parsed.query
and not parsed.fragment
and model.strip().lower() == "k3"
and model.strip().lower() in {"k3", "kimi-k3", "kimi-k3-cot"}
):
return 1_048_576
return None
@@ -567,8 +667,13 @@ def _skip_persistent_context_cache(base_url: str, provider: str) -> bool:
LM Studio excludes caching because loaded context is transient — the user
can reload the model with a different context_length at any time.
"""
return provider == "lmstudio"
Codex OAuth excludes caching because its context window is account- and
entitlement-specific metadata supplied by the authenticated /models
endpoint. A fallback value written after a transient probe failure must
not prevent a later live probe from observing an updated allocation.
"""
return (provider or "").strip().lower() in {"lmstudio", "openai-codex"}
def _maybe_cache_local_context_length(
@@ -730,6 +835,13 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
if cached is not None and (time.monotonic() - cached[1]) < _ENDPOINT_PROBE_TTL_SECONDS:
return cached[0]
# Disk L2: a fresh cross-process verdict skips the HTTP waterfall
# entirely (back-to-back CLI invocations, cron ticks).
disk_hit = _local_probe_disk_get("server_type", server_url)
if isinstance(disk_hit, str):
_endpoint_probe_path_cache[server_url] = (disk_hit, time.monotonic())
return disk_hit
headers = _auth_headers(api_key)
result: Optional[str] = None
@@ -782,6 +894,7 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
if result is not None:
_endpoint_probe_path_cache[server_url] = (result, time.monotonic())
_local_probe_disk_put("server_type", server_url, result)
return result
@@ -904,6 +1017,7 @@ def fetch_model_metadata(force_refresh: bool = False) -> Dict[str, Dict[str, Any
return _model_metadata_cache
try:
_ensure_requests()
# Tuple (connect, read) — flat timeout=10 means urllib3 can block 10s per
# retry stage through proxies that 403 CONNECT, ballooning to minutes
# (#46620). 5s connect / 10s read fails fast on unreachable hosts.
@@ -960,6 +1074,7 @@ def fetch_endpoint_model_metadata(
normalized = _normalize_base_url(base_url)
if not normalized or _is_openrouter_base_url(normalized):
return {}
_ensure_requests()
if not force_refresh:
cached = _endpoint_model_metadata_cache.get(normalized)
@@ -1501,6 +1616,13 @@ def query_ollama_num_ctx(model: str, base_url: str, api_key: str = "") -> Option
if server_type != "ollama":
return None
# Disk L2: /api/show results are stable for a given (model, server) on
# human timescales — skip the HTTP roundtrip on fresh cross-process hits.
_disk_key = f"{server_url}|{bare_model}"
disk_hit = _local_probe_disk_get("ollama_num_ctx", _disk_key)
if isinstance(disk_hit, int) and disk_hit > 0:
return disk_hit
headers = _auth_headers(api_key)
try:
@@ -1518,7 +1640,9 @@ def query_ollama_num_ctx(model: str, base_url: str, api_key: str = "") -> Option
parts = line.strip().split()
if len(parts) >= 2:
try:
return int(parts[-1])
_ctx = int(parts[-1])
_local_probe_disk_put("ollama_num_ctx", _disk_key, _ctx)
return _ctx
except ValueError:
pass
@@ -1526,7 +1650,9 @@ def query_ollama_num_ctx(model: str, base_url: str, api_key: str = "") -> Option
model_info = data.get("model_info", {})
for key, value in model_info.items():
if "context_length" in key and isinstance(value, (int, float)):
return int(value)
_ctx = int(value)
_local_probe_disk_put("ollama_num_ctx", _disk_key, _ctx)
return _ctx
except Exception:
pass
return None
@@ -1860,6 +1986,7 @@ def _query_anthropic_context_length(model: str, base_url: str, api_key: str) ->
"x-api-key": api_key,
"anthropic-version": "2023-06-01",
}
_ensure_requests()
resp = requests.get(url, headers=headers, timeout=(5, 10), verify=_resolve_requests_verify())
if resp.status_code != 200:
return None
@@ -1904,32 +2031,73 @@ _CODEX_OAUTH_CONTEXT_FALLBACK: Dict[str, int] = {
}
_codex_oauth_context_cache: Dict[str, int] = {}
_codex_oauth_context_cache_time: float = 0.0
_codex_oauth_context_cache: Dict[str, Tuple[Dict[str, int], float]] = {}
_CODEX_OAUTH_CONTEXT_CACHE_TTL = 3600 # 1 hour
def _fetch_codex_oauth_context_lengths(access_token: str) -> Dict[str, int]:
"""Probe the ChatGPT Codex /models endpoint for per-slug context windows.
def _codex_oauth_token_fingerprint(access_token: str) -> str:
"""Return a non-secret cache key for a Codex OAuth access token."""
return hashlib.sha256(access_token.encode("utf-8")).hexdigest()[:16]
Codex OAuth imposes its own context limits that differ from the direct
OpenAI API (e.g. gpt-5.5 is 1.05M on the API, 272K on Codex). The
`context_window` field in each model entry is the authoritative source.
Returns a ``{slug: context_window}`` dict. Empty on failure.
def _extract_chatgpt_account_id(access_token: str) -> Optional[str]:
"""Extract ``chatgpt_account_id`` from the Codex OAuth JWT.
The Codex ``/backend-api/codex/models`` endpoint returns the per-account
catalog only when the ``ChatGPT-Account-Id`` header is present; without
it, the endpoint returns ``{"models":[]}`` (HTTP 200) and the context
probe falls back to the hardcoded defaults — which can be stale or
wrong for the active account's plan. Mirrors the same extraction done
in ``auxiliary_client.py`` for the request path.
Returns ``None`` on any parse error rather than raising, so a bad
token still surfaces as a normal probe failure instead of crashing
the metadata resolver.
"""
global _codex_oauth_context_cache, _codex_oauth_context_cache_time
try:
parts = access_token.split(".")
if len(parts) < 2:
return None
payload_b64 = parts[1] + "=" * (-len(parts[1]) % 4)
claims = json.loads(base64.urlsafe_b64decode(payload_b64))
if not isinstance(claims, dict):
return None
acct_id = claims.get("https://api.openai.com/auth", {}).get("chatgpt_account_id")
return acct_id if isinstance(acct_id, str) and acct_id else None
except Exception:
return None
def _fetch_codex_oauth_context_lengths_with_source(
access_token: str,
) -> Tuple[Dict[str, int], bool]:
"""Fetch Codex catalogue data and report whether it came from HTTP.
The in-process cache is scoped by token fingerprint because Codex model
availability and context windows can vary by account entitlement. The raw
token is never retained in the cache key. The boolean is false for a
same-token in-process hit, which must not be treated as a fresh provider
confirmation when deciding whether to update persistent state.
"""
global _codex_oauth_context_cache
now = time.time()
if (
_codex_oauth_context_cache
and now - _codex_oauth_context_cache_time < _CODEX_OAUTH_CONTEXT_CACHE_TTL
):
return _codex_oauth_context_cache
cache_key = _codex_oauth_token_fingerprint(access_token)
cached = _codex_oauth_context_cache.get(cache_key)
if cached is not None:
cached_models, cached_at = cached
if now - cached_at < _CODEX_OAUTH_CONTEXT_CACHE_TTL:
return cached_models, False
headers = {"Authorization": f"Bearer {access_token}"}
acct_id = _extract_chatgpt_account_id(access_token)
if acct_id:
headers["ChatGPT-Account-Id"] = acct_id
try:
_ensure_requests()
resp = requests.get(
"https://chatgpt.com/backend-api/codex/models?client_version=1.0.0",
headers={"Authorization": f"Bearer {access_token}"},
headers=headers,
timeout=(5, 10),
verify=_resolve_requests_verify(),
)
@@ -1938,11 +2106,11 @@ def _fetch_codex_oauth_context_lengths(access_token: str) -> Dict[str, int]:
"Codex /models probe returned HTTP %s; falling back to hardcoded defaults",
resp.status_code,
)
return {}
return {}, False
data = resp.json()
except Exception as exc:
logger.debug("Codex /models probe failed: %s", exc)
return {}
return {}, False
entries = data.get("models", []) if isinstance(data, dict) else []
result: Dict[str, int] = {}
@@ -1955,32 +2123,50 @@ def _fetch_codex_oauth_context_lengths(access_token: str) -> Dict[str, int]:
result[slug.strip()] = ctx
if result:
_codex_oauth_context_cache = result
_codex_oauth_context_cache_time = now
_codex_oauth_context_cache[cache_key] = (result, now)
return result, True
def _fetch_codex_oauth_context_lengths(access_token: str) -> Dict[str, int]:
"""Probe the ChatGPT Codex /models endpoint for per-slug context windows.
Codex OAuth imposes its own context limits that differ from the direct
OpenAI API (e.g. gpt-5.5 is 1.05M on the API, 272K on Codex). The
`context_window` field in each model entry is the authoritative source.
Returns a ``{slug: context_window}`` dict. Empty on failure.
"""
result, _fresh = _fetch_codex_oauth_context_lengths_with_source(access_token)
return result
def _resolve_codex_oauth_context_length(
def _resolve_codex_oauth_context_length_with_source(
model: str, access_token: str = ""
) -> Optional[int]:
) -> Tuple[Optional[int], str]:
"""Resolve a Codex OAuth model's real context window.
Prefers a live probe of chatgpt.com/backend-api/codex/models (when we
have a bearer token), then falls back to ``_CODEX_OAUTH_CONTEXT_FALLBACK``.
Returns ``(context_length, source)`` where source is ``"live"`` for a
value returned by a fresh authenticated endpoint probe, ``"memory"`` for
a same-token in-process catalogue hit, or ``"fallback"`` for the static
conservative table. Only ``"live"`` is eligible for persistent writes.
"""
model_bare = _strip_provider_prefix(model).strip()
if not model_bare:
return None
return None, ""
if access_token:
live = _fetch_codex_oauth_context_lengths(access_token)
live, fresh_probe = _fetch_codex_oauth_context_lengths_with_source(access_token)
live_source = "live" if fresh_probe else "memory"
if model_bare in live:
return live[model_bare]
return live[model_bare], live_source
# Case-insensitive match in case casing drifts
model_lower = model_bare.lower()
for slug, ctx in live.items():
if slug.lower() == model_lower:
return ctx
return ctx, live_source
# Fallback: longest-key-first substring match over hardcoded defaults.
model_lower = model_bare.lower()
@@ -1988,9 +2174,19 @@ def _resolve_codex_oauth_context_length(
_CODEX_OAUTH_CONTEXT_FALLBACK.items(), key=lambda x: len(x[0]), reverse=True
):
if slug in model_lower:
return ctx
return ctx, "fallback"
return None
return None, ""
def _resolve_codex_oauth_context_length(
model: str, access_token: str = ""
) -> Optional[int]:
"""Resolve a Codex OAuth model's context length (compatibility wrapper)."""
context_length, _source = _resolve_codex_oauth_context_length_with_source(
model, access_token=access_token,
)
return context_length
def _resolve_nous_context_length(
@@ -2080,9 +2276,9 @@ def get_model_context_length(
Resolution order:
0. Explicit config override (model.context_length or custom_providers per-model)
0c. Endpoint-scoped metadata for models validated on one multiplexed endpoint
1. Persistent cache (previously discovered via probing). Nous URLs
bypass the cache here so step 5b can always reconcile against
the authoritative portal /v1/models response.
1. Persistent cache (previously discovered via probing). Nous URLs,
LM Studio, and Codex OAuth bypass the cache here so their provider
metadata can be reconciled against the authoritative live source.
1b. AWS Bedrock static table (must precede custom-endpoint probe)
2. Active endpoint metadata (/models for explicit custom endpoints)
3. Local server query (for local endpoints)
@@ -2112,11 +2308,18 @@ def get_model_context_length(
# acting context, so they're ignored here.
if (provider or "").strip().lower() == "moa":
try:
from hermes_cli.config import load_config
from hermes_cli.config import (
get_compatible_custom_providers,
load_config,
)
from hermes_cli.moa_config import resolve_moa_preset
from hermes_cli.runtime_provider import resolve_runtime_provider
preset = resolve_moa_preset(load_config().get("moa") or {}, model)
config = load_config()
effective_custom_providers = custom_providers
if effective_custom_providers is None:
effective_custom_providers = get_compatible_custom_providers(config)
preset = resolve_moa_preset(config.get("moa") or {}, model)
agg = preset.get("aggregator") or {}
agg_provider = str(agg.get("provider") or "").strip()
agg_model = str(agg.get("model") or "").strip()
@@ -2126,7 +2329,8 @@ def get_model_context_length(
agg_model,
base_url=rt.get("base_url", "") or "",
api_key=rt.get("api_key", "") or "",
provider=agg_provider,
provider=rt.get("provider") or agg_provider,
custom_providers=effective_custom_providers,
)
except Exception:
logger.debug("MoA aggregator context-length resolution failed", exc_info=True)
@@ -2172,28 +2376,23 @@ def get_model_context_length(
if endpoint_context is not None:
return endpoint_context
is_bedrock_context = provider == "bedrock" or (
base_url
and base_url_hostname(base_url).startswith("bedrock-runtime.")
and base_url_host_matches(base_url, "amazonaws.com")
)
# 1. Check persistent cache (model+provider)
# LM Studio is excluded — its loaded context length is transient (the
# user can reload the model with a different context_length at any time
# via /api/v1/models/load), so a stale cached value would mask reloads.
# Codex OAuth is excluded because the authenticated /models catalogue is
# account-specific and a fallback must never suppress later revalidation.
if base_url and not _skip_persistent_context_cache(base_url, provider):
cached = get_cached_context_length(model, base_url)
if cached is not None:
# Invalidate stale Codex OAuth cache entries: pre-PR #14935 builds
# resolved gpt-5.x to the direct-API value (e.g. 1.05M) via
# models.dev and persisted it. Codex OAuth caps at 272K for every
# slug, so any cached Codex entry at or above 400K is a leftover
# from the old resolution path. Drop it and fall through to the
# live /models probe in step 5 below.
if provider == "openai-codex" and cached >= 400_000:
logger.info(
"Dropping stale Codex cache entry %s@%s -> %s (pre-fix value); "
"re-resolving via live /models probe",
model, base_url, f"{cached:,}",
)
_invalidate_cached_context_length(model, base_url)
# Invalidate stale 32k cache entries for Kimi-family models.
elif cached <= 32768 and _model_name_suggests_kimi(model):
if cached <= 32768 and _model_name_suggests_kimi(model):
logger.info(
"Dropping stale Kimi cache entry %s@%s -> %s (OpenRouter underreport); "
"re-resolving via hardcoded defaults",
@@ -2240,6 +2439,30 @@ def get_model_context_length(
model, base_url,
)
# Fall through; step 5b reconciles and overwrites if portal responds.
# Invalidate stale Bedrock entries seeded before the Claude 4.6+
# long-context table was corrected to 1M. The static table is a
# FLOOR, not an override: probe-derived cache entries (step 1b)
# may legitimately exceed the table (real window read from
# Bedrock's length-validation error), so only under-reporting
# entries are dropped — never a cached value above the table.
elif is_bedrock_context:
try:
from agent.bedrock_adapter import get_bedrock_context_length
bedrock_ctx = get_bedrock_context_length(model)
if cached < bedrock_ctx:
logger.info(
"Dropping stale Bedrock cache entry %s@%s -> %s; "
"using static Bedrock table value %s",
model,
base_url,
f"{cached:,}",
f"{bedrock_ctx:,}",
)
_invalidate_cached_context_length(model, base_url)
return bedrock_ctx
except ImportError:
pass
return cached
else:
if is_local_endpoint(base_url):
return _reconcile_local_cached_context_length(
@@ -2250,22 +2473,50 @@ def get_model_context_length(
# 1b. AWS Bedrock — use static context length table.
# Bedrock's ListFoundationModels API doesn't expose context window sizes,
# so we maintain a curated table in bedrock_adapter.py that reflects
# AWS-imposed limits (e.g. 200K for Claude models vs 1M on the native
# Anthropic API). This must run BEFORE the custom-endpoint probe at
# Bedrock-hosted model limits (e.g. older Claude 4 at 200K; Claude
# Opus/Sonnet 4.6+ at 1M). This must run BEFORE the custom-endpoint probe at
# step 2 — bedrock-runtime.<region>.amazonaws.com is not in
# _URL_TO_PROVIDER, so it would otherwise be treated as a custom endpoint,
# fail the /models probe (Bedrock doesn't expose that shape), and fall
# back to the 128K default before reaching the original step 4b branch.
if provider == "bedrock" or (
base_url
and base_url_hostname(base_url).startswith("bedrock-runtime.")
and base_url_host_matches(base_url, "amazonaws.com")
):
if is_bedrock_context:
try:
from agent.bedrock_adapter import get_bedrock_context_length
return get_bedrock_context_length(model)
from agent.bedrock_adapter import (
get_bedrock_context_length,
resolve_bedrock_region,
)
except ImportError:
pass # boto3 not installed — fall through to generic resolution
else:
# Bedrock does not expose the context window via any metadata API,
# so get_bedrock_context_length() probes the live endpoint (one
# fast, pre-inference length rejection) to read the real window.
# Cache the probe result per model so we pay that cost once, not
# every turn — keyed by base_url when present, else a synthetic
# bedrock:// key so display/offline paths share the entry.
cache_key_url = base_url or "bedrock://"
cached = get_cached_context_length(model, cache_key_url)
if cached is not None:
return cached
# Resolve region from the base_url host first, then the standard
# AWS region chain. An empty region disables probing (table only).
region = ""
if base_url:
_m = re.search(r"bedrock-runtime\.([a-z0-9-]+)\.", base_url)
if _m:
region = _m.group(1)
if not region:
try:
region = resolve_bedrock_region()
except Exception:
region = ""
ctx = get_bedrock_context_length(model, region=region, probe=bool(region))
if ctx and region:
# Only persist probe-derived values (region present); a pure
# table fallback shouldn't poison the cache against a later
# successful probe.
save_context_length(model, cache_key_url, ctx)
return ctx
if provider == "novita" or (base_url and base_url_host_matches(base_url, "api.novita.ai")):
ctx = _resolve_endpoint_context_length(model, base_url or "https://api.novita.ai/openai/v1", api_key=api_key)
@@ -2284,21 +2535,27 @@ def get_model_context_length(
if context_length is not None:
return context_length
if not _is_known_provider_base_url(base_url):
# 2b. Ollama native /api/show — any URL might be an Ollama server
# (local, cloud, or custom hosting). Non-Ollama servers return
# 404/405 quickly. Fall through on failure.
ctx = _query_ollama_api_show(model, base_url, api_key=api_key)
if ctx is not None:
if not _skip_persistent_context_cache(base_url, provider):
save_context_length(model, base_url, ctx)
return ctx
# 3. Try querying local server directly
# For local endpoints, run the probe that respects configured
# Modelfile context values first. _query_local_context_length
# prefers num_ctx from Modelfile, while _query_ollama_api_show
# returns the GGUF training max first which can be larger and
# would create a false-safe window for compression (#63122).
# Non-local endpoints preserve the existing GGUF-first behavior.
if is_local_endpoint(base_url):
local_ctx = _query_local_context_length(model, base_url, api_key=api_key)
if local_ctx and local_ctx > 0:
if not _skip_persistent_context_cache(base_url, provider):
_maybe_cache_local_context_length(model, base_url, local_ctx)
return local_ctx
# 2b. Ollama native /api/show — non-local endpoints preserve
# the existing generic /api/show GGUF-first behavior.
# Non-Ollama servers return 404/405 quickly.
ctx = _query_ollama_api_show(model, base_url, api_key=api_key)
if ctx is not None:
if not _skip_persistent_context_cache(base_url, provider):
save_context_length(model, base_url, ctx)
return ctx
# 3. Probe-down fallback after endpoint-specific detection failed
logger.info(
"Could not detect context length for model %r at %s — "
"defaulting to %s tokens (probe-down). Set model.context_length "
@@ -2380,9 +2637,14 @@ def get_model_context_length(
# Codex OAuth enforces lower context limits than the direct OpenAI
# API for the same slug (e.g. gpt-5.5 is 1.05M on the API but 272K
# on Codex). Authoritative source is Codex's own /models endpoint.
codex_ctx = _resolve_codex_oauth_context_length(model, access_token=api_key or "")
codex_ctx, codex_source = _resolve_codex_oauth_context_length_with_source(
model, access_token=api_key or "",
)
if codex_ctx:
if base_url:
# Only a successful authenticated catalogue response is safe to
# persist. The static fallback is deliberately runtime-only so a
# transient OAuth/network failure cannot poison future probes.
if base_url and codex_source == "live":
save_context_length(model, base_url, codex_ctx)
return codex_ctx
if effective_provider == "gmi" and base_url:
@@ -2525,16 +2787,61 @@ async def get_model_context_length_async(
)
def _is_cjk_token_dense_char(ch: str) -> bool:
code = ord(ch)
return (
0x1100 <= code <= 0x11FF # Hangul Jamo
or 0x2E80 <= code <= 0x9FFF # CJK radicals/ideographs
or 0xA960 <= code <= 0xA97F # Hangul Jamo Extended-A
or 0xAC00 <= code <= 0xD7AF # Hangul Syllables
or 0xF900 <= code <= 0xFAFF # CJK compatibility ideographs
or 0xFF00 <= code <= 0xFFEF # Fullwidth forms / halfwidth kana
)
# Same codepoint ranges as _is_cjk_token_dense_char, as a compiled character
# class so dense-char counting runs in C (``len(text) - len(re.sub(...))``)
# instead of a per-char Python loop. MUST stay in sync with
# _is_cjk_token_dense_char.
_CJK_DENSE_RE = re.compile(
"[\u1100-\u11ff" # Hangul Jamo
"\u2e80-\u9fff" # CJK radicals/ideographs
"\ua960-\ua97f" # Hangul Jamo Extended-A
"\uac00-\ud7af" # Hangul Syllables
"\uf900-\ufaff" # CJK compatibility ideographs
"\uff00-\uffef]" # Fullwidth forms / halfwidth kana
)
def estimate_tokens_rough(text: str) -> int:
"""Rough token estimate (~4 chars/token) for pre-flight checks.
"""Rough token estimate for pre-flight checks.
Uses ceiling division so short texts (1-3 chars) never estimate as
0 tokens, which would cause the compressor and pre-flight checks to
systematically undercount when many short tool results are present.
CJK/Hangul/Kana text is much denser than English under common LLM
tokenizers, so count those codepoints as roughly one token each instead
of applying the English-centric ~4 chars/token rule.
Perf: this runs on every message in every preflight/compaction walk,
including MB-scale tool outputs, so the common all-ASCII case must stay
O(1). ``str.isascii()`` is a flag check on CPython's compact unicode
representation (no scan), and the CJK counting itself is a single
C-level ``re.findall`` rather than a per-character Python loop.
"""
if not text:
return 0
return (len(text) + 3) // 4
text = str(text)
if text.isascii():
# O(1) fast path — ASCII text cannot contain token-dense CJK chars.
return (len(text) + 3) // 4
dense = len(text) - len(_CJK_DENSE_RE.sub("", text))
if not dense:
# Non-ASCII but no CJK (accents, Cyrillic, emoji, ...): keep the
# classic ~4 chars/token rule.
return (len(text) + 3) // 4
sparse = len(text) - dense
return dense + ((sparse + 3) // 4)
def estimate_messages_tokens_rough(messages: List[Dict[str, Any]]) -> int:
@@ -2544,14 +2851,92 @@ def estimate_messages_tokens_rough(messages: List[Dict[str, Any]]) -> int:
image — the Anthropic pricing model — instead of counting raw base64
character length. Without this, a single ~1MB screenshot would be
estimated at ~250K tokens and trigger premature context compression.
Per-message results are memoized (see ``_estimate_message_tokens_cached``)
keyed on a deep *identity fingerprint* of the message, so re-walking a
long history every iteration only pays for messages whose object graph
actually changed. The memo is exact: equal fingerprints imply identical
leaf objects and structure, hence an identical estimate.
"""
_IMAGE_TOKEN_COST = 1500
total_chars = 0
image_tokens = 0
total = 0
for msg in messages:
total_chars += _estimate_message_chars(msg)
image_tokens += _count_image_tokens(msg, _IMAGE_TOKEN_COST)
return ((total_chars + 3) // 4) + image_tokens
total += _estimate_message_tokens_cached(msg, _IMAGE_TOKEN_COST)
return total
# --- Per-message token-estimate memo -------------------------------------
#
# ``estimate_messages_tokens_rough`` is called on the full history every
# loop iteration (conversation_loop preflight), repeatedly during compaction
# telemetry, and inside an O(n^2) shrink loop in moa_loop. The per-message
# helpers are pure functions of the message's value, so a memo keyed on a
# fingerprint that uniquely determines the value is exactly equivalent.
#
# Fingerprint design (soundness argument):
# * strings are fingerprinted by ``id()`` AND pinned (a strong reference is
# stored in the cache entry). While the entry lives, that id cannot be
# reused by another object, so id-equality implies object-equality —
# strings are immutable, so value-equality too (no #50372-style aliasing).
# * ints/floats/bools/None are fingerprinted by value.
# * dicts/lists recurse structurally, preserving key order — ``str(shadow)``
# depends on insertion order, so order is part of the key.
# * any other type aborts the memo and falls through to a direct compute.
# Equal fingerprints therefore imply deep-equal messages built from identical
# immutable leaves ⇒ identical ``str(shadow)`` bytes ⇒ identical estimate.
#
# Because the api_messages build shallow-copies history dicts each iteration,
# the copies share the same content strings — so unchanged history messages
# hit the memo even though the outer dicts are fresh objects every turn.
_MSG_TOKENS_CACHE: Dict[Any, Tuple[list, int]] = {}
_MSG_TOKENS_CACHE_MAX = 4096
def _msg_fingerprint(value: Any, pins: list) -> Any:
if value is None or value is True or value is False:
return value
t = type(value)
if t is str:
pins.append(value)
return ("s", id(value))
if t is int or t is float:
return ("n", t.__name__, value)
if t is dict:
return ("d", tuple(
(_msg_fingerprint(k, pins), _msg_fingerprint(v, pins))
for k, v in value.items()
))
if t is list:
return ("l", tuple(_msg_fingerprint(v, pins) for v in value))
if t is tuple:
return ("t", tuple(_msg_fingerprint(v, pins) for v in value))
raise ValueError("unfingerprintable message value")
def _estimate_message_tokens_cached(msg: Any, image_cost: int) -> int:
try:
pins: list = []
key = _msg_fingerprint(msg, pins)
hash(key)
except Exception:
return (
_estimate_message_tokens_without_images(msg)
+ _count_image_tokens(msg, image_cost)
)
cached = _MSG_TOKENS_CACHE.get(key)
if cached is not None:
return cached[1]
tokens = (
_estimate_message_tokens_without_images(msg)
+ _count_image_tokens(msg, image_cost)
)
_MSG_TOKENS_CACHE[key] = (pins, tokens)
while len(_MSG_TOKENS_CACHE) > _MSG_TOKENS_CACHE_MAX:
try:
_MSG_TOKENS_CACHE.pop(next(iter(_MSG_TOKENS_CACHE)))
except (StopIteration, KeyError, RuntimeError):
break
return tokens
def _count_image_tokens(msg: Dict[str, Any], cost_per_image: int) -> int:
@@ -2613,6 +2998,35 @@ def _estimate_message_chars(msg: Dict[str, Any]) -> int:
return len(str(shadow))
def _estimate_message_tokens_without_images(msg: Dict[str, Any]) -> int:
"""Token estimate for a message shadow with image payloads stripped."""
if not isinstance(msg, dict):
return estimate_tokens_rough(str(msg))
shadow: Dict[str, Any] = {}
for k, v in msg.items():
if k == "_anthropic_content_blocks":
continue
if k == "content":
if isinstance(v, list):
cleaned = []
for part in v:
if isinstance(part, dict):
if part.get("type") in {"image", "image_url", "input_image"}:
cleaned.append({"type": part.get("type"), "image": "[stripped]"})
else:
cleaned.append(part)
else:
cleaned.append(part)
shadow[k] = cleaned
elif isinstance(v, dict) and v.get("_multimodal"):
shadow[k] = v.get("text_summary", "")
else:
shadow[k] = v
else:
shadow[k] = v
return estimate_tokens_rough(str(shadow))
def estimate_request_tokens_rough(
messages: List[Dict[str, Any]],
*,
@@ -2629,7 +3043,7 @@ def estimate_request_tokens_rough(
"""
total = 0
if system_prompt:
total += (len(system_prompt) + 3) // 4
total += estimate_tokens_rough(system_prompt)
if messages:
total += estimate_messages_tokens_rough(messages)
if tools:
+236 -58
View File
@@ -8,11 +8,15 @@ of 4000+ models across 109+ providers. Provides:
(reasoning, tools, vision, PDF, audio), modalities, knowledge cutoff,
open-weights flag, family grouping, deprecation status
Data resolution order (like TypeScript OpenCode):
1. Bundled snapshot (ships with the package — offline-first)
2. Disk cache (~/.hermes/models_dev_cache.json)
3. Network fetch (https://models.dev/api.json)
4. Background refresh every 60 minutes
Data resolution order:
1. In-memory cache (fresh, or stale served immediately while a single
background daemon thread refreshes)
2. Disk cache (~/.hermes/models_dev_cache.json — any age; stale data is
served rather than blocking callers on the network)
3. Network fetch (https://models.dev/api.json) — only when no cache
exists at all; failed refreshes back off for 5 minutes process-wide
Latency-sensitive callers (gateway route-identity checks) pass
``allow_network=False`` and never touch the network.
Other modules should import the dataclasses and query functions from here
rather than parsing the raw JSON themselves.
@@ -20,6 +24,7 @@ rather than parsing the raw JSON themselves.
import json
import logging
import threading
import time
from dataclasses import dataclass
from pathlib import Path
@@ -33,10 +38,15 @@ logger = logging.getLogger(__name__)
MODELS_DEV_URL = "https://models.dev/api.json"
_MODELS_DEV_CACHE_TTL = 3600 # 1 hour in-memory
_MODELS_DEV_RETRY_DELAY = 300 # 5 minutes after a failed refresh
# In-memory cache
_models_dev_cache: Dict[str, Any] = {}
_models_dev_cache_time: float = 0
_models_dev_retry_after: float = 0
_models_dev_fetch_lock = threading.Lock()
_models_dev_refresh_lock = threading.Lock()
_models_dev_refresh_in_flight = False
# ---------------------------------------------------------------------------
@@ -158,6 +168,7 @@ PROVIDER_TO_MODELS_DEV: Dict[str, str] = {
"alibaba": "alibaba",
"qwen-oauth": "alibaba",
"copilot": "github-copilot",
"ai-gateway": "vercel",
"opencode-zen": "opencode",
"opencode-go": "opencode-go",
"kilocode": "kilo",
@@ -237,27 +248,157 @@ def _save_disk_cache(data: Dict[str, Any]) -> None:
logger.debug("Failed to save models.dev disk cache: %s", e)
def fetch_models_dev(force_refresh: bool = False) -> Dict[str, Any]:
def _fetch_models_dev_from_network() -> Dict[str, Any]:
"""Fetch the live models.dev registry without touching local caches.
Raises on network errors and on an empty/invalid registry payload.
"""
# Tuple (connect, read): a flat timeout=15 let a blackholed connect
# stall the first-turn critical path for the full 15 s. 5 s connect
# fails fast on unreachable hosts; 10 s read still tolerates a slow
# registry response (matches the OpenRouter fetch convention in
# agent/model_metadata.py).
response = requests.get(MODELS_DEV_URL, timeout=(5, 10))
response.raise_for_status()
data = response.json()
if not isinstance(data, dict) or not data:
raise ValueError("models.dev returned an empty or invalid registry")
return data
def _mark_stale_cache_grace() -> None:
"""Give stale cache data a short in-memory grace before retrying refresh.
Only ever moves the timestamp forward: if a background refresh completed
between the caller's staleness check and this call, the fresh timestamp
is preserved instead of being rewound to a 5-minute grace.
"""
global _models_dev_cache_time
grace_time = time.time() - _MODELS_DEV_CACHE_TTL + _MODELS_DEV_RETRY_DELAY
if grace_time > _models_dev_cache_time:
_models_dev_cache_time = grace_time
def _commit_registry(data: Dict[str, Any], *, where: str) -> None:
"""Persist a freshly fetched registry: disk + in-mem + clear backoff.
Callers must hold ``_models_dev_fetch_lock`` so a failing refresh on one
path can never stomp the state a succeeding refresh on the other path
just committed (e.g. a failing background worker re-arming the backoff
immediately after a successful ``force_refresh``).
"""
global _models_dev_cache, _models_dev_cache_time, _models_dev_retry_after
_save_disk_cache(data)
_models_dev_cache = data
_models_dev_cache_time = time.time()
_models_dev_retry_after = 0
logger.debug(
"Refreshed models.dev registry (%s): %d providers, %d total models",
where,
len(data),
sum(len(p.get("models", {})) for p in data.values() if isinstance(p, dict)),
)
def _note_refresh_failure(exc: Exception, *, where: str) -> None:
"""Record a failed refresh: arm the process-wide 5-minute backoff.
Callers must hold ``_models_dev_fetch_lock`` (see ``_commit_registry``).
"""
global _models_dev_retry_after
_models_dev_retry_after = time.time() + _MODELS_DEV_RETRY_DELAY
logger.debug(
"models.dev refresh failed (%s); retry suppressed for %ds: %s",
where,
_MODELS_DEV_RETRY_DELAY,
exc,
)
def _background_refresh_models_dev() -> None:
"""Best-effort refresh after serving stale cache data."""
global _models_dev_refresh_in_flight
try:
data = _fetch_models_dev_from_network()
with _models_dev_fetch_lock:
_commit_registry(data, where="background")
except Exception as e:
with _models_dev_fetch_lock:
_note_refresh_failure(e, where="background")
finally:
with _models_dev_refresh_lock:
_models_dev_refresh_in_flight = False
def _start_background_refresh_models_dev() -> None:
"""Start one daemon refresh worker if none is already running.
Honors the process-wide failure backoff: after a failed refresh,
no new background worker is spawned until ``_models_dev_retry_after``.
"""
global _models_dev_refresh_in_flight
if time.time() < _models_dev_retry_after:
return
with _models_dev_refresh_lock:
if _models_dev_refresh_in_flight:
return
_models_dev_refresh_in_flight = True
thread = threading.Thread(
target=_background_refresh_models_dev,
name="models-dev-refresh",
daemon=True,
)
try:
thread.start()
except Exception as e:
# Thread/fd exhaustion: clear the flag so refresh isn't disabled
# for the rest of the process lifetime. Callers still get stale data.
with _models_dev_refresh_lock:
_models_dev_refresh_in_flight = False
logger.debug("Failed to start models.dev refresh thread: %s", e)
def fetch_models_dev(
force_refresh: bool = False, *, allow_network: bool = True
) -> Dict[str, Any]:
"""Fetch models.dev registry. Cache hierarchy: in-mem → disk → network.
Returns the full registry dict keyed by provider ID, or empty dict on failure.
Cache hierarchy (when ``force_refresh=False``):
1. In-memory cache, populated and < TTL old → return immediately.
2. **Disk cache file < TTL old by mtime → load, populate in-mem, return.**
No network call. Saves ~500 ms per cold-start agent construction;
``models.dev`` only changes when providers add new models, so a
1 hour staleness window is acceptable (same TTL as in-mem cache).
3. Network fetch → on success, save to disk + in-mem and return.
4. Network fails → fall back to ANY available disk cache (even stale)
with a short 5 min in-mem grace period before retrying network.
1. Fresh in-memory cache → return immediately.
2. Stale in-memory cache → return immediately and refresh in a single
background daemon thread. Callers never block on the network while
any cache exists; ``models.dev`` only changes when providers add
new models, so stale data is preferable to a foreground timeout.
3. Disk cache file (any age) → load, populate in-mem, return
immediately. Stale disk caches trigger the same background refresh.
4. No cache at all → singleflight foreground network fetch. On
success, save to disk + in-mem and return.
5. Any failed refresh (foreground or background) suppresses further
automatic refreshes for 5 minutes process-wide.
When ``force_refresh=True`` (used by ``hermes config refresh``, the
\"refresh model catalog\" code path), stages 1 and 2 are skipped. The
function always hits the network and only falls back to disk if the
network call fails.
\"refresh model catalog\" code path), cache fast paths and the failure
backoff are bypassed; the function hits the network and only falls back
to cached data if the call fails. When ``allow_network=False``, any
memory or disk cache is returned regardless of age and no request is
made — used by latency-sensitive paths (gateway route-identity checks)
that must never wait on the network.
"""
global _models_dev_cache, _models_dev_cache_time
global _models_dev_cache, _models_dev_cache_time, _models_dev_retry_after
if not allow_network:
if _models_dev_cache:
return _models_dev_cache
disk_data = _load_disk_cache()
if disk_data:
_models_dev_cache = disk_data
disk_age = _disk_cache_age_seconds()
_models_dev_cache_time = (
time.time() - disk_age if disk_age is not None else 0
)
return _models_dev_cache
# Stage 1: fresh in-memory cache wins. This is the hot path on
# long-lived processes — no I/O, no system calls.
@@ -268,54 +409,82 @@ def fetch_models_dev(force_refresh: bool = False) -> Dict[str, Any]:
):
return _models_dev_cache
# Stage 2: fresh-by-mtime disk cache short-circuits the network call.
# Only kicks in on cold-start processes (in-mem cache is empty or
# expired) and only when the user hasn't asked for a forced refresh.
# Skipped if the disk cache file is missing, unreadable, or older
# than _MODELS_DEV_CACHE_TTL.
# Stage 2: stale in-memory cache is still better than blocking provider
# resolution on a foreground network timeout. Refresh it in the background.
if not force_refresh and _models_dev_cache:
_mark_stale_cache_grace()
_start_background_refresh_models_dev()
logger.debug(
"Using stale in-memory models.dev cache; refreshing in background"
)
return _models_dev_cache
# Stage 3: disk cache short-circuits the network call.
# Only kicks in on cold-start processes (in-mem cache is empty) and only
# when the user hasn't asked for a forced refresh. A stale disk cache is
# deliberately usable: provider/model resolution should not hang just
# because models.dev is unreachable.
if not force_refresh:
disk_age = _disk_cache_age_seconds()
if disk_age is not None and disk_age < _MODELS_DEV_CACHE_TTL:
if disk_age is not None:
disk_data = _load_disk_cache()
if disk_data:
_models_dev_cache = disk_data
# Anchor in-mem TTL to the disk file's age so we don't
# extend an already-aging cache by another full hour.
_models_dev_cache_time = time.time() - disk_age
logger.debug(
"Loaded models.dev from fresh disk cache "
"(%d providers, age=%.0fs)", len(disk_data), disk_age,
)
if disk_age < _MODELS_DEV_CACHE_TTL:
# Anchor in-mem TTL to the disk file's age so we don't
# extend an already-aging cache by another full hour.
_models_dev_cache_time = time.time() - disk_age
logger.debug(
"Loaded models.dev from fresh disk cache "
"(%d providers, age=%.0fs)", len(disk_data), disk_age,
)
else:
_mark_stale_cache_grace()
_start_background_refresh_models_dev()
logger.debug(
"Using stale models.dev disk cache (age=%.0fs); "
"refreshing in background",
disk_age,
)
return _models_dev_cache
# Stage 3: network fetch.
try:
response = requests.get(MODELS_DEV_URL, timeout=15)
response.raise_for_status()
data = response.json()
if isinstance(data, dict) and data:
_models_dev_cache = data
_models_dev_cache_time = time.time()
_save_disk_cache(data)
logger.debug(
"Fetched models.dev registry: %d providers, %d total models",
len(data),
sum(len(p.get("models", {})) for p in data.values() if isinstance(p, dict)),
)
# Failed automatic refreshes are process-wide. Avoid making every caller
# retry the same unreachable endpoint while no usable cache exists.
if not force_refresh and time.time() < _models_dev_retry_after:
return _models_dev_cache
# Stage 4: singleflight foreground network fetch — only reached when no
# memory or disk cache exists (or on force_refresh). Recheck state after
# acquiring the lock because another caller may have refreshed or
# established backoff while we waited.
with _models_dev_fetch_lock:
now = time.time()
if not force_refresh:
if _models_dev_cache:
return _models_dev_cache
if now < _models_dev_retry_after:
return _models_dev_cache
try:
data = _fetch_models_dev_from_network()
_commit_registry(data, where="foreground")
return data
except Exception as e:
logger.debug("Failed to fetch models.dev: %s", e)
except Exception as e:
_note_refresh_failure(e, where="foreground")
# Stage 4: network failed — fall back to whatever disk cache exists,
# even if it's stale. Give it a short 5 min in-mem TTL so we retry
# the network soon instead of serving stale data for a full hour.
if not _models_dev_cache:
_models_dev_cache = _load_disk_cache()
if _models_dev_cache:
_models_dev_cache_time = time.time() - _MODELS_DEV_CACHE_TTL + 300
logger.debug("Loaded models.dev from disk cache (%d providers)", len(_models_dev_cache))
# Stage 5: network failed — return any stale memory/disk cache. Cache
# freshness remains expired; the retry-after timestamp controls when
# the next automatic request is allowed.
if not _models_dev_cache:
_models_dev_cache = _load_disk_cache()
_models_dev_cache_time = 0
if _models_dev_cache:
logger.debug(
"Loaded stale models.dev disk cache (%d providers)",
len(_models_dev_cache),
)
return _models_dev_cache
return _models_dev_cache
def lookup_models_dev_context(provider: str, model: str) -> Optional[int]:
@@ -671,7 +840,9 @@ def _parse_provider_info(provider_id: str, raw: Dict[str, Any]) -> ProviderInfo:
# Provider-level queries
# ---------------------------------------------------------------------------
def get_provider_info(provider_id: str) -> Optional[ProviderInfo]:
def get_provider_info(
provider_id: str, *, allow_network: bool = True
) -> Optional[ProviderInfo]:
"""Get full provider metadata from models.dev.
Accepts either a Hermes provider ID (e.g. "kilocode") or a models.dev
@@ -680,7 +851,14 @@ def get_provider_info(provider_id: str) -> Optional[ProviderInfo]:
# Resolve Hermes ID → models.dev ID
mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id)
data = fetch_models_dev()
# NOTE: keep the zero-argument call on the default path. Dozens of test
# sites monkeypatch fetch_models_dev with zero-arg lambdas; passing the
# kwarg unconditionally would break them all (they raise TypeError).
data = (
fetch_models_dev()
if allow_network
else fetch_models_dev(allow_network=False)
)
raw = data.get(mdev_id)
if not isinstance(raw, dict):
return None
+29
View File
@@ -0,0 +1,29 @@
"""Hermes gateway monitoring.
Service health monitoring plus redacted operational diagnostics for the
gateway daemon, exported over OTLP to an operator-configured endpoint.
``emitter`` is the in-process event bus: producers (gateway status hooks,
the diagnostic log handler) hand typed events to a fire-and-forget queue,
and subscribers (the OTLP streamers) consume them off the hot path. The
emitter never blocks or raises into gateway code (the hot-path invariant),
and nothing is persisted locally — monitoring is an egress path, not a store.
Deliberately out of scope here: run/model/tool trajectory capture, usage
analytics, and any content-bearing signal. Those planes are served by the
NeMo Relay integration and its Hermes-owned subscribers.
"""
from __future__ import annotations
from . import emitter, events
emit = emitter.emit
get_emitter = emitter.get_emitter
__all__ = [
"emitter",
"events",
"emit",
"get_emitter",
]
+201
View File
@@ -0,0 +1,201 @@
"""Content-free cron service-health and execution telemetry projection."""
from __future__ import annotations
import hashlib
import logging
import re
from dataclasses import dataclass
from datetime import datetime
from typing import Any, Optional
from agent.monitoring.events import CronExecutionEvent
from agent.monitoring.gateway_health import GatewayHealthSnapshot, GatewayMetric
from cron.jobs import (
_compute_grace_seconds,
get_catch_up_occurrence_count,
get_ticker_heartbeat_age,
get_ticker_success_age,
load_jobs,
)
from cron.scheduler import get_running_job_ids
from hermes_time import now as _hermes_now
logger = logging.getLogger(__name__)
_KNOWN_STATUSES = {"claimed", "running", "completed", "failed", "unknown"}
_KNOWN_SOURCES = {"builtin", "direct", "external"}
_KNOWN_DELIVERY_OUTCOMES = {"delivered", "failed", "suppressed", "not_configured"}
@dataclass(frozen=True, slots=True)
class CronHealthSnapshot:
metrics: list[GatewayMetric]
events: list[CronExecutionEvent]
def _now() -> datetime:
return _hermes_now()
def _job_key(raw: Any) -> str:
value = str(raw or "unknown").encode("utf-8", errors="replace")
return f"sha256:{hashlib.sha256(value).hexdigest()[:24]}"
def classify_cron_error(raw: Any) -> str:
text = str(raw or "").lower()
if (
re.search(r"\b(?:authentication|authenticated|authenticate|authorization|authorized|authorize|unauthorized|forbidden)\b", text)
or re.search(r"\bbearer\b", text)
or re.search(r"\b(?:access|api|refresh) token\b", text)
or re.search(r"\b(?:401|403)\b", text)
):
return "auth_failed"
if "rate limit" in text or "429" in text or "quota" in text:
return "rate_limited"
if "timeout" in text or "timed out" in text:
return "timeout"
if any(value in text for value in ("network", "connection", "dns", "socket", "unreachable")):
return "network_error"
if "dispatch" in text or "executor" in text:
return "dispatch_failed"
if "interrupt" in text or "owner exited" in text or "restarted" in text:
return "interrupted"
if "empty response" in text:
return "empty_response"
if any(value in text for value in ("config", "missing", "invalid")):
return "invalid_config"
return "unknown"
def _parse_time(raw: Any) -> Optional[datetime]:
try:
return datetime.fromisoformat(str(raw)) if raw else None
except (TypeError, ValueError):
return None
def _duration_ms(record: dict[str, Any]) -> Optional[int]:
start = _parse_time(record.get("started_at")) or _parse_time(record.get("claimed_at"))
finish = _parse_time(record.get("finished_at"))
if start is None or finish is None:
return None
try:
duration = int((finish - start).total_seconds() * 1000)
except (TypeError, ValueError):
return None
return max(0, duration)
def project_execution_event(
record: dict[str, Any], *, delivery_outcome: Optional[str] = None
) -> CronExecutionEvent:
status = str(record.get("status") or "unknown").lower()
source = str(record.get("source") or "unknown").lower()
if source not in _KNOWN_SOURCES and source != "unknown":
source = "external"
outcome = str(delivery_outcome).lower() if delivery_outcome is not None else None
return CronExecutionEvent(
status=status if status in _KNOWN_STATUSES else "unknown",
job_key=_job_key(record.get("job_id")),
source=source if source in _KNOWN_SOURCES else "unknown",
duration_ms=_duration_ms(record),
delivery_outcome=(
outcome if outcome in _KNOWN_DELIVERY_OUTCOMES else None
),
error_class=(
classify_cron_error(record.get("error"))
if status in {"failed", "unknown"}
else None
),
)
def emit_execution_state(
record: Optional[dict[str, Any]], *, delivery_outcome: Optional[str] = None
) -> None:
"""Best-effort lifecycle emit; terminal states synchronously cross the queue barrier."""
if not record:
return
try:
from agent.monitoring import emitter
event = project_execution_event(record, delivery_outcome=delivery_outcome)
target = emitter.get_emitter()
target.emit(event)
if event.status in {"completed", "failed", "unknown"}:
target.flush(timeout=1.0)
except Exception:
logger.debug("cron execution telemetry emit failed", exc_info=True)
def _is_overdue(job: dict[str, Any], now: datetime) -> bool:
if not job.get("enabled", True):
return False
next_run = _parse_time(job.get("next_run_at"))
schedule = job.get("schedule")
if next_run is None or not isinstance(schedule, dict):
return False
try:
if next_run.tzinfo is None and now.tzinfo is not None:
next_run = next_run.replace(tzinfo=now.tzinfo)
lateness = (now - next_run).total_seconds()
return lateness > _compute_grace_seconds(schedule)
except (TypeError, ValueError):
return False
def build_cron_health_snapshot() -> CronHealthSnapshot:
metrics: list[GatewayMetric] = []
for name, reader in (
("hermes.cron.scheduler.heartbeat_age_seconds", get_ticker_heartbeat_age),
("hermes.cron.scheduler.last_success_age_seconds", get_ticker_success_age),
):
try:
value = reader()
if value is not None:
metrics.append(GatewayMetric(name, max(0.0, float(value)), {}))
except Exception:
logger.debug("cron freshness metric unavailable", exc_info=True)
try:
metrics.append(
GatewayMetric(
"hermes.cron.scheduler.catch_up_occurrences",
get_catch_up_occurrence_count(),
{},
)
)
except Exception:
logger.debug("cron catch-up metric unavailable", exc_info=True)
try:
jobs = load_jobs()
enabled = [job for job in jobs if job.get("enabled", True)]
metrics.append(GatewayMetric("hermes.cron.jobs.enabled", len(enabled), {}))
metrics.append(
GatewayMetric(
"hermes.cron.jobs.overdue",
sum(1 for job in enabled if _is_overdue(job, _now())),
{},
)
)
except Exception:
logger.debug("cron job metrics unavailable", exc_info=True)
try:
metrics.append(
GatewayMetric("hermes.cron.jobs.running", len(get_running_job_ids()), {})
)
except Exception:
logger.debug("cron running-job metric unavailable", exc_info=True)
return CronHealthSnapshot(metrics=metrics, events=[])
__all__ = [
"CronHealthSnapshot",
"build_cron_health_snapshot",
"classify_cron_error",
"emit_execution_state",
"project_execution_event",
]
+211
View File
@@ -0,0 +1,211 @@
"""Monitoring emitter: fire-and-forget queue + background dispatcher.
The emitter is the single seam between producers (gateway status hooks, the
diagnostic log handler) and consumers (the OTLP streamers). Its contract is
the hot-path invariant:
``emit()`` MUST return in O(microseconds), MUST NOT block on disk/network,
and MUST NEVER raise into the caller. A monitoring failure is logged
locally and dropped — it can never affect the gateway or a session.
Mechanism:
* ``emit(event)`` does a non-blocking ``queue.put_nowait`` wrapped in a bare
except. On a full queue it drops the *oldest* event and counts the drop.
* A daemon thread drains the queue and fans each batch out to subscribers
(the OTLP metric/span/log streamers). Each subscriber is fail-isolated —
a slow or raising subscriber never affects the hot path or its peers.
Nothing is persisted here. Monitoring is an egress path, not a local store;
if no subscriber is attached, events simply age out of the ring buffer.
"""
from __future__ import annotations
import logging
import queue
import threading
import time
from typing import Any, Dict, Optional
logger = logging.getLogger(__name__)
_MAX_QUEUE = 10_000 # ring-buffer depth; oldest dropped when full
_DRAIN_BATCH = 256
class MonitoringEmitter:
"""Owns the queue, the dispatcher thread, and the subscriber list."""
def __init__(self, *, enabled: bool = True) -> None:
self._enabled = enabled
self._q: "queue.Queue[Dict[str, Any]]" = queue.Queue(maxsize=_MAX_QUEUE)
self._dropped = 0
self._dispatched = 0
self._stop = threading.Event()
self._started = False
self._lock = threading.Lock()
self._thread: Optional[threading.Thread] = None
# Live subscribers (the OTLP streamers). Called from the dispatcher
# thread, fully fail-isolated. Each subscriber is callable(batch: list[dict]).
self._subscribers: list = []
# ── public API (hot path) ───────────────────────────────────────────────
def emit(self, event: Any) -> None:
"""Enqueue an event. Never blocks, never raises.
``event`` may be a dataclass with ``to_dict()`` or a plain dict.
"""
if not self._enabled:
return
try:
payload = event.to_dict() if hasattr(event, "to_dict") else dict(event)
payload.setdefault("ts_ns", time.time_ns())
self._ensure_started()
try:
self._q.put_nowait(payload)
except queue.Full:
# Drop oldest to make room — bounded memory, newest-wins.
try:
self._q.get_nowait()
self._q.task_done()
self._dropped += 1
self._q.put_nowait(payload)
except Exception:
self._dropped += 1
except Exception: # the hot-path invariant: never propagate
logger.debug("monitoring emit failed", exc_info=True)
# ── lifecycle ───────────────────────────────────────────────────────────
def _ensure_started(self) -> None:
if self._started:
return
with self._lock:
if self._started:
return
self._thread = threading.Thread(
target=self._run, name="hermes-monitoring-dispatch", daemon=True
)
self._thread.start()
self._started = True
def _run(self) -> None:
while not self._stop.is_set():
try:
first = self._q.get(timeout=0.5)
except queue.Empty:
continue
batch = [first]
while len(batch) < _DRAIN_BATCH:
try:
batch.append(self._q.get_nowait())
except queue.Empty:
break
try:
self._dispatch(batch)
finally:
for _ in batch:
self._q.task_done()
def _dispatch(self, batch) -> None:
# Fan-out to subscribers (OTLP streamers) — fully fail-isolated.
for sub in list(self._subscribers):
try:
sub(batch)
except Exception:
logger.debug("monitoring subscriber failed", exc_info=True)
self._dispatched += len(batch)
def subscribe(self, callback) -> None:
"""Register a live batch subscriber (callable(batch: list[dict]))."""
if callback not in self._subscribers:
self._subscribers.append(callback)
self._enabled = True
def unsubscribe(self, callback) -> None:
try:
self._subscribers.remove(callback)
except ValueError:
pass
if not self._subscribers:
self._enabled = False
# ── introspection / shutdown (tests, CLI) ───────────────────────────────
def flush(self, timeout: float = 2.0) -> None:
"""Wait boundedly for queued and in-flight batches to finish dispatch."""
if timeout <= 0:
return
finished = threading.Event()
def _wait_for_completion() -> None:
self._q.join()
finished.set()
waiter = threading.Thread(
target=_wait_for_completion,
name="hermes-monitoring-flush",
daemon=True,
)
waiter.start()
finished.wait(timeout=timeout)
def stats(self) -> Dict[str, int]:
return {
"queued": self._q.qsize(),
"dispatched": self._dispatched,
"dropped": self._dropped,
"subscribers": len(self._subscribers),
}
def close(self) -> None:
self._stop.set()
if self._thread is not None:
self._thread.join(timeout=2.0)
self._started = False
# ── process-wide singleton ──────────────────────────────────────────────────
_EMITTER: Optional[MonitoringEmitter] = None
_EMITTER_LOCK = threading.Lock()
def get_emitter() -> MonitoringEmitter:
"""Return the process-wide monitoring emitter."""
global _EMITTER
if _EMITTER is not None:
return _EMITTER
with _EMITTER_LOCK:
if _EMITTER is None:
# Collection is opt-in. A plane exporter enables the singleton by
# attaching its first subscriber; until then producers are no-ops.
_EMITTER = MonitoringEmitter(enabled=False)
return _EMITTER
def emit(event: Any) -> None:
"""Module-level convenience: emit via the singleton."""
get_emitter().emit(event)
def reset_emitter_for_tests(emitter: Optional[MonitoringEmitter] = None) -> None:
"""Swap the singleton (tests only)."""
global _EMITTER
with _EMITTER_LOCK:
if _EMITTER is not None and emitter is not _EMITTER:
try:
_EMITTER.close()
except Exception:
pass
_EMITTER = emitter
# Back-compat alias for the salvaged class name used in emozilla's tests.
TelemetryEmitter = MonitoringEmitter
__all__ = [
"MonitoringEmitter",
"TelemetryEmitter",
"get_emitter",
"emit",
"reset_emitter_for_tests",
]
+86
View File
@@ -0,0 +1,86 @@
"""Typed gateway monitoring events.
Content-free service-health and redacted diagnostic events for the gateway
daemon. These are the only event shapes the monitoring plane emits: no
prompts, messages, tool args/results, session history, or usage analytics.
"""
from __future__ import annotations
import time
from dataclasses import dataclass, field, asdict
from typing import Any, Dict, Optional
def _now_ns() -> int:
return time.time_ns()
@dataclass(slots=True)
class GatewayHealthEvent:
"""Content-free gateway health snapshot or lifecycle event."""
name: str
gateway_state: Optional[str] = None
old_state: Optional[str] = None
new_state: Optional[str] = None
exit_reason: Optional[str] = None
restart_requested: Optional[bool] = None
active_agents: int = 0
gateway_busy: bool = False
gateway_drainable: bool = False
platform_count: int = 0
fatal_platform_count: int = 0
profile: Optional[str] = None
install_id: Optional[str] = None
version: Optional[str] = None
supervision_mode: Optional[str] = None
pid: Optional[int] = None
ts_ns: int = field(default_factory=_now_ns)
def to_dict(self) -> Dict[str, Any]:
return {"event": "gateway_health", **asdict(self)}
@dataclass(slots=True)
class GatewayDiagnosticEvent:
"""Redacted gateway diagnostic event for operator-owned observability."""
name: str
subsystem: str
error_class: str = "unknown"
error_code: Optional[str] = None
platform: Optional[str] = None
old_state: Optional[str] = None
new_state: Optional[str] = None
profile: Optional[str] = None
version: Optional[str] = None
severity: str = "warning"
ts_ns: int = field(default_factory=_now_ns)
source_logger: Optional[str] = None
def to_dict(self) -> Dict[str, Any]:
return {"event": "gateway_diagnostic", **asdict(self)}
@dataclass(slots=True)
class CronExecutionEvent:
"""Content-free durable cron execution lifecycle projection."""
status: str
job_key: str
source: str = "unknown"
duration_ms: Optional[int] = None
delivery_outcome: Optional[str] = None
error_class: Optional[str] = None
ts_ns: int = field(default_factory=_now_ns)
def to_dict(self) -> Dict[str, Any]:
return {"event": "cron_execution", **asdict(self)}
__all__ = [
"GatewayHealthEvent",
"GatewayDiagnosticEvent",
"CronExecutionEvent",
]
+469
View File
@@ -0,0 +1,469 @@
"""Gateway health and diagnostics signal producer.
This module keeps the plane narrow: service health monitoring plus
redacted operational diagnostics. It reuses the existing gateway runtime-status
contract and emits content-free metrics/events. No prompts, messages, tool args,
session history, audit records, or product analytics belong here.
"""
from __future__ import annotations
import hashlib
import logging
import re
from dataclasses import dataclass
from typing import Any, Dict, List, Optional
from agent.monitoring.events import GatewayDiagnosticEvent, GatewayHealthEvent
@dataclass(frozen=True, slots=True)
class GatewayMetric:
name: str
value: int | float
attributes: Dict[str, str]
@dataclass(frozen=True, slots=True)
class GatewayHealthSnapshot:
metrics: List[GatewayMetric]
events: List[GatewayHealthEvent | GatewayDiagnosticEvent]
_RUNNING_PLATFORM_STATES = {"running", "connected", "ok", "ready"}
_FATAL_PLATFORM_STATES = {"fatal", "degraded", "error", "failed"}
_KNOWN_GATEWAY_STATES = {
"starting", "draining", "stopping", "stopped", "startup_failed", "unknown"
} | _RUNNING_PLATFORM_STATES | _FATAL_PLATFORM_STATES
_KNOWN_PLATFORM_STATES = _RUNNING_PLATFORM_STATES | _FATAL_PLATFORM_STATES | {
"connecting", "disconnected", "disabled", "paused", "retrying", "unknown"
}
_SUPERVISION_MODES = {"systemd", "s6", "container", "launchd", "manual", "unknown"}
_SOURCE_LOGGER_RE = re.compile(r"^gateway(?:\.[A-Za-z_][A-Za-z0-9_]*)*$")
def _allowed_logger(name: str) -> bool:
return name == "gateway" or name.startswith("gateway.")
def source_logger_for_export(name: Any) -> Optional[str]:
"""Return a bounded source-controlled gateway logger name for OTLP scope."""
value = str(name or "")
return value if len(value) <= 128 and _SOURCE_LOGGER_RE.fullmatch(value) else None
def redact_gateway_message(message: Any) -> str:
"""Redact gateway diagnostic free text for operator-owned export.
Single scrub path: everything goes through
``agent.monitoring.redaction.redact_for_export`` (unconditional
secrets + PII), then is length-bounded.
"""
try:
from agent.monitoring.redaction import redact_for_export
redacted = redact_for_export(str(message or "")) or ""
except Exception:
redacted = "[redaction-unavailable]"
return redacted[:500]
def classify_gateway_error(raw: Any) -> str:
s = str(raw or "").lower()
if any(k in s for k in ("auth", "token", "unauthorized", "forbidden", "401", "403")):
return "auth_failed"
if "rate" in s and "limit" in s:
return "rate_limited"
if "timeout" in s or "timed out" in s:
return "timeout"
if any(
k in s
for k in (
"network",
"connection",
"dns",
"socket",
"connect call failed",
"failed to connect",
"cannot connect",
"unreachable",
"name resolution",
)
):
return "network_error"
if any(k in s for k in ("config", "missing", "invalid")):
return "invalid_config"
if "startup" in s:
return "startup_failed"
if "fatal" in s:
return "platform_fatal"
return "unknown"
def classify_exit_reason(
raw: Any, *, state: Any, restart_requested: bool
) -> Optional[str]:
"""Reduce free-form shutdown text to a bounded operational class."""
if restart_requested:
return "restart_requested"
state_name = str(state or "").lower()
if raw is None and state_name != "startup_failed":
return None
classified = classify_gateway_error(raw)
if state_name == "startup_failed":
return classified if classified != "unknown" else "startup_failed"
text = str(raw or "").lower()
if "signal" in text or "sigterm" in text or "sigint" in text:
return "signal"
if state_name == "stopped" and any(word in text for word in ("shutdown", "stop")):
return "planned_stop"
return classified
def _bounded_state(raw: Any, *, allowed: set[str]) -> str:
state = str(raw or "unknown").lower()
return state if state in allowed else "unknown"
def _safe_metric_value(raw: Any, *, limit: int = 128) -> str:
try:
from agent.monitoring.redaction import redact_for_export
value = redact_for_export(str(raw or "")) or "unknown"
except Exception:
return "unknown"
return value[:limit]
def _safe_instance_id(raw: Any) -> str:
"""Return a stable opaque instance key without exporting the source ID."""
value = str(raw or "unknown").encode("utf-8", errors="replace")
return f"sha256:{hashlib.sha256(value).hexdigest()[:24]}"
def subsystem_for_logger(logger_name: str) -> str:
if logger_name == "gateway.relay" or logger_name.startswith("gateway.relay."):
return "platform.relay"
if logger_name.startswith("gateway.platforms."):
parts = logger_name.split(".")
if len(parts) >= 3 and parts[2]:
return f"platform.{parts[2]}"
if logger_name.startswith("gateway.platforms"):
return "platform"
if logger_name.startswith("gateway"):
return "gateway"
return "gateway"
def platform_for_subsystem(subsystem: str) -> Optional[str]:
if subsystem.startswith("platform."):
return subsystem.split(".", 1)[1] or None
return None
def _parse_active_agents(raw: Any) -> int:
try:
from gateway.status import parse_active_agents
return parse_active_agents(raw)
except Exception:
try:
return max(0, int(raw))
except (TypeError, ValueError):
return 0
def _derive_busy(gateway_running: bool, gateway_state: Any, active_agents: Any) -> bool:
try:
from gateway.status import derive_gateway_busy
return derive_gateway_busy(
gateway_running=gateway_running,
gateway_state=gateway_state,
active_agents=active_agents,
)
except Exception:
return bool(gateway_running and gateway_state == "running" and _parse_active_agents(active_agents) > 0)
def _derive_drainable(gateway_running: bool, gateway_state: Any) -> bool:
try:
from gateway.status import derive_gateway_drainable
return derive_gateway_drainable(gateway_running=gateway_running, gateway_state=gateway_state)
except Exception:
return bool(gateway_running and gateway_state == "running")
def _base_attrs(*, profile: str, install_id: str, version: str, supervision_mode: str) -> Dict[str, str]:
mode = str(supervision_mode or "unknown").lower()
return {
"service.instance.id": _safe_instance_id(install_id),
"service.version": _safe_metric_value(version, limit=64),
"hermes.supervision_mode": mode if mode in _SUPERVISION_MODES else "unknown",
}
def _metric(name: str, value: int | float, attrs: Dict[str, str], **extra: str) -> GatewayMetric:
out = dict(attrs)
for key, val in extra.items():
if val is not None:
out[key] = _safe_metric_value(val)
return GatewayMetric(name=name, value=value, attributes=out)
def build_gateway_health_snapshot(
runtime: Optional[dict[str, Any]],
*,
gateway_running: bool,
profile: str,
install_id: str,
version: str,
supervision_mode: str = "unknown",
) -> GatewayHealthSnapshot:
"""Convert gateway_state.json-compatible runtime state into P0 signals."""
runtime = runtime or {}
gateway_state = _bounded_state(
runtime.get("gateway_state"), allowed=_KNOWN_GATEWAY_STATES
)
active_agents = _parse_active_agents(runtime.get("active_agents", 0))
busy = _derive_busy(gateway_running, gateway_state, active_agents)
drainable = _derive_drainable(gateway_running, gateway_state)
platforms = runtime.get("platforms") if isinstance(runtime.get("platforms"), dict) else {}
base = _base_attrs(profile=profile, install_id=install_id, version=version, supervision_mode=supervision_mode)
metrics: list[GatewayMetric] = [
_metric("hermes.gateway.up", 1 if gateway_running else 0, base),
_metric("hermes.gateway.active_agents", active_agents, base),
_metric("hermes.gateway.busy", 1 if busy else 0, base),
_metric("hermes.gateway.drainable", 1 if drainable else 0, base),
_metric("hermes.gateway.restart_requested", 1 if runtime.get("restart_requested") else 0, base),
]
if gateway_state:
metrics.append(_metric("hermes.gateway.state", 1, base, **{"hermes.gateway.state": str(gateway_state)}))
fatal_count = 0
events: list[GatewayHealthEvent | GatewayDiagnosticEvent] = []
for platform, pdata in platforms.items():
pdata = pdata if isinstance(pdata, dict) else {}
state = _bounded_state(
pdata.get("state"), allowed=_KNOWN_PLATFORM_STATES
)
raw_error = pdata.get("error_code") or pdata.get("error_message")
error_code = classify_gateway_error(raw_error)
is_up = state in _RUNNING_PLATFORM_STATES
is_degraded = state in _FATAL_PLATFORM_STATES
if is_degraded:
fatal_count += 1
metrics.append(_metric(
"hermes.platform.up",
1 if is_up else 0,
base,
**{"hermes.platform": str(platform), "hermes.platform.state": state},
))
metrics.append(_metric(
"hermes.platform.degraded",
1 if is_degraded else 0,
base,
**{"hermes.platform": str(platform), "hermes.platform.state": state, "hermes.error_code": error_code},
))
if is_degraded:
events.append(GatewayDiagnosticEvent(
name="platform.fatal",
subsystem=f"platform.{platform}",
platform=str(platform),
error_code=error_code,
error_class=classify_gateway_error(error_code or pdata.get("error_message")),
profile=profile,
version=version,
severity="error" if state == "fatal" else "warning",
))
events.insert(0, GatewayHealthEvent(
name="gateway.health_snapshot",
gateway_state=str(gateway_state) if gateway_state is not None else None,
active_agents=active_agents,
gateway_busy=busy,
gateway_drainable=drainable,
platform_count=len(platforms),
fatal_platform_count=fatal_count,
profile=profile,
install_id=install_id,
version=version,
supervision_mode=supervision_mode,
pid=_coerce_pid(runtime.get("pid")),
))
return GatewayHealthSnapshot(metrics=metrics, events=events)
def _safe_profile() -> str:
try:
from hermes_cli.profiles import get_active_profile_name
return str(get_active_profile_name() or "default")
except Exception:
return "default"
def _safe_version() -> str:
try:
from hermes_cli import __version__
return str(__version__)
except Exception:
return "unknown"
def emit_runtime_status_transition(previous: Optional[dict[str, Any]], current: dict[str, Any]) -> None:
"""Emit immediate content-free gateway events for runtime status changes.
Called by gateway.status.write_runtime_status after persisting the new status.
Fully fail-open: failures never affect gateway status writes.
"""
try:
from agent.monitoring import emitter
out: list[GatewayHealthEvent | GatewayDiagnosticEvent] = []
profile = _safe_profile()
version = _safe_version()
old_gateway_state = _bounded_state(
(previous or {}).get("gateway_state"), allowed=_KNOWN_GATEWAY_STATES
) if (previous or {}).get("gateway_state") is not None else None
new_gateway_state = _bounded_state(
current.get("gateway_state"), allowed=_KNOWN_GATEWAY_STATES
) if current.get("gateway_state") is not None else None
if old_gateway_state != new_gateway_state and new_gateway_state:
out.append(GatewayHealthEvent(
name="gateway.lifecycle",
gateway_state=new_gateway_state,
old_state=old_gateway_state,
new_state=new_gateway_state,
exit_reason=classify_exit_reason(
current.get("exit_reason"),
state=new_gateway_state,
restart_requested=bool(current.get("restart_requested")),
),
restart_requested=bool(current.get("restart_requested")),
active_agents=_parse_active_agents(current.get("active_agents", 0)),
profile=profile,
version=version,
pid=_coerce_pid(current.get("pid")),
))
if new_gateway_state == "startup_failed":
out.append(GatewayDiagnosticEvent(
name="gateway.startup_failed",
subsystem="gateway",
error_class=classify_gateway_error(current.get("exit_reason") or "startup_failed"),
error_code=classify_gateway_error(current.get("exit_reason") or "startup_failed"),
profile=profile,
version=version,
severity="error",
))
if new_gateway_state == "stopped":
out.append(GatewayHealthEvent(
name="gateway.exit",
gateway_state=new_gateway_state,
old_state=old_gateway_state,
new_state=new_gateway_state,
exit_reason=classify_exit_reason(
current.get("exit_reason"),
state=new_gateway_state,
restart_requested=bool(current.get("restart_requested")),
),
restart_requested=bool(current.get("restart_requested")),
active_agents=_parse_active_agents(current.get("active_agents", 0)),
profile=profile,
version=version,
pid=_coerce_pid(current.get("pid")),
))
old_platforms_raw = (previous or {}).get("platforms")
new_platforms_raw = current.get("platforms")
old_platforms = old_platforms_raw if isinstance(old_platforms_raw, dict) else {}
new_platforms = new_platforms_raw if isinstance(new_platforms_raw, dict) else {}
for platform, pdata in new_platforms.items():
pdata = pdata if isinstance(pdata, dict) else {}
prev_raw = old_platforms.get(platform, {})
prev = prev_raw if isinstance(prev_raw, dict) else {}
old_state = _bounded_state(
prev.get("state"), allowed=_KNOWN_PLATFORM_STATES
) if prev.get("state") is not None else None
new_state = _bounded_state(
pdata.get("state"), allowed=_KNOWN_PLATFORM_STATES
) if pdata.get("state") is not None else None
if old_state == new_state or not new_state:
continue
error_code = classify_gateway_error(pdata.get("error_code") or pdata.get("error_message"))
severity = "error" if new_state.lower() in {"fatal", "failed", "error"} else "warning"
out.append(GatewayDiagnosticEvent(
name="platform.state_change",
subsystem=f"platform.{platform}",
platform=str(platform),
old_state=old_state,
new_state=new_state,
error_code=error_code,
error_class=error_code,
profile=profile,
version=version,
severity=severity,
))
if new_state.lower() in _FATAL_PLATFORM_STATES:
out.append(GatewayDiagnosticEvent(
name="platform.fatal",
subsystem=f"platform.{platform}",
platform=str(platform),
error_code=error_code,
error_class=error_code,
profile=profile,
version=version,
severity=severity,
))
for ev in out:
emitter.emit(ev)
except Exception:
logging.getLogger(__name__).debug("gateway runtime status transition emit failed", exc_info=True)
def _coerce_pid(raw: Any) -> Optional[int]:
try:
pid = int(raw)
except (TypeError, ValueError):
return None
return pid if pid > 0 else None
class GatewayDiagnosticLogHandler(logging.Handler):
"""Allowlisted warning/error bridge for gateway-owned diagnostics."""
def __init__(self, *, profile: str = "default", version: str = "unknown") -> None:
super().__init__(level=logging.WARNING)
self.profile = profile
self.version = version
def emit(self, record: logging.LogRecord) -> None:
try:
if record.levelno < logging.WARNING:
return
if not _allowed_logger(record.name):
return
subsystem = subsystem_for_logger(record.name)
message = record.getMessage()
error_class = classify_gateway_error(message)
event = GatewayDiagnosticEvent(
name=f"gateway.log.{record.levelname.lower()}",
subsystem=subsystem,
source_logger=source_logger_for_export(record.name),
platform=platform_for_subsystem(subsystem),
error_class=error_class,
error_code=error_class,
profile=self.profile,
version=self.version,
severity=record.levelname.lower(),
)
from agent.monitoring import emitter
emitter.get_emitter().emit(event)
except Exception:
logging.getLogger(__name__).debug("gateway diagnostic emit failed", exc_info=True)
__all__ = [
"GatewayMetric",
"GatewayHealthSnapshot",
"GatewayDiagnosticLogHandler",
"build_gateway_health_snapshot",
"classify_gateway_error",
"source_logger_for_export",
"redact_gateway_message",
]
+643
View File
@@ -0,0 +1,643 @@
"""Gateway Health & Diagnostics OTLP export runtime.
This exporter emits operator-owned gateway service-health metrics plus
narrow redacted diagnostic events. It is deliberately in-process and fail-open so
it works under systemd, launchd, s6, containers, tmux, nohup, or a simple shell
without a sidecar/watchdog dependency.
"""
from __future__ import annotations
import logging
import os
import re
import threading
from dataclasses import dataclass
from typing import Any, Dict, Optional
logger = logging.getLogger(__name__)
_DEFAULT_DIAGNOSTIC_SCOPE = "hermes.gateway.diagnostics"
_RESOURCE_ATTRIBUTE_KEYS = frozenset({
"service.name",
"service.namespace",
"service.version",
"service.instance.id",
"deployment.environment.name",
"cloud.provider",
"cloud.platform",
"cloud.region",
"telemetry.scope",
})
_DIAGNOSTIC_ATTRIBUTE_KEYS = frozenset({
"name",
"subsystem",
"error_class",
"error_code",
"platform",
"old_state",
"new_state",
"version",
"severity",
})
_SAFE_RESOURCE_VALUE = re.compile(r"^[A-Za-z0-9._:/-]{1,128}$")
def _redact_string(raw: Any, *, limit: int = 500) -> str:
try:
from agent.monitoring.redaction import redact_for_export
return (redact_for_export(str(raw or "")) or "[redacted]")[:limit]
except Exception:
return "[redaction-unavailable]"
def _safe_resource_attributes(raw: Any) -> Dict[str, str]:
"""Allowlist bounded resource labels and reject values changed by redaction."""
attrs: Dict[str, str] = {}
if not isinstance(raw, dict):
return attrs
for key, value in raw.items():
key = str(key)
if key not in _RESOURCE_ATTRIBUTE_KEYS or value is None:
continue
if key == "service.instance.id":
from agent.monitoring.gateway_health import _safe_instance_id
attrs[key] = _safe_instance_id(value)
continue
text = str(value)
if not _SAFE_RESOURCE_VALUE.fullmatch(text):
continue
if _redact_string(text, limit=128) != text:
continue
attrs[key] = text
return attrs
def _runtime_resource_attributes(
config: Dict[str, Any], *, telemetry_scope: str
) -> Dict[str, str]:
"""Build the safe OTLP resource shared by metrics and diagnostic logs."""
gh = _gateway_health_config(config)
attrs = _safe_resource_attributes(gh.get("resource_attributes"))
from agent.monitoring.gateway_health import _safe_instance_id
attrs["service.name"] = "hermes-gateway"
attrs["service.instance.id"] = _safe_instance_id(_install_id(config))
attrs["telemetry.scope"] = telemetry_scope
return attrs
def _diagnostic_log_attributes(event: Dict[str, Any]) -> Dict[str, Any]:
attrs: Dict[str, Any] = {}
for key in _DIAGNOSTIC_ATTRIBUTE_KEYS:
value = event.get(key)
if value is None:
continue
attrs[f"hermes.{key}"] = _redact_string(value) if isinstance(value, str) else value
return attrs
@dataclass(slots=True)
class GatewayHealthExportRuntime:
enabled: bool
reason: str = "disabled"
streamer: Any = None
metric_provider: Any = None
log_handler: Any = None
log_streamer: Any = None
thread: Optional[threading.Thread] = None
stop_event: Optional[threading.Event] = None
def shutdown(self) -> None:
if self.stop_event is not None:
self.stop_event.set()
if self.thread is not None:
self.thread.join(timeout=0.25)
if self.log_handler is not None:
try:
logging.getLogger().removeHandler(self.log_handler)
except Exception:
pass
# All producers above are now stopped. Drain queued and in-flight
# events before detaching subscribers so the terminal lifecycle event
# cannot race exporter shutdown. The barrier is bounded and fail-open.
try:
from agent.monitoring.emitter import get_emitter
emitter = get_emitter()
emitter.flush(timeout=1.0)
if self.streamer is not None:
emitter.unsubscribe(self.streamer)
if self.log_streamer is not None:
emitter.unsubscribe(self.log_streamer)
except Exception:
pass
# Network flush/close runs under one bounded daemon-thread deadline and
# can never delay gateway teardown indefinitely.
closeables = [
item for item in (self.streamer, self.log_streamer, self.metric_provider)
if item is not None
]
def _close() -> None:
for item in closeables:
try:
item.shutdown()
except Exception:
pass
if closeables:
worker = threading.Thread(
target=_close,
name="hermes-gateway-health-export-shutdown",
daemon=True,
)
worker.start()
worker.join(timeout=2.0)
self.streamer = None
self.log_streamer = None
self.metric_provider = None
self.thread = None
self.stop_event = None
def _gateway_health_config(config: Dict[str, Any]) -> Dict[str, Any]:
mon = (config or {}).get("monitoring") or {}
return mon.get("gateway_health_export") or {}
def _otlp_config(config: Dict[str, Any]) -> Dict[str, Any]:
mon = (config or {}).get("monitoring") or {}
export = mon.get("export") or {}
return export.get("otlp") or {}
def _enabled(config: Dict[str, Any]) -> bool:
gh = _gateway_health_config(config)
otlp = _otlp_config(config)
return bool(gh.get("enabled") and otlp.get("enabled") and otlp.get("endpoint"))
def _require_metrics_sdk(*, auto_install: bool = True, prompt: bool = False) -> Dict[str, Any]:
if auto_install:
try:
from tools.lazy_deps import ensure as _lazy_ensure
_lazy_ensure("export.otlp", prompt=prompt)
except Exception:
pass
try:
from opentelemetry.exporter.otlp.proto.http._log_exporter import OTLPLogExporter
from opentelemetry.exporter.otlp.proto.http.metric_exporter import OTLPMetricExporter
from opentelemetry.metrics import Observation
from opentelemetry.trace import INVALID_SPAN_ID, INVALID_TRACE_ID, TraceFlags
from opentelemetry._logs import LogRecord
from opentelemetry._logs.severity import SeverityNumber
from opentelemetry.sdk._logs import LoggerProvider
from opentelemetry.sdk._logs.export import BatchLogRecordProcessor
from opentelemetry.sdk.metrics import MeterProvider
from opentelemetry.sdk.metrics.export import PeriodicExportingMetricReader
from opentelemetry.sdk.resources import Resource
return {
"OTLPLogExporter": OTLPLogExporter,
"OTLPMetricExporter": OTLPMetricExporter,
"Observation": Observation,
"LogRecord": LogRecord,
"LoggerProvider": LoggerProvider,
"INVALID_SPAN_ID": INVALID_SPAN_ID,
"INVALID_TRACE_ID": INVALID_TRACE_ID,
"TraceFlags": TraceFlags,
"SeverityNumber": SeverityNumber,
"BatchLogRecordProcessor": BatchLogRecordProcessor,
"MeterProvider": MeterProvider,
"PeriodicExportingMetricReader": PeriodicExportingMetricReader,
"Resource": Resource,
}
except Exception as exc:
raise RuntimeError(f"OTLP metrics SDK unavailable: {exc}") from exc
def _resolve_headers(headers_env: Optional[Dict[str, str]]) -> Dict[str, str]:
resolved: Dict[str, str] = {}
for header_name, env_name in (headers_env or {}).items():
val = os.environ.get(str(env_name))
if val:
resolved[str(header_name)] = val
return resolved
def _metric_endpoint(endpoint: str) -> str:
if endpoint.endswith("/v1/traces"):
return endpoint[: -len("/v1/traces")] + "/v1/metrics"
return endpoint
def _logs_endpoint(endpoint: str) -> str:
if endpoint.endswith("/v1/traces"):
return endpoint[: -len("/v1/traces")] + "/v1/logs"
if endpoint.endswith("/v1/metrics"):
return endpoint[: -len("/v1/metrics")] + "/v1/logs"
return endpoint
def _version() -> str:
try:
from hermes_cli import __version__
return str(__version__)
except Exception:
return "unknown"
def _profile() -> str:
try:
from hermes_cli.profiles import get_active_profile_name
return str(get_active_profile_name() or "default")
except Exception:
return "default"
def _install_id(config: Dict[str, Any]) -> str:
try:
from agent.monitoring.policy import ensure_install_id
return str(ensure_install_id(config))
except Exception:
return "unknown"
def _supervision_mode() -> str:
if os.environ.get("INVOCATION_ID"):
return "systemd"
if os.environ.get("S6_CMD_ARG0") or os.environ.get("S6_VERSION"):
return "s6"
if os.environ.get("container") or os.path.exists("/.dockerenv"):
return "container"
if os.environ.get("LAUNCHD_SOCKET"):
return "launchd"
return "manual"
def _read_gateway_snapshot(config: Dict[str, Any]):
from agent.monitoring.gateway_health import build_gateway_health_snapshot
try:
from gateway.status import read_runtime_status
runtime = read_runtime_status() or {}
except Exception:
runtime = {}
return build_gateway_health_snapshot(
runtime,
gateway_running=True,
profile=_profile(),
install_id=_install_id(config),
version=_version(),
supervision_mode=_supervision_mode(),
)
def _read_cron_snapshot():
from agent.monitoring.cron_health import build_cron_health_snapshot
return build_cron_health_snapshot()
def _read_background_work_count() -> int:
"""Count live background/subagent work that ``active_agents`` does NOT include.
``hermes.gateway.active_agents`` counts foreground turns + in-flight cron
jobs + API runs, but deliberately excludes backgrounded ``delegate_task``
subagents, ``terminal(background=true)`` processes, kanban workers, and the
runner's own background tasks (they are tracked only for the scale-to-zero
suspend guard, ``_scale_to_zero_has_live_background_work``). Without this
metric a peer churning through delegated subagents shows ``active_agents=0``
on the fleet dashboard. Best-effort and content-free: a single integer,
no job/task identity. Returns 0 if a source can't be imported.
Delegation is counted TASK-granular (``active_task_count``): a fan-out batch
of N subagents contributes N, not 1, so the metric reflects real concurrent
subagent load rather than dispatch-unit/pool-slot count. This intentionally
differs from the async pool's capacity accounting (one batch = one slot).
"""
total = 0
try:
from tools.async_delegation import active_task_count
total += max(0, int(active_task_count()))
except Exception:
logger.debug("background-work async-delegation count failed", exc_info=True)
try:
from tools.process_registry import process_registry
total += max(0, int(process_registry.count_running()))
except Exception:
logger.debug("background-work process-registry count failed", exc_info=True)
return total
def _read_background_delegations_count() -> int:
"""Count live async delegation UNITS (dispatch/pool slots).
Complements ``_read_background_work_count`` (which is task-granular): this
counts each ``delegate_task`` dispatch as ONE regardless of fan-out width,
matching the async pool's capacity accounting (a batch = one slot). Together
the two metrics let an operator see both slot pressure
(``background_delegations``, alert vs ``max_concurrent_children``) and real
concurrent subagent load (``background_work``). Delegations only — it does
not include ``terminal(background)`` / kanban work, which are already folded
into ``background_work``. Best-effort; 0 if the source can't be imported.
"""
try:
from tools.async_delegation import active_count
return max(0, int(active_count()))
except Exception:
logger.debug("background-delegations count failed", exc_info=True)
return 0
def _read_runtime_snapshot(config: Dict[str, Any]):
gateway_snapshot = _read_gateway_snapshot(config)
# Background/subagent work — a distinct metric from active_agents (which
# never counts it). Appended to the gateway snapshot so it rides the same
# base resource attributes (service.instance.id etc.).
try:
from agent.monitoring.gateway_health import GatewayMetric
base = dict(gateway_snapshot.metrics[0].attributes) if gateway_snapshot.metrics else {}
gateway_snapshot.metrics.append(
GatewayMetric(
name="hermes.gateway.background_work",
value=_read_background_work_count(),
attributes=base,
)
)
gateway_snapshot.metrics.append(
GatewayMetric(
name="hermes.gateway.background_delegations",
value=_read_background_delegations_count(),
attributes=base,
)
)
except Exception as exc:
logger.warning(
"background-work snapshot unavailable; metric not exported (error_type=%s)",
type(exc).__name__,
)
logger.debug("background-work snapshot traceback", exc_info=True)
try:
cron_snapshot = _read_cron_snapshot()
except Exception as exc:
# Content-free visibility: cron telemetry silently dropping out is a
# release-relevant regression, so surface it at WARNING with only the
# exception *type* name (never the message, which could carry paths or
# other environment detail). exc_info stays on the DEBUG record.
logger.warning(
"cron health snapshot unavailable; cron telemetry not exported (error_type=%s)",
type(exc).__name__,
)
logger.debug("cron health snapshot traceback", exc_info=True)
return gateway_snapshot
gateway_snapshot.metrics.extend(cron_snapshot.metrics)
return gateway_snapshot
def _emit_snapshot_events(config: Dict[str, Any]) -> None:
gh = _gateway_health_config(config)
if not gh.get("diagnostic_events_enabled", True):
return
try:
from agent.monitoring import emitter
snapshot = _read_runtime_snapshot(config)
for event in snapshot.events:
emitter.emit(event)
except Exception:
logger.debug("gateway health snapshot emit failed", exc_info=True)
def _start_metric_provider(config: Dict[str, Any], sdk: Dict[str, Any]) -> Any:
gh = _gateway_health_config(config)
if not gh.get("metrics_enabled", True):
return None
otlp = _otlp_config(config)
endpoint = _metric_endpoint(str(otlp.get("endpoint")))
headers = _resolve_headers(otlp.get("headers_env"))
exporter = sdk["OTLPMetricExporter"](endpoint=endpoint, headers=headers or None)
interval_ms = max(5, int(gh.get("export_interval_seconds", 60))) * 1000
reader = sdk["PeriodicExportingMetricReader"](exporter, export_interval_millis=interval_ms)
resource_attrs = _runtime_resource_attributes(
config, telemetry_scope="gateway_health"
)
provider = sdk["MeterProvider"](
metric_readers=[reader],
resource=sdk["Resource"].create(resource_attrs),
)
meter = provider.get_meter("hermes.gateway.health")
Observation = sdk["Observation"]
metric_names = [
"hermes.gateway.up",
"hermes.gateway.state",
"hermes.gateway.active_agents",
"hermes.gateway.busy",
"hermes.gateway.drainable",
"hermes.gateway.restart_requested",
"hermes.gateway.background_work",
"hermes.gateway.background_delegations",
"hermes.platform.up",
"hermes.platform.degraded",
"hermes.cron.scheduler.heartbeat_age_seconds",
"hermes.cron.scheduler.last_success_age_seconds",
"hermes.cron.scheduler.catch_up_occurrences",
"hermes.cron.jobs.enabled",
"hermes.cron.jobs.running",
"hermes.cron.jobs.overdue",
]
def callback(name: str):
def _cb(_options=None):
try:
snapshot = _read_runtime_snapshot(config)
return [Observation(m.value, m.attributes) for m in snapshot.metrics if m.name == name]
except Exception:
logger.debug("gateway metric callback failed", exc_info=True)
return []
return _cb
for metric_name in metric_names:
meter.create_observable_gauge(metric_name, callbacks=[callback(metric_name)])
return provider
def _severity_number(sdk: Dict[str, Any], severity: Any) -> Any:
SeverityNumber = sdk["SeverityNumber"]
sev = str(severity or "warning").lower()
if sev in {"critical", "fatal"}:
return SeverityNumber.FATAL
if sev == "error":
return SeverityNumber.ERROR
if sev in {"info", "information"}:
return SeverityNumber.INFO
if sev == "debug":
return SeverityNumber.DEBUG
return SeverityNumber.WARN
class GatewayDiagnosticLogStreamer:
"""Emitter subscriber that sends gateway diagnostic events as OTLP logs."""
def __init__(self, config: Dict[str, Any], sdk: Dict[str, Any]):
otlp = _otlp_config(config)
headers = _resolve_headers(otlp.get("headers_env"))
endpoint = _logs_endpoint(str(otlp.get("endpoint")))
resource_attrs = _runtime_resource_attributes(
config, telemetry_scope="gateway_diagnostics"
)
self._provider = sdk["LoggerProvider"](resource=sdk["Resource"].create(resource_attrs))
self._processor = sdk["BatchLogRecordProcessor"](
sdk["OTLPLogExporter"](endpoint=endpoint, headers=headers or None)
)
self._provider.add_log_record_processor(self._processor)
self._logger = self._provider.get_logger(_DEFAULT_DIAGNOSTIC_SCOPE)
self._LogRecord = sdk["LogRecord"]
self._sdk = sdk
self.exported = 0
def __call__(self, batch: list[Dict[str, Any]]) -> None:
from agent.monitoring.gateway_health import source_logger_for_export
for ev in batch:
if ev.get("event") != "gateway_diagnostic":
continue
attrs = _diagnostic_log_attributes(ev)
# Preserve the source-controlled Python logger as the OTel
# instrumentation scope. This adds precise code attribution without
# turning a fluid module layout into a maintained subsystem enum.
# Rendered messages stay out because they may contain arbitrary IDs,
# names, paths, or configured strings. A future, separately gated
# ``diagnostic_detail: redacted_message`` mode may add best-effort
# free text when an observability plane defines that privacy policy.
source_logger = source_logger_for_export(ev.get("source_logger"))
otel_logger = (
self._provider.get_logger(source_logger)
if source_logger is not None
else self._logger
)
body = "gateway diagnostic"
record = self._LogRecord(
timestamp=ev.get("ts_ns"),
trace_id=self._sdk["INVALID_TRACE_ID"],
span_id=self._sdk["INVALID_SPAN_ID"],
trace_flags=self._sdk["TraceFlags"].DEFAULT,
severity_text=str(ev.get("severity") or "warning").upper(),
severity_number=_severity_number(self._sdk, ev.get("severity")),
body=_redact_string(body),
attributes=attrs,
)
otel_logger.emit(record)
self.exported += 1
def shutdown(self) -> None:
try:
from agent.monitoring.emitter import get_emitter
get_emitter().unsubscribe(self)
except Exception:
pass
try:
self._processor.force_flush()
self._provider.shutdown()
except Exception:
pass
def _start_diagnostic_log_streamer(config: Dict[str, Any], sdk: Dict[str, Any]) -> GatewayDiagnosticLogStreamer:
from agent.monitoring.emitter import get_emitter
streamer = GatewayDiagnosticLogStreamer(config, sdk)
get_emitter().subscribe(streamer)
return streamer
def _start_snapshot_thread(config: Dict[str, Any], stop_event: threading.Event) -> threading.Thread:
interval = max(5, int(_gateway_health_config(config).get("logs_export_interval_seconds", 5)))
def _run() -> None:
while not stop_event.wait(interval):
_emit_snapshot_events(config)
thread = threading.Thread(target=_run, name="hermes-gateway-health-export", daemon=True)
thread.start()
return thread
def _attach_log_handler(config: Dict[str, Any]) -> Any:
gh = _gateway_health_config(config)
if not gh.get("diagnostic_events_enabled", True) or not gh.get("warning_error_events_enabled", True):
return None
from agent.monitoring.gateway_health import GatewayDiagnosticLogHandler
handler = GatewayDiagnosticLogHandler(profile=_profile(), version=_version())
root = logging.getLogger()
if handler not in root.handlers:
root.addHandler(handler)
return handler
def _gateway_health_event(ev: Dict[str, Any]) -> bool:
return ev.get("event") in {"gateway_health", "cron_execution"}
def start_gateway_health_export(config: Dict[str, Any]) -> GatewayHealthExportRuntime:
"""Start P0 gateway health export if configured. Never raises."""
if not _enabled(config):
return GatewayHealthExportRuntime(enabled=False, reason="disabled")
gh = _gateway_health_config(config)
runtime = GatewayHealthExportRuntime(enabled=True, reason="enabled")
sdk: Optional[Dict[str, Any]] = None
if gh.get("metrics_enabled", True) or gh.get("diagnostic_events_enabled", True):
try:
sdk = _require_metrics_sdk(prompt=False)
except Exception:
logger.warning(
"monitoring.gateway_health_export.enabled but OTLP SDK is unavailable; "
"install 'hermes-agent[otlp]'",
exc_info=True,
)
return GatewayHealthExportRuntime(enabled=False, reason="otlp_unavailable")
if gh.get("metrics_enabled", True) and sdk is not None:
try:
runtime.metric_provider = _start_metric_provider(config, sdk)
except Exception:
logger.warning("gateway health OTLP metrics failed to start", exc_info=True)
runtime.shutdown()
return GatewayHealthExportRuntime(enabled=False, reason="metrics_start_failed")
if gh.get("diagnostic_events_enabled", True) and sdk is not None:
try:
from agent.monitoring import otlp_exporter
runtime.streamer = otlp_exporter.start_streaming(config, event_filter=_gateway_health_event)
if runtime.streamer is None:
raise RuntimeError("gateway health span streamer did not start")
runtime.log_streamer = _start_diagnostic_log_streamer(config, sdk)
except Exception:
logger.debug("gateway diagnostic OTLP export failed to start", exc_info=True)
runtime.shutdown()
return GatewayHealthExportRuntime(enabled=False, reason="diagnostics_start_failed")
try:
runtime.log_handler = _attach_log_handler(config)
except Exception:
logger.debug("gateway diagnostic log handler failed to attach", exc_info=True)
if gh.get("diagnostic_events_enabled", True):
try:
_emit_snapshot_events(config)
runtime.stop_event = threading.Event()
runtime.thread = _start_snapshot_thread(config, runtime.stop_event)
except Exception:
logger.debug("gateway health snapshot thread failed to start", exc_info=True)
return runtime
__all__ = [
"GatewayHealthExportRuntime",
"start_gateway_health_export",
]
+272
View File
@@ -0,0 +1,272 @@
"""Export monitoring events to an OpenTelemetry Collector over OTLP/HTTP.
Maps gateway monitoring events to OTel spans and sends them to the endpoint
configured under ``monitoring.export.otlp``. Lets an operator stream Hermes
gateway health into their own observability stack (OTEL Collector, DataDog,
and similar).
Notes:
* The destination is operator-configured; this module only sends to that
endpoint. No default destination ships.
* ``opentelemetry-sdk`` + ``opentelemetry-exporter-otlp-proto-http`` are an
optional extra (``pip install hermes-agent[otlp]``), imported lazily so the
dependency is only required when OTLP export is actually used.
* ``headers_env`` maps a header name to an environment variable name; values
are read from the environment at export time and never logged or stored.
* The continuous subscriber runs in the emitter's dispatcher thread and is
fail-isolated, so an export error cannot affect the gateway.
Only monitoring events (gateway_health / gateway_diagnostic) exist on this
plane; the ``event_filter`` seam is kept so future planes sharing the emitter
cannot silently ride along on this exporter.
"""
from __future__ import annotations
import logging
import os
from typing import Any, Callable, Dict, List, Optional
logger = logging.getLogger(__name__)
class OTLPUnavailable(RuntimeError):
"""Raised when the optional OpenTelemetry SDK isn't installed."""
def _require_sdk(*, auto_install: bool = True, prompt: bool = True):
"""Import the OTel SDK, lazily installing it on first use if needed.
Routes through tools.lazy_deps (feature 'export.otlp') so a missing SDK
triggers the standard venv install flow — same as every other optional
backend — gated by security.allow_lazy_installs and TTY-prompted. Falls back
to OTLPUnavailable (with a manual install hint) when the SDK can't be made
importable (lazy installs disabled, install failed, or auto_install=False).
``auto_install``: attempt the lazy install when missing (default True).
``prompt``: ask before installing when interactive (default True); pass
False from non-interactive contexts like the continuous streamer.
"""
if auto_install:
try:
from tools.lazy_deps import ensure as _lazy_ensure
_lazy_ensure("export.otlp", prompt=prompt)
except ImportError:
pass # lazy_deps unavailable — fall through to the import attempt
except Exception:
# FeatureUnavailable (lazy installs disabled / declined / failed) —
# fall through; the import below raises OTLPUnavailable with the hint.
pass
try:
from opentelemetry.sdk.trace import TracerProvider
from opentelemetry.sdk.trace.export import BatchSpanProcessor
from opentelemetry.sdk.resources import Resource
from opentelemetry.exporter.otlp.proto.http.trace_exporter import (
OTLPSpanExporter,
)
from opentelemetry.trace import SpanKind
return {
"TracerProvider": TracerProvider,
"BatchSpanProcessor": BatchSpanProcessor,
"Resource": Resource,
"OTLPSpanExporter": OTLPSpanExporter,
"SpanKind": SpanKind,
}
except Exception as e: # ImportError or partial install
raise OTLPUnavailable(
"OTLP export requires the optional dependency. Install with:\n"
" pip install 'hermes-agent[otlp]'\n"
f"(import error: {e})"
)
def _resolve_headers(headers_env: Optional[Dict[str, str]]) -> Dict[str, str]:
"""Resolve {header_name: ENV_VAR_NAME} -> {header_name: value} from env.
The config stores environment variable names, not secret values; values are
read from the environment here. Missing variables are skipped (and noted at
debug level without the value).
"""
resolved: Dict[str, str] = {}
for header_name, env_name in (headers_env or {}).items():
val = os.environ.get(str(env_name))
if val:
resolved[str(header_name)] = val
else:
logger.debug("OTLP header %s: env var %s not set; skipping",
header_name, env_name)
return resolved
def _otlp_config(config: Dict[str, Any]) -> Dict[str, Any]:
mon = (config or {}).get("monitoring") or {}
export = mon.get("export") or {}
return export.get("otlp") or {}
def build_exporter(config: Dict[str, Any]):
"""Construct an OTLP span exporter from config. Raises OTLPUnavailable if no SDK."""
sdk = _require_sdk()
otlp = _otlp_config(config)
endpoint = otlp.get("endpoint")
if not endpoint:
raise ValueError("monitoring.export.otlp.endpoint is not set")
headers = _resolve_headers(otlp.get("headers_env"))
return sdk["OTLPSpanExporter"](endpoint=endpoint, headers=headers or None)
def _resource_attributes(config: Dict[str, Any]) -> Dict[str, str]:
from agent.monitoring.gateway_health import _safe_instance_id
from agent.monitoring.policy import ensure_install_id
return {
"service.name": "hermes-gateway",
"service.instance.id": _safe_instance_id(ensure_install_id(config)),
"telemetry.scope": "gateway_monitoring",
}
def _make_provider(config: Dict[str, Any]):
sdk = _require_sdk()
resource = sdk["Resource"].create(_resource_attributes(config))
provider = sdk["TracerProvider"](resource=resource)
processor = sdk["BatchSpanProcessor"](build_exporter(config))
provider.add_span_processor(processor)
return provider, processor
# ── event -> span attribute mapping ──────────────────────────────────────────
def _span_attrs(ev: Dict[str, Any]) -> Dict[str, Any]:
"""Span attributes for a monitoring event (content-free by construction)."""
kind = ev.get("event")
attrs: Dict[str, Any] = {"hermes.event": kind or "unknown"}
keep_by_kind = {
"gateway_health": ("name", "gateway_state", "old_state", "new_state",
"exit_reason", "restart_requested", "active_agents",
"gateway_busy", "gateway_drainable", "platform_count",
"fatal_platform_count", "version",
"supervision_mode", "pid"),
"gateway_diagnostic": ("name", "subsystem", "error_class", "error_code",
"platform", "old_state", "new_state",
"version", "severity"),
"cron_execution": ("status", "job_key", "source", "duration_ms",
"delivery_outcome", "error_class"),
}
for col in keep_by_kind.get(kind, ()): # type: ignore[arg-type]
v = ev.get(col)
if v is not None:
if isinstance(v, str):
try:
from agent.monitoring.redaction import redact_for_export
v = (redact_for_export(v) or "[redacted]")[:500]
except Exception:
v = "[redaction-unavailable]"
attrs[f"hermes.{col}"] = v
return attrs
def export_batch(provider, batch: List[Dict[str, Any]]) -> int:
"""Map a batch of events to OTel spans. Returns spans created."""
tracer = provider.get_tracer("hermes.monitoring")
n = 0
for ev in batch:
try:
name = f"hermes.{ev.get('event', 'event')}"
span = tracer.start_span(name, attributes=_span_attrs(ev))
span.end()
n += 1
except Exception:
logger.debug("OTLP span map failed", exc_info=True)
return n
# ── continuous streaming subscriber ─────────────────────────────────────────
class OTLPStreamer:
"""A live subscriber that pushes each emitter batch to OTLP as it lands.
Register with ``emitter.subscribe(streamer)``. Fail-isolated by the emitter.
"""
def __init__(
self,
config: Dict[str, Any],
*,
event_filter: Optional[Callable[[Dict[str, Any]], bool]] = None,
):
self._provider, self._processor = _make_provider(config)
self._event_filter = event_filter
self.exported = 0
def __call__(self, batch: List[Dict[str, Any]]) -> None:
if self._event_filter is not None:
batch = [ev for ev in batch if self._event_filter(ev)]
if not batch:
return
self.exported += export_batch(self._provider, batch)
def shutdown(self) -> None:
try:
from agent.monitoring.emitter import get_emitter
get_emitter().unsubscribe(self)
except Exception:
pass
try:
self._processor.force_flush()
self._provider.shutdown()
except Exception:
pass
def is_available() -> bool:
"""True when the OTel SDK is already importable. Does NOT auto-install —
this is a pure check (e.g. for status display)."""
try:
_require_sdk(auto_install=False)
return True
except OTLPUnavailable:
return False
def is_enabled(config: Dict[str, Any]) -> bool:
otlp = _otlp_config(config)
return bool(otlp.get("enabled") and otlp.get("endpoint"))
def start_streaming(
config: Dict[str, Any],
*,
event_filter: Optional[Callable[[Dict[str, Any]], bool]] = None,
) -> Optional[OTLPStreamer]:
"""If OTLP is enabled, attach a streamer to the singleton emitter.
``event_filter`` scopes the exporter to its plane, e.g. gateway-health
export, so enabling one plane cannot silently export unrelated events.
Non-interactive context (startup): attempts a lazy install with prompt=False
so a configured-but-missing SDK is installed once (gated by
security.allow_lazy_installs), then streams. If it still can't load, logs and
no-ops — never blocks or raises into startup.
"""
if not is_enabled(config):
return None
try:
_require_sdk(prompt=False)
except OTLPUnavailable:
logger.warning("monitoring.export.otlp.enabled but the OTel SDK could not "
"be installed/imported; install 'hermes-agent[otlp]'")
return None
from agent.monitoring.emitter import get_emitter
streamer = OTLPStreamer(config, event_filter=event_filter)
get_emitter().subscribe(streamer)
return streamer
__all__ = [
"OTLPUnavailable",
"OTLPStreamer",
"build_exporter",
"export_batch",
"is_available",
"is_enabled",
"start_streaming",
]
+57
View File
@@ -0,0 +1,57 @@
"""Install identity for gateway monitoring.
The install id is a stable, resettable pseudonymous identifier attached to
exported health signals so an operator can tell instances apart in their
collector. It carries no account identity and can be rotated by clearing
``monitoring.install_id`` in config.
"""
from __future__ import annotations
import logging
import uuid
from typing import Any, Dict
logger = logging.getLogger(__name__)
def ensure_install_id(config: Dict[str, Any]) -> str:
"""Return a stable install id, minting and persisting one when empty.
The id must survive gateway restarts (it becomes ``service.instance.id``
on exported signals), so a freshly minted UUID is written back to
config.yaml immediately. The write is fail-open: if persisting fails
(read-only home, managed scope), the ephemeral id is still returned and
a new one is minted next start.
Clearing ``monitoring.install_id`` (e.g. ``hermes config set
monitoring.install_id ""``) rotates the id on the next gateway start.
"""
mon = config.get("monitoring") if isinstance(config, dict) else None
existing = (mon or {}).get("install_id") if isinstance(mon, dict) else None
if isinstance(existing, str) and existing.strip():
return existing
minted = str(uuid.uuid4())
try:
from hermes_cli.config import load_config, save_config
fresh = load_config()
if isinstance(fresh, dict):
slot = fresh.setdefault("monitoring", {})
if isinstance(slot, dict) and not str(slot.get("install_id") or "").strip():
slot["install_id"] = minted
save_config(fresh)
except Exception:
logger.debug("install_id persist failed; using ephemeral id", exc_info=True)
# Keep the in-memory config consistent for this process either way.
if isinstance(config, dict):
config.setdefault("monitoring", {})
if isinstance(config["monitoring"], dict):
config["monitoring"]["install_id"] = minted
return minted
__all__ = [
"ensure_install_id",
]
+71
View File
@@ -0,0 +1,71 @@
"""Redaction applied to monitoring data before egress.
One unconditional scrub, no modes, no knobs. Every string that leaves the
process passes through ``redact_for_export``:
* Secrets first — wraps ``agent/redact.py::redact_sensitive_text(force=True)``
plus bearer/token-shape patterns, and fails CLOSED: if the redactor cannot
run, the raw string is never emitted.
* PII second — e-mail addresses, phone numbers, and UUID-shaped identifiers
are rewritten to ``[email]`` / ``[phone]`` / ``[id]``.
There is deliberately no setting to weaken this. The monitoring plane is
content-free by design: rendered log messages are not exported, and bounded
structured strings are still scrubbed as defense-in-depth. This redactor also
remains available for a future, explicitly gated redacted-message detail mode.
"""
from __future__ import annotations
import re
from typing import Optional
# ── secret shapes (belt-and-suspenders on top of agent/redact.py) ───────────
_BEARER_RE = re.compile(r"\bBearer\s+[A-Za-z0-9._~+\-/]+=*", re.IGNORECASE)
_TOKEN_RE = re.compile(
r"\b(xox[baprs]-[A-Za-z0-9-]+|sk-[A-Za-z0-9_-]{8,}|gh[pousr]_[A-Za-z0-9_]{8,})\b"
)
_SECRET_LITERAL_RE = re.compile(r"\*{3,}")
_BEARER_RESIDUE_RE = re.compile(r"\bBearer\s+\[[^\]]+\]", re.IGNORECASE)
# ── PII shapes ───────────────────────────────────────────────────────────────
_EMAIL_RE = re.compile(r"[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}")
# E.164-ish and common separators; conservative to avoid nuking code/IDs.
_PHONE_RE = re.compile(
r"(?<!\w)(?:\+?\d{1,3}[\s.\-]?)?(?:\(\d{2,4}\)[\s.\-]?)?\d{3}[\s.\-]?\d{3,4}(?:[\s.\-]?\d{2,4})?(?!\w)"
)
# Long opaque hex/uuid-ish user identifiers.
_UUID_RE = re.compile(
r"\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b"
)
def _secret_redact(text: str) -> str:
"""Always-on secret redaction. force=True so user config can't disable it."""
try:
from agent.redact import redact_sensitive_text
out = redact_sensitive_text(text, force=True)
except Exception:
# Fail CLOSED: if the redactor can't run, do not emit the raw string.
return "[redaction-unavailable]"
out = _BEARER_RE.sub("[redacted]", out)
out = _TOKEN_RE.sub("[redacted]", out)
out = _SECRET_LITERAL_RE.sub("[redacted]", out)
out = _BEARER_RESIDUE_RE.sub("[redacted]", out)
return out
def redact_for_export(text: Optional[str]) -> Optional[str]:
"""Scrub a string for egress: secrets, then PII. Unconditional."""
if text is None:
return None
out = _secret_redact(str(text))
out = _EMAIL_RE.sub("[email]", out)
out = _UUID_RE.sub("[id]", out)
out = _PHONE_RE.sub("[phone]", out)
return out
__all__ = [
"redact_for_export",
]
+33 -2
View File
@@ -15,6 +15,9 @@ and MoonshotAI/kimi-cli#1595:
2. When ``anyOf`` is used, ``type`` must be on the ``anyOf`` children, not
the parent. Presence of both causes "type should be defined in anyOf
items instead of the parent schema".
3. Every object schema must carry a ``required`` array, even an empty one.
Standard JSON Schema allows omitting it; Moonshot 400s with
"required must be an array".
The ``#/definitions/...`` → ``#/$defs/...`` rewrite for draft-07 refs is
handled separately in ``tools/mcp_tool._normalize_mcp_input_schema`` so it
@@ -130,9 +133,32 @@ def _repair_schema(node: Any, is_schema: bool = True) -> Any:
else:
repaired.pop("enum")
# Rule 4: object schemas must carry a `required` array, even when empty.
if repaired.get("type") == "object":
repaired = _ensure_required_array(repaired)
return repaired
def _ensure_required_array(node: Dict[str, Any]) -> Dict[str, Any]:
"""Guarantee an object schema carries a ``required`` array (Moonshot rule).
Standard JSON Schema lets you omit ``required`` when nothing is required;
Moonshot 400s on that ("required must be an array"). Ensure the key is a
list. When ``properties`` is known, prune ``required`` entries that don't
name a real property — defensive against dangling names, which Moonshot
also rejects. Mutates and returns ``node``.
"""
props = node.get("properties")
req = node.get("required")
if isinstance(req, list):
if isinstance(props, dict):
node["required"] = [r for r in req if r in props]
else:
node["required"] = []
return node
def _fill_missing_type(node: Dict[str, Any]) -> Dict[str, Any]:
"""Infer a reasonable ``type`` if this schema node has none."""
node_type = node.get("type")
@@ -174,17 +200,18 @@ def sanitize_moonshot_tool_parameters(parameters: Any) -> Dict[str, Any]:
applied. Input is not mutated.
"""
if not isinstance(parameters, dict):
return {"type": "object", "properties": {}}
return {"type": "object", "properties": {}, "required": []}
repaired = _repair_schema(copy.deepcopy(parameters), is_schema=True)
if not isinstance(repaired, dict):
return {"type": "object", "properties": {}}
return {"type": "object", "properties": {}, "required": []}
# Top-level must be an object schema
if repaired.get("type") != "object":
repaired["type"] = "object"
if "properties" not in repaired:
repaired["properties"] = {}
_ensure_required_array(repaired)
return repaired
@@ -232,6 +259,10 @@ def is_moonshot_model(model: str | None) -> bool:
tail = bare.rsplit("/", 1)[-1]
if tail.startswith("kimi-") or tail == "kimi":
return True
# Kimi Coding Plan serves K3 under the bare slug ``k3`` (plus dated /
# suffixed variants like ``k3.1`` or ``k3-turbo``).
if tail == "k3" or tail.startswith(("k3.", "k3-")):
return True
# Vendor-prefixed forms commonly used on aggregators
if "moonshot" in bare or "/kimi" in bare or bare.startswith("kimi"):
return True

Some files were not shown because too many files have changed in this diff Show More