Merge upstream/main into linux-keychain-auto-detect
Resolves conflicts from upstream's DEFAULT_CONFIG extraction into hermes_cli/config_defaults.py (password_store default moved there) and the test-pruning waves (dropped the pruned pre-existing launch-option tests; kept the new password-store tests). Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -97,9 +97,6 @@ packaging/
|
||||
plans/
|
||||
.plans/
|
||||
|
||||
# ACP registry manifest (icon + agent.json) — not consumed at runtime
|
||||
acp_registry/
|
||||
|
||||
# Repo-level dotfiles that are git-only or dev-tooling config
|
||||
.env.example
|
||||
.envrc
|
||||
|
||||
@@ -7,7 +7,7 @@ description: >-
|
||||
|
||||
inputs:
|
||||
github-token:
|
||||
description: Token for the GitHub API (gh CLI). Pass secrets.AUTOFIX_BOT_PAT from the calling workflow.
|
||||
description: Token for the GitHub API (gh CLI). Pass steps.app-token.outputs.token from the calling workflow.
|
||||
required: false
|
||||
default: ${{ github.token }}
|
||||
|
||||
@@ -39,6 +39,9 @@ outputs:
|
||||
ci_review:
|
||||
description: Require CI-sensitive file review label.
|
||||
value: ${{ steps.classify.outputs.ci_review }}
|
||||
ci_review_files:
|
||||
description: JSON list of CI-sensitive files changed by the pull request.
|
||||
value: ${{ steps.classify.outputs.ci_review_files }}
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
@@ -72,12 +75,19 @@ runs:
|
||||
# Retried: a rate-limit blip or eventual-consistency 404 on a
|
||||
# freshly-pushed HEAD would otherwise silently fall open (all lanes
|
||||
# run — safe, but wasteful and it masks the API failure).
|
||||
#
|
||||
# `.files[]?` (null-safe): with --paginate, a PR more than 100
|
||||
# commits ahead of its merge-base paginates the compare, and pages
|
||||
# after the first carry `files: null` — bare `.files[]` makes jq
|
||||
# die with "cannot iterate over: null", which fails every retry
|
||||
# and forces the fail-open path (seen on stacked PRs). The full
|
||||
# file list (up to the API's 300-file cap) is on page one.
|
||||
CHANGED=""
|
||||
for i in 1 2 3; do
|
||||
if CHANGED="$(gh api \
|
||||
--paginate \
|
||||
"repos/${REPO}/compare/${BASE_SHA}...${HEAD_SHA}" \
|
||||
--jq '.files[].filename')"; then
|
||||
--jq '.files[]?.filename')"; then
|
||||
break
|
||||
fi
|
||||
if [ "$i" = 3 ]; then
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
name: Get GitHub App Token
|
||||
description: >-
|
||||
Mint a short-lived (1-hour) installation access token from the repo's
|
||||
GitHub App, replacing the long-lived AUTOFIX_BOT_PAT. App tokens get
|
||||
5,000 req/hr per installation (vs 1,000 for the default GITHUB_TOKEN)
|
||||
and are scoped to the App's installation permissions, not a user account.
|
||||
|
||||
Callers must source App credentials from a protected, main-only environment.
|
||||
Never pass an App private key to a pull_request job, a local action, or a
|
||||
reusable workflow resolved from an untrusted PR ref. The fallback keeps a
|
||||
trusted caller functional when its protected environment is misconfigured.
|
||||
|
||||
Composite actions cannot access contexts directly, so callers pass the
|
||||
public vars.APP_CLIENT_ID and protected secrets.APP_PRIVATE_KEY as inputs.
|
||||
When the private key is empty, the fallback fires.
|
||||
|
||||
inputs:
|
||||
client-id:
|
||||
description: GitHub App Client ID. Pass vars.APP_CLIENT_ID from the calling workflow.
|
||||
required: false
|
||||
default: ''
|
||||
private-key:
|
||||
description: GitHub App private key PEM. Pass secrets.APP_PRIVATE_KEY from the calling workflow.
|
||||
required: false
|
||||
default: ''
|
||||
owner:
|
||||
description: GitHub App installation owner. Empty scopes the token to the current repository.
|
||||
required: false
|
||||
default: ''
|
||||
repositories:
|
||||
description: Comma- or newline-separated repositories to scope within the installation owner.
|
||||
required: false
|
||||
default: ''
|
||||
|
||||
outputs:
|
||||
token:
|
||||
description: A GitHub App installation access token (1-hour TTL), or GITHUB_TOKEN on forks.
|
||||
value: ${{ steps.app-token.outputs.token || steps.fallback.outputs.token }}
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Check if App credentials exist
|
||||
id: check
|
||||
shell: bash
|
||||
env:
|
||||
CLIENT_ID: ${{ inputs.client-id }}
|
||||
run: |
|
||||
if [ -n "$CLIENT_ID" ]; then
|
||||
echo "has_app=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "has_app=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Create GitHub App token
|
||||
id: app-token
|
||||
if: steps.check.outputs.has_app == 'true'
|
||||
uses: actions/create-github-app-token@bcd2ba49218906704ab6c1aa796996da409d3eb1 # v3.2.0
|
||||
with:
|
||||
client-id: ${{ inputs.client-id }}
|
||||
private-key: ${{ inputs.private-key }}
|
||||
owner: ${{ inputs.owner }}
|
||||
repositories: ${{ inputs.repositories }}
|
||||
|
||||
- name: Fall back to GITHUB_TOKEN
|
||||
id: fallback
|
||||
if: steps.check.outputs.has_app != 'true'
|
||||
shell: bash
|
||||
run: echo "token=${{ github.token }}" >> "$GITHUB_OUTPUT"
|
||||
+156
-29
@@ -9,6 +9,10 @@ name: CI
|
||||
# definitions, matrices, and concurrency settings. They no longer have
|
||||
# ``push:`` / ``pull_request:`` triggers of their own — everything flows
|
||||
# through this file.
|
||||
#
|
||||
# SECURITY: this workflow runs PR-controlled actions, workflows, and code.
|
||||
# Do not add ``secrets: inherit`` or GitHub App credentials here. Trusted
|
||||
# main-only automation uses protected environments in its own workflows.
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
@@ -17,7 +21,7 @@ on:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # needed by lint (PR comment) + supply-chain (PR comment)
|
||||
pull-requests: write # needed by lint (PR comment) + supply-chain review_status
|
||||
actions: read # needed by osv-scanner (SARIF upload)
|
||||
security-events: write # needed by osv-scanner (SARIF upload)
|
||||
packages: write # needed by docker build
|
||||
@@ -46,6 +50,7 @@ jobs:
|
||||
docker_meta: ${{ steps.classify.outputs.docker_meta }}
|
||||
mcp_catalog: ${{ steps.classify.outputs.mcp_catalog }}
|
||||
ci_review: ${{ steps.classify.outputs.ci_review }}
|
||||
ci_review_files: ${{ steps.classify.outputs.ci_review_files }}
|
||||
event_name: ${{ github.event_name }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
@@ -53,9 +58,7 @@ jobs:
|
||||
id: classify
|
||||
uses: ./.github/actions/detect-changes
|
||||
with:
|
||||
# Forks get no repo secrets (AUTOFIX_BOT_PAT is empty); fall back to
|
||||
# the built-in read-only token so classification still works there.
|
||||
github-token: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
|
||||
github-token: ${{ github.token }}
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
# Lane-gated sub-workflows. Each runs in parallel after detect finishes.
|
||||
@@ -68,89 +71,177 @@ jobs:
|
||||
uses: ./.github/workflows/tests.yml
|
||||
with:
|
||||
slice_count: 8
|
||||
secrets: inherit
|
||||
|
||||
lint:
|
||||
name: Python lints
|
||||
needs: detect
|
||||
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.ci_review == 'true'
|
||||
if: needs.detect.outputs.python == 'true'
|
||||
uses: ./.github/workflows/lint.yml
|
||||
with:
|
||||
event_name: ${{ needs.detect.outputs.event_name }}
|
||||
ci_review: ${{ needs.detect.outputs.ci_review == 'true' }}
|
||||
secrets: inherit
|
||||
|
||||
js-tests:
|
||||
name: JS & TS checks
|
||||
needs: detect
|
||||
if: needs.detect.outputs.frontend == 'true'
|
||||
uses: ./.github/workflows/js-tests.yml
|
||||
secrets: inherit
|
||||
|
||||
e2e-desktop:
|
||||
name: Desktop E2E
|
||||
needs: detect
|
||||
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true'
|
||||
uses: ./.github/workflows/e2e-desktop.yml
|
||||
|
||||
docs-site:
|
||||
name: Docs Site
|
||||
needs: detect
|
||||
if: needs.detect.outputs.site == 'true'
|
||||
uses: ./.github/workflows/docs-site-checks.yml
|
||||
secrets: inherit
|
||||
|
||||
history-check:
|
||||
name: Deny unrelated histories
|
||||
needs: detect
|
||||
if: needs.detect.outputs.event_name == 'pull_request'
|
||||
uses: ./.github/workflows/history-check.yml
|
||||
secrets: inherit
|
||||
|
||||
contributor-check:
|
||||
name: Check contributors
|
||||
needs: detect
|
||||
if: needs.detect.outputs.python == 'true'
|
||||
uses: ./.github/workflows/contributor-check.yml
|
||||
secrets: inherit
|
||||
|
||||
uv-lockfile:
|
||||
name: Check uv.lock
|
||||
needs: detect
|
||||
uses: ./.github/workflows/uv-lockfile-check.yml
|
||||
secrets: inherit
|
||||
|
||||
infographic-check:
|
||||
name: Check no committed infographics
|
||||
needs: detect
|
||||
uses: ./.github/workflows/infographic-check.yml
|
||||
|
||||
lockfile-diff:
|
||||
name: package-lock.json diff
|
||||
needs: detect
|
||||
if: needs.detect.outputs.event_name == 'pull_request' && needs.detect.outputs.npm_lock == 'true'
|
||||
uses: ./.github/workflows/lockfile-diff.yml
|
||||
secrets: inherit
|
||||
|
||||
docker-lint:
|
||||
name: Lint Docker scripts
|
||||
needs: detect
|
||||
if: needs.detect.outputs.docker_meta == 'true'
|
||||
uses: ./.github/workflows/docker-lint.yml
|
||||
secrets: inherit
|
||||
|
||||
docker:
|
||||
name: Build&Test Docker image
|
||||
needs: detect
|
||||
if: needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true' || needs.detect.outputs.docker_meta == 'true'
|
||||
# Trusted main pushes run docker.yml directly so its container-publish
|
||||
# environment secrets never cross this reusable-workflow call. PR runs
|
||||
# remain build/test-only and secret-free.
|
||||
if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.python == 'true' || needs.detect.outputs.frontend == 'true' || needs.detect.outputs.docker_meta == 'true')
|
||||
uses: ./.github/workflows/docker.yml
|
||||
secrets: inherit
|
||||
|
||||
supply-chain:
|
||||
name: Supply-chain scan
|
||||
needs: detect
|
||||
if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.scan == 'true' || needs.detect.outputs.deps == 'true' || needs.detect.outputs.mcp_catalog == 'true')
|
||||
if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.scan == 'true' || needs.detect.outputs.deps == 'true')
|
||||
uses: ./.github/workflows/supply-chain-audit.yml
|
||||
with:
|
||||
event_name: ${{ needs.detect.outputs.event_name }}
|
||||
scan: ${{ needs.detect.outputs.scan == 'true' }}
|
||||
deps: ${{ needs.detect.outputs.deps == 'true' }}
|
||||
|
||||
review-labels:
|
||||
name: Review label gate
|
||||
needs: [detect, supply-chain]
|
||||
if: always() && needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.ci_review == 'true' || needs.detect.outputs.mcp_catalog == 'true' || needs.supply-chain.outputs.critical_findings == 'true')
|
||||
uses: ./.github/workflows/review-labels.yml
|
||||
with:
|
||||
ci_review: ${{ needs.detect.outputs.ci_review == 'true' }}
|
||||
ci_review_files: ${{ needs.detect.outputs.ci_review_files }}
|
||||
mcp_catalog: ${{ needs.detect.outputs.mcp_catalog == 'true' }}
|
||||
secrets: inherit
|
||||
supply_chain: ${{ needs.supply-chain.outputs.critical_findings == 'true' }}
|
||||
|
||||
osv-scanner:
|
||||
name: OSV scan
|
||||
uses: ./.github/workflows/osv-scanner.yml
|
||||
secrets: inherit
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
# Live-updating PR review comment.
|
||||
#
|
||||
# A single ``comment-live`` job polls the GitHub Actions API every 15s
|
||||
# for job statuses in this run, re-assembles the review comment from
|
||||
# whatever results are available, and upserts it via the
|
||||
# ``<!-- hermes-ci-review-bot -->`` marker.
|
||||
#
|
||||
# When the visible job set goes quiet, the poller waits 10 seconds and polls
|
||||
# once more so downstream jobs created by an aggregate gate get included.
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
comment-live:
|
||||
name: CI review comment (live)
|
||||
needs: [detect, review-labels, lockfile-diff, supply-chain, osv-scanner, uv-lockfile, history-check, contributor-check, e2e-desktop]
|
||||
if: always() && github.event_name == 'pull_request' && github.event.pull_request.head.repo.fork != true
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 40
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
ref: ${{ github.event.repository.default_branch }}
|
||||
persist-credentials: false
|
||||
|
||||
- name: Run live comment poller
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GITHUB_REPOSITORY: ${{ github.repository }}
|
||||
GITHUB_RUN_ID: ${{ github.run_id }}
|
||||
PR_NUMBER: ${{ github.event.pull_request.number }}
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
# Commit info for the review comment header.
|
||||
COMMIT_SHA: ${{ github.event.pull_request.head.sha }}
|
||||
COMMIT_MESSAGE: ${{ github.event.pull_request.head.commit.message }}
|
||||
COMMIT_URL: ${{ github.server_url }}/${{ github.repository }}/pull/${{ github.event.pull_request.number }}/commits/${{ github.event.pull_request.head.sha }}
|
||||
# Structured review statuses from workflow_call jobs.
|
||||
# Each job outputs a JSON array of {source, results: [...]} objects
|
||||
# that the assembler renders directly — no hardcoded job-name
|
||||
# matching. We merge all available outputs into one array.
|
||||
REVIEW_STATUSES: ${{ toJSON(needs.*.outputs.review_status) }}
|
||||
run: |
|
||||
set -uo pipefail
|
||||
|
||||
# REVIEW_STATUSES is a JSON array of strings (some may be empty
|
||||
# when a job was skipped). Parse each string and merge into one
|
||||
# flat array for the assembler.
|
||||
python3 - <<'PYEOF'
|
||||
import json, os, sys
|
||||
|
||||
raw = os.environ.get("REVIEW_STATUSES", "")
|
||||
merged = []
|
||||
if raw:
|
||||
try:
|
||||
arr = json.loads(raw)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
arr = []
|
||||
for item in arr:
|
||||
if not item:
|
||||
continue
|
||||
try:
|
||||
statuses = json.loads(item)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
continue
|
||||
if isinstance(statuses, list):
|
||||
merged.extend(statuses)
|
||||
|
||||
# Write merged array to a temp file the poller reads.
|
||||
with open("/tmp/review_statuses.json", "w") as f:
|
||||
json.dump(merged, f)
|
||||
print(f"Merged {len(merged)} review status entries")
|
||||
PYEOF
|
||||
|
||||
python3 scripts/ci/live_comment.py \
|
||||
--interval 15 \
|
||||
--timeout 2100 \
|
||||
--review-statuses-file /tmp/review_statuses.json
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
# Gate: runs after everything. ``if: always()`` ensures it reports a
|
||||
@@ -158,13 +249,18 @@ jobs:
|
||||
# results cause it to fail; ``skipped`` is treated as success.
|
||||
#
|
||||
# Branch protection should require ONLY this check.
|
||||
#
|
||||
# Outputs ``needs-json`` — a compact ``{job_name: result}`` dict — so
|
||||
# the live comment poller can list failed jobs in the PR comment.
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
all-checks-pass:
|
||||
name: All required checks pass
|
||||
needs:
|
||||
- detect
|
||||
- tests
|
||||
- lint
|
||||
- js-tests
|
||||
- e2e-desktop
|
||||
- docs-site
|
||||
- history-check
|
||||
- contributor-check
|
||||
@@ -172,20 +268,30 @@ jobs:
|
||||
- lockfile-diff
|
||||
- docker-lint
|
||||
- supply-chain
|
||||
- review-labels
|
||||
- osv-scanner
|
||||
# comment-live is a polling job — it doesn't block the gate.
|
||||
# we don't require docker to pass rn because it's so slow lol
|
||||
# - docker
|
||||
if: always()
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
outputs:
|
||||
needs-json: ${{ steps.evaluate.outputs.needs-json }}
|
||||
steps:
|
||||
- name: Evaluate job results
|
||||
id: evaluate
|
||||
env:
|
||||
NEEDS: ${{ toJSON(needs) }}
|
||||
run: |
|
||||
echo "$NEEDS" | python3 -c "
|
||||
import json, sys
|
||||
needs = json.load(sys.stdin)
|
||||
# Emit compact {job_name: result} for the comment assembler.
|
||||
compact = {name: info['result'] for name, info in needs.items()}
|
||||
print(f'needs-json={json.dumps(compact)}')
|
||||
with open('$GITHUB_OUTPUT', 'a') as f:
|
||||
f.write(f'needs-json={json.dumps(compact)}\n')
|
||||
failed = [name for name, info in needs.items() if info['result'] == 'failure']
|
||||
for name, info in sorted(needs.items()):
|
||||
result = info['result']
|
||||
@@ -202,6 +308,9 @@ jobs:
|
||||
# cache them on main (as a baseline), and on PRs generate an HTML diff
|
||||
# report with a gantt chart + per-step breakdown. The report is uploaded
|
||||
# as an artifact and a markdown summary is written to $GITHUB_STEP_SUMMARY.
|
||||
#
|
||||
# The live comment poller can read the standalone review-status artifact
|
||||
# after the HTML report is uploaded, so its link points straight at that report.
|
||||
# ─────────────────────────────────────────────────────────────────────
|
||||
ci-timings:
|
||||
name: CI timing report
|
||||
@@ -226,10 +335,7 @@ jobs:
|
||||
|
||||
- name: Collect timings and generate report
|
||||
env:
|
||||
# Forks get no repo secrets (AUTOFIX_BOT_PAT is empty); fall back to
|
||||
# the built-in read-only token so the timings API read still works
|
||||
# there instead of hard-failing this advisory job on every fork PR.
|
||||
GITHUB_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
python3 scripts/ci/timings_report.py \
|
||||
--baseline ci-timings-baseline.json \
|
||||
@@ -241,19 +347,40 @@ jobs:
|
||||
# Advisory report — artifact-service blips must not fail the job.
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
id: ci-timings-artifact
|
||||
id: ci-timings-html
|
||||
with:
|
||||
name: ci-timings-report
|
||||
path: ci-timings-report.html
|
||||
retention-days: 14
|
||||
archive: false
|
||||
|
||||
- name: Build linked review status
|
||||
if: hashFiles('ci-timings.json') != ''
|
||||
env:
|
||||
CI_TIMINGS_REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url }}
|
||||
run: |
|
||||
python3 scripts/ci/timings_report.py \
|
||||
--from-json ci-timings.json \
|
||||
--baseline ci-timings-baseline.json \
|
||||
--review-status-out review-status.json \
|
||||
--review-status-only
|
||||
|
||||
- name: Upload review status
|
||||
if: hashFiles('review-status.json') != ''
|
||||
continue-on-error: true
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: ci-timings-review-status
|
||||
path: review-status.json
|
||||
retention-days: 14
|
||||
|
||||
- name: Output summary
|
||||
env:
|
||||
REPORT_URL: ${{ steps.ci-timings-artifact.outputs.artifact-url}}
|
||||
REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url}}
|
||||
run: |
|
||||
echo "# CI Timing report" >> "$GITHUB_STEP_SUMMARY"
|
||||
echo "[View the full interactive report]($REPORT_URL)" >> "$GITHUB_STEP_SUMMARY"
|
||||
{
|
||||
echo "# CI Timing report"
|
||||
echo "[View the full interactive report]($REPORT_URL)"
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
cat ci-timings-summary.md >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
- name: Save baseline cache (main only)
|
||||
|
||||
@@ -2,6 +2,10 @@ name: Contributor Attribution Check
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
outputs:
|
||||
review_status:
|
||||
description: "JSON array of review status objects"
|
||||
value: ${{ jobs.check-attribution.outputs.review_status }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -10,12 +14,15 @@ jobs:
|
||||
check-attribution:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
outputs:
|
||||
review_status: ${{ steps.check-emails.outputs.review_status }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
fetch-depth: 0 # Full history needed for git log
|
||||
|
||||
- name: Check for unmapped contributor emails
|
||||
id: check-emails
|
||||
run: |
|
||||
# Get the merge base between this PR and main
|
||||
MERGE_BASE=$(git merge-base origin/main HEAD)
|
||||
@@ -25,6 +32,7 @@ jobs:
|
||||
|
||||
if [ -z "$NEW_EMAILS" ]; then
|
||||
echo "No new commits to check."
|
||||
echo "review_status=[]" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
@@ -67,6 +75,16 @@ jobs:
|
||||
echo ""
|
||||
echo "To find the GitHub username for an email:"
|
||||
echo " gh api 'search/users?q=EMAIL+in:email' --jq '.items[0].login'"
|
||||
|
||||
# Emit review_status for unmapped emails
|
||||
DETAIL=$(echo -e "$MISSING" | sed '/^$/d; s/^ //')
|
||||
HOW_TO_FIX=$'Add mappings to scripts/release.py AUTHOR_MAP:\n```\n"<email>": "<github-username>",\n```\nTo find the GitHub username for an email:\n```\ngh api \'search/users?q=EMAIL+in:email\' --jq \'.items[0].login\'\n```\n'
|
||||
REVIEW_STATUS=$(jq -nc \
|
||||
--arg detail "$DETAIL" \
|
||||
--arg how_to_fix "$HOW_TO_FIX" \
|
||||
'[{"source":"contributor attribution","results":[{"kind":"action_required","title":"Unmapped contributor email(s)","summary":"New contributor email(s) are not in AUTHOR_MAP.","detail":$detail,"how_to_fix":$how_to_fix}]}]')
|
||||
echo "review_status=$REVIEW_STATUS" >> "$GITHUB_OUTPUT"
|
||||
|
||||
exit 1
|
||||
else
|
||||
echo "✅ All contributor emails are mapped."
|
||||
|
||||
@@ -56,6 +56,13 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Get GitHub App token
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ vars.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 22
|
||||
@@ -73,8 +80,8 @@ jobs:
|
||||
|
||||
- name: Prepare skills index (unified multi-source catalog)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
GITHUB_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
GH_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
SKILLS_INDEX_RUN_ID: ${{ github.event.inputs.skills_index_run_id || '' }}
|
||||
REBUILD_SKILLS_INDEX: ${{ github.event.inputs.rebuild_skills_index || 'false' }}
|
||||
run: |
|
||||
|
||||
+113
-45
@@ -1,8 +1,15 @@
|
||||
name: Docker Build, Test, and Publish
|
||||
|
||||
on:
|
||||
# Trusted main pushes run this workflow directly so environment-scoped
|
||||
# Docker Hub secrets are resolved by the top-level workflow, never across
|
||||
# a reusable-workflow boundary.
|
||||
push:
|
||||
branches: [main]
|
||||
release:
|
||||
types: [published]
|
||||
# CI calls this only for untrusted PR build/test coverage. Those runs never
|
||||
# reach the protected publish or merge jobs below.
|
||||
workflow_call:
|
||||
|
||||
permissions:
|
||||
@@ -20,7 +27,9 @@ env:
|
||||
IMAGE_NAME: nousresearch/hermes-agent
|
||||
|
||||
jobs:
|
||||
# Build, test, and optionally push the image for each architecture.
|
||||
# Build and test the image for each architecture. This job runs PR code,
|
||||
# so it must remain secret-free. Publishing happens in the separate,
|
||||
# protected publish job after these tests pass.
|
||||
build:
|
||||
if: github.repository == 'NousResearch/hermes-agent'
|
||||
strategy:
|
||||
@@ -44,7 +53,19 @@ jobs:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
# Retry once on transient Docker Hub / buildkit pull failures
|
||||
# (connection reset, auth token timeout, rate limiting). The action
|
||||
# generates a unique builder name per invocation so the retry doesn't
|
||||
# collide with the failed first attempt. A genuine persistent failure
|
||||
# still fails the job — only the first attempt has continue-on-error.
|
||||
# Refs: docker/setup-buildx-action#510
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
continue-on-error: true
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
- name: Set up Docker Buildx (retry)
|
||||
if: steps.buildx.outcome == 'failure'
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
# Build once, load into the local daemon for testing. Cached
|
||||
@@ -62,49 +83,6 @@ jobs:
|
||||
cache-from: ${{ matrix.cache-from }}
|
||||
cache-to: ${{ (github.event_name != 'pull_request') && matrix.cache-to || '' }}
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
|
||||
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
# Push by digest only (no tag). The merge job assembles the
|
||||
# tagged manifest list. `push-by-digest=true` is docker's recommended
|
||||
# pattern for multi-runner multi-platform builds.
|
||||
- name: Push ${{ matrix.arch }} by digest
|
||||
id: push
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
|
||||
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
|
||||
with:
|
||||
context: .
|
||||
file: Dockerfile
|
||||
platforms: ${{ matrix.platform }}
|
||||
labels: |
|
||||
org.opencontainers.image.revision=${{ github.sha }}
|
||||
build-args: |
|
||||
HERMES_GIT_SHA=${{ github.sha }}
|
||||
outputs: type=image,name=${{ env.IMAGE_NAME }},push-by-digest=true,name-canonical=true,push=true
|
||||
cache-from: ${{ matrix.cache-from }}
|
||||
cache-to: ${{ matrix.cache-to }}
|
||||
|
||||
# Write the digest to a file and upload it as an artifact so the
|
||||
# merge job can stitch both per-arch digests into a manifest list.
|
||||
- name: Export digest
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
|
||||
run: |
|
||||
mkdir -p /tmp/digests
|
||||
digest="${{ steps.push.outputs.digest }}"
|
||||
touch "/tmp/digests/${digest#sha256:}"
|
||||
|
||||
- name: Upload digest artifact
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release'
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: digest-${{ matrix.arch }}
|
||||
path: /tmp/digests/*
|
||||
if-no-files-found: error
|
||||
retention-days: 1
|
||||
|
||||
# Run the docker-integration test suite against the freshly-built
|
||||
# image already loaded into the local daemon (`:test`).
|
||||
@@ -122,6 +100,11 @@ jobs:
|
||||
# ---------------------------------------------------------------------
|
||||
- name: Install uv (for docker tests)
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
|
||||
# raw.githubusercontent.com every job; transient fetch failures
|
||||
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
|
||||
version: "0.9.28"
|
||||
|
||||
- name: Set up Python 3.11 (for docker tests)
|
||||
run: uv python install 3.11
|
||||
@@ -147,6 +130,82 @@ jobs:
|
||||
run: |
|
||||
scripts/run_tests.sh tests/docker/ --file-timeout 600
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Rebuild and push each architecture only after the unprivileged build/test
|
||||
# matrix passes. This job is the sole Docker Hub credential boundary.
|
||||
# ---------------------------------------------------------------------------
|
||||
publish:
|
||||
if: github.repository == 'NousResearch/hermes-agent' && (github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release')
|
||||
needs: [build]
|
||||
environment: container-publish
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- arch: amd64
|
||||
runner: ubuntu-latest
|
||||
platform: linux/amd64
|
||||
cache-from: type=gha,scope=docker-amd64
|
||||
cache-to: type=gha,mode=max,scope=docker-amd64
|
||||
- arch: arm64
|
||||
runner: ubuntu-24.04-arm
|
||||
platform: linux/arm64
|
||||
cache-from: type=gha,scope=docker-arm64
|
||||
cache-to: type=gha,mode=max,scope=docker-arm64
|
||||
runs-on: ${{ matrix.runner }}
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout trusted source
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
# Retry once on transient Docker Hub / buildkit pull failures.
|
||||
# See build job for rationale; same pattern.
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
continue-on-error: true
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
- name: Set up Docker Buildx (retry)
|
||||
if: steps.buildx.outcome == 'failure'
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
# Push by digest only (no tag). The merge job assembles the tagged
|
||||
# manifest list after both architecture publishers complete.
|
||||
- name: Push ${{ matrix.arch }} by digest
|
||||
id: push
|
||||
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
|
||||
with:
|
||||
context: .
|
||||
file: Dockerfile
|
||||
platforms: ${{ matrix.platform }}
|
||||
labels: |
|
||||
org.opencontainers.image.revision=${{ github.sha }}
|
||||
build-args: |
|
||||
HERMES_GIT_SHA=${{ github.sha }}
|
||||
outputs: type=image,name=${{ env.IMAGE_NAME }},push-by-digest=true,name-canonical=true,push=true
|
||||
cache-from: ${{ matrix.cache-from }}
|
||||
cache-to: ${{ matrix.cache-to }}
|
||||
|
||||
- name: Export digest
|
||||
run: |
|
||||
mkdir -p /tmp/digests
|
||||
digest="${{ steps.push.outputs.digest }}"
|
||||
touch "/tmp/digests/${digest#sha256:}"
|
||||
|
||||
- name: Upload digest artifact
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: digest-${{ matrix.arch }}
|
||||
path: /tmp/digests/*
|
||||
if-no-files-found: error
|
||||
retention-days: 1
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stitch both per-arch digests into a single tagged multi-arch manifest.
|
||||
# This is a registry-side operation — no building, no layer re-push —
|
||||
@@ -158,8 +217,9 @@ jobs:
|
||||
merge:
|
||||
if: github.repository == 'NousResearch/hermes-agent' && (github.event_name == 'push' && github.ref == 'refs/heads/main' || github.event_name == 'release')
|
||||
runs-on: ubuntu-latest
|
||||
needs: [build]
|
||||
needs: [publish]
|
||||
timeout-minutes: 10
|
||||
environment: container-publish
|
||||
steps:
|
||||
- name: Download digests
|
||||
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
|
||||
@@ -168,7 +228,15 @@ jobs:
|
||||
pattern: digest-*
|
||||
merge-multiple: true
|
||||
|
||||
# Retry once on transient Docker Hub / buildkit pull failures.
|
||||
# See build job for rationale; same pattern.
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
continue-on-error: true
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
- name: Set up Docker Buildx (retry)
|
||||
if: steps.buildx.outcome == 'failure'
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
|
||||
@@ -0,0 +1,260 @@
|
||||
name: E2E Desktop
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
outputs:
|
||||
review_status:
|
||||
description: Screenshot and visual-diff status for the CI review comment.
|
||||
value: ${{ jobs.e2e.outputs.review_status }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: e2e-desktop-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
e2e:
|
||||
name: Playwright E2E (Linux)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
outputs:
|
||||
review_status: ${{ steps.review-status.outputs.review_status }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
# ── System deps for Electron on headless Ubuntu ───────────────────
|
||||
# Electron needs GTK, NSS,atk, etc. even under xvfb. Playwright's
|
||||
# install-deps covers browsers; for Electron we install the apt
|
||||
# packages directly.
|
||||
- name: Install system dependencies for Electron
|
||||
run: |
|
||||
sudo apt-get update -qq
|
||||
sudo apt-get install -y -qq \
|
||||
xvfb \
|
||||
libgtk-3-0 libnotify4 libnss3 libxss1 libxtst6 \
|
||||
xdg-utils libatspi2.0-0 libdrm2 libgbm1 libasound2t64
|
||||
|
||||
# ── Node ───────────────────────────────────────────────────────────
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 22
|
||||
cache: npm
|
||||
# Full npm ci (not --ignore-scripts): electron's postinstall
|
||||
# downloads the binary we launch, and node-pty's native build is
|
||||
# needed for the terminal pane.
|
||||
- uses: ./.github/actions/retry
|
||||
with:
|
||||
command: npm ci
|
||||
|
||||
# ── Python (for the hermes serve backend) ──────────────────────────
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pin the uv version: unpinned, setup-uv resolves "latest" by
|
||||
# fetching a manifest from raw.githubusercontent.com on EVERY job —
|
||||
# a transient fetch failure fails the whole job (2026-07-28 slice-5
|
||||
# incident). Pinned, the binary downloads directly; no manifest hop.
|
||||
version: "0.9.28"
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
pyproject.toml
|
||||
uv.lock
|
||||
- name: Set up Python 3.11
|
||||
run: uv python install 3.11
|
||||
- name: Install Python dependencies
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev
|
||||
|
||||
# ── Build desktop app ─────────────────────────────────────────────
|
||||
# The Playwright step below runs `npm run build` before testing so
|
||||
# dist/ is always fresh — no separate build step needed here.
|
||||
|
||||
# ── Restore visual baseline screenshots from main ──────────────────
|
||||
# Baselines are generated on main (via --update-snapshots) and cached.
|
||||
# On PRs, we restore them so toHaveScreenshot has something to compare
|
||||
# against. The cache key is keyed on the desktop source files so a
|
||||
# UI change naturally invalidates it — but we fall back to the main
|
||||
# cache to avoid cold starts on unrelated PRs.
|
||||
- name: Restore visual baseline screenshots
|
||||
id: restore-baselines
|
||||
uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4
|
||||
with:
|
||||
path: apps/desktop/e2e/*-snapshots
|
||||
key: visual-baselines-${{ github.ref_name }}
|
||||
restore-keys: |
|
||||
visual-baselines-main
|
||||
|
||||
# ── Run Playwright E2E under xvfb ─────────────────────────────────
|
||||
# xvfb runs at a fixed 1280x1024 screen so the 1220x800 Electron
|
||||
# window always has a consistent viewport for screenshot comparison.
|
||||
# On main, we run with --update-snapshots to generate baselines.
|
||||
# `npm run test:e2e` builds dist/ as a pretest hook so the renderer
|
||||
# is always fresh — no separate build step needed.
|
||||
- name: Run Playwright E2E tests
|
||||
working-directory: apps/desktop
|
||||
run: |
|
||||
if [ "${{ github.ref_name }}" = "main" ]; then
|
||||
echo "On main — generating/updating baseline screenshots"
|
||||
npm run build && xvfb-run -a --server-args="-screen 0 1280x1024x24" \
|
||||
npx playwright test --reporter=list --update-snapshots
|
||||
else
|
||||
echo "On PR — comparing against cached baselines"
|
||||
npm run build && xvfb-run -a --server-args="-screen 0 1280x1024x24" \
|
||||
npx playwright test --reporter=list
|
||||
fi
|
||||
env:
|
||||
CI: "true"
|
||||
# Ensure no real API keys leak into the test env.
|
||||
OPENROUTER_API_KEY: ""
|
||||
OPENAI_API_KEY: ""
|
||||
NOUS_API_KEY: ""
|
||||
|
||||
# ── Save updated baselines to cache (main only) ───────────────────
|
||||
- name: Save updated baselines to cache
|
||||
if: github.ref_name == 'main' && always()
|
||||
uses: actions/cache/save@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4
|
||||
with:
|
||||
path: apps/desktop/e2e/*-snapshots
|
||||
key: visual-baselines-main
|
||||
|
||||
# ── Upload Playwright report (HTML + traces) ──────────────────────
|
||||
- name: Upload Playwright report
|
||||
id: upload-report
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: playwright-report-${{ github.sha }}
|
||||
path: apps/desktop/playwright-report
|
||||
retention-days: 14
|
||||
overwrite: true
|
||||
|
||||
# ── Upload test results (screenshots, traces, diffs) ───────────────
|
||||
- name: Upload test results
|
||||
id: upload-results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: playwright-test-results-${{ github.sha }}
|
||||
path: apps/desktop/test-results
|
||||
retention-days: 14
|
||||
overwrite: true
|
||||
|
||||
# ── Upload just the visual diffs (small, fast to review) ──────────
|
||||
- name: Upload visual diffs
|
||||
id: upload-diffs
|
||||
if: always() && github.ref_name != 'main'
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: visual-diffs-${{ github.sha }}
|
||||
path: |
|
||||
apps/desktop/test-results/**/*-diff.png
|
||||
apps/desktop/test-results/**/*-actual.png
|
||||
apps/desktop/test-results/**/*-expected.png
|
||||
retention-days: 14
|
||||
overwrite: true
|
||||
if-no-files-found: ignore
|
||||
|
||||
- name: Build screenshot review status
|
||||
id: review-status
|
||||
if: always()
|
||||
working-directory: apps/desktop
|
||||
env:
|
||||
RESULTS_URL: ${{ steps.upload-results.outputs.artifact-url }}
|
||||
run: |
|
||||
python3 ../../scripts/ci/e2e_screenshot_status.py \
|
||||
--results-dir test-results \
|
||||
--manifest-output /tmp/e2e-screenshot-manifest.json \
|
||||
--evidence-dir /tmp/e2e-evidence \
|
||||
--artifact-url "$RESULTS_URL" \
|
||||
--output /tmp/e2e-review-status.json
|
||||
{
|
||||
echo 'review_status<<__E2E_REVIEW_STATUS__'
|
||||
cat /tmp/e2e-review-status.json
|
||||
echo '__E2E_REVIEW_STATUS__'
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The trusted workflow_run publisher consumes only this flat, bounded
|
||||
# artifact. It turns selected images into GitHub attachment URLs; it
|
||||
# never checks out or runs this PR's code.
|
||||
- name: Upload inline E2E evidence
|
||||
if: always() && github.ref_name != 'main'
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: e2e-evidence-${{ github.sha }}
|
||||
path: /tmp/e2e-evidence
|
||||
retention-days: 14
|
||||
overwrite: true
|
||||
if-no-files-found: error
|
||||
|
||||
# ── Generate step summary with visual diff info ───────────────────
|
||||
# Parse the JSON report + scan for diff images, then post a summary
|
||||
# to the GitHub Actions step output so reviewers can see what changed
|
||||
# without downloading artifacts. Runs AFTER uploads so it can link
|
||||
# the artifact download URLs from their step outputs.
|
||||
- name: Generate visual diff summary
|
||||
if: always()
|
||||
working-directory: apps/desktop
|
||||
env:
|
||||
REPORT_URL: ${{ steps.upload-report.outputs.artifact-url }}
|
||||
RESULTS_URL: ${{ steps.upload-results.outputs.artifact-url }}
|
||||
DIFFS_URL: ${{ steps.upload-diffs.outputs.artifact-url }}
|
||||
run: |
|
||||
{
|
||||
echo "## Desktop E2E — Visual Diff Report"
|
||||
echo ""
|
||||
|
||||
# Count diff images (playwright writes *-diff.png on mismatch)
|
||||
DIFF_COUNT=$(find test-results -name '*-diff.png' 2>/dev/null | wc -l)
|
||||
ACTUAL_COUNT=$(find test-results -name '*-actual.png' 2>/dev/null | wc -l)
|
||||
|
||||
if [ "$DIFF_COUNT" -eq 0 ]; then
|
||||
echo "✅ All $ACTUAL_COUNT screenshot(s) matched their baselines (or no baselines existed yet)."
|
||||
else
|
||||
echo "📸 **$DIFF_COUNT of $ACTUAL_COUNT screenshot(s) differ from baseline:**"
|
||||
echo ""
|
||||
echo "| Test | Diff | Actual | Expected |"
|
||||
echo "|------|------|--------|----------|"
|
||||
|
||||
# List each diff image with a link to the artifact
|
||||
for diff in $(find test-results -name '*-diff.png' 2>/dev/null | sort); do
|
||||
base=${diff%-diff.png}
|
||||
test_name=$(basename "$base")
|
||||
echo "| $test_name | [diff]($diff) | [actual](${base}-actual.png) | [expected](${base}-expected.png) |"
|
||||
done
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "📥 **Artifacts:**"
|
||||
echo ""
|
||||
if [ -n "$RESULTS_URL" ]; then
|
||||
echo "- [playwright-test-results]($RESULTS_URL) — all screenshots (actual + expected + diff) + traces"
|
||||
fi
|
||||
if [ -n "$REPORT_URL" ]; then
|
||||
echo "- [playwright-report]($REPORT_URL) — interactive HTML report"
|
||||
fi
|
||||
if [ -n "$DIFFS_URL" ]; then
|
||||
echo "- [visual-diffs]($DIFFS_URL) — just the diffed screenshots (small, fast to review)"
|
||||
fi
|
||||
echo ""
|
||||
echo "**To update baselines:** merge to main (baselines auto-update on main runs) or run \`npx playwright test --update-snapshots\` locally."
|
||||
|
||||
# Also parse the JSON report for pass/fail counts
|
||||
if [ -f playwright-report/results.json ]; then
|
||||
echo ""
|
||||
echo "### Test Results"
|
||||
echo ""
|
||||
node -e "
|
||||
const r = require('./playwright-report/results.json');
|
||||
const stats = r.stats || {};
|
||||
console.log('| Status | Count |');
|
||||
console.log('|--------|-------|');
|
||||
console.log('| ✅ Passed | ' + (stats.expected || 0) + ' |');
|
||||
console.log('| ❌ Failed | ' + (stats.unexpected || 0) + ' |');
|
||||
console.log('| ⏭️ Skipped | ' + (stats.skipped || 0) + ' |');
|
||||
console.log('| 🔄 Flaky | ' + (stats.flaky || 0) + ' |');
|
||||
" 2>/dev/null || true
|
||||
fi
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
@@ -15,6 +15,10 @@ name: History Check
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
outputs:
|
||||
review_status:
|
||||
description: "JSON array of review_status objects for the synthesizer."
|
||||
value: ${{ jobs.check-common-ancestor.outputs.review_status }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -23,18 +27,23 @@ jobs:
|
||||
check-common-ancestor:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
outputs:
|
||||
review_status: ${{ steps.merge-base-check.outputs.review_status }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
fetch-depth: 0 # full history both sides for merge-base
|
||||
|
||||
- name: Reject PRs with no common ancestor on main
|
||||
- id: merge-base-check
|
||||
name: Reject PRs with no common ancestor on main
|
||||
run: |
|
||||
# `git merge-base` exits non-zero AND prints nothing when the two
|
||||
# commits share no ancestor. We check both conditions explicitly
|
||||
# so the failure message is clear regardless of which signal fires
|
||||
# first.
|
||||
if ! BASE=$(git merge-base origin/main HEAD 2>/dev/null) || [ -z "$BASE" ]; then
|
||||
STATUS='[{"source":"unrelated histories","results":[{"kind":"action_required","title":"Unrelated histories","summary":"This PR has no common ancestor with main.","detail":"","how_to_fix":"Rebase your changes onto current main:\n```\ngit fetch origin main\ngit checkout -b fix-branch origin/main\n# re-apply your changes (cherry-pick, copy files, etc.)\ngit push -f origin fix-branch\n```\n"}]}]'
|
||||
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
|
||||
echo ""
|
||||
echo "::error::This PR has no common ancestor with main."
|
||||
echo ""
|
||||
@@ -56,3 +65,4 @@ jobs:
|
||||
exit 1
|
||||
fi
|
||||
echo "::notice::Common ancestor with main: $BASE"
|
||||
echo "review_status=[]" >> "$GITHUB_OUTPUT"
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
name: Infographic Check
|
||||
|
||||
# Rejects PRs that commit PR-infographic images into the repo.
|
||||
#
|
||||
# PR infographics are rendered to an image-provider URL (fal.media) and
|
||||
# embedded in the PR *description*. The PR body is the archive; the binary
|
||||
# never belongs in git history.
|
||||
#
|
||||
# This has now leaked twice. PR #48261 removed the first batch, PR #54564
|
||||
# removed a second batch and added `infographic/` to `.gitignore` — but
|
||||
# `.gitignore` only stops *accidental* `git add`. It does nothing against
|
||||
# `git add -f`, and it does nothing for a path that does not literally match
|
||||
# the ignore pattern. Nine more PNGs (~14MB) were committed in the four
|
||||
# weeks AFTER that rule landed, plus PR #70552 caught an `infograficos/`
|
||||
# spelling that sidestepped the pattern entirely.
|
||||
#
|
||||
# A passive ignore rule cannot enforce a policy. This check can.
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
outputs:
|
||||
review_status:
|
||||
description: "JSON array of review_status objects for the synthesizer."
|
||||
value: ${{ jobs.check-no-committed-infographics.outputs.review_status }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
check-no-committed-infographics:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
outputs:
|
||||
review_status: ${{ steps.infographic-check.outputs.review_status }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- id: infographic-check
|
||||
name: Reject committed PR-infographic images
|
||||
run: |
|
||||
# Match on the IMAGE, not on a directory name. Keying this to
|
||||
# `infographic/` is what let `infograficos/` through in #70552 —
|
||||
# any localized or typo'd directory would sidestep it again.
|
||||
# Instead: find tracked raster images whose path contains an
|
||||
# infographic-ish segment, in any spelling, at any depth.
|
||||
#
|
||||
# `docs/assets` and `website/` legitimately hold product imagery
|
||||
# and are excluded; those are referenced from shipped docs pages.
|
||||
OFFENDERS=$(git ls-files -z \
|
||||
| tr '\0' '\n' \
|
||||
| grep -iE '(^|/)(infograph|infograf)[^/]*/' \
|
||||
| grep -iE '\.(png|jpe?g|webp|gif)$' \
|
||||
|| true)
|
||||
|
||||
if [ -n "$OFFENDERS" ]; then
|
||||
COUNT=$(printf '%s\n' "$OFFENDERS" | wc -l | tr -d ' ')
|
||||
STATUS='[{"source":"committed infographics","results":[{"kind":"action_required","title":"PR infographic committed to the repo","summary":"Infographic images belong in the PR description, never in git.","detail":"","how_to_fix":"Untrack the image and reference the provider URL from the PR body instead:\n```\ngit rm --cached <path-to-image>\n```\nThen put it in the PR description:\n```\n## Infographic\n\n\n```\n"}]}]'
|
||||
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
|
||||
echo ""
|
||||
echo "::error::${COUNT} PR-infographic image(s) are tracked in git."
|
||||
echo ""
|
||||
printf '%s\n' "$OFFENDERS" | sed 's/^/ /'
|
||||
echo ""
|
||||
echo "PR infographics are rendered to an image-provider URL and"
|
||||
echo "embedded in the PR DESCRIPTION. The PR body is the archive —"
|
||||
echo "the binary never enters git history."
|
||||
echo ""
|
||||
echo "This rule has been re-established twice already (#48261,"
|
||||
echo "#54564) and leaked both times, because .gitignore cannot stop"
|
||||
echo "'git add -f' or a differently-spelled directory (#70552)."
|
||||
echo ""
|
||||
echo "To fix:"
|
||||
echo " git rm --cached <path> # keeps your local copy"
|
||||
echo " # then embed the provider URL in the PR description"
|
||||
exit 1
|
||||
fi
|
||||
echo "::notice::No committed PR-infographic images."
|
||||
echo "review_status=[]" >> "$GITHUB_OUTPUT"
|
||||
@@ -7,7 +7,7 @@ name: auto-fix lint issues & formatting
|
||||
# auto-corrected on merge so PRs aren't blocked by them. The PR-time eslint
|
||||
# check in typecheck.yml fails only when un-fixable errors remain.
|
||||
#
|
||||
# NOTE: AUTOFIX_BOT_PAT pushes DO trigger further workflow runs (unlike
|
||||
# NOTE: App token pushes DO trigger further workflow runs (unlike
|
||||
# secrets.GITHUB_TOKEN). The concurrency group (ts-autofix-${{ github.ref }})
|
||||
# with cancel-in-progress: true prevents an infinite loop — a re-triggered
|
||||
# run cancels the in-flight one, and since the second run finds no new fixes
|
||||
@@ -122,12 +122,20 @@ jobs:
|
||||
if: needs.generate-patch.outputs.has-fixes == 'true'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
environment: trusted-automation
|
||||
permissions:
|
||||
contents: write # needed to push to bot/js-autofix
|
||||
pull-requests: write # needed for PR creation + auto-merge
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Get GitHub App token
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ vars.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- name: Download patch
|
||||
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
|
||||
with:
|
||||
@@ -170,7 +178,7 @@ jobs:
|
||||
|
||||
- name: Create/update PR and enable auto-merge
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
GH_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
BOT_BRANCH: bot/js-autofix
|
||||
run: |
|
||||
set -euo pipefail
|
||||
@@ -193,7 +201,7 @@ jobs:
|
||||
|
||||
- name: Wait for merge, auto-close on failure or stale
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
GH_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
START_SHA: ${{ github.sha }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
@@ -10,7 +10,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
outputs:
|
||||
packages: ${{ steps.set-matrix.outputs.packages }}
|
||||
checks: ${{ steps.set-matrix.outputs.checks }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
@@ -22,21 +22,40 @@ jobs:
|
||||
command: npm ci --ignore-scripts
|
||||
- id: set-matrix
|
||||
run: |
|
||||
PACKAGES=$(npm query .workspace | jq -c '[.[].location]')
|
||||
if [ "$PACKAGES" = "[]" ] || [ -z "$PACKAGES" ]; then
|
||||
echo "::error::Workspace discovery produced an empty package list — refusing to emit a zero-length matrix (would skip all JS/TS checks silently)."
|
||||
exit 1
|
||||
fi
|
||||
echo "packages=$PACKAGES" >> "$GITHUB_OUTPUT"
|
||||
node -e '
|
||||
const { execSync } = require("child_process");
|
||||
const pkgs = JSON.parse(execSync("npm query .workspace", { encoding: "utf-8" }));
|
||||
if (pkgs.length === 0) {
|
||||
console.error("::error::Workspace discovery produced an empty package list — refusing to emit a zero-length matrix (would skip all JS/TS checks silently).");
|
||||
process.exit(1);
|
||||
}
|
||||
const checks = [];
|
||||
for (const pkg of pkgs) {
|
||||
const scripts = pkg.scripts || {};
|
||||
const subs = Object.keys(scripts).filter(s => /^check:.+$/.test(s));
|
||||
if (subs.length > 0) {
|
||||
for (const script of subs) {
|
||||
checks.push({ package: pkg.location, script });
|
||||
}
|
||||
} else if (scripts.check) {
|
||||
checks.push({ package: pkg.location, script: "check" });
|
||||
}
|
||||
}
|
||||
if (checks.length === 0) {
|
||||
console.error("::error::No check scripts found in any workspace package.");
|
||||
process.exit(1);
|
||||
}
|
||||
process.stdout.write("checks=" + JSON.stringify(checks) + "\n");
|
||||
' >> "$GITHUB_OUTPUT"
|
||||
|
||||
check:
|
||||
name: Typecheck & Test
|
||||
name: ${{ matrix.package }} / ${{ matrix.script }}
|
||||
needs: workspaces
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
strategy:
|
||||
matrix:
|
||||
package: ${{ fromJson(needs.workspaces.outputs.packages) }}
|
||||
include: ${{ fromJson(needs.workspaces.outputs.checks) }}
|
||||
fail-fast: false # report all failures, not just the first one
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
@@ -47,5 +66,4 @@ jobs:
|
||||
- uses: ./.github/actions/retry
|
||||
with:
|
||||
command: npm ci
|
||||
- run: npm run --prefix ${{ matrix.package }} check
|
||||
- run: npm run --prefix ${{ matrix.package }} fix
|
||||
- run: npm run --prefix ${{ matrix.package }} ${{ matrix.script }}
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
name: Label rerun
|
||||
|
||||
# When the ``ci-reviewed`` label is added to a PR, rerun all failed jobs in
|
||||
# the latest CI run. This re-evaluates ``review-labels`` (which now sees the
|
||||
# label) and GitHub automatically reruns dependent jobs (``comment-live``,
|
||||
# ``all-checks-pass``) — so the review comment gets updated too.
|
||||
#
|
||||
# If the CI run is still in progress when the label is added, we wait for it
|
||||
# to finish before rerunning (``gh run rerun`` only works on completed runs).
|
||||
# The wait can be long (20+ min for a full CI run), but it's better than
|
||||
# silently failing and leaving the reviewer stuck.
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [labeled]
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
pull-requests: read
|
||||
|
||||
concurrency:
|
||||
group: label-rerun-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
rerun-review-labels:
|
||||
name: Rerun review-labels job
|
||||
if: github.event.label.name == 'ci-reviewed'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 40
|
||||
steps:
|
||||
- name: Wait for CI run to finish, then rerun failed jobs
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
REPO: ${{ github.repository }}
|
||||
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
|
||||
run: |
|
||||
set -uo pipefail
|
||||
|
||||
# Find the latest CI run for this PR's head SHA.
|
||||
RUN_ID=$(gh run list \
|
||||
--repo "$REPO" \
|
||||
--commit "$HEAD_SHA" \
|
||||
--workflow ci.yml \
|
||||
--limit 1 \
|
||||
--json databaseId,status \
|
||||
--jq '.[0] | "\(.databaseId) \(.status)"' 2>/dev/null || true)
|
||||
|
||||
if [ -z "$RUN_ID" ]; then
|
||||
echo "No CI run found for this PR — nothing to rerun."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Split "RUN_ID STATUS" into two vars.
|
||||
RUN_ID="${RUN_ID%% *}"
|
||||
STATUS="${RUN_ID##* }"
|
||||
|
||||
echo "Latest CI run: $RUN_ID (status: $STATUS)"
|
||||
|
||||
# If the run is still in progress, wait for it to finish.
|
||||
# gh run rerun only works on completed runs — if we try while it's
|
||||
# running, GitHub rejects with "cannot be rerun; This workflow is
|
||||
# already running".
|
||||
if [ "$STATUS" != "completed" ]; then
|
||||
echo "Run is $STATUS — waiting for completion (this may take a while)..."
|
||||
# gh run watch --exit-status exits non-zero if the run fails,
|
||||
# which is expected (the label gate fails). Don't let that kill
|
||||
# the workflow — we WANT to rerun failed jobs.
|
||||
timeout 2100 gh run watch "$RUN_ID" --repo "$REPO" --interval 15 || true
|
||||
|
||||
# Verify it's actually completed now.
|
||||
STATUS=$(gh run view "$RUN_ID" --repo "$REPO" --json status --jq '.status' 2>/dev/null || echo "unknown")
|
||||
if [ "$STATUS" != "completed" ]; then
|
||||
echo "Run is still $STATUS after wait — giving up."
|
||||
exit 0
|
||||
fi
|
||||
fi
|
||||
|
||||
echo "Run completed. Rerunning all failed jobs..."
|
||||
gh run rerun "$RUN_ID" --repo "$REPO" --failed || true
|
||||
echo "Done. GitHub will rerun review-labels and all dependent jobs."
|
||||
+16
-120
@@ -2,11 +2,14 @@ name: Lint (ruff + ty)
|
||||
|
||||
# Two things here:
|
||||
# 1. Advisory diff — ruff + ty diagnostics as a diff vs the target branch.
|
||||
# Posts a Markdown summary and a PR comment. Exit zero always.
|
||||
# Writes a Markdown summary to the run page. Exit zero always.
|
||||
# 2. Blocking ``ruff check .`` — enforces the explicit rules in
|
||||
# ``[tool.ruff.lint.select]`` (currently PLW1514). Failure blocks merge.
|
||||
# Separate job so the advisory diff still runs and posts even when
|
||||
# enforcement fails.
|
||||
# Separate job so the advisory diff still runs even when enforcement
|
||||
# fails.
|
||||
#
|
||||
# CI-sensitive file review was previously here as a ``ci-review`` job but
|
||||
# has moved to ``review-labels.yml`` so it can be rerun independently.
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
@@ -15,14 +18,9 @@ on:
|
||||
description: The event name from the calling orchestrator (pull_request or push).
|
||||
type: string
|
||||
required: true
|
||||
ci_review:
|
||||
description: Whether CI-sensitive files (eslint config, workflows, actions) changed and require a review label.
|
||||
type: boolean
|
||||
default: false
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # needed to post/update PR comments
|
||||
|
||||
concurrency:
|
||||
group: lint-${{ github.ref }}
|
||||
@@ -42,6 +40,11 @@ jobs:
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
|
||||
# raw.githubusercontent.com every job; transient fetch failures
|
||||
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
|
||||
version: "0.9.28"
|
||||
|
||||
- name: Install ruff + ty
|
||||
uses: ./.github/actions/retry
|
||||
@@ -131,6 +134,11 @@ jobs:
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
|
||||
# raw.githubusercontent.com every job; transient fetch failures
|
||||
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
|
||||
version: "0.9.28"
|
||||
|
||||
- name: Install ruff
|
||||
uses: ./.github/actions/retry
|
||||
@@ -162,115 +170,3 @@ jobs:
|
||||
|
||||
- name: Run footgun checker
|
||||
run: python scripts/check-windows-footguns.py --all
|
||||
|
||||
ci-review:
|
||||
# Require explicit maintainer review when CI-sensitive files change:
|
||||
# eslint config, workflow YAMLs, or composite actions. These files
|
||||
# influence what code the js-autofix job executes and pushes to
|
||||
# main, so a malicious PR could inject arbitrary code via a custom eslint
|
||||
# rule's `fix` function. The label gate ensures a human reviews before
|
||||
# merge. Mirrors the mcp-catalog-reviewed pattern in supply-chain-audit.yml.
|
||||
name: CI-sensitive file review
|
||||
if: inputs.event_name == 'pull_request' && inputs.ci_review
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 2
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- name: Require ci-reviewed label
|
||||
id: label-check
|
||||
env:
|
||||
# Read-only label lookup. Use the built-in GITHUB_TOKEN (present and
|
||||
# read-only on forks) so the gate works on fork PRs; fall back to it
|
||||
# when AUTOFIX_BOT_PAT is empty. `|| true` degrades an API blip to
|
||||
# "label absent" rather than hard-failing the step.
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PR="${{ github.event.pull_request.number }}"
|
||||
LABELS=$(gh pr view "$PR" --json labels --jq '.labels[].name' || true)
|
||||
if echo "$LABELS" | grep -Fxq 'ci-reviewed'; then
|
||||
echo "reviewed=true" >> "$GITHUB_OUTPUT"
|
||||
echo "ci-reviewed label present."
|
||||
exit 0
|
||||
fi
|
||||
echo "reviewed=false" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# On failure: find the bot's previous comment and edit it, or create
|
||||
# a new one if none exists. Using an HTML comment marker so we can
|
||||
# locate it reliably across runs without parsing the body text.
|
||||
# Skipped on fork PRs — GITHUB_TOKEN is read-only there, so the API
|
||||
# call would fail. The label gate still holds via the step below.
|
||||
- name: Post or update review warning
|
||||
if: steps.label-check.outputs.reviewed != 'true' && github.event.pull_request.head.repo.fork != true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PR="${{ github.event.pull_request.number }}"
|
||||
MARKER="<!-- ci-review-bot -->"
|
||||
BODY="${MARKER}
|
||||
## ⚠️ CI-sensitive file review required
|
||||
|
||||
This PR changes CI-sensitive files (eslint config, workflow YAMLs,
|
||||
or composite actions). These files influence what code the
|
||||
js-autofix job executes and pushes to main.
|
||||
|
||||
A maintainer should verify:
|
||||
- no new eslint rules with custom \`fix\` functions that write outside linted paths,
|
||||
- no workflow changes that widen permissions or remove guards,
|
||||
- no composite action changes that alter what gets executed.
|
||||
|
||||
After review, add the \`ci-reviewed\` label and re-run this check."
|
||||
|
||||
# Find an existing comment with our marker.
|
||||
COMMENT_ID=$(gh api \
|
||||
"repos/${{ github.repository }}/issues/${PR}/comments" \
|
||||
--paginate --jq ".[] | select(.body | contains(\"${MARKER}\")) | .id" \
|
||||
| head -1 || true)
|
||||
|
||||
if [ -n "$COMMENT_ID" ]; then
|
||||
gh api --method PATCH \
|
||||
"repos/${{ github.repository }}/issues/comments/${COMMENT_ID}" \
|
||||
-f body="$BODY"
|
||||
else
|
||||
gh pr comment "$PR" --body "$BODY"
|
||||
fi
|
||||
|
||||
# Fail the job when the label is missing — always runs (including
|
||||
# fork PRs) so the security gate holds even when the comment step
|
||||
# was skipped above.
|
||||
- name: Fail on missing label
|
||||
if: steps.label-check.outputs.reviewed != 'true'
|
||||
run: |
|
||||
echo "::error::CI-sensitive changes require the ci-reviewed label."
|
||||
exit 1
|
||||
|
||||
# On success: if a previous warning comment exists, edit it to show
|
||||
# the review passed so the PR doesn't have a stale ⚠️ sitting around.
|
||||
# Skipped on fork PRs — no comment was ever posted to update.
|
||||
- name: Update previous warning to passed
|
||||
if: steps.label-check.outputs.reviewed == 'true' && github.event.pull_request.head.repo.fork != true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PR="${{ github.event.pull_request.number }}"
|
||||
MARKER="<!-- ci-review-bot -->"
|
||||
|
||||
# Find an existing comment with our marker.
|
||||
COMMENT_ID=$(gh api \
|
||||
"repos/${{ github.repository }}/issues/${PR}/comments" \
|
||||
--paginate --jq ".[] | select(.body | contains(\"${MARKER}\")) | .id" \
|
||||
| head -1 || true)
|
||||
|
||||
if [ -n "$COMMENT_ID" ]; then
|
||||
BODY="${MARKER}
|
||||
## ✅ CI-sensitive file review passed
|
||||
|
||||
The \`ci-reviewed\` label is present on this PR."
|
||||
|
||||
gh api --method PATCH \
|
||||
"repos/${{ github.repository }}/issues/comments/${COMMENT_ID}" \
|
||||
-f body="$BODY"
|
||||
fi
|
||||
|
||||
@@ -7,22 +7,25 @@ name: Lockfile diff
|
||||
# the ``packages`` map at the merge base and at HEAD and set-diffs the
|
||||
# {install path: version} maps instead.
|
||||
#
|
||||
# The comment is upserted: the script embeds a hidden HTML marker and the
|
||||
# workflow PATCHes the existing comment when one is found, so a PR gets
|
||||
# exactly one lockfile-diff comment that tracks the latest push instead
|
||||
# of a stack of stale ones. When a later push reverts all lockfile
|
||||
# changes, the comment is updated to say so (deleting it would be more
|
||||
# surprising than telling the reviewer it's resolved).
|
||||
# The semantic diff is exposed as a workflow_call output ``review_status``
|
||||
# (a JSON array in the unified status format) and an artifact
|
||||
# (``lockfile-diff`` containing the markdown fragment) for the step
|
||||
# summary.
|
||||
#
|
||||
# Never blocking — this is review signal, not enforcement. Exit is 0 even
|
||||
# when commenting fails (fork PRs get a read-only GITHUB_TOKEN).
|
||||
# Never blocking — this is review signal, not enforcement.
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
outputs:
|
||||
changed:
|
||||
description: Whether package-lock.json changed relative to the target branch.
|
||||
value: ${{ jobs.diff.outputs.changed }}
|
||||
review_status:
|
||||
description: JSON array of review status objects for the unified PR comment.
|
||||
value: ${{ jobs.diff.outputs.review_status }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: write # post/update the diff comment
|
||||
|
||||
concurrency:
|
||||
group: lockfile-diff-${{ github.event.pull_request.number || github.ref }}
|
||||
@@ -33,6 +36,9 @@ jobs:
|
||||
name: package-lock.json semantic diff
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
outputs:
|
||||
changed: ${{ steps.diff.outputs.changed }}
|
||||
review_status: ${{ steps.emit-status.outputs.review_status }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
@@ -54,45 +60,36 @@ jobs:
|
||||
--output /tmp/lockfile-diff.md
|
||||
if [ -s /tmp/lockfile-diff.md ]; then
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
cat /tmp/lockfile-diff.md >> "$GITHUB_STEP_SUMMARY"
|
||||
{
|
||||
echo "## package-lock.json semantic diff"
|
||||
echo ""
|
||||
cat /tmp/lockfile-diff.md
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
else
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
: > /tmp/lockfile-diff.md
|
||||
fi
|
||||
|
||||
- name: Post or update PR comment
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
REPO: ${{ github.repository }}
|
||||
PR: ${{ github.event.pull_request.number }}
|
||||
CHANGED: ${{ steps.diff.outputs.changed }}
|
||||
- name: Emit review_status
|
||||
id: emit-status
|
||||
run: |
|
||||
set -euo pipefail
|
||||
MARKER='<!-- hermes-lockfile-diff -->'
|
||||
CHANGED="${{ steps.diff.outputs.changed }}"
|
||||
STATUS="[]"
|
||||
|
||||
# Find our previous comment (paginated — busy PRs exceed one page).
|
||||
EXISTING=$(gh api --paginate "repos/${REPO}/issues/${PR}/comments" \
|
||||
--jq ".[] | select(.body | startswith(\"$MARKER\")) | .id" \
|
||||
| head -1 || true)
|
||||
|
||||
if [ "$CHANGED" != "true" ]; then
|
||||
if [ -n "$EXISTING" ]; then
|
||||
# A previous push changed the lockfile but the latest one
|
||||
# doesn't — update the comment rather than leave stale info.
|
||||
printf '%s\n✅ package-lock.json changes from an earlier push have been reverted — locked versions now match the target branch.\n' "$MARKER" > /tmp/lockfile-diff.md
|
||||
else
|
||||
echo "No lockfile changes and no existing comment — nothing to do."
|
||||
exit 0
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ -n "$EXISTING" ]; then
|
||||
echo "Updating existing comment ${EXISTING}"
|
||||
gh api --method PATCH "repos/${REPO}/issues/comments/${EXISTING}" \
|
||||
-F body=@/tmp/lockfile-diff.md > /dev/null \
|
||||
|| echo "::warning::Could not update PR comment (expected for fork PRs — GITHUB_TOKEN is read-only)"
|
||||
if [ "$CHANGED" = "true" ]; then
|
||||
CONTENT=$(cat /tmp/lockfile-diff.md | python3 -c "import sys,json; print(json.dumps(sys.stdin.read()))")
|
||||
STATUS="[{\"source\":\"lockfile-diff\",\"results\":[{\"kind\":\"action_required\",\"title\":\"package-lock.json\",\"summary\":\"Locked npm dependency versions changed.\",\"detail\":${CONTENT},\"how_to_fix\":\"Add the \`ci-reviewed\` label after verifying the version changes are expected.\"}]}"
|
||||
else
|
||||
echo "Creating new comment"
|
||||
gh api "repos/${REPO}/issues/${PR}/comments" \
|
||||
-F body=@/tmp/lockfile-diff.md > /dev/null \
|
||||
|| echo "::warning::Could not post PR comment (expected for fork PRs — GITHUB_TOKEN is read-only)"
|
||||
STATUS="[]"
|
||||
fi
|
||||
|
||||
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Upload diff artifact
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: lockfile-diff
|
||||
path: /tmp/lockfile-diff.md
|
||||
retention-days: 1
|
||||
overwrite: true
|
||||
|
||||
@@ -14,14 +14,14 @@ name: OSV-Scanner
|
||||
# code patterns in PR diffs) by covering the orthogonal "currently-pinned
|
||||
# dep became known-vulnerable" case.
|
||||
#
|
||||
# Steps below are inlined from Google's officially-recommended reusable
|
||||
# workflow (google/osv-scanner-action/.github/workflows/osv-scanner-reusable.yml),
|
||||
# rather than called via `uses:` so we can set a `timeout-minutes` in the
|
||||
# degenerate case where this job hangs.
|
||||
|
||||
# Uses Google's officially-recommended reusable workflow, pinned by SHA.
|
||||
# Findings land in the repo's Security tab (Code Scanning > OSV-Scanner).
|
||||
# fail-on-vuln is disabled so the job does not block merges on pre-existing
|
||||
# vulnerabilities in pinned deps that we may need to patch deliberately.
|
||||
#
|
||||
# The reusable workflow can't emit custom outputs, so a wrapper job
|
||||
# downloads the SARIF result and summarizes the vulnerability count into
|
||||
# a review_status for the unified PR comment.
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
@@ -40,62 +40,88 @@ permissions:
|
||||
jobs:
|
||||
scan:
|
||||
name: Scan lockfiles
|
||||
uses: google/osv-scanner-action/.github/workflows/osv-scanner-reusable.yml@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8
|
||||
with:
|
||||
# Scan explicit lockfiles rather than recursing, so we only look at
|
||||
# the three sources of truth and skip vendored / test / worktree dirs.
|
||||
scan-args: |-
|
||||
--lockfile=uv.lock
|
||||
--lockfile=package-lock.json
|
||||
--lockfile=website/package-lock.json
|
||||
# The upstream reusable workflow uploads this exact file under its
|
||||
# fixed artifact name, which the wrapper downloads below.
|
||||
results-file-name: osv-results.sarif
|
||||
fail-on-vuln: false
|
||||
|
||||
emit-status:
|
||||
name: Emit review status
|
||||
runs-on: ubuntu-latest
|
||||
needs: scan
|
||||
if: always()
|
||||
outputs:
|
||||
review_status: ${{ steps.emit.outputs.review_status }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: 'Run scanner'
|
||||
uses: google/osv-scanner-action/osv-scanner-action@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8
|
||||
with:
|
||||
# Scan explicit lockfiles rather than recursing, so we only look at
|
||||
# the three sources of truth and skip vendored / test / worktree dirs.
|
||||
scan-args: |-
|
||||
--output=results.json
|
||||
--format=json
|
||||
--lockfile=uv.lock
|
||||
--lockfile=package-lock.json
|
||||
--lockfile=website/package-lock.json
|
||||
continue-on-error: true
|
||||
|
||||
- name: 'Run osv-scanner-reporter'
|
||||
uses: google/osv-scanner-action/osv-reporter-action@9a498708959aeaef5ef730655706c5a1df1edbc2 # v2.3.8
|
||||
with:
|
||||
scan-args: |-
|
||||
--output=results.sarif
|
||||
--new=results.json
|
||||
--gh-annotations=false
|
||||
--fail-on-vuln=false
|
||||
|
||||
# Upload the results as artifacts (optional). Commenting out will disable uploads of run results in SARIF
|
||||
# format to the repository Actions tab.
|
||||
- name: 'Upload artifact'
|
||||
id: 'upload_artifact'
|
||||
if: ${{ !cancelled() }}
|
||||
uses: actions/upload-artifact@bbbca2ddaa5d8feaa63e36b76fdaad77386f024f # v7.0.0
|
||||
- name: Download SARIF result
|
||||
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
|
||||
with:
|
||||
name: OSV Scanner SARIF file
|
||||
path: results.sarif
|
||||
retention-days: 5
|
||||
path: /tmp/osv-results
|
||||
continue-on-error: true
|
||||
|
||||
# Upload the results to GitHub's code scanning dashboard.
|
||||
- name: 'Upload to code-scanning'
|
||||
if: ${{ !cancelled() }}
|
||||
uses: github/codeql-action/upload-sarif@cdefb33c0f6224e58673d9004f47f7cb3e328b89 # v4.31.10
|
||||
with:
|
||||
sarif_file: results.sarif
|
||||
|
||||
- name: 'Print Code Scanning URL'
|
||||
if: ${{ !cancelled() }}
|
||||
- name: Emit review_status
|
||||
id: emit
|
||||
run: |
|
||||
echo "View the OSV-Scanner results in the 'Security' tab, using the following link:"
|
||||
echo "${{ github.server_url }}/${{ github.repository }}/security/code-scanning?query=is%3Aopen+branch%3A${GITHUB_REF_NAME}+tool%3Aosv-scanner"
|
||||
env:
|
||||
GITHUB_REF_NAME: ${{ github.ref_name }}
|
||||
set -euo pipefail
|
||||
STATUS="[]"
|
||||
|
||||
- name: 'Error troubleshooter'
|
||||
if: ${{ always() && steps.upload_artifact.outcome == 'failure' }}
|
||||
run: |
|
||||
echo "::error::Artifact upload failed. This is most likely caused by a error during scanning earlier in the workflow."
|
||||
exit 1
|
||||
if [ -f /tmp/osv-results/osv-results.sarif ]; then
|
||||
# Count vulnerabilities from the SARIF file
|
||||
VULN_COUNT=$(python3 -c "
|
||||
import json, sys
|
||||
try:
|
||||
with open('/tmp/osv-results/osv-results.sarif') as f:
|
||||
data = json.load(f)
|
||||
count = 0
|
||||
vulns = []
|
||||
for run in data.get('runs', []):
|
||||
for result in run.get('results', []):
|
||||
count += 1
|
||||
rule_id = result.get('ruleId', 'unknown')
|
||||
message = result.get('message', {}).get('text', '')
|
||||
loc = result.get('locations', [{}])[0].get('physicalLocation', {}).get('artifactLocation', {}).get('uri', '')
|
||||
vulns.append(f'- {rule_id} in {loc}: {message}')
|
||||
print(count)
|
||||
if vulns:
|
||||
print('\n'.join(vulns[:20]), file=sys.stderr)
|
||||
except Exception:
|
||||
print(0)
|
||||
")
|
||||
|
||||
VULN_DETAIL=""
|
||||
if [ "$VULN_COUNT" -gt 0 ] 2>/dev/null; then
|
||||
VULN_PLURAL=$([ "$VULN_COUNT" -eq 1 ] && echo "y" || echo "ies")
|
||||
VULN_DETAIL=$(python3 -c "
|
||||
import json, sys
|
||||
try:
|
||||
with open('/tmp/osv-results/osv-results.sarif') as f:
|
||||
data = json.load(f)
|
||||
vulns = []
|
||||
for run in data.get('runs', []):
|
||||
for result in run.get('results', []):
|
||||
rule_id = result.get('ruleId', 'unknown')
|
||||
loc = result.get('locations', [{}])[0].get('physicalLocation', {}).get('artifactLocation', {}).get('uri', '')
|
||||
vulns.append(f'- {rule_id} in {loc}')
|
||||
print(json.dumps('\n'.join(vulns[:20])))
|
||||
except Exception:
|
||||
print(json.dumps(''))
|
||||
")
|
||||
STATUS="[{\"source\":\"osv scan\",\"results\":[{\"kind\":\"warning\",\"title\":\"OSV vulnerability scan\",\"summary\":\"${VULN_COUNT} known vulnerabilit${VULN_PLURAL} found in pinned dependencies.\",\"detail\":${VULN_DETAIL},\"how_to_fix\":\"Review the findings in the [Security tab](../../security/code-scanning). Update the affected dependencies if a patched version is available.\"}]}]"
|
||||
else
|
||||
STATUS="[]"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo "review_status=${STATUS}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
name: Publish E2E evidence
|
||||
|
||||
# This runs only from the default branch after CI completes. It intentionally
|
||||
# checks out main, never the PR ref, and treats the downloaded artifact as
|
||||
# untrusted input before uploading validated GitHub attachments.
|
||||
on:
|
||||
workflow_run:
|
||||
workflows: [CI]
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
pull-requests: write
|
||||
|
||||
concurrency:
|
||||
group: publish-e2e-evidence-${{ github.event.workflow_run.id }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
name: Publish inline E2E evidence
|
||||
if: github.event.workflow_run.event == 'pull_request'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
environment: gh-image
|
||||
steps:
|
||||
- name: Check out trusted publisher
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
ref: ${{ github.event.repository.default_branch }}
|
||||
persist-credentials: false
|
||||
|
||||
# v1.2.0 resolves to 44f4b93ecbbe22de6c45fa2f62f519aee564ca8c.
|
||||
- name: Install gh-image
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: gh extension install drogers0/gh-image --pin v1.2.0
|
||||
|
||||
- name: Download and attach evidence
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
GH_SESSION_TOKEN: ${{ secrets.GH_IMAGE_SESSION_TOKEN }}
|
||||
SOURCE_REPO: ${{ github.repository }}
|
||||
SOURCE_RUN_ID: ${{ github.event.workflow_run.id }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
PR_NUMBER=$(gh api "repos/$SOURCE_REPO/actions/runs/$SOURCE_RUN_ID" --jq '.pull_requests[0].number // empty')
|
||||
if [ -z "$PR_NUMBER" ]; then
|
||||
echo "No pull request is associated with CI run $SOURCE_RUN_ID."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
ARTIFACT_NAME=$(gh api "repos/$SOURCE_REPO/actions/runs/$SOURCE_RUN_ID/artifacts" \
|
||||
--jq '.artifacts[] | select(.expired == false and (.name | startswith("e2e-evidence-"))) | .name' \
|
||||
| python3 -c 'import sys; print(next(iter(sys.stdin), "").strip())')
|
||||
if [ -z "$ARTIFACT_NAME" ]; then
|
||||
echo "No E2E evidence artifact was produced for CI run $SOURCE_RUN_ID."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
EVIDENCE_DIR="$RUNNER_TEMP/e2e-evidence"
|
||||
mkdir -p "$EVIDENCE_DIR"
|
||||
gh run download "$SOURCE_RUN_ID" --repo "$SOURCE_REPO" --name "$ARTIFACT_NAME" --dir "$EVIDENCE_DIR"
|
||||
|
||||
python3 scripts/ci/publish_e2e_evidence.py \
|
||||
--evidence-dir "$EVIDENCE_DIR" \
|
||||
--source-repo "$SOURCE_REPO" \
|
||||
--pr-number "$PR_NUMBER"
|
||||
@@ -0,0 +1,109 @@
|
||||
name: Review labels
|
||||
|
||||
# Require explicit maintainer review when CI-sensitive files or the MCP
|
||||
# catalog change. Previously this was split across two jobs in two
|
||||
# workflows: ``ci-review`` in lint.yml (gated on ``ci_review``) and
|
||||
# ``mcp-catalog-review`` in supply-chain-audit.yml (gated on
|
||||
# ``mcp_catalog``). Both checked for their own label.
|
||||
#
|
||||
# Now consolidated: a single ``ci-reviewed`` label covers both. The
|
||||
# comment sections tell the reviewer exactly what to verify per area,
|
||||
# so one label is enough — the human reads the comment, not the label
|
||||
# name.
|
||||
#
|
||||
# Outputs:
|
||||
# ci_reviewed — "true" / "false" / "" (empty when neither lane ran)
|
||||
# review_status — JSON array of status objects consumed by the review
|
||||
# comment assembler. See scripts/ci/emit_review_status.py.
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
ci_review:
|
||||
description: Whether CI-sensitive files (eslint config, workflows, actions) changed.
|
||||
type: boolean
|
||||
default: false
|
||||
ci_review_files:
|
||||
description: JSON list of CI-sensitive files changed by the pull request.
|
||||
type: string
|
||||
default: '[]'
|
||||
mcp_catalog:
|
||||
description: Whether the MCP catalog / installer changed.
|
||||
type: boolean
|
||||
default: false
|
||||
supply_chain:
|
||||
description: Whether the critical supply-chain scan found a risk requiring review.
|
||||
type: boolean
|
||||
default: false
|
||||
outputs:
|
||||
ci_reviewed:
|
||||
description: Whether the ci-reviewed label is present. Empty when neither input was true.
|
||||
value: ${{ jobs.check.outputs.ci_reviewed }}
|
||||
review_status:
|
||||
description: JSON array of status objects for the review comment assembler.
|
||||
value: ${{ jobs.check.outputs.review_status }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: read # read PR labels
|
||||
|
||||
jobs:
|
||||
check:
|
||||
name: Review label gate
|
||||
if: inputs.ci_review || inputs.mcp_catalog || inputs.supply_chain
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 2
|
||||
outputs:
|
||||
ci_reviewed: ${{ steps.label-check.outputs.ci_reviewed }}
|
||||
review_status: ${{ steps.build-status.outputs.review_status }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Check ci-reviewed label
|
||||
id: label-check
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
REPO: ${{ github.repository }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PR="${{ github.event.pull_request.number }}"
|
||||
LABELS=$(gh pr view "$PR" --repo "$REPO" --json labels --jq '.labels[].name' || true)
|
||||
|
||||
if echo "$LABELS" | grep -Fxq 'ci-reviewed'; then
|
||||
echo "ci-reviewed label present."
|
||||
echo "ci_reviewed=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "ci-reviewed label missing."
|
||||
echo "ci_reviewed=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Build review_status JSON
|
||||
id: build-status
|
||||
env:
|
||||
CI_REVIEW: ${{ inputs.ci_review }}
|
||||
CI_REVIEW_FILES: ${{ inputs.ci_review_files }}
|
||||
MCP_CATALOG: ${{ inputs.mcp_catalog }}
|
||||
SUPPLY_CHAIN: ${{ inputs.supply_chain }}
|
||||
LABEL_PRESENT: ${{ steps.label-check.outputs.ci_reviewed }}
|
||||
REPO_URL: ${{ github.server_url }}/${{ github.repository }}
|
||||
BASE_SHA: ${{ github.event.pull_request.base.sha }}
|
||||
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
args=()
|
||||
if [ "$CI_REVIEW" = "true" ]; then args+=(--ci-review); fi
|
||||
args+=(--ci-review-files "$CI_REVIEW_FILES")
|
||||
if [ "$MCP_CATALOG" = "true" ]; then args+=(--mcp-catalog); fi
|
||||
if [ "$SUPPLY_CHAIN" = "true" ]; then args+=(--supply-chain); fi
|
||||
if [ "$LABEL_PRESENT" = "true" ]; then args+=(--label-present); fi
|
||||
|
||||
python3 scripts/ci/emit_review_status.py "${args[@]}" \
|
||||
--repo-url "$REPO_URL" --base-sha "$BASE_SHA" --head-sha "$HEAD_SHA" \
|
||||
--output "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Fail on missing label
|
||||
if: steps.label-check.outputs.ci_reviewed != 'true'
|
||||
run: |
|
||||
echo "::error::CI-sensitive changes require the ci-reviewed label. Add the label and re-run this check."
|
||||
exit 1
|
||||
@@ -21,6 +21,7 @@ jobs:
|
||||
if: github.repository == 'NousResearch/hermes-agent'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
environment: trusted-automation
|
||||
steps:
|
||||
- name: Probe live index
|
||||
id: probe
|
||||
@@ -108,10 +109,18 @@ jobs:
|
||||
echo "Summary: ${{ steps.probe.outputs.summary }}"
|
||||
fi
|
||||
|
||||
- name: Get GitHub App token
|
||||
if: steps.probe.outputs.status != 'ok'
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ vars.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- name: Open issue on degraded / failed probe
|
||||
if: steps.probe.outputs.status != 'ok'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
GH_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
STATUS: ${{ steps.probe.outputs.status }}
|
||||
DETAIL: ${{ steps.probe.outputs.detail }}
|
||||
run: |
|
||||
|
||||
@@ -21,9 +21,17 @@ jobs:
|
||||
if: github.repository == 'NousResearch/hermes-agent'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
environment: trusted-automation
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Get GitHub App token
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ vars.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
|
||||
- uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
with:
|
||||
python-version: "3.11"
|
||||
@@ -35,7 +43,7 @@ jobs:
|
||||
|
||||
- name: Build skills index
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
GITHUB_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
run: python scripts/build_skills_index.py
|
||||
|
||||
- name: Upload index artifact
|
||||
@@ -53,8 +61,15 @@ jobs:
|
||||
if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
environment: trusted-automation
|
||||
steps:
|
||||
- name: Get GitHub App token
|
||||
id: app-token
|
||||
uses: ./.github/actions/get-app-token
|
||||
with:
|
||||
client-id: ${{ vars.APP_CLIENT_ID }}
|
||||
private-key: ${{ secrets.APP_PRIVATE_KEY }}
|
||||
- name: Trigger Deploy Site workflow
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
GH_TOKEN: ${{ steps.app-token.outputs.token }}
|
||||
run: gh workflow run deploy-site.yml --repo ${{ github.repository }} -f skills_index_run_id=${{ github.run_id }}
|
||||
|
||||
@@ -10,9 +10,18 @@ name: Supply Chain Audit
|
||||
# advisory-only workflow instead.
|
||||
#
|
||||
# Path-gating is handled centrally by the ``ci.yml`` orchestrator's
|
||||
# ``detect`` job. The orchestrator passes ``scan`` / ``deps`` /
|
||||
# ``mcp_catalog`` booleans as inputs; this workflow's jobs gate on those
|
||||
# inputs instead of re-computing the diff.
|
||||
# ``detect`` job. The orchestrator passes ``scan`` / ``deps`` booleans as
|
||||
# inputs; this workflow's jobs gate on those inputs instead of re-computing
|
||||
# the diff. MCP catalog review was previously here but has moved to
|
||||
# ``review-labels.yml`` so it can be rerun independently.
|
||||
#
|
||||
# Outputs:
|
||||
# review_status — JSON array of status objects consumed by the review
|
||||
# comment assembler (scripts/ci/assemble_review_comment.py).
|
||||
# critical_findings — "true" when the narrow critical-pattern scan found
|
||||
# something. The review-label gate consumes this and
|
||||
# owns the action-required result, so adding
|
||||
# ``ci-reviewed`` can heal the run on rerun.
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
@@ -29,10 +38,13 @@ on:
|
||||
description: Whether pyproject.toml changed.
|
||||
type: boolean
|
||||
required: true
|
||||
mcp_catalog:
|
||||
description: Whether the MCP catalog / installer changed.
|
||||
type: boolean
|
||||
required: true
|
||||
outputs:
|
||||
review_status:
|
||||
description: JSON array of review status objects for the review comment assembler.
|
||||
value: ${{ jobs.aggregate.outputs.review_status }}
|
||||
critical_findings:
|
||||
description: Whether the critical-pattern scan found a risk requiring maintainer review.
|
||||
value: ${{ jobs.aggregate.outputs.critical_findings }}
|
||||
|
||||
permissions:
|
||||
pull-requests: write
|
||||
@@ -44,6 +56,9 @@ jobs:
|
||||
if: inputs.scan
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
outputs:
|
||||
review_status: ${{ steps.emit-status.outputs.review_status }}
|
||||
critical_findings: ${{ steps.scan.outputs.found }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
@@ -53,7 +68,8 @@ jobs:
|
||||
- name: Scan diff for critical patterns
|
||||
id: scan
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
CI_REVIEWED: ${{ contains(github.event.pull_request.labels.*.name, 'ci-reviewed') }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
@@ -61,7 +77,7 @@ jobs:
|
||||
HEAD="${{ github.event.pull_request.head.sha }}"
|
||||
|
||||
# Added lines only, excluding lockfiles.
|
||||
# Three-dot diff (base...head) diffs from the merge base to HEAD,
|
||||
# Three-point diff (base...head) diffs from the merge base to HEAD,
|
||||
# so only changes introduced by this PR are included — not changes
|
||||
# that landed on main after the PR branched off.
|
||||
DIFF=$(git diff "$BASE"..."$HEAD" -- . ':!uv.lock' ':!*.lock' ':!package-lock.json' ':!yarn.lock' || true)
|
||||
@@ -71,7 +87,7 @@ jobs:
|
||||
# --- .pth files (auto-execute on Python startup) ---
|
||||
# The exact mechanism used in the litellm supply chain attack:
|
||||
# https://github.com/BerriAI/litellm/issues/24512
|
||||
PTH_FILES=$(git diff --name-only "$BASE"..."$HEAD" | grep '\.pth$' || true)
|
||||
PTH_FILES=$(git diff --diff-filter=d --name-only "$BASE"..."$HEAD" | grep '\.pth$' || true)
|
||||
if [ -n "$PTH_FILES" ]; then
|
||||
FINDINGS="${FINDINGS}
|
||||
### 🚨 CRITICAL: .pth file added or modified
|
||||
@@ -119,8 +135,11 @@ jobs:
|
||||
# auto-loaded by the interpreter via site.py. Any nested file with the
|
||||
# same name (e.g. hermes_cli/setup.py — the CLI setup wizard) is unrelated
|
||||
# and produced false positives that trained reviewers to ignore the scanner.
|
||||
SETUP_HITS=$(git diff --name-only "$BASE"..."$HEAD" | grep -E '^(setup\.py|setup\.cfg|sitecustomize\.py|usercustomize\.py|__init__\.pth)$' || true)
|
||||
if [ -n "$SETUP_HITS" ]; then
|
||||
SETUP_HITS=$(git diff --diff-filter=d --name-only "$BASE"..."$HEAD" | grep -E '^(setup\.py|setup\.cfg|sitecustomize\.py|usercustomize\.py|__init__\.pth)$' || true)
|
||||
# A maintainer-applied ci-reviewed label records the manual review
|
||||
# required for intentional changes to an install hook. The scanner
|
||||
# still blocks every unreviewed addition or modification.
|
||||
if [ -n "$SETUP_HITS" ] && [ "$CI_REVIEWED" != "true" ]; then
|
||||
FINDINGS="${FINDINGS}
|
||||
### 🚨 CRITICAL: Install-hook file added or modified
|
||||
These files can execute code during package installation or interpreter startup.
|
||||
@@ -139,33 +158,32 @@ jobs:
|
||||
echo "found=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Post critical finding comment
|
||||
if: steps.scan.outputs.found == 'true'
|
||||
- name: Emit review_status
|
||||
id: emit-status
|
||||
if: always()
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
FOUND: ${{ steps.scan.outputs.found }}
|
||||
run: |
|
||||
BODY="## 🚨 CRITICAL Supply Chain Risk Detected
|
||||
python3 - <<'PYEOF'
|
||||
import json, os
|
||||
|
||||
This PR contains a pattern that has been used in real supply chain attacks. A maintainer must review the flagged code carefully before merging.
|
||||
# The review-label gate renders and blocks critical findings. Keep
|
||||
# this scan a fact-finder so adding ci-reviewed can rerun the gate
|
||||
# without requiring the scanner itself to fail again.
|
||||
status = []
|
||||
|
||||
$(cat /tmp/findings.md)
|
||||
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as f:
|
||||
f.write(f"review_status={json.dumps(status)}\n")
|
||||
PYEOF
|
||||
|
||||
---
|
||||
*Scanner only fires on high-signal indicators: .pth files, base64+exec/eval combos, subprocess with encoded commands, or install-hook files. Low-signal warnings were removed intentionally — if you're seeing this comment, the finding is worth inspecting.*"
|
||||
|
||||
gh pr comment "${{ github.event.pull_request.number }}" --body "$BODY" || echo "::warning::Could not post PR comment (expected for fork PRs — GITHUB_TOKEN is read-only)"
|
||||
|
||||
- name: Fail on critical findings
|
||||
if: steps.scan.outputs.found == 'true'
|
||||
run: |
|
||||
echo "::error::CRITICAL supply chain risk patterns detected in this PR. See the PR comment for details."
|
||||
exit 1
|
||||
|
||||
dep-bounds:
|
||||
name: Check PyPI dependency upper bounds
|
||||
if: inputs.deps
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
outputs:
|
||||
review_status: ${{ steps.emit-status.outputs.review_status }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
@@ -188,7 +206,7 @@ jobs:
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Match PyPI dep specs that have >= but no < ceiling.
|
||||
# Match PyPI dep specs that have >= and no < ceiling.
|
||||
# Pattern: "package>=version" without a following ",<" bound.
|
||||
# Excludes git+ URLs (which use commit SHAs) and comments.
|
||||
UNBOUNDED=$(echo "$ADDED" | grep -oE '"[a-zA-Z0-9_-]+(\[[^\]]*\])?>=[ 0-9.]+"' | grep -v ',<' || true)
|
||||
@@ -200,26 +218,36 @@ jobs:
|
||||
echo "found=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Post unbounded dep warning
|
||||
if: steps.bounds.outputs.found == 'true'
|
||||
- name: Emit review_status
|
||||
id: emit-status
|
||||
if: always()
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
FOUND: ${{ steps.bounds.outputs.found }}
|
||||
run: |
|
||||
BODY="## ⚠️ Unbounded PyPI Dependency Detected
|
||||
python3 - <<'PYEOF'
|
||||
import json, os
|
||||
|
||||
This PR adds PyPI dependencies without a \`<next_major\` upper bound. Per our [supply chain policy](../blob/main/CONTRIBUTING.md#dependency-pinning-policy-supply-chain-hardening), all PyPI deps must be pinned as \`>=floor,<next_major\`.
|
||||
found = os.environ.get("FOUND", "") == "true"
|
||||
|
||||
**Unbounded specs found:**
|
||||
\`\`\`
|
||||
$(cat /tmp/unbounded.txt)
|
||||
\`\`\`
|
||||
if found:
|
||||
with open("/tmp/unbounded.txt", encoding="utf-8") as f:
|
||||
detail = f.read()
|
||||
status = [{
|
||||
"source": "supply chain",
|
||||
"results": [{
|
||||
"kind": "action_required",
|
||||
"title": "Unbounded PyPI dependencies",
|
||||
"summary": "This PR adds PyPI dependencies without upper bounds.",
|
||||
"detail": detail,
|
||||
"how_to_fix": 'Add a `<next_major` upper bound, e.g. `"package>=1.2.0,<2"`. See CONTRIBUTING.md dependency pinning policy.'
|
||||
}]
|
||||
}]
|
||||
else:
|
||||
status = []
|
||||
|
||||
**Fix:** Add an upper bound, e.g. \`"package>=1.2.0,<2"\`
|
||||
|
||||
---
|
||||
*See PR #2810 and CONTRIBUTING.md for the full policy rationale.*"
|
||||
|
||||
gh pr comment "${{ github.event.pull_request.number }}" --body "$BODY" || echo "::warning::Could not post PR comment (expected for fork PRs)"
|
||||
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as f:
|
||||
f.write(f"review_status={json.dumps(status)}\n")
|
||||
PYEOF
|
||||
|
||||
- name: Fail on unbounded deps
|
||||
if: steps.bounds.outputs.found == 'true'
|
||||
@@ -227,45 +255,39 @@ jobs:
|
||||
echo "::error::PyPI dependencies without upper bounds detected. Add <next_major ceiling per CONTRIBUTING.md policy."
|
||||
exit 1
|
||||
|
||||
mcp-catalog-review:
|
||||
name: MCP catalog security review
|
||||
if: inputs.mcp_catalog
|
||||
aggregate:
|
||||
name: Aggregate review statuses
|
||||
needs: [scan, dep-bounds]
|
||||
if: always()
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
outputs:
|
||||
review_status: ${{ steps.merge.outputs.review_status }}
|
||||
critical_findings: ${{ steps.merge.outputs.critical_findings }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Require explicit MCP catalog review label
|
||||
- name: Merge review statuses
|
||||
id: merge
|
||||
env:
|
||||
# Read-only label lookup. Use the built-in GITHUB_TOKEN (present and
|
||||
# read-only on forks) so the gate works on fork PRs; fall back to it
|
||||
# when AUTOFIX_BOT_PAT is empty. `|| true` degrades an API blip to
|
||||
# "label absent" rather than hard-failing the step.
|
||||
GH_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT || github.token }}
|
||||
SCAN_STATUS: ${{ needs.scan.outputs.review_status }}
|
||||
DEP_STATUS: ${{ needs.dep-bounds.outputs.review_status }}
|
||||
CRITICAL_FINDINGS: ${{ needs.scan.outputs.critical_findings }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PR="${{ github.event.pull_request.number }}"
|
||||
LABELS=$(gh pr view "$PR" --json labels --jq '.labels[].name' || true)
|
||||
if echo "$LABELS" | grep -Fxq 'mcp-catalog-reviewed'; then
|
||||
echo "MCP catalog review label present."
|
||||
exit 0
|
||||
fi
|
||||
python3 - <<'PYEOF'
|
||||
import json, os
|
||||
|
||||
BODY="## ⚠️ MCP catalog security review required
|
||||
merged = []
|
||||
for key in ("SCAN_STATUS", "DEP_STATUS"):
|
||||
raw = os.environ.get(key, "")
|
||||
if not raw:
|
||||
continue
|
||||
try:
|
||||
data = json.loads(raw)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
continue
|
||||
if isinstance(data, list):
|
||||
merged.extend(data)
|
||||
|
||||
This PR changes the bundled MCP catalog or MCP catalog installer code. MCP entries can define local commands that users later install into \`mcp_servers\`, so this needs explicit maintainer review before merge.
|
||||
|
||||
A maintainer should verify:
|
||||
- any new/changed \`optional-mcps/**/manifest.yaml\` command and args are expected,
|
||||
- stdio transports do not use shell+egress/exfiltration payloads,
|
||||
- git install refs are pinned and bootstrap commands are minimal,
|
||||
- requested env vars/secrets match the upstream MCP's documented needs.
|
||||
|
||||
After review, add the \`mcp-catalog-reviewed\` label and re-run this check."
|
||||
|
||||
gh pr comment "$PR" --body "$BODY" || echo "::warning::Could not post PR comment (expected for fork PRs)"
|
||||
echo "::error::MCP catalog changes require the mcp-catalog-reviewed label."
|
||||
exit 1
|
||||
with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as f:
|
||||
f.write(f"review_status={json.dumps(merged)}\n")
|
||||
f.write("critical_findings=" + os.environ.get("CRITICAL_FINDINGS", "false") + "\n")
|
||||
PYEOF
|
||||
|
||||
@@ -74,6 +74,11 @@ jobs:
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pin the uv version: unpinned, setup-uv resolves "latest" by
|
||||
# fetching a manifest from raw.githubusercontent.com on EVERY job —
|
||||
# a transient fetch failure fails the whole job (2026-07-28 slice-5
|
||||
# incident). Pinned, the binary downloads directly; no manifest hop.
|
||||
version: "0.9.28"
|
||||
# Persist uv's download/wheel cache (~/.cache/uv) across runs.
|
||||
# Keyed on the dependency manifests, so the cache is reused until
|
||||
# pyproject.toml or uv.lock changes. `uv sync` still runs every
|
||||
@@ -92,9 +97,18 @@ jobs:
|
||||
# fails if the lock is out of sync with pyproject.toml), giving a
|
||||
# reproducible env. It also creates .venv itself, so no separate
|
||||
# `uv venv` step is needed.
|
||||
#
|
||||
# The trailing extras beyond all/dev are the lazy-install features
|
||||
# (tools/lazy_deps.py) that tests exercise for real: provider.anthropic,
|
||||
# stt/tts.mistral, image.fal, terminal.modal, terminal.daytona,
|
||||
# memory.hindsight, search.parallel. The hermetic test env forbids
|
||||
# mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
|
||||
# tests/conftest.py), so the SDKs those tests need must be in the
|
||||
# venv up front — resolved from uv.lock like everything else, which
|
||||
# also honors the exact supply-chain pins these extras carry.
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
|
||||
|
||||
- name: Minimize uv cache
|
||||
# Optimized for CI: prunes pre-built wheels that are cheap to
|
||||
@@ -188,6 +202,11 @@ jobs:
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pin the uv version: unpinned, setup-uv resolves "latest" by
|
||||
# fetching a manifest from raw.githubusercontent.com on EVERY job —
|
||||
# a transient fetch failure fails the whole job (2026-07-28 slice-5
|
||||
# incident). Pinned, the binary downloads directly; no manifest hop.
|
||||
version: "0.9.28"
|
||||
# Persist uv's download/wheel cache (~/.cache/uv) across runs.
|
||||
# Keyed on the dependency manifests, so the cache is reused until
|
||||
# pyproject.toml or uv.lock changes. `uv sync` still runs every
|
||||
@@ -206,20 +225,20 @@ jobs:
|
||||
# fails if the lock is out of sync with pyproject.toml), giving a
|
||||
# reproducible env. It also creates .venv itself, so no separate
|
||||
# `uv venv` step is needed.
|
||||
#
|
||||
# Same extras as the test job's sync above: the hermetic test env
|
||||
# forbids mid-run pip installs (HERMES_DISABLE_LAZY_INSTALLS=1 in
|
||||
# tests/conftest.py), so lazy-install SDKs exercised by tests must be
|
||||
# in the venv up front.
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev
|
||||
command: uv sync --locked --python 3.11 --extra all --extra dev --extra anthropic --extra mistral --extra fal --extra modal --extra daytona --extra hindsight --extra parallel-web
|
||||
|
||||
- name: Minimize uv cache
|
||||
# Optimized for CI: prunes pre-built wheels that are cheap to
|
||||
# re-download, keeping the persisted cache small and fast to restore.
|
||||
run: uv cache prune --ci
|
||||
|
||||
- name: Packaged-wheel i18n smoke test
|
||||
run: |
|
||||
source .venv/bin/activate
|
||||
python -m pytest -m integration tests/test_wheel_locales_e2e.py -v
|
||||
|
||||
- name: Run e2e tests
|
||||
run: |
|
||||
source .venv/bin/activate
|
||||
|
||||
@@ -1,181 +0,0 @@
|
||||
name: Publish to PyPI
|
||||
|
||||
# Triggered by CalVer tag pushes from scripts/release.py (e.g. v2026.5.15)
|
||||
# Can also be triggered manually from the Actions tab as an escape hatch.
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- "v20*" # CalVer tags: v2026.5.15, v2026.5.15.2, etc.
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
confirm_tag:
|
||||
description: "Tag to publish (e.g. v2026.5.15). Must already exist."
|
||||
required: true
|
||||
type: string
|
||||
|
||||
# Restrict default token to read-only; each job escalates as needed.
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# Prevent overlapping publishes (e.g. two same-day tags pushed quickly).
|
||||
concurrency:
|
||||
group: pypi-publish
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: Build distribution 📦
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
# On workflow_dispatch, check out the confirmed tag.
|
||||
ref: ${{ inputs.confirm_tag || github.ref }}
|
||||
fetch-tags: true
|
||||
|
||||
- name: Validate tag exists
|
||||
if: github.event_name == 'workflow_dispatch'
|
||||
run: |
|
||||
if ! git tag -l "${{ inputs.confirm_tag }}" | grep -q .; then
|
||||
echo "::error::Tag '${{ inputs.confirm_tag }}' does not exist in the repo"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
|
||||
with:
|
||||
python-version: "3.13"
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: "22"
|
||||
|
||||
- name: Build web dashboard
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: npm ci
|
||||
working-directory: web
|
||||
|
||||
- name: Compile web dashboard
|
||||
run: npm run build
|
||||
working-directory: web
|
||||
|
||||
- name: Build TUI bundle
|
||||
uses: ./.github/actions/retry
|
||||
with:
|
||||
command: npm ci
|
||||
working-directory: ui-tui
|
||||
|
||||
- name: Compile TUI bundle
|
||||
run: npm run build
|
||||
working-directory: ui-tui
|
||||
|
||||
- name: Bundle TUI into hermes_cli
|
||||
run: |
|
||||
mkdir -p hermes_cli/tui_dist
|
||||
cp ui-tui/dist/entry.js hermes_cli/tui_dist/entry.js
|
||||
|
||||
- name: Verify frontend assets exist
|
||||
run: |
|
||||
test -f hermes_cli/web_dist/index.html || { echo "ERROR: web_dist not built"; exit 1; }
|
||||
test -f hermes_cli/tui_dist/entry.js || { echo "ERROR: tui_dist not built"; exit 1; }
|
||||
|
||||
- name: Bundle install scripts into wheel
|
||||
run: |
|
||||
mkdir -p hermes_cli/scripts
|
||||
cp scripts/install.sh hermes_cli/scripts/install.sh
|
||||
cp scripts/install.ps1 hermes_cli/scripts/install.ps1
|
||||
|
||||
- name: Build wheel and sdist
|
||||
run: uv build --sdist --wheel
|
||||
|
||||
- name: Upload distribution artifacts
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: python-package-distributions
|
||||
path: dist/
|
||||
|
||||
publish:
|
||||
name: Publish to PyPI
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
environment:
|
||||
name: pypi
|
||||
url: https://pypi.org/p/hermes-agent
|
||||
permissions:
|
||||
id-token: write # OIDC trusted publishing
|
||||
|
||||
steps:
|
||||
- name: Download distribution artifacts
|
||||
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
|
||||
with:
|
||||
name: python-package-distributions
|
||||
path: dist/
|
||||
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # v1.14.0
|
||||
with:
|
||||
skip-existing: true
|
||||
|
||||
sign:
|
||||
name: Sign and attach to GitHub Release
|
||||
# Only runs on tag pushes — release.py creates the GitHub Release,
|
||||
# and workflow_dispatch won't have a matching release to attach to.
|
||||
if: startsWith(github.ref, 'refs/tags/')
|
||||
needs: publish
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: write # attach assets to the existing release
|
||||
id-token: write # sigstore signing
|
||||
|
||||
steps:
|
||||
- name: Download distribution artifacts
|
||||
uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
|
||||
with:
|
||||
name: python-package-distributions
|
||||
path: dist/
|
||||
|
||||
- name: Wait for GitHub Release to exist
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
# release.py creates the GitHub Release after pushing the tag,
|
||||
# but this workflow starts from the tag push — wait for it.
|
||||
run: |
|
||||
for i in $(seq 1 30); do
|
||||
if gh release view "$GITHUB_REF_NAME" --repo "$GITHUB_REPOSITORY" >/dev/null 2>&1; then
|
||||
echo "Release $GITHUB_REF_NAME found"
|
||||
exit 0
|
||||
fi
|
||||
echo "Waiting for release... ($i/30)"
|
||||
sleep 10
|
||||
done
|
||||
echo "::warning::Release $GITHUB_REF_NAME not found after 5 minutes — skipping signature upload"
|
||||
echo "skip_sign=true" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Sign with Sigstore
|
||||
if: env.skip_sign != 'true'
|
||||
uses: sigstore/gh-action-sigstore-python@04cffa1d795717b140764e8b640de88853c92acc # v3.3.0
|
||||
with:
|
||||
inputs: >-
|
||||
./dist/*.tar.gz
|
||||
./dist/*.whl
|
||||
|
||||
- name: Attach signed artifacts to GitHub Release
|
||||
if: env.skip_sign != 'true'
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.AUTOFIX_BOT_PAT }}
|
||||
# release.py already created the GitHub Release — just upload
|
||||
# the Sigstore signatures alongside the existing assets.
|
||||
run: >-
|
||||
gh release upload
|
||||
"$GITHUB_REF_NAME" dist/*.sigstore.json
|
||||
--repo "$GITHUB_REPOSITORY"
|
||||
--clobber
|
||||
@@ -45,6 +45,10 @@ name: uv.lock check
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
outputs:
|
||||
review_status:
|
||||
description: "JSON review status for the review-status aggregator"
|
||||
value: ${{ jobs.check.outputs.review_status }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -58,12 +62,19 @@ jobs:
|
||||
name: uv lock --check
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
outputs:
|
||||
review_status: ${{ steps.verify.outputs.review_status }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0
|
||||
with:
|
||||
# Pinned: unpinned setup-uv fetches a 'latest' manifest from
|
||||
# raw.githubusercontent.com every job; transient fetch failures
|
||||
# fail the job (2026-07-28 incident). Keep in sync with tests.yml.
|
||||
version: "0.9.28"
|
||||
|
||||
# `uv lock --check` re-resolves the project from pyproject.toml and
|
||||
# compares the result to uv.lock, exiting non-zero if they disagree.
|
||||
@@ -73,6 +84,7 @@ jobs:
|
||||
# of this file) — failures often mean "your branch is behind main,
|
||||
# rebase and regenerate uv.lock."
|
||||
- name: Verify uv.lock is up-to-date
|
||||
id: verify
|
||||
run: |
|
||||
# uv lock --check re-resolves against PyPI (network). Retry so a
|
||||
# registry blip doesn't read as "lockfile stale". A genuinely stale
|
||||
@@ -117,5 +129,9 @@ jobs:
|
||||
on `main` post-merge.
|
||||
EOF
|
||||
echo "::error title=uv.lock out of sync::Run \`uv lock\` locally and commit the result. If on a PR, sync with main first."
|
||||
review_status='[{"source":"uv.lock check","results":[{"kind":"action_required","title":"uv.lock out of sync","summary":"uv.lock is out of sync with pyproject.toml.","how_to_fix":"Run `uv lock` locally and commit the result. If on a PR, sync with main first:\n```\ngit fetch origin main\ngit rebase origin/main\nuv lock\ngit add uv.lock\ngit commit -m \"chore: refresh uv.lock\"\n```\n"}]}]'
|
||||
echo "review_status=${review_status}" >> "$GITHUB_OUTPUT"
|
||||
exit 1
|
||||
fi
|
||||
review_status='[]'
|
||||
echo "review_status=${review_status}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
+37
-1
@@ -1,9 +1,13 @@
|
||||
.DS_Store
|
||||
/venv/
|
||||
/venv.old/
|
||||
/venv.stale.runtime-*/
|
||||
/.hermes-runtime/
|
||||
/_pycache/
|
||||
*.pyc*
|
||||
__pycache__/
|
||||
act/
|
||||
.act-sandbox-agent.*
|
||||
.venv/
|
||||
.venv
|
||||
.vscode/
|
||||
@@ -42,7 +46,10 @@ run_datagen_sonnet.sh
|
||||
source-data/*
|
||||
run_datagen_megascience_glm4-6.sh
|
||||
data/*
|
||||
node_modules/
|
||||
# No trailing slash: also matches node_modules SYMLINKS (worktrees often
|
||||
# symlink node_modules to the main checkout; the dir-only pattern let one
|
||||
# slip into a commit and break `npm ci` on CI with ENOTDIR).
|
||||
node_modules
|
||||
browser-use/
|
||||
agent-browser/
|
||||
# Private keys
|
||||
@@ -54,6 +61,10 @@ __pycache__/
|
||||
hermes_agent.egg-info/
|
||||
wandb/
|
||||
testlogs
|
||||
playwright-report/
|
||||
test-results/
|
||||
# Playwright visual regression baselines — cached from main in CI, not committed
|
||||
*-snapshots/
|
||||
|
||||
# CLI config (may contain sensitive SSH paths)
|
||||
cli-config.yaml
|
||||
@@ -66,6 +77,8 @@ environments/benchmarks/evals/
|
||||
|
||||
# Web UI build output
|
||||
hermes_cli/web_dist/
|
||||
# Cross-process web UI build lock (flock target, always empty)
|
||||
.web_ui_build.lock
|
||||
apps/desktop/build/
|
||||
apps/desktop/dist/
|
||||
|
||||
@@ -139,6 +152,16 @@ docs/superpowers/*
|
||||
.update-incomplete
|
||||
.update-incomplete.lock
|
||||
|
||||
# Checkout fingerprint the __pycache__ tree was last validated against
|
||||
# (launch-time stale-bytecode sweep). Runtime state, never a code change.
|
||||
.bytecode-fingerprint
|
||||
.bytecode-fingerprint.tmp
|
||||
|
||||
# Installer-written method stamp in the managed checkout root (scripts/install.sh).
|
||||
# Runtime metadata only — never a code change. Ignore so `git status` stays clean
|
||||
# and `hermes update`'s untracked autostash does not treat it as a local edit (#66189 / #54855).
|
||||
/.install_method
|
||||
|
||||
# Tool Search live-test harness output — non-deterministic model transcripts,
|
||||
# regenerated by scripts/tool_search_livetest.py. Never an artifact of the repo.
|
||||
scripts/out/
|
||||
@@ -157,4 +180,17 @@ apps/desktop/demo/
|
||||
# image-provider (fal.media) URL — they are NEVER committed to the repo. The
|
||||
# PR body is the archive. See the hermes-agent-dev skill's
|
||||
# pr-infographic-workflow reference (storage rule + lapse #8 / #COMMIT-1).
|
||||
#
|
||||
# Spelling variants are listed because a single `infographic/` pattern was
|
||||
# sidestepped by an `infograficos/` directory (#70552). .gitignore is only
|
||||
# the first line of defence and cannot stop `git add -f` at all — the
|
||||
# infographic-check CI job is what actually enforces this.
|
||||
infographic/
|
||||
infographics/
|
||||
infograficos/
|
||||
infografico/
|
||||
native/fts5_cjk/*.so
|
||||
# Runtime marker written by hermes update when a lazy dependency refresh is
|
||||
# interrupted; consumed by launch-time recovery. Never commit it (was tracked
|
||||
# by accident via 3a69e34702, removed in the #72002 salvage).
|
||||
.lazy-refresh-incomplete
|
||||
|
||||
@@ -0,0 +1,146 @@
|
||||
# Message reactions (desktop tapbacks)
|
||||
|
||||
Two-way emoji reactions on individual messages in the desktop transcript: the
|
||||
user reacts to any message, the agent reacts to a user message, and both sides
|
||||
read the other's reactions as conversational signal.
|
||||
|
||||
## What already exists
|
||||
|
||||
Hermes already models reactions on the **platform** side — the desktop is the
|
||||
only surface without them.
|
||||
|
||||
| Surface | Reaction support | Where |
|
||||
|---|---|---|
|
||||
| Agent → platform message | `send_message(action="react"/"unreact")` | `tools/send_message_tool.py:266` `_handle_react()` |
|
||||
| Photon / iMessage | tapbacks in + out, routed only for messages we sent | `plugins/platforms/photon/adapter.py:1240-1283` |
|
||||
| Telegram | `setMessageReaction`, config-gated | `plugins/platforms/telegram/adapter.py:9669+` |
|
||||
| Slack / Matrix / Feishu / Discord | inbound reaction events → hooks | `gateway/run.py:4688` `_handle_reaction_event()` → `HookRegistry.emit("reaction:added")` |
|
||||
| Adapter contract | `add_reaction()` / `remove_reaction()` coroutines, `set_reaction_handler()` | `gateway/platforms/base.py:3330` |
|
||||
| Core "affection" detector | regex on user text → `vibe`, drives CLI pet / TUI heart / desktop hearts | `agent/reactions.py`, `agent/turn_context.py:592-604` |
|
||||
|
||||
Two things follow from that table:
|
||||
|
||||
1. **The agent-facing verb already exists.** `send_message(action="react")` is
|
||||
the established shape. A desktop reaction should extend that tool, not add a
|
||||
new core tool — every new tool ships on every API call (AGENTS.md footprint
|
||||
ladder).
|
||||
2. **The inbound convention already exists.** Photon turns a tapback into a
|
||||
normal message event with `reply_to_message_id` + `reply_to_is_own_message`,
|
||||
and the gateway prefixes `[Replying to your previous message: "…"]`
|
||||
(`gateway/run.py:13125-13132`). Desktop reactions should read the same way to
|
||||
the model.
|
||||
|
||||
Nothing exists on the desktop side: `grep -ri reaction` across `apps/desktop`
|
||||
finds only the pet-overlay hearts.
|
||||
|
||||
## Prior art
|
||||
|
||||
**iOS Tapback** ([Apple](https://support.apple.com/guide/iphone/react-with-tapbacks-iph018d3c336/ios)):
|
||||
double-tap or touch-and-hold a message → floating pill above the bubble with
|
||||
heart / thumbs-up / thumbs-down / haha / ‼️ / ❓, swipe left for suggested emoji
|
||||
and stickers, or tap the emoji button for the full keyboard. **One tapback per
|
||||
message per person** — tapping the same one again removes it, tapping a
|
||||
different one replaces it. Multiple people's tapbacks stack on the badge.
|
||||
|
||||
**Platform data models** converge on the same shape:
|
||||
|
||||
| Platform | Model | Add / remove |
|
||||
|---|---|---|
|
||||
| Slack | `{name, count, users[]}` | [`reactions.add`](https://docs.slack.dev/reference/methods/reactions.add) / `reactions.remove`, emits `reaction_added` |
|
||||
| Discord | `{emoji, count, me}` on the message object | `PUT`/`DELETE .../reactions/{emoji}/@me` |
|
||||
| Telegram | `reaction: [{type:"emoji", emoji:"👍"}]` — replaces the whole set | `setMessageReaction`, `is_big` for the big animation |
|
||||
|
||||
Telegram's "set the whole array" is the closest match to iOS semantics and the
|
||||
simplest thing to persist.
|
||||
|
||||
**assistant-ui has no reaction primitive.** `@assistant-ui/react` 0.14.24 (MIT,
|
||||
vendored at `apps/desktop/node_modules`): zero hits for "reaction" in `core/src`,
|
||||
`react/src`, `dist/`, or the 2.2 MB `llms-full.txt` docs dump. What exists is a
|
||||
hard-coded binary `FeedbackAdapter` (`"positive" | "negative"`,
|
||||
`core/src/adapters/feedback.ts`) that throws when unconfigured and only writes
|
||||
back onto assistant messages. Not usable for emoji, not usable on user messages.
|
||||
|
||||
**But `metadata.custom` is the supported extension channel** and this repo
|
||||
already uses it: `ThreadUserMessage`/`ThreadAssistantMessage`/`ThreadSystemMessage`
|
||||
all carry `metadata.custom: Record<string, unknown>` (`core/src/types/message.ts:319-366`),
|
||||
and `chat-runtime.ts:397` already ships `custom: { attachmentRefs }` through it.
|
||||
|
||||
**Emoji picker survey** (npm week of 2026-07-22, sizes measured from the
|
||||
published ESM entry):
|
||||
|
||||
| Library | License | Weekly DL | gzip | Headless | Latest |
|
||||
|---|---|---|---|---|---|
|
||||
| **frimousse** | MIT | 573k | **8.5 kB** | ✅ fully unstyled, composable parts | 0.3.0 · 2025-07-15 |
|
||||
| emoji-picker-react | MIT | 1.31M | 87 kB | ❌ own CSS-in-JS (flairup) | 4.19.1 · 2026-04-27 |
|
||||
| emoji-mart | MIT | 2.22M | ~120 kB w/ data | ❌ Preact + shadow styling | 5.6.0 · **2024-04-25**, 217 open issues |
|
||||
| emoji-picker-element | Apache-2.0 | 183k | — | ❌ Web Component / Shadow DOM | 1.29.1 · 2026-03-01 |
|
||||
|
||||
No picker is currently a dependency (only `emoji-regex`, transitive). Already
|
||||
paid for and reusable: `radix-ui` (Popover), `motion`, `@tanstack/react-virtual`,
|
||||
Tailwind v4.
|
||||
|
||||
## Recommendation
|
||||
|
||||
**Hand-roll the tapback pill; add frimousse only behind the "+".** Six fixed
|
||||
emoji in a pill is ~40 lines of JSX against existing tokens — pulling 87 kB of
|
||||
`emoji-picker-react` to render six buttons, plus a CSS engine that fights
|
||||
`DESIGN.md`, is backwards. frimousse is headless, dependency-free, 10× smaller,
|
||||
and exposes `emojibaseUrl` so the data can be bundled as a Vite asset instead of
|
||||
hitting jsDelivr (Electron must work offline).
|
||||
|
||||
### Data model
|
||||
|
||||
One reaction per author per message, Telegram-style whole-set replacement:
|
||||
|
||||
```ts
|
||||
type MessageReaction = { emoji: string; author: 'user' | 'agent'; at: number }
|
||||
```
|
||||
|
||||
Persisted in the existing `messages.display_metadata` JSON column
|
||||
(`hermes_state_common.py:215`) — no new table. It already survives insert,
|
||||
compaction, and every read projection, and
|
||||
`set_latest_matching_message_display_kind()` (`hermes_state.py:5292`) is the
|
||||
precedent for stamping metadata onto an already-persisted row.
|
||||
|
||||
### Model context
|
||||
|
||||
Reactions must reach the model **without breaking prompt caching**. The
|
||||
`api_messages` build loop strips `display_metadata` from every outgoing copy
|
||||
(`agent/conversation_loop.py:1443-1446`) precisely so display state never
|
||||
becomes a provider field. Two candidate paths:
|
||||
|
||||
| Path | Cache impact | Notes |
|
||||
|---|---|---|
|
||||
| Rewrite the reacted-to message's content to carry the annotation | **Breaks the cached prefix** — mutates past context | Rejected. AGENTS.md: prompt caching is sacred. |
|
||||
| Deliver the reaction as the *next* turn's leading annotation, mirroring photon | Prefix untouched; only the new turn carries it | Matches `[Replying to your previous message: "…"]` (`gateway/run.py:13125`), which the agent already understands |
|
||||
|
||||
The second is the same trick the platform adapters already use, so the model
|
||||
sees a familiar shape and no existing conversation is rewritten.
|
||||
|
||||
### Attach points
|
||||
|
||||
| Concern | File | Lines |
|
||||
|---|---|---|
|
||||
| Assistant hover bar | `apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx` | 134–175 |
|
||||
| User hover cluster | `apps/desktop/src/components/assistant-ui/thread/user-message.tsx` | 296–336 |
|
||||
| Callback threading (ref caveat 79–99) | `apps/desktop/src/components/assistant-ui/thread/index.tsx` | 109–133 |
|
||||
| `metadata.custom` → runtime | `apps/desktop/src/lib/chat-runtime.ts` | 384–432 |
|
||||
| RPC client ↔ server pattern | `sidebar/session-actions-menu.tsx:62-89` ↔ `tui_gateway/server.py:8322` | — |
|
||||
| Persistence | `hermes_state_common.py:192-216`, `hermes_state.py:5292-5324` | — |
|
||||
| Prompt injection / strip | `agent/conversation_loop.py` | 1430–1529 |
|
||||
|
||||
### Known gaps to solve first
|
||||
|
||||
- **No durable message id crosses the gateway RPC path.** `_history_to_messages()`
|
||||
(`tui_gateway/server.py:6545`) builds `{"role", "text"}` and drops the id. The
|
||||
REST path carries `messages.id` incidentally via `SELECT *` but TS
|
||||
`SessionMessage` (`types/hermes.ts:513-533`) doesn't declare it. Renderer ids
|
||||
are ephemeral and change shape between rehydrated (`<ts>-<i>-<role>`), live
|
||||
(`assistant-<ms>`), and optimistic (`user-<ms>-<rand>`) messages. A reaction
|
||||
needs a stable key — this is the first thing to fix.
|
||||
- **WeakMap identity cache** in `apps/desktop/src/app/chat/runtime-repository.ts:26-66`
|
||||
keys normalized `ThreadMessage` by `ChatMessage` identity. A reaction change
|
||||
must produce a **new** `ChatMessage` object or the UI renders stale.
|
||||
- **Rewind rewrites rows** (`replace_messages`), so anything keyed by row id
|
||||
needs cascade handling — an argument for keeping reactions in
|
||||
`display_metadata` on the row itself rather than a side table.
|
||||
@@ -325,7 +325,7 @@ class AIAgent:
|
||||
provider: str = None,
|
||||
api_mode: str = None, # "chat_completions" | "codex_responses" | ...
|
||||
model: str = "", # empty → resolved from config/provider later
|
||||
max_iterations: int = 90, # tool-calling iterations (shared with subagents)
|
||||
max_iterations: int = 500, # tool-calling iterations (shared with subagents)
|
||||
enabled_toolsets: list = None,
|
||||
disabled_toolsets: list = None,
|
||||
quiet_mode: bool = False,
|
||||
@@ -998,7 +998,8 @@ Two shapes:
|
||||
Roles:
|
||||
|
||||
- `role="leaf"` (default) — focused worker. Cannot call `delegate_task`,
|
||||
`clarify`, `memory`, `send_message`, `execute_code`.
|
||||
`clarify`, `memory`, `send_message`, `cronjob`. Retains `execute_code`
|
||||
(programmatic tool calling).
|
||||
- `role="orchestrator"` — retains `delegate_task` so it can spawn its
|
||||
own workers. Gated by `delegation.orchestrator_enabled` (default true)
|
||||
and bounded by `delegation.max_spawn_depth` (default 2).
|
||||
@@ -1283,14 +1284,15 @@ def profile_env(tmp_path, monkeypatch):
|
||||
### Python
|
||||
**ALWAYS use `scripts/run_tests.sh`** — do not call `pytest` directly. The script enforces
|
||||
hermetic environment parity with CI (unset credential vars, TZ=UTC, LANG=C.UTF-8,
|
||||
`-n auto` xdist workers, in-tree subprocess-isolation plugin). Direct `pytest`
|
||||
per-file subprocess isolation via `scripts/run_tests_parallel.py` — no xdist,
|
||||
worker count auto-scaled from CPU count). Direct `pytest`
|
||||
on a 16+ core developer machine with API keys set diverges from CI in ways
|
||||
that have caused multiple "works locally, fails in CI" incidents (and the reverse).
|
||||
|
||||
```bash
|
||||
scripts/run_tests.sh # full suite, CI-parity
|
||||
scripts/run_tests.sh tests/gateway/ # one directory
|
||||
scripts/run_tests.sh tests/agent/test_foo.py::test_x # one test
|
||||
scripts/run_tests.sh tests/agent/test_foo.py -k test_x # one test (file + -k; the runner is file-granular)
|
||||
scripts/run_tests.sh -v --tb=long # pass-through pytest flags
|
||||
```
|
||||
|
||||
|
||||
+3
-2
@@ -201,7 +201,8 @@ ln -sf "$(pwd)/venv/bin/hermes" ~/.local/bin/hermes
|
||||
### Run tests
|
||||
|
||||
```bash
|
||||
# Preferred — matches CI (hermetic env, 4 xdist workers); see AGENTS.md
|
||||
# Preferred — matches CI (hermetic `env -i`, per-file subprocess isolation
|
||||
# via run_tests_parallel.py, worker count auto-scaled); see AGENTS.md
|
||||
scripts/run_tests.sh
|
||||
|
||||
# Alternative (activate the venv first). The wrapper is still recommended
|
||||
@@ -848,7 +849,7 @@ that touches the OS, assume *any* platform can hit your code path.
|
||||
Tests that use POSIX-only syscalls need a skip marker. Common ones:
|
||||
- Symlinks → `@pytest.mark.skipif(sys.platform == "win32", ...)`
|
||||
- `0o600` file modes → `@pytest.mark.skipif(sys.platform.startswith("win"), ...)`
|
||||
- `signal.SIGALRM` → Unix-only (see `tests/conftest.py::_enforce_test_timeout`)
|
||||
- `signal.SIGALRM` → Unix-only (per-test timeouts no longer use it directly; see the win32 timeout-method shim in `tests/conftest.py::pytest_configure`)
|
||||
- `os.setsid` / `os.fork` → Unix-only
|
||||
- Live Winsock / Windows-specific regression tests →
|
||||
`@pytest.mark.skipif(sys.platform != "win32", reason="Windows-specific regression")`
|
||||
|
||||
+81
-2
@@ -1,3 +1,45 @@
|
||||
# Debian 13 still ships SQLite 3.46.1, which contains the upstream WAL-reset
|
||||
# corruption bug. Build a pinned shared library for the runtime image instead
|
||||
# of relying on a distro backport that trixie does not currently provide.
|
||||
# See #70480 and https://sqlite.org/wal.html#walresetbug.
|
||||
FROM debian:13.4 AS sqlite_build
|
||||
ARG SQLITE_AUTOCONF_VERSION=3530400
|
||||
ARG SQLITE_SHA256=0e9483900e92cd5de8fd48d16bf9200145a61f7fd5be542a5ac81d8a9516eb9c
|
||||
RUN apt-get -o Acquire::Retries=3 update && \
|
||||
apt-get -o Acquire::Retries=3 install -y --no-install-recommends \
|
||||
build-essential ca-certificates curl && \
|
||||
rm -rf /var/lib/apt/lists/* && \
|
||||
(curl -fsSL --retry 1 --retry-all-errors --connect-timeout 15 --max-time 60 \
|
||||
-o /tmp/sqlite.tar.gz \
|
||||
"https://sqlite.org/2026/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}.tar.gz" || \
|
||||
curl -fsSL --retry 3 --retry-all-errors --connect-timeout 15 --max-time 120 \
|
||||
-o /tmp/sqlite.tar.gz \
|
||||
"https://sources.buildroot.net/sqlite/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}.tar.gz") && \
|
||||
printf '%s %s\n' "${SQLITE_SHA256}" /tmp/sqlite.tar.gz > /tmp/sqlite.sha256 && \
|
||||
sha256sum -c /tmp/sqlite.sha256 && \
|
||||
tar -xzf /tmp/sqlite.tar.gz -C /tmp && \
|
||||
cd "/tmp/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}" && \
|
||||
CFLAGS="-O2 \
|
||||
-DSQLITE_ENABLE_FTS3 \
|
||||
-DSQLITE_ENABLE_FTS3_PARENTHESIS \
|
||||
-DSQLITE_ENABLE_FTS4 \
|
||||
-DSQLITE_ENABLE_FTS5 \
|
||||
-DSQLITE_ENABLE_RTREE \
|
||||
-DSQLITE_ENABLE_GEOPOLY \
|
||||
-DSQLITE_ENABLE_COLUMN_METADATA \
|
||||
-DSQLITE_ENABLE_UNLOCK_NOTIFY \
|
||||
-DSQLITE_ENABLE_DBSTAT_VTAB \
|
||||
-DSQLITE_ENABLE_DBPAGE_VTAB \
|
||||
-DSQLITE_ENABLE_MATH_FUNCTIONS \
|
||||
-DSQLITE_ENABLE_PREUPDATE_HOOK \
|
||||
-DSQLITE_ENABLE_SESSION \
|
||||
-DSQLITE_SECURE_DELETE \
|
||||
-DSQLITE_THREADSAFE=1 \
|
||||
-DSQLITE_MAX_VARIABLE_NUMBER=250000" \
|
||||
./configure --prefix=/opt/sqlite-fixed --disable-static && \
|
||||
make -j"$(nproc)" && \
|
||||
make install
|
||||
|
||||
FROM ghcr.io/astral-sh/uv:0.11.6-python3.13-trixie@sha256:b3c543b6c4f23a5f2df22866bd7857e5d304b67a564f4feab6ac22044dde719b AS uv_source
|
||||
# Node 22 LTS source stage. Debian trixie's bundled nodejs is pinned to 20.x
|
||||
# which reached EOL in April 2026 — we copy node + npm + corepack from the
|
||||
@@ -31,6 +73,23 @@ RUN apt-get -o Acquire::Retries=3 update && \
|
||||
ca-certificates curl iputils-ping python3 python-is-python3 ripgrep ffmpeg gcc g++ make cmake python3-dev python3-venv libffi-dev libolm-dev procps git openssh-client docker-cli xz-utils && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Prefer the fixed SQLite over Debian's vulnerable libsqlite3.so.0. Keep the
|
||||
# public library name stable so both the system interpreter and the uv-created
|
||||
# venv resolve the replacement without changing Python import paths.
|
||||
COPY --from=sqlite_build /opt/sqlite-fixed/lib/libsqlite3.so.3.53.4 /usr/local/lib/
|
||||
RUN ln -sf libsqlite3.so.3.53.4 /usr/local/lib/libsqlite3.so.0 && \
|
||||
ln -sf libsqlite3.so.3.53.4 /usr/local/lib/libsqlite3.so && \
|
||||
printf '/usr/local/lib\n' > /etc/ld.so.conf.d/000-sqlite-fixed.conf && \
|
||||
ldconfig && \
|
||||
python3 -c "import sqlite3, sys; \
|
||||
v = sqlite3.sqlite_version_info; \
|
||||
sys.exit(f'linked SQLite {sqlite3.sqlite_version} still has the WAL-reset bug') if v < (3, 51, 3) else None; \
|
||||
db = sqlite3.connect(':memory:'); \
|
||||
db.execute(\"CREATE VIRTUAL TABLE docs USING fts5(content, tokenize='trigram')\"); \
|
||||
db.execute(\"INSERT INTO docs VALUES ('hermes')\"); \
|
||||
sys.exit('SQLite FTS5 trigram self-test failed') if db.execute(\"SELECT count(*) FROM docs WHERE docs MATCH 'erm'\").fetchone()[0] != 1 else None; \
|
||||
db.close()"
|
||||
|
||||
# ---------- s6-overlay install ----------
|
||||
# s6-overlay provides supervision for the main hermes process, the dashboard,
|
||||
# and per-profile gateways. /init becomes PID 1 below — see ENTRYPOINT.
|
||||
@@ -141,6 +200,22 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
|
||||
done && \
|
||||
npm cache clean --force
|
||||
|
||||
# ---------- Photon iMessage sidecar deps (baked, NS-606) ----------
|
||||
# The photon plugin's Node sidecar needs its own node_modules
|
||||
# (spectrum-ts). The install tree is immutable at runtime, so a lazy
|
||||
# `npm ci` on first connect would hit EROFS — bake the deps here instead
|
||||
# (deterministic installs, NS-559). The patch script is copied alongside
|
||||
# the manifests because package.json's postinstall runs it, which also
|
||||
# means the spectrum-ts patch is applied at build time. Layer-cached:
|
||||
# only re-runs when the sidecar manifests/patch change.
|
||||
COPY plugins/platforms/photon/sidecar/package.json \
|
||||
plugins/platforms/photon/sidecar/package-lock.json \
|
||||
plugins/platforms/photon/sidecar/patch-spectrum-mixed-attachments.mjs \
|
||||
plugins/platforms/photon/sidecar/
|
||||
RUN cd plugins/platforms/photon/sidecar && \
|
||||
npm ci --no-audit --fetch-retries=5 && \
|
||||
npm cache clean --force
|
||||
|
||||
# ---------- Layer-cached Python dependency install ----------
|
||||
# Copy only pyproject.toml + uv.lock so the Python dep resolve + wheel
|
||||
# download + native-extension compile layer is cached unless those inputs
|
||||
@@ -152,7 +227,7 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
|
||||
# frontend stats the readme path during dep resolution, so we `touch` an
|
||||
# empty placeholder — the real README is restored by `COPY . .` below.
|
||||
#
|
||||
# `uv sync --frozen --no-install-project --extra all --extra messaging`
|
||||
# `uv sync --frozen --no-install-project --extra all --extra messaging --extra otlp`
|
||||
# installs the deps reachable through the composite `[all]` extra
|
||||
# (handpicked set intended for the production image — excludes `[dev]`),
|
||||
# plus gateway messaging adapters that should work in the published image
|
||||
@@ -165,6 +240,10 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
|
||||
# so Docker users can use these providers without requiring runtime
|
||||
# lazy-install access to PyPI (often blocked in containerized envs).
|
||||
#
|
||||
# The [otlp] extra contains the SDK/exporter imported by Hermes when Gateway
|
||||
# Health export is enabled. Collector and observability-backend dependencies
|
||||
# remain external and are not part of the Hermes production image.
|
||||
#
|
||||
# The hindsight memory provider's client (hindsight-client) is baked in
|
||||
# for the same reason: it lazy-installs into /opt/hermes/.venv at first
|
||||
# use, which lives inside the (immutable) image layer rather than the
|
||||
@@ -182,7 +261,7 @@ RUN npm install --prefer-offline --no-audit --fetch-retries=5 && \
|
||||
# The editable link is created after the source copy below.
|
||||
COPY pyproject.toml uv.lock ./
|
||||
RUN touch ./README.md
|
||||
RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra anthropic --extra bedrock --extra azure-identity --extra hindsight --extra matrix
|
||||
RUN uv sync --frozen --no-install-project --extra all --extra messaging --extra otlp --extra anthropic --extra bedrock --extra azure-identity --extra hindsight --extra matrix
|
||||
|
||||
# ---------- Frontend build (cached independently from Python source) ----------
|
||||
# Copy only the frontend source trees first so that Python-only changes don't
|
||||
|
||||
-13
@@ -1,13 +0,0 @@
|
||||
graft skills
|
||||
graft optional-skills
|
||||
graft optional-mcps
|
||||
graft locales
|
||||
# Bundled plugin manifests (plugin.yaml / plugin.yml). Without these the
|
||||
# PluginManager scan (hermes_cli/plugins.py) finds zero plugins on installs
|
||||
# built from the sdist (e.g. Homebrew, downstream packagers). package-data
|
||||
# below covers the wheel; this covers the sdist. See #34034 / #28149.
|
||||
recursive-include plugins plugin.yaml plugin.yml
|
||||
# Gateway assets include images plus YAML catalogs such as status_phrases.yaml.
|
||||
recursive-include gateway/assets *
|
||||
global-exclude __pycache__
|
||||
global-exclude *.py[cod]
|
||||
@@ -26,7 +26,7 @@ Use any model you want — [Nous Portal](https://portal.nousresearch.com), OpenR
|
||||
<tr><td><b>A closed learning loop</b></td><td>Agent-curated memory with periodic nudges. Autonomous skill creation after complex tasks. Skills self-improve during use. FTS5 session search with LLM summarization for cross-session recall. <a href="https://github.com/plastic-labs/honcho">Honcho</a> dialectic user modeling. Compatible with the <a href="https://agentskills.io">agentskills.io</a> open standard.</td></tr>
|
||||
<tr><td><b>Scheduled automations</b></td><td>Built-in cron scheduler with delivery to any platform. Daily reports, nightly backups, weekly audits — all in natural language, running unattended.</td></tr>
|
||||
<tr><td><b>Delegates and parallelizes</b></td><td>Spawn isolated subagents for parallel workstreams. Write Python scripts that call tools via RPC, collapsing multi-step pipelines into zero-context-cost turns.</td></tr>
|
||||
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Six terminal backends — local, Docker, SSH, Singularity, Modal, and Daytona. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
|
||||
<tr><td><b>Runs anywhere, not just your laptop</b></td><td>Seven terminal backends — local, Docker, SSH, Singularity, Modal, Daytona, and Vercel Sandbox. Daytona and Modal offer serverless persistence — your agent's environment hibernates when idle and wakes on demand, costing nearly nothing between sessions. Run it on a $5 VPS or a GPU cluster.</td></tr>
|
||||
<tr><td><b>Research-ready</b></td><td>Batch trajectory generation, trajectory compression for training the next generation of tool-calling models.</td></tr>
|
||||
</table>
|
||||
|
||||
|
||||
+7
-3
@@ -173,9 +173,13 @@ modelo de autorización, pero las reglas a continuación se aplican uniformement
|
||||
|
||||
**Superficies en Hermes Agent:**
|
||||
|
||||
- **Adaptadores de plataforma del gateway.** Integraciones de mensajería en
|
||||
`gateway/platforms/` (Telegram, Discord, Slack, email, SMS, etc.)
|
||||
y adaptadores análogos incluidos como plugins.
|
||||
- **Adaptadores de plataforma del gateway.** La mayoría de las integraciones
|
||||
de mensajería se distribuyen como plugins empaquetados en
|
||||
`plugins/platforms/<name>/` (Telegram, Discord, Slack, email, SMS, etc.).
|
||||
Los tipos base compartidos y un conjunto menor de adaptadores
|
||||
legacy/directos viven en `gateway/platforms/` (`base.py`, Signal, servidor
|
||||
API, webhooks, …), con descubrimiento y carga diferida vía
|
||||
`gateway/platform_registry.py`.
|
||||
- **Superficies HTTP expuestas en red.** El adaptador del servidor API, el
|
||||
plugin del dashboard, los endpoints HTTP del plugin kanban, y cualquier
|
||||
otro plugin que vincule un socket de escucha.
|
||||
|
||||
+6
-3
@@ -177,9 +177,12 @@ authorization model, but the rules below apply uniformly.
|
||||
|
||||
**Surfaces in Hermes Agent:**
|
||||
|
||||
- **Gateway platform adapters.** Messaging integrations in
|
||||
`gateway/platforms/` (Telegram, Discord, Slack, email, SMS, etc.)
|
||||
and analogous adapters shipped as plugins.
|
||||
- **Gateway platform adapters.** Most messaging integrations ship as
|
||||
bundled plugins under `plugins/platforms/<name>/` (Telegram, Discord,
|
||||
Slack, email, SMS, etc.). Shared base types and a smaller set of
|
||||
legacy/direct adapters live under `gateway/platforms/`
|
||||
(`base.py`, Signal, API server, webhooks, …), with discovery and
|
||||
deferred loading via `gateway/platform_registry.py`.
|
||||
- **Network-exposed HTTP surfaces.** The API server adapter, the
|
||||
dashboard plugin, the kanban plugin's HTTP endpoints, and any
|
||||
other plugin that binds a listening socket.
|
||||
|
||||
@@ -32,6 +32,7 @@ else:
|
||||
import argparse
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from hermes_constants import get_hermes_home
|
||||
@@ -190,7 +191,7 @@ def _run_setup_browser(assume_yes: bool = False) -> int:
|
||||
"""Bootstrap agent-browser + Chromium.
|
||||
|
||||
Routes through dep_ensure -> install.{sh,ps1} --ensure, sharing code
|
||||
with ``hermes postinstall`` and the runtime lazy installer.
|
||||
with the runtime lazy installer.
|
||||
|
||||
Returns 0 on success, 1 on failure.
|
||||
"""
|
||||
@@ -251,11 +252,13 @@ def main(argv: list[str] | None = None) -> None:
|
||||
# MCP servers dynamically via asyncio.to_thread inside the event
|
||||
# loop; that path is unaffected.) Moved from model_tools.py module
|
||||
# scope to avoid freezing the gateway's loop on lazy import (#16856).
|
||||
try:
|
||||
from tools.mcp_tool import discover_mcp_tools
|
||||
discover_mcp_tools()
|
||||
except Exception:
|
||||
logger.debug("MCP tool discovery failed at ACP startup", exc_info=True)
|
||||
# Metadata-only hosts can opt out of unrelated global MCP startup.
|
||||
if os.environ.get("HERMES_ACP_SKIP_CONFIGURED_MCP", "").strip() != "1":
|
||||
try:
|
||||
from tools.mcp_tool import discover_mcp_tools
|
||||
discover_mcp_tools()
|
||||
except Exception:
|
||||
logger.debug("MCP tool discovery failed at ACP startup", exc_info=True)
|
||||
|
||||
agent = HermesACPAgent()
|
||||
try:
|
||||
|
||||
+373
-58
@@ -74,6 +74,10 @@ from acp_adapter.permissions import make_approval_callback
|
||||
from acp_adapter.provenance import session_provenance_meta
|
||||
from acp_adapter.session import SessionManager, SessionState, _expand_acp_enabled_toolsets
|
||||
from acp_adapter.tools import build_tool_complete, build_tool_start
|
||||
from agent.context_compressor import (
|
||||
COMPRESSED_SUMMARY_METADATA_KEY,
|
||||
ContextCompressor,
|
||||
)
|
||||
from tools.approval import (
|
||||
reset_hermes_interactive_context,
|
||||
set_hermes_interactive_context,
|
||||
@@ -81,6 +85,110 @@ from tools.approval import (
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, str]]]]:
|
||||
"""Return ``(slug, label, [(model_id, description), ...])`` for named endpoints.
|
||||
|
||||
Covers both the v12 ``providers:`` mapping and the legacy
|
||||
``custom_providers:`` list. These endpoints never appear in canonical
|
||||
provider enumeration, so without this the ACP model selector hides every
|
||||
named endpoint that the TUI ``/model`` picker already renders (#47039
|
||||
implemented named-endpoint rows for the TUI surface only).
|
||||
|
||||
Model lists come from the entry's declared models (``default_model`` +
|
||||
``models``), refreshed from the endpoint's live ``/models`` listing when a
|
||||
credential is available and ``discover_models`` is not disabled. Declared
|
||||
models are kept even when live discovery fails — some OpenAI-compatible
|
||||
endpoints (e.g. Bedrock Mantle Responses) expose no ``/models`` route at
|
||||
all yet serve the declared models fine.
|
||||
|
||||
Slugs use the ``custom:<name>`` shape that ``parse_model_input`` and
|
||||
``resolve_runtime_provider`` already resolve, so encoded choice ids
|
||||
(``custom:<name>:<model>``) round-trip through ``set_session_model``
|
||||
unchanged.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.config import (
|
||||
get_compatible_custom_providers,
|
||||
is_provider_enabled,
|
||||
load_config,
|
||||
)
|
||||
from hermes_cli.models import fetch_api_models
|
||||
from hermes_cli.providers import custom_provider_slug
|
||||
except ImportError:
|
||||
return []
|
||||
|
||||
try:
|
||||
cfg = load_config()
|
||||
entries = get_compatible_custom_providers(cfg)
|
||||
except Exception:
|
||||
logger.debug("Could not load named custom providers", exc_info=True)
|
||||
return []
|
||||
|
||||
# ``get_compatible_custom_providers`` drops the ``enabled`` flag during
|
||||
# normalization, so collect explicitly disabled provider keys from the
|
||||
# raw config and skip their entries below.
|
||||
disabled_keys: set[str] = set()
|
||||
raw_providers = cfg.get("providers") if isinstance(cfg, dict) else None
|
||||
if isinstance(raw_providers, dict):
|
||||
for raw_key, raw_entry in raw_providers.items():
|
||||
if isinstance(raw_entry, dict) and not is_provider_enabled(raw_entry):
|
||||
disabled_keys.add(str(raw_key).strip().lower())
|
||||
|
||||
catalogs: list[tuple[str, str, list[tuple[str, str]]]] = []
|
||||
for entry in entries:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
provider_key = str(entry.get("provider_key", "") or "").strip()
|
||||
if provider_key.lower() in disabled_keys:
|
||||
continue
|
||||
name = str(entry.get("name", "") or "").strip()
|
||||
base_url = str(entry.get("base_url", "") or "").strip()
|
||||
if not name or not base_url:
|
||||
continue
|
||||
slug = custom_provider_slug(name, provider_key)
|
||||
|
||||
api_key = str(entry.get("api_key", "") or "").strip()
|
||||
if not api_key:
|
||||
key_env = str(entry.get("key_env", "") or "").strip()
|
||||
api_key = os.environ.get(key_env, "").strip() if key_env else ""
|
||||
|
||||
declared: list[str] = []
|
||||
default_model = str(entry.get("model", "") or "").strip()
|
||||
if default_model:
|
||||
declared.append(default_model)
|
||||
models_cfg = entry.get("models")
|
||||
if isinstance(models_cfg, dict):
|
||||
for mid in models_cfg:
|
||||
mid = str(mid or "").strip()
|
||||
if mid and mid not in declared:
|
||||
declared.append(mid)
|
||||
|
||||
if not api_key and not declared:
|
||||
# No credential to discover with and nothing declared:
|
||||
# not addressable from the selector.
|
||||
continue
|
||||
|
||||
model_ids = list(declared)
|
||||
discover = entry.get("discover_models", True)
|
||||
if isinstance(discover, str):
|
||||
discover = discover.lower() not in {"false", "no", "0"}
|
||||
if discover and api_key:
|
||||
try:
|
||||
live = fetch_api_models(
|
||||
api_key, base_url, api_mode=entry.get("api_mode")
|
||||
)
|
||||
except Exception:
|
||||
live = None
|
||||
if live:
|
||||
model_ids = declared + [m for m in live if m not in declared]
|
||||
|
||||
if not model_ids:
|
||||
continue
|
||||
catalogs.append((slug, name, [(mid, "") for mid in model_ids]))
|
||||
|
||||
return catalogs
|
||||
|
||||
try:
|
||||
from hermes_cli import __version__ as HERMES_VERSION
|
||||
except Exception:
|
||||
@@ -93,6 +201,13 @@ _executor = ThreadPoolExecutor(max_workers=4, thread_name_prefix="acp-agent")
|
||||
# does not expose a client-side limit, so this is a fixed cap that clients
|
||||
# paginate against using `cursor` / `next_cursor`.
|
||||
_LIST_SESSIONS_PAGE_SIZE = 50
|
||||
# Per-provider cap for the ACP model selector. ACP clients (Zed, Buzz) render
|
||||
# the whole `availableModels` array in one dropdown, so an unbounded
|
||||
# cross-provider catalog degrades the picker. Mirrors the cap the MoA picker
|
||||
# already uses (`hermes_cli/moa_cmd.py`). This bounds each provider's row, not
|
||||
# the total; aggregator providers stay intentionally uncapped inside the shared
|
||||
# inventory, and the current model is always kept via the fallback insert below.
|
||||
ACP_MAX_MODELS_PER_PROVIDER = 200
|
||||
_MAX_ACP_RESOURCE_BYTES = 512 * 1024
|
||||
_TEXT_RESOURCE_MIME_PREFIXES = ("text/",)
|
||||
_TEXT_RESOURCE_MIME_TYPES = {
|
||||
@@ -456,7 +571,7 @@ class HermesACPAgent(acp.Agent):
|
||||
"tools": "List available tools",
|
||||
"context": "Show conversation context info",
|
||||
"reset": "Clear conversation history",
|
||||
"compact": "Compress conversation context",
|
||||
"compress": "Compress conversation context",
|
||||
"steer": "Inject guidance into the currently running agent turn",
|
||||
"queue": "Queue a prompt to run after the current turn finishes",
|
||||
"version": "Show Hermes version",
|
||||
@@ -485,7 +600,7 @@ class HermesACPAgent(acp.Agent):
|
||||
"description": "Clear conversation history",
|
||||
},
|
||||
{
|
||||
"name": "compact",
|
||||
"name": "compress",
|
||||
"description": "Compress conversation context",
|
||||
},
|
||||
{
|
||||
@@ -581,46 +696,108 @@ class HermesACPAgent(acp.Agent):
|
||||
return f"{raw_provider}:{raw_model}"
|
||||
|
||||
def _build_model_state(self, state: SessionState) -> SessionModelState | None:
|
||||
"""Return the ACP model selector payload for editors like Zed."""
|
||||
"""Return authenticated providers and their models for ACP clients.
|
||||
|
||||
The shared Hermes inventory is also used by ``hermes model``, the TUI,
|
||||
and the dashboard. Keeping ACP on that substrate prevents its selector
|
||||
from silently collapsing to the current provider's curated list.
|
||||
"""
|
||||
model = str(state.model or getattr(state.agent, "model", "") or "").strip()
|
||||
provider = getattr(state.agent, "provider", None) or detect_provider() or "openrouter"
|
||||
|
||||
try:
|
||||
from hermes_cli.models import curated_models_for_provider, normalize_provider, provider_label
|
||||
from hermes_cli.inventory import build_models_payload, load_picker_context
|
||||
from hermes_cli.models import normalize_provider, provider_label
|
||||
|
||||
normalized_provider = normalize_provider(provider)
|
||||
provider_name = provider_label(normalized_provider)
|
||||
context = load_picker_context().with_overrides(
|
||||
current_provider=normalized_provider,
|
||||
current_model=model,
|
||||
current_base_url=str(getattr(state.agent, "base_url", "") or ""),
|
||||
)
|
||||
payload = build_models_payload(
|
||||
context,
|
||||
explicit_only=True,
|
||||
include_unconfigured=False,
|
||||
picker_hints=False,
|
||||
canonical_order=True,
|
||||
pricing=False,
|
||||
capabilities=False,
|
||||
refresh=False,
|
||||
probe_custom_providers=False,
|
||||
probe_current_custom_provider=False,
|
||||
max_models=ACP_MAX_MODELS_PER_PROVIDER,
|
||||
)
|
||||
|
||||
available_models: list[ModelInfo] = []
|
||||
seen_ids: set[str] = set()
|
||||
|
||||
for model_id, description in curated_models_for_provider(normalized_provider):
|
||||
rendered_model = str(model_id or "").strip()
|
||||
if not rendered_model:
|
||||
for row in payload.get("providers") or []:
|
||||
row_provider = normalize_provider(str(row.get("slug") or "").strip())
|
||||
if not row_provider:
|
||||
continue
|
||||
choice_id = self._encode_model_choice(normalized_provider, rendered_model)
|
||||
if choice_id in seen_ids:
|
||||
continue
|
||||
desc_parts = [f"Provider: {provider_name}"]
|
||||
if description:
|
||||
desc_parts.append(str(description).strip())
|
||||
if rendered_model == model:
|
||||
desc_parts.append("current")
|
||||
available_models.append(
|
||||
ModelInfo(
|
||||
model_id=choice_id,
|
||||
name=rendered_model,
|
||||
description=" • ".join(part for part in desc_parts if part),
|
||||
)
|
||||
provider_name = str(row.get("name") or "").strip() or provider_label(
|
||||
row_provider
|
||||
)
|
||||
seen_ids.add(choice_id)
|
||||
for model_entry in row.get("models") or []:
|
||||
if isinstance(model_entry, dict):
|
||||
rendered_model = str(
|
||||
model_entry.get("id")
|
||||
or model_entry.get("model")
|
||||
or model_entry.get("name")
|
||||
or ""
|
||||
).strip()
|
||||
else:
|
||||
rendered_model = str(model_entry or "").strip()
|
||||
if not rendered_model:
|
||||
continue
|
||||
choice_id = self._encode_model_choice(row_provider, rendered_model)
|
||||
if choice_id in seen_ids:
|
||||
continue
|
||||
is_current = (
|
||||
row_provider == normalized_provider and rendered_model == model
|
||||
)
|
||||
description = f"Provider: {provider_name}"
|
||||
if is_current:
|
||||
description += " • current"
|
||||
available_models.append(
|
||||
ModelInfo(
|
||||
model_id=choice_id,
|
||||
name=f"{provider_name} · {rendered_model}",
|
||||
description=description,
|
||||
)
|
||||
)
|
||||
seen_ids.add(choice_id)
|
||||
|
||||
# Named user-defined endpoints (providers: / custom_providers:)
|
||||
# are invisible to canonical provider enumeration — append them
|
||||
# so editor clients can select them like the TUI /model picker.
|
||||
for named_slug, named_label, named_catalog in _named_custom_provider_catalogs():
|
||||
for named_model, named_desc in named_catalog:
|
||||
named_choice = self._encode_model_choice(named_slug, named_model)
|
||||
if not named_choice or named_choice in seen_ids:
|
||||
continue
|
||||
named_parts = [f"Provider: {named_label}"]
|
||||
if named_desc:
|
||||
named_parts.append(str(named_desc).strip())
|
||||
if named_slug == normalized_provider and named_model == model:
|
||||
named_parts.append("current")
|
||||
available_models.append(
|
||||
ModelInfo(
|
||||
model_id=named_choice,
|
||||
name=named_model,
|
||||
description=" • ".join(part for part in named_parts if part),
|
||||
)
|
||||
)
|
||||
seen_ids.add(named_choice)
|
||||
|
||||
current_model_id = self._encode_model_choice(normalized_provider, model)
|
||||
if current_model_id and current_model_id not in seen_ids:
|
||||
provider_name = provider_label(normalized_provider)
|
||||
available_models.insert(
|
||||
0,
|
||||
ModelInfo(
|
||||
model_id=current_model_id,
|
||||
name=model,
|
||||
name=f"{provider_name} · {model}",
|
||||
description=f"Provider: {provider_name} • current",
|
||||
),
|
||||
)
|
||||
@@ -969,11 +1146,49 @@ class HermesACPAgent(acp.Agent):
|
||||
return text
|
||||
return ""
|
||||
|
||||
@staticmethod
|
||||
def _history_summary_meta(message: dict[str, Any], text: str) -> dict[str, Any] | None:
|
||||
"""Build the ``_meta`` payload for a replayed compaction summary.
|
||||
|
||||
Compaction summaries are persisted as ordinary history messages —
|
||||
standalone handoffs under ``role="user"`` OR ``role="assistant"``
|
||||
(the compressor picks whichever role keeps alternation valid), and
|
||||
merge-into-tail messages where the summary is appended after the
|
||||
first preserved tail message's real content. Without a wire flag,
|
||||
ACP frontends render all of these as ordinary turns.
|
||||
|
||||
Two distinct keys under ``_meta.hermes`` (ACP's extensibility
|
||||
channel), so clients cannot accidentally hide real content:
|
||||
|
||||
* ``compactionSummary: true`` — the entire chunk is the handoff
|
||||
summary. Safe to restyle or collapse wholesale.
|
||||
* ``containsCompactionSummary: true`` — a merged-tail message: real
|
||||
preserved turn content followed by the summary. Clients may style
|
||||
it, but collapsing the whole chunk would hide the preserved
|
||||
content, hence the separate key.
|
||||
|
||||
Detection honors the in-process ``_compressed_summary`` flag and
|
||||
falls back to content classification, so it also works for a
|
||||
DB-reloaded session that lost the in-memory flag.
|
||||
"""
|
||||
kind = ContextCompressor.classify_summary_content(text)
|
||||
if kind is None and message.get(COMPRESSED_SUMMARY_METADATA_KEY):
|
||||
# Flagged in-process but content didn't classify (e.g. future
|
||||
# prefix drift): treat as a standalone summary — the flag is only
|
||||
# ever set on summary-bearing messages.
|
||||
kind = "standalone"
|
||||
if kind == "standalone":
|
||||
return {"hermes": {"compactionSummary": True}}
|
||||
if kind == "merged":
|
||||
return {"hermes": {"containsCompactionSummary": True}}
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _history_message_update(
|
||||
*,
|
||||
role: str,
|
||||
text: str,
|
||||
field_meta: dict[str, Any] | None = None,
|
||||
) -> UserMessageChunk | AgentMessageChunk | None:
|
||||
"""Build an ACP history replay update for a user/assistant message."""
|
||||
block = TextContentBlock(type="text", text=text)
|
||||
@@ -981,11 +1196,13 @@ class HermesACPAgent(acp.Agent):
|
||||
return UserMessageChunk(
|
||||
session_update="user_message_chunk",
|
||||
content=block,
|
||||
field_meta=field_meta,
|
||||
)
|
||||
if role == "assistant":
|
||||
return AgentMessageChunk(
|
||||
session_update="agent_message_chunk",
|
||||
content=block,
|
||||
field_meta=field_meta,
|
||||
)
|
||||
return None
|
||||
|
||||
@@ -1056,7 +1273,11 @@ class HermesACPAgent(acp.Agent):
|
||||
if role == "user":
|
||||
text = self._history_message_text(message)
|
||||
if text:
|
||||
update = self._history_message_update(role=role, text=text)
|
||||
update = self._history_message_update(
|
||||
role=role,
|
||||
text=text,
|
||||
field_meta=self._history_summary_meta(message, text),
|
||||
)
|
||||
if update is not None and not await _send(update):
|
||||
return
|
||||
continue
|
||||
@@ -1068,7 +1289,11 @@ class HermesACPAgent(acp.Agent):
|
||||
|
||||
text = self._history_message_text(message)
|
||||
if text:
|
||||
update = self._history_message_update(role=role, text=text)
|
||||
update = self._history_message_update(
|
||||
role=role,
|
||||
text=text,
|
||||
field_meta=self._history_summary_meta(message, text),
|
||||
)
|
||||
if update is not None and not await _send(update):
|
||||
return
|
||||
|
||||
@@ -1218,12 +1443,19 @@ class HermesACPAgent(acp.Agent):
|
||||
with state.runtime_lock:
|
||||
if state.is_running and state.current_prompt_text:
|
||||
state.interrupted_prompt_text = state.current_prompt_text
|
||||
state.cancel_event.set()
|
||||
try:
|
||||
if getattr(state, "agent", None) and hasattr(state.agent, "interrupt"):
|
||||
state.agent.interrupt()
|
||||
except Exception:
|
||||
logger.debug("Failed to interrupt ACP session %s", session_id, exc_info=True)
|
||||
# Publish cancellation and hard-stop the agent before another
|
||||
# prompt can acquire this lock and mistake the turn for
|
||||
# redirectable work.
|
||||
state.cancel_event.set()
|
||||
try:
|
||||
if getattr(state, "agent", None) and hasattr(state.agent, "interrupt"):
|
||||
state.agent.interrupt()
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"Failed to interrupt ACP session %s",
|
||||
session_id,
|
||||
exc_info=True,
|
||||
)
|
||||
logger.info("Cancelled session %s", session_id)
|
||||
|
||||
async def fork_session(
|
||||
@@ -1352,6 +1584,26 @@ class HermesACPAgent(acp.Agent):
|
||||
elif rewrite_idle:
|
||||
user_text = steer_text
|
||||
user_content = steer_text
|
||||
elif (
|
||||
text_only_prompt
|
||||
and isinstance(user_content, str)
|
||||
and not user_text.startswith("/")
|
||||
):
|
||||
# Some ACP clients implement "stop and send" as two protocol calls:
|
||||
# cancel the active prompt, then submit plain correction text. Keep
|
||||
# the cancelled request attached so deictic follow-ups ("not that
|
||||
# file") still have an explicit target.
|
||||
interrupted_prompt = ""
|
||||
with state.runtime_lock:
|
||||
if not state.is_running and state.interrupted_prompt_text:
|
||||
interrupted_prompt = state.interrupted_prompt_text
|
||||
state.interrupted_prompt_text = ""
|
||||
if interrupted_prompt:
|
||||
user_text = (
|
||||
f"{interrupted_prompt}\n\n"
|
||||
f"User correction/guidance after interrupt: {user_text}"
|
||||
)
|
||||
user_content = user_text
|
||||
|
||||
# Intercept slash commands — handle locally without calling the LLM.
|
||||
# Slash commands are text-only; if the client included images/resources,
|
||||
@@ -1366,23 +1618,54 @@ class HermesACPAgent(acp.Agent):
|
||||
await self._send_usage_update(state)
|
||||
return PromptResponse(stop_reason="end_turn")
|
||||
|
||||
# If Zed sends another regular prompt while the same ACP session is
|
||||
# still running, queue it instead of racing two AIAgent loops against
|
||||
# the same state.history. /steer and /queue are handled above and can
|
||||
# land immediately.
|
||||
# If the client sends another regular text prompt while this ACP session
|
||||
# is running, route it through the core active-turn redirect. Rich media
|
||||
# and older runtimes retain the proven next-turn queue fallback.
|
||||
redirected = False
|
||||
queued_depth: int | None = None
|
||||
with state.runtime_lock:
|
||||
if state.is_running:
|
||||
queued_text = user_text or "[Image attachment]"
|
||||
state.queued_prompts.append(queued_text)
|
||||
depth = len(state.queued_prompts)
|
||||
if self._conn:
|
||||
update = acp.update_agent_message_text(
|
||||
f"Queued for the next turn. ({depth} queued)"
|
||||
if (
|
||||
text_only_prompt
|
||||
and isinstance(user_content, str)
|
||||
and getattr(
|
||||
state.agent,
|
||||
"_supports_active_turn_redirect",
|
||||
False,
|
||||
)
|
||||
await self._conn.session_update(session_id, update)
|
||||
return PromptResponse(stop_reason="end_turn")
|
||||
state.is_running = True
|
||||
state.current_prompt_text = user_text or "[Image attachment]"
|
||||
is True
|
||||
and hasattr(state.agent, "redirect")
|
||||
):
|
||||
try:
|
||||
redirected = bool(state.agent.redirect(user_content))
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"ACP active-turn redirect failed for %s",
|
||||
session_id,
|
||||
exc_info=True,
|
||||
)
|
||||
if not redirected:
|
||||
queued_text = user_text or "[Image attachment]"
|
||||
state.queued_prompts.append(queued_text)
|
||||
queued_depth = len(state.queued_prompts)
|
||||
else:
|
||||
state.is_running = True
|
||||
state.current_prompt_text = user_text or "[Image attachment]"
|
||||
|
||||
if redirected:
|
||||
if self._conn:
|
||||
update = acp.update_agent_message_text(
|
||||
"Redirected the active turn with your correction."
|
||||
)
|
||||
await self._conn.session_update(session_id, update)
|
||||
return PromptResponse(stop_reason="end_turn")
|
||||
if queued_depth is not None:
|
||||
if self._conn:
|
||||
update = acp.update_agent_message_text(
|
||||
f"Queued for the next turn. ({queued_depth} queued)"
|
||||
)
|
||||
await self._conn.session_update(session_id, update)
|
||||
return PromptResponse(stop_reason="end_turn")
|
||||
|
||||
logger.info("Prompt on session %s: %s", session_id, user_text[:100])
|
||||
|
||||
@@ -1478,7 +1761,16 @@ class HermesACPAgent(acp.Agent):
|
||||
clear_session_vars,
|
||||
set_session_vars,
|
||||
)
|
||||
session_tokens = set_session_vars(session_key=session_id)
|
||||
# ``cwd`` pins the logical working directory for this context,
|
||||
# which is what the system prompt's "Current working directory"
|
||||
# line reports (agent/prompt_builder.py -> resolve_agent_cwd).
|
||||
# Without it the prompt advertises the global Hermes workspace
|
||||
# while the tools are rooted at the client's project, so the
|
||||
# model emits absolute paths under ~/.hermes/workspace and the
|
||||
# edit silently lands outside the editor's workspace.
|
||||
session_tokens = set_session_vars(
|
||||
session_key=session_id, cwd=state.cwd,
|
||||
)
|
||||
except Exception:
|
||||
session_tokens = None
|
||||
clear_session_vars = None # type: ignore[assignment]
|
||||
@@ -1756,7 +2048,7 @@ class HermesACPAgent(acp.Agent):
|
||||
"tools": self._cmd_tools,
|
||||
"context": self._cmd_context,
|
||||
"reset": self._cmd_reset,
|
||||
"compact": self._cmd_compact,
|
||||
"compress": self._cmd_compress,
|
||||
"steer": self._cmd_steer,
|
||||
"queue": self._cmd_queue,
|
||||
"version": self._cmd_version,
|
||||
@@ -1765,8 +2057,26 @@ class HermesACPAgent(acp.Agent):
|
||||
if handler is None:
|
||||
return None # not a known command — let the LLM handle it
|
||||
|
||||
try:
|
||||
# Slash handlers run on the event-loop thread, OUTSIDE the per-turn
|
||||
# contextvars.copy_context() that pins the session cwd for the agent
|
||||
# call. ``/compress`` and ``/model`` reach code that REBUILDS the
|
||||
# system prompt (agent._build_system_prompt -> resolve_agent_cwd), so
|
||||
# an unpinned handler bakes the Hermes install tree into the session's
|
||||
# cached prompt — persisted, and therefore poisoning every later turn
|
||||
# even though the turn itself is pinned. Pin inside a fresh context so
|
||||
# the write can't leak into other concurrent ACP sessions and needs no
|
||||
# teardown.
|
||||
def _dispatch() -> str | None:
|
||||
try:
|
||||
from agent.runtime_cwd import set_session_cwd
|
||||
|
||||
set_session_cwd(state.cwd)
|
||||
except Exception:
|
||||
logger.debug("Could not pin ACP session cwd for slash command", exc_info=True)
|
||||
return handler(args, state)
|
||||
|
||||
try:
|
||||
return contextvars.copy_context().run(_dispatch)
|
||||
except Exception as e:
|
||||
logger.error("Slash command /%s error: %s", cmd, e, exc_info=True)
|
||||
return f"Error executing /{cmd}: {e}"
|
||||
@@ -1826,8 +2136,8 @@ class HermesACPAgent(acp.Agent):
|
||||
return "No tools available."
|
||||
lines = [f"Available tools ({len(tools)}):"]
|
||||
for t in tools:
|
||||
name = t.get("function", {}).get("name", "?")
|
||||
desc = t.get("function", {}).get("description", "")
|
||||
name = (t.get("function") or {}).get("name", "?")
|
||||
desc = (t.get("function") or {}).get("description", "")
|
||||
# Truncate long descriptions
|
||||
if len(desc) > 80:
|
||||
desc = desc[:77] + "..."
|
||||
@@ -1898,7 +2208,7 @@ class HermesACPAgent(acp.Agent):
|
||||
lines.append(
|
||||
f"Compression: due now (threshold ~{threshold_tokens:,}"
|
||||
+ (f", {threshold_pct:.0f}%" if threshold_pct else "")
|
||||
+ "). Run /compact."
|
||||
+ "). Run /compress."
|
||||
)
|
||||
else:
|
||||
lines.append(
|
||||
@@ -1911,9 +2221,12 @@ class HermesACPAgent(acp.Agent):
|
||||
lines.append(f"Compression threshold: ~{threshold_tokens:,} tokens")
|
||||
|
||||
if getattr(agent, "compression_enabled", True) is False:
|
||||
lines.append("Compression is disabled for this agent.")
|
||||
lines.append(
|
||||
"Auto-compaction is disabled (compression.enabled: false); "
|
||||
"/compress still compresses manually."
|
||||
)
|
||||
else:
|
||||
lines.append("Tip: run /compact to compress manually before the threshold.")
|
||||
lines.append("Tip: run /compress to compress manually before the threshold.")
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
@@ -1933,13 +2246,14 @@ class HermesACPAgent(acp.Agent):
|
||||
return "Conversation history cleared. Agent session state reset failed; see logs."
|
||||
return "Conversation history cleared."
|
||||
|
||||
def _cmd_compact(self, args: str, state: SessionState) -> str:
|
||||
def _cmd_compress(self, args: str, state: SessionState) -> str:
|
||||
if not state.history:
|
||||
return "Nothing to compress — conversation is empty."
|
||||
try:
|
||||
agent = state.agent
|
||||
if not getattr(agent, "compression_enabled", True):
|
||||
return "Context compression is disabled for this agent."
|
||||
# No compression_enabled gate: the flag disables *automatic*
|
||||
# compaction only; manual /compress must keep working (matches
|
||||
# the CLI /compress and gateway handlers).
|
||||
if not hasattr(agent, "_compress_context"):
|
||||
return "Context compression not available for this agent."
|
||||
|
||||
@@ -1964,6 +2278,7 @@ class HermesACPAgent(acp.Agent):
|
||||
getattr(agent, "_cached_system_prompt", "") or "",
|
||||
approx_tokens=approx_tokens,
|
||||
task_id=state.session_id,
|
||||
force=True,
|
||||
)
|
||||
finally:
|
||||
agent._session_db = original_session_db
|
||||
|
||||
@@ -1,16 +0,0 @@
|
||||
{
|
||||
"id": "hermes-agent",
|
||||
"name": "Hermes Agent",
|
||||
"version": "0.18.2",
|
||||
"description": "Self-improving open-source AI agent by Nous Research with ACP editor integration, persistent memory, skills, and rich tool support.",
|
||||
"repository": "https://github.com/NousResearch/hermes-agent",
|
||||
"website": "https://hermes-agent.nousresearch.com/docs/user-guide/features/acp",
|
||||
"authors": ["Nous Research"],
|
||||
"license": "MIT",
|
||||
"distribution": {
|
||||
"uvx": {
|
||||
"package": "hermes-agent[acp]==0.18.2",
|
||||
"args": ["hermes-acp"]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,8 +0,0 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 16 16" width="16" height="16" fill="none">
|
||||
<path d="M8 1.5v13" stroke="currentColor" stroke-width="1.5" stroke-linecap="round"/>
|
||||
<path d="M8 3.25c-2.35-1.4-4.7-.95-6.25.35 1.85-.2 3.8.2 5.55 1.55" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
|
||||
<path d="M8 3.25c2.35-1.4 4.7-.95 6.25.35-1.85-.2-3.8.2-5.55 1.55" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
|
||||
<path d="M8 13.25c-2.3-1-3.05-2.65-1.35-4.15-2 .8-2.35 2.95-.35 4" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
|
||||
<path d="M8 13.25c2.3-1 3.05-2.65 1.35-4.15 2 .8 2.35 2.95.35 4" stroke="currentColor" stroke-width="1.1" stroke-linecap="round" stroke-linejoin="round"/>
|
||||
<circle cx="8" cy="1.8" r="1.1" fill="currentColor"/>
|
||||
</svg>
|
||||
|
Before Width: | Height: | Size: 882 B |
@@ -701,6 +701,18 @@ def redeem_codex_reset_credit(
|
||||
remaining = max(0, available - 1)
|
||||
plural = "s" if remaining != 1 else ""
|
||||
if code == "reset":
|
||||
# The redeemed reset restores the account's quota upstream — lift any
|
||||
# persisted pool cooldowns so Hermes doesn't keep the credential
|
||||
# frozen behind the now-stale ``last_error_reset_at`` (issue #43747).
|
||||
try:
|
||||
from hermes_cli.auth import clear_codex_pool_quota_cooldowns
|
||||
|
||||
clear_codex_pool_quota_cooldowns()
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"Failed to clear Codex pool cooldowns after reset redemption",
|
||||
exc_info=True,
|
||||
)
|
||||
return CodexResetRedeemResult(
|
||||
status="reset",
|
||||
message=(
|
||||
|
||||
+617
-74
@@ -28,7 +28,7 @@ import time
|
||||
import uuid
|
||||
from datetime import datetime
|
||||
from typing import Any, Callable, Dict, List, Optional
|
||||
from urllib.parse import urlparse, parse_qs, urlunparse
|
||||
from urllib.parse import parse_qs, urlparse, urlunparse
|
||||
|
||||
from agent.context_compressor import ContextCompressor
|
||||
from agent.iteration_budget import IterationBudget
|
||||
@@ -48,6 +48,7 @@ from agent.tool_guardrails import (
|
||||
ToolGuardrailDecision,
|
||||
)
|
||||
from hermes_cli.config import cfg_get
|
||||
from hermes_cli.route_identity import normalize_route_base_url
|
||||
from hermes_cli.timeouts import get_provider_request_timeout
|
||||
from hermes_constants import get_hermes_home
|
||||
from utils import base_url_host_matches, is_truthy_value
|
||||
@@ -68,18 +69,188 @@ def _ra():
|
||||
return run_agent
|
||||
|
||||
|
||||
def _build_codex_gpt5_autoraise_notice(autoraise: Dict[str, Any]) -> str:
|
||||
def _moa_reference_output_allowed(agent: Any) -> bool:
|
||||
"""Keep MoA display events off only the machine-readable ``-Q`` surface."""
|
||||
return not (
|
||||
getattr(agent, "platform", None) == "cli"
|
||||
and getattr(agent, "tool_progress_mode", "all") == "off"
|
||||
)
|
||||
|
||||
|
||||
def _relay_moa_reference_event(agent: Any, event: str, **kwargs: Any) -> None:
|
||||
"""Relay MoA display events while preserving the ``-Q`` stdout contract."""
|
||||
if not _moa_reference_output_allowed(agent):
|
||||
return
|
||||
cb = getattr(agent, "tool_progress_callback", None)
|
||||
if cb is None:
|
||||
return
|
||||
try:
|
||||
if event == "moa.reference":
|
||||
cb(
|
||||
"moa.reference",
|
||||
str(kwargs.get("label") or ""),
|
||||
str(kwargs.get("text") or ""),
|
||||
None,
|
||||
moa_index=kwargs.get("index"),
|
||||
moa_count=kwargs.get("count"),
|
||||
)
|
||||
elif event == "moa.aggregating":
|
||||
cb(
|
||||
"moa.aggregating",
|
||||
str(kwargs.get("aggregator") or ""),
|
||||
None,
|
||||
None,
|
||||
moa_ref_count=kwargs.get("ref_count"),
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _normalize_route_base_url(base_url: Any) -> str:
|
||||
"""Canonicalize an endpoint URL for model-route identity comparisons."""
|
||||
return normalize_route_base_url(base_url)
|
||||
|
||||
|
||||
def _provider_default_routes(provider: str) -> set[str]:
|
||||
"""Return known exact default routes for a canonical provider id."""
|
||||
routes: set[str] = set()
|
||||
try:
|
||||
from hermes_cli.providers import HERMES_OVERLAYS, get_provider
|
||||
|
||||
overlay = HERMES_OVERLAYS.get(provider)
|
||||
provider_def = get_provider(provider, allow_network=False)
|
||||
for value in (
|
||||
getattr(overlay, "base_url_override", ""),
|
||||
getattr(provider_def, "base_url", ""),
|
||||
):
|
||||
route = _normalize_route_base_url(value)
|
||||
if route:
|
||||
routes.add(route)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
try:
|
||||
from providers import get_provider_profile
|
||||
|
||||
profile = get_provider_profile(provider)
|
||||
route = _normalize_route_base_url(
|
||||
getattr(profile, "base_url", "")
|
||||
)
|
||||
if route:
|
||||
routes.add(route)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
try:
|
||||
from hermes_cli.auth import PROVIDER_REGISTRY
|
||||
from hermes_cli.models import normalize_provider as normalize_model_provider
|
||||
from hermes_cli.providers import normalize_provider as normalize_registry_provider
|
||||
|
||||
for provider_id, config in PROVIDER_REGISTRY.items():
|
||||
canonical_id = normalize_registry_provider(
|
||||
normalize_model_provider(provider_id)
|
||||
)
|
||||
if canonical_id != provider:
|
||||
continue
|
||||
route = _normalize_route_base_url(
|
||||
getattr(config, "inference_base_url", "")
|
||||
)
|
||||
if route:
|
||||
routes.add(route)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if provider == "gemini":
|
||||
routes.update(
|
||||
f"{route.rstrip('/')}/openai"
|
||||
for route in list(routes)
|
||||
)
|
||||
return routes
|
||||
|
||||
|
||||
def _context_route_mismatch(
|
||||
configured_base_url: Any,
|
||||
active_base_url: Any,
|
||||
configured_provider: Any,
|
||||
active_provider: Any,
|
||||
*,
|
||||
already_normalized: bool = False,
|
||||
) -> bool:
|
||||
"""Return whether a context pin's configured route differs from runtime."""
|
||||
if already_normalized:
|
||||
configured_route = str(configured_base_url or "")
|
||||
active_route = str(active_base_url or "")
|
||||
else:
|
||||
configured_route = _normalize_route_base_url(configured_base_url)
|
||||
active_route = _normalize_route_base_url(active_base_url)
|
||||
if configured_route:
|
||||
return configured_route != active_route
|
||||
|
||||
configured_provider = str(configured_provider or "").strip()
|
||||
active_provider = str(active_provider or "").strip()
|
||||
if not configured_provider:
|
||||
return False
|
||||
try:
|
||||
from hermes_cli.models import normalize_provider as normalize_model_provider
|
||||
|
||||
configured_provider = normalize_model_provider(configured_provider)
|
||||
active_provider = normalize_model_provider(active_provider)
|
||||
except Exception:
|
||||
configured_provider = configured_provider.lower()
|
||||
active_provider = active_provider.lower()
|
||||
try:
|
||||
from hermes_cli.providers import normalize_provider as normalize_registry_provider
|
||||
|
||||
configured_provider = normalize_registry_provider(configured_provider)
|
||||
active_provider = normalize_registry_provider(active_provider)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if active_route:
|
||||
configured_routes = _provider_default_routes(configured_provider)
|
||||
return not configured_routes or active_route not in configured_routes
|
||||
return bool(
|
||||
configured_provider
|
||||
and active_provider
|
||||
and configured_provider != active_provider
|
||||
)
|
||||
|
||||
|
||||
def _normalize_custom_provider_name(value: Any) -> str:
|
||||
"""Mirror runtime normalization for a requested custom-provider identity."""
|
||||
return str(value or "").strip().lower().replace(" ", "-")
|
||||
|
||||
|
||||
def _custom_provider_runtime_ids(value: Any) -> set[str]:
|
||||
"""Return raw/menu identities that runtime accepts for a configured name."""
|
||||
normalized = _normalize_custom_provider_name(value)
|
||||
if not normalized:
|
||||
return set()
|
||||
return {normalized, f"custom:{normalized}"}
|
||||
|
||||
|
||||
def _build_codex_gpt5_autoraise_notice(
|
||||
autoraise: Dict[str, Any], context_length: Optional[int] = None
|
||||
) -> str:
|
||||
"""Build the one-time notice shown when Codex gpt-5.x raises compaction.
|
||||
|
||||
``autoraise`` is ``{"model": <slug>, "from": <old_ratio>, "to": <new_ratio>}``.
|
||||
The same text is printed inline for CLI users and replayed via
|
||||
``context_length`` is the live-resolved window from the context compressor
|
||||
(Codex's /models catalog is authoritative and can change server-side, e.g.
|
||||
the gpt-5.6 family's 272K → 372K → 272K shifts in July 2026), so the banner
|
||||
reports what this session actually got rather than a hardcoded cap. The
|
||||
same text is printed inline for CLI users and replayed via
|
||||
``status_callback`` for gateway users, so it must be self-contained and
|
||||
include the exact opt-back-out command.
|
||||
"""
|
||||
model = str(autoraise.get("model") or "gpt-5.4/5.5").strip().lower().rsplit("/", 1)[-1]
|
||||
# gpt-5.3-codex-spark has a native 128K window; the gpt-5.4/5.5/5.6 family
|
||||
# is capped at 272K by the Codex OAuth backend.
|
||||
cap = "128K" if model.startswith("gpt-5.3-codex-spark") else "272K"
|
||||
if isinstance(context_length, int) and context_length > 0:
|
||||
cap = f"{round(context_length / 1000)}K"
|
||||
else:
|
||||
# Static fallback when the resolved window isn't available:
|
||||
# gpt-5.3-codex-spark has a native 128K window; the gpt-5.4/5.5/5.6
|
||||
# family is capped at 272K by the Codex OAuth backend.
|
||||
cap = "128K" if model.startswith("gpt-5.3-codex-spark") else "272K"
|
||||
from_pct = int(round(autoraise["from"] * 100))
|
||||
to_pct = int(round(autoraise["to"] * 100))
|
||||
return (
|
||||
@@ -285,7 +456,6 @@ def init_agent(
|
||||
args: list[str] | None = None,
|
||||
model: str = "",
|
||||
max_iterations: int = 90, # Default tool-calling iterations (shared with subagents)
|
||||
tool_delay: float = 1.0,
|
||||
enabled_toolsets: List[str] = None,
|
||||
disabled_toolsets: List[str] = None,
|
||||
save_trajectories: bool = False,
|
||||
@@ -346,6 +516,7 @@ def init_agent(
|
||||
checkpoint_max_total_size_mb: int = 500,
|
||||
checkpoint_max_file_size_mb: int = 10,
|
||||
pass_session_id: bool = False,
|
||||
requested_provider: str = None,
|
||||
):
|
||||
"""
|
||||
Initialize the AI Agent.
|
||||
@@ -354,10 +525,10 @@ def init_agent(
|
||||
base_url (str): Base URL for the model API (optional)
|
||||
api_key (str): API key for authentication (optional, uses env var if not provided)
|
||||
provider (str): Provider identifier (optional; used for telemetry/routing hints)
|
||||
requested_provider (str): Original provider identity before runtime canonicalization
|
||||
api_mode (str): API mode override: "chat_completions" or "codex_responses"
|
||||
model (str): Model name to use (default: "anthropic/claude-opus-4.6")
|
||||
max_iterations (int): Maximum number of tool calling iterations (default: 90)
|
||||
tool_delay (float): Delay between tool calls in seconds (default: 1.0)
|
||||
enabled_toolsets (List[str]): Only enable tools from these toolsets (optional)
|
||||
disabled_toolsets (List[str]): Disable tools from these toolsets (optional)
|
||||
save_trajectories (bool): Whether to save conversation trajectories to JSONL files (default: False)
|
||||
@@ -403,7 +574,6 @@ def init_agent(
|
||||
# Shared iteration budget — parent creates, children inherit.
|
||||
# Consumed by every LLM turn across parent + all subagents.
|
||||
agent.iteration_budget = iteration_budget or IterationBudget(max_iterations)
|
||||
agent.tool_delay = tool_delay
|
||||
agent.save_trajectories = save_trajectories
|
||||
agent.verbose_logging = verbose_logging
|
||||
agent.quiet_mode = quiet_mode
|
||||
@@ -434,6 +604,11 @@ def init_agent(
|
||||
agent.base_url = base_url or ""
|
||||
provider_name = provider.strip().lower() if isinstance(provider, str) and provider.strip() else None
|
||||
agent.provider = provider_name or ""
|
||||
agent.requested_provider = (
|
||||
requested_provider.strip().lower()
|
||||
if isinstance(requested_provider, str) and requested_provider.strip()
|
||||
else agent.provider
|
||||
)
|
||||
agent._credential_pool = credential_pool
|
||||
agent.acp_command = acp_command or command
|
||||
agent.acp_args = list(acp_args or args or [])
|
||||
@@ -467,6 +642,13 @@ def init_agent(
|
||||
# AWS Bedrock — auto-detect from provider name or base URL
|
||||
# (bedrock-runtime.<region>.amazonaws.com).
|
||||
agent.api_mode = "bedrock_converse"
|
||||
elif agent.provider in {"nous", "nous-portal", "nousresearch"}:
|
||||
# Portal is dual-wire: anthropic/* → Messages, everything else →
|
||||
# chat_completions. Callers that already pass api_mode win above;
|
||||
# this covers direct AIAgent construction without a resolved runtime.
|
||||
from hermes_cli.providers import nous_api_mode
|
||||
|
||||
agent.api_mode = nous_api_mode(agent.model)
|
||||
else:
|
||||
agent.api_mode = "chat_completions"
|
||||
|
||||
@@ -586,6 +768,8 @@ def init_agent(
|
||||
agent._execution_thread_id: int | None = None # Set at run_conversation() start
|
||||
agent._interrupt_thread_signal_pending = False
|
||||
agent._client_lock = threading.RLock()
|
||||
agent._model_request_active = threading.Event()
|
||||
agent._supports_active_turn_redirect = True
|
||||
|
||||
# /steer mechanism — inject a user note into the next tool result
|
||||
# without interrupting the agent. Unlike interrupt(), steer() does
|
||||
@@ -597,6 +781,13 @@ def init_agent(
|
||||
agent._pending_steer: Optional[str] = None
|
||||
agent._pending_steer_lock = threading.Lock()
|
||||
|
||||
# Active-turn redirect mechanism. A regular follow-up sent while the model
|
||||
# is generating is different from a hard /stop: preserve the valid turn
|
||||
# prefix, cancel only the in-flight model request, and rebuild its tail with
|
||||
# the correction. The loop drains this slot at a role-safe boundary.
|
||||
agent._pending_redirect: Optional[str] = None
|
||||
agent._pending_redirect_lock = threading.Lock()
|
||||
|
||||
# Concurrent-tool worker thread tracking. `_execute_tool_calls_concurrent`
|
||||
# runs each tool on its own ThreadPoolExecutor worker — those worker
|
||||
# threads have tids distinct from `_execution_thread_id`, so
|
||||
@@ -636,9 +827,10 @@ def init_agent(
|
||||
# Anthropic prompt caching: auto-enabled for Claude models on native
|
||||
# Anthropic, OpenRouter, and third-party gateways that speak the
|
||||
# Anthropic protocol (``api_mode == 'anthropic_messages'``). Reduces
|
||||
# input costs by ~75% on multi-turn conversations. Uses system_and_3
|
||||
# strategy (4 breakpoints). See ``_anthropic_prompt_cache_policy``
|
||||
# for the layout-vs-transport decision.
|
||||
# input costs by ~75% on multi-turn conversations. Uses four breakpoints:
|
||||
# the static system prefix, full system prompt, and last two messages
|
||||
# (falling back to system-and-3 when no static prefix is available). See
|
||||
# ``_anthropic_prompt_cache_policy`` for the layout-vs-transport decision.
|
||||
agent._use_prompt_caching, agent._use_native_cache_layout = (
|
||||
agent._anthropic_prompt_cache_policy()
|
||||
)
|
||||
@@ -648,7 +840,7 @@ def init_agent(
|
||||
# sessions with >5-minute pauses between turns (#14971).
|
||||
agent._cache_ttl = "5m"
|
||||
try:
|
||||
from hermes_cli.config import load_config as _load_pc_cfg
|
||||
from hermes_cli.config import load_config_readonly as _load_pc_cfg
|
||||
|
||||
_pc_cfg = _load_pc_cfg().get("prompt_caching", {}) or {}
|
||||
_ttl = _pc_cfg.get("cache_ttl", "5m")
|
||||
@@ -692,8 +884,10 @@ def init_agent(
|
||||
# report cumulative micros spent. Surfaced behind HERMES_DEV_CREDITS.
|
||||
agent._credits_state = None
|
||||
agent._credits_session_start_micros = None
|
||||
# Threshold-notice latch (L4): active sticky-notice keys + the warn90 crossing gate.
|
||||
agent._credits_latch = {"active": set(), "seen_below_90": False, "usage_band": None}
|
||||
# Threshold-notice latch (L4): active sticky-notice keys + the crossing gates.
|
||||
from agent.credits_tracker import new_credits_latch
|
||||
|
||||
agent._credits_latch = new_credits_latch()
|
||||
|
||||
# OpenRouter response cache hit counter — incremented when
|
||||
# X-OpenRouter-Cache-Status: HIT is seen in streaming response headers.
|
||||
@@ -869,49 +1063,20 @@ def init_agent(
|
||||
elif isinstance(effective_key, str) and len(effective_key) > 12:
|
||||
print(f"🔑 Using token: {effective_key[:8]}...{effective_key[-4:]}")
|
||||
elif agent.provider == "moa":
|
||||
from agent.moa_loop import MoAClient
|
||||
from agent.moa_loop import build_moa_facade
|
||||
agent.api_mode = "chat_completions"
|
||||
|
||||
# Route reference-model outputs to the agent's tool_progress_callback so
|
||||
# build_moa_facade wires the reference relay that routes
|
||||
# reference-model outputs to the agent's tool_progress_callback so
|
||||
# every surface that already consumes it (CLI spinner/scrollback, TUI,
|
||||
# desktop, gateway) can show each reference's answer as a labelled block
|
||||
# before the aggregator acts. The facade emits "moa.reference" and
|
||||
# "moa.aggregating" events; we forward them through the same callback
|
||||
# the tool lifecycle uses. Best-effort and cache-safe — these are
|
||||
# display-only events, they never touch the message history.
|
||||
def _moa_reference_relay(event: str, **kwargs: Any) -> None:
|
||||
cb = getattr(agent, "tool_progress_callback", None)
|
||||
if cb is None:
|
||||
return
|
||||
try:
|
||||
if event == "moa.reference":
|
||||
label = str(kwargs.get("label") or "")
|
||||
text = str(kwargs.get("text") or "")
|
||||
idx = kwargs.get("index")
|
||||
count = kwargs.get("count")
|
||||
cb(
|
||||
"moa.reference",
|
||||
label,
|
||||
text,
|
||||
None,
|
||||
moa_index=idx,
|
||||
moa_count=count,
|
||||
)
|
||||
elif event == "moa.aggregating":
|
||||
cb(
|
||||
"moa.aggregating",
|
||||
str(kwargs.get("aggregator") or ""),
|
||||
None,
|
||||
None,
|
||||
moa_ref_count=kwargs.get("ref_count"),
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
agent.client = MoAClient(
|
||||
agent.model or "default",
|
||||
reference_callback=_moa_reference_relay,
|
||||
)
|
||||
# desktop, gateway) can show each reference's answer as a labelled
|
||||
# block before the aggregator acts. The facade emits "moa.reference",
|
||||
# "moa.progress", "moa.phase", and "moa.aggregating" events, forwarded
|
||||
# through the same callback the tool lifecycle uses. Best-effort and
|
||||
# cache-safe — display-only events, they never touch the message
|
||||
# history. The factory is shared with the fallback-restore/recovery
|
||||
# paths so a restored facade keeps emitting these events (#53802).
|
||||
agent.client = build_moa_facade(agent, agent.model)
|
||||
agent._client_kwargs = {}
|
||||
agent.api_key = api_key or "moa-virtual-provider"
|
||||
agent.base_url = "moa://local"
|
||||
@@ -925,7 +1090,7 @@ def init_agent(
|
||||
# Guardrail config — read from config.yaml at init time.
|
||||
agent._bedrock_guardrail_config = None
|
||||
try:
|
||||
from hermes_cli.config import load_config as _load_br_cfg
|
||||
from hermes_cli.config import load_config_readonly as _load_br_cfg
|
||||
_gr = _load_br_cfg().get("bedrock", {}).get("guardrail", {})
|
||||
if _gr.get("guardrail_identifier") and _gr.get("guardrail_version"):
|
||||
agent._bedrock_guardrail_config = {
|
||||
@@ -990,10 +1155,14 @@ def init_agent(
|
||||
elif base_url_host_matches(effective_base, "chatgpt.com"):
|
||||
from agent.auxiliary_client import _codex_cloudflare_headers
|
||||
client_kwargs["default_headers"] = _codex_cloudflare_headers(api_key)
|
||||
elif base_url_host_matches(effective_base, "x.ai"):
|
||||
from tools.xai_http import hermes_xai_default_headers
|
||||
|
||||
client_kwargs["default_headers"] = hermes_xai_default_headers()
|
||||
elif "default_headers" not in client_kwargs:
|
||||
# Fall back to profile.default_headers for providers that
|
||||
# declare custom headers (e.g. Kimi User-Agent on non-kimi.com
|
||||
# endpoints).
|
||||
# declare custom headers (e.g. Vercel AI Gateway attribution,
|
||||
# Kimi User-Agent on non-kimi.com endpoints).
|
||||
try:
|
||||
from providers import get_provider_profile as _gpf
|
||||
_ph = _gpf(agent.provider)
|
||||
@@ -1177,6 +1346,13 @@ def init_agent(
|
||||
print("⚠️ Warning: API key appears invalid or missing")
|
||||
except Exception as e:
|
||||
raise RuntimeError(f"Failed to initialize OpenAI client: {e}")
|
||||
|
||||
# Keep a stable identity for the pool entry that supplied this runtime.
|
||||
# OAuth refreshes can replace the runtime token before a failed request is
|
||||
# recovered, so the mutable API-key value alone cannot reliably attribute
|
||||
# the failure to its source entry.
|
||||
from agent.agent_runtime_helpers import sync_credential_pool_entry_id
|
||||
sync_credential_pool_entry_id(agent)
|
||||
|
||||
# Provider fallback chain — ordered list of backup providers tried
|
||||
# when the primary is exhausted (rate-limit, overload, connection
|
||||
@@ -1301,7 +1477,7 @@ def init_agent(
|
||||
# reads the JSON files directly. See run_agent._save_session_log.
|
||||
agent._session_json_enabled = False
|
||||
try:
|
||||
from hermes_cli.config import load_config as _load_sess_cfg
|
||||
from hermes_cli.config import load_config_readonly as _load_sess_cfg
|
||||
_sess_cfg = (_load_sess_cfg().get("sessions") or {})
|
||||
agent._session_json_enabled = bool(_sess_cfg.get("write_json_snapshots", False))
|
||||
except Exception:
|
||||
@@ -1323,6 +1499,9 @@ def init_agent(
|
||||
|
||||
# Cached system prompt -- built once per session, only rebuilt on compression
|
||||
agent._cached_system_prompt: Optional[str] = None
|
||||
# Cross-session-stable prefix of the cached prompt. It remains separate
|
||||
# from the persisted string and is used only to place an early cache marker.
|
||||
agent._cached_system_prompt_static: Optional[str] = None
|
||||
|
||||
# Filesystem checkpoint manager (transparent — not a tool)
|
||||
from tools.checkpoint_manager import CheckpointManager
|
||||
@@ -1369,7 +1548,7 @@ def init_agent(
|
||||
|
||||
# Load config once for memory, skills, and compression sections
|
||||
try:
|
||||
from hermes_cli.config import load_config as _load_agent_config
|
||||
from hermes_cli.config import load_config_readonly as _load_agent_config
|
||||
_agent_cfg = _load_agent_config()
|
||||
except Exception:
|
||||
_agent_cfg = {}
|
||||
@@ -1427,7 +1606,14 @@ def init_agent(
|
||||
agent._memory_nudge_interval = 10
|
||||
agent._turns_since_memory = 0
|
||||
agent._iters_since_skill = 0
|
||||
if not skip_memory:
|
||||
# A flush/background agent may pass skip_memory=True to avoid spinning up an
|
||||
# external memory *provider*, but if the caller also explicitly enables the
|
||||
# "memory" toolset it still needs the built-in file-backed store — otherwise
|
||||
# the memory tool dispatches with store=None and every call fails (#65429).
|
||||
# So the built-in store is created unless memory is globally disabled, while
|
||||
# the external-provider block below stays gated on skip_memory.
|
||||
_memory_toolset_requested = "memory" in (agent.enabled_toolsets or [])
|
||||
if not skip_memory or _memory_toolset_requested:
|
||||
try:
|
||||
mem_config = _agent_cfg.get("memory", {})
|
||||
agent._memory_enabled = mem_config.get("memory_enabled", False)
|
||||
@@ -1647,6 +1833,89 @@ def init_agent(
|
||||
compression_enabled = str(_compression_cfg.get("enabled", True)).lower() in {"true", "1", "yes"}
|
||||
compression_target_ratio = float(_compression_cfg.get("target_ratio", 0.20))
|
||||
compression_protect_last = int(_compression_cfg.get("protect_last_n", 20))
|
||||
# Minimum REAL (actionable) user messages guaranteed to survive in the
|
||||
# uncompressed tail (compression.min_tail_user_messages). Default 1
|
||||
# preserves current behavior exactly — the existing single-user tail
|
||||
# anchor. Values > 1 extend the guarantee to the last N actionable
|
||||
# user turns. Booleans rejected (bool subclasses int), non-int-like
|
||||
# values fall back to 1, floor at 1.
|
||||
_raw_min_tail_users = _compression_cfg.get("min_tail_user_messages", 1)
|
||||
if isinstance(_raw_min_tail_users, bool):
|
||||
compression_min_tail_users = 1
|
||||
elif isinstance(_raw_min_tail_users, int):
|
||||
compression_min_tail_users = _raw_min_tail_users
|
||||
elif isinstance(_raw_min_tail_users, float):
|
||||
compression_min_tail_users = (
|
||||
int(_raw_min_tail_users) if _raw_min_tail_users.is_integer() else 1
|
||||
)
|
||||
else:
|
||||
try:
|
||||
compression_min_tail_users = int(str(_raw_min_tail_users).strip())
|
||||
except (TypeError, ValueError):
|
||||
compression_min_tail_users = 1
|
||||
if compression_min_tail_users < 1:
|
||||
compression_min_tail_users = 1
|
||||
# Cap on compression retry rounds before a turn gives up with "max
|
||||
# compression attempts reached" (compression.max_attempts). Hardcoding 3
|
||||
# strands sessions that legitimately need more rounds — e.g. a restart
|
||||
# history reload whose incompressible tool schemas keep the request
|
||||
# estimate above the threshold even though the messages compress fine
|
||||
# (the #62605 failure class). Default 3 preserves current behavior, so
|
||||
# an unset key is behavior-neutral; validated >= 1, hard-capped at 10,
|
||||
# and any non-int-like value falls back to 3. Booleans are rejected
|
||||
# (bool subclasses int, so int(True) would silently become 1) and
|
||||
# fractional floats are rejected rather than truncated — "4.7 attempts"
|
||||
# is a config mistake, not a request for 4.
|
||||
_raw_max_attempts = _compression_cfg.get("max_attempts", 3)
|
||||
if isinstance(_raw_max_attempts, bool):
|
||||
compression_max_attempts = 3
|
||||
elif isinstance(_raw_max_attempts, int):
|
||||
compression_max_attempts = _raw_max_attempts
|
||||
elif isinstance(_raw_max_attempts, float):
|
||||
compression_max_attempts = (
|
||||
int(_raw_max_attempts) if _raw_max_attempts.is_integer() else 3
|
||||
)
|
||||
else:
|
||||
try:
|
||||
compression_max_attempts = int(str(_raw_max_attempts).strip())
|
||||
except (TypeError, ValueError):
|
||||
compression_max_attempts = 3
|
||||
if compression_max_attempts < 1:
|
||||
compression_max_attempts = 3
|
||||
compression_max_attempts = min(compression_max_attempts, 10)
|
||||
|
||||
def _parse_prune_int(raw, default):
|
||||
# Same parser semantics as compression.max_attempts above: reject
|
||||
# booleans (bool subclasses int — YAML `true` would coerce to 1),
|
||||
# reject fractional floats rather than truncating them, accept
|
||||
# integral floats and numeric strings, fall back to the default on
|
||||
# anything else.
|
||||
if isinstance(raw, bool):
|
||||
return default
|
||||
if isinstance(raw, int):
|
||||
return raw
|
||||
if isinstance(raw, float):
|
||||
return int(raw) if raw.is_integer() else default
|
||||
try:
|
||||
return int(str(raw).strip())
|
||||
except (TypeError, ValueError):
|
||||
return default
|
||||
|
||||
# Opt-in proactive tool-result prune trigger (0 = disabled — the
|
||||
# default, so an unset key is behavior-neutral). Negative values are
|
||||
# treated as disabled rather than erroring.
|
||||
compression_proactive_prune_tokens = max(
|
||||
0, _parse_prune_int(_compression_cfg.get("proactive_prune_tokens", 0), 0)
|
||||
)
|
||||
compression_proactive_prune_min_chars = _parse_prune_int(
|
||||
_compression_cfg.get("proactive_prune_min_result_chars", 8000), 8000
|
||||
)
|
||||
compression_proactive_prune_min_reclaim = max(
|
||||
0,
|
||||
_parse_prune_int(
|
||||
_compression_cfg.get("proactive_prune_min_reclaim_tokens", 4096), 4096
|
||||
),
|
||||
)
|
||||
# protect_first_n is the number of non-system messages to protect at
|
||||
# the head, in addition to the system prompt (which is always
|
||||
# implicitly protected by the compressor). Floor at 0 — a value of
|
||||
@@ -1659,13 +1928,65 @@ def init_agent(
|
||||
compression_abort_on_summary_failure = str(
|
||||
_compression_cfg.get("abort_on_summary_failure", False)
|
||||
).lower() in {"true", "1", "yes"}
|
||||
# Per-model threshold overrides: keys are substring-matched against the
|
||||
# model name (longest match wins). Empty dict = use the global threshold
|
||||
# for all models (backward compatible).
|
||||
_raw_model_thresholds = _compression_cfg.get("model_thresholds", {})
|
||||
if isinstance(_raw_model_thresholds, dict):
|
||||
compression_model_thresholds = {
|
||||
str(k): float(v) for k, v in _raw_model_thresholds.items()
|
||||
if isinstance(v, (int, float)) and not isinstance(v, bool)
|
||||
}
|
||||
else:
|
||||
compression_model_thresholds = {}
|
||||
# Absolute token cap: when set, compression triggers at the lower of
|
||||
# the ratio-based threshold and this absolute count. Clamped to the
|
||||
# model's context length at apply-time so a cap above the window is
|
||||
# a no-op (ratio-based threshold wins).
|
||||
compression_threshold_tokens = _compression_cfg.get("threshold_tokens")
|
||||
if compression_threshold_tokens is not None:
|
||||
try:
|
||||
compression_threshold_tokens = int(compression_threshold_tokens)
|
||||
if compression_threshold_tokens <= 0:
|
||||
compression_threshold_tokens = None
|
||||
except (TypeError, ValueError):
|
||||
compression_threshold_tokens = None
|
||||
# In-place compaction: when True, compress_context() rewrites the message
|
||||
# list + rebuilds the system prompt WITHOUT rotating the session id (no
|
||||
# parent_session_id chain, no `name #N` renumber). See #38763 and
|
||||
# agent/conversation_compression.py. Consumed by compress_context(), not the
|
||||
# compressor, so it rides on the agent.
|
||||
# Default True must match DEFAULT_CONFIG["compression"]["in_place"]
|
||||
# (#38763). default=False here previously flipped agents into rotation
|
||||
# mode whenever the merged config omitted the key (partial configs,
|
||||
# load_config failure → {}), re-arming the pre-lease drift abort.
|
||||
compression_in_place = is_truthy_value(
|
||||
_compression_cfg.get("in_place"), default=False
|
||||
_compression_cfg.get("in_place"), default=True
|
||||
)
|
||||
# Opt-in (default False): a micro-compaction pass rewrites already-sent
|
||||
# history every turn, which breaks the provider prompt-cache prefix on a
|
||||
# per-turn cadence rather than at an episodic boundary. That is the cost
|
||||
# `proactive_prune_min_reclaim_tokens` exists to amortize, so the feature
|
||||
# stays off until an operator opts in and accepts the tradeoff.
|
||||
compression_micro_compact = is_truthy_value(
|
||||
_compression_cfg.get("micro_compact"), default=False
|
||||
)
|
||||
# How often a pass runs, in completed turns. Each pass rewrites
|
||||
# already-sent history and costs one prompt-cache break, so this is the
|
||||
# dial for how often that cost is paid: 1 = every turn (most aggressive
|
||||
# reclaim), 5 = one break per five turns. Clamped to >= 1.
|
||||
compression_micro_compact_every_n_turns = max(
|
||||
1,
|
||||
_parse_prune_int(_compression_cfg.get("micro_compact_every_n_turns", 1), 1),
|
||||
)
|
||||
# Rolling-summary defrag threshold, in tokens. Lived on the compressor as
|
||||
# a hardcoded attribute with no path from config until now.
|
||||
compression_micro_compact_defrag_tokens = max(
|
||||
1,
|
||||
_parse_prune_int(
|
||||
_compression_cfg.get("micro_compact_defrag_threshold_tokens", 2000),
|
||||
2000,
|
||||
),
|
||||
)
|
||||
codex_app_server_auto_compaction = str(
|
||||
_compression_cfg.get("codex_app_server_auto", "native") or "native"
|
||||
@@ -1677,6 +1998,12 @@ def init_agent(
|
||||
codex_app_server_auto_compaction,
|
||||
)
|
||||
codex_app_server_auto_compaction = "native"
|
||||
# Opt-in idle compaction: compact a session up front when it resumes after
|
||||
# this many seconds of inactivity (0 = disabled). Time-based, so it
|
||||
# complements the size-based threshold above. Consumed by build_turn_context().
|
||||
compression_idle_compact_after_seconds = max(
|
||||
0, int(_compression_cfg.get("idle_compact_after_seconds", 0))
|
||||
)
|
||||
|
||||
# Read optional explicit context_length override for the auxiliary
|
||||
# compression model. Custom endpoints often cannot report this via
|
||||
@@ -1747,8 +2074,9 @@ def init_agent(
|
||||
)
|
||||
_config_context_length = None
|
||||
|
||||
# Resolve custom_providers list once for reuse below (startup
|
||||
# context-length override and plugin context-engine init).
|
||||
# Resolve custom_providers once before route-scoping a global context pin:
|
||||
# a named custom provider may keep its base URL only in this list rather
|
||||
# than repeating it under ``model``.
|
||||
try:
|
||||
from hermes_cli.config import get_compatible_custom_providers
|
||||
_custom_providers = get_compatible_custom_providers(_agent_cfg)
|
||||
@@ -1757,6 +2085,163 @@ def init_agent(
|
||||
if not isinstance(_custom_providers, list):
|
||||
_custom_providers = []
|
||||
|
||||
# ``model.context_length`` describes the configured default model. A
|
||||
# process launched directly with ``--model`` / ``-m`` has already replaced
|
||||
# ``agent.model`` before this initializer loads config, so carrying the
|
||||
# default model's explicit window into that different runtime is stale. The
|
||||
# live switch/fallback paths already clear this override; keep direct-start
|
||||
# overrides consistent with them and let provider metadata resolve the
|
||||
# active model's window instead.
|
||||
if _config_context_length is not None and isinstance(_model_cfg, dict):
|
||||
_configured_default_model = str(_model_cfg.get("default") or "").strip()
|
||||
_configured_default_runtime_model = _configured_default_model
|
||||
_active_runtime_model = agent.model
|
||||
if _configured_default_model:
|
||||
try:
|
||||
from hermes_cli.model_normalize import normalize_model_for_provider
|
||||
|
||||
_configured_default_runtime_model = normalize_model_for_provider(
|
||||
_configured_default_model, agent.provider
|
||||
)
|
||||
_active_runtime_model = normalize_model_for_provider(
|
||||
agent.model, agent.provider
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
_configured_provider = str(_model_cfg.get("provider") or "").strip()
|
||||
_configured_base_url = _normalize_route_base_url(
|
||||
_model_cfg.get("base_url")
|
||||
)
|
||||
_configured_provider_norm = _normalize_custom_provider_name(
|
||||
_configured_provider
|
||||
)
|
||||
_custom_provider_candidate = bool(_configured_provider_norm)
|
||||
_runtime_first_provider_ids = {
|
||||
"auto",
|
||||
"moa",
|
||||
"vertex",
|
||||
"google-vertex",
|
||||
"vertex-ai",
|
||||
"gcp-vertex",
|
||||
"vertexai",
|
||||
}
|
||||
if _configured_provider_norm in _runtime_first_provider_ids:
|
||||
_custom_provider_candidate = False
|
||||
elif (
|
||||
_custom_provider_candidate
|
||||
and _configured_provider_norm != "custom"
|
||||
and not _configured_provider_norm.startswith("custom:")
|
||||
):
|
||||
try:
|
||||
from hermes_cli.auth import resolve_provider as resolve_auth_provider
|
||||
|
||||
_resolved_auth_provider = resolve_auth_provider(
|
||||
_configured_provider_norm
|
||||
)
|
||||
_custom_provider_candidate = (
|
||||
str(_resolved_auth_provider or "").strip().lower()
|
||||
!= _configured_provider_norm
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
if not _configured_base_url and _custom_provider_candidate:
|
||||
_configured_custom_provider = _normalize_custom_provider_name(
|
||||
_configured_provider
|
||||
)
|
||||
_user_providers = _agent_cfg.get("providers")
|
||||
_disabled_custom_provider_ids: set[str] = set()
|
||||
if isinstance(_user_providers, dict):
|
||||
from hermes_cli.config import is_provider_enabled
|
||||
|
||||
for _provider_key, _provider_entry in _user_providers.items():
|
||||
if not isinstance(_provider_entry, dict):
|
||||
continue
|
||||
_entry_name = str(
|
||||
_provider_entry.get("name") or ""
|
||||
).strip()
|
||||
_entry_provider_ids = _custom_provider_runtime_ids(
|
||||
_provider_key
|
||||
) | _custom_provider_runtime_ids(_entry_name)
|
||||
if not is_provider_enabled(_provider_entry):
|
||||
_disabled_custom_provider_ids.update(
|
||||
provider_id
|
||||
for provider_id in _entry_provider_ids
|
||||
if provider_id
|
||||
)
|
||||
continue
|
||||
if _configured_custom_provider not in _entry_provider_ids:
|
||||
continue
|
||||
_configured_base_url = _normalize_route_base_url(
|
||||
_provider_entry.get("api")
|
||||
or _provider_entry.get("url")
|
||||
or _provider_entry.get("base_url")
|
||||
)
|
||||
if _configured_base_url:
|
||||
break
|
||||
if not _configured_base_url:
|
||||
for _provider_entry in _custom_providers:
|
||||
if not isinstance(_provider_entry, dict):
|
||||
continue
|
||||
_entry_name = str(
|
||||
_provider_entry.get("name") or ""
|
||||
).strip()
|
||||
_entry_provider_key = str(
|
||||
_provider_entry.get("provider_key") or ""
|
||||
).strip().lower()
|
||||
_entry_provider_ids = _custom_provider_runtime_ids(
|
||||
_entry_name
|
||||
) | _custom_provider_runtime_ids(_entry_provider_key)
|
||||
if (
|
||||
_entry_provider_key
|
||||
and _custom_provider_runtime_ids(_entry_provider_key)
|
||||
& _disabled_custom_provider_ids
|
||||
):
|
||||
continue
|
||||
if _configured_custom_provider not in _entry_provider_ids:
|
||||
continue
|
||||
_configured_base_url = _normalize_route_base_url(
|
||||
_provider_entry.get("base_url")
|
||||
)
|
||||
if _configured_base_url:
|
||||
break
|
||||
_active_route_url = str(agent.base_url or "")
|
||||
_requested_route_url = str(base_url or "")
|
||||
if "?" in _requested_route_url.split("#", 1)[0]:
|
||||
try:
|
||||
_requested_parts = urlparse(_requested_route_url)
|
||||
_requested_without_query = urlunparse(
|
||||
_requested_parts._replace(query="")
|
||||
)
|
||||
if _normalize_route_base_url(
|
||||
_requested_without_query
|
||||
) == _normalize_route_base_url(_active_route_url):
|
||||
_active_route_url = _requested_route_url
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
_active_base_url = _normalize_route_base_url(_active_route_url)
|
||||
_route_mismatch = _context_route_mismatch(
|
||||
_configured_base_url,
|
||||
_active_base_url,
|
||||
_configured_provider,
|
||||
agent.provider,
|
||||
already_normalized=True,
|
||||
)
|
||||
_model_mismatch = bool(
|
||||
_configured_default_runtime_model
|
||||
and _configured_default_runtime_model != _active_runtime_model
|
||||
)
|
||||
if _model_mismatch or _route_mismatch:
|
||||
_ra().logger.debug(
|
||||
"Ignoring model.context_length=%s for startup runtime %s at %s "
|
||||
"(configured default is %s at %s)",
|
||||
_config_context_length,
|
||||
agent.model,
|
||||
_active_base_url or agent.provider,
|
||||
_configured_default_model,
|
||||
_configured_base_url or _model_cfg.get("provider"),
|
||||
)
|
||||
_config_context_length = None
|
||||
|
||||
# Store for reuse by _check_compression_model_feasibility (auxiliary
|
||||
# compression model context-length detection needs the same list).
|
||||
agent._custom_providers = _custom_providers
|
||||
@@ -1779,11 +2264,11 @@ def init_agent(
|
||||
# Surface a clear warning if the user set a context_length but it
|
||||
# wasn't a valid positive int — the helper silently skips those.
|
||||
if _config_context_length is None:
|
||||
_target = agent.base_url.rstrip("/") if agent.base_url else ""
|
||||
_target = _normalize_route_base_url(agent.base_url)
|
||||
for _cp_entry in _custom_providers:
|
||||
if not isinstance(_cp_entry, dict):
|
||||
continue
|
||||
_cp_url = (_cp_entry.get("base_url") or "").rstrip("/")
|
||||
_cp_url = _normalize_route_base_url(_cp_entry.get("base_url"))
|
||||
if _target and _cp_url == _target:
|
||||
_cp_models = _cp_entry.get("models", {})
|
||||
if isinstance(_cp_models, dict):
|
||||
@@ -1815,7 +2300,18 @@ def init_agent(
|
||||
# AFTER the custom_providers branch so per-model overrides aren't lost.
|
||||
agent._config_context_length = _config_context_length
|
||||
|
||||
agent._ensure_lmstudio_runtime_loaded(_config_context_length)
|
||||
_lmstudio_runtime_context_length = agent._ensure_lmstudio_runtime_loaded(
|
||||
_config_context_length
|
||||
)
|
||||
if agent._lmstudio_load_was_unverified(_lmstudio_runtime_context_length):
|
||||
_ra().logger.warning(
|
||||
"LM Studio model activation was rejected or completed without a "
|
||||
"verifiable active context length; falling back to configured context"
|
||||
)
|
||||
_effective_context_length = agent._effective_lmstudio_context_length(
|
||||
_config_context_length,
|
||||
_lmstudio_runtime_context_length,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -1892,10 +2388,20 @@ def init_agent(
|
||||
agent.model,
|
||||
base_url=agent.base_url,
|
||||
api_key=getattr(agent, "api_key", ""),
|
||||
config_context_length=_config_context_length,
|
||||
config_context_length=_effective_context_length,
|
||||
provider=agent.provider,
|
||||
custom_providers=_custom_providers,
|
||||
)
|
||||
# Per-model threshold overrides are part of the explicit
|
||||
# context-engine contract: assign them BEFORE the initial
|
||||
# update_model() call so the first resolution (which derives
|
||||
# threshold_percent/threshold_tokens for the initial model) already
|
||||
# sees the overrides. Assigning after update_model() left the initial
|
||||
# model on the engine's global threshold until the first /model
|
||||
# switch. Engines that override update_model() own their own policy
|
||||
# and may ignore the attribute.
|
||||
if compression_model_thresholds:
|
||||
agent.context_compressor.model_thresholds = compression_model_thresholds
|
||||
agent.context_compressor.update_model(
|
||||
model=agent.model,
|
||||
context_length=_plugin_ctx_len,
|
||||
@@ -1917,11 +2423,17 @@ def init_agent(
|
||||
quiet_mode=agent.quiet_mode,
|
||||
base_url=agent.base_url,
|
||||
api_key=getattr(agent, "api_key", ""),
|
||||
config_context_length=_config_context_length,
|
||||
config_context_length=_effective_context_length,
|
||||
provider=agent.provider,
|
||||
api_mode=agent.api_mode,
|
||||
abort_on_summary_failure=compression_abort_on_summary_failure,
|
||||
max_tokens=agent.max_tokens,
|
||||
model_thresholds=compression_model_thresholds,
|
||||
threshold_tokens_cap=compression_threshold_tokens,
|
||||
proactive_prune_tokens=compression_proactive_prune_tokens,
|
||||
proactive_prune_min_result_chars=compression_proactive_prune_min_chars,
|
||||
proactive_prune_min_reclaim_tokens=compression_proactive_prune_min_reclaim,
|
||||
min_tail_user_messages=compression_min_tail_users,
|
||||
)
|
||||
_bind_session_state = getattr(agent.context_compressor, "bind_session_state", None)
|
||||
if callable(_bind_session_state):
|
||||
@@ -1931,12 +2443,32 @@ def init_agent(
|
||||
pass
|
||||
agent.compression_enabled = compression_enabled
|
||||
agent.compression_in_place = compression_in_place
|
||||
# Apply micro-compaction settings to the compressor (feature is opt-in)
|
||||
_cc = getattr(agent, "context_compressor", None)
|
||||
if _cc is not None and hasattr(_cc, "_micro_compact_enabled"):
|
||||
_cc._micro_compact_enabled = compression_micro_compact
|
||||
if _cc is not None and hasattr(_cc, "_micro_compact_every_n_turns"):
|
||||
_cc._micro_compact_every_n_turns = compression_micro_compact_every_n_turns
|
||||
if _cc is not None and hasattr(_cc, "_micro_compact_defrag_threshold_tokens"):
|
||||
_cc._micro_compact_defrag_threshold_tokens = (
|
||||
compression_micro_compact_defrag_tokens
|
||||
)
|
||||
agent.codex_app_server_auto_compaction = codex_app_server_auto_compaction
|
||||
agent.max_compression_attempts = compression_max_attempts
|
||||
agent.compression_idle_compact_after_seconds = (
|
||||
compression_idle_compact_after_seconds
|
||||
)
|
||||
|
||||
# Reject models whose context window is below the minimum required
|
||||
# for reliable tool-calling workflows (64K tokens).
|
||||
_ctx = getattr(agent.context_compressor, "context_length", 0)
|
||||
if _ctx and _ctx < MINIMUM_CONTEXT_LENGTH:
|
||||
_allow_lmstudio_explicit_below_floor = (
|
||||
str(getattr(agent, "provider", "") or "").strip().lower() == "lmstudio"
|
||||
and isinstance(agent._config_context_length, int)
|
||||
and not isinstance(agent._config_context_length, bool)
|
||||
and agent._config_context_length > 0
|
||||
)
|
||||
if _ctx and _ctx < MINIMUM_CONTEXT_LENGTH and not _allow_lmstudio_explicit_below_floor:
|
||||
raise ValueError(
|
||||
f"Model {agent.model} has a context window of {_ctx:,} tokens, "
|
||||
f"which is below the minimum {MINIMUM_CONTEXT_LENGTH:,} required "
|
||||
@@ -2116,7 +2648,7 @@ def init_agent(
|
||||
# autoraised model) updates the marker state and re-notifies once. The
|
||||
# config display gate (compression.codex_gpt55_autoraise_notice) still
|
||||
# suppresses the banner entirely without disabling the threshold autoraise.
|
||||
_autoraise = getattr(agent, "_compression_threshold_autoraised", None)
|
||||
_autoraise = getattr(agent, "_compression_threshold_autoraised", None) or {}
|
||||
_show_autoraise_notice = (
|
||||
bool(_autoraise)
|
||||
and compression_enabled
|
||||
@@ -2132,14 +2664,21 @@ def init_agent(
|
||||
_active_threshold_pct = getattr(
|
||||
agent.context_compressor, "threshold_percent", compression_threshold
|
||||
)
|
||||
print(f"📊 Context limit: {agent.context_compressor.context_length:,} tokens (compress at {int(_active_threshold_pct*100)}% = {agent.context_compressor.threshold_tokens:,})")
|
||||
_cap_note = ""
|
||||
_cap = getattr(agent.context_compressor, "threshold_tokens_cap", None)
|
||||
if _cap and _cap > 0:
|
||||
_cap_note = f" (capped at {_cap:,} tokens)"
|
||||
print(f"📊 Context limit: {agent.context_compressor.context_length:,} tokens (compress at {int(_active_threshold_pct*100)}% = {agent.context_compressor.threshold_tokens:,}{_cap_note})")
|
||||
else:
|
||||
print(f"📊 Context limit: {agent.context_compressor.context_length:,} tokens (auto-compression disabled)")
|
||||
# Notice with the exact opt-back-out command. Printed inline at startup
|
||||
# for CLI users; gateway users get the same text replayed via
|
||||
# _compression_warning on turn 1 (set below).
|
||||
if _show_autoraise_notice:
|
||||
print(_build_codex_gpt5_autoraise_notice(_autoraise))
|
||||
print(_build_codex_gpt5_autoraise_notice(
|
||||
_autoraise,
|
||||
context_length=getattr(agent.context_compressor, "context_length", None),
|
||||
))
|
||||
|
||||
# Check immediately so CLI users see the warning at startup.
|
||||
# Gateway status_callback is not yet wired, so any warning is stored
|
||||
@@ -2149,7 +2688,10 @@ def init_agent(
|
||||
# above only reaches the CLI, so stash the same text here to be replayed
|
||||
# through status_callback on the first turn (Telegram/Discord/Slack/etc.).
|
||||
if _show_autoraise_notice:
|
||||
agent._compression_warning = _build_codex_gpt5_autoraise_notice(_autoraise)
|
||||
agent._compression_warning = _build_codex_gpt5_autoraise_notice(
|
||||
_autoraise,
|
||||
context_length=getattr(agent.context_compressor, "context_length", None),
|
||||
)
|
||||
|
||||
# Mark shown so repeated inits in this profile (e.g. every gateway message)
|
||||
# stay silent. Recorded once, whether the notice went to the CLI print or
|
||||
@@ -2172,6 +2714,7 @@ def init_agent(
|
||||
agent._primary_runtime = {
|
||||
"model": agent.model,
|
||||
"provider": agent.provider,
|
||||
"requested_provider": agent.requested_provider,
|
||||
"base_url": agent.base_url,
|
||||
"api_mode": agent.api_mode,
|
||||
"api_key": getattr(agent, "api_key", ""),
|
||||
|
||||
+613
-215
File diff suppressed because it is too large
Load Diff
+297
-43
@@ -23,7 +23,7 @@ from urllib.parse import urlparse
|
||||
|
||||
from hermes_constants import get_hermes_home
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
from utils import base_url_host_matches, normalize_proxy_env_vars
|
||||
from utils import base_url_host_matches, base_url_hostname, normalize_proxy_env_vars
|
||||
|
||||
# NOTE: `import anthropic` is deliberately NOT at module top — the SDK pulls
|
||||
# ~220 ms of imports (anthropic.types, anthropic.lib.tools._beta_runner, etc.)
|
||||
@@ -127,6 +127,8 @@ _FAST_MODE_SUPPORTED_SUBSTRINGS = ("opus-4-6", "opus-4.6")
|
||||
_ANTHROPIC_OUTPUT_LIMITS = {
|
||||
# Mythos-class named models (claude-fable-5, …) — 1M context, reasoning
|
||||
"claude-fable": 128_000,
|
||||
# Claude Sonnet 5
|
||||
"claude-sonnet-5": 128_000,
|
||||
# Claude 4.8
|
||||
"claude-opus-4-8": 128_000,
|
||||
# Claude 4.7
|
||||
@@ -247,7 +249,13 @@ def _supports_adaptive_thinking(model: str) -> bool:
|
||||
only returns False for the explicit legacy list of older Claude families
|
||||
that require manual budget-based thinking. Non-Claude Anthropic-Messages
|
||||
models (minimax, qwen3, …) return False so they keep the manual path.
|
||||
|
||||
Kimi / Moonshot models are the exception: their Anthropic-compatible
|
||||
endpoints implement the adaptive contract (``thinking.type="adaptive"``
|
||||
+ ``output_config.effort``, including ``xhigh`` and ``display``).
|
||||
"""
|
||||
if _model_name_is_kimi_family(model):
|
||||
return True
|
||||
if not _is_claude_model(model):
|
||||
return False
|
||||
m = model.lower()
|
||||
@@ -360,7 +368,7 @@ def _detect_claude_code_version() -> str:
|
||||
try:
|
||||
result = _sp.run(
|
||||
[cmd, "--version"],
|
||||
capture_output=True, text=True, timeout=5,
|
||||
capture_output=True, text=True, encoding='utf-8', errors='replace', timeout=5,
|
||||
)
|
||||
if result.returncode == 0 and result.stdout.strip():
|
||||
# Output is like "2.1.74 (Claude Code)" or just "2.1.74"
|
||||
@@ -449,7 +457,8 @@ def _is_kimi_coding_endpoint(base_url: str | None) -> bool:
|
||||
|
||||
# Model-name prefixes that identify the Kimi / Moonshot family. Covers
|
||||
# - official slugs: ``kimi-k2.5``, ``kimi_thinking``, ``moonshot-v1-8k``
|
||||
# - common release lines: ``k1.5-...``, ``k2-thinking``, ``k25-...``, ``k2.5-...``
|
||||
# - common release lines: ``k1.5-...``, ``k2-thinking``, ``k25-...``, ``k2.5-...``,
|
||||
# and the bare Coding Plan slug ``k3`` (plus ``k3.x``/``k3-...`` variants)
|
||||
# Matched case-insensitively against the post-``normalize_model_name`` form,
|
||||
# so a caller's ``provider/vendor/model`` slug is handled the same as a
|
||||
# bare name.
|
||||
@@ -459,8 +468,14 @@ _KIMI_FAMILY_MODEL_PREFIXES = (
|
||||
"k1.", "k1-",
|
||||
"k2.", "k2-",
|
||||
"k25", "k2.5",
|
||||
"k3.", "k3-",
|
||||
)
|
||||
|
||||
# Bare release slugs with no separator suffix (Kimi Coding Plan serves K3
|
||||
# as the exact slug ``k3``). Kept exact-match so unrelated model names that
|
||||
# merely start with the same characters don't get misclassified.
|
||||
_KIMI_FAMILY_EXACT_SLUGS = frozenset({"k3"})
|
||||
|
||||
|
||||
def _model_name_is_kimi_family(model: str | None) -> bool:
|
||||
if not isinstance(model, str):
|
||||
@@ -471,6 +486,8 @@ def _model_name_is_kimi_family(model: str | None) -> bool:
|
||||
# Strip vendor prefix (e.g. ``moonshotai/kimi-k2.5`` → ``kimi-k2.5``)
|
||||
if "/" in m:
|
||||
m = m.rsplit("/", 1)[-1]
|
||||
if m in _KIMI_FAMILY_EXACT_SLUGS:
|
||||
return True
|
||||
return m.startswith(_KIMI_FAMILY_MODEL_PREFIXES)
|
||||
|
||||
|
||||
@@ -529,15 +546,49 @@ def _is_deepseek_anthropic_endpoint(base_url: str | None) -> bool:
|
||||
return "/anthropic" in normalized.rstrip("/").lower()
|
||||
|
||||
|
||||
def _is_nous_portal_endpoint(base_url: str | None) -> bool:
|
||||
"""Return True for Nous Portal's Anthropic Messages route.
|
||||
|
||||
Portal serves its ``anthropic/*`` catalog natively at
|
||||
``https://inference-api.nousresearch.com/v1/messages``. Portal-specific
|
||||
behaviours key off this: Bearer JWT auth, verbatim catalog model ids,
|
||||
and native thinking-signature replay.
|
||||
|
||||
Trusted hosts only:
|
||||
|
||||
1. Prod hostname ``inference-api.nousresearch.com``
|
||||
2. The operator-set ``NOUS_INFERENCE_BASE_URL`` hostname (staging/preview)
|
||||
|
||||
Lookalikes such as ``inference-api.nousresearch.com.attacker.test`` are
|
||||
rejected (hostname match, not substring).
|
||||
"""
|
||||
if base_url_host_matches(base_url or "", "inference-api.nousresearch.com"):
|
||||
return True
|
||||
try:
|
||||
from hermes_cli.auth import _nous_inference_env_override
|
||||
|
||||
override = _nous_inference_env_override()
|
||||
except Exception:
|
||||
return False
|
||||
if not override:
|
||||
return False
|
||||
# Exact host equality (not subdomain) so the env override can't broaden
|
||||
# into sibling hosts the operator did not set.
|
||||
override_host = base_url_hostname(override)
|
||||
return bool(override_host) and base_url_hostname(base_url or "") == override_host
|
||||
|
||||
|
||||
def _requires_bearer_auth(base_url: str | None) -> bool:
|
||||
"""Return True for Anthropic-compatible providers that require Bearer auth.
|
||||
|
||||
Some third-party /anthropic endpoints implement Anthropic's Messages API but
|
||||
require Authorization: Bearer instead of Anthropic's native x-api-key header.
|
||||
MiniMax's global and China Anthropic-compatible endpoints, Azure AI
|
||||
Foundry's Anthropic-style endpoint, and Palantir Foundry's LLM proxy
|
||||
follow this pattern.
|
||||
Foundry's Anthropic-style endpoint, Palantir Foundry's LLM proxy, and Nous
|
||||
Portal's Messages route follow this pattern.
|
||||
"""
|
||||
if _is_nous_portal_endpoint(base_url):
|
||||
return True
|
||||
normalized = _normalize_base_url_text(base_url)
|
||||
if not normalized:
|
||||
return False
|
||||
@@ -704,7 +755,11 @@ def _build_anthropic_client_with_bearer_hook(
|
||||
if common_betas:
|
||||
kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)}
|
||||
|
||||
return _anthropic_sdk.Anthropic(**kwargs)
|
||||
client = _anthropic_sdk.Anthropic(**kwargs)
|
||||
# Same env-inference trap as build_anthropic_client: auth_token-only
|
||||
# construction would otherwise also send ANTHROPIC_API_KEY as X-Api-Key.
|
||||
client.api_key = None
|
||||
return client
|
||||
|
||||
|
||||
def build_anthropic_client(
|
||||
@@ -833,7 +888,16 @@ def build_anthropic_client(
|
||||
if common_betas:
|
||||
kwargs["default_headers"] = {"anthropic-beta": ",".join(common_betas)}
|
||||
|
||||
return _anthropic_sdk.Anthropic(**kwargs)
|
||||
client = _anthropic_sdk.Anthropic(**kwargs)
|
||||
# Bearer-only construction leaves ``api_key`` unset, so the SDK fills it
|
||||
# from ``ANTHROPIC_API_KEY`` (Hermes loads that into the process env from
|
||||
# ``~/.hermes/.env``). The result is dual auth —
|
||||
# ``X-Api-Key: sk-ant-…`` *and* ``Authorization: Bearer <portal-jwt>`` —
|
||||
# on every Portal / MiniMax / OAuth Messages request. Clear the env-filled
|
||||
# key whenever we intentionally authenticated via auth_token alone.
|
||||
if "auth_token" in kwargs and "api_key" not in kwargs:
|
||||
client.api_key = None
|
||||
return client
|
||||
|
||||
|
||||
def build_anthropic_bedrock_client(region: str):
|
||||
@@ -897,7 +961,7 @@ def _read_claude_code_credentials_from_keychain() -> Optional[Dict[str, Any]]:
|
||||
"-s", "Claude Code-credentials",
|
||||
"-w"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
timeout=5,
|
||||
stdin=subprocess.DEVNULL,
|
||||
)
|
||||
@@ -1574,7 +1638,10 @@ def _is_bedrock_model_id(model: str) -> bool:
|
||||
"""
|
||||
lower = model.lower()
|
||||
# Regional inference-profile prefixes
|
||||
if any(lower.startswith(p) for p in ("global.", "us.", "eu.", "ap.", "jp.")):
|
||||
if any(lower.startswith(p) for p in (
|
||||
"global.", "us.", "eu.", "apac.", "ap.", "au.", "jp.",
|
||||
"ca.", "sa.", "me.", "af.",
|
||||
)):
|
||||
return True
|
||||
# Bare Bedrock model IDs: provider.model-family
|
||||
if lower.startswith("anthropic."):
|
||||
@@ -1861,6 +1928,28 @@ def _content_parts_to_anthropic_blocks(parts: Any) -> List[Dict[str, Any]]:
|
||||
return out
|
||||
|
||||
|
||||
_EMPTY_TEXT_PLACEHOLDER = "(empty)"
|
||||
|
||||
|
||||
def _safe_text(text: Any) -> str:
|
||||
"""Return ``text`` if it's non-whitespace, else a non-whitespace placeholder.
|
||||
|
||||
The Anthropic Messages API rejects requests where a text content block is
|
||||
empty or whitespace-only (HTTP 400 "text content blocks must contain
|
||||
non-whitespace text"). When such a block gets stored in session history —
|
||||
e.g. produced by context compression — it is replayed verbatim on every
|
||||
subsequent turn, permanently wedging the session. Coercing to a
|
||||
non-whitespace placeholder is self-healing: the next API call recovers.
|
||||
|
||||
Mirrors ``bedrock_adapter._safe_text`` (#9486); ref #69512.
|
||||
"""
|
||||
if text is None:
|
||||
return _EMPTY_TEXT_PLACEHOLDER
|
||||
if not isinstance(text, str):
|
||||
text = str(text)
|
||||
return text if text.strip() else _EMPTY_TEXT_PLACEHOLDER
|
||||
|
||||
|
||||
def _sanitize_replay_block(b: Dict[str, Any]) -> Optional[Dict[str, Any]]:
|
||||
"""Strip output-only fields from a stored Anthropic content block so it is
|
||||
valid as REQUEST input on replay.
|
||||
@@ -1878,7 +1967,18 @@ def _sanitize_replay_block(b: Dict[str, Any]) -> Optional[Dict[str, Any]]:
|
||||
return None
|
||||
btype = b.get("type")
|
||||
if btype == "text":
|
||||
out: Dict[str, Any] = {"type": "text", "text": b.get("text", "")}
|
||||
text_val = b.get("text", "")
|
||||
# Bedrock and strict Anthropic-compatible endpoints reject text
|
||||
# blocks where "text" is empty or whitespace-only (#69512). Drop the
|
||||
# blank block (the caller relocates any cache_control it carried and
|
||||
# falls back to a non-whitespace placeholder when nothing survives)
|
||||
# rather than coercing in place — a coerced "(empty)" block would be
|
||||
# model-visible noise next to surviving thinking/tool_use blocks.
|
||||
# Type-safe: captured blocks can carry text=None from an invalid
|
||||
# upstream payload, which a bare .strip() would crash on.
|
||||
if not isinstance(text_val, str) or not text_val.strip():
|
||||
return None
|
||||
out: Dict[str, Any] = {"type": "text", "text": text_val}
|
||||
# citations is input-valid ONLY when it's a non-empty list; the SDK
|
||||
# emits citations=None on responses, which the input schema rejects.
|
||||
cits = b.get("citations")
|
||||
@@ -1966,9 +2066,17 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
|
||||
parsed_args = {}
|
||||
redacted_input_by_id[_sanitize_tool_id(tc.get("id", ""))] = parsed_args
|
||||
replayed: List[Dict[str, Any]] = []
|
||||
_relocated_replay_cache_control = None
|
||||
_dropped_blank_text = False
|
||||
for b in ordered_blocks:
|
||||
clean = _sanitize_replay_block(b)
|
||||
if clean is None:
|
||||
if isinstance(b, dict) and b.get("type") == "text":
|
||||
_dropped_blank_text = True
|
||||
if isinstance(b, dict) and isinstance(b.get("cache_control"), dict):
|
||||
# A dropped blank text block can still carry the cache
|
||||
# breakpoint marker -- relocate it rather than losing it.
|
||||
_relocated_replay_cache_control = b["cache_control"]
|
||||
continue
|
||||
if clean.get("type") == "tool_use":
|
||||
# Override raw (un-redacted) input with the redacted copy when
|
||||
@@ -1978,20 +2086,90 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
|
||||
if redacted is not None:
|
||||
clean["input"] = redacted
|
||||
replayed.append(clean)
|
||||
# When every text block was blank and nothing cacheable survived
|
||||
# (e.g. signed thinking + a blank text block, or a SOLE blank
|
||||
# cache-marked block), emit the non-whitespace placeholder so the
|
||||
# replayed message stays schema-valid (#69512) and a relocated cache
|
||||
# marker still has a carrier instead of being silently lost.
|
||||
_has_cacheable_replay = any(
|
||||
isinstance(b, dict) and b.get("type") in {"text", "tool_use"}
|
||||
for b in replayed
|
||||
)
|
||||
if not _has_cacheable_replay and (
|
||||
_dropped_blank_text or _relocated_replay_cache_control is not None
|
||||
):
|
||||
replayed.append({"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER})
|
||||
if replayed:
|
||||
if _relocated_replay_cache_control is not None:
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
replayed, _relocated_replay_cache_control
|
||||
)
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
replayed, m.get("cache_control")
|
||||
)
|
||||
# apply_anthropic_cache_control marks an assistant turn with
|
||||
# non-empty text by writing cache_control INTO ``content`` (see
|
||||
# _apply_cache_marker's list branch), not at the top level. This
|
||||
# branch rebuilds the message from ordered_blocks and never reads
|
||||
# ``content``, so that marker would be dropped -- and because
|
||||
# _can_carry_marker already counted this message as a carrier, the
|
||||
# breakpoint is burned rather than relocated. #56195 covered the
|
||||
# complementary shape (blank content -> top-level marker); this is
|
||||
# the interleaved thinking + preamble-text + tool_use shape.
|
||||
_inline_cc = None
|
||||
_msg_content = m.get("content")
|
||||
if isinstance(_msg_content, list):
|
||||
for _blk in _msg_content:
|
||||
if isinstance(_blk, dict) and isinstance(
|
||||
_blk.get("cache_control"), dict
|
||||
):
|
||||
_inline_cc = _blk["cache_control"]
|
||||
break
|
||||
if _inline_cc is not None:
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
replayed, _inline_cc
|
||||
)
|
||||
return {"role": "assistant", "content": replayed}
|
||||
|
||||
blocks = _extract_preserved_thinking_blocks(m)
|
||||
# Cache markers dropped along with a blank block are relocated onto the
|
||||
# last surviving cacheable block below (via
|
||||
# _apply_assistant_cache_control_to_last_cacheable_block), rather than
|
||||
# lost -- prompt_caching.py's _apply_cache_marker() sets cache_control
|
||||
# directly on content[-1] for list content, so if that last part happens
|
||||
# to be blank text, dropping it silently would lose the breakpoint.
|
||||
_relocated_cache_control = None
|
||||
if content:
|
||||
if isinstance(content, list):
|
||||
converted_content = _convert_content_to_anthropic(content)
|
||||
if isinstance(converted_content, list):
|
||||
blocks.extend(converted_content)
|
||||
# Bedrock and strict Anthropic-compatible endpoints reject
|
||||
# text blocks where "text" is empty or whitespace-only. The
|
||||
# ordered-replay path enforces the same invariant via
|
||||
# _sanitize_replay_block(). Type-safe against ANY invalid
|
||||
# "text" value from an upstream payload -- None, or a
|
||||
# truthy non-string like an int -- not just None: checking
|
||||
# isinstance() first (rather than `blk.get("text") or ""`)
|
||||
# means a non-string value is treated as blank/invalid
|
||||
# instead of reaching .strip() and raising AttributeError.
|
||||
for blk in converted_content:
|
||||
_blk_text = blk.get("text") if isinstance(blk, dict) else None
|
||||
if (
|
||||
isinstance(blk, dict)
|
||||
and blk.get("type") == "text"
|
||||
and (not isinstance(_blk_text, str) or not _blk_text.strip())
|
||||
):
|
||||
if isinstance(blk.get("cache_control"), dict):
|
||||
_relocated_cache_control = blk["cache_control"]
|
||||
continue
|
||||
blocks.append(blk)
|
||||
else:
|
||||
blocks.append({"type": "text", "text": str(content)})
|
||||
# Scalar (non-list) content: a whitespace-only string is the
|
||||
# same invalid-payload case as an empty list block -- drop it
|
||||
# rather than emitting a blank text block.
|
||||
text_str = str(content)
|
||||
if text_str.strip():
|
||||
blocks.append({"type": "text", "text": text_str})
|
||||
for tc in m.get("tool_calls", []):
|
||||
if not tc or not isinstance(tc, dict):
|
||||
continue
|
||||
@@ -2007,9 +2185,6 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"name": fn.get("name", ""),
|
||||
"input": parsed_args,
|
||||
})
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
blocks, m.get("cache_control")
|
||||
)
|
||||
# Kimi's /coding endpoint (Anthropic protocol) requires assistant
|
||||
# tool-call messages to carry reasoning_content when thinking is
|
||||
# enabled server-side. Preserve it as a thinking block so Kimi
|
||||
@@ -2035,10 +2210,26 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]:
|
||||
)
|
||||
if isinstance(reasoning_content, str) and not _already_has_thinking:
|
||||
blocks.insert(0, {"type": "thinking", "thinking": reasoning_content})
|
||||
# Anthropic rejects empty assistant content
|
||||
effective = blocks or content
|
||||
if not effective or effective == "":
|
||||
effective = [{"type": "text", "text": "(empty)"}]
|
||||
# Anthropic rejects empty assistant content. IMPORTANT: fall back only
|
||||
# to the placeholder, never to the raw `content` variable -- `content`
|
||||
# is the UNFILTERED original message content, and can itself be exactly
|
||||
# the blank/whitespace-only payload the filtering above just removed
|
||||
# (a sole blank text block, or scalar whitespace with no tool_calls).
|
||||
# `blocks or content` there would silently restore the invalid provider
|
||||
# payload this function exists to prevent (#69512).
|
||||
effective = blocks if blocks else [{"type": "text", "text": _EMPTY_TEXT_PLACEHOLDER}]
|
||||
# Applied here (after the empty-fallback resolution) rather than
|
||||
# earlier against `blocks` directly, so a cache_control relocated from
|
||||
# a dropped blank block that was the ONLY block still lands on the
|
||||
# (empty) placeholder instead of being silently lost when blocks was
|
||||
# empty at the point the marker would otherwise have been applied.
|
||||
if _relocated_cache_control is not None:
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
effective, _relocated_cache_control
|
||||
)
|
||||
_apply_assistant_cache_control_to_last_cacheable_block(
|
||||
effective, m.get("cache_control")
|
||||
)
|
||||
return {"role": "assistant", "content": effective}
|
||||
|
||||
|
||||
@@ -2272,16 +2463,21 @@ def _manage_thinking_signatures(
|
||||
replayed assistant tool-call messages. See hermes-agent#13848 (Kimi) and
|
||||
hermes-agent#16748 (DeepSeek).
|
||||
|
||||
Nous Portal's ``/v1/messages`` route is the exception among third-party
|
||||
hosts: it proxies Claude to Anthropic/Vertex/Bedrock and validates the
|
||||
same signed thinking blocks. Sticky ``session_id`` keeps a conversation
|
||||
on one upstream instance so those signatures stay warm — stripping them
|
||||
here would 400 the first tool-loop turn ("thinking must be passed back").
|
||||
Portal therefore takes the native Anthropic replay path below.
|
||||
|
||||
Mutates ``result`` in place.
|
||||
"""
|
||||
_THINKING_TYPES = frozenset(("thinking", "redacted_thinking"))
|
||||
_is_third_party = _is_third_party_anthropic_endpoint(base_url)
|
||||
# Kimi / DeepSeek share a contract: strip signed Anthropic blocks
|
||||
# (neither upstream can validate Anthropic signatures), preserve unsigned
|
||||
# ones synthesised from reasoning_content. See #13848, #16748.
|
||||
_preserve_unsigned_thinking = (
|
||||
_is_kimi_family_endpoint(base_url, model)
|
||||
or _is_deepseek_anthropic_endpoint(base_url)
|
||||
# Portal speaks Anthropic's thinking contract end-to-end; do not treat it
|
||||
# as a signature-blind proxy even though the host is not anthropic.com.
|
||||
_is_third_party = (
|
||||
_is_third_party_anthropic_endpoint(base_url)
|
||||
and not _is_nous_portal_endpoint(base_url)
|
||||
)
|
||||
|
||||
last_assistant_idx = None
|
||||
@@ -2294,8 +2490,12 @@ def _manage_thinking_signatures(
|
||||
if m.get("role") != "assistant" or not isinstance(m.get("content"), list):
|
||||
continue
|
||||
|
||||
if _preserve_unsigned_thinking:
|
||||
# Kimi / DeepSeek: strip signed, preserve unsigned.
|
||||
if _is_kimi_family_endpoint(base_url, model):
|
||||
# Kimi does not enforce thinking signatures — replay as-is
|
||||
# (shared cleanup below still strips cache markers + the internal flag).
|
||||
pass
|
||||
elif _is_deepseek_anthropic_endpoint(base_url):
|
||||
# DeepSeek: strip signed, preserve unsigned.
|
||||
new_content = []
|
||||
for b in m["content"]:
|
||||
if not isinstance(b, dict) or b.get("type") not in _THINKING_TYPES:
|
||||
@@ -2395,6 +2595,24 @@ def _evict_old_screenshots(result: List[Dict[str, Any]]) -> None:
|
||||
]
|
||||
|
||||
|
||||
def _ensure_leading_user_turn(result: List[Dict[str, Any]]) -> None:
|
||||
"""Anthropic requires messages[0] to have role=user.
|
||||
|
||||
After a second context compaction on the auto path the summary can be
|
||||
emitted as role=assistant with nothing in front of it (the system prompt
|
||||
lives outside messages[] or is extracted into the separate ``system``
|
||||
param), so messages[0] ends up assistant and the Messages API rejects
|
||||
the request with HTTP 400 — often masked by a misleading
|
||||
"tool_use ids were found without tool_result blocks" error (#52160).
|
||||
|
||||
Mirror the Bedrock Converse adapter, which unconditionally prepends a
|
||||
minimal user turn when the first message is not user
|
||||
(convert_messages_to_converse).
|
||||
"""
|
||||
if result and result[0].get("role") != "user":
|
||||
result.insert(0, {"role": "user", "content": [{"type": "text", "text": " "}]})
|
||||
|
||||
|
||||
def convert_messages_to_anthropic(
|
||||
messages: List[Dict],
|
||||
base_url: str | None = None,
|
||||
@@ -2453,6 +2671,7 @@ def convert_messages_to_anthropic(
|
||||
|
||||
_strip_orphaned_tool_blocks(result)
|
||||
result = _merge_consecutive_roles(result)
|
||||
_ensure_leading_user_turn(result)
|
||||
_manage_thinking_signatures(result, base_url, model)
|
||||
_evict_old_screenshots(result)
|
||||
|
||||
@@ -2516,7 +2735,12 @@ def build_anthropic_kwargs(
|
||||
)
|
||||
anthropic_tools = convert_tools_to_anthropic(tools) if tools else []
|
||||
|
||||
model = normalize_model_name(model, preserve_dots=preserve_dots)
|
||||
# Nous Portal routes on its own catalog ids (``anthropic/claude-opus-4.8``);
|
||||
# normalizing to the bare Anthropic slug would make the model unresolvable
|
||||
# there. Skipping the call preserves the prefix AND the dots, so
|
||||
# ``preserve_dots`` stays irrelevant for Portal.
|
||||
if not _is_nous_portal_endpoint(base_url):
|
||||
model = normalize_model_name(model, preserve_dots=preserve_dots)
|
||||
# effective_max_tokens = output cap for this call (≠ total context window)
|
||||
# Use the resolver helper so non-positive values (negative ints,
|
||||
# fractional floats, NaN, non-numeric) fail locally with a clear error
|
||||
@@ -2627,25 +2851,19 @@ def build_anthropic_kwargs(
|
||||
# MiniMax Anthropic-compat endpoints support thinking (manual mode only,
|
||||
# not adaptive). Haiku does NOT support extended thinking — skip entirely.
|
||||
#
|
||||
# Kimi's /coding endpoint speaks the Anthropic Messages protocol but has
|
||||
# its own thinking semantics: when ``thinking.enabled`` is sent, Kimi
|
||||
# validates the message history and requires every prior assistant
|
||||
# tool-call message to carry OpenAI-style ``reasoning_content``. The
|
||||
# Anthropic path never populates that field, and
|
||||
# ``convert_messages_to_anthropic`` strips all Anthropic thinking blocks
|
||||
# on third-party endpoints — so the request fails with HTTP 400
|
||||
# "thinking is enabled but reasoning_content is missing in assistant
|
||||
# tool call message at index N". Kimi's reasoning is driven server-side
|
||||
# on the /coding route, so skip Anthropic's thinking parameter entirely
|
||||
# for that host. (Kimi on chat_completions enables thinking via
|
||||
# extra_body in the ChatCompletionsTransport — see #13503.)
|
||||
# Kimi / Moonshot models also use adaptive thinking: their
|
||||
# Anthropic-compatible endpoints (api.moonshot.cn/anthropic,
|
||||
# api.kimi.com/coding) accept ``thinking.type="adaptive"`` +
|
||||
# ``output_config.effort``, and the replay-validation 400s that
|
||||
# originally motivated dropping the parameter (#13848) no longer
|
||||
# occur. (Kimi on chat_completions enables thinking via extra_body
|
||||
# in the ChatCompletionsTransport — see #13503.)
|
||||
#
|
||||
# On 4.7+ the `thinking.display` field defaults to "omitted", which
|
||||
# silently hides reasoning text that Hermes surfaces in its CLI. We
|
||||
# request "summarized" so the reasoning blocks stay populated — matching
|
||||
# 4.6 behavior and preserving the activity-feed UX during long tool runs.
|
||||
_is_kimi_coding = _is_kimi_family_endpoint(base_url, model)
|
||||
if reasoning_config and isinstance(reasoning_config, dict) and not _is_kimi_coding:
|
||||
if reasoning_config and isinstance(reasoning_config, dict):
|
||||
if reasoning_config.get("enabled") is not False and "haiku" not in model.lower():
|
||||
effort = str(reasoning_config.get("effort", "medium")).lower()
|
||||
budget = THINKING_BUDGET.get(effort, 8000)
|
||||
@@ -2761,6 +2979,8 @@ def create_anthropic_message(
|
||||
*,
|
||||
log_prefix: str = "",
|
||||
prefer_stream: bool = True,
|
||||
on_stream_event=None,
|
||||
on_response=None,
|
||||
) -> Any:
|
||||
"""Create an Anthropic message, aggregating via stream when available.
|
||||
|
||||
@@ -2770,6 +2990,20 @@ def create_anthropic_message(
|
||||
crash on ``.content``. Prefer ``messages.stream().get_final_message()`` to
|
||||
match the main turn path, falling back to ``create()`` only for providers
|
||||
that explicitly do not support streaming, such as restricted Bedrock roles.
|
||||
|
||||
``on_stream_event``: optional callable invoked once per streamed event
|
||||
(best-effort, exceptions swallowed). Lets callers report forward progress
|
||||
to liveness watchdogs — e.g. the auxiliary compression path ticking its
|
||||
progress hook so a slow-but-generating summary model isn't treated as
|
||||
hung. Only fires on the streaming path; the ``create()`` fallback has no
|
||||
events to report.
|
||||
|
||||
``on_response``: optional callable invoked once with the underlying httpx
|
||||
response before the message is aggregated (best-effort, exceptions
|
||||
swallowed). Response *headers* carry out-of-band provider state that the
|
||||
parsed ``Message`` drops — Nous Portal's ``x-nous-credits-*`` balance family
|
||||
in particular. Only fires on the streaming path, which is the one the main
|
||||
turn loop takes.
|
||||
"""
|
||||
sanitize_anthropic_kwargs(api_kwargs, log_prefix=log_prefix)
|
||||
|
||||
@@ -2780,6 +3014,26 @@ def create_anthropic_message(
|
||||
stream_kwargs.pop("stream", None)
|
||||
try:
|
||||
with stream_fn(**stream_kwargs) as stream:
|
||||
if callable(on_response):
|
||||
try:
|
||||
on_response(getattr(stream, "response", None))
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"%son_response callback failed",
|
||||
log_prefix, exc_info=True,
|
||||
)
|
||||
if callable(on_stream_event):
|
||||
# Consume the event stream manually so each event can
|
||||
# tick the caller's progress callback; get_final_message
|
||||
# then returns the accumulated snapshot.
|
||||
for _event in stream:
|
||||
try:
|
||||
on_stream_event(_event)
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"%son_stream_event callback failed",
|
||||
log_prefix, exc_info=True,
|
||||
)
|
||||
return stream.get_final_message()
|
||||
except Exception as exc:
|
||||
if not _is_stream_unavailable_error(exc):
|
||||
|
||||
@@ -66,3 +66,19 @@ def safe_schedule_threadsafe(
|
||||
coro.close()
|
||||
log.log(log_level, "%s: %s", log_message, exc)
|
||||
return None
|
||||
|
||||
|
||||
def consume_detached_task_result(task: "asyncio.Future[Any]") -> None:
|
||||
"""Retrieve a detached task's result without surfacing cancellation.
|
||||
|
||||
Used as an ``add_done_callback`` on tasks that were cancelled and
|
||||
detached (e.g. an adapter close path that swallows ``CancelledError``
|
||||
past its teardown deadline). Observing ``task.exception()`` prevents
|
||||
"exception was never retrieved" noise on the event loop; cancellation
|
||||
and any terminal error are deliberately swallowed — the task's owner
|
||||
already gave up on it.
|
||||
"""
|
||||
try:
|
||||
task.exception()
|
||||
except (asyncio.CancelledError, Exception):
|
||||
pass
|
||||
|
||||
+1314
-119
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,204 @@
|
||||
"""Single owner for backend identity and failure-scoped skip decisions.
|
||||
|
||||
Every fallback / dedup / skip / quarantine decision in Hermes ultimately asks
|
||||
one question: **"is this candidate the same backend as the one that failed,
|
||||
along the axis that failure invalidated?"** Before this module, that
|
||||
question was re-implemented inline at six call sites across four subsystems,
|
||||
each comparing whatever string was locally convenient (provider label,
|
||||
provider+model, base_url+model, ...). Each incident fixed one site while the
|
||||
others kept the bug: #22548 (same-shim aliases), #70893 (xai-oauth vs xai —
|
||||
same host, distinct credential), #59561 (aux chain skipped sibling models),
|
||||
#72468 (aux main-model safety net, same bug three weeks later), #62984 /
|
||||
#54250 / #57584 (dedup ignoring base_url strands multi-endpoint pools).
|
||||
|
||||
The root insight: "provider" conflates three independent identity axes, and
|
||||
each failure class invalidates a different one:
|
||||
|
||||
* **credential surface** — auth 401 / payment 402 kill everything sharing the
|
||||
credential (every model, every host reached with that key/token).
|
||||
* **endpoint** — DNS failure / connection refused kill everything behind the
|
||||
URL, regardless of model or credential.
|
||||
* **model deployment** — timeout / overload / rate limit / model-incompatible
|
||||
kill ONE model's deployment. A sibling model behind the same URL is an
|
||||
independent deployment (real incident: aux ``glm-5.2`` hung and timed out
|
||||
while main ``macaron-v1-venti`` on the identical endpoint was serving
|
||||
448K-token turns).
|
||||
|
||||
Call sites should build :class:`BackendIdentity` values, classify the failure
|
||||
with :func:`classify_failure_scope`, and ask :func:`should_skip_candidate`.
|
||||
Do not re-implement any comparison inline — extend THIS module instead.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from typing import Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class FailureScope(Enum):
|
||||
"""Which identity axis a failure invalidates."""
|
||||
|
||||
#: Timeout, overload/429, connection blip, model-incompatible, invalid
|
||||
#: response: evidence against ONE model deployment only.
|
||||
MODEL = "model"
|
||||
#: Auth 401 / payment 402: evidence against the shared credential —
|
||||
#: every model reached with it is equally dead.
|
||||
CREDENTIAL = "credential"
|
||||
#: DNS / connection-refused / unreachable host: evidence against the
|
||||
#: endpoint — every model behind the URL is equally dead.
|
||||
ENDPOINT = "endpoint"
|
||||
|
||||
|
||||
#: Reason strings already used by auxiliary_client's except-chain, mapped to
|
||||
#: scopes. Unknown reasons default to MODEL — the least-invalidating scope —
|
||||
#: so an unrecognized failure never over-skips viable candidates.
|
||||
_REASON_SCOPES = {
|
||||
"auth error": FailureScope.CREDENTIAL,
|
||||
"payment error": FailureScope.CREDENTIAL,
|
||||
"rate limit": FailureScope.MODEL,
|
||||
"model incompatible with route": FailureScope.MODEL,
|
||||
"invalid provider response": FailureScope.MODEL,
|
||||
"connection error": FailureScope.MODEL,
|
||||
"timeout": FailureScope.MODEL,
|
||||
}
|
||||
|
||||
|
||||
def classify_failure_scope(reason: Optional[str]) -> FailureScope:
|
||||
"""Map a human-readable failure reason to the identity axis it kills."""
|
||||
return _REASON_SCOPES.get((reason or "").strip().lower(), FailureScope.MODEL)
|
||||
|
||||
|
||||
def _norm_provider(value: Optional[str]) -> str:
|
||||
return (value or "").strip().lower()
|
||||
|
||||
|
||||
def _norm_model(value: Optional[str]) -> str:
|
||||
return (value or "").strip().lower()
|
||||
|
||||
|
||||
def _norm_base_url(value: Optional[str]) -> str:
|
||||
return (value or "").strip().rstrip("/").lower()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BackendIdentity:
|
||||
"""Normalized identity of one (provider, model, endpoint) deployment.
|
||||
|
||||
Empty fields mean "unknown" — comparisons treat an unknown axis as
|
||||
non-distinguishing (it can neither prove sameness nor difference on its
|
||||
own; the remaining axes decide).
|
||||
"""
|
||||
|
||||
provider: str = ""
|
||||
model: str = ""
|
||||
base_url: str = ""
|
||||
|
||||
@classmethod
|
||||
def build(
|
||||
cls,
|
||||
provider: Optional[str] = None,
|
||||
model: Optional[str] = None,
|
||||
base_url: Optional[str] = None,
|
||||
) -> "BackendIdentity":
|
||||
return cls(
|
||||
provider=_norm_provider(provider),
|
||||
model=_norm_model(model),
|
||||
base_url=_norm_base_url(base_url),
|
||||
)
|
||||
|
||||
|
||||
def _both_first_class(a: BackendIdentity, b: BackendIdentity) -> bool:
|
||||
"""True when both providers are distinct registered first-class providers.
|
||||
|
||||
Two different registry providers have distinct credential surfaces even
|
||||
when they share an inference host (xai-oauth vs xai, openai-codex vs
|
||||
openai-api) — #70893. Custom/shim aliases are NOT in the registry, so
|
||||
two aliases pointing at one URL still count as the same backend (#22548).
|
||||
"""
|
||||
if not a.provider or not b.provider or a.provider == b.provider:
|
||||
return False
|
||||
try:
|
||||
from hermes_cli.auth import PROVIDER_REGISTRY
|
||||
|
||||
return a.provider in PROVIDER_REGISTRY and b.provider in PROVIDER_REGISTRY
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def same_credential_surface(a: BackendIdentity, b: BackendIdentity) -> bool:
|
||||
"""Do two identities share the credential a 401/402 just invalidated?
|
||||
|
||||
Conservative on purpose: an unprovable axis must answer "different"
|
||||
(try the candidate — worst case one wasted RTT) rather than "same"
|
||||
(skip — worst case stranded failover). Two distinct custom labels at
|
||||
one URL may carry different per-entry api_keys, so a shared URL alone
|
||||
never proves a shared credential; it is only used as a weak signal
|
||||
when a provider label is missing entirely.
|
||||
"""
|
||||
if a.provider and b.provider:
|
||||
# Same label = same configured credential. Different labels =
|
||||
# different credential config (first-class registry providers
|
||||
# explicitly so — #70893; custom entries can each carry their own
|
||||
# api_key, so sameness is unprovable and we must not skip).
|
||||
return a.provider == b.provider
|
||||
# Provider unknown on a side: same explicit URL is the best signal left.
|
||||
return bool(a.base_url and a.base_url == b.base_url)
|
||||
|
||||
|
||||
def same_endpoint(a: BackendIdentity, b: BackendIdentity) -> bool:
|
||||
"""Do two identities sit behind the endpoint that just went unreachable?"""
|
||||
if a.base_url and b.base_url:
|
||||
return a.base_url == b.base_url
|
||||
# An unknown base_url inherits the provider default → same provider
|
||||
# label implies the same default endpoint.
|
||||
return bool(a.provider and a.provider == b.provider)
|
||||
|
||||
|
||||
def same_deployment(a: BackendIdentity, b: BackendIdentity) -> bool:
|
||||
"""Are these the exact same model deployment (the thing a timeout kills)?
|
||||
|
||||
Provider+model must match; the base_url axis distinguishes only when BOTH
|
||||
sides carry an explicit URL (#62984: same provider+model on two different
|
||||
explicit URLs is two deployments — a pool). A side with an unknown URL
|
||||
inherits the provider default and cannot prove difference.
|
||||
"""
|
||||
if not (a.provider and b.provider and a.provider == b.provider):
|
||||
# Same-host different-label shims: same URL + same model IS the same
|
||||
# deployment even when the alias labels differ (#22548) — unless both
|
||||
# labels are first-class registry providers (#70893).
|
||||
if (
|
||||
a.base_url
|
||||
and a.base_url == b.base_url
|
||||
and a.model
|
||||
and a.model == b.model
|
||||
and not _both_first_class(a, b)
|
||||
):
|
||||
return True
|
||||
return False
|
||||
if not (a.model and b.model and a.model == b.model):
|
||||
return False
|
||||
if a.base_url and b.base_url and a.base_url != b.base_url:
|
||||
return False # distinct explicit endpoints — a pool, not a dup
|
||||
return True
|
||||
|
||||
|
||||
def should_skip_candidate(
|
||||
candidate: BackendIdentity,
|
||||
failed: BackendIdentity,
|
||||
scope: FailureScope = FailureScope.MODEL,
|
||||
) -> bool:
|
||||
"""THE skip predicate: would trying ``candidate`` just repeat the failure?
|
||||
|
||||
True when the candidate is the same backend as ``failed`` along the axis
|
||||
``scope`` says the failure invalidated. Every fallback/dedup/skip site
|
||||
must call this instead of comparing labels inline.
|
||||
"""
|
||||
if scope is FailureScope.CREDENTIAL:
|
||||
return same_credential_surface(candidate, failed)
|
||||
if scope is FailureScope.ENDPOINT:
|
||||
return same_endpoint(candidate, failed)
|
||||
return same_deployment(candidate, failed)
|
||||
+30
-12
@@ -70,8 +70,8 @@ def _resolve_review_runtime(agent: Any) -> Dict[str, Any]:
|
||||
"routed": False,
|
||||
}
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
cfg = load_config()
|
||||
from hermes_cli.config import load_config_readonly
|
||||
cfg = load_config_readonly()
|
||||
except Exception:
|
||||
return parent
|
||||
aux = cfg.get("auxiliary", {}) if isinstance(cfg.get("auxiliary"), dict) else {}
|
||||
@@ -209,7 +209,10 @@ _SKILL_REVIEW_PROMPT = (
|
||||
"conversation for skills the user loaded via /skill-name or you "
|
||||
"read via skill_view. If any of them covers the territory of the "
|
||||
"new learning, PATCH that one first. It is the skill that was in "
|
||||
"play, so it's the right one to extend.\n"
|
||||
"play, so it's the right one to extend — but only if it is "
|
||||
"curator-managed. Bundled, hub, pinned, and user-owned skills are "
|
||||
"off-limits to you no matter how relevant (see Protected skills "
|
||||
"below); for those, fall through to the next option.\n"
|
||||
" 2. UPDATE AN EXISTING UMBRELLA (via skills_list + skill_view). "
|
||||
"If no loaded skill fits but an existing class-level skill does, "
|
||||
"patch it. Add a subsection, a pitfall, or broaden a trigger.\n"
|
||||
@@ -251,10 +254,18 @@ _SKILL_REVIEW_PROMPT = (
|
||||
"Protected skills (DO NOT edit these):\n"
|
||||
" • Bundled skills (shipped with Hermes, e.g. 'hermes-agent').\n"
|
||||
" • Hub-installed skills (installed via 'hermes skills install').\n"
|
||||
"Pinned skills (marked via 'hermes curator pin') CAN be improved — "
|
||||
"pin only blocks deletion/archive/consolidation by the curator, not "
|
||||
"content updates. Patch them when a pitfall or missing step turns up, "
|
||||
"same as any other agent-created skill.\n"
|
||||
" • Skills in skills.external_dirs (externally owned).\n"
|
||||
" • PINNED skills (marked via 'hermes curator pin'). You are an "
|
||||
"autonomous no-user-present actor, so pin blocks your writes too — "
|
||||
"content updates included. Only the user, in a foreground session, "
|
||||
"can change a pinned skill.\n"
|
||||
" • USER-OWNED skills — anything not curator-managed. A skill the "
|
||||
"user hand-wrote, installed by URL, or asked a foreground agent to "
|
||||
"create is theirs, not yours; your writes to it WILL be refused. "
|
||||
"This includes skills that were loaded or consulted this session: "
|
||||
"being in play does not make one yours to edit. If such a skill is "
|
||||
"wrong or outdated, say so in your reply and recommend "
|
||||
"'hermes curator adopt <name>' — do not try to patch it.\n"
|
||||
"If the only skills that need updating are protected, say\n"
|
||||
"'Nothing to save.' and stop.\n\n"
|
||||
"Do NOT capture (these become persistent self-imposed constraints "
|
||||
@@ -309,7 +320,9 @@ _COMBINED_REVIEW_PROMPT = (
|
||||
" 1. UPDATE A CURRENTLY-LOADED SKILL. Check what skills were "
|
||||
"loaded via /skill-name or skill_view in the conversation. If one "
|
||||
"of them covers the learning, PATCH it first. It was in play; "
|
||||
"it's the right place.\n"
|
||||
"it's the right place — provided it is curator-managed. Protected "
|
||||
"and user-owned skills are off-limits however relevant; fall "
|
||||
"through when one of those is the best fit.\n"
|
||||
" 2. UPDATE AN EXISTING UMBRELLA (skills_list + skill_view to "
|
||||
"find the right one). Patch it.\n"
|
||||
" 3. ADD A SUPPORT FILE under an existing umbrella via "
|
||||
@@ -337,10 +350,15 @@ _COMBINED_REVIEW_PROMPT = (
|
||||
"Protected skills (DO NOT edit these):\n"
|
||||
" • Bundled skills (shipped with Hermes, e.g. 'hermes-agent').\n"
|
||||
" • Hub-installed skills (installed via 'hermes skills install').\n"
|
||||
"Pinned skills (marked via 'hermes curator pin') CAN be improved — "
|
||||
"pin only blocks deletion/archive/consolidation by the curator, not "
|
||||
"content updates. Patch them when a pitfall or missing step turns up, "
|
||||
"same as any other agent-created skill.\n"
|
||||
" • Skills in skills.external_dirs (externally owned).\n"
|
||||
" • PINNED skills (marked via 'hermes curator pin'). Pin blocks "
|
||||
"autonomous writes entirely — content updates included — because no "
|
||||
"user is present to consent. Only a foreground session can change one.\n"
|
||||
" • USER-OWNED skills — anything not curator-managed (hand-written, "
|
||||
"URL-installed, or created by a foreground agent at the user's "
|
||||
"request). Your writes to these WILL be refused, including to skills "
|
||||
"loaded or consulted this session. If one is wrong, say so in your "
|
||||
"reply and recommend 'hermes curator adopt <name>' instead.\n"
|
||||
"If the only skills that need updating are protected, say\n"
|
||||
"'Nothing to save.' and stop.\n\n"
|
||||
"Do NOT capture as skills (these become persistent self-imposed "
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
"""System-battery read-out for the CLI/TUI status bar.
|
||||
|
||||
Reads the host battery through ``psutil`` (already a Hermes dependency) and
|
||||
exposes a compact, colour-coded label. Everything degrades to "unavailable"
|
||||
when there is no battery (desktops, servers, VMs) or when the read fails, so
|
||||
callers can render the result unconditionally and simply show nothing.
|
||||
|
||||
The status bar repaints often (every keystroke and on a ~1s idle refresh), so
|
||||
:func:`read_battery` memoises the last reading for a few seconds instead of
|
||||
hitting ``psutil`` on every frame.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BatteryStatus:
|
||||
"""A single battery reading.
|
||||
|
||||
``available`` is False on machines without a battery (or when the read
|
||||
failed). ``percent`` is clamped to 0-100. ``plugged`` is True when on AC
|
||||
power, False on battery, and None when the platform can't tell.
|
||||
"""
|
||||
|
||||
available: bool
|
||||
percent: Optional[int] = None
|
||||
plugged: Optional[bool] = None
|
||||
|
||||
@property
|
||||
def charging(self) -> bool:
|
||||
return bool(self.plugged)
|
||||
|
||||
|
||||
UNAVAILABLE = BatteryStatus(available=False)
|
||||
|
||||
# Colour buckets, mirroring the status-bar context styles but inverted (a full
|
||||
# battery is "good", an empty one is "critical").
|
||||
CATEGORY_GOOD = "good"
|
||||
CATEGORY_WARN = "warn"
|
||||
CATEGORY_BAD = "bad"
|
||||
CATEGORY_CRITICAL = "critical"
|
||||
CATEGORY_DIM = "dim"
|
||||
|
||||
_CACHE_TTL_SECONDS = 8.0
|
||||
_cache: Optional[tuple[float, BatteryStatus]] = None
|
||||
|
||||
|
||||
def _read_battery_uncached() -> BatteryStatus:
|
||||
try:
|
||||
import psutil
|
||||
except Exception:
|
||||
return UNAVAILABLE
|
||||
|
||||
# ``sensors_battery`` is missing on some platforms/builds of psutil.
|
||||
reader = getattr(psutil, "sensors_battery", None)
|
||||
if reader is None:
|
||||
return UNAVAILABLE
|
||||
|
||||
try:
|
||||
batt = reader()
|
||||
except Exception:
|
||||
return UNAVAILABLE
|
||||
|
||||
if batt is None:
|
||||
return UNAVAILABLE
|
||||
|
||||
percent: Optional[int] = None
|
||||
raw_percent = getattr(batt, "percent", None)
|
||||
if raw_percent is not None:
|
||||
try:
|
||||
percent = max(0, min(100, int(round(float(raw_percent)))))
|
||||
except (TypeError, ValueError):
|
||||
percent = None
|
||||
|
||||
plugged = getattr(batt, "power_plugged", None)
|
||||
if plugged is not None:
|
||||
plugged = bool(plugged)
|
||||
|
||||
return BatteryStatus(available=True, percent=percent, plugged=plugged)
|
||||
|
||||
|
||||
def read_battery(use_cache: bool = True) -> BatteryStatus:
|
||||
"""Return the current battery status (cached for a few seconds)."""
|
||||
global _cache
|
||||
if use_cache and _cache is not None:
|
||||
ts, cached = _cache
|
||||
if time.monotonic() - ts < _CACHE_TTL_SECONDS:
|
||||
return cached
|
||||
|
||||
status = _read_battery_uncached()
|
||||
_cache = (time.monotonic(), status)
|
||||
return status
|
||||
|
||||
|
||||
def clear_cache() -> None:
|
||||
"""Drop the memoised reading (used by tests)."""
|
||||
global _cache
|
||||
_cache = None
|
||||
|
||||
|
||||
def battery_category(status: BatteryStatus) -> str:
|
||||
"""Bucket a reading into a colour category: good/warn/bad/critical/dim."""
|
||||
if not status.available or status.percent is None:
|
||||
return CATEGORY_DIM
|
||||
# On AC power the level isn't a concern — always read as healthy.
|
||||
if status.charging:
|
||||
return CATEGORY_GOOD
|
||||
pct = status.percent
|
||||
if pct <= 10:
|
||||
return CATEGORY_CRITICAL
|
||||
if pct <= 20:
|
||||
return CATEGORY_BAD
|
||||
if pct <= 50:
|
||||
return CATEGORY_WARN
|
||||
return CATEGORY_GOOD
|
||||
|
||||
|
||||
def battery_glyph(status: BatteryStatus) -> str:
|
||||
"""Return the leading glyph: a bolt while charging, else a battery."""
|
||||
return "\u26a1" if status.charging else "\U0001f50b" # ⚡ / 🔋
|
||||
|
||||
|
||||
def format_battery(status: BatteryStatus) -> str:
|
||||
"""Return a compact label like ``🔋 82%`` / ``⚡ 82%`` (empty if N/A)."""
|
||||
if not status.available or status.percent is None:
|
||||
return ""
|
||||
return f"{battery_glyph(status)} {status.percent}%"
|
||||
+255
-33
@@ -433,6 +433,29 @@ def _model_supports_tool_use(model_id: str) -> bool:
|
||||
return not any(pattern in model_lower for pattern in _NON_TOOL_CALLING_PATTERNS)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Prompt-cache capability detection (Converse API cachePoint)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Claude on Bedrock already gets prompt caching through the AnthropicBedrock
|
||||
# SDK path (see is_anthropic_bedrock_model / runtime_provider.py's dual-path
|
||||
# routing) — it never reaches build_converse_kwargs unless bearer-token auth
|
||||
# forces the Converse path (#28156). This allowlist covers the Converse API
|
||||
# itself: sending an unsupported model a cachePoint block raises a
|
||||
# ValidationException, so — like _model_supports_tool_use but inverted —
|
||||
# unknown models default to NOT receiving cache markers until confirmed.
|
||||
# Ref: https://docs.aws.amazon.com/bedrock/latest/userguide/prompt-caching.html
|
||||
_CACHE_POINT_PATTERNS = [
|
||||
"anthropic.claude", # bearer-token fallback path
|
||||
"amazon.nova",
|
||||
]
|
||||
|
||||
|
||||
def _model_supports_prompt_cache(model_id: str) -> bool:
|
||||
"""Return True if the model accepts a Converse API cachePoint block."""
|
||||
model_lower = model_id.lower()
|
||||
return any(pattern in model_lower for pattern in _CACHE_POINT_PATTERNS)
|
||||
|
||||
|
||||
def is_anthropic_bedrock_model(model_id: str) -> bool:
|
||||
"""Return True if the model is an Anthropic Claude model on Bedrock.
|
||||
|
||||
@@ -448,7 +471,10 @@ def is_anthropic_bedrock_model(model_id: str) -> bool:
|
||||
"""
|
||||
model_lower = model_id.lower()
|
||||
# Strip regional prefix if present
|
||||
for prefix in ("us.", "global.", "eu.", "ap.", "jp."):
|
||||
for prefix in (
|
||||
"global.", "us.", "eu.", "apac.", "ap.", "au.", "jp.",
|
||||
"ca.", "sa.", "me.", "af.",
|
||||
):
|
||||
if model_lower.startswith(prefix):
|
||||
model_lower = model_lower[len(prefix):]
|
||||
break
|
||||
@@ -490,6 +516,26 @@ def convert_tools_to_converse(tools: List[Dict]) -> List[Dict]:
|
||||
return result
|
||||
|
||||
|
||||
# Bedrock's Converse API rejects any text content block whose text is empty
|
||||
# OR whitespace-only (ValidationException: "text content blocks must contain
|
||||
# non-whitespace text"). A lone space is whitespace and is rejected too — the
|
||||
# placeholder MUST itself be non-whitespace. Ref: issue #9486.
|
||||
_EMPTY_TEXT_PLACEHOLDER = "(empty)"
|
||||
|
||||
|
||||
def _safe_text(text) -> str:
|
||||
"""Return ``text`` if it's non-whitespace, else a non-whitespace placeholder.
|
||||
|
||||
Handles None, empty string, and whitespace-only string (spaces, tabs,
|
||||
newlines) — all of which Bedrock's Converse API rejects as text content.
|
||||
"""
|
||||
if text is None:
|
||||
return _EMPTY_TEXT_PLACEHOLDER
|
||||
if not isinstance(text, str):
|
||||
text = str(text)
|
||||
return text if text.strip() else _EMPTY_TEXT_PLACEHOLDER
|
||||
|
||||
|
||||
def _convert_content_to_converse(content) -> List[Dict]:
|
||||
"""Convert OpenAI message content (string or list) to Converse content blocks.
|
||||
|
||||
@@ -497,26 +543,27 @@ def _convert_content_to_converse(content) -> List[Dict]:
|
||||
- Plain text strings → [{"text": "..."}]
|
||||
- Content arrays with text/image_url parts → mixed text/image blocks
|
||||
|
||||
Filters out empty text blocks — Bedrock's Converse API rejects messages
|
||||
where a text content block has an empty ``text`` field (ValidationException:
|
||||
"text content blocks must be non-empty"). Ref: issue #9486.
|
||||
Replaces empty/whitespace-only text blocks with a non-whitespace
|
||||
placeholder — Bedrock's Converse API rejects messages where a text
|
||||
content block is empty or whitespace-only (ValidationException:
|
||||
"text content blocks must contain non-whitespace text"). Ref: issue #9486.
|
||||
"""
|
||||
if content is None:
|
||||
return [{"text": " "}]
|
||||
return [{"text": _safe_text(content)}]
|
||||
if isinstance(content, str):
|
||||
return [{"text": content}] if content.strip() else [{"text": " "}]
|
||||
return [{"text": _safe_text(content)}]
|
||||
if isinstance(content, list):
|
||||
blocks = []
|
||||
for part in content:
|
||||
if isinstance(part, str):
|
||||
blocks.append({"text": part})
|
||||
blocks.append({"text": _safe_text(part)})
|
||||
continue
|
||||
if not isinstance(part, dict):
|
||||
continue
|
||||
part_type = part.get("type", "")
|
||||
if part_type == "text":
|
||||
text = part.get("text", "")
|
||||
blocks.append({"text": text if text else " "})
|
||||
blocks.append({"text": _safe_text(text)})
|
||||
elif part_type == "image_url":
|
||||
image_url = part.get("image_url", {})
|
||||
url = image_url.get("url", "") if isinstance(image_url, dict) else ""
|
||||
@@ -547,8 +594,8 @@ def _convert_content_to_converse(content) -> List[Dict]:
|
||||
# Remote URL — Converse doesn't support URLs directly,
|
||||
# include as text reference for the model.
|
||||
blocks.append({"text": f"[Image: {url}]"})
|
||||
return blocks if blocks else [{"text": " "}]
|
||||
return [{"text": str(content)}]
|
||||
return blocks if blocks else [{"text": _EMPTY_TEXT_PLACEHOLDER}]
|
||||
return [{"text": _safe_text(content)}]
|
||||
|
||||
|
||||
def convert_messages_to_converse(
|
||||
@@ -578,14 +625,18 @@ def convert_messages_to_converse(
|
||||
content = msg.get("content")
|
||||
|
||||
if role == "system":
|
||||
# System messages become the system prompt
|
||||
# System messages become the system prompt. Blank/whitespace-only
|
||||
# parts are dropped entirely (not placeholder-filled) since a
|
||||
# system prompt made up of only placeholder text is meaningless.
|
||||
if isinstance(content, str) and content.strip():
|
||||
system_blocks.append({"text": content})
|
||||
elif isinstance(content, list):
|
||||
for part in content:
|
||||
if isinstance(part, dict) and part.get("type") == "text":
|
||||
system_blocks.append({"text": part.get("text", "")})
|
||||
elif isinstance(part, str):
|
||||
text = part.get("text", "")
|
||||
if isinstance(text, str) and text.strip():
|
||||
system_blocks.append({"text": text})
|
||||
elif isinstance(part, str) and part.strip():
|
||||
system_blocks.append({"text": part})
|
||||
continue
|
||||
|
||||
@@ -596,7 +647,7 @@ def convert_messages_to_converse(
|
||||
tool_result_block = {
|
||||
"toolResult": {
|
||||
"toolUseId": tool_call_id,
|
||||
"content": [{"text": result_content}],
|
||||
"content": [{"text": _safe_text(result_content)}],
|
||||
}
|
||||
}
|
||||
# In Converse, tool results go in a "user" role message
|
||||
@@ -635,7 +686,7 @@ def convert_messages_to_converse(
|
||||
})
|
||||
|
||||
if not content_blocks:
|
||||
content_blocks = [{"text": " "}]
|
||||
content_blocks = [{"text": _EMPTY_TEXT_PLACEHOLDER}]
|
||||
|
||||
# Merge with previous assistant message if needed (strict alternation)
|
||||
if converse_msgs and converse_msgs[-1]["role"] == "assistant":
|
||||
@@ -661,11 +712,11 @@ def convert_messages_to_converse(
|
||||
|
||||
# Converse requires the first message to be from the user
|
||||
if converse_msgs and converse_msgs[0]["role"] != "user":
|
||||
converse_msgs.insert(0, {"role": "user", "content": [{"text": " "}]})
|
||||
converse_msgs.insert(0, {"role": "user", "content": [{"text": _EMPTY_TEXT_PLACEHOLDER}]})
|
||||
|
||||
# Converse requires the last message to be from the user
|
||||
if converse_msgs and converse_msgs[-1]["role"] != "user":
|
||||
converse_msgs.append({"role": "user", "content": [{"text": " "}]})
|
||||
converse_msgs.append({"role": "user", "content": [{"text": _EMPTY_TEXT_PLACEHOLDER}]})
|
||||
|
||||
return (system_blocks if system_blocks else None, converse_msgs)
|
||||
|
||||
@@ -736,14 +787,22 @@ def normalize_converse_response(response: Dict) -> SimpleNamespace:
|
||||
reasoning_content="\n\n".join(reasoning_parts) if reasoning_parts else None,
|
||||
)
|
||||
|
||||
# Build usage stats
|
||||
# Build usage stats. Converse's inputTokens excludes cache read/write
|
||||
# tokens (unlike OpenAI's prompt_tokens, which includes them) — restore
|
||||
# the OpenAI-style "total includes cache" convention here so downstream
|
||||
# normalize_usage() can subtract them back out consistently, and surface
|
||||
# the Anthropic-named fields it already falls back to for cache reads.
|
||||
usage_data = response.get("usage", {})
|
||||
input_tokens = usage_data.get("inputTokens", 0)
|
||||
cache_read_tokens = usage_data.get("cacheReadInputTokens", 0)
|
||||
cache_write_tokens = usage_data.get("cacheWriteInputTokens", 0)
|
||||
output_tokens = usage_data.get("outputTokens", 0)
|
||||
usage = SimpleNamespace(
|
||||
prompt_tokens=usage_data.get("inputTokens", 0),
|
||||
completion_tokens=usage_data.get("outputTokens", 0),
|
||||
total_tokens=(
|
||||
usage_data.get("inputTokens", 0) + usage_data.get("outputTokens", 0)
|
||||
),
|
||||
prompt_tokens=input_tokens + cache_read_tokens + cache_write_tokens,
|
||||
completion_tokens=output_tokens,
|
||||
total_tokens=input_tokens + cache_read_tokens + cache_write_tokens + output_tokens,
|
||||
cache_read_input_tokens=cache_read_tokens,
|
||||
cache_creation_input_tokens=cache_write_tokens,
|
||||
)
|
||||
|
||||
finish_reason = _converse_stop_reason_to_openai(stop_reason)
|
||||
@@ -789,6 +848,7 @@ def stream_converse_with_callbacks(
|
||||
on_tool_start=None,
|
||||
on_reasoning_delta=None,
|
||||
on_interrupt_check=None,
|
||||
on_event=None,
|
||||
) -> SimpleNamespace:
|
||||
"""Process a Bedrock ConverseStream event stream with real-time callbacks.
|
||||
|
||||
@@ -808,6 +868,12 @@ def stream_converse_with_callbacks(
|
||||
on supported models (Claude 4.6+).
|
||||
on_interrupt_check: Called on each event. Should return True if the
|
||||
agent has been interrupted and streaming should stop.
|
||||
on_event: Called once at the top of the loop body for EVERY yielded
|
||||
Bedrock event (text/tool-input/reasoning/metadata deltas alike),
|
||||
before any branching. Provides a wire-level liveness signal so an
|
||||
external watchdog can distinguish "still receiving events" from
|
||||
"stream wedged with no data". Errors raised by the callback are
|
||||
swallowed so a liveness hook can never abort the stream.
|
||||
|
||||
Returns:
|
||||
An OpenAI-compatible SimpleNamespace response, identical in shape to
|
||||
@@ -823,6 +889,15 @@ def stream_converse_with_callbacks(
|
||||
usage_data: Dict[str, int] = {}
|
||||
|
||||
for event in event_stream.get("stream", []):
|
||||
# Wire-level liveness signal: fire on EVERY yielded event (text, tool
|
||||
# input, reasoning, metadata) before branching so an external watchdog
|
||||
# can tell a still-flowing stream from a wedged one. Best-effort — a
|
||||
# liveness callback must never be able to abort the stream.
|
||||
if on_event is not None:
|
||||
try:
|
||||
on_event()
|
||||
except Exception:
|
||||
pass
|
||||
# Check for interrupt
|
||||
if on_interrupt_check and on_interrupt_check():
|
||||
break
|
||||
@@ -892,6 +967,8 @@ def stream_converse_with_callbacks(
|
||||
usage_data = {
|
||||
"inputTokens": meta_usage.get("inputTokens", 0),
|
||||
"outputTokens": meta_usage.get("outputTokens", 0),
|
||||
"cacheReadInputTokens": meta_usage.get("cacheReadInputTokens", 0),
|
||||
"cacheWriteInputTokens": meta_usage.get("cacheWriteInputTokens", 0),
|
||||
}
|
||||
|
||||
# Flush remaining text
|
||||
@@ -905,12 +982,16 @@ def stream_converse_with_callbacks(
|
||||
reasoning_content="\n\n".join(reasoning_parts) if reasoning_parts else None,
|
||||
)
|
||||
|
||||
input_tokens = usage_data.get("inputTokens", 0)
|
||||
cache_read_tokens = usage_data.get("cacheReadInputTokens", 0)
|
||||
cache_write_tokens = usage_data.get("cacheWriteInputTokens", 0)
|
||||
output_tokens = usage_data.get("outputTokens", 0)
|
||||
usage = SimpleNamespace(
|
||||
prompt_tokens=usage_data.get("inputTokens", 0),
|
||||
completion_tokens=usage_data.get("outputTokens", 0),
|
||||
total_tokens=(
|
||||
usage_data.get("inputTokens", 0) + usage_data.get("outputTokens", 0)
|
||||
),
|
||||
prompt_tokens=input_tokens + cache_read_tokens + cache_write_tokens,
|
||||
completion_tokens=output_tokens,
|
||||
total_tokens=input_tokens + cache_read_tokens + cache_write_tokens + output_tokens,
|
||||
cache_read_input_tokens=cache_read_tokens,
|
||||
cache_creation_input_tokens=cache_write_tokens,
|
||||
)
|
||||
|
||||
finish_reason = _converse_stop_reason_to_openai(stop_reason)
|
||||
@@ -949,6 +1030,7 @@ def build_converse_kwargs(
|
||||
Converts OpenAI-format inputs to Converse API parameters.
|
||||
"""
|
||||
system_prompt, converse_messages = convert_messages_to_converse(messages)
|
||||
cache_enabled = _model_supports_prompt_cache(model)
|
||||
|
||||
kwargs: Dict[str, Any] = {
|
||||
"modelId": model,
|
||||
@@ -959,6 +1041,8 @@ def build_converse_kwargs(
|
||||
}
|
||||
|
||||
if system_prompt:
|
||||
if cache_enabled:
|
||||
system_prompt = system_prompt + [{"cachePoint": {"type": "default"}}]
|
||||
kwargs["system"] = system_prompt
|
||||
|
||||
from agent.anthropic_adapter import _forbids_sampling_params
|
||||
@@ -982,6 +1066,8 @@ def build_converse_kwargs(
|
||||
# Strip tools for known non-tool-calling models and warn the user.
|
||||
# Ref: PR #7920 feedback from @ptlally, pattern from PR #4346.
|
||||
if _model_supports_tool_use(model):
|
||||
if cache_enabled:
|
||||
converse_tools = converse_tools + [{"cachePoint": {"type": "default"}}]
|
||||
kwargs["toolConfig"] = {"tools": converse_tools}
|
||||
else:
|
||||
logger.warning(
|
||||
@@ -989,6 +1075,14 @@ def build_converse_kwargs(
|
||||
"The agent will operate in text-only mode.", model
|
||||
)
|
||||
|
||||
if cache_enabled and len(converse_messages) >= 2:
|
||||
# Checkpoint everything up to (not including) the newest turn, so the
|
||||
# marker survives unchanged across requests as only the tail grows —
|
||||
# mirroring the Anthropic system_and_3 strategy in prompt_caching.py.
|
||||
content = converse_messages[-2].get("content")
|
||||
if isinstance(content, list) and content:
|
||||
content.append({"cachePoint": {"type": "default"}})
|
||||
|
||||
if guardrail_config:
|
||||
kwargs["guardrailConfig"] = guardrail_config
|
||||
|
||||
@@ -1305,9 +1399,24 @@ def classify_bedrock_error(error_message: str) -> str:
|
||||
# detection is unavailable.
|
||||
|
||||
BEDROCK_CONTEXT_LENGTHS: Dict[str, int] = {
|
||||
# Anthropic Claude models on Bedrock
|
||||
"anthropic.claude-opus-4-6": 200_000,
|
||||
"anthropic.claude-sonnet-4-6": 200_000,
|
||||
# Anthropic Claude models on Bedrock.
|
||||
# Context windows per Anthropic's official models comparison
|
||||
# (https://platform.claude.com/docs/en/about-claude/models/overview).
|
||||
# Fable / Sonnet 5 / Opus 4.8 / 4.7 / 4.6 / Sonnet 4.6 have 1M generally
|
||||
# available (no beta header required as of April 2026). Sonnet 4.5 and
|
||||
# Sonnet 4 had their `context-1m-2025-08-07` beta retired on
|
||||
# April 30, 2026, so they are standard 200K; Haiku 4.5 is 200K.
|
||||
# These 1M entries must match agent/model_metadata.py
|
||||
# DEFAULT_CONTEXT_LENGTHS or the agent compresses context prematurely.
|
||||
# Keys are matched by longest-substring, so the versioned 4-6/4-7/4-8
|
||||
# entries win over the generic "anthropic.claude-opus-4" fallback.
|
||||
"anthropic.claude-fable-5": 1_000_000,
|
||||
"anthropic.claude-fable": 1_000_000,
|
||||
"anthropic.claude-sonnet-5": 1_000_000,
|
||||
"anthropic.claude-opus-4-8": 1_000_000,
|
||||
"anthropic.claude-opus-4-7": 1_000_000,
|
||||
"anthropic.claude-opus-4-6": 1_000_000,
|
||||
"anthropic.claude-sonnet-4-6": 1_000_000,
|
||||
"anthropic.claude-sonnet-4-5": 200_000,
|
||||
"anthropic.claude-haiku-4-5": 200_000,
|
||||
"anthropic.claude-opus-4": 200_000,
|
||||
@@ -1334,9 +1443,22 @@ BEDROCK_CONTEXT_LENGTHS: Dict[str, int] = {
|
||||
# Default for unknown Bedrock models
|
||||
BEDROCK_DEFAULT_CONTEXT_LENGTH = 128_000
|
||||
|
||||
# Probe tiers (in tokens). We send a request padded just past each tier and
|
||||
# read the real window from Bedrock's length-validation error. Two reasons
|
||||
# this is tiered rather than one giant request:
|
||||
# 1. A wildly oversized payload (e.g. 5M tokens) makes Bedrock return an
|
||||
# opaque InternalServerException after retries instead of a clean
|
||||
# ValidationException — so we must stay within a sane overage.
|
||||
# 2. Stepping up lets us discover larger windows (2M+) without over-padding
|
||||
# smaller ones.
|
||||
# Each tier value is the *padding target*; the error reports the true maximum,
|
||||
# which is what we actually return.
|
||||
_BEDROCK_PROBE_TIERS = (1_300_000, 2_200_000)
|
||||
_WORDS_PER_TOKEN = 0.9 # conservative: ensures the padded prompt clears the tier
|
||||
|
||||
def get_bedrock_context_length(model_id: str) -> int:
|
||||
"""Look up the context window size for a Bedrock model.
|
||||
|
||||
def _static_bedrock_context_length(model_id: str) -> int:
|
||||
"""Longest-substring-match lookup against the static fallback table.
|
||||
|
||||
Uses substring matching so versioned IDs like
|
||||
``anthropic.claude-sonnet-4-6-20250514-v1:0`` resolve correctly.
|
||||
@@ -1349,3 +1471,103 @@ def get_bedrock_context_length(model_id: str) -> int:
|
||||
best_key = key
|
||||
best_val = val
|
||||
return best_val
|
||||
|
||||
|
||||
def probe_bedrock_context_length(model_id: str, region: str) -> Optional[int]:
|
||||
"""Discover a Bedrock model's real context window by provoking a length error.
|
||||
|
||||
Bedrock does not expose the context window via any metadata API
|
||||
(``get-foundation-model`` omits it, ``Converse`` metrics omit it,
|
||||
``CountTokens`` is unsupported on several models). The only authoritative
|
||||
source is the ``ValidationException`` raised when a prompt exceeds the
|
||||
window:
|
||||
|
||||
"The model returned the following errors: prompt is too long:
|
||||
1300032 tokens > 1000000 maximum"
|
||||
|
||||
Length validation happens *before* inference, so an oversized request is
|
||||
rejected immediately and cheaply — no tokens are generated and no input is
|
||||
actually processed. We pad a request just past each tier in
|
||||
``_BEDROCK_PROBE_TIERS`` and parse the reported ``maximum``. Tiers exist
|
||||
because (a) a *wildly* oversized payload makes Bedrock fail with an opaque
|
||||
InternalServerException instead of a clean length error, and (b) stepping
|
||||
up discovers larger windows without over-padding smaller ones.
|
||||
|
||||
Returns the detected window, or ``None`` if the probe could not run
|
||||
(missing credentials, network error, or no parseable limit) so the caller
|
||||
can fall back to the static table.
|
||||
"""
|
||||
try:
|
||||
from agent.model_metadata import parse_context_limit_from_error
|
||||
except ImportError: # pragma: no cover — same package
|
||||
return None
|
||||
|
||||
try:
|
||||
client = _get_bedrock_runtime_client(region)
|
||||
except Exception as exc: # boto3 missing / credential resolution failure
|
||||
logger.debug("Bedrock context probe skipped for %s: %s", model_id, exc)
|
||||
return None
|
||||
|
||||
last_error = ""
|
||||
for tier_tokens in _BEDROCK_PROBE_TIERS:
|
||||
pad_words = int(tier_tokens / _WORDS_PER_TOKEN)
|
||||
oversized = "data " * pad_words
|
||||
try:
|
||||
client.converse(
|
||||
modelId=model_id,
|
||||
messages=[{"role": "user", "content": [{"text": oversized}]}],
|
||||
inferenceConfig={"maxTokens": 8},
|
||||
)
|
||||
# Accepted a prompt this large → the window is at least this tier.
|
||||
# Returning the tier as a lower bound is safe and avoids inventing
|
||||
# a number we can't confirm.
|
||||
logger.debug(
|
||||
"Bedrock context probe for %s accepted ~%s-token prompt; "
|
||||
"window is at least that", model_id, f"{tier_tokens:,}",
|
||||
)
|
||||
return tier_tokens
|
||||
except Exception as exc:
|
||||
msg = str(exc)
|
||||
last_error = msg
|
||||
limit = parse_context_limit_from_error(msg)
|
||||
if limit and limit >= 1024:
|
||||
logger.info(
|
||||
"Probed Bedrock context window for %s: %s tokens",
|
||||
model_id, f"{limit:,}",
|
||||
)
|
||||
return limit
|
||||
# No parseable limit at this tier (opaque server error, auth,
|
||||
# throttle). Try the next, smaller-overage strategy is N/A here —
|
||||
# tiers ascend — so just continue; if all fail we return None.
|
||||
continue
|
||||
|
||||
logger.debug(
|
||||
"Bedrock context probe for %s returned no parseable limit: %s",
|
||||
model_id, last_error[:200],
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def get_bedrock_context_length(model_id: str, region: str = "", probe: bool = True) -> int:
|
||||
"""Resolve the context window for a Bedrock model.
|
||||
|
||||
Resolution order:
|
||||
1. Live probe against Bedrock (authoritative; cached by the caller).
|
||||
2. Static fallback table (longest-substring match).
|
||||
3. Conservative default.
|
||||
|
||||
The static table is intentionally a *fallback*, not the primary source:
|
||||
AWS ships new model versions (opus-4-7, opus-4-8, ...) faster than the
|
||||
table can track, and a stale entry silently caps the window (e.g. a
|
||||
1M-token Opus pinned to 200K via an ``opus-4`` substring match). The
|
||||
probe asks Bedrock directly so every model — current or future — gets its
|
||||
real window with no table maintenance.
|
||||
|
||||
``probe=False`` (or an empty ``region``) skips the network call and uses
|
||||
the static table only — used by pure-offline/display code paths.
|
||||
"""
|
||||
if probe and region:
|
||||
probed = probe_bedrock_context_length(model_id, region)
|
||||
if probed:
|
||||
return probed
|
||||
return _static_bedrock_context_length(model_id)
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
"""Provider-agnostic billing/credit recovery links.
|
||||
|
||||
Maps a billing-classified failure onto a recovery link + label. *Detection*
|
||||
is not done here — that is :mod:`agent.error_classifier`
|
||||
(``FailoverReason.billing``), the single source of truth for "credit wall vs.
|
||||
rate limit / auth / transport". The resulting :class:`BillingBlock` rides the
|
||||
turn result and the gateway ``message.complete`` event so every surface (CLI,
|
||||
TUI, desktop) renders one structured signal instead of re-parsing error text.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import asdict, dataclass
|
||||
from typing import Optional
|
||||
|
||||
from utils import base_url_host_matches
|
||||
|
||||
|
||||
@dataclass
|
||||
class BillingBlock:
|
||||
"""Structured billing-wall descriptor shared across every surface.
|
||||
|
||||
``is_nous`` is the routing bit: Nous has a first-class in-app billing surface
|
||||
(desktop Settings → Billing, TUI/CLI ``/topup``), so surfaces prefer that over
|
||||
``billing_url``; third-party providers have no in-app flow, so ``billing_url``
|
||||
is the deep link the user actually needs.
|
||||
"""
|
||||
|
||||
provider: str
|
||||
provider_label: str
|
||||
model: str
|
||||
billing_url: Optional[str]
|
||||
is_nous: bool
|
||||
message: str
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _Provider:
|
||||
label: str
|
||||
url: str
|
||||
slugs: tuple[str, ...]
|
||||
hosts: tuple[str, ...] = ()
|
||||
|
||||
|
||||
# Single source of truth: internal slug(s) + base_url host(s) → billing page.
|
||||
# Curated "add credits / manage billing" landing pages, not marketing homes.
|
||||
# Hosts back the OpenAI-compatible fallback where the slug is a generic bucket
|
||||
# (e.g. "openai_compatible") but base_url reveals the real upstream. An unknown
|
||||
# provider degrades to a readable label with no invented URL.
|
||||
_PROVIDERS: tuple[_Provider, ...] = (
|
||||
_Provider("OpenAI", "https://platform.openai.com/settings/organization/billing", ("openai",), ("api.openai.com",)),
|
||||
_Provider("Anthropic", "https://console.anthropic.com/settings/billing", ("anthropic",), ("api.anthropic.com",)),
|
||||
_Provider("OpenRouter", "https://openrouter.ai/settings/credits", ("openrouter",), ("openrouter.ai",)),
|
||||
_Provider("xAI", "https://console.x.ai/team/default/billing", ("xai", "xai-oauth"), ("api.x.ai",)),
|
||||
_Provider("DeepSeek", "https://platform.deepseek.com/top_up", ("deepseek",), ("api.deepseek.com",)),
|
||||
_Provider("Groq", "https://console.groq.com/settings/billing", ("groq",), ("api.groq.com",)),
|
||||
_Provider("Mistral", "https://console.mistral.ai/billing", ("mistral",), ("api.mistral.ai",)),
|
||||
_Provider("Together AI", "https://api.together.ai/settings/billing", ("together",), ("api.together.ai", "api.together.xyz")),
|
||||
_Provider("Fireworks AI", "https://fireworks.ai/account/billing", ("fireworks",), ("fireworks.ai",)),
|
||||
_Provider("Perplexity", "https://www.perplexity.ai/settings/api", ("perplexity",), ("perplexity.ai",)),
|
||||
_Provider("Google AI", "https://aistudio.google.com/app/billing", ("google", "gemini"), ("generativelanguage.googleapis.com",)),
|
||||
_Provider("Cohere", "https://dashboard.cohere.com/billing", ("cohere",)),
|
||||
_Provider("Moonshot AI", "https://platform.moonshot.ai/console/pay", ("moonshot",)),
|
||||
_Provider("NVIDIA", "https://build.nvidia.com/settings/billing", ("nvidia",)),
|
||||
)
|
||||
|
||||
_BY_SLUG: dict[str, _Provider] = {slug: p for p in _PROVIDERS for slug in p.slugs}
|
||||
|
||||
|
||||
def is_nous_inference_route(provider: str, base_url: str) -> bool:
|
||||
"""True when the failing route is the Nous-managed inference gateway."""
|
||||
if (provider or "").strip().lower() == "nous":
|
||||
return True
|
||||
return base_url_host_matches(str(base_url or ""), "inference-api.nousresearch.com")
|
||||
|
||||
|
||||
def _nous_billing_url() -> Optional[str]:
|
||||
"""Best-effort Nous portal billing URL (text-surface fallback; Nous prefers the in-app flow)."""
|
||||
try:
|
||||
from hermes_cli.nous_account import nous_portal_billing_url
|
||||
|
||||
return nous_portal_billing_url(None)
|
||||
except Exception:
|
||||
return "https://portal.nousresearch.com/billing"
|
||||
|
||||
|
||||
def _resolve_provider_link(slug: str, base_url: str) -> tuple[str, Optional[str]]:
|
||||
"""Resolve ``(label, url)``: exact slug → base_url host → readable-label fallback."""
|
||||
hit = _BY_SLUG.get(slug)
|
||||
if hit:
|
||||
return hit.label, hit.url
|
||||
|
||||
base = str(base_url or "")
|
||||
for p in _PROVIDERS:
|
||||
if any(base_url_host_matches(base, host) for host in p.hosts):
|
||||
return p.label, p.url
|
||||
|
||||
return slug.replace("_", " ").replace("-", " ").strip().title() or "your provider", None
|
||||
|
||||
|
||||
def build_billing_block(
|
||||
*,
|
||||
provider: str,
|
||||
base_url: str,
|
||||
model: str,
|
||||
message: str = "",
|
||||
) -> BillingBlock:
|
||||
"""Build the billing descriptor for a billing-classified failure.
|
||||
|
||||
``message`` is the guidance already assembled by the agent loop
|
||||
(:func:`agent.conversation_loop._billing_or_entitlement_message`), carried
|
||||
through unchanged so every surface shows identical copy.
|
||||
"""
|
||||
slug = (provider or "").strip().lower()
|
||||
model = (model or "").strip()
|
||||
|
||||
if is_nous_inference_route(slug, base_url):
|
||||
return BillingBlock(slug or "nous", "Nous Portal", model, _nous_billing_url(), True, message or "")
|
||||
|
||||
label, url = _resolve_provider_link(slug, base_url)
|
||||
return BillingBlock(slug, label, model, url, False, message or "")
|
||||
@@ -34,7 +34,7 @@ from __future__ import annotations
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
+55
-2
@@ -1,4 +1,4 @@
|
||||
"""Surface-agnostic core for the Phase 2b terminal-billing screens.
|
||||
"""Surface-agnostic core for the Phase 2b Remote Spending screens.
|
||||
|
||||
One fetch/parse per concern, consumed identically by the CLI handler
|
||||
(``cli.py::_show_billing``), the TUI JSON-RPC methods
|
||||
@@ -17,7 +17,7 @@ from __future__ import annotations
|
||||
import logging
|
||||
import os
|
||||
import uuid
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Optional
|
||||
|
||||
@@ -107,6 +107,22 @@ class CardInfo:
|
||||
return f"{self.masked} — {label}" if label else self.masked
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PaymentMethodInfo:
|
||||
"""The payment method on file. `kind` is "card", "link", or "unknown"
|
||||
— anything else is normalised to "unknown" at parse time, so consumers
|
||||
only ever see fields that belong to the kind they are looking at."""
|
||||
|
||||
kind: str
|
||||
brand: Optional[str] = None
|
||||
last4: Optional[str] = None
|
||||
wallet: Optional[str] = None
|
||||
email: Optional[str] = None
|
||||
resolved_via: Optional[str] = None
|
||||
#: What the server called it, when we did not recognise the kind.
|
||||
raw_kind: Optional[str] = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MonthlyCap:
|
||||
limit_usd: Optional[Decimal] = None
|
||||
@@ -150,6 +166,7 @@ class BillingState:
|
||||
min_usd: Optional[Decimal] = None
|
||||
max_usd: Optional[Decimal] = None
|
||||
card: Optional[CardInfo] = None
|
||||
payment_method: Optional[PaymentMethodInfo] = None
|
||||
monthly_cap: Optional[MonthlyCap] = None
|
||||
auto_reload: Optional[AutoReload] = None
|
||||
portal_url: Optional[str] = None
|
||||
@@ -201,6 +218,41 @@ def _parse_card(raw: Any) -> Optional[CardInfo]:
|
||||
return CardInfo(brand=brand, last4=last4, resolved_via=resolved_via)
|
||||
|
||||
|
||||
def _parse_payment_method(raw: Any) -> Optional[PaymentMethodInfo]:
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
kind = raw.get("kind")
|
||||
if not isinstance(kind, str):
|
||||
return None
|
||||
|
||||
def _optional_string(key: str) -> Optional[str]:
|
||||
value = raw.get(key)
|
||||
return value if isinstance(value, str) else None
|
||||
|
||||
resolved_via = _optional_string("resolvedVia")
|
||||
brand = _optional_string("brand")
|
||||
last4 = _optional_string("last4")
|
||||
# Settle the kind here, the way _parse_card settles a card, so nothing
|
||||
# downstream has to re-check which fields this kind is allowed to have.
|
||||
if kind == "card" and brand and last4:
|
||||
return PaymentMethodInfo(
|
||||
kind="card",
|
||||
brand=brand,
|
||||
last4=last4,
|
||||
wallet=_optional_string("wallet"),
|
||||
resolved_via=resolved_via,
|
||||
)
|
||||
if kind == "link":
|
||||
return PaymentMethodInfo(
|
||||
kind="link",
|
||||
email=_optional_string("email"),
|
||||
resolved_via=resolved_via,
|
||||
)
|
||||
return PaymentMethodInfo(
|
||||
kind="unknown", raw_kind=kind, resolved_via=resolved_via
|
||||
)
|
||||
|
||||
|
||||
def _parse_monthly_cap(raw: Any) -> Optional[MonthlyCap]:
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
@@ -274,6 +326,7 @@ def billing_state_from_payload(
|
||||
min_usd=parse_money(bounds.get("minUsd")),
|
||||
max_usd=parse_money(bounds.get("maxUsd")),
|
||||
card=_parse_card(payload.get("card")),
|
||||
payment_method=_parse_payment_method(payload.get("paymentMethod")),
|
||||
monthly_cap=_parse_monthly_cap(payload.get("monthlyCap")),
|
||||
auto_reload=_parse_auto_reload(payload.get("autoReload")),
|
||||
portal_url=portal_url,
|
||||
|
||||
+1263
-401
File diff suppressed because it is too large
Load Diff
@@ -18,6 +18,7 @@ import uuid
|
||||
from types import SimpleNamespace
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from agent.message_sanitization import deterministic_call_id
|
||||
from agent.prompt_builder import DEFAULT_AGENT_IDENTITY
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -182,12 +183,34 @@ def _summarize_user_message_for_log(content: Any, *, sep: str = " ") -> str:
|
||||
def _deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
|
||||
"""Generate a deterministic call_id from tool call content.
|
||||
|
||||
Used as a fallback when the API doesn't provide a call_id.
|
||||
Thin wrapper over the single policy owner
|
||||
``agent.message_sanitization.deterministic_call_id`` (audit F4) — kept
|
||||
as a module-level name because run_agent and tests import it from here.
|
||||
Deterministic IDs prevent cache invalidation — random UUIDs would
|
||||
make every API call's prefix unique, breaking OpenAI's prompt cache.
|
||||
"""
|
||||
seed = f"{fn_name}:{arguments}:{index}"
|
||||
digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12]
|
||||
return deterministic_call_id(fn_name, arguments, index)
|
||||
|
||||
|
||||
def _clamp_responses_call_id(call_id: str) -> str:
|
||||
"""Keep a ``call_id`` within the Responses API's 64-char limit (#73492).
|
||||
|
||||
The codex app-server namespaces MCP tool call ids as
|
||||
``codex_mcp__<server>__<tool>_<codex_call_id>``; with an ``exec-<uuid>``
|
||||
component the built-in ``hermes-tools`` server already overflows 64 chars,
|
||||
and the Responses API rejects the whole payload with a non-retryable HTTP
|
||||
400 that then replays every turn — permanently bricking the session.
|
||||
|
||||
Sibling defect to #10788 (which clamped ``input[*].id``), applied here to
|
||||
``call_id``. The surrogate is a pure, deterministic function of the
|
||||
original, so the ``function_call`` and its matching ``function_call_output``
|
||||
— which carry the same original id — map to the same surrogate and stay
|
||||
paired without correlating the two items. Short ids pass through unchanged,
|
||||
preserving prompt-cache prefixes.
|
||||
"""
|
||||
if len(call_id) <= _MAX_RESPONSES_ITEM_ID_LENGTH:
|
||||
return call_id
|
||||
digest = hashlib.sha256(call_id.encode("utf-8", errors="replace")).hexdigest()[:32]
|
||||
return f"call_{digest}"
|
||||
|
||||
|
||||
@@ -546,7 +569,7 @@ def _chat_messages_to_responses_input(
|
||||
|
||||
items.append({
|
||||
"type": "function_call",
|
||||
"call_id": call_id,
|
||||
"call_id": _clamp_responses_call_id(call_id),
|
||||
"name": fn_name,
|
||||
"arguments": arguments,
|
||||
})
|
||||
@@ -589,7 +612,7 @@ def _chat_messages_to_responses_input(
|
||||
|
||||
items.append({
|
||||
"type": "function_call_output",
|
||||
"call_id": call_id,
|
||||
"call_id": _clamp_responses_call_id(call_id),
|
||||
"output": output_value,
|
||||
})
|
||||
|
||||
@@ -912,7 +935,8 @@ def _preflight_codex_api_kwargs(
|
||||
allowed_keys = {
|
||||
"model", "instructions", "input", "tools", "store",
|
||||
"reasoning", "include", "max_output_tokens", "temperature",
|
||||
"tool_choice", "parallel_tool_calls", "prompt_cache_key", "service_tier",
|
||||
"tool_choice", "parallel_tool_calls", "prompt_cache_key",
|
||||
"prompt_cache_retention", "service_tier",
|
||||
"extra_headers", "extra_body", "timeout",
|
||||
}
|
||||
normalized: Dict[str, Any] = {
|
||||
@@ -950,8 +974,13 @@ def _preflight_codex_api_kwargs(
|
||||
if isinstance(temperature, (int, float)):
|
||||
normalized["temperature"] = float(temperature)
|
||||
|
||||
# Pass through tool_choice, parallel_tool_calls, prompt_cache_key
|
||||
for passthrough_key in ("tool_choice", "parallel_tool_calls", "prompt_cache_key"):
|
||||
# Pass through cache routing/retention and tool-dispatch hints.
|
||||
for passthrough_key in (
|
||||
"tool_choice",
|
||||
"parallel_tool_calls",
|
||||
"prompt_cache_key",
|
||||
"prompt_cache_retention",
|
||||
):
|
||||
val = api_kwargs.get(passthrough_key)
|
||||
if val is not None:
|
||||
normalized[passthrough_key] = val
|
||||
|
||||
+155
-38
@@ -18,7 +18,6 @@ from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from types import SimpleNamespace
|
||||
from typing import Any, Callable, Dict, List
|
||||
@@ -74,7 +73,10 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
|
||||
try:
|
||||
if not agent._session_db_created:
|
||||
agent._ensure_db_session()
|
||||
agent._session_db.update_token_counts(
|
||||
# Enqueued for the SessionDB background writer — keeps the
|
||||
# per-call accounting write off the turn thread (see
|
||||
# conversation_loop's queue_token_counts call).
|
||||
agent._session_db.queue_token_counts(
|
||||
agent.session_id,
|
||||
model=agent.model,
|
||||
billing_provider=agent.provider,
|
||||
@@ -154,7 +156,8 @@ def _record_codex_app_server_usage(agent, turn) -> dict[str, Any]:
|
||||
try:
|
||||
if not agent._session_db_created:
|
||||
agent._ensure_db_session()
|
||||
agent._session_db.update_token_counts(
|
||||
# Enqueued for the SessionDB background writer (see above).
|
||||
agent._session_db.queue_token_counts(
|
||||
agent.session_id,
|
||||
input_tokens=canonical_usage.input_tokens,
|
||||
output_tokens=canonical_usage.output_tokens,
|
||||
@@ -702,6 +705,16 @@ def run_codex_app_server_turn(
|
||||
except Exception:
|
||||
pass
|
||||
agent._codex_session = None
|
||||
_user_interrupted = bool(
|
||||
getattr(agent, "_interrupt_requested", False)
|
||||
)
|
||||
_interrupt_message = (
|
||||
getattr(agent, "_interrupt_message", None)
|
||||
if _user_interrupted
|
||||
else None
|
||||
)
|
||||
if _user_interrupted:
|
||||
agent.clear_interrupt()
|
||||
return {
|
||||
"final_response": (
|
||||
f"Codex app-server turn failed: {exc}. "
|
||||
@@ -711,9 +724,27 @@ def run_codex_app_server_turn(
|
||||
"api_calls": 0,
|
||||
"completed": False,
|
||||
"partial": True,
|
||||
"interrupted": _user_interrupted,
|
||||
**(
|
||||
{"interrupt_message": _interrupt_message}
|
||||
if _interrupt_message
|
||||
else {}
|
||||
),
|
||||
"error": str(exc),
|
||||
}
|
||||
|
||||
# This runtime bypasses the normal conversation-loop finalizer. Mirror its
|
||||
# interrupt handoff/cleanup so a hard stop cannot poison the next turn and a
|
||||
# message-bearing compatibility interrupt can still be replayed by callers.
|
||||
_user_interrupted = bool(
|
||||
turn.interrupted and getattr(agent, "_interrupt_requested", False)
|
||||
)
|
||||
_interrupt_message = (
|
||||
getattr(agent, "_interrupt_message", None) if _user_interrupted else None
|
||||
)
|
||||
if _user_interrupted:
|
||||
agent.clear_interrupt()
|
||||
|
||||
# If the turn signalled the underlying client is wedged (deadline
|
||||
# blown, post-tool watchdog tripped, OAuth refresh died, subprocess
|
||||
# exited), retire the session so the next turn respawns codex
|
||||
@@ -750,12 +781,27 @@ def run_codex_app_server_turn(
|
||||
# the already-flushed user turn). See gateway/run.py agent_persisted.
|
||||
if getattr(agent, "_session_db", None) is not None:
|
||||
try:
|
||||
agent._flush_messages_to_session_db(messages)
|
||||
_codex_flush_ok = agent._flush_messages_to_session_db(messages)
|
||||
except Exception:
|
||||
logger.debug(
|
||||
_codex_flush_ok = False
|
||||
logger.warning(
|
||||
"codex app-server projected-message flush failed",
|
||||
exc_info=True,
|
||||
)
|
||||
if _codex_flush_ok is False:
|
||||
# Unlike the chat-completions loop (which fails closed BEFORE
|
||||
# projection — see conversation_loop session_persistence_failed),
|
||||
# codex output has already streamed to the user by the time this
|
||||
# flush runs, so there is nothing left to withhold. We cannot
|
||||
# flip agent_persisted=False either: the gateway fallback write
|
||||
# would re-INSERT the already-flushed user turn (#860/#42039).
|
||||
# Surface the durability gap loudly instead of a silent debug.
|
||||
logger.warning(
|
||||
"codex app-server turn was delivered but could NOT be "
|
||||
"persisted to the session DB (session=%s) — this turn "
|
||||
"will be missing after restart/resume",
|
||||
getattr(agent, "session_id", None),
|
||||
)
|
||||
|
||||
|
||||
# Counter ticks for the agent-improvement loop.
|
||||
@@ -819,6 +865,12 @@ def run_codex_app_server_turn(
|
||||
"api_calls": api_calls,
|
||||
"completed": not turn.interrupted and turn.error is None,
|
||||
"partial": turn.interrupted or turn.error is not None,
|
||||
"interrupted": _user_interrupted,
|
||||
**(
|
||||
{"interrupt_message": _interrupt_message}
|
||||
if _interrupt_message
|
||||
else {}
|
||||
),
|
||||
"error": turn.error,
|
||||
# The codex app-server runtime IS an early-return path that bypasses
|
||||
# conversation_loop, but we flush the projected assistant/tool messages
|
||||
@@ -1187,6 +1239,8 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
|
||||
"""
|
||||
import httpx as _httpx
|
||||
|
||||
from agent import relay_llm
|
||||
|
||||
active_client = client or agent._ensure_primary_openai_client(reason="codex_stream_direct")
|
||||
max_stream_retries = 1
|
||||
# Accumulate streamed text so callers / compat shims can read it.
|
||||
@@ -1211,48 +1265,88 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
|
||||
if agent._interrupt_requested:
|
||||
raise InterruptedError("Agent interrupted before Codex stream retry")
|
||||
|
||||
stream_kwargs = dict(api_kwargs)
|
||||
stream_kwargs["stream"] = True
|
||||
intercepted_events = []
|
||||
writer_token = {"value": None}
|
||||
|
||||
def _open_codex_stream(next_api_kwargs: dict[str, Any]):
|
||||
stream_kwargs = dict(next_api_kwargs)
|
||||
stream_kwargs["stream"] = True
|
||||
return active_client.responses.create(**stream_kwargs)
|
||||
|
||||
def _codex_stream_created(_raw_stream: Any) -> None:
|
||||
# Claim the delta sink for THIS physical attempt. A newer attempt
|
||||
# supersedes this token and fences late deltas out of the turn.
|
||||
writer_token["value"] = claim_stream_writer(agent)
|
||||
|
||||
def _accept_codex_chunk(_chunk: Any) -> bool:
|
||||
token = writer_token["value"]
|
||||
if token is None or stream_writer_is_current(agent, token):
|
||||
return True
|
||||
logger.warning(
|
||||
"Codex streaming attempt superseded by a newer stream; "
|
||||
"stopping consumption to preserve the single-writer "
|
||||
"invariant (model=%s).",
|
||||
api_kwargs.get("model", "unknown"),
|
||||
)
|
||||
return False
|
||||
|
||||
def _finalize_codex_stream() -> Any:
|
||||
return _consume_codex_event_stream(
|
||||
list(intercepted_events),
|
||||
model=api_kwargs.get("model"),
|
||||
)
|
||||
|
||||
try:
|
||||
event_stream = active_client.responses.create(**stream_kwargs)
|
||||
except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc:
|
||||
event_stream = relay_llm.stream(
|
||||
dict(api_kwargs),
|
||||
_open_codex_stream,
|
||||
session_id=str(getattr(agent, "session_id", "") or ""),
|
||||
name=str(getattr(agent, "provider", "") or "codex"),
|
||||
model_name=str(api_kwargs.get("model") or ""),
|
||||
finalizer=_finalize_codex_stream,
|
||||
on_stream_created=_codex_stream_created,
|
||||
on_chunk=intercepted_events.append,
|
||||
chunk_adapter=lambda chunk: chunk,
|
||||
accept_chunk=_accept_codex_chunk,
|
||||
completed_response_predicate=lambda response: bool(
|
||||
hasattr(response, "output") and not hasattr(response, "__iter__")
|
||||
),
|
||||
metadata={
|
||||
"api_mode": "codex_responses",
|
||||
"api_request_id": getattr(agent, "_current_api_request_id", None),
|
||||
"call_role": (
|
||||
"delegated"
|
||||
if getattr(agent, "is_subagent", False)
|
||||
else "fallback"
|
||||
if int(getattr(agent, "_fallback_index", 0) or 0) > 0
|
||||
else "primary"
|
||||
),
|
||||
"retry_count": attempt,
|
||||
},
|
||||
defer_logical_completion=True,
|
||||
)
|
||||
except (
|
||||
_httpx.RemoteProtocolError,
|
||||
_httpx.ReadTimeout,
|
||||
_httpx.ConnectError,
|
||||
ConnectionError,
|
||||
) as exc:
|
||||
if attempt < max_stream_retries:
|
||||
logger.debug(
|
||||
"Codex Responses stream connect failed (attempt %s/%s); retrying. %s error=%s",
|
||||
attempt + 1, max_stream_retries + 1,
|
||||
agent._client_log_context(), exc,
|
||||
"Codex Responses stream connect failed (attempt %s/%s); "
|
||||
"retrying. %s error=%s",
|
||||
attempt + 1,
|
||||
max_stream_retries + 1,
|
||||
agent._client_log_context(),
|
||||
exc,
|
||||
)
|
||||
continue
|
||||
raise
|
||||
|
||||
# Claim the delta sink for THIS attempt (#65991) — parity with the
|
||||
# chat_completions/anthropic/bedrock paths. If a prior attempt's
|
||||
# stream is somehow still alive, this claim supersedes it so its
|
||||
# late deltas are fenced out of the turn; conversely, a newer
|
||||
# attempt supersedes us and the interrupt_check below stops our
|
||||
# consumption immediately.
|
||||
_writer_token = claim_stream_writer(agent)
|
||||
|
||||
def _interrupt_or_superseded(_tok=_writer_token) -> bool:
|
||||
if agent._interrupt_requested:
|
||||
return True
|
||||
if not stream_writer_is_current(agent, _tok):
|
||||
logger.warning(
|
||||
"Codex streaming attempt superseded by a newer stream; "
|
||||
"stopping consumption to preserve the single-writer "
|
||||
"invariant (model=%s).",
|
||||
api_kwargs.get("model", "unknown"),
|
||||
)
|
||||
return True
|
||||
return False
|
||||
def _interrupt_or_superseded() -> bool:
|
||||
return bool(agent._interrupt_requested)
|
||||
|
||||
try:
|
||||
# Compatibility: some mocks/providers return a concrete response
|
||||
# instead of an iterable. Pass it straight through.
|
||||
if hasattr(event_stream, "output") and not hasattr(event_stream, "__iter__"):
|
||||
return event_stream
|
||||
|
||||
try:
|
||||
final = _consume_codex_event_stream(
|
||||
event_stream,
|
||||
@@ -1271,6 +1365,12 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
|
||||
on_event=_on_event,
|
||||
interrupt_check=_interrupt_or_superseded,
|
||||
)
|
||||
# The terminal SSE frame is contractually last. Request the
|
||||
# end-of-stream marker so Relay can run its response finalizer
|
||||
# and close the physical attempt scope before Hermes returns.
|
||||
if not agent._interrupt_requested:
|
||||
for _ignored in event_stream:
|
||||
pass
|
||||
except (_httpx.RemoteProtocolError, _httpx.ReadTimeout, _httpx.ConnectError, ConnectionError) as exc:
|
||||
if attempt < max_stream_retries:
|
||||
logger.debug(
|
||||
@@ -1281,6 +1381,10 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
|
||||
)
|
||||
continue
|
||||
raise
|
||||
except RuntimeError:
|
||||
if event_stream.final_response is not None:
|
||||
return event_stream.final_response
|
||||
raise
|
||||
|
||||
if final.status in {"incomplete", "failed"}:
|
||||
logger.warning(
|
||||
@@ -1298,7 +1402,20 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
|
||||
try:
|
||||
close_fn()
|
||||
except Exception:
|
||||
pass
|
||||
# A failed close can leave this response's connection
|
||||
# checked out of the httpx pool while the caller's finally
|
||||
# reports a reuse-reason close (e.g. interrupt_check broke
|
||||
# the event loop with collected output) — caching the
|
||||
# client with the leaked connection. Poison the slot so
|
||||
# that close really closes the pool (owner-thread abort;
|
||||
# mirrors the chat-streaming interrupt-break handling).
|
||||
# ``client is None`` means the shared primary client,
|
||||
# which is never reuse-cached and must not have its
|
||||
# sockets force-shut here.
|
||||
if client is not None:
|
||||
agent._abort_request_openai_client(
|
||||
active_client, reason="codex_stream_close_failed"
|
||||
)
|
||||
|
||||
|
||||
def run_codex_create_stream_fallback(agent, api_kwargs: dict, client: Any = None):
|
||||
|
||||
+48
-24
@@ -55,13 +55,12 @@ import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import tempfile
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
from hermes_cli._subprocess_compat import IS_WINDOWS, windows_hide_flags
|
||||
from hermes_cli._subprocess_compat import bounded_git_probe
|
||||
|
||||
logger = logging.getLogger("hermes.coding_context")
|
||||
|
||||
@@ -338,9 +337,9 @@ def _coding_mode(config: Optional[dict[str, Any]]) -> str:
|
||||
"""Return the normalized ``agent.coding_context`` mode (auto/focus/on/off)."""
|
||||
if config is None:
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
from hermes_cli.config import load_config_readonly
|
||||
|
||||
config = load_config()
|
||||
config = load_config_readonly()
|
||||
except Exception:
|
||||
config = {}
|
||||
raw = ((config or {}).get("agent", {}) or {}).get("coding_context", "auto")
|
||||
@@ -521,30 +520,46 @@ class RuntimeMode:
|
||||
return None
|
||||
return [self.profile.toolset, *_enabled_mcp_servers(config)]
|
||||
|
||||
def system_blocks(self) -> list[str]:
|
||||
"""Stable system-prompt blocks for this posture (brief + workspace).
|
||||
def system_prompt_parts(self) -> tuple[list[str], list[str], list[str]]:
|
||||
"""Return prefix, workspace, and trailing posture blocks separately.
|
||||
|
||||
The operating brief carries a model-family edit-format nudge appended
|
||||
to it (one cached string, not a separate block) so the model is steered
|
||||
toward the `patch` mode it handles best — see ``_edit_format_line``.
|
||||
|
||||
The three lists preserve the historical flat prompt order: the brief,
|
||||
the live workspace snapshot, then configured operator instructions.
|
||||
Prompt assembly can therefore put a cache boundary before the snapshot
|
||||
without changing the persisted system-prompt bytes.
|
||||
"""
|
||||
if not self.is_coding:
|
||||
return []
|
||||
blocks: list[str] = []
|
||||
return [], [], []
|
||||
prefix: list[str] = []
|
||||
workspace_parts: list[str] = []
|
||||
trailing: list[str] = []
|
||||
if self.profile.guidance:
|
||||
brief = self.profile.guidance
|
||||
edit_line = _edit_format_line(self.model)
|
||||
if edit_line:
|
||||
brief = f"{brief}\n{edit_line}"
|
||||
blocks.append(brief)
|
||||
prefix.append(brief)
|
||||
workspace = build_coding_workspace_block(self.cwd)
|
||||
if workspace:
|
||||
blocks.append(workspace)
|
||||
workspace_parts.append(workspace)
|
||||
# Operator instructions ride their own block so the brief (block 0) stays
|
||||
# byte-stable and cache-keyed independently of user config.
|
||||
if self.instructions:
|
||||
blocks.append(f"Operator instructions (from config):\n{self.instructions}")
|
||||
return blocks
|
||||
trailing.append(f"Operator instructions (from config):\n{self.instructions}")
|
||||
return prefix, workspace_parts, trailing
|
||||
|
||||
def system_blocks(self) -> list[str]:
|
||||
"""Return posture blocks in their historical display order.
|
||||
|
||||
``system_prompt_parts`` is the cache-aware API. This compatibility
|
||||
helper retains the public flat list for callers outside prompt assembly.
|
||||
"""
|
||||
prefix, workspace, trailing = self.system_prompt_parts()
|
||||
return [*prefix, *workspace, *trailing]
|
||||
|
||||
def compact_skill_categories(self) -> frozenset[str]:
|
||||
"""Skill categories to demote to names-only in the prompt's skill index.
|
||||
@@ -645,6 +660,19 @@ def coding_system_blocks(
|
||||
).system_blocks()
|
||||
|
||||
|
||||
def coding_system_prompt_parts(
|
||||
*,
|
||||
platform: Optional[str] = None,
|
||||
cwd: Optional[str | Path] = None,
|
||||
config: Optional[dict[str, Any]] = None,
|
||||
model: Optional[str] = None,
|
||||
) -> tuple[list[str], list[str], list[str]]:
|
||||
"""Return coding prefix, workspace snapshot, and trailing guidance."""
|
||||
return resolve_runtime_mode(
|
||||
platform=platform, cwd=cwd, config=config, model=model
|
||||
).system_prompt_parts()
|
||||
|
||||
|
||||
def coding_compact_skill_categories(
|
||||
*,
|
||||
platform: Optional[str] = None,
|
||||
@@ -689,18 +717,14 @@ def _enabled_mcp_servers(config: Optional[dict[str, Any]]) -> list[str]:
|
||||
|
||||
|
||||
def _git(cwd: Path, *args: str) -> str:
|
||||
_popen_kwargs = {"creationflags": windows_hide_flags()} if IS_WINDOWS else {}
|
||||
try:
|
||||
out = subprocess.run(
|
||||
["git", "-C", str(cwd), *args],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=_GIT_TIMEOUT,
|
||||
**_popen_kwargs,
|
||||
)
|
||||
except (OSError, subprocess.SubprocessError):
|
||||
return ""
|
||||
return out.stdout.strip() if out.returncode == 0 else ""
|
||||
"""``git -C <cwd> <args>`` → stripped stdout, or ``""`` on any failure.
|
||||
|
||||
Uses the shared :func:`bounded_git_probe` so the post-kill cleanup is bounded
|
||||
on Windows — a plain ``subprocess.run(timeout=...)`` here deadlocked the agent
|
||||
turn inside ``build_coding_workspace_block`` when a killed git left a suspended
|
||||
descendant holding the pipe handles (issue #66037).
|
||||
"""
|
||||
return bounded_git_probe(["git", "-C", str(cwd), *args], timeout=_GIT_TIMEOUT)
|
||||
|
||||
|
||||
def _parse_status(porcelain: str) -> tuple[dict[str, str], dict[str, int]]:
|
||||
|
||||
@@ -154,3 +154,207 @@ def compute_session_context_breakdown(
|
||||
"estimated_total": estimated_total,
|
||||
"model": getattr(agent, "model", "") or "",
|
||||
}
|
||||
|
||||
|
||||
# ── /context rendering (CLI + gateway) ──────────────────────────────────────
|
||||
#
|
||||
# Pure text renderers over the payload above. The CLI shows a glyph block-grid
|
||||
# plus a category table; the gateway uses the same table without the grid
|
||||
# (proportional monospace is not guaranteed on messaging platforms).
|
||||
|
||||
_CATEGORY_GLYPHS = {
|
||||
"system_prompt": "■",
|
||||
"tool_definitions": "▣",
|
||||
"rules": "▩",
|
||||
"skills": "▤",
|
||||
"mcp": "▥",
|
||||
"subagent_definitions": "▦",
|
||||
"memory": "▧",
|
||||
"conversation": "▨",
|
||||
}
|
||||
_FREE_GLYPH = "·"
|
||||
_GRID_COLUMNS = 20
|
||||
_GRID_ROWS = 5 # 100 cells → 1 cell per percent of the context window
|
||||
|
||||
# Human-readable tables cap the expanded listings; nothing is dropped from
|
||||
# the underlying data.
|
||||
_DETAILS_TABLE_LIMIT = 15
|
||||
|
||||
|
||||
def _bytes_to_tokens(size: Optional[int]) -> Optional[int]:
|
||||
if size is None:
|
||||
return None
|
||||
return (int(size) + 3) // 4
|
||||
|
||||
|
||||
def compute_context_details(agent: Any) -> Dict[str, Any]:
|
||||
"""Expanded per-skill / per-toolset cost listing for ``/context all``.
|
||||
|
||||
Reuses the ``hermes prompt-size`` attribution mechanism (PR #66656):
|
||||
per-skill index-line bytes parsed from the live ``<available_skills>``
|
||||
block, and per-toolset schema bytes attributed via the tool registry's
|
||||
canonical tool→toolset map. Byte figures are converted to the same
|
||||
chars/4 token heuristic the categories above use.
|
||||
"""
|
||||
from hermes_cli.prompt_size import (
|
||||
_compute_skills_breakdown,
|
||||
_compute_toolsets_breakdown,
|
||||
)
|
||||
from agent.system_prompt import build_system_prompt_parts
|
||||
|
||||
parts = build_system_prompt_parts(agent)
|
||||
stable = parts.get("stable", "") or ""
|
||||
skills_match = _SKILLS_BLOCK_RE.search(stable)
|
||||
skills_block = skills_match.group(0) if skills_match else ""
|
||||
|
||||
skills: List[Dict[str, Any]] = []
|
||||
if skills_block:
|
||||
for entry in _compute_skills_breakdown(skills_block):
|
||||
skills.append({
|
||||
"name": entry.get("name", ""),
|
||||
"index_tokens": _bytes_to_tokens(entry.get("index_line_bytes")) or 0,
|
||||
"skill_md_tokens": _bytes_to_tokens(entry.get("skill_md_bytes")),
|
||||
})
|
||||
|
||||
toolsets: List[Dict[str, Any]] = []
|
||||
tools = list(getattr(agent, "tools", None) or [])
|
||||
if tools:
|
||||
for group in _compute_toolsets_breakdown(tools):
|
||||
toolsets.append({
|
||||
"toolset": group.get("toolset", ""),
|
||||
"tool_count": int(group.get("tool_count", 0) or 0),
|
||||
"schema_tokens": _bytes_to_tokens(group.get("json_bytes")) or 0,
|
||||
})
|
||||
|
||||
return {"skills": skills, "toolsets": toolsets}
|
||||
|
||||
|
||||
def render_context_grid(payload: Dict[str, Any]) -> List[str]:
|
||||
"""Render the payload as a Claude Code-style glyph block grid.
|
||||
|
||||
100 cells (5×20), each one percent of the model context window. Categories
|
||||
fill in declaration order; the remainder renders as free space.
|
||||
"""
|
||||
context_max = int(payload.get("context_max") or 0)
|
||||
categories = payload.get("categories") or []
|
||||
total_cells = _GRID_COLUMNS * _GRID_ROWS
|
||||
|
||||
cells: List[str] = []
|
||||
if context_max > 0:
|
||||
for cat in categories:
|
||||
tokens = int(cat.get("tokens") or 0)
|
||||
n = round(tokens / context_max * total_cells)
|
||||
if tokens > 0 and n == 0:
|
||||
n = 1 # never render a nonzero category as invisible
|
||||
glyph = _CATEGORY_GLYPHS.get(str(cat.get("id") or ""), "▪")
|
||||
cells.extend([glyph] * n)
|
||||
cells = cells[:total_cells]
|
||||
cells.extend([_FREE_GLYPH] * (total_cells - len(cells)))
|
||||
|
||||
return [
|
||||
" ".join(cells[row * _GRID_COLUMNS:(row + 1) * _GRID_COLUMNS])
|
||||
for row in range(_GRID_ROWS)
|
||||
]
|
||||
|
||||
|
||||
def render_context_category_lines(payload: Dict[str, Any]) -> List[str]:
|
||||
"""Render the 'Estimated usage by category' table as plain-text lines."""
|
||||
categories = payload.get("categories") or []
|
||||
context_max = int(payload.get("context_max") or 0)
|
||||
estimated_total = int(payload.get("estimated_total") or 0)
|
||||
denom = context_max or estimated_total
|
||||
|
||||
lines = ["Estimated usage by category"]
|
||||
if not categories:
|
||||
lines.append(" (no data yet — send a message first)")
|
||||
return lines
|
||||
|
||||
width = max(len(str(cat.get("label") or "")) for cat in categories)
|
||||
width = max(width, len("Free space"))
|
||||
for cat in categories:
|
||||
tokens = int(cat.get("tokens") or 0)
|
||||
glyph = _CATEGORY_GLYPHS.get(str(cat.get("id") or ""), "▪")
|
||||
pct = tokens / denom * 100 if denom else 0.0
|
||||
label = str(cat.get("label") or cat.get("id") or "")
|
||||
lines.append(f"{glyph} {label:<{width}} {tokens:>9,} tokens {pct:>5.1f}%")
|
||||
if context_max > 0:
|
||||
free = max(0, context_max - estimated_total)
|
||||
pct = free / context_max * 100
|
||||
lines.append(f"{_FREE_GLYPH} {'Free space':<{width}} {free:>9,} tokens {pct:>5.1f}%")
|
||||
return lines
|
||||
|
||||
|
||||
def render_context_details_lines(details: Dict[str, Any]) -> List[str]:
|
||||
"""Render the expanded ``/context all`` per-skill / per-toolset tables."""
|
||||
lines: List[str] = []
|
||||
|
||||
toolsets = details.get("toolsets") or []
|
||||
if toolsets:
|
||||
lines.append("Toolsets by schema cost (largest first)")
|
||||
for group in toolsets[:_DETAILS_TABLE_LIMIT]:
|
||||
lines.append(
|
||||
f" {group['toolset']:<24} {group['tool_count']:>3} tools"
|
||||
f" {group['schema_tokens']:>8,} tokens"
|
||||
)
|
||||
remaining = len(toolsets) - _DETAILS_TABLE_LIMIT
|
||||
if remaining > 0:
|
||||
lines.append(f" … and {remaining} more")
|
||||
|
||||
skills = details.get("skills") or []
|
||||
if skills:
|
||||
if lines:
|
||||
lines.append("")
|
||||
lines.append("Skills by cost (index = always-on; SKILL.md = cost when loaded)")
|
||||
for entry in skills[:_DETAILS_TABLE_LIMIT]:
|
||||
name = str(entry.get("name") or "")
|
||||
if len(name) > 28:
|
||||
name = name[:27] + "…"
|
||||
md = entry.get("skill_md_tokens")
|
||||
md_str = f"{md:>8,}" if md is not None else f"{'n/a':>8}"
|
||||
lines.append(
|
||||
f" {name:<28} index {entry['index_tokens']:>6,}"
|
||||
f" SKILL.md {md_str} tokens"
|
||||
)
|
||||
remaining = len(skills) - _DETAILS_TABLE_LIMIT
|
||||
if remaining > 0:
|
||||
lines.append(f" … and {remaining} more")
|
||||
|
||||
return lines
|
||||
|
||||
|
||||
def render_context_breakdown_lines(
|
||||
payload: Dict[str, Any],
|
||||
*,
|
||||
details: Optional[Dict[str, Any]] = None,
|
||||
grid: bool = True,
|
||||
) -> List[str]:
|
||||
"""Render the full /context view as plain-text lines.
|
||||
|
||||
``grid=True`` (CLI) prepends the glyph block grid; the gateway passes
|
||||
``grid=False`` and keeps its own gauge. ``details`` (from
|
||||
:func:`compute_context_details`) appends the expanded listings.
|
||||
"""
|
||||
lines: List[str] = []
|
||||
if grid:
|
||||
lines.extend(render_context_grid(payload))
|
||||
lines.append("")
|
||||
lines.extend(render_context_category_lines(payload))
|
||||
|
||||
context_max = int(payload.get("context_max") or 0)
|
||||
context_used = int(payload.get("context_used") or 0)
|
||||
if context_max > 0:
|
||||
pct = int(payload.get("context_percent") or 0)
|
||||
lines.append("")
|
||||
lines.append(
|
||||
f"Context window: {context_used:,} / {context_max:,} tokens ({pct}%)"
|
||||
)
|
||||
|
||||
if details is not None:
|
||||
detail_lines = render_context_details_lines(details)
|
||||
if detail_lines:
|
||||
lines.append("")
|
||||
lines.extend(detail_lines)
|
||||
else:
|
||||
lines.append("")
|
||||
lines.append("Use /context all for per-skill and per-toolset costs.")
|
||||
return lines
|
||||
|
||||
+3246
-275
File diff suppressed because it is too large
Load Diff
+261
-3
@@ -26,7 +26,64 @@ Lifecycle:
|
||||
"""
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from typing import Any, Dict, List
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from agent.redact import redact_sensitive_text
|
||||
|
||||
|
||||
MEMORY_CONTEXT_MAX_CHARS = 6_000
|
||||
_MEMORY_CONTEXT_HEAD_CHARS = 4_000
|
||||
_MEMORY_CONTEXT_TAIL_CHARS = 1_500
|
||||
_MEMORY_CONTEXT_TRUNCATION_MARKER = "\n...[memory provider context truncated]...\n"
|
||||
|
||||
|
||||
def sanitize_memory_context(memory_context: str) -> str:
|
||||
"""Prepare provider context for a context-engine/LLM egress boundary."""
|
||||
sanitized = redact_sensitive_text(
|
||||
memory_context.strip(),
|
||||
force=True,
|
||||
redact_url_credentials=True,
|
||||
)
|
||||
if len(sanitized) <= MEMORY_CONTEXT_MAX_CHARS:
|
||||
return sanitized
|
||||
return (
|
||||
sanitized[:_MEMORY_CONTEXT_HEAD_CHARS]
|
||||
+ _MEMORY_CONTEXT_TRUNCATION_MARKER
|
||||
+ sanitized[-_MEMORY_CONTEXT_TAIL_CHARS:]
|
||||
)
|
||||
|
||||
|
||||
def automatic_compaction_status_message(
|
||||
engine: Any,
|
||||
*,
|
||||
phase: str,
|
||||
default_message: str,
|
||||
**context: Any,
|
||||
) -> str | None:
|
||||
"""Resolve host-visible status for an automatic compaction event.
|
||||
|
||||
Engines can suppress routine automatic status with
|
||||
``emit_automatic_compaction_status = False`` or customize it by defining
|
||||
``get_automatic_compaction_status_message(...)``. Empty strings and
|
||||
``None`` mean "do not emit a lifecycle status".
|
||||
"""
|
||||
if not getattr(engine, "emit_automatic_compaction_status", True):
|
||||
return None
|
||||
|
||||
formatter = getattr(engine, "get_automatic_compaction_status_message", None)
|
||||
if callable(formatter):
|
||||
message = formatter(
|
||||
phase=phase,
|
||||
default_message=default_message,
|
||||
**context,
|
||||
)
|
||||
else:
|
||||
message = default_message
|
||||
|
||||
if message is None:
|
||||
return None
|
||||
message = str(message).strip()
|
||||
return message or None
|
||||
|
||||
|
||||
class ContextEngine(ABC):
|
||||
@@ -65,6 +122,12 @@ class ContextEngine(ABC):
|
||||
protect_first_n: int = 3
|
||||
protect_last_n: int = 6
|
||||
|
||||
# User-visible lifecycle status for automatic host-triggered compaction.
|
||||
# Alternative engines that treat compaction as routine background
|
||||
# maintenance can set this false to keep successful automatic passes silent;
|
||||
# warnings, errors, and explicit manual commands should still surface.
|
||||
emit_automatic_compaction_status: bool = True
|
||||
|
||||
# -- Core interface ----------------------------------------------------
|
||||
|
||||
@abstractmethod
|
||||
@@ -83,12 +146,27 @@ class ContextEngine(ABC):
|
||||
def should_compress(self, prompt_tokens: int = None) -> bool:
|
||||
"""Return True if compaction should fire this turn."""
|
||||
|
||||
def should_compress_info(self, prompt_tokens: int = None) -> "tuple[bool, str | None]":
|
||||
"""Return ``(should_compress, reason)``.
|
||||
|
||||
The base implementation is backward-compatible: engines that only
|
||||
implement ``should_compress`` get ``(should_compress(prompt_tokens),
|
||||
None)``. Concrete engines with richer block reasons (e.g. a
|
||||
summary-LLM cooldown or an anti-thrashing guard) override this to
|
||||
surface a human-readable reason so callers can warn the user instead
|
||||
of silently skipping compression. Added for the silent-overflow
|
||||
warning fix (#62625) so plugin engines don't raise AttributeError.
|
||||
"""
|
||||
return self.should_compress(prompt_tokens), None
|
||||
|
||||
@abstractmethod
|
||||
def compress(
|
||||
self,
|
||||
messages: List[Dict[str, Any]],
|
||||
current_tokens: int = None,
|
||||
focus_topic: str = None,
|
||||
current_tokens: Optional[int] = None,
|
||||
focus_topic: Optional[str] = None,
|
||||
force: bool = False,
|
||||
memory_context: str = "",
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Compact the message list and return the new message list.
|
||||
|
||||
@@ -103,8 +181,152 @@ class ContextEngine(ABC):
|
||||
Engines that support guided compression should prioritise
|
||||
preserving information related to this topic. Engines that
|
||||
don't support it may simply ignore this argument.
|
||||
force: Whether a user-requested compression should bypass an
|
||||
engine-owned cooldown. Engines without cooldowns may ignore it.
|
||||
memory_context: Text returned by memory providers immediately before
|
||||
compaction. Summarizing engines should include non-empty text in
|
||||
their handoff prompt. Older engines may omit this parameter; the
|
||||
host filters unsupported optional arguments by signature.
|
||||
"""
|
||||
|
||||
# -- Optional: proactive tool-result prune -----------------------------
|
||||
|
||||
def prune_tool_results_only(
|
||||
self,
|
||||
messages: List[Dict[str, Any]],
|
||||
current_tokens: int | None = None,
|
||||
) -> tuple[List[Dict[str, Any]], int]:
|
||||
"""Deterministically trim old tool-result payloads without an LLM call.
|
||||
|
||||
Runs on a low, cost-oriented trigger independent of ``should_compress``
|
||||
so large-window engines can reclaim re-sent tool output long before full
|
||||
compaction would fire. Returns ``(messages, n_pruned)``.
|
||||
|
||||
Default is a safe no-op: the list is returned unchanged with ``0``
|
||||
pruned. Engines that don't implement a cheap prune — and any engine that
|
||||
predates this hook — inherit this default, so the agent loop's
|
||||
post-tool-call prune path never raises ``AttributeError`` on them. The
|
||||
built-in ContextCompressor overrides this with the real implementation.
|
||||
"""
|
||||
return messages, 0
|
||||
|
||||
# -- Optional: per-turn context selection (distinct from compression) --
|
||||
|
||||
def select_context(
|
||||
self,
|
||||
request_messages: List[Dict[str, Any]],
|
||||
*,
|
||||
conversation_messages: List[Dict[str, Any]] = None,
|
||||
incoming_message: Dict[str, Any] = None,
|
||||
budget_tokens: int = 0,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Optionally choose/replace the context for THIS request, pre-generation.
|
||||
|
||||
Called every turn after the request message list is assembled and
|
||||
before it is dispatched to the provider — independent of
|
||||
``should_compress()``. This lets an engine *select* which context
|
||||
enters the prompt (retrieval, topic routing, role/branch switching)
|
||||
rather than *shrink* context that is already there. The two verbs are
|
||||
orthogonal:
|
||||
|
||||
- ``compress()`` : context is too long -> make it shorter.
|
||||
- ``select_context()``: this turn belongs to a different context
|
||||
-> use that one instead.
|
||||
|
||||
Without this hook, engines that need per-turn access to the message
|
||||
list have to force ``should_compress()`` to return ``True`` so that
|
||||
``compress()`` is invoked every turn purely as a callback — which
|
||||
conflates selection with compression and degrades behaviour when the
|
||||
engine's backend is unavailable. ``select_context()`` removes the need
|
||||
for that workaround.
|
||||
|
||||
The returned list is request-only: it replaces the messages sent to
|
||||
the provider for this single call and MUST NOT be treated as persisted
|
||||
transcript state. The conversation history in the session DB is left
|
||||
untouched, so nothing leaks across turns. Return ``None`` to leave the
|
||||
request unchanged.
|
||||
|
||||
Unlike the ``pre_llm_call`` plugin hook (which appends to the user
|
||||
message and intentionally never rewrites the list, to preserve the
|
||||
cache prefix), ``select_context()`` may *replace* the message list.
|
||||
|
||||
Ordering / cache contract: the host runs this hook **before** prompt
|
||||
cache-control and **before** every request sanitizer (orphaned-tool
|
||||
cleanup, thinking-only/role normalization, whitespace/JSON
|
||||
normalization). So (a) whatever the hook returns still passes through
|
||||
the same validation as any request — a malformed replacement cannot
|
||||
reach the provider — and (b) prompt-cache stability (an AGENTS.md
|
||||
invariant) is preserved: the default no-op leaves the request
|
||||
byte-identical, so cache behaviour is unchanged for the built-in
|
||||
compressor and any non-implementing engine. An engine that *does*
|
||||
replace the list changes its own cache prefix by definition; that is
|
||||
the engine's concern, and cache-control breakpoints are re-derived on
|
||||
the selected list. The hook is evaluated per provider request (so it
|
||||
re-runs on retries within a turn), consistent with "select the context
|
||||
for THIS request".
|
||||
|
||||
Args:
|
||||
request_messages: The assembled request message list (system
|
||||
prompt + history + any ephemeral prefill), in OpenAI format.
|
||||
conversation_messages: The unmodified persisted conversation
|
||||
history, for reference only (do not mutate).
|
||||
incoming_message: The current turn's user message, if available.
|
||||
budget_tokens: The active model's context length, or 0 if unknown.
|
||||
|
||||
Default returns ``None`` (no-op) — zero impact on the built-in
|
||||
compressor or any existing engine.
|
||||
"""
|
||||
return None
|
||||
|
||||
def on_turn_complete(
|
||||
self,
|
||||
messages: List[Dict[str, Any]],
|
||||
usage: Dict[str, Any] = None,
|
||||
**kwargs: Any,
|
||||
) -> None:
|
||||
"""Observe a finished user turn (post-turn ingestion / observation).
|
||||
|
||||
Called from the standard turn-finalization path once the assistant/tool
|
||||
loop completes, with the finalized in-memory transcript snapshot. This
|
||||
is the complement to ``select_context()``: selection happens *before*
|
||||
the request, while observation happens *after* the turn. It lets an
|
||||
engine ingest, index, summarize, or update routing / topic / session
|
||||
state from what actually happened — so the next ``select_context()``
|
||||
can act on it.
|
||||
|
||||
Coverage: this fires from the normal finalization seam. Some abnormal
|
||||
early-return paths in the loop (e.g. a content-policy block or a
|
||||
provider terminal failure) persist and return without routing through
|
||||
finalization, and therefore do not currently emit this hook. Treat it
|
||||
as a best-effort post-turn observation for completed turns, not a
|
||||
guaranteed callback for every possible early exit; unifying all
|
||||
terminal paths behind one finalization seam is a separate follow-up.
|
||||
|
||||
Together the two hooks remove the need to abuse ``should_compress()`` /
|
||||
``compress()`` as a generic per-turn callback just to observe history,
|
||||
and they cover the case where a turn finishes and there may be no next
|
||||
request from which to infer the previous turn.
|
||||
|
||||
``messages`` is a shallow copy and should be treated as read-only:
|
||||
return values are ignored and this hook must not rely on transcript
|
||||
mutation for persistence. ``kwargs`` may include ``turn_id``,
|
||||
``task_id``, ``api_call_count``, ``interrupted``, ``failed``, and
|
||||
``turn_exit_reason``.
|
||||
|
||||
``usage`` carries the completed turn's canonical token usage (the same
|
||||
dict shape passed to ``update_from_response`` — ``prompt_tokens`` /
|
||||
``completion_tokens`` / ``total_tokens`` plus the canonical
|
||||
``input_tokens`` / ``output_tokens`` / ``cache_read_tokens`` /
|
||||
``cache_write_tokens`` / ``reasoning_tokens`` buckets) so an engine can
|
||||
weigh how large/expensive the selected context actually was when
|
||||
deciding the next ``select_context()``. It is ``None`` on finalized
|
||||
turns that never reached a provider response (e.g. interrupt); engines
|
||||
must treat it as optional.
|
||||
|
||||
Default is a no-op.
|
||||
"""
|
||||
return None
|
||||
|
||||
# -- Optional: pre-flight check ----------------------------------------
|
||||
|
||||
def should_compress_preflight(self, messages: List[Dict[str, Any]]) -> bool:
|
||||
@@ -124,6 +346,27 @@ class ContextEngine(ABC):
|
||||
"""
|
||||
return False
|
||||
|
||||
def get_automatic_compaction_status_message(
|
||||
self,
|
||||
*,
|
||||
phase: str,
|
||||
default_message: str,
|
||||
**context: Any,
|
||||
) -> str | None:
|
||||
"""Return user-visible status for automatic host-triggered compaction.
|
||||
|
||||
Return ``None`` to suppress successful automatic lifecycle status for
|
||||
this compaction event. ``phase`` identifies the host call site (for
|
||||
example ``"preflight"`` or ``"compress"``). ``context`` contains
|
||||
best-effort fields such as ``approx_tokens`` and ``threshold_tokens``.
|
||||
|
||||
This hook does not control warning/error messages or explicit manual
|
||||
commands such as ``/compress``.
|
||||
"""
|
||||
if not self.emit_automatic_compaction_status:
|
||||
return None
|
||||
return default_message
|
||||
|
||||
# -- Optional: manual /compress preflight ------------------------------
|
||||
|
||||
def has_content_to_compress(self, messages: List[Dict[str, Any]]) -> bool:
|
||||
@@ -228,4 +471,19 @@ class ContextEngine(ABC):
|
||||
(e.g. recalculate DAG budgets, switch summary models).
|
||||
"""
|
||||
self.context_length = context_length
|
||||
# Apply per-model threshold overrides if set (longest substring match).
|
||||
# Falls back to _config_threshold_percent (the raw config value) when
|
||||
# no override matches. Plugin engines that override update_model() can
|
||||
# call resolve_model_threshold() for the same logic.
|
||||
from agent.context_compressor import resolve_model_threshold
|
||||
if not hasattr(self, "_config_threshold_percent"):
|
||||
# Snapshot the pre-override percent ONCE so repeated model
|
||||
# switches fall back to the engine's configured value, not the
|
||||
# previous model's override.
|
||||
self._config_threshold_percent = self.threshold_percent
|
||||
self._base_threshold_percent = resolve_model_threshold(
|
||||
model, getattr(self, "model_thresholds", {}),
|
||||
self._config_threshold_percent,
|
||||
)
|
||||
self.threshold_percent = self._base_threshold_percent
|
||||
self.threshold_tokens = int(context_length * self.threshold_percent)
|
||||
|
||||
+24
-17
@@ -19,6 +19,7 @@ REFERENCE_PATTERN = re.compile(
|
||||
rf"(?<![\w/])@(?:(?P<simple>diff|staged)\b|(?P<kind>file|folder|git|url):(?P<value>{_QUOTED_REFERENCE_VALUE}(?::\d+(?:-\d+)?)?|\S+))"
|
||||
)
|
||||
TRAILING_PUNCTUATION = ",.;!?"
|
||||
_NEEDS_QUOTING = re.compile(r"""[\s()\[\]{}<>"'`]""")
|
||||
_SENSITIVE_HOME_DIRS = (".ssh", ".aws", ".gnupg", ".kube", ".docker", ".azure", ".config/gh")
|
||||
_SENSITIVE_HERMES_DIRS = (Path("skills") / ".hub",)
|
||||
_SENSITIVE_HOME_FILES = (
|
||||
@@ -60,6 +61,21 @@ class ContextReferenceResult:
|
||||
blocked: bool = False
|
||||
|
||||
|
||||
def format_reference_value(value: str) -> str:
|
||||
"""Quote a reference value so ``REFERENCE_PATTERN`` reads it back whole.
|
||||
|
||||
The unquoted alternative in the pattern is ``\\S+``, so a path containing a
|
||||
space parses as a truncated ref with the tail left behind as loose text.
|
||||
Mirrors ``formatRefValue`` in the desktop's directive-text.tsx.
|
||||
"""
|
||||
if not _NEEDS_QUOTING.search(value):
|
||||
return value
|
||||
for quote in ("`", '"', "'"):
|
||||
if quote not in value:
|
||||
return f"{quote}{value}{quote}"
|
||||
return value
|
||||
|
||||
|
||||
def parse_context_references(message: str) -> list[ContextReference]:
|
||||
refs: list[ContextReference] = []
|
||||
if not message:
|
||||
@@ -197,8 +213,12 @@ async def preprocess_context_references_async(
|
||||
f"@ context injection warning: {injected_tokens} tokens exceeds the 25% soft limit ({soft_limit})."
|
||||
)
|
||||
|
||||
stripped = _remove_reference_tokens(message, refs)
|
||||
final = stripped
|
||||
# Leave the `@file:`/`@folder:` tokens where the user typed them. The token
|
||||
# IS the reference, not scaffolding around it: clients render each one as an
|
||||
# inline chip, so stripping them left a sentence with a hole in it ("review
|
||||
# and ship") and made the desktop re-derive the refs from the attached block
|
||||
# to show them as a detached list above the prose.
|
||||
final = message
|
||||
if warnings:
|
||||
final = f"{final}\n\n--- Context Warnings ---\n" + "\n".join(f"- {warning}" for warning in warnings)
|
||||
if blocks:
|
||||
@@ -308,7 +328,7 @@ def _expand_git_reference(
|
||||
["git", *args],
|
||||
cwd=cwd,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
timeout=30,
|
||||
stdin=subprocess.DEVNULL,
|
||||
**_popen_kwargs,
|
||||
@@ -457,19 +477,6 @@ def _parse_file_reference_value(value: str) -> tuple[str, int | None, int | None
|
||||
return _strip_reference_wrappers(value), None, None
|
||||
|
||||
|
||||
def _remove_reference_tokens(message: str, refs: list[ContextReference]) -> str:
|
||||
pieces: list[str] = []
|
||||
cursor = 0
|
||||
for ref in refs:
|
||||
pieces.append(message[cursor:ref.start])
|
||||
cursor = ref.end
|
||||
pieces.append(message[cursor:])
|
||||
text = "".join(pieces)
|
||||
text = re.sub(r"\s{2,}", " ", text)
|
||||
text = re.sub(r"\s+([,.;:!?])", r"\1", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
def _is_binary_file(path: Path) -> bool:
|
||||
mime, _ = mimetypes.guess_type(path.name)
|
||||
if mime and not mime.startswith("text/") and not any(
|
||||
@@ -534,7 +541,7 @@ def _rg_files(path: Path, cwd: Path, limit: int) -> list[Path] | None:
|
||||
["rg", "--files", str(path.relative_to(cwd))],
|
||||
cwd=cwd,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
timeout=10,
|
||||
stdin=subprocess.DEVNULL,
|
||||
**_popen_kwargs,
|
||||
|
||||
+1558
-218
File diff suppressed because it is too large
Load Diff
+1512
-151
File diff suppressed because it is too large
Load Diff
@@ -503,15 +503,20 @@ class CopilotACPClient:
|
||||
|
||||
def _run_prompt(self, prompt_text: str, *, timeout_seconds: float) -> tuple[str, str]:
|
||||
try:
|
||||
# Hide the console the CLI child would otherwise flash on Windows
|
||||
# (#56747). Hide-only — stdio pipes stay intact for the ACP wire.
|
||||
from hermes_cli._subprocess_compat import windows_hide_flags
|
||||
|
||||
proc = subprocess.Popen(
|
||||
[self._acp_command] + self._acp_args,
|
||||
stdin=subprocess.PIPE,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
text=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
bufsize=1,
|
||||
cwd=self._acp_cwd,
|
||||
env=_build_subprocess_env(),
|
||||
creationflags=windows_hide_flags(),
|
||||
)
|
||||
except FileNotFoundError as exc:
|
||||
raise RuntimeError(
|
||||
@@ -703,7 +708,7 @@ class CopilotACPClient:
|
||||
if block_error:
|
||||
raise PermissionError(block_error)
|
||||
try:
|
||||
content = path.read_text()
|
||||
content = path.read_text(encoding="utf-8")
|
||||
except FileNotFoundError:
|
||||
content = ""
|
||||
line = params.get("line")
|
||||
@@ -731,7 +736,7 @@ class CopilotACPClient:
|
||||
if denied:
|
||||
raise PermissionError(denied)
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(str(params.get("content") or ""))
|
||||
path.write_text(str(params.get("content") or ""), encoding="utf-8")
|
||||
response = {
|
||||
"jsonrpc": "2.0",
|
||||
"id": message_id,
|
||||
|
||||
+290
-85
@@ -594,22 +594,64 @@ class CredentialPool:
|
||||
# Re-armed to None on every successful selection so a recover→re-exhaust
|
||||
# transition logs promptly instead of being swallowed by a stale window.
|
||||
self._last_no_entries_log_at: Optional[float] = None
|
||||
# #70401: consecutive mark_exhausted_and_rotate() calls whose supplied
|
||||
# credential identity matched no pool entry (OAuth wrappers whose
|
||||
# runtime key rotates, entries pruned by another process, ...). These
|
||||
# rotations mark nothing exhausted, so without a cap the pool can
|
||||
# never converge to "no available entries" and the caller's 401 retry
|
||||
# loop runs unbounded and non-interruptible. Reset whenever a real
|
||||
# entry is identified or an escape path returns None.
|
||||
self._unmatched_rotation_streak: int = 0
|
||||
|
||||
def has_credentials(self) -> bool:
|
||||
return bool(self._entries)
|
||||
with self._lock:
|
||||
return bool(self._entries)
|
||||
|
||||
def has_available(self) -> bool:
|
||||
"""True if at least one entry is not currently in exhaustion cooldown."""
|
||||
return bool(self._available_entries())
|
||||
# ``_available_entries`` is not read-only: it prunes aged-out DEAD
|
||||
# manual entries (rebinding ``self._entries``) and persists. It must
|
||||
# run under ``self._lock`` like every other caller (``select`` etc.),
|
||||
# otherwise a status probe here can race a concurrent ``select`` /
|
||||
# rotation and tear ``self._entries`` or double-write auth.json.
|
||||
with self._lock:
|
||||
return bool(self._available_entries())
|
||||
|
||||
def entries(self) -> List[PooledCredential]:
|
||||
return list(self._entries)
|
||||
with self._lock:
|
||||
return list(self._entries)
|
||||
|
||||
def current(self) -> Optional[PooledCredential]:
|
||||
def _current_unlocked(self) -> Optional[PooledCredential]:
|
||||
if not self._current_id:
|
||||
return None
|
||||
return next((entry for entry in self._entries if entry.id == self._current_id), None)
|
||||
|
||||
def current(self) -> Optional[PooledCredential]:
|
||||
with self._lock:
|
||||
return self._current_unlocked()
|
||||
|
||||
def entry_id_for_api_key(self, api_key_hint: Any = None) -> Optional[str]:
|
||||
"""Return the stable id for the runtime credential in use.
|
||||
|
||||
Prefer the current selection when it still supplies ``api_key_hint``.
|
||||
If the cursor was cleared, fall back to an unambiguous key match.
|
||||
"""
|
||||
with self._lock:
|
||||
current = self._current_unlocked()
|
||||
if current is not None and (
|
||||
api_key_hint is None
|
||||
or current.runtime_api_key == api_key_hint
|
||||
):
|
||||
return current.id
|
||||
if api_key_hint is None:
|
||||
return None
|
||||
matches = [
|
||||
entry
|
||||
for entry in self._entries
|
||||
if entry.runtime_api_key == api_key_hint
|
||||
]
|
||||
return matches[0].id if len(matches) == 1 else None
|
||||
|
||||
def _replace_entry(self, old: PooledCredential, new: PooledCredential) -> None:
|
||||
"""Swap an entry in-place by id, preserving sort order."""
|
||||
for idx, entry in enumerate(self._entries):
|
||||
@@ -652,6 +694,8 @@ class CredentialPool:
|
||||
entry: PooledCredential,
|
||||
status_code: Optional[int],
|
||||
error_context: Optional[Dict[str, Any]] = None,
|
||||
*,
|
||||
persist: bool = True,
|
||||
) -> PooledCredential:
|
||||
normalized_error = _normalize_error_context(error_context)
|
||||
# Permanent OAuth failures (token_invalidated, token_revoked, etc.)
|
||||
@@ -675,7 +719,8 @@ class CredentialPool:
|
||||
last_error_reset_at=normalized_error.get("reset_at"),
|
||||
)
|
||||
self._replace_entry(entry, updated)
|
||||
self._persist()
|
||||
if persist:
|
||||
self._persist()
|
||||
return updated
|
||||
|
||||
def _sync_anthropic_entry_from_credentials_file(self, entry: PooledCredential) -> PooledCredential:
|
||||
@@ -1484,6 +1529,43 @@ class CredentialPool:
|
||||
self._sync_device_code_entry_to_auth_store(updated)
|
||||
return updated
|
||||
|
||||
def _codex_quota_restored_upstream(self, entry: PooledCredential) -> bool:
|
||||
"""Live-check whether an exhausted Codex entry's quota reset early.
|
||||
|
||||
A Codex 429 persists a ``last_error_reset_at`` that can be days in
|
||||
the future (weekly windows), but the upstream window can reopen
|
||||
before then — the user redeems a banked rate-limit reset via the
|
||||
Codex CLI / ChatGPT UI, upgrades their plan, or OpenAI resets the
|
||||
window. Without this check the pool keeps the credential frozen
|
||||
until the stale timestamp elapses even though the account is
|
||||
usable (issue #43747).
|
||||
|
||||
Only fires for openai-codex entries frozen by a 429/quota-shaped
|
||||
error. The underlying probe is throttled per token (5 min) so this
|
||||
is safe on the hot selection path.
|
||||
"""
|
||||
if self.provider != "openai-codex" or entry.last_status != STATUS_EXHAUSTED:
|
||||
return False
|
||||
if not auth_mod._is_codex_rate_limit_shaped(
|
||||
entry.last_error_code,
|
||||
entry.last_error_reason,
|
||||
entry.last_error_message,
|
||||
):
|
||||
return False
|
||||
token = entry.access_token or ""
|
||||
if not token:
|
||||
return False
|
||||
try:
|
||||
return bool(
|
||||
auth_mod._probe_codex_quota_restored(
|
||||
token,
|
||||
base_url=entry.base_url,
|
||||
)
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("Codex quota-restored probe failed", exc_info=True)
|
||||
return False
|
||||
|
||||
def _entry_needs_refresh(self, entry: PooledCredential) -> bool:
|
||||
if entry.auth_type != AUTH_TYPE_OAUTH:
|
||||
return False
|
||||
@@ -1510,7 +1592,13 @@ class CredentialPool:
|
||||
|
||||
def select(self) -> Optional[PooledCredential]:
|
||||
with self._lock:
|
||||
return self._select_unlocked()
|
||||
entry = self._select_unlocked()
|
||||
if entry is not None:
|
||||
# A normal (non-recovery) selection starts a fresh episode —
|
||||
# don't let a leftover unmatched-rotation streak from an old
|
||||
# failure trip the #70401 bound early next time.
|
||||
self._unmatched_rotation_streak = 0
|
||||
return entry
|
||||
|
||||
def _available_entries(self, *, clear_expired: bool = False, refresh: bool = False) -> List[PooledCredential]:
|
||||
"""Return entries not currently in exhaustion cooldown.
|
||||
@@ -1605,7 +1693,18 @@ class CredentialPool:
|
||||
if entry.last_status == STATUS_EXHAUSTED:
|
||||
exhausted_until = _exhausted_until(entry)
|
||||
if exhausted_until is not None and now < exhausted_until:
|
||||
continue
|
||||
# Codex quota windows can reopen EARLY: the user redeems a
|
||||
# banked rate-limit reset (Codex CLI / ChatGPT UI), upgrades
|
||||
# their plan, or OpenAI resets the window. The persisted
|
||||
# ``last_error_reset_at`` can then be days in the future
|
||||
# while the account is already usable again — a throttled
|
||||
# live probe of the Codex usage endpoint detects that and
|
||||
# lifts the stale cooldown (issue #43747).
|
||||
if not (
|
||||
clear_expired
|
||||
and self._codex_quota_restored_upstream(entry)
|
||||
):
|
||||
continue
|
||||
if clear_expired:
|
||||
cleared = replace(
|
||||
entry,
|
||||
@@ -1678,18 +1777,21 @@ class CredentialPool:
|
||||
self._entries = [replace(candidate, priority=idx) for idx, candidate in enumerate(rotated)]
|
||||
self._persist()
|
||||
self._current_id = entry.id
|
||||
return self.current() or entry
|
||||
return self._current_unlocked() or entry
|
||||
|
||||
entry = available[0]
|
||||
self._current_id = entry.id
|
||||
return entry
|
||||
|
||||
def peek(self) -> Optional[PooledCredential]:
|
||||
current = self.current()
|
||||
if current is not None:
|
||||
return current
|
||||
available = self._available_entries()
|
||||
return available[0] if available else None
|
||||
# Single lock acquisition for the whole read; call the unlocked
|
||||
# helpers so we don't re-enter the non-reentrant ``self._lock``.
|
||||
with self._lock:
|
||||
current = self._current_unlocked()
|
||||
if current is not None:
|
||||
return current
|
||||
available = self._available_entries()
|
||||
return available[0] if available else None
|
||||
|
||||
def mark_exhausted_and_rotate(
|
||||
self,
|
||||
@@ -1697,10 +1799,17 @@ class CredentialPool:
|
||||
status_code: Optional[int],
|
||||
error_context: Optional[Dict[str, Any]] = None,
|
||||
api_key_hint: Optional[str] = None,
|
||||
credential_id: Optional[str] = None,
|
||||
) -> Optional[PooledCredential]:
|
||||
with self._lock:
|
||||
entry = None
|
||||
if api_key_hint:
|
||||
identity_supplied = bool(credential_id or api_key_hint)
|
||||
if credential_id:
|
||||
entry = next(
|
||||
(e for e in self._entries if e.id == credential_id),
|
||||
None,
|
||||
)
|
||||
if entry is None and api_key_hint:
|
||||
# Prefer the specific entry whose API key matches the one that
|
||||
# actually failed. When this pool was freshly loaded from disk
|
||||
# (another process already rotated), current() is None and
|
||||
@@ -1709,12 +1818,89 @@ class CredentialPool:
|
||||
(e for e in self._entries if e.runtime_api_key == api_key_hint),
|
||||
None,
|
||||
)
|
||||
if entry is None and identity_supplied:
|
||||
# The failed credential is identifiable but matches no entry
|
||||
# (rotated away, or a wrapper whose runtime key differs).
|
||||
# Falling through to current()/_select_unlocked() would mark an
|
||||
# innocent healthy key exhausted for the full cooldown TTL.
|
||||
#
|
||||
# #70401: this branch must still be BOUNDED. With OAuth-token
|
||||
# auth the upstream 401's key hint never matches any entry's
|
||||
# ``runtime_api_key``, so every retry lands here, nothing is
|
||||
# ever marked exhausted, and the pool can never reach the
|
||||
# "no available entries" state — the caller retries the same
|
||||
# dead token forever (~6/sec, starving the event loop so chat
|
||||
# interrupts are never processed). The single-entry case
|
||||
# below already escapes; multi-entry pools could still
|
||||
# ping-pong A→B→A indefinitely without marking anything.
|
||||
# Cap consecutive no-mark rotations at one full lap of the
|
||||
# available entries: past that, every candidate has been
|
||||
# handed back at least once without recovery, so stop
|
||||
# guessing and surface the error (no cooldown is written for
|
||||
# anybody — healthy keys stay available for the next turn).
|
||||
self._unmatched_rotation_streak += 1
|
||||
available_count = len(self._available_entries())
|
||||
if self._unmatched_rotation_streak > max(available_count, 1):
|
||||
logger.warning(
|
||||
"credential pool: failed credential identity matched no "
|
||||
"%s entry for %d consecutive rotations (pool size %d) — "
|
||||
"surfacing the error instead of rotating again",
|
||||
self.provider,
|
||||
self._unmatched_rotation_streak,
|
||||
available_count,
|
||||
)
|
||||
self._unmatched_rotation_streak = 0
|
||||
self._current_id = None
|
||||
return None
|
||||
logger.info(
|
||||
"credential pool: failed credential identity matched no %s "
|
||||
"entry; rotating without marking any credential exhausted",
|
||||
self.provider,
|
||||
)
|
||||
self._current_id = None
|
||||
next_entry = self._select_unlocked()
|
||||
if next_entry is not None and len(self._available_entries()) == 1:
|
||||
# A single-entry pool cannot rotate. Returning its only
|
||||
# entry reports a successful recovery without changing
|
||||
# the credential, so the caller retries the same 401
|
||||
# indefinitely. Let fallback/error propagation proceed.
|
||||
self._unmatched_rotation_streak = 0
|
||||
self._current_id = None
|
||||
return None
|
||||
return next_entry
|
||||
# A real entry was identified — any prior unmatched-rotation
|
||||
# streak is stale (this mark WILL advance pool state).
|
||||
self._unmatched_rotation_streak = 0
|
||||
if entry is None:
|
||||
entry = self.current() or self._select_unlocked()
|
||||
entry = self._current_unlocked() or self._select_unlocked()
|
||||
if entry is None:
|
||||
return None
|
||||
_label = entry.label or entry.id[:8]
|
||||
self._mark_exhausted(entry, status_code, error_context)
|
||||
# A 402/429/401 is an API-key–level failure: the account is out of
|
||||
# balance, rate-limited, or its key is rejected. The same key can
|
||||
# back more than one pool entry (e.g. an explicit pool entry plus a
|
||||
# ``model_config`` entry auto-seeded from ``model.api_key`` — both
|
||||
# carry the identical ``runtime_api_key``). Marking only the first
|
||||
# match leaves the sibling entries OK, so ``_select_unlocked()``
|
||||
# keeps handing back the same depleted key and rotation never
|
||||
# converges — the caller ``continue``s forever until the client
|
||||
# disconnects (a ~2.5min hang with no error surfaced to the user).
|
||||
# Mark every entry sharing the failed key so the pool can reach the
|
||||
# "no available entries" state and let the error propagate.
|
||||
failed_runtime_key = getattr(entry, "runtime_api_key", None)
|
||||
if identity_supplied and failed_runtime_key:
|
||||
siblings_marked = False
|
||||
for sibling in self._entries:
|
||||
if sibling.id == entry.id:
|
||||
continue
|
||||
if sibling.runtime_api_key == failed_runtime_key:
|
||||
self._mark_exhausted(
|
||||
sibling, status_code, error_context, persist=False
|
||||
)
|
||||
siblings_marked = True
|
||||
if siblings_marked:
|
||||
self._persist()
|
||||
# Re-read the updated entry to log the correct terminal state.
|
||||
updated_entry = next(
|
||||
(e for e in self._entries if e.id == entry.id), entry,
|
||||
@@ -1782,9 +1968,11 @@ class CredentialPool:
|
||||
return self._try_refresh_current_unlocked()
|
||||
|
||||
def try_refresh_matching(
|
||||
self, api_key_hint: Optional[str] = None
|
||||
self,
|
||||
api_key_hint: Optional[str] = None,
|
||||
credential_id: Optional[str] = None,
|
||||
) -> Optional[PooledCredential]:
|
||||
"""Force-refresh the entry that supplied ``api_key_hint``.
|
||||
"""Force-refresh the entry that supplied the failed request.
|
||||
|
||||
Direct provider integrations may reload the pool after a request has
|
||||
already failed, so they cannot rely on ``current_id`` identifying the
|
||||
@@ -1794,24 +1982,36 @@ class CredentialPool:
|
||||
"""
|
||||
with self._lock:
|
||||
entry = None
|
||||
if api_key_hint:
|
||||
if credential_id:
|
||||
entry = next(
|
||||
(
|
||||
candidate
|
||||
for candidate in self._entries
|
||||
if candidate.runtime_api_key == api_key_hint
|
||||
if candidate.id == credential_id
|
||||
),
|
||||
None,
|
||||
)
|
||||
else:
|
||||
entry = self.current() or self._select_unlocked(refresh=False)
|
||||
if entry is None:
|
||||
if api_key_hint:
|
||||
entry = next(
|
||||
(
|
||||
candidate
|
||||
for candidate in self._entries
|
||||
if candidate.runtime_api_key == api_key_hint
|
||||
),
|
||||
None,
|
||||
)
|
||||
else:
|
||||
entry = self._current_unlocked() or self._select_unlocked(
|
||||
refresh=False
|
||||
)
|
||||
if entry is None:
|
||||
return None
|
||||
self._current_id = entry.id
|
||||
return self._try_refresh_current_unlocked()
|
||||
|
||||
def _try_refresh_current_unlocked(self) -> Optional[PooledCredential]:
|
||||
entry = self.current()
|
||||
entry = self._current_unlocked()
|
||||
if entry is None:
|
||||
return None
|
||||
refreshed = self._refresh_entry(entry, force=True)
|
||||
@@ -1820,76 +2020,80 @@ class CredentialPool:
|
||||
return refreshed
|
||||
|
||||
def reset_statuses(self) -> int:
|
||||
count = 0
|
||||
new_entries = []
|
||||
for entry in self._entries:
|
||||
if entry.last_status or entry.last_status_at or entry.last_error_code:
|
||||
new_entries.append(
|
||||
replace(
|
||||
entry,
|
||||
last_status=None,
|
||||
last_status_at=None,
|
||||
last_error_code=None,
|
||||
last_error_reason=None,
|
||||
last_error_message=None,
|
||||
last_error_reset_at=None,
|
||||
with self._lock:
|
||||
count = 0
|
||||
new_entries = []
|
||||
for entry in self._entries:
|
||||
if entry.last_status or entry.last_status_at or entry.last_error_code:
|
||||
new_entries.append(
|
||||
replace(
|
||||
entry,
|
||||
last_status=None,
|
||||
last_status_at=None,
|
||||
last_error_code=None,
|
||||
last_error_reason=None,
|
||||
last_error_message=None,
|
||||
last_error_reset_at=None,
|
||||
)
|
||||
)
|
||||
)
|
||||
count += 1
|
||||
else:
|
||||
new_entries.append(entry)
|
||||
if count:
|
||||
self._entries = new_entries
|
||||
self._persist()
|
||||
return count
|
||||
count += 1
|
||||
else:
|
||||
new_entries.append(entry)
|
||||
if count:
|
||||
self._entries = new_entries
|
||||
self._persist()
|
||||
return count
|
||||
|
||||
def remove_index(self, index: int) -> Optional[PooledCredential]:
|
||||
if index < 1 or index > len(self._entries):
|
||||
return None
|
||||
removed = self._entries.pop(index - 1)
|
||||
self._entries = [
|
||||
replace(entry, priority=new_priority)
|
||||
for new_priority, entry in enumerate(self._entries)
|
||||
]
|
||||
write_credential_pool(
|
||||
self.provider,
|
||||
[entry.to_dict() for entry in self._entries],
|
||||
removed_ids=[removed.id],
|
||||
)
|
||||
if self._current_id == removed.id:
|
||||
self._current_id = None
|
||||
return removed
|
||||
with self._lock:
|
||||
if index < 1 or index > len(self._entries):
|
||||
return None
|
||||
removed = self._entries.pop(index - 1)
|
||||
self._entries = [
|
||||
replace(entry, priority=new_priority)
|
||||
for new_priority, entry in enumerate(self._entries)
|
||||
]
|
||||
write_credential_pool(
|
||||
self.provider,
|
||||
[entry.to_dict() for entry in self._entries],
|
||||
removed_ids=[removed.id],
|
||||
)
|
||||
if self._current_id == removed.id:
|
||||
self._current_id = None
|
||||
return removed
|
||||
|
||||
def resolve_target(self, target: Any) -> Tuple[Optional[int], Optional[PooledCredential], Optional[str]]:
|
||||
raw = str(target or "").strip()
|
||||
if not raw:
|
||||
return None, None, "No credential target provided."
|
||||
|
||||
for idx, entry in enumerate(self._entries, start=1):
|
||||
if entry.id == raw:
|
||||
return idx, entry, None
|
||||
with self._lock:
|
||||
for idx, entry in enumerate(self._entries, start=1):
|
||||
if entry.id == raw:
|
||||
return idx, entry, None
|
||||
|
||||
label_matches = [
|
||||
(idx, entry)
|
||||
for idx, entry in enumerate(self._entries, start=1)
|
||||
if entry.label.strip().lower() == raw.lower()
|
||||
]
|
||||
if len(label_matches) == 1:
|
||||
return label_matches[0][0], label_matches[0][1], None
|
||||
if len(label_matches) > 1:
|
||||
return None, None, f'Ambiguous credential label "{raw}". Use the numeric index or entry id instead.'
|
||||
if raw.isdigit():
|
||||
index = int(raw)
|
||||
if 1 <= index <= len(self._entries):
|
||||
return index, self._entries[index - 1], None
|
||||
return None, None, f"No credential #{index}."
|
||||
return None, None, f'No credential matching "{raw}".'
|
||||
label_matches = [
|
||||
(idx, entry)
|
||||
for idx, entry in enumerate(self._entries, start=1)
|
||||
if entry.label.strip().lower() == raw.lower()
|
||||
]
|
||||
if len(label_matches) == 1:
|
||||
return label_matches[0][0], label_matches[0][1], None
|
||||
if len(label_matches) > 1:
|
||||
return None, None, f'Ambiguous credential label "{raw}". Use the numeric index or entry id instead.'
|
||||
if raw.isdigit():
|
||||
index = int(raw)
|
||||
if 1 <= index <= len(self._entries):
|
||||
return index, self._entries[index - 1], None
|
||||
return None, None, f"No credential #{index}."
|
||||
return None, None, f'No credential matching "{raw}".'
|
||||
|
||||
def add_entry(self, entry: PooledCredential) -> PooledCredential:
|
||||
entry = replace(entry, priority=_next_priority(self._entries))
|
||||
self._entries.append(entry)
|
||||
self._persist()
|
||||
return entry
|
||||
with self._lock:
|
||||
entry = replace(entry, priority=_next_priority(self._entries))
|
||||
self._entries.append(entry)
|
||||
self._persist()
|
||||
return entry
|
||||
|
||||
|
||||
def _upsert_entry(entries: List[PooledCredential], provider: str, source: str, payload: Dict[str, Any]) -> bool:
|
||||
@@ -2309,9 +2513,10 @@ def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool
|
||||
def _get_env_prefer_dotenv(key: str) -> str:
|
||||
env_file = load_env()
|
||||
raw = env_file.get(key, "").strip()
|
||||
env_val = os.environ.get(key, "").strip()
|
||||
scoped_value = (_get_secret(key, "") or "").strip()
|
||||
# If .env contains an unresolved op:// reference, prefer the
|
||||
# already-resolved value from os.environ (set by
|
||||
# already-resolved value supplied by the active secret scope (or by
|
||||
# os.environ in legacy single-profile mode), set by
|
||||
# load_hermes_dotenv() -> apply_onepassword_secrets()). The raw
|
||||
# "op://Vault/Item/field" string would otherwise win and every
|
||||
# provider auth attempt would receive a URL instead of a key. This
|
||||
@@ -2319,9 +2524,9 @@ def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool
|
||||
# references straight into .env rather than the secrets.onepassword
|
||||
# config block. For every non-op:// value the original
|
||||
# .env-takes-precedence behaviour is preserved unchanged.
|
||||
if raw.startswith("op://") and env_val:
|
||||
return env_val
|
||||
return raw or _get_secret(key, "") or env_val
|
||||
if raw.startswith("op://") and scoped_value:
|
||||
return scoped_value
|
||||
return raw or scoped_value
|
||||
|
||||
# Honour user suppression — `hermes auth remove <provider> <N>` for an
|
||||
# env-seeded credential marks the env:<VAR> source as suppressed so it
|
||||
|
||||
@@ -164,7 +164,7 @@ def _remove_env_source(provider: str, removed) -> RemovalResult:
|
||||
if env_path.exists():
|
||||
env_in_dotenv = any(
|
||||
line.strip().startswith(f"{env_var}=")
|
||||
for line in env_path.read_text(errors="replace").splitlines()
|
||||
for line in env_path.read_text(errors="replace", encoding="utf-8").splitlines()
|
||||
)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
@@ -170,6 +170,27 @@ CREDITS_USAGE_BANDS: tuple[tuple[float, str, int], ...] = (
|
||||
)
|
||||
CREDITS_USAGE_KEY = "credits.usage" # single key for the escalating usage notice
|
||||
|
||||
# Minimum subscription balance that counts as "grant not yet spent" for the
|
||||
# grant_spent crossing gate (see evaluate_credits_notices). 1¢: portal-seeded
|
||||
# states derive micros from float dollars and can carry sub-cent residue where
|
||||
# the inference headers report exactly 0 — without this floor such a seed
|
||||
# opens the gate and the first header re-creates the at-open nag.
|
||||
GRANT_UNSPENT_MIN_MICROS = 10_000
|
||||
|
||||
|
||||
def new_credits_latch() -> dict:
|
||||
"""Fresh notice latch in the shape :func:`evaluate_credits_notices` expects.
|
||||
|
||||
The policy owns this schema — every producer (agent build, lazy re-init,
|
||||
tests) must build the latch through here so a new gate key lands everywhere
|
||||
at once instead of drifting across hand-rolled literals."""
|
||||
return {
|
||||
"active": set(),
|
||||
"seen_below_90": False,
|
||||
"usage_band": None,
|
||||
"seen_grant_unspent": False,
|
||||
}
|
||||
|
||||
|
||||
# ── AgentNotice (out-of-band notice payload; driver-agnostic) ────────────────
|
||||
|
||||
@@ -250,7 +271,8 @@ def evaluate_credits_notices(
|
||||
) -> tuple[list[AgentNotice], list[str]]:
|
||||
"""Reconcile credits notices against the latch. Mutates ``latch`` IN PLACE.
|
||||
|
||||
latch = {"active": set[str], "seen_below_90": bool, "usage_band": Optional[int]}.
|
||||
latch = {"active": set[str], "seen_below_90": bool, "usage_band": Optional[int],
|
||||
"seen_grant_unspent": bool}.
|
||||
|
||||
``model_is_free``: True when the session's active model is a Nous free-tier
|
||||
model (see :func:`is_free_tier_model`). Suppresses the ``credits.depleted``
|
||||
@@ -277,6 +299,18 @@ def evaluate_credits_notices(
|
||||
if uf is not None and uf < _lowest_band:
|
||||
latch["seen_below_90"] = True # gate opened: usage-band notices may now fire
|
||||
|
||||
# Grant-spent crossing gate: grant_spent may fire only after this session
|
||||
# has OBSERVED the grant meaningfully unspent (≥1¢ left — see
|
||||
# GRANT_UNSPENT_MIN_MICROS). Opening at grant-spent is a steady STATE, not
|
||||
# an event — /usage carries it; only a live in-session crossing announces.
|
||||
# Unlike seen_below_90, seeds must NOT prime this gate.
|
||||
if (
|
||||
uf is not None
|
||||
and uf < 1.0
|
||||
and state.subscription_micros >= GRANT_UNSPENT_MIN_MICROS
|
||||
):
|
||||
latch["seen_grant_unspent"] = True
|
||||
|
||||
active = latch["active"]
|
||||
|
||||
# ── Conditions ───────────────────────────────────────────────────────────
|
||||
@@ -316,12 +350,21 @@ def evaluate_credits_notices(
|
||||
active.discard(CREDITS_USAGE_KEY)
|
||||
if target_band is not None:
|
||||
# Belt-and-suspenders: a producer could set subscription_limit_micros
|
||||
# without subscription_limit_usd. Render "$? cap" rather than "$None cap".
|
||||
# without subscription_limit_usd. Render "$?" rather than "$None".
|
||||
_cap_usd = state.subscription_limit_usd or "?"
|
||||
_level = current_band[1] # type: ignore[index] (current_band set when target_band set)
|
||||
# Report absolute dollars used, not a bare "N% used": the percentage is
|
||||
# only meaningful against a Nous subscription cap (no cap → never fires),
|
||||
# so dollars are clearer and don't imply a universal %. Used = cap −
|
||||
# remaining (micros, money-safe), clamped to [0, cap]. Re-emits on band
|
||||
# change (50 → 75 → 90), not every turn — a snapshot, not a live ticker.
|
||||
_lim = state.subscription_limit_micros or 0
|
||||
_used_micros = max(0, min(_lim, _lim - state.subscription_micros))
|
||||
_used_usd = f"{_used_micros / 1_000_000:.2f}" if _lim else "?"
|
||||
_glyph = "⚠" if _level == "warn" else "•"
|
||||
to_show.append(
|
||||
AgentNotice(
|
||||
text=f"{'⚠' if _level == 'warn' else '•'} Credits {target_band}% used · ${_cap_usd} cap",
|
||||
text=f"{_glyph} You've used ${_used_usd} of your ${_cap_usd} cap",
|
||||
level=_level,
|
||||
kind=CREDITS_NOTICE_KIND,
|
||||
key=CREDITS_USAGE_KEY,
|
||||
@@ -332,7 +375,17 @@ def evaluate_credits_notices(
|
||||
latch["usage_band"] = target_band
|
||||
|
||||
# ── grant_spent ──────────────────────────────────────────────────────────
|
||||
if grant_cond and "credits.grant_spent" not in active:
|
||||
# The crossing gate guards only the SHOW and is CONSUMED by it — one
|
||||
# announcement per crossing. A header flicker (uf → None → back to 1.0)
|
||||
# clears the sticky line via grant_cond but cannot re-announce; only a
|
||||
# renewal that re-opens the gate (a fresh ≥1¢ observation) arms the next
|
||||
# announcement. .get(): default closed for any hand-built latch missing
|
||||
# the key, so a first observation can never fire this notice.
|
||||
if (
|
||||
grant_cond
|
||||
and "credits.grant_spent" not in active
|
||||
and latch.get("seen_grant_unspent", False)
|
||||
):
|
||||
to_show.append(
|
||||
AgentNotice(
|
||||
text=f"• Grant spent · ${state.purchased_usd} top-up left",
|
||||
@@ -343,6 +396,7 @@ def evaluate_credits_notices(
|
||||
)
|
||||
)
|
||||
active.add("credits.grant_spent")
|
||||
latch["seen_grant_unspent"] = False
|
||||
elif "credits.grant_spent" in active and not grant_cond:
|
||||
to_clear.append("credits.grant_spent")
|
||||
active.discard("credits.grant_spent")
|
||||
@@ -618,7 +672,8 @@ _DEV_FIXTURES: dict[str, dict] = {
|
||||
subscription_limit_micros=20_000_000, subscription_limit_usd="20.00",
|
||||
denominator_kind="subscription_cap", paid_access=True,
|
||||
),
|
||||
"grant_exhausted": dict( # used_fraction == 1.0 + purchased>0 → credits.grant_spent
|
||||
"grant_exhausted": dict( # uf == 1.0 + purchased>0 → SILENT at open (crossing-gated);
|
||||
# flip healthy → grant_exhausted via the fixture-file path to see credits.grant_spent
|
||||
remaining_micros=12_340_000, remaining_usd="12.34",
|
||||
subscription_micros=0, subscription_usd="0.00",
|
||||
subscription_limit_micros=20_000_000, subscription_limit_usd="20.00",
|
||||
@@ -732,6 +787,9 @@ def _hydrate_seed_state(agent, state) -> None:
|
||||
agent._credits_session_start_micros = state.remaining_micros
|
||||
_latch = getattr(agent, "_credits_latch", None)
|
||||
if isinstance(_latch, dict) and state.used_fraction is not None:
|
||||
# Prime ONLY seen_below_90 (open-high band warnings are wanted at open).
|
||||
# Never prime seen_grant_unspent here: a seed observing grant-spent is a
|
||||
# steady state, and priming it would revive the every-session nag.
|
||||
_latch["seen_below_90"] = True
|
||||
emit = getattr(agent, "_emit_credits_notices", None)
|
||||
if callable(emit):
|
||||
|
||||
+18
-16
@@ -138,8 +138,8 @@ def is_paused() -> bool:
|
||||
def _load_config() -> Dict[str, Any]:
|
||||
"""Read curator.* config from ~/.hermes/config.yaml. Tolerates missing file."""
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
cfg = load_config()
|
||||
from hermes_cli.config import load_config_readonly
|
||||
cfg = load_config_readonly()
|
||||
except Exception as e:
|
||||
logger.debug("Failed to load config for curator: %s", e)
|
||||
return {}
|
||||
@@ -325,7 +325,7 @@ def apply_automatic_transitions(now: Optional[datetime] = None) -> Dict[str, int
|
||||
|
||||
counts = {"marked_stale": 0, "archived": 0, "reactivated": 0, "checked": 0, "seeded": 0}
|
||||
|
||||
for row in _u.agent_created_report():
|
||||
for row in _u.curated_report():
|
||||
counts["checked"] += 1
|
||||
name = row["name"]
|
||||
if row.get("pinned"):
|
||||
@@ -422,7 +422,9 @@ CURATOR_REVIEW_PROMPT = (
|
||||
"INSTRUCTIONS AND EXPERIENTIAL KNOWLEDGE. A collection of hundreds of "
|
||||
"narrow skills where each one captures one session's specific bug is "
|
||||
"a FAILURE of the library — not a feature. An agent searching skills "
|
||||
"matches on descriptions, not on exact names; one broad umbrella "
|
||||
"matches on descriptions, not on exact names (note: long descriptions "
|
||||
"are truncated to 57 chars in the system prompt skill index — keep the "
|
||||
"trigger class in that window). One broad umbrella "
|
||||
"skill with labeled subsections beats five narrow siblings for "
|
||||
"discoverability, not the other way around.\n\n"
|
||||
"The right target shape is CLASS-LEVEL skills with rich SKILL.md "
|
||||
@@ -900,7 +902,6 @@ def _reconcile_classification(
|
||||
Every removed skill is placed in exactly one bucket.
|
||||
"""
|
||||
heur_cons = {e["name"]: e for e in heuristic.get("consolidated", [])}
|
||||
heur_pruned = {e["name"] for e in heuristic.get("pruned", [])}
|
||||
|
||||
model_cons = {e["from"]: e for e in model_block.get("consolidations", [])}
|
||||
model_pruned = {e["name"]: e for e in model_block.get("prunings", [])}
|
||||
@@ -1470,15 +1471,16 @@ def _render_report_markdown(p: Dict[str, Any]) -> str:
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _render_candidate_list() -> str:
|
||||
"""Human/agent-readable list of agent-created skills with usage stats."""
|
||||
rows = skill_usage.agent_created_report()
|
||||
"""Human/agent-readable list of curator-managed skills with usage stats."""
|
||||
rows = skill_usage.curated_report()
|
||||
if not rows:
|
||||
return "No agent-created skills to review."
|
||||
return "No curator-managed skills to review."
|
||||
cron_referenced = _cron_referenced_skills()
|
||||
lines = [f"Agent-created skills ({len(rows)}):\n"]
|
||||
lines = [f"Curator-managed skills ({len(rows)}):\n"]
|
||||
for r in rows:
|
||||
lines.append(
|
||||
f"- {r['name']} "
|
||||
f"provenance={r.get('provenance', 'agent')} "
|
||||
f"state={r['state']} "
|
||||
f"pinned={'yes' if r.get('pinned') else 'no'} "
|
||||
f"cron={'yes' if r['name'] in cron_referenced else 'no'} "
|
||||
@@ -1531,7 +1533,7 @@ def run_curator_review(
|
||||
if dry_run:
|
||||
# Count candidates without mutating state.
|
||||
try:
|
||||
report = skill_usage.agent_created_report()
|
||||
report = skill_usage.curated_report()
|
||||
counts = {
|
||||
"checked": len(report),
|
||||
"marked_stale": 0,
|
||||
@@ -1584,7 +1586,7 @@ def run_curator_review(
|
||||
nonlocal auto_summary
|
||||
# Snapshot skill state BEFORE the LLM pass so the report can diff.
|
||||
try:
|
||||
before_report = skill_usage.agent_created_report()
|
||||
before_report = skill_usage.curated_report()
|
||||
except Exception:
|
||||
before_report = []
|
||||
before_names = {r.get("name") for r in before_report if isinstance(r, dict)}
|
||||
@@ -1610,7 +1612,7 @@ def run_curator_review(
|
||||
state2["last_run_duration_seconds"] = elapsed
|
||||
state2["last_run_summary"] = final_summary
|
||||
try:
|
||||
after_report = skill_usage.agent_created_report()
|
||||
after_report = skill_usage.curated_report()
|
||||
except Exception:
|
||||
after_report = []
|
||||
try:
|
||||
@@ -1697,7 +1699,7 @@ def run_curator_review(
|
||||
try:
|
||||
rename_lines = _build_rename_summary(
|
||||
before_names=before_names,
|
||||
after_report=skill_usage.agent_created_report(),
|
||||
after_report=skill_usage.curated_report(),
|
||||
tool_calls=llm_meta.get("tool_calls", []) or [],
|
||||
model_final=llm_meta.get("final", "") or "",
|
||||
)
|
||||
@@ -1715,7 +1717,7 @@ def run_curator_review(
|
||||
# reporting bug never breaks the curator itself. Report path is
|
||||
# recorded in state so `hermes curator status` can point at it.
|
||||
try:
|
||||
after_report = skill_usage.agent_created_report()
|
||||
after_report = skill_usage.curated_report()
|
||||
except Exception:
|
||||
after_report = []
|
||||
try:
|
||||
@@ -1873,9 +1875,9 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
|
||||
_acp_args = None
|
||||
_model_name = ""
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
from hermes_cli.config import load_config_readonly
|
||||
from hermes_cli.runtime_provider import resolve_runtime_provider
|
||||
_cfg = load_config()
|
||||
_cfg = load_config_readonly()
|
||||
_binding = _resolve_review_runtime(_cfg)
|
||||
_provider, _model_name = _binding.provider, _binding.model
|
||||
_rp = resolve_runtime_provider(
|
||||
|
||||
@@ -147,8 +147,8 @@ def _utc_id(now: Optional[datetime] = None) -> str:
|
||||
|
||||
def _load_config() -> Dict[str, Any]:
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
cfg = load_config()
|
||||
from hermes_cli.config import load_config_readonly
|
||||
cfg = load_config_readonly()
|
||||
except Exception as e:
|
||||
logger.debug("Failed to load config for curator backup: %s", e)
|
||||
return {}
|
||||
|
||||
@@ -0,0 +1,85 @@
|
||||
"""Context-local state for delegate_task child execution.
|
||||
|
||||
The parent Hermes process may itself be a Kanban dispatcher worker with
|
||||
HERMES_KANBAN_* variables in process env. delegate_task children run inside the
|
||||
same Python process, but they are not dispatcher-owned Kanban workers. This
|
||||
module lets code paths that resolve tool schemas or spawn subprocesses fail
|
||||
closed for delegated children without mutating global os.environ for the parent.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from contextlib import contextmanager
|
||||
from contextvars import ContextVar
|
||||
from typing import Iterator, Mapping, MutableMapping
|
||||
|
||||
_DELEGATED_CHILD_CONTEXT: ContextVar[bool] = ContextVar(
|
||||
"hermes_delegated_child_context",
|
||||
default=False,
|
||||
)
|
||||
|
||||
DELEGATED_CHILD_ENV_MARKER = "HERMES_DELEGATED_CHILD_CONTEXT"
|
||||
|
||||
KANBAN_ENV_KEYS: tuple[str, ...] = (
|
||||
"HERMES_KANBAN_TASK",
|
||||
"HERMES_KANBAN_RUN_ID",
|
||||
"HERMES_KANBAN_WORKSPACE",
|
||||
"HERMES_KANBAN_WORKSPACES_ROOT",
|
||||
"HERMES_KANBAN_CLAIM_LOCK",
|
||||
"HERMES_KANBAN_BOARD",
|
||||
"HERMES_KANBAN_DB",
|
||||
)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def delegated_child_context() -> Iterator[None]:
|
||||
"""Mark the current execution context as a delegate_task child."""
|
||||
token = _DELEGATED_CHILD_CONTEXT.set(True)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
_DELEGATED_CHILD_CONTEXT.reset(token)
|
||||
|
||||
|
||||
def is_delegated_child_context() -> bool:
|
||||
"""Return True while code is running for a delegate_task child."""
|
||||
return bool(_DELEGATED_CHILD_CONTEXT.get())
|
||||
|
||||
|
||||
def is_delegated_child_process_context() -> bool:
|
||||
"""Return True in this process or a subprocess spawned by a child."""
|
||||
import os
|
||||
|
||||
return bool(_DELEGATED_CHILD_CONTEXT.get()) or bool(
|
||||
os.environ.get(DELEGATED_CHILD_ENV_MARKER)
|
||||
)
|
||||
|
||||
|
||||
def scrub_kanban_env(env: Mapping[str, str] | MutableMapping[str, str]) -> dict[str, str]:
|
||||
"""Return *env* with dispatcher-only Kanban variables removed."""
|
||||
cleaned = dict(env)
|
||||
for key in KANBAN_ENV_KEYS:
|
||||
cleaned.pop(key, None)
|
||||
cleaned[DELEGATED_CHILD_ENV_MARKER] = "1"
|
||||
return cleaned
|
||||
|
||||
|
||||
def delegated_child_subprocess_env(
|
||||
env: Mapping[str, str] | MutableMapping[str, str] | None = None,
|
||||
) -> dict[str, str] | None:
|
||||
"""Return an env override only when delegated-child lineage must cross fork.
|
||||
|
||||
Most subprocess call sites historically used ``env=None`` to inherit the
|
||||
process environment. In a ``delegate_task`` child, inheriting as-is leaks
|
||||
parent dispatcher ``HERMES_KANBAN_*`` vars while losing the ContextVar in
|
||||
the new process. This helper preserves normal ``env=None`` semantics for
|
||||
non-delegated calls, and only materializes a scrubbed env when the lineage
|
||||
marker must be propagated across a child-process boundary.
|
||||
"""
|
||||
if not is_delegated_child_process_context():
|
||||
return None if env is None else dict(env)
|
||||
|
||||
if env is None:
|
||||
import os
|
||||
|
||||
env = os.environ
|
||||
return scrub_kanban_env(env)
|
||||
@@ -645,6 +645,52 @@ def verb_drops_preview(tool_name: str) -> bool:
|
||||
return tool_name in _TOOL_VERBS_NO_PREVIEW
|
||||
|
||||
|
||||
def build_status_phrase(tool_name: str, args: dict | None, max_len: int = 49) -> str | None:
|
||||
"""Build a short present-tense status phrase for platform status surfaces.
|
||||
|
||||
Used by text-rendering "typing" indicators (Slack's
|
||||
``assistant.threads.setStatus`` line) to show what the agent is doing
|
||||
right now: ``is running scripts/run_tests.sh…`` instead of a static
|
||||
``is thinking...``. The phrase is phrased to follow the bot's display
|
||||
name ("Hermes is running …"), so it starts lowercase with "is".
|
||||
|
||||
Pass ``args=None`` for a verb-only phrase (``is running…``) — used when
|
||||
``display.live_status`` is ``verb`` to keep argument previews out of
|
||||
shared channels.
|
||||
|
||||
Returns None for the ``_thinking`` pseudo-tool and when friendly labels
|
||||
are disabled (callers fall back to their static default). ``max_len``
|
||||
caps the total phrase length; Slack truncates its status line around 50
|
||||
characters, so the default stays just under that.
|
||||
"""
|
||||
if not tool_name or tool_name == "_thinking":
|
||||
return None
|
||||
if not _friendly_tool_labels:
|
||||
return None
|
||||
|
||||
verb = _TOOL_VERBS.get(tool_name)
|
||||
if verb:
|
||||
head = f"is {verb[0].lower()}{verb[1:]}"
|
||||
else:
|
||||
# Custom / plugin / MCP tools: generic but still informative.
|
||||
head = f"is using {tool_name}"
|
||||
|
||||
phrase = head
|
||||
if args and verb and tool_name not in _TOOL_VERBS_NO_PREVIEW:
|
||||
preview = build_tool_preview(tool_name, args, max_len=None)
|
||||
if preview:
|
||||
# Previews can contain newlines (terminal commands); keep the
|
||||
# status to the first line.
|
||||
preview = preview.splitlines()[0].strip()
|
||||
phrase = f"{head}{tool_verb_connector(tool_name)}{preview}"
|
||||
|
||||
if len(phrase) > max_len - 1:
|
||||
phrase = phrase[: max_len - 2].rstrip() + "…"
|
||||
else:
|
||||
phrase = phrase + "…"
|
||||
return phrase
|
||||
|
||||
|
||||
def build_tool_label(tool_name: str, args: dict, max_len: int | None = None) -> str | None:
|
||||
"""Build a human-phrased status label for a tool call.
|
||||
|
||||
|
||||
+147
-3
@@ -159,6 +159,14 @@ _RATE_LIMIT_PATTERNS = [
|
||||
"throttlingexception",
|
||||
"too many concurrent requests",
|
||||
"servicequotaexceededexception",
|
||||
# Generic throttle prefix — Bedrock (and some proxies) surface throttling
|
||||
# as "Throttling error: Too many tokens, please wait before trying
|
||||
# again." Without this entry the message falls through to the
|
||||
# context-overflow list (which contains "too many tokens") and the retry
|
||||
# loop compresses a healthy session instead of backing off. Matched
|
||||
# BEFORE _CONTEXT_OVERFLOW_PATTERNS in the message-only path, so the
|
||||
# throttle wins. (port of anomalyco/opencode#37848's exclusion guard)
|
||||
"throttling",
|
||||
]
|
||||
|
||||
# Patterns that indicate provider-side overload, NOT a per-credential rate
|
||||
@@ -212,6 +220,12 @@ _PAYLOAD_TOO_LARGE_PATTERNS = [
|
||||
"request entity too large",
|
||||
"payload too large",
|
||||
"error code: 413",
|
||||
# Anthropic's structured 413 error type. Normally arrives with an HTTP
|
||||
# 413 status (handled by the status path), but aggregators/proxies can
|
||||
# re-wrap it into a plain message with no status attribute — route it to
|
||||
# the same compression recovery. (port of anomalyco/opencode#37848)
|
||||
"request_too_large",
|
||||
"request exceeds the maximum size",
|
||||
]
|
||||
|
||||
# Image-size patterns. Matched against 400 bodies (not 413) because most
|
||||
@@ -269,6 +283,11 @@ _CONTEXT_OVERFLOW_PATTERNS = [
|
||||
"context window",
|
||||
"prompt is too long",
|
||||
"prompt exceeds max length",
|
||||
# NOTE: bare "max_tokens" is load-bearing — the output-cap-retry path keys
|
||||
# off it (e.g. "max_tokens: 65536 > context_window: 200000 ..."). Do NOT
|
||||
# remove it. Provider empty-response advisories also contain "very low
|
||||
# max_tokens", but those are intercepted by _EMPTY_PROVIDER_RESPONSE_PATTERNS
|
||||
# BEFORE this list is consulted, so they never mis-route into compression.
|
||||
"max_tokens",
|
||||
"maximum number of tokens",
|
||||
# vLLM / local inference server patterns
|
||||
@@ -293,6 +312,10 @@ _CONTEXT_OVERFLOW_PATTERNS = [
|
||||
"max input token",
|
||||
"input token",
|
||||
"exceeds the maximum number of input tokens",
|
||||
# Together/Fireworks-style: "Input length 131393 exceeds the maximum
|
||||
# allowed input length of 131040 tokens." No other pattern in this list
|
||||
# matches that wording. (port of anomalyco/opencode#37848)
|
||||
"maximum allowed input length",
|
||||
]
|
||||
|
||||
# Model not found patterns
|
||||
@@ -316,6 +339,30 @@ _MODEL_NOT_FOUND_PATTERNS = [
|
||||
"no endpoints found that support tool use",
|
||||
]
|
||||
|
||||
# Malformed-message-array 400s. Deterministic request-shape rejections that
|
||||
# describe the *transcript* being invalid, not a parameter. The canonical
|
||||
# case: a stream dies mid-response and Hermes persists a content-less
|
||||
# assistant stub; on the next turn the Anthropic message schema (and the
|
||||
# litellm/Bedrock proxies in front of it) reject the whole request with
|
||||
# "all messages must have non-empty content except for the optional final
|
||||
# assistant message" / errorCode INVALID_REQUEST_BODY
|
||||
# These are NOT context overflow — the input may be tiny — but a large
|
||||
# session used to mis-route them into the compression loop via the generic
|
||||
# "400 + large session" heuristic below, ending in "Cannot compress further"
|
||||
# every retry (the input is unchanged, so compression cannot help). Match
|
||||
# the message-shape signals explicitly and fail fast as a format_error so the
|
||||
# loop stops looping. The empty-stub creation is the root cause (fixed in
|
||||
# chat_completion_helpers); this pattern stops the misclassification symptom
|
||||
# for transcripts that already contain a poisoned stub.
|
||||
_INVALID_MESSAGE_BODY_PATTERNS = [
|
||||
"must have non-empty content",
|
||||
"messages must have non-empty",
|
||||
"invalid_request_body",
|
||||
"text content blocks must be non-empty",
|
||||
"content field is required",
|
||||
"messages: at least one message is required",
|
||||
]
|
||||
|
||||
# Request-validation patterns — the request is malformed and will fail
|
||||
# identically on every retry. Some OpenAI-compatible gateways (notably
|
||||
# codex.nekos.me) return these as 5xx instead of the standard 4xx, which
|
||||
@@ -408,6 +455,7 @@ _CONTENT_POLICY_BLOCKED_PATTERNS = [
|
||||
_AUTH_PATTERNS = [
|
||||
"invalid api key",
|
||||
"invalid_api_key",
|
||||
"gateway_auth_failed",
|
||||
"authentication",
|
||||
"unauthorized",
|
||||
"forbidden",
|
||||
@@ -426,6 +474,19 @@ _THINKING_SIG_PATTERNS = [
|
||||
# the exception type is generic (e.g. RuntimeError from a local shim that
|
||||
# wraps a subprocess timeout). Checked before the type-based transport
|
||||
# heuristics so custom-provider "timed out" errors don't fall through to
|
||||
# Provider empty-response advisories (OpenRouter / nano-gpt / similar).
|
||||
# Checked before context-overflow matching because the advisory text often
|
||||
# mentions "max_tokens" as a possible cause, which historically sat in
|
||||
# _CONTEXT_OVERFLOW_PATTERNS and sent healthy sessions into a compression
|
||||
# death spiral ending in "Cannot compress further".
|
||||
_EMPTY_PROVIDER_RESPONSE_PATTERNS = [
|
||||
"returned an empty response",
|
||||
"empty response despite retries",
|
||||
"provider returned an empty response",
|
||||
"model returning empty responses",
|
||||
"empty response stream",
|
||||
]
|
||||
|
||||
# the unknown bucket and get misreported as empty responses.
|
||||
_TIMEOUT_MESSAGE_PATTERNS = [
|
||||
"timed out",
|
||||
@@ -1077,6 +1138,14 @@ def _classify_by_status(
|
||||
# remaining explicit context-overflow signal routes into the
|
||||
# compression-and-retry path (mirroring _classify_400) instead of
|
||||
# blind server_error retries that exhaust and drop the turn.
|
||||
# Empty-response advisories that mention "max_tokens" must not enter
|
||||
# that compression path.
|
||||
if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS):
|
||||
return result_fn(
|
||||
FailoverReason.server_error,
|
||||
retryable=True,
|
||||
should_compress=False,
|
||||
)
|
||||
if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS):
|
||||
return result_fn(
|
||||
FailoverReason.context_overflow,
|
||||
@@ -1090,6 +1159,12 @@ def _classify_by_status(
|
||||
# Cloudflare/Tailscale hop relabeling the status). Route explicit
|
||||
# overflow bodies into compression; otherwise treat as transient
|
||||
# overload and retry.
|
||||
if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS):
|
||||
return result_fn(
|
||||
FailoverReason.server_error,
|
||||
retryable=True,
|
||||
should_compress=False,
|
||||
)
|
||||
if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS):
|
||||
return result_fn(
|
||||
FailoverReason.context_overflow,
|
||||
@@ -1215,8 +1290,8 @@ def _classify_400(
|
||||
# returns:
|
||||
# "Unsupported parameter: 'max_tokens' is not supported with this model.
|
||||
# Use 'max_completion_tokens' instead."
|
||||
# That string contains the literal substring "max_tokens", which is one of
|
||||
# the _CONTEXT_OVERFLOW_PATTERNS — so without this guard the 400 is
|
||||
# That string contains the literal substring "max_tokens", which historically
|
||||
# sat in _CONTEXT_OVERFLOW_PATTERNS — so without this guard the 400 is
|
||||
# misclassified as context_overflow, routed into the compression loop,
|
||||
# re-sent with the same bad parameter, and ends in "Cannot compress
|
||||
# further". These errors are deterministic (every retry gets the identical
|
||||
@@ -1238,6 +1313,44 @@ def _classify_400(
|
||||
should_fallback=True,
|
||||
)
|
||||
|
||||
# Malformed message array (empty-content assistant stub, etc.). Must be
|
||||
# checked BEFORE context_overflow: the input can be tiny, so the generic
|
||||
# "400 + large session" heuristic would otherwise mis-route it into the
|
||||
# compression loop and thrash until "Cannot compress further" on every
|
||||
# retry (the request is unchanged, so compression cannot fix it). This is
|
||||
# a deterministic request-shape rejection — fail fast as a non-retryable
|
||||
# format_error and fall back. Checked against the message text AND the
|
||||
# structured error code, since proxies (litellm/Bedrock) surface the
|
||||
# signal in errorCode=INVALID_REQUEST_BODY.
|
||||
if (
|
||||
any(p in error_msg for p in _INVALID_MESSAGE_BODY_PATTERNS)
|
||||
or error_code_lower == "invalid_request_body"
|
||||
):
|
||||
logger.warning(
|
||||
"Malformed message array 400 (invalid request body) classified as "
|
||||
"format_error, NOT context overflow — failing fast + falling back "
|
||||
"instead of entering the compression loop. This usually means an "
|
||||
"empty-content assistant stub is in the transcript; num_messages=%s "
|
||||
"approx_tokens=%s. error=%.200s",
|
||||
num_messages, approx_tokens, error_msg,
|
||||
)
|
||||
return result_fn(
|
||||
FailoverReason.format_error,
|
||||
retryable=False,
|
||||
should_fallback=True,
|
||||
)
|
||||
|
||||
# Empty-provider-response advisories must not enter compression. They
|
||||
# often mention "max_tokens" as a possible cause and used to match the
|
||||
# bare overflow pattern, then thrash compress until "Cannot compress
|
||||
# further" on an otherwise healthy session (custom endpoints / nano-gpt).
|
||||
if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS):
|
||||
return result_fn(
|
||||
FailoverReason.server_error,
|
||||
retryable=True,
|
||||
should_compress=False,
|
||||
)
|
||||
|
||||
# Context overflow from 400
|
||||
if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS):
|
||||
return result_fn(
|
||||
@@ -1287,6 +1400,18 @@ def _classify_400(
|
||||
# Responses API (and some providers) use flat body: {"message": "..."}
|
||||
if not err_body_msg:
|
||||
err_body_msg = str(body.get("message") or "").strip().lower()
|
||||
# litellm / Bedrock proxies use a custom shape: {"errorMessage": "...",
|
||||
# "errorCode": "...", "errorArgs": {"reason": "..."}}. Without these
|
||||
# keys err_body_msg stays "" and a long, descriptive rejection is
|
||||
# wrongly treated as a "generic" (bare) error below, which — on a
|
||||
# large session — mis-routes into the compression loop. Recognize
|
||||
# them so the is_generic heuristic sees the real message length.
|
||||
if not err_body_msg:
|
||||
err_body_msg = str(body.get("errorMessage") or "").strip().lower()
|
||||
if not err_body_msg:
|
||||
_args = body.get("errorArgs")
|
||||
if isinstance(_args, dict):
|
||||
err_body_msg = str(_args.get("reason") or "").strip().lower()
|
||||
is_generic = len(err_body_msg) < 30 or err_body_msg in {"error", ""}
|
||||
# Absolute token/message-count thresholds are only a proxy for smaller
|
||||
# context windows. Large-context sessions can have many messages while
|
||||
@@ -1441,6 +1566,15 @@ def _classify_by_message(
|
||||
should_fallback=True,
|
||||
)
|
||||
|
||||
# Empty-provider-response advisories (often mention "max_tokens") must
|
||||
# retry without compression — see the matching 400-path guard above.
|
||||
if any(p in error_msg for p in _EMPTY_PROVIDER_RESPONSE_PATTERNS):
|
||||
return result_fn(
|
||||
FailoverReason.server_error,
|
||||
retryable=True,
|
||||
should_compress=False,
|
||||
)
|
||||
|
||||
# Context overflow patterns
|
||||
if any(p in error_msg for p in _CONTEXT_OVERFLOW_PATTERNS):
|
||||
return result_fn(
|
||||
@@ -1576,7 +1710,7 @@ def _extract_error_code(body: dict) -> str:
|
||||
return nested_code
|
||||
|
||||
# Top-level code
|
||||
code = body.get("code") or body.get("error_code") or ""
|
||||
code = body.get("code") or body.get("error_code") or body.get("errorCode") or ""
|
||||
if isinstance(code, (str, int)):
|
||||
text = str(code).strip()
|
||||
if text and text != "400":
|
||||
@@ -1596,6 +1730,16 @@ def _extract_message(error: Exception, body: dict) -> str:
|
||||
msg = body.get("message", "")
|
||||
if isinstance(msg, str) and msg.strip():
|
||||
return msg.strip()[:500]
|
||||
# litellm / Bedrock proxy shape: {"errorMessage": "...",
|
||||
# "errorArgs": {"reason": "..."}}.
|
||||
msg = body.get("errorMessage", "")
|
||||
if isinstance(msg, str) and msg.strip():
|
||||
return msg.strip()[:500]
|
||||
args = body.get("errorArgs")
|
||||
if isinstance(args, dict):
|
||||
reason = args.get("reason", "")
|
||||
if isinstance(reason, str) and reason.strip():
|
||||
return reason.strip()[:500]
|
||||
# Fallback to str(error)
|
||||
return str(error)[:500]
|
||||
|
||||
|
||||
@@ -46,6 +46,9 @@ def build_write_denied_paths(home: str) -> set[str]:
|
||||
# Top-level Anthropic PKCE credential store remains sensitive even
|
||||
# when a profile is active; default/non-profile sessions still read it.
|
||||
str(hermes_root / ".anthropic_oauth.json"),
|
||||
# Bitwarden Secrets Manager encrypted disk cache.
|
||||
str(hermes_home / "cache" / "bws_cache.enc.json"),
|
||||
str(hermes_root / "cache" / "bws_cache.enc.json"),
|
||||
os.path.join(home, ".netrc"),
|
||||
os.path.join(home, ".pgpass"),
|
||||
os.path.join(home, ".npmrc"),
|
||||
|
||||
@@ -73,7 +73,7 @@ def probe_gemini_tier(
|
||||
api_key: str,
|
||||
base_url: str = DEFAULT_GEMINI_BASE_URL,
|
||||
*,
|
||||
model: str = "gemini-2.5-flash",
|
||||
model: str = "gemini-3.6-flash",
|
||||
timeout: float = 10.0,
|
||||
) -> str:
|
||||
"""Probe a Google AI Studio API key and return its tier.
|
||||
@@ -154,8 +154,8 @@ def is_free_tier_quota_error(error_message: str) -> bool:
|
||||
|
||||
|
||||
_FREE_TIER_GUIDANCE = (
|
||||
"\n\nYour Google API key is on the free tier (<= 250 requests/day for "
|
||||
"gemini-2.5-flash). Hermes typically makes 3-10 API calls per user turn, "
|
||||
"\n\nYour Google API key is on the free tier (a few hundred requests/day "
|
||||
"for Gemini Flash models). Hermes typically makes 3-10 API calls per user turn, "
|
||||
"so the free tier is exhausted in a handful of messages and cannot sustain "
|
||||
"an agent session. Enable billing on your Google Cloud project and "
|
||||
"regenerate the key in a billing-enabled project: "
|
||||
@@ -163,6 +163,42 @@ _FREE_TIER_GUIDANCE = (
|
||||
)
|
||||
|
||||
|
||||
def is_standard_key_auth_error(
|
||||
status: int, error_message: str, reason: str = ""
|
||||
) -> bool:
|
||||
"""Return True when a Gemini 401 indicates Google rejected the key TYPE.
|
||||
|
||||
Google began rejecting unrestricted legacy "Standard" Google Cloud API
|
||||
keys on the Gemini API on June 19, 2026, and ALL Standard keys stop
|
||||
working in September 2026. The rejection surfaces as a misleading 401
|
||||
telling the user to supply an OAuth 2 access token ("Request had invalid
|
||||
authentication credentials. Expected OAuth 2 access token, login cookie
|
||||
or other valid authentication credential."), optionally carrying
|
||||
``google.rpc.ErrorInfo`` reason ``ACCESS_TOKEN_TYPE_UNSUPPORTED``.
|
||||
|
||||
Scoped narrowly so a plain bad key (reason ``API_KEY_INVALID``,
|
||||
"API key not valid") keeps its existing message.
|
||||
"""
|
||||
if status != 401:
|
||||
return False
|
||||
if reason == "ACCESS_TOKEN_TYPE_UNSUPPORTED":
|
||||
return True
|
||||
return "expected oauth 2 access token" in (error_message or "").lower()
|
||||
|
||||
|
||||
_STANDARD_KEY_GUIDANCE = (
|
||||
"\n\nGoogle Gemini rejected this API key's type — you do NOT need OAuth. "
|
||||
"Google began rejecting legacy 'Standard' Google Cloud keys for the "
|
||||
"Gemini API on June 19, 2026, and all Standard keys stop working in "
|
||||
"September 2026. Open https://aistudio.google.com/api-keys, check the "
|
||||
"key's type and status, and create a replacement Gemini API key (or, as "
|
||||
"a temporary bridge, restrict the Standard key to "
|
||||
"generativelanguage.googleapis.com). Then update GEMINI_API_KEY / "
|
||||
"GOOGLE_API_KEY in ~/.hermes/.env and restart your session. "
|
||||
"Details: https://ai.google.dev/gemini-api/docs/api-key"
|
||||
)
|
||||
|
||||
|
||||
class GeminiAPIError(Exception):
|
||||
"""Error shape compatible with Hermes retry/error classification."""
|
||||
|
||||
@@ -270,8 +306,12 @@ def _translate_tool_call_to_gemini(tool_call: Dict[str, Any]) -> Dict[str, Any]:
|
||||
}
|
||||
}
|
||||
thought_signature = _tool_call_extra_signature(tool_call)
|
||||
if thought_signature:
|
||||
part["thoughtSignature"] = thought_signature
|
||||
# Fallback sentinel for cross-provider tool_calls (e.g. fallback from
|
||||
# xAI/Anthropic to Gemini, where the original tool_call carries no
|
||||
# Gemini thoughtSignature). Mirrors gemini_cloudcode_adapter.py:106.
|
||||
# Without this, Gemini 3 thinking models reject replayed history with
|
||||
# 400 INVALID_ARGUMENT on the missing thoughtSignature.
|
||||
part["thoughtSignature"] = thought_signature or "skip_thought_signature_validator"
|
||||
return part
|
||||
|
||||
|
||||
@@ -281,9 +321,13 @@ def _translate_tool_result_to_gemini(
|
||||
) -> Dict[str, Any]:
|
||||
tool_name_by_call_id = tool_name_by_call_id or {}
|
||||
tool_call_id = str(message.get("tool_call_id") or "")
|
||||
# A tool result can carry the unwrapped internal tool name (for example,
|
||||
# an MCP tool invoked through the `tool_call` bridge). Gemini requires
|
||||
# functionResponse.name to echo the matching functionCall.name, so the
|
||||
# call-id mapping must take precedence over the internal result name.
|
||||
name = str(
|
||||
message.get("name")
|
||||
or tool_name_by_call_id.get(tool_call_id)
|
||||
tool_name_by_call_id.get(tool_call_id)
|
||||
or message.get("name")
|
||||
or tool_call_id
|
||||
or "tool"
|
||||
)
|
||||
@@ -820,6 +864,12 @@ def gemini_http_error(
|
||||
if status == 429 and is_free_tier_quota_error(err_message or body_text):
|
||||
message = message + _FREE_TIER_GUIDANCE
|
||||
|
||||
# Legacy "Standard" Google Cloud key rejection (June 19, 2026 onward) ->
|
||||
# Google's raw 401 misleadingly tells the user to use OAuth. Append the
|
||||
# actual fix (mint a new Gemini API key in AI Studio).
|
||||
if is_standard_key_auth_error(status, err_message or body_text, reason):
|
||||
message = message + _STANDARD_KEY_GUIDANCE
|
||||
|
||||
return GeminiAPIError(
|
||||
message,
|
||||
code=code,
|
||||
@@ -930,7 +980,7 @@ class GeminiNativeClient:
|
||||
def _create_chat_completion(
|
||||
self,
|
||||
*,
|
||||
model: str = "gemini-2.5-flash",
|
||||
model: str = "gemini-3.6-flash",
|
||||
messages: Optional[List[Dict[str, Any]]] = None,
|
||||
stream: bool = False,
|
||||
tools: Any = None,
|
||||
|
||||
+23
-6
@@ -2,6 +2,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Any, Dict
|
||||
|
||||
# Gemini's ``FunctionDeclaration.parameters`` field accepts the ``Schema``
|
||||
@@ -76,15 +77,31 @@ def sanitize_gemini_schema(schema: Any) -> Dict[str, Any]:
|
||||
|
||||
# Gemini's Schema validator requires every ``enum`` entry to be a string,
|
||||
# even when the parent ``type`` is ``integer`` / ``number`` / ``boolean``.
|
||||
# OpenAI / OpenRouter / Anthropic accept typed enums (e.g. Discord's
|
||||
# ``auto_archive_duration: {type: integer, enum: [60, 1440, 4320, 10080]}``),
|
||||
# so we only drop the ``enum`` when it would collide with Gemini's rule.
|
||||
# Keeping ``type: integer`` plus the human-readable description gives the
|
||||
# model enough guidance; the tool handler still validates the value.
|
||||
# Preserve those constraints by stringifying scalar values while keeping
|
||||
# the declared type intact; Gemini uses the strings as schema metadata and
|
||||
# still emits typed tool arguments at runtime.
|
||||
enum_val = cleaned.get("enum")
|
||||
type_val = cleaned.get("type")
|
||||
if isinstance(enum_val, list) and type_val in {"integer", "number", "boolean"}:
|
||||
if any(not isinstance(item, str) for item in enum_val):
|
||||
stringified = []
|
||||
for item in enum_val:
|
||||
if isinstance(item, str):
|
||||
value = item
|
||||
elif isinstance(item, bool):
|
||||
value = "true" if item else "false"
|
||||
elif (
|
||||
isinstance(item, (int, float))
|
||||
and not isinstance(item, bool)
|
||||
and math.isfinite(item)
|
||||
):
|
||||
value = str(item)
|
||||
else:
|
||||
continue
|
||||
if value not in stringified:
|
||||
stringified.append(value)
|
||||
if stringified:
|
||||
cleaned["enum"] = stringified
|
||||
else:
|
||||
cleaned.pop("enum", None)
|
||||
|
||||
# Gemini validates ``required`` strictly against the same node's
|
||||
|
||||
+9
-29
@@ -25,14 +25,14 @@ Language resolution order:
|
||||
3. ``display.language`` from config.yaml
|
||||
4. ``"en"`` (baseline)
|
||||
|
||||
Supported languages: en, zh, ja, de, es, fr, tr, uk. Unknown values fall back to en.
|
||||
Supported languages: en, zh, zh-hant, ja, de, es, fr, tr, uk, af, ko, it, ga,
|
||||
pt, ru, hu, ar. Unknown values fall back to en.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sysconfig
|
||||
import threading
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
@@ -42,7 +42,7 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
SUPPORTED_LANGUAGES: tuple[str, ...] = (
|
||||
"en", "zh", "zh-hant", "ja", "de", "es", "fr", "tr", "uk",
|
||||
"af", "ko", "it", "ga", "pt", "ru", "hu",
|
||||
"af", "ko", "it", "ga", "pt", "ru", "hu", "ar",
|
||||
)
|
||||
DEFAULT_LANGUAGE = "en"
|
||||
|
||||
@@ -79,6 +79,9 @@ _LANGUAGE_ALIASES: dict[str, str] = {
|
||||
"russian": "ru", "русский": "ru", "ru-ru": "ru",
|
||||
# Hungarian
|
||||
"hungarian": "hu", "magyar": "hu", "hu-hu": "hu",
|
||||
# Arabic — bare "arabic"/endonym plus the common regional BCP-47 tags.
|
||||
"arabic": "ar", "العربية": "ar",
|
||||
"ar-sa": "ar", "ar-eg": "ar", "ar-ae": "ar", "ar-ma": "ar", "ar-dz": "ar",
|
||||
}
|
||||
|
||||
_catalog_cache: dict[str, dict[str, str]] = {}
|
||||
@@ -92,12 +95,8 @@ def _locales_dir() -> Path:
|
||||
|
||||
1. ``HERMES_BUNDLED_LOCALES`` env var -- set by the Nix wrapper (or any
|
||||
sealed-packaging system) to point at the installed catalog directory.
|
||||
2. ``<repo-root>/locales`` -- source checkouts and ``pip install -e .``,
|
||||
2. ``<repo-root>/locales`` -- source checkouts and editable installs,
|
||||
where the working tree sits next to ``agent/``.
|
||||
3. ``<sysconfig data|purelib|platlib>/locales`` -- pip wheel installs.
|
||||
setuptools ``data-files`` extracts ``locales/*.yaml`` under the
|
||||
interpreter's ``data`` scheme; the other schemes are checked as a
|
||||
safety net for nonstandard layouts.
|
||||
|
||||
Falling through to the source-style path (even when missing) keeps
|
||||
``_load_catalog`` error messages informative -- it logs the path it
|
||||
@@ -116,25 +115,6 @@ def _locales_dir() -> Path:
|
||||
|
||||
# agent/i18n.py -> agent/ -> repo root (source checkout, editable install)
|
||||
source_dir = Path(__file__).resolve().parent.parent / "locales"
|
||||
if source_dir.is_dir():
|
||||
return source_dir
|
||||
|
||||
# pip wheel install: data-files lands under the interpreter data scheme.
|
||||
# ``data`` (== sys.prefix in a venv) is where setuptools data-files extract
|
||||
# and is checked first. ``purelib``/``platlib`` (site-packages) are a safety
|
||||
# net for nonstandard layouts. NOTE: this does NOT cover ``pip install
|
||||
# --user`` (user scheme, ~/.local/locales) or ``pip install --target`` --
|
||||
# both are out of scope; see the plan header.
|
||||
for scheme in ("data", "purelib", "platlib"):
|
||||
raw = sysconfig.get_path(scheme)
|
||||
if not raw:
|
||||
continue
|
||||
candidate = Path(raw) / "locales"
|
||||
if candidate.is_dir():
|
||||
return candidate
|
||||
|
||||
# Last resort: return the source-style path so _load_catalog's catalog-missing
|
||||
# log (logger.debug "i18n catalog missing for %s at %s") stays informative.
|
||||
return source_dir
|
||||
|
||||
|
||||
@@ -217,8 +197,8 @@ def _config_language_cached() -> str | None:
|
||||
(e.g. after the setup wizard).
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
cfg = load_config()
|
||||
from hermes_cli.config import load_config_readonly
|
||||
cfg = load_config_readonly()
|
||||
lang = (cfg.get("display") or {}).get("language")
|
||||
if lang:
|
||||
return _normalize_lang(lang)
|
||||
|
||||
@@ -91,9 +91,9 @@ def get_active_provider() -> Optional[ImageGenProvider]:
|
||||
"""
|
||||
configured: Optional[str] = None
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
from hermes_cli.config import load_config_readonly
|
||||
|
||||
cfg = load_config()
|
||||
cfg = load_config_readonly()
|
||||
section = cfg.get("image_gen") if isinstance(cfg, dict) else None
|
||||
if isinstance(section, dict):
|
||||
raw = section.get("provider")
|
||||
|
||||
+89
-35
@@ -181,6 +181,8 @@ def _supports_vision_override(
|
||||
cfg: Optional[Dict[str, Any]],
|
||||
provider: str,
|
||||
model: str,
|
||||
*,
|
||||
requested_provider: str = "",
|
||||
) -> Optional[bool]:
|
||||
"""Resolve user-declared vision capability from config.yaml.
|
||||
|
||||
@@ -188,9 +190,14 @@ def _supports_vision_override(
|
||||
1. ``model.supports_vision`` (top-level shortcut for the active model)
|
||||
2. ``providers.<provider>.models.<model>.supports_vision``
|
||||
(named custom providers — ``provider`` may be the runtime-resolved
|
||||
value ``"custom"`` and/or the user-declared name under
|
||||
``model.provider``; both are tried. For ``custom:<name>`` syntax,
|
||||
the stripped ``<name>`` is also tried as a provider key.)
|
||||
value ``"custom"``, the runtime's originally requested provider,
|
||||
and/or the user-declared name under ``model.provider``; all are
|
||||
tried. For ``custom:<name>`` syntax, the stripped ``<name>`` is also
|
||||
tried as a provider key.)
|
||||
2b. ``custom_providers`` (legacy list form) ``.models.<model>``
|
||||
|
||||
Under (2) and (2b), the per-model capability key may be written as
|
||||
either ``supports_vision`` or the shorter ``vision`` alias; both work.
|
||||
|
||||
Returns None when no override is set, so the caller falls through to
|
||||
models.dev. Returns False explicitly only when the user wrote a
|
||||
@@ -210,23 +217,30 @@ def _supports_vision_override(
|
||||
# get rewritten to provider="custom" at runtime
|
||||
# (hermes_cli/runtime_provider.py:_resolve_named_custom_runtime), so the
|
||||
# config still holds the user-declared name under model.provider. Try
|
||||
# both as candidate provider keys, plus the stripped suffix from
|
||||
# "custom:<name>" (where <name> is the key under providers:).
|
||||
# both as candidate provider keys. Either identity may use the
|
||||
# "custom:<name>" form while providers: is keyed by bare <name>.
|
||||
config_provider = str(model_cfg.get("provider") or "").strip()
|
||||
# Extract the stripped name from "custom:<name>" if present
|
||||
stripped_suffix = ""
|
||||
if config_provider.startswith("custom:"):
|
||||
stripped_suffix = config_provider[len("custom:"):]
|
||||
provider_candidates: List[str] = []
|
||||
for candidate in (requested_provider, provider, config_provider):
|
||||
if not candidate:
|
||||
continue
|
||||
provider_candidates.append(candidate)
|
||||
if candidate.startswith("custom:"):
|
||||
stripped_candidate = candidate[len("custom:"):]
|
||||
if stripped_candidate:
|
||||
provider_candidates.append(stripped_candidate)
|
||||
providers_raw = cfg.get("providers")
|
||||
providers_cfg: Dict[str, Any] = providers_raw if isinstance(providers_raw, dict) else {}
|
||||
for p in dict.fromkeys(filter(None, (provider, config_provider, stripped_suffix))):
|
||||
for p in dict.fromkeys(provider_candidates):
|
||||
entry_raw = providers_cfg.get(p)
|
||||
entry: Dict[str, Any] = entry_raw if isinstance(entry_raw, dict) else {}
|
||||
models_raw = entry.get("models")
|
||||
models_cfg: Dict[str, Any] = models_raw if isinstance(models_raw, dict) else {}
|
||||
per_model_raw = models_cfg.get(model)
|
||||
per_model: Dict[str, Any] = per_model_raw if isinstance(per_model_raw, dict) else {}
|
||||
coerced = _coerce_capability_bool(per_model.get("supports_vision"))
|
||||
coerced = _coerce_capability_bool(
|
||||
per_model.get("supports_vision", per_model.get("vision"))
|
||||
)
|
||||
if coerced is not None:
|
||||
return coerced
|
||||
|
||||
@@ -235,28 +249,26 @@ def _supports_vision_override(
|
||||
# may appear as the raw name or "custom:<name>" at runtime).
|
||||
custom_providers = cfg.get("custom_providers")
|
||||
if isinstance(custom_providers, list):
|
||||
# Build candidate names: the provider value and the config provider
|
||||
# value, both raw and with "custom:" prefix stripped/added.
|
||||
candidate_names: set = set()
|
||||
for p in filter(None, (provider, config_provider)):
|
||||
candidate_names.add(p)
|
||||
if p.startswith("custom:"):
|
||||
candidate_names.add(p[len("custom:"):])
|
||||
else:
|
||||
candidate_names.add(f"custom:{p}")
|
||||
for entry_raw in custom_providers:
|
||||
if not isinstance(entry_raw, dict):
|
||||
continue
|
||||
entry_name = str(entry_raw.get("name") or "").strip()
|
||||
if entry_name not in candidate_names:
|
||||
continue
|
||||
models_raw = entry_raw.get("models")
|
||||
models_cfg = models_raw if isinstance(models_raw, dict) else {}
|
||||
per_model_raw = models_cfg.get(model)
|
||||
per_model = per_model_raw if isinstance(per_model_raw, dict) else {}
|
||||
coerced = _coerce_capability_bool(per_model.get("supports_vision"))
|
||||
if coerced is not None:
|
||||
return coerced
|
||||
# Candidate priority matters when the CLI-selected provider differs
|
||||
# from model.provider. Walk identities first, then config entries, so
|
||||
# list order cannot let the persisted default shadow the live route.
|
||||
for candidate in dict.fromkeys(provider_candidates):
|
||||
candidate_name = candidate.strip().lower()
|
||||
for entry_raw in custom_providers:
|
||||
if not isinstance(entry_raw, dict):
|
||||
continue
|
||||
entry_name = str(entry_raw.get("name") or "").strip().lower()
|
||||
if entry_name != candidate_name:
|
||||
continue
|
||||
models_raw = entry_raw.get("models")
|
||||
models_cfg = models_raw if isinstance(models_raw, dict) else {}
|
||||
per_model_raw = models_cfg.get(model)
|
||||
per_model = per_model_raw if isinstance(per_model_raw, dict) else {}
|
||||
coerced = _coerce_capability_bool(
|
||||
per_model.get("supports_vision", per_model.get("vision"))
|
||||
)
|
||||
if coerced is not None:
|
||||
return coerced
|
||||
|
||||
return None
|
||||
|
||||
@@ -376,6 +388,8 @@ def _lookup_supports_vision(
|
||||
provider: str,
|
||||
model: str,
|
||||
cfg: Optional[Dict[str, Any]] = None,
|
||||
*,
|
||||
requested_provider: str = "",
|
||||
) -> Optional[bool]:
|
||||
"""Return True/False if we can resolve caps, None if unknown.
|
||||
|
||||
@@ -383,7 +397,34 @@ def _lookup_supports_vision(
|
||||
(so custom/local models declared as vision-capable don't fall through to
|
||||
text routing in ``auto`` mode), then falls back to models.dev.
|
||||
"""
|
||||
override = _supports_vision_override(cfg, provider, model)
|
||||
# Named custom providers are canonicalized to ``provider="custom"`` by
|
||||
# runtime resolution. The original CLI/config name is carried in the
|
||||
# context-local main runtime so capability lookup can still select the
|
||||
# exact custom_providers entry. Require an exact provider+model match:
|
||||
# background/auxiliary lookups must never borrow another turn's identity.
|
||||
if not requested_provider:
|
||||
try:
|
||||
from agent.auxiliary_client import _runtime_main_value
|
||||
|
||||
runtime_provider = str(
|
||||
_runtime_main_value("provider") or ""
|
||||
).strip().lower()
|
||||
runtime_model = str(_runtime_main_value("model") or "").strip()
|
||||
lookup_provider = str(provider or "").strip().lower()
|
||||
lookup_model = str(model or "").strip()
|
||||
if runtime_provider == lookup_provider and runtime_model == lookup_model:
|
||||
requested_provider = str(
|
||||
_runtime_main_value("requested_provider") or ""
|
||||
).strip()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
override = _supports_vision_override(
|
||||
cfg,
|
||||
provider,
|
||||
model,
|
||||
requested_provider=requested_provider,
|
||||
)
|
||||
if override is not None:
|
||||
return override
|
||||
if not provider or not model:
|
||||
@@ -421,6 +462,8 @@ def decide_image_input_mode(
|
||||
provider: str,
|
||||
model: str,
|
||||
cfg: Optional[Dict[str, Any]],
|
||||
*,
|
||||
requested_provider: str = "",
|
||||
) -> str:
|
||||
"""Return ``"native"`` or ``"text"`` for the given turn.
|
||||
|
||||
@@ -428,6 +471,7 @@ def decide_image_input_mode(
|
||||
provider: active inference provider ID (e.g. ``"anthropic"``, ``"openrouter"``).
|
||||
model: active model slug as it would be sent to the provider.
|
||||
cfg: loaded config.yaml dict, or None. When None, behaves as auto.
|
||||
requested_provider: provider identity before runtime canonicalization.
|
||||
"""
|
||||
mode_cfg = "auto"
|
||||
if isinstance(cfg, dict):
|
||||
@@ -444,7 +488,17 @@ def decide_image_input_mode(
|
||||
# explicit auxiliary.vision config acts as a *fallback* for text-only
|
||||
# main models — it should not preempt native vision on a model that
|
||||
# can natively inspect the pixels (issue #29135).
|
||||
supports = _lookup_supports_vision(provider, model, cfg)
|
||||
if requested_provider:
|
||||
supports = _lookup_supports_vision(
|
||||
provider,
|
||||
model,
|
||||
cfg,
|
||||
requested_provider=requested_provider,
|
||||
)
|
||||
else:
|
||||
# Keep the long-standing three-argument call contract for callers and
|
||||
# tests that replace the capability lookup hook.
|
||||
supports = _lookup_supports_vision(provider, model, cfg)
|
||||
if supports is True:
|
||||
return "native"
|
||||
if _explicit_aux_vision_override(cfg):
|
||||
|
||||
@@ -113,6 +113,13 @@ class InsightsEngine:
|
||||
"""
|
||||
cutoff = time.time() - (days * 86400)
|
||||
|
||||
# Token/cost totals may still sit on the SessionDB's async
|
||||
# accounting queue; drain so the report reflects exact counters.
|
||||
# (self.db may be a raw sqlite3 connection in tests — guard.)
|
||||
flush = getattr(self.db, "flush_token_counts", None)
|
||||
if callable(flush):
|
||||
flush()
|
||||
|
||||
# Gather raw data
|
||||
sessions = self._get_sessions(cutoff, source)
|
||||
tool_usage = self._get_tool_usage(cutoff, source)
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
Extracted from ``run_agent.py``. Each ``AIAgent`` instance (parent or
|
||||
subagent) holds an :class:`IterationBudget`; the parent's cap comes from
|
||||
``max_iterations`` (default 90), each subagent's cap comes from
|
||||
``max_iterations`` (default 500), each subagent's cap comes from
|
||||
``delegation.max_iterations`` (default 50).
|
||||
|
||||
``run_agent`` re-exports ``IterationBudget`` so existing
|
||||
@@ -18,7 +18,7 @@ class IterationBudget:
|
||||
"""Thread-safe iteration counter for an agent.
|
||||
|
||||
Each agent (parent or subagent) gets its own ``IterationBudget``.
|
||||
The parent's budget is capped at ``max_iterations`` (default 90).
|
||||
The parent's budget is capped at ``max_iterations`` (default 500).
|
||||
Each subagent gets an independent budget capped at
|
||||
``delegation.max_iterations`` (default 50) — this means total
|
||||
iterations across parent + subagents can exceed the parent's cap.
|
||||
|
||||
@@ -403,7 +403,6 @@ def _category_counts(payload: dict[str, Any]) -> list[tuple[str, int]]:
|
||||
def category_color_map(payload: dict[str, Any]) -> dict[str, str]:
|
||||
"""Deterministic, evenly-spread hue per skill category (theme-independent)."""
|
||||
clusters = _category_counts(payload)
|
||||
n = max(1, len(clusters))
|
||||
# Golden-angle hue spacing so adjacent categories never collide in color.
|
||||
return {cat: rgb_to_hex(_hsl_to_rgb((i * 137.508) % 360, 0.55, 0.62)) for i, (cat, _c) in enumerate(clusters)}
|
||||
|
||||
|
||||
+1
-1
@@ -55,7 +55,7 @@ def register_subparser(subparsers: argparse._SubParsersAction) -> None:
|
||||
help="Even attempt servers marked manual-install (best effort)",
|
||||
)
|
||||
|
||||
sub_restart = sub.add_parser(
|
||||
sub.add_parser(
|
||||
"restart",
|
||||
help="Tear down running LSP clients (next edit re-spawns)",
|
||||
)
|
||||
|
||||
+149
-71
@@ -18,9 +18,15 @@ into it via :func:`agent.lsp.manager.LSPService.touch_file`.
|
||||
|
||||
Implementation notes:
|
||||
|
||||
- Push diagnostics are stored per-URI in :attr:`_push_diagnostics` from
|
||||
``textDocument/publishDiagnostics`` notifications. Pull diagnostics
|
||||
go in :attr:`_pull_diagnostics`. The merged view dedupes by content.
|
||||
- All per-document state lives in one :class:`_DocState` keyed by
|
||||
absolute path. Freshness is tracked with **document versions**,
|
||||
not timestamps: every didChange bumps ``version``, and each stored
|
||||
push/pull result is tagged with the version it describes. A
|
||||
result is fresh iff its tag >= the version being waited on, so a
|
||||
didChange implicitly invalidates everything older — no clearing,
|
||||
no clock comparisons, no race windows. This is what prevents
|
||||
"ghost diagnostics": a slow server's leftovers from the previous
|
||||
edit can never masquerade as a verdict on the current content.
|
||||
|
||||
- Whole-document sync. Even when the server advertises incremental
|
||||
sync, we send a single ``contentChanges`` entry replacing the
|
||||
@@ -45,10 +51,13 @@ import asyncio
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Awaitable, Callable, Dict, List, Optional, Set
|
||||
from urllib.parse import quote, unquote
|
||||
|
||||
from hermes_cli._subprocess_compat import windows_hide_flags
|
||||
|
||||
from agent.lsp.protocol import (
|
||||
ERROR_CONTENT_MODIFIED,
|
||||
ERROR_METHOD_NOT_FOUND,
|
||||
@@ -124,6 +133,40 @@ def _end_position(text: str) -> Dict[str, int]:
|
||||
return {"line": last_line, "character": last_col}
|
||||
|
||||
|
||||
@dataclass
|
||||
class _DocState:
|
||||
"""Everything the client tracks for one open document.
|
||||
|
||||
``version`` is the LSP document version we last sent (didOpen=0,
|
||||
each didChange +1). It doubles as the freshness token: stored
|
||||
push/pull results are tagged with the version they describe
|
||||
(``push_version`` / ``pull_version``), and a result is *fresh*
|
||||
iff its tag has caught up to ``version``. Bumping the version on
|
||||
didChange therefore invalidates all older results implicitly —
|
||||
no store-clearing, no timestamps.
|
||||
|
||||
``push_version``/``pull_version`` start at -1 = "no data yet".
|
||||
Servers that echo a document version in publishDiagnostics get
|
||||
exact tagging; those that don't are credited with the current
|
||||
version at receipt time (a push observed after we sent the
|
||||
change describes the changed content or newer).
|
||||
"""
|
||||
|
||||
version: int = 0
|
||||
text: str = ""
|
||||
push: List[Dict[str, Any]] = field(default_factory=list)
|
||||
pull: List[Dict[str, Any]] = field(default_factory=list)
|
||||
push_version: int = -1
|
||||
pull_version: int = -1
|
||||
seed_seen: bool = False
|
||||
|
||||
def fresh_push(self, version: Optional[int] = None) -> bool:
|
||||
return self.push_version >= (self.version if version is None else version)
|
||||
|
||||
def fresh_pull(self, version: Optional[int] = None) -> bool:
|
||||
return self.pull_version >= (self.version if version is None else version)
|
||||
|
||||
|
||||
class LSPClient:
|
||||
"""Async LSP client tied to one server process and one workspace root.
|
||||
|
||||
@@ -186,18 +229,10 @@ class LSPClient:
|
||||
# is silently dropped by default.
|
||||
}
|
||||
|
||||
# Tracked file state — required for didChange version bumps.
|
||||
self._files: Dict[str, Dict[str, Any]] = {}
|
||||
# Diagnostic stores, keyed by file path (NOT URI).
|
||||
self._push_diagnostics: Dict[str, List[Dict[str, Any]]] = {}
|
||||
self._pull_diagnostics: Dict[str, List[Dict[str, Any]]] = {}
|
||||
# Per-path "last published" time so wait-for-fresh logic works.
|
||||
self._published: Dict[str, float] = {}
|
||||
# Per-path version of the latest push (matches our didChange
|
||||
# version when the server respects it).
|
||||
self._published_version: Dict[str, int] = {}
|
||||
# First-push seen flag, for typescript-style seed-on-first-push.
|
||||
self._first_push_seen: Set[str] = set()
|
||||
# Per-document state (version, text, diagnostic stores, and
|
||||
# their freshness tags), keyed by absolute file path (NOT URI).
|
||||
# See _DocState for the version-based freshness model.
|
||||
self._docs: Dict[str, _DocState] = {}
|
||||
# Capability registrations — only diagnostic ones are tracked.
|
||||
self._diagnostic_registrations: Dict[str, Dict[str, Any]] = {}
|
||||
|
||||
@@ -261,6 +296,12 @@ class LSPClient:
|
||||
cmd = self._command
|
||||
if sys.platform == "win32":
|
||||
cmd = self._win_wrap_cmd(cmd)
|
||||
# Suppress the cmd.exe console window that would otherwise flash
|
||||
# every time we launch a ``.cmd``-wrapped language server
|
||||
# (e.g. pyright-langserver.CMD) from a console-less host such as
|
||||
# a VS Code/Zed extension running the ACP adapter.
|
||||
# windows_hide_flags() is CREATE_NO_WINDOW on Windows, 0 on POSIX.
|
||||
creationflags = windows_hide_flags()
|
||||
|
||||
try:
|
||||
# start_new_session=True detaches the LSP server into its own
|
||||
@@ -279,6 +320,7 @@ class LSPClient:
|
||||
env=env,
|
||||
cwd=self._cwd,
|
||||
start_new_session=True,
|
||||
creationflags=creationflags,
|
||||
)
|
||||
except FileNotFoundError as e:
|
||||
raise LSPProtocolError(
|
||||
@@ -647,25 +689,25 @@ class LSPClient:
|
||||
if not isinstance(diagnostics, list):
|
||||
diagnostics = []
|
||||
version = params.get("version")
|
||||
loop_time = asyncio.get_event_loop().time()
|
||||
|
||||
if self._seed_first_push and path not in self._first_push_seen:
|
||||
# First push: seed without firing the event so a waiter
|
||||
# doesn't resolve on the very first push (which arrives
|
||||
# before the user-triggered didChange could've produced
|
||||
# fresh diagnostics).
|
||||
self._first_push_seen.add(path)
|
||||
self._push_diagnostics[path] = diagnostics
|
||||
self._published[path] = loop_time
|
||||
if isinstance(version, int):
|
||||
self._published_version[path] = version
|
||||
doc = self._docs.setdefault(path, _DocState(version=-1))
|
||||
if self._seed_first_push and not doc.seed_seen:
|
||||
# First push: seed the store WITHOUT a freshness tag. It
|
||||
# arrives before the user-triggered didChange could've
|
||||
# produced fresh diagnostics, so it must never satisfy a
|
||||
# waiter — it's baseline data only.
|
||||
doc.seed_seen = True
|
||||
doc.push = diagnostics
|
||||
return
|
||||
|
||||
self._push_diagnostics[path] = diagnostics
|
||||
self._published[path] = loop_time
|
||||
if isinstance(version, int):
|
||||
self._published_version[path] = version
|
||||
self._first_push_seen.add(path)
|
||||
doc.seed_seen = True
|
||||
doc.push = diagnostics
|
||||
# Tag with the echoed document version when the server provides
|
||||
# one; otherwise credit the current version — a push observed
|
||||
# after we sent the change describes the changed content (or
|
||||
# newer). Note doc.version is -1 for never-opened paths
|
||||
# (e.g. relatedDocuments spillover), keeping them unfresh.
|
||||
doc.push_version = version if isinstance(version, int) else doc.version
|
||||
# Bump the monotonic push counter and wake every waiter. We
|
||||
# keep the Event sticky-set so any wait already in progress
|
||||
# resolves; waiters re-check their predicate after waking and
|
||||
@@ -694,16 +736,16 @@ class LSPClient:
|
||||
raise LSPProtocolError(f"cannot read {abs_path}: {e}") from e
|
||||
|
||||
uri = file_uri(abs_path)
|
||||
existing = self._files.get(abs_path)
|
||||
doc = self._docs.get(abs_path)
|
||||
|
||||
if existing is not None:
|
||||
if doc is not None and doc.version >= 0:
|
||||
# Re-open: bump version, fire didChangeWatchedFiles + didChange.
|
||||
await self._send_notification(
|
||||
"workspace/didChangeWatchedFiles",
|
||||
{"changes": [{"uri": uri, "type": 2}]}, # 2 = CHANGED
|
||||
)
|
||||
new_version = existing["version"] + 1
|
||||
old_text = existing["text"]
|
||||
new_version = doc.version + 1
|
||||
old_text = doc.text
|
||||
content_changes: List[Dict[str, Any]]
|
||||
if self._sync_kind == 2:
|
||||
content_changes = [
|
||||
@@ -724,7 +766,11 @@ class LSPClient:
|
||||
"contentChanges": content_changes,
|
||||
},
|
||||
)
|
||||
self._files[abs_path] = {"version": new_version, "text": text}
|
||||
# Bumping the version is the whole invalidation story:
|
||||
# every stored result tagged with an older version is now
|
||||
# stale by definition (see _DocState).
|
||||
doc.version = new_version
|
||||
doc.text = text
|
||||
return new_version
|
||||
|
||||
# First open: didChangeWatchedFiles CREATED + didOpen.
|
||||
@@ -732,12 +778,9 @@ class LSPClient:
|
||||
"workspace/didChangeWatchedFiles",
|
||||
{"changes": [{"uri": uri, "type": 1}]}, # 1 = CREATED
|
||||
)
|
||||
# Clear any stale push/pull entries — fresh open should start
|
||||
# from scratch.
|
||||
self._push_diagnostics.pop(abs_path, None)
|
||||
self._pull_diagnostics.pop(abs_path, None)
|
||||
self._published.pop(abs_path, None)
|
||||
self._published_version.pop(abs_path, None)
|
||||
# Fresh doc state — anything stashed under this path by a
|
||||
# pre-open push (relatedDocuments spillover etc.) is discarded.
|
||||
self._docs[abs_path] = _DocState(version=0, text=text)
|
||||
await self._send_notification(
|
||||
"textDocument/didOpen",
|
||||
{
|
||||
@@ -749,7 +792,6 @@ class LSPClient:
|
||||
}
|
||||
},
|
||||
)
|
||||
self._files[abs_path] = {"version": 0, "text": text}
|
||||
return 0
|
||||
|
||||
async def save_file(self, path: str) -> None:
|
||||
@@ -769,12 +811,19 @@ class LSPClient:
|
||||
async def _pull_document_diagnostics(self, path: str) -> None:
|
||||
"""Send ``textDocument/diagnostic`` for one file.
|
||||
|
||||
Stores results into :attr:`_pull_diagnostics`. Silently
|
||||
no-ops on errors (server may not support the pull endpoint).
|
||||
Stores results into the doc's pull store, tagged with the
|
||||
document version captured at request send time. If a didChange
|
||||
races past the in-flight request, the version bump makes the
|
||||
stored result stale automatically — no explicit invalidation.
|
||||
Silently no-ops on errors (server may not support the pull
|
||||
endpoint).
|
||||
"""
|
||||
abs_path = os.path.abspath(path)
|
||||
doc = self._docs.get(abs_path)
|
||||
sent_version = doc.version if doc else -1
|
||||
try:
|
||||
params: Dict[str, Any] = {
|
||||
"textDocument": {"uri": file_uri(os.path.abspath(path))}
|
||||
"textDocument": {"uri": file_uri(abs_path)}
|
||||
}
|
||||
result = await self._send_request_with_retry(
|
||||
"textDocument/diagnostic",
|
||||
@@ -788,7 +837,9 @@ class LSPClient:
|
||||
return
|
||||
items = result.get("items")
|
||||
if isinstance(items, list):
|
||||
self._pull_diagnostics[os.path.abspath(path)] = items
|
||||
doc = self._docs.setdefault(abs_path, _DocState(version=-1))
|
||||
doc.pull = items
|
||||
doc.pull_version = sent_version
|
||||
related = result.get("relatedDocuments")
|
||||
if isinstance(related, dict):
|
||||
for uri, sub in related.items():
|
||||
@@ -796,7 +847,11 @@ class LSPClient:
|
||||
continue
|
||||
sub_items = sub.get("items")
|
||||
if isinstance(sub_items, list):
|
||||
self._pull_diagnostics[uri_to_path(uri)] = sub_items
|
||||
rel = self._docs.setdefault(uri_to_path(uri), _DocState(version=-1))
|
||||
rel.pull = sub_items
|
||||
# Same send-anchored tagging: fresh only if that
|
||||
# doc hasn't changed since the request went out.
|
||||
rel.pull_version = rel.version
|
||||
|
||||
async def wait_for_diagnostics(
|
||||
self,
|
||||
@@ -804,22 +859,36 @@ class LSPClient:
|
||||
version: int,
|
||||
*,
|
||||
mode: str = "document",
|
||||
) -> None:
|
||||
timeout: Optional[float] = None,
|
||||
) -> bool:
|
||||
"""Wait for the server to publish diagnostics for ``path`` at ``version``.
|
||||
|
||||
``mode`` is ``"document"`` (5s budget, document pulls) or
|
||||
``"full"`` (10s budget, also workspace pulls). Best-effort —
|
||||
returns silently on timeout. Does NOT throw if the server
|
||||
doesn't support pull diagnostics; we still get the push side.
|
||||
``"full"`` (10s budget, also workspace pulls). ``timeout``
|
||||
overrides the mode's default budget when provided — this is
|
||||
how the user's ``lsp.wait_timeout`` config reaches the wait
|
||||
loop (slow servers like tsserver on big projects need more
|
||||
than the 5s default).
|
||||
|
||||
Returns ``True`` when *fresh* diagnostics arrived (a push at
|
||||
or after our didChange, or a pull answered after it) and
|
||||
``False`` on timeout. Callers must treat ``False`` as "no
|
||||
data", NOT as "no errors" — the diagnostic stores may still
|
||||
hold stale entries from the previous edit at that point.
|
||||
Best-effort — never throws if the server doesn't support pull
|
||||
diagnostics; we still get the push side.
|
||||
"""
|
||||
budget = DIAGNOSTICS_FULL_WAIT if mode == "full" else DIAGNOSTICS_DOCUMENT_WAIT
|
||||
if timeout is not None and timeout > 0:
|
||||
budget = timeout
|
||||
else:
|
||||
budget = DIAGNOSTICS_FULL_WAIT if mode == "full" else DIAGNOSTICS_DOCUMENT_WAIT
|
||||
deadline = asyncio.get_event_loop().time() + budget
|
||||
abs_path = os.path.abspath(path)
|
||||
|
||||
while True:
|
||||
remaining = deadline - asyncio.get_event_loop().time()
|
||||
if remaining <= 0:
|
||||
return
|
||||
return False
|
||||
|
||||
# Concurrent: document pull + push wait.
|
||||
pull_task = asyncio.create_task(self._pull_document_diagnostics(abs_path))
|
||||
@@ -838,26 +907,24 @@ class LSPClient:
|
||||
pass
|
||||
|
||||
# If we got a fresh push for our version, we're done.
|
||||
current_v = self._published_version.get(abs_path)
|
||||
if abs_path in self._published and (
|
||||
current_v is None or current_v >= version
|
||||
):
|
||||
return
|
||||
doc = self._docs.get(abs_path)
|
||||
if doc and doc.fresh_push(version):
|
||||
return True
|
||||
|
||||
# Pull may have populated _pull_diagnostics — that's also
|
||||
# success.
|
||||
if abs_path in self._pull_diagnostics:
|
||||
return
|
||||
# Pull may have answered for the current version — that's
|
||||
# also success.
|
||||
if doc and doc.fresh_pull(version):
|
||||
return True
|
||||
|
||||
# Loop until budget runs out.
|
||||
|
||||
async def _wait_for_fresh_push(self, path: str, version: int, timeout: float) -> None:
|
||||
"""Wait until a publishDiagnostics arrives for ``path`` at ``version``+."""
|
||||
"""Wait until a fresh publishDiagnostics arrives for ``path`` at ``version``+."""
|
||||
deadline = asyncio.get_event_loop().time() + timeout
|
||||
baseline = self._push_counter
|
||||
while True:
|
||||
current_v = self._published_version.get(path)
|
||||
if path in self._published and (current_v is None or current_v >= version):
|
||||
doc = self._docs.get(path)
|
||||
if doc and doc.fresh_push(version):
|
||||
# Debounce — wait a tick in case more diagnostics arrive
|
||||
# immediately after. TS often emits in pairs. We
|
||||
# snapshot the counter so we wake on a *new* push, not
|
||||
@@ -888,17 +955,28 @@ class LSPClient:
|
||||
except asyncio.TimeoutError:
|
||||
continue
|
||||
|
||||
def diagnostics_for(self, path: str) -> List[Dict[str, Any]]:
|
||||
def diagnostics_for(self, path: str, *, fresh_only: bool = False) -> List[Dict[str, Any]]:
|
||||
"""Return current merged + deduped diagnostics for one file.
|
||||
|
||||
Diagnostics from push and pull stores are concatenated and
|
||||
deduplicated by ``(severity, code, message, range)`` content
|
||||
key. Empty list if the server hasn't published anything.
|
||||
|
||||
With ``fresh_only=True``, a store only contributes when its
|
||||
version tag has caught up to the document's current version —
|
||||
stale leftovers from the previous edit cycle are excluded.
|
||||
This is what report paths should use: after an edit, "stale
|
||||
errors" and "no errors" must not be conflated.
|
||||
"""
|
||||
abs_path = os.path.abspath(path)
|
||||
push = self._push_diagnostics.get(abs_path) or []
|
||||
pull = self._pull_diagnostics.get(abs_path) or []
|
||||
return _dedupe(push, pull)
|
||||
doc = self._docs.get(os.path.abspath(path))
|
||||
if doc is None:
|
||||
return []
|
||||
if fresh_only:
|
||||
return _dedupe(
|
||||
doc.push if doc.fresh_push() else [],
|
||||
doc.pull if doc.fresh_pull() else [],
|
||||
)
|
||||
return _dedupe(doc.push, doc.pull)
|
||||
|
||||
|
||||
def _dedupe(*lists: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
|
||||
+21
-1
@@ -40,7 +40,7 @@ from __future__ import annotations
|
||||
import logging
|
||||
import os
|
||||
import threading
|
||||
from typing import Tuple
|
||||
from typing import List, Tuple
|
||||
|
||||
# Dedicated logger name so the documented grep recipe survives a
|
||||
# ``logging.getLogger(__name__)`` rename of any internal module.
|
||||
@@ -188,6 +188,25 @@ def log_spawn_failed(server_id: str, workspace_root: str, exc: BaseException) ->
|
||||
)
|
||||
|
||||
|
||||
def log_reaped(keys: List[Tuple[str, str]], idle_timeout: float) -> None:
|
||||
"""Idle clients were shut down by the reaper. INFO — one line per
|
||||
sweep so users can correlate memory drops with LSP activity.
|
||||
|
||||
Also clears the ``log_active`` announce cache for the reaped keys so
|
||||
a later respawn re-announces at INFO instead of logging a misleading
|
||||
DEBUG "reused client".
|
||||
"""
|
||||
with _announce_lock:
|
||||
for key in keys:
|
||||
_announced_active.discard(key)
|
||||
summary = ", ".join(f"{sid} ({root})" for sid, root in keys)
|
||||
_emit(
|
||||
"reaper",
|
||||
logging.INFO,
|
||||
f"reaped {len(keys)} idle client(s) after {idle_timeout:.0f}s: {summary}",
|
||||
)
|
||||
|
||||
|
||||
def reset_announce_caches() -> None:
|
||||
"""Test-only: clear the dedup caches. Production code never calls this."""
|
||||
with _announce_lock:
|
||||
@@ -209,5 +228,6 @@ __all__ = [
|
||||
"log_timeout",
|
||||
"log_server_error",
|
||||
"log_spawn_failed",
|
||||
"log_reaped",
|
||||
"reset_announce_caches",
|
||||
]
|
||||
|
||||
@@ -30,11 +30,12 @@ import logging
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from hermes_cli._subprocess_compat import windows_hide_flags
|
||||
|
||||
logger = logging.getLogger("agent.lsp.install")
|
||||
|
||||
# Package-name → install-strategy hint registry. Each entry is a
|
||||
@@ -122,10 +123,9 @@ def _is_windows() -> bool:
|
||||
|
||||
def hermes_lsp_bin_dir() -> Path:
|
||||
"""Return the Hermes-owned bin staging dir for LSP servers."""
|
||||
home = os.environ.get("HERMES_HOME")
|
||||
if home is None:
|
||||
home = os.path.join(os.path.expanduser("~"), ".hermes")
|
||||
p = Path(home) / "lsp" / "bin"
|
||||
from hermes_constants import get_hermes_home
|
||||
|
||||
p = get_hermes_home() / "lsp" / "bin"
|
||||
p.mkdir(parents=True, exist_ok=True)
|
||||
return p
|
||||
|
||||
@@ -265,9 +265,10 @@ def _install_npm(
|
||||
[npm, "install", "--prefix", str(staging), "--silent", "--no-fund", "--no-audit", *install_targets],
|
||||
check=False,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
text=True, encoding="utf-8", errors="replace",
|
||||
timeout=300,
|
||||
stdin=subprocess.DEVNULL,
|
||||
creationflags=windows_hide_flags(),
|
||||
)
|
||||
if proc.returncode != 0:
|
||||
logger.warning(
|
||||
@@ -313,10 +314,11 @@ def _install_go(pkg: str, bin_name: str) -> Optional[str]:
|
||||
[go, "install", pkg],
|
||||
check=False,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
text=True, encoding="utf-8", errors="replace",
|
||||
timeout=600,
|
||||
env=env,
|
||||
stdin=subprocess.DEVNULL,
|
||||
creationflags=windows_hide_flags(),
|
||||
)
|
||||
if proc.returncode != 0:
|
||||
logger.warning(
|
||||
|
||||
+120
-15
@@ -59,6 +59,7 @@ from agent.lsp.workspace import (
|
||||
logger = logging.getLogger("agent.lsp.manager")
|
||||
|
||||
DEFAULT_IDLE_TIMEOUT = 600 # seconds; servers idle for >10min get reaped
|
||||
MIN_IDLE_TIMEOUT = 30 # floor for config values; must exceed any per-op wait budget
|
||||
|
||||
|
||||
class _BackgroundLoop:
|
||||
@@ -176,6 +177,7 @@ class LSPService:
|
||||
self._spawning: Dict[Tuple[str, str], asyncio.Future] = {}
|
||||
self._last_used: Dict[Tuple[str, str], float] = {}
|
||||
self._state_lock = threading.Lock()
|
||||
self._idle_reaper_task: Optional[asyncio.Task] = None
|
||||
|
||||
# Delta baseline: file path → snapshot of diagnostics taken
|
||||
# immediately before a write. ``get_diagnostics_sync`` filters
|
||||
@@ -183,6 +185,9 @@ class LSPService:
|
||||
# introduced by the current edit.
|
||||
self._delta_baseline: Dict[str, List[Dict[str, Any]]] = {}
|
||||
|
||||
if self._enabled and self._idle_timeout > 0:
|
||||
self._loop.run(self._start_idle_reaper(), timeout=2.0)
|
||||
|
||||
@classmethod
|
||||
def create_from_config(cls) -> Optional["LSPService"]:
|
||||
"""Build a service from ``hermes_cli.config`` settings.
|
||||
@@ -191,8 +196,8 @@ class LSPService:
|
||||
itself returns ``is_active()`` False when LSP is disabled.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
cfg = load_config()
|
||||
from hermes_cli.config import load_config_readonly
|
||||
cfg = load_config_readonly()
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("LSP config load failed: %s", e)
|
||||
return None
|
||||
@@ -205,6 +210,16 @@ class LSPService:
|
||||
wait_mode = lsp_cfg.get("wait_mode", "document")
|
||||
wait_timeout = float(lsp_cfg.get("wait_timeout", DIAGNOSTICS_DOCUMENT_WAIT))
|
||||
install_strategy = lsp_cfg.get("install_strategy", "auto")
|
||||
try:
|
||||
idle_timeout = float(lsp_cfg.get("idle_timeout", DEFAULT_IDLE_TIMEOUT))
|
||||
except (TypeError, ValueError):
|
||||
idle_timeout = DEFAULT_IDLE_TIMEOUT
|
||||
if 0 < idle_timeout < MIN_IDLE_TIMEOUT:
|
||||
# A timeout below the per-operation wait budget could reap a
|
||||
# client mid-flight; the resulting outer timeout would then
|
||||
# mark the (server, workspace) pair broken for the process
|
||||
# lifetime. Clamp to a safe floor (0 still disables).
|
||||
idle_timeout = MIN_IDLE_TIMEOUT
|
||||
servers_cfg = lsp_cfg.get("servers") or {}
|
||||
disabled = []
|
||||
binary_overrides: Dict[str, List[str]] = {}
|
||||
@@ -235,6 +250,7 @@ class LSPService:
|
||||
env_overrides=env_overrides,
|
||||
init_overrides=init_overrides,
|
||||
disabled_servers=disabled,
|
||||
idle_timeout=idle_timeout,
|
||||
)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
@@ -292,7 +308,10 @@ class LSPService:
|
||||
if not self.enabled_for(file_path):
|
||||
return
|
||||
try:
|
||||
diags = self._loop.run(self._snapshot_async(file_path), timeout=8.0)
|
||||
# Outer join budget must exceed the inner wait budget or a
|
||||
# slow-but-alive server gets falsely marked broken.
|
||||
t = max(8.0, self._wait_timeout + 3.0)
|
||||
diags = self._loop.run(self._snapshot_async(file_path), timeout=t)
|
||||
self._delta_baseline[os.path.abspath(file_path)] = diags or []
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("baseline snapshot failed for %s: %s", file_path, e)
|
||||
@@ -341,7 +360,7 @@ class LSPService:
|
||||
|
||||
try:
|
||||
t = timeout if timeout is not None else self._wait_timeout + 2.0
|
||||
diags = self._loop.run(self._open_and_wait_async(file_path), timeout=t) or []
|
||||
diags = self._loop.run(self._open_and_wait_async(file_path), timeout=t)
|
||||
except asyncio.TimeoutError as e:
|
||||
eventlog.log_timeout(server_id, file_path)
|
||||
logger.debug("LSP diagnostics timeout for %s: %s", file_path, e)
|
||||
@@ -353,6 +372,17 @@ class LSPService:
|
||||
self._mark_broken_for_file(file_path, e)
|
||||
return []
|
||||
|
||||
if diags is None:
|
||||
# The server is alive but never produced diagnostics for the
|
||||
# post-edit content within the wait budget (common for
|
||||
# tsserver on large projects). Report "no data" rather than
|
||||
# whatever stale state is in the stores — surfacing the
|
||||
# previous edit's errors as if they were current is the
|
||||
# ghost-diagnostics bug. The server is NOT marked broken:
|
||||
# slow is not dead, and the next edit may well succeed.
|
||||
eventlog.log_timeout(server_id, file_path, kind="fresh diagnostics")
|
||||
return []
|
||||
|
||||
abs_path = os.path.abspath(file_path)
|
||||
if delta:
|
||||
baseline = self._delta_baseline.get(abs_path) or []
|
||||
@@ -420,6 +450,7 @@ class LSPService:
|
||||
# ``_clients`` with a half-initialized state.
|
||||
with self._state_lock:
|
||||
client = self._clients.pop(key, None)
|
||||
self._last_used.pop(key, None)
|
||||
if client is not None:
|
||||
try:
|
||||
# Fire-and-forget shutdown — give it a second to cleanup,
|
||||
@@ -452,26 +483,43 @@ class LSPService:
|
||||
return []
|
||||
try:
|
||||
version = await client.open_file(file_path, language_id=language_id_for(file_path))
|
||||
await client.wait_for_diagnostics(file_path, version, mode=self._wait_mode)
|
||||
fresh = await client.wait_for_diagnostics(file_path, version, mode=self._wait_mode)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("snapshot open/wait failed: %s", e)
|
||||
return []
|
||||
self._last_used[(client.server_id, client.workspace_root)] = time.time()
|
||||
return list(client.diagnostics_for(file_path))
|
||||
self._touch(client)
|
||||
if not fresh:
|
||||
# No fresh data for the pre-edit content — an empty baseline
|
||||
# is safe: worst case the delta filter removes less, never
|
||||
# more. Never seed the baseline from stale stores.
|
||||
return []
|
||||
return list(client.diagnostics_for(file_path, fresh_only=True))
|
||||
|
||||
async def _open_and_wait_async(self, file_path: str) -> List[Dict[str, Any]]:
|
||||
async def _open_and_wait_async(self, file_path: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Open + wait for FRESH diagnostics.
|
||||
|
||||
Returns the fresh diagnostic list, or ``None`` when the server
|
||||
never produced post-change data within the wait budget. The
|
||||
distinction matters: ``[]`` means "server checked the new
|
||||
content, it's clean", ``None`` means "no verdict" — the caller
|
||||
must not substitute stale data for either.
|
||||
"""
|
||||
client = await self._get_or_spawn(file_path)
|
||||
if client is None:
|
||||
return []
|
||||
return None
|
||||
try:
|
||||
version = await client.open_file(file_path, language_id=language_id_for(file_path))
|
||||
await client.save_file(file_path)
|
||||
await client.wait_for_diagnostics(file_path, version, mode=self._wait_mode)
|
||||
fresh = await client.wait_for_diagnostics(
|
||||
file_path, version, mode=self._wait_mode, timeout=self._wait_timeout
|
||||
)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("open/wait failed for %s: %s", file_path, e)
|
||||
return []
|
||||
self._last_used[(client.server_id, client.workspace_root)] = time.time()
|
||||
return list(client.diagnostics_for(file_path))
|
||||
return None
|
||||
self._touch(client)
|
||||
if not fresh:
|
||||
return None
|
||||
return list(client.diagnostics_for(file_path, fresh_only=True))
|
||||
|
||||
async def _current_diags_async(self, file_path: str) -> List[Dict[str, Any]]:
|
||||
ws, gated = resolve_workspace_for_file(file_path)
|
||||
@@ -482,7 +530,7 @@ class LSPService:
|
||||
client = self._clients.get((srv.server_id, ws))
|
||||
if client is None:
|
||||
return []
|
||||
return list(client.diagnostics_for(file_path))
|
||||
return list(client.diagnostics_for(file_path, fresh_only=True))
|
||||
|
||||
async def _get_or_spawn(self, file_path: str) -> Optional[LSPClient]:
|
||||
srv = find_server_for_file(file_path)
|
||||
@@ -508,6 +556,7 @@ class LSPService:
|
||||
with self._state_lock:
|
||||
client = self._clients.get(key)
|
||||
if client is not None and client.is_running:
|
||||
self._last_used[key] = time.time()
|
||||
eventlog.log_active(srv.server_id, per_server_root)
|
||||
return client
|
||||
spawning = self._spawning.get(key)
|
||||
@@ -558,7 +607,7 @@ class LSPService:
|
||||
return None
|
||||
with self._state_lock:
|
||||
self._clients[key] = client
|
||||
self._last_used[key] = time.time()
|
||||
self._last_used[key] = time.time()
|
||||
eventlog.log_active(srv.server_id, per_server_root)
|
||||
spawn_future.set_result(client)
|
||||
return client
|
||||
@@ -566,7 +615,63 @@ class LSPService:
|
||||
with self._state_lock:
|
||||
self._spawning.pop(key, None)
|
||||
|
||||
async def _start_idle_reaper(self) -> None:
|
||||
self._idle_reaper_task = asyncio.create_task(self._idle_reaper_loop())
|
||||
|
||||
def _touch(self, client: LSPClient) -> None:
|
||||
"""Refresh the last-used timestamp for a client we just used.
|
||||
|
||||
Guarded on membership so a reaped-mid-operation client can't
|
||||
resurrect an orphan ``_last_used`` entry after the reaper popped
|
||||
the key. All writers and the reaper run on the background loop
|
||||
thread; the lock keeps this consistent with the reader anyway.
|
||||
"""
|
||||
key = (client.server_id, client.workspace_root)
|
||||
with self._state_lock:
|
||||
if key in self._clients:
|
||||
self._last_used[key] = time.time()
|
||||
|
||||
async def _idle_reaper_loop(self) -> None:
|
||||
interval = min(60.0, self._idle_timeout)
|
||||
while True:
|
||||
await asyncio.sleep(interval)
|
||||
try:
|
||||
await self._reap_idle_once()
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception as e: # noqa: BLE001
|
||||
# A transient sweep error must not kill the reaper —
|
||||
# otherwise one bad shutdown permanently re-opens the
|
||||
# unbounded-accumulation leak this loop exists to fix.
|
||||
logger.debug("LSP idle reaper sweep error: %s", e)
|
||||
|
||||
async def _reap_idle_once(self) -> None:
|
||||
cutoff = time.time() - self._idle_timeout
|
||||
with self._state_lock:
|
||||
idle_keys = [
|
||||
key
|
||||
for key in self._clients
|
||||
if self._last_used.get(key, 0) < cutoff
|
||||
]
|
||||
clients = [self._clients.pop(key) for key in idle_keys]
|
||||
for key in idle_keys:
|
||||
self._last_used.pop(key, None)
|
||||
if clients:
|
||||
eventlog.log_reaped(
|
||||
[(c.server_id, c.workspace_root) for c in clients],
|
||||
self._idle_timeout,
|
||||
)
|
||||
await asyncio.gather(
|
||||
*(client.shutdown() for client in clients),
|
||||
return_exceptions=True,
|
||||
)
|
||||
|
||||
async def _shutdown_async(self) -> None:
|
||||
reaper = self._idle_reaper_task
|
||||
self._idle_reaper_task = None
|
||||
if reaper is not None:
|
||||
reaper.cancel()
|
||||
await asyncio.gather(reaper, return_exceptions=True)
|
||||
with self._state_lock:
|
||||
clients = list(self._clients.values())
|
||||
self._clients.clear()
|
||||
|
||||
@@ -710,9 +710,9 @@ def _find_pses_bundle(ctx: ServerContext) -> Optional[str]:
|
||||
env_path = os.environ.get("PSES_BUNDLE_PATH")
|
||||
if env_path:
|
||||
candidates.append(env_path)
|
||||
home = os.environ.get("HERMES_HOME") or os.path.join(
|
||||
os.path.expanduser("~"), ".hermes"
|
||||
)
|
||||
from hermes_constants import get_hermes_home
|
||||
|
||||
home = str(get_hermes_home())
|
||||
candidates.append(os.path.join(home, "lsp", "PowerShellEditorServices"))
|
||||
|
||||
for cand in candidates:
|
||||
@@ -796,9 +796,9 @@ def _spawn_powershell_es(root: str, ctx: ServerContext) -> Optional[SpawnSpec]:
|
||||
|
||||
def hermes_lsp_session_dir() -> str:
|
||||
"""Return (and create) the dir for PSES session/log scratch files."""
|
||||
home = os.environ.get("HERMES_HOME") or os.path.join(
|
||||
os.path.expanduser("~"), ".hermes"
|
||||
)
|
||||
from hermes_constants import get_hermes_home
|
||||
|
||||
home = str(get_hermes_home())
|
||||
d = os.path.join(home, "lsp", "pses")
|
||||
os.makedirs(d, exist_ok=True)
|
||||
return d
|
||||
|
||||
@@ -7,6 +7,36 @@ from typing import Any, Sequence
|
||||
from agent.redact import redact_sensitive_text
|
||||
|
||||
|
||||
def describe_compression_lock_skip(lock_signal: Any) -> str:
|
||||
"""User-facing text for a manual /compress skipped by the compression lock.
|
||||
|
||||
``lock_signal`` is ``agent._compression_skipped_due_to_lock`` (or the
|
||||
``holder`` carried by the TUI's ``CompressionLockHeld``): a descriptive
|
||||
holder string when another compressor CONFIRMED holds the lock, or
|
||||
``True``/``None`` when acquisition failed without a confirmed holder
|
||||
(``hermes_state.try_acquire_compression_lock`` catches ``sqlite3.Error``
|
||||
internally and returns ``False``, so a failed acquire is NOT proof that
|
||||
another compression is running). The two cases must be worded
|
||||
differently: claiming "already in progress" on an unconfirmed failure
|
||||
misdirects the user when the real problem is a broken lock subsystem.
|
||||
"""
|
||||
holder = (
|
||||
lock_signal
|
||||
if isinstance(lock_signal, str) and lock_signal.strip()
|
||||
else None
|
||||
)
|
||||
if holder:
|
||||
return (
|
||||
f"⏳ Compression already in progress for this session "
|
||||
f"(holder: {holder}). Please wait for it to finish."
|
||||
)
|
||||
return (
|
||||
"⏳ Compression skipped: could not acquire this session's "
|
||||
"compression lock. Another compression may still be running, or "
|
||||
"the lock check failed — try again shortly."
|
||||
)
|
||||
|
||||
|
||||
def summarize_manual_compression(
|
||||
before_messages: Sequence[dict[str, Any]],
|
||||
after_messages: Sequence[dict[str, Any]],
|
||||
|
||||
+14
-4
@@ -80,8 +80,17 @@ def normalize_tool_schema(schema: Any) -> Optional[Dict[str, Any]]:
|
||||
return schema
|
||||
|
||||
|
||||
def memory_provider_tools_enabled(enabled_toolsets: Optional[List[str]]) -> bool:
|
||||
def memory_provider_tools_enabled(
|
||||
enabled_toolsets: Optional[List[str]],
|
||||
disabled_toolsets: Optional[List[str]] = None,
|
||||
*,
|
||||
memory_tool_present: bool = False,
|
||||
) -> bool:
|
||||
"""Return whether external memory-provider tools should be exposed."""
|
||||
if disabled_toolsets and "memory" in disabled_toolsets:
|
||||
return False
|
||||
if memory_tool_present:
|
||||
return True
|
||||
if enabled_toolsets is None:
|
||||
return True
|
||||
if not enabled_toolsets:
|
||||
@@ -110,9 +119,10 @@ def inject_memory_provider_tools(agent: Any) -> int:
|
||||
for tool in tools
|
||||
if isinstance(tool, dict)
|
||||
}
|
||||
if (
|
||||
"memory" not in existing_tool_names
|
||||
and not memory_provider_tools_enabled(getattr(agent, "enabled_toolsets", None))
|
||||
if not memory_provider_tools_enabled(
|
||||
getattr(agent, "enabled_toolsets", None),
|
||||
getattr(agent, "disabled_toolsets", None),
|
||||
memory_tool_present="memory" in existing_tool_names,
|
||||
):
|
||||
return 0
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@ re-exports from ``run_agent`` remain in place so existing imports
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
@@ -474,4 +475,378 @@ __all__ = [
|
||||
"_sanitize_tools_non_ascii",
|
||||
"_strip_images_from_messages",
|
||||
"_sanitize_structure_non_ascii",
|
||||
# call_id policy owners (F4 consolidation)
|
||||
"deterministic_call_id",
|
||||
"coalesce_tool_call_id",
|
||||
"uniquify_tool_call_ids",
|
||||
# reasoning_content policy owners (F4 consolidation)
|
||||
"reasoning_echo_family",
|
||||
"matches_reasoning_echo_family",
|
||||
"needs_reasoning_echo",
|
||||
"apply_reasoning_content_policy",
|
||||
"reapply_reasoning_echo",
|
||||
]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# call_id policy — single owner (audit F4, incident chain I4)
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# Three forked policy sites converged here:
|
||||
# * agent/codex_responses_adapter.py `_deterministic_call_id` — hash
|
||||
# synthesis when a provider omits call_id (fa3ab2ffd0 → e45f2b39e2).
|
||||
# * run_agent.AIAgent._get_tool_call_id_static — `call_id or id`
|
||||
# coalescing for dicts and SDK objects.
|
||||
# * run_agent.AIAgent._uniquify_tool_call_ids — duplicate-id repair with
|
||||
# deterministic `_d<n>` suffixes (#58327 loss class).
|
||||
#
|
||||
# NOT consolidated (different scheme on purpose):
|
||||
# agent/transports/codex_event_projector._deterministic_call_id maps codex
|
||||
# app-server ITEM ids (`codex_<type>_<item_id>`), not chat tool-call
|
||||
# content; merging the two would change ids and invalidate prompt caches.
|
||||
#
|
||||
# HARD INVARIANT: everything here must stay deterministic (never uuid4) and
|
||||
# byte-identical for existing inputs — these ids feed prompt-cache prefixes.
|
||||
|
||||
|
||||
def deterministic_call_id(fn_name: str, arguments: str, index: int = 0) -> str:
|
||||
"""Generate a deterministic call_id from tool call content.
|
||||
|
||||
Used as a fallback when the API doesn't provide a call_id.
|
||||
Deterministic IDs prevent cache invalidation — random UUIDs would
|
||||
make every API call's prefix unique, breaking OpenAI's prompt cache.
|
||||
"""
|
||||
seed = f"{fn_name}:{arguments}:{index}"
|
||||
digest = hashlib.sha256(seed.encode("utf-8", errors="replace")).hexdigest()[:12]
|
||||
return f"call_{digest}"
|
||||
|
||||
|
||||
def coalesce_tool_call_id(tc: Any) -> str:
|
||||
"""Extract the effective call ID from a tool_call entry (dict or object).
|
||||
|
||||
Single owner for the ``call_id or id`` coalescing rule: Codex Responses
|
||||
tool calls carry ``call_id`` (authoritative pairing key), Chat
|
||||
Completions ones carry ``id`` only. Returns ``""`` when neither is set.
|
||||
"""
|
||||
if isinstance(tc, dict):
|
||||
return (tc.get("call_id", "") or tc.get("id", "") or "").strip()
|
||||
return (getattr(tc, "call_id", "") or getattr(tc, "id", "") or "").strip()
|
||||
|
||||
|
||||
def uniquify_tool_call_ids(tool_calls: list) -> list:
|
||||
"""Ensure every tool call in a single assistant turn has a distinct id.
|
||||
|
||||
Some models/providers reuse one call id across different calls in a
|
||||
single batch (observed with native Kimi Responses replays, Ollama-
|
||||
compatible endpoints, and degraded models at long context; same bug
|
||||
class as openclaw/openclaw#110518 / #110956). Duplicate ids are lossy
|
||||
downstream: the pre-API sanitizer keeps only the first call/result
|
||||
pair per id (#58327), so the later call's result silently vanishes
|
||||
from every replayed payload, and strict providers (Anthropic
|
||||
tool_use, DeepSeek) reject duplicate ids outright.
|
||||
|
||||
The first occurrence keeps its id; later collisions get a
|
||||
deterministic ``<id>_d<n>`` suffix — never a random UUID, which would
|
||||
break prompt-cache prefix stability across replays. Mutates the
|
||||
entries in place (SDK models / SimpleNamespace / dicts) and returns
|
||||
the same list. Blank/missing ids are left for the deterministic
|
||||
fallback in ``build_assistant_message``.
|
||||
"""
|
||||
seen: set = set()
|
||||
for tc in tool_calls or []:
|
||||
# Same coalescing rule as ``coalesce_tool_call_id`` but tolerant of
|
||||
# non-string ids (degraded models can emit ints/None here).
|
||||
if isinstance(tc, dict):
|
||||
raw = tc.get("call_id") or tc.get("id") or ""
|
||||
else:
|
||||
raw = getattr(tc, "call_id", None) or getattr(tc, "id", None) or ""
|
||||
raw = raw.strip() if isinstance(raw, str) else ""
|
||||
if not raw:
|
||||
continue
|
||||
# Composite Responses ids ("call_x|fc_y") collide on the call
|
||||
# half — that's the pairing key providers enforce per turn.
|
||||
cid = raw.split("|", 1)[0]
|
||||
if not cid:
|
||||
continue
|
||||
if cid not in seen:
|
||||
seen.add(cid)
|
||||
continue
|
||||
n = 2
|
||||
new_id = f"{cid}_d{n}"
|
||||
while new_id in seen:
|
||||
n += 1
|
||||
new_id = f"{cid}_d{n}"
|
||||
seen.add(new_id)
|
||||
|
||||
def _renamed(value):
|
||||
# Preserve a composite id's response-item half so the
|
||||
# provider's real fc_/item id survives the rename.
|
||||
if isinstance(value, str) and "|" in value:
|
||||
return f"{new_id}|{value.split('|', 1)[1]}"
|
||||
return new_id
|
||||
|
||||
try:
|
||||
if isinstance(tc, dict):
|
||||
if tc.get("id"):
|
||||
tc["id"] = _renamed(tc["id"])
|
||||
else:
|
||||
tc["id"] = new_id
|
||||
if tc.get("call_id"):
|
||||
tc["call_id"] = new_id
|
||||
else:
|
||||
tc.id = _renamed(getattr(tc, "id", None))
|
||||
if getattr(tc, "call_id", None):
|
||||
tc.call_id = new_id
|
||||
except Exception:
|
||||
logger.warning(
|
||||
"Could not uniquify duplicate tool call id %s", cid
|
||||
)
|
||||
continue
|
||||
_fn = tc.get("function") if isinstance(tc, dict) else getattr(tc, "function", None)
|
||||
_fn_name = (_fn.get("name") if isinstance(_fn, dict) else getattr(_fn, "name", None)) or "?"
|
||||
logger.warning(
|
||||
"Model reused tool call id %s within one turn; renamed the "
|
||||
"duplicate to %s (tool=%s) to keep call/result pairing "
|
||||
"lossless.", cid, new_id, _fn_name,
|
||||
)
|
||||
return tool_calls
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# reasoning_content policy — single owner (audit F4)
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# The strip-vs-repad decision was previously forked across the wire files in
|
||||
# separate incident commits (2b3a4f0af8 strip for strict providers,
|
||||
# b5495db701 re-pad for require-side, 94b3131be7/9a9f8a6d99 kimi pad). The
|
||||
# POLICY — which provider direction gets which treatment — lives here as one
|
||||
# rule table + apply functions; adapters keep only SYNTAX mapping (e.g.
|
||||
# anthropic_adapter turning reasoning_content into a thinking block).
|
||||
#
|
||||
# Direction table:
|
||||
# require-side (echo-back enforced; replays 400 without the field):
|
||||
# kimi — provider kimi-coding/kimi-coding-cn, or host api.kimi.com /
|
||||
# moonshot.ai / moonshot.cn. Host-driven on purpose:
|
||||
# aggregators re-exporting kimi models reject the echo.
|
||||
# deepseek — provider "deepseek", model contains "deepseek", or host
|
||||
# api.deepseek.com (#15250; V4 rejects empty-string pads,
|
||||
# hence the " " single-space pad, #17341).
|
||||
# mimo — provider "xiaomi", model contains "mimo", or host
|
||||
# *.xiaomimimo.com.
|
||||
# strict side (field rejected with 400/422 "Extra inputs are not
|
||||
# permitted"): everyone else — Mistral, Cerebras, Groq, SambaNova, …
|
||||
# (#45655). Strip the key entirely, even a single-space pad.
|
||||
|
||||
_REASONING_ECHO_RULES: tuple = (
|
||||
# (family, exact providers (raw), exact providers (lowered),
|
||||
# model substrings (lowered), base_url hosts)
|
||||
("kimi", frozenset({"kimi-coding", "kimi-coding-cn"}), frozenset(), (),
|
||||
("api.kimi.com", "moonshot.ai", "moonshot.cn")),
|
||||
("deepseek", frozenset(), frozenset({"deepseek"}), ("deepseek",),
|
||||
("api.deepseek.com",)),
|
||||
("mimo", frozenset(), frozenset({"xiaomi"}), ("mimo",),
|
||||
("api.xiaomimimo.com", "xiaomimimo.com")),
|
||||
)
|
||||
|
||||
|
||||
def _family_rule(family: str) -> tuple:
|
||||
for rule in _REASONING_ECHO_RULES:
|
||||
if rule[0] == family:
|
||||
return rule
|
||||
raise KeyError(family)
|
||||
|
||||
|
||||
def matches_reasoning_echo_family(
|
||||
family: str, provider: Any, model: Any, base_url: Any
|
||||
) -> bool:
|
||||
"""True when (provider, model, base_url) matches one echo-back family.
|
||||
|
||||
Families can overlap (e.g. a deepseek-named model pointed at a kimi
|
||||
host); this membership test is independent per family so per-family
|
||||
predicates keep their original semantics.
|
||||
"""
|
||||
from utils import base_url_host_matches
|
||||
|
||||
_, raw_providers, lowered_providers, model_subs, hosts = _family_rule(family)
|
||||
provider_lower = (provider or "").lower()
|
||||
model_lower = (model or "").lower()
|
||||
if provider in raw_providers or provider_lower in lowered_providers:
|
||||
return True
|
||||
if any(sub in model_lower for sub in model_subs):
|
||||
return True
|
||||
return any(base_url_host_matches(base_url, host) for host in hosts)
|
||||
|
||||
|
||||
def reasoning_echo_family(provider: Any, model: Any, base_url: Any) -> "str | None":
|
||||
"""Classify the provider direction for the reasoning_content echo policy.
|
||||
|
||||
Returns ``"kimi"``, ``"deepseek"``, or ``"mimo"`` (first match in table
|
||||
order) when the target endpoint enforces reasoning_content echo-back on
|
||||
assistant turns, else ``None`` (strict/indifferent side — the field must
|
||||
be stripped).
|
||||
"""
|
||||
for rule in _REASONING_ECHO_RULES:
|
||||
if matches_reasoning_echo_family(rule[0], provider, model, base_url):
|
||||
return rule[0]
|
||||
return None
|
||||
|
||||
|
||||
def needs_reasoning_echo(provider: Any, model: Any, base_url: Any) -> bool:
|
||||
"""True when the endpoint requires reasoning_content echo-back."""
|
||||
return reasoning_echo_family(provider, model, base_url) is not None
|
||||
|
||||
|
||||
def apply_reasoning_content_policy(
|
||||
source_msg: dict, api_msg: dict, needs_thinking_pad: bool
|
||||
) -> None:
|
||||
"""Copy provider-facing reasoning fields onto an API replay message.
|
||||
|
||||
``needs_thinking_pad`` is the require-side flag (see
|
||||
``needs_reasoning_echo`` / the agent's cached
|
||||
``_needs_thinking_reasoning_pad``). Mutates ``api_msg`` in place.
|
||||
"""
|
||||
if source_msg.get("role") != "assistant":
|
||||
return
|
||||
|
||||
# 1. Explicit reasoning_content already set.
|
||||
#
|
||||
# When the active provider enforces the thinking-mode echo-back
|
||||
# (DeepSeek / Kimi / MiMo), preserve it verbatim — that includes their
|
||||
# own space-placeholder written at creation time and any valid reasoning
|
||||
# from the same provider. Sessions persisted BEFORE #17341 have
|
||||
# empty-string placeholders pinned at creation time; DeepSeek V4 Pro
|
||||
# rejects those with HTTP 400, so upgrade "" → " " on replay.
|
||||
#
|
||||
# When the active provider does NOT enforce echo-back, strip the field
|
||||
# entirely. Strict OpenAI-compatible providers (Mistral, Cerebras, Groq,
|
||||
# SambaNova, …) reject ANY reasoning_content key in input messages with
|
||||
# HTTP 400/422 ("Extra inputs are not permitted"), even an empty string
|
||||
# or a single-space pad. This is the cross-provider fallback case: a
|
||||
# reasoning primary (DeepSeek/Kimi/MiMo) pads history with " ", then a
|
||||
# fallback to a strict provider replays that pad and 422s. Stripping
|
||||
# here covers the rebuild path; ``reapply_reasoning_echo`` covers the
|
||||
# already-built api_messages path. Refs #45655.
|
||||
existing = source_msg.get("reasoning_content")
|
||||
if isinstance(existing, str):
|
||||
if not needs_thinking_pad:
|
||||
api_msg.pop("reasoning_content", None)
|
||||
elif existing == "":
|
||||
api_msg["reasoning_content"] = " "
|
||||
else:
|
||||
api_msg["reasoning_content"] = existing
|
||||
return
|
||||
|
||||
# 2. Cross-provider poisoned history (#15748): on DeepSeek/Kimi,
|
||||
# if the source turn has tool_calls AND a 'reasoning' field but no
|
||||
# 'reasoning_content' key, the 'reasoning' text was written by a
|
||||
# prior provider (e.g. MiniMax) — DeepSeek's own _build_assistant_message
|
||||
# pins reasoning_content at creation time for tool-call turns, so the
|
||||
# shape (reasoning set, reasoning_content absent, tool_calls present)
|
||||
# is unreachable from same-provider DeepSeek history after this fix.
|
||||
# Inject a single space to satisfy the API without leaking another
|
||||
# provider's chain of thought to DeepSeek/Kimi. Space (not "")
|
||||
# because DeepSeek V4 Pro rejects empty-string reasoning_content
|
||||
# in thinking mode (refs #17341).
|
||||
normalized_reasoning = source_msg.get("reasoning")
|
||||
if (
|
||||
needs_thinking_pad
|
||||
and source_msg.get("tool_calls")
|
||||
and isinstance(normalized_reasoning, str)
|
||||
and normalized_reasoning
|
||||
):
|
||||
api_msg["reasoning_content"] = " "
|
||||
return
|
||||
|
||||
# 3. Healthy session: promote 'reasoning' field to 'reasoning_content'
|
||||
# for providers that use the internal 'reasoning' key.
|
||||
# This must happen before the unconditional empty-string fallback so
|
||||
# genuine reasoning content is not overwritten (#15812 regression in
|
||||
# PR #15478). Only promote for providers that enforce echo-back —
|
||||
# strict providers reject the field (refs #45655).
|
||||
if isinstance(normalized_reasoning, str) and normalized_reasoning:
|
||||
if needs_thinking_pad:
|
||||
api_msg["reasoning_content"] = normalized_reasoning
|
||||
else:
|
||||
api_msg.pop("reasoning_content", None)
|
||||
return
|
||||
|
||||
# 4. DeepSeek / Kimi thinking mode: all assistant messages need
|
||||
# reasoning_content. Inject a single space to satisfy the provider's
|
||||
# requirement when no explicit reasoning content is present. Covers
|
||||
# both tool-call turns (already-poisoned history with no reasoning
|
||||
# at all) and plain text turns. Space (not "") because DeepSeek V4
|
||||
# Pro tightened validation and rejects empty string with HTTP 400
|
||||
# ("The reasoning content in the thinking mode must be passed back
|
||||
# to the API"). Refs #17341.
|
||||
if needs_thinking_pad:
|
||||
api_msg["reasoning_content"] = " "
|
||||
return
|
||||
|
||||
# 5. reasoning_content was present but not a string (e.g. None after
|
||||
# context compaction). Don't pass null to the API.
|
||||
api_msg.pop("reasoning_content", None)
|
||||
|
||||
|
||||
def reapply_reasoning_echo(api_messages: list, needs_thinking_pad: bool) -> int:
|
||||
"""Re-pad (or strip) assistant turns' reasoning_content for the active provider.
|
||||
|
||||
``api_messages`` is built once, before the retry loop, while the *primary*
|
||||
provider is active. A mid-conversation fallback can then switch providers,
|
||||
so the reasoning fields baked into ``api_messages`` are shaped for the
|
||||
*prior* provider and must be reconciled against the *current* one:
|
||||
|
||||
* Switching TO a require-side provider (DeepSeek / Kimi / MiMo thinking
|
||||
mode): assistant turns built when the prior provider did NOT need the
|
||||
echo-back go out without ``reasoning_content`` and the new provider
|
||||
rejects them with HTTP 400 ("The reasoning_content in the thinking mode
|
||||
must be passed back"). Re-apply the pad.
|
||||
|
||||
* Switching TO a strict provider that rejects the field (Mistral,
|
||||
Cerebras, Groq, SambaNova, …): assistant turns built under a reasoning
|
||||
primary carry a ``reasoning_content`` pad (often a single space ``" "``),
|
||||
and the strict provider rejects it with HTTP 400/422 ("Extra inputs are
|
||||
not permitted"). Strip the field. This is the exact cross-provider
|
||||
fallback bug from #45655 — a DeepSeek primary pads history with ``" "``,
|
||||
the request falls back to Mistral, and Mistral 422s on the stale pad.
|
||||
|
||||
Calling this immediately before building the request kwargs reconciles the
|
||||
fields against the *current* provider. It is idempotent and safe to call
|
||||
every iteration; it covers every fallback path.
|
||||
|
||||
Returns the number of assistant turns whose reasoning_content was added or
|
||||
removed.
|
||||
"""
|
||||
changed = 0
|
||||
for api_msg in api_messages:
|
||||
if api_msg.get("role") != "assistant":
|
||||
continue
|
||||
if needs_thinking_pad:
|
||||
if api_msg.get("reasoning_content"):
|
||||
continue
|
||||
apply_reasoning_content_policy(api_msg, api_msg, needs_thinking_pad)
|
||||
if api_msg.get("reasoning_content"):
|
||||
changed += 1
|
||||
else:
|
||||
# Strict provider — strip any stale reasoning_content pad left
|
||||
# over from a reasoning primary so the fallback request doesn't
|
||||
# 400/422 on it.
|
||||
if "reasoning_content" in api_msg:
|
||||
api_msg.pop("reasoning_content", None)
|
||||
changed += 1
|
||||
return changed
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Image / multimodal parts — evaluated, NOT consolidated (verdict: syntax)
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# The per-adapter image handling is format-specific SYNTAX, not shared policy:
|
||||
# * anthropic_adapter (~1817): data-URL → Anthropic `source: {type: base64}`
|
||||
# block mapping — Anthropic wire shape only.
|
||||
# * codex_responses_adapter (~113/165/812): chat `image_url` parts →
|
||||
# Responses `input_image` items and image counting for log summaries —
|
||||
# Responses wire shape only.
|
||||
# * transports/chat_completions: pass-through (native format).
|
||||
# The one genuinely shared image POLICY — removing images when a server
|
||||
# rejects them while preserving tool_call_id pairing — already has a single
|
||||
# owner here: ``_strip_images_from_messages`` above.
|
||||
|
||||
+1231
-230
File diff suppressed because it is too large
Load Diff
+500
-86
@@ -4,6 +4,8 @@ Pure utility functions with no AIAgent dependency. Used by ContextCompressor
|
||||
and run_agent.py for pre-flight context checks.
|
||||
"""
|
||||
|
||||
import base64
|
||||
import hashlib
|
||||
import ipaddress
|
||||
import json
|
||||
import logging
|
||||
@@ -11,18 +13,40 @@ import os
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
from typing import Any, Dict, List, Optional, Tuple, TYPE_CHECKING
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
import yaml
|
||||
|
||||
if TYPE_CHECKING: # pragma: no cover — runtime import is lazy (see below)
|
||||
import requests
|
||||
|
||||
from utils import atomic_json_write, base_url_host_matches, base_url_hostname
|
||||
|
||||
from hermes_constants import OPENROUTER_MODELS_URL
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# ``requests`` (with urllib3) costs ~27 ms of the `import cli` waterfall and
|
||||
# is only used inside the fetch functions below. It's resolved lazily:
|
||||
# ``_ensure_requests()`` populates the module global on the runtime path, and
|
||||
# the PEP 562 ``__getattr__`` covers external attribute access — notably
|
||||
# ``patch("agent.model_metadata.requests.get")`` in tests, which resolves the
|
||||
# attribute at patch time.
|
||||
|
||||
|
||||
def _ensure_requests():
|
||||
if "requests" not in globals():
|
||||
import requests as _requests
|
||||
globals()["requests"] = _requests
|
||||
return globals()["requests"]
|
||||
|
||||
|
||||
def __getattr__(name: str):
|
||||
if name == "requests":
|
||||
return _ensure_requests()
|
||||
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
||||
|
||||
|
||||
def _resolve_requests_verify() -> bool | str:
|
||||
"""Resolve SSL verify setting for `requests` calls from env vars.
|
||||
@@ -48,7 +72,7 @@ def _resolve_requests_verify() -> bool | str:
|
||||
_PROVIDER_PREFIXES: frozenset[str] = frozenset({
|
||||
"openrouter", "nous", "openai-codex", "copilot", "copilot-acp",
|
||||
"gemini", "ollama-cloud", "zai", "kimi-coding", "kimi-coding-cn", "stepfun", "minimax", "minimax-oauth", "minimax-cn", "anthropic", "deepseek", "deepinfra",
|
||||
"opencode-zen", "opencode-go", "kilocode", "alibaba", "novita",
|
||||
"opencode-zen", "opencode-go", "ai-gateway", "kilocode", "alibaba", "novita",
|
||||
"qwen-oauth",
|
||||
"xiaomi",
|
||||
"arcee",
|
||||
@@ -60,7 +84,7 @@ _PROVIDER_PREFIXES: frozenset[str] = frozenset({
|
||||
"glm", "z-ai", "z.ai", "zhipu", "github", "github-copilot",
|
||||
"github-models", "kimi", "moonshot", "kimi-cn", "moonshot-cn", "claude", "deep-seek", "deep-infra",
|
||||
"ollama",
|
||||
"stepfun", "opencode", "zen", "go", "kilo", "dashscope", "aliyun", "qwen",
|
||||
"stepfun", "opencode", "zen", "go", "vercel", "kilo", "dashscope", "aliyun", "qwen",
|
||||
"mimo", "xiaomi-mimo",
|
||||
"tencent", "tokenhub", "tencent-cloud", "tencentmaas",
|
||||
"arcee-ai", "arceeai",
|
||||
@@ -121,6 +145,67 @@ _ENDPOINT_MODEL_CACHE_TTL = 300
|
||||
_ENDPOINT_PROBE_TTL_SECONDS = 3600.0
|
||||
_endpoint_probe_path_cache: Dict[str, tuple] = {}
|
||||
|
||||
# ── Disk L2 for local-endpoint probe results ────────────────────────────────
|
||||
# The in-process caches above die with the process, so every CLI cold start
|
||||
# with a local model re-paid the probe waterfall in AIAgent.__init__:
|
||||
# detect_local_server_type (up to 4 HTTP GETs, ≤2 s each on a hung server)
|
||||
# + /api/show (≤3 s). A short-TTL disk cache makes back-to-back CLI
|
||||
# invocations hit disk instead of the network. Only SUCCESSFUL probes are
|
||||
# persisted (a down server must not pin a negative verdict), and the TTL is
|
||||
# short enough that swapping the server on a port (stop Ollama, start
|
||||
# LM Studio) is picked up within minutes — strictly fresher than the 1 h
|
||||
# in-process TTL that already accepts that staleness.
|
||||
_LOCAL_PROBE_DISK_TTL_SECONDS = 300.0
|
||||
|
||||
|
||||
def _local_probe_disk_cache_path() -> Path:
|
||||
from hermes_constants import get_hermes_home
|
||||
return get_hermes_home() / "cache" / "local_endpoint_probes.json"
|
||||
|
||||
|
||||
def _load_local_probe_disk_cache() -> Dict[str, Any]:
|
||||
try:
|
||||
with _local_probe_disk_cache_path().open("r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
return data if isinstance(data, dict) else {}
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
|
||||
def _local_probe_disk_get(kind: str, key: str) -> Optional[Any]:
|
||||
"""Return a fresh cached value for ``kind:key``, else None."""
|
||||
entry = _load_local_probe_disk_cache().get(f"{kind}:{key}")
|
||||
if not isinstance(entry, dict):
|
||||
return None
|
||||
try:
|
||||
if (time.time() - float(entry["ts"])) >= _LOCAL_PROBE_DISK_TTL_SECONDS:
|
||||
return None
|
||||
return entry["value"]
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _local_probe_disk_put(kind: str, key: str, value: Any) -> None:
|
||||
"""Persist a successful probe result. Best-effort; prunes stale entries."""
|
||||
try:
|
||||
now = time.time()
|
||||
data = _load_local_probe_disk_cache()
|
||||
data = {
|
||||
k: v
|
||||
for k, v in data.items()
|
||||
if isinstance(v, dict)
|
||||
and (now - float(v.get("ts", 0))) < _LOCAL_PROBE_DISK_TTL_SECONDS
|
||||
}
|
||||
data[f"{kind}:{key}"] = {"value": value, "ts": now}
|
||||
atomic_json_write(
|
||||
_local_probe_disk_cache_path(),
|
||||
data,
|
||||
indent=0,
|
||||
separators=(",", ":"),
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug("Failed to save local probe disk cache: %s", e)
|
||||
|
||||
|
||||
def _get_model_metadata_cache_path() -> Path:
|
||||
"""Return path to the OpenRouter model metadata disk cache."""
|
||||
@@ -213,6 +298,8 @@ DEFAULT_CONTEXT_LENGTHS = {
|
||||
# OpenRouter-prefixed models resolve via OpenRouter live API or models.dev.
|
||||
"claude-fable-5": 1000000,
|
||||
"claude-fable": 1000000,
|
||||
"claude-opus-5": 1000000,
|
||||
"claude-sonnet-5": 1000000,
|
||||
"claude-opus-4-8": 1000000,
|
||||
"claude-opus-4.8": 1000000,
|
||||
"claude-opus-4-7": 1000000,
|
||||
@@ -275,8 +362,10 @@ DEFAULT_CONTEXT_LENGTHS = {
|
||||
# Qwen — specific model families before the catch-all.
|
||||
# Official docs: https://help.aliyun.com/zh/model-studio/developer-reference/
|
||||
"qwen3.6-plus": 1048576, # 1M context (DashScope/Alibaba & OpenRouter)
|
||||
"qwen3.7-plus": 1048576, # 1M context (DashScope/Alibaba)
|
||||
"qwen3-coder-plus": 1000000, # 1M context
|
||||
"qwen3-coder": 262144, # 256K context
|
||||
"qwen3-max": 262144, # 256K context (qwen3-max-2026-01-23 snapshot, Coding Plan)
|
||||
"qwen": 131072,
|
||||
# MiniMax — M3 is 1M context (max output 512K); M2.x series is 204,800.
|
||||
# Keys use substring matching (longest-first), so "minimax-m3" wins over
|
||||
@@ -316,7 +405,12 @@ DEFAULT_CONTEXT_LENGTHS = {
|
||||
"grok-3": 131072, # grok-3, grok-3-mini, grok-3-fast, grok-3-mini-fast
|
||||
"grok-2": 131072, # grok-2, grok-2-1212, grok-2-latest
|
||||
"grok": 131072, # catch-all (grok-beta, unknown grok-*)
|
||||
# Kimi
|
||||
# Kimi — K3 ships with a 1 Mi context window (1,048,576; verified against
|
||||
# models.dev and OpenRouter live metadata, matching the endpoint-scoped
|
||||
# override in _endpoint_scoped_context_length). Longest-key-first substring
|
||||
# matching ensures "kimi-k3" resolves to 1M while older/unknown Kimi models
|
||||
# still hit the generic 256K fallback.
|
||||
"kimi-k3": 1_048_576,
|
||||
"kimi": 262144,
|
||||
# Upstage Solar — api.upstage.ai/v1/models does not return context_length,
|
||||
# so these fallbacks keep token budgeting / compression from probing down
|
||||
@@ -540,7 +634,13 @@ def _is_known_provider_base_url(base_url: str) -> bool:
|
||||
|
||||
|
||||
def _endpoint_scoped_context_length(model: str, base_url: str) -> Optional[int]:
|
||||
"""Return metadata confirmed only for one provider endpoint."""
|
||||
"""Return metadata confirmed only for the Kimi Coding endpoint.
|
||||
|
||||
Kimi Coding serves K3 under the bare slug ``k3``, but users may also
|
||||
configure or select the public-facing aliases ``kimi-k3`` and
|
||||
``kimi-k3-cot``. Only canonical ``https://api.kimi.com/coding`` endpoints
|
||||
(legacy Moonshot keys do not serve K3) get the 1 Mi context window.
|
||||
"""
|
||||
normalized = _normalize_base_url(base_url)
|
||||
try:
|
||||
parsed = urlparse(normalized)
|
||||
@@ -556,7 +656,7 @@ def _endpoint_scoped_context_length(model: str, base_url: str) -> Optional[int]:
|
||||
and parsed.path.rstrip("/") in {"/coding", "/coding/v1"}
|
||||
and not parsed.query
|
||||
and not parsed.fragment
|
||||
and model.strip().lower() == "k3"
|
||||
and model.strip().lower() in {"k3", "kimi-k3", "kimi-k3-cot"}
|
||||
):
|
||||
return 1_048_576
|
||||
return None
|
||||
@@ -567,8 +667,13 @@ def _skip_persistent_context_cache(base_url: str, provider: str) -> bool:
|
||||
|
||||
LM Studio excludes caching because loaded context is transient — the user
|
||||
can reload the model with a different context_length at any time.
|
||||
"""
|
||||
return provider == "lmstudio"
|
||||
|
||||
Codex OAuth excludes caching because its context window is account- and
|
||||
entitlement-specific metadata supplied by the authenticated /models
|
||||
endpoint. A fallback value written after a transient probe failure must
|
||||
not prevent a later live probe from observing an updated allocation.
|
||||
"""
|
||||
return (provider or "").strip().lower() in {"lmstudio", "openai-codex"}
|
||||
|
||||
|
||||
def _maybe_cache_local_context_length(
|
||||
@@ -730,6 +835,13 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
|
||||
if cached is not None and (time.monotonic() - cached[1]) < _ENDPOINT_PROBE_TTL_SECONDS:
|
||||
return cached[0]
|
||||
|
||||
# Disk L2: a fresh cross-process verdict skips the HTTP waterfall
|
||||
# entirely (back-to-back CLI invocations, cron ticks).
|
||||
disk_hit = _local_probe_disk_get("server_type", server_url)
|
||||
if isinstance(disk_hit, str):
|
||||
_endpoint_probe_path_cache[server_url] = (disk_hit, time.monotonic())
|
||||
return disk_hit
|
||||
|
||||
headers = _auth_headers(api_key)
|
||||
|
||||
result: Optional[str] = None
|
||||
@@ -782,6 +894,7 @@ def detect_local_server_type(base_url: str, api_key: str = "") -> Optional[str]:
|
||||
|
||||
if result is not None:
|
||||
_endpoint_probe_path_cache[server_url] = (result, time.monotonic())
|
||||
_local_probe_disk_put("server_type", server_url, result)
|
||||
return result
|
||||
|
||||
|
||||
@@ -904,6 +1017,7 @@ def fetch_model_metadata(force_refresh: bool = False) -> Dict[str, Dict[str, Any
|
||||
return _model_metadata_cache
|
||||
|
||||
try:
|
||||
_ensure_requests()
|
||||
# Tuple (connect, read) — flat timeout=10 means urllib3 can block 10s per
|
||||
# retry stage through proxies that 403 CONNECT, ballooning to minutes
|
||||
# (#46620). 5s connect / 10s read fails fast on unreachable hosts.
|
||||
@@ -960,6 +1074,7 @@ def fetch_endpoint_model_metadata(
|
||||
normalized = _normalize_base_url(base_url)
|
||||
if not normalized or _is_openrouter_base_url(normalized):
|
||||
return {}
|
||||
_ensure_requests()
|
||||
|
||||
if not force_refresh:
|
||||
cached = _endpoint_model_metadata_cache.get(normalized)
|
||||
@@ -1501,6 +1616,13 @@ def query_ollama_num_ctx(model: str, base_url: str, api_key: str = "") -> Option
|
||||
if server_type != "ollama":
|
||||
return None
|
||||
|
||||
# Disk L2: /api/show results are stable for a given (model, server) on
|
||||
# human timescales — skip the HTTP roundtrip on fresh cross-process hits.
|
||||
_disk_key = f"{server_url}|{bare_model}"
|
||||
disk_hit = _local_probe_disk_get("ollama_num_ctx", _disk_key)
|
||||
if isinstance(disk_hit, int) and disk_hit > 0:
|
||||
return disk_hit
|
||||
|
||||
headers = _auth_headers(api_key)
|
||||
|
||||
try:
|
||||
@@ -1518,7 +1640,9 @@ def query_ollama_num_ctx(model: str, base_url: str, api_key: str = "") -> Option
|
||||
parts = line.strip().split()
|
||||
if len(parts) >= 2:
|
||||
try:
|
||||
return int(parts[-1])
|
||||
_ctx = int(parts[-1])
|
||||
_local_probe_disk_put("ollama_num_ctx", _disk_key, _ctx)
|
||||
return _ctx
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
@@ -1526,7 +1650,9 @@ def query_ollama_num_ctx(model: str, base_url: str, api_key: str = "") -> Option
|
||||
model_info = data.get("model_info", {})
|
||||
for key, value in model_info.items():
|
||||
if "context_length" in key and isinstance(value, (int, float)):
|
||||
return int(value)
|
||||
_ctx = int(value)
|
||||
_local_probe_disk_put("ollama_num_ctx", _disk_key, _ctx)
|
||||
return _ctx
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
@@ -1860,6 +1986,7 @@ def _query_anthropic_context_length(model: str, base_url: str, api_key: str) ->
|
||||
"x-api-key": api_key,
|
||||
"anthropic-version": "2023-06-01",
|
||||
}
|
||||
_ensure_requests()
|
||||
resp = requests.get(url, headers=headers, timeout=(5, 10), verify=_resolve_requests_verify())
|
||||
if resp.status_code != 200:
|
||||
return None
|
||||
@@ -1904,32 +2031,73 @@ _CODEX_OAUTH_CONTEXT_FALLBACK: Dict[str, int] = {
|
||||
}
|
||||
|
||||
|
||||
_codex_oauth_context_cache: Dict[str, int] = {}
|
||||
_codex_oauth_context_cache_time: float = 0.0
|
||||
_codex_oauth_context_cache: Dict[str, Tuple[Dict[str, int], float]] = {}
|
||||
_CODEX_OAUTH_CONTEXT_CACHE_TTL = 3600 # 1 hour
|
||||
|
||||
|
||||
def _fetch_codex_oauth_context_lengths(access_token: str) -> Dict[str, int]:
|
||||
"""Probe the ChatGPT Codex /models endpoint for per-slug context windows.
|
||||
def _codex_oauth_token_fingerprint(access_token: str) -> str:
|
||||
"""Return a non-secret cache key for a Codex OAuth access token."""
|
||||
return hashlib.sha256(access_token.encode("utf-8")).hexdigest()[:16]
|
||||
|
||||
Codex OAuth imposes its own context limits that differ from the direct
|
||||
OpenAI API (e.g. gpt-5.5 is 1.05M on the API, 272K on Codex). The
|
||||
`context_window` field in each model entry is the authoritative source.
|
||||
|
||||
Returns a ``{slug: context_window}`` dict. Empty on failure.
|
||||
def _extract_chatgpt_account_id(access_token: str) -> Optional[str]:
|
||||
"""Extract ``chatgpt_account_id`` from the Codex OAuth JWT.
|
||||
|
||||
The Codex ``/backend-api/codex/models`` endpoint returns the per-account
|
||||
catalog only when the ``ChatGPT-Account-Id`` header is present; without
|
||||
it, the endpoint returns ``{"models":[]}`` (HTTP 200) and the context
|
||||
probe falls back to the hardcoded defaults — which can be stale or
|
||||
wrong for the active account's plan. Mirrors the same extraction done
|
||||
in ``auxiliary_client.py`` for the request path.
|
||||
|
||||
Returns ``None`` on any parse error rather than raising, so a bad
|
||||
token still surfaces as a normal probe failure instead of crashing
|
||||
the metadata resolver.
|
||||
"""
|
||||
global _codex_oauth_context_cache, _codex_oauth_context_cache_time
|
||||
try:
|
||||
parts = access_token.split(".")
|
||||
if len(parts) < 2:
|
||||
return None
|
||||
payload_b64 = parts[1] + "=" * (-len(parts[1]) % 4)
|
||||
claims = json.loads(base64.urlsafe_b64decode(payload_b64))
|
||||
if not isinstance(claims, dict):
|
||||
return None
|
||||
acct_id = claims.get("https://api.openai.com/auth", {}).get("chatgpt_account_id")
|
||||
return acct_id if isinstance(acct_id, str) and acct_id else None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _fetch_codex_oauth_context_lengths_with_source(
|
||||
access_token: str,
|
||||
) -> Tuple[Dict[str, int], bool]:
|
||||
"""Fetch Codex catalogue data and report whether it came from HTTP.
|
||||
|
||||
The in-process cache is scoped by token fingerprint because Codex model
|
||||
availability and context windows can vary by account entitlement. The raw
|
||||
token is never retained in the cache key. The boolean is false for a
|
||||
same-token in-process hit, which must not be treated as a fresh provider
|
||||
confirmation when deciding whether to update persistent state.
|
||||
"""
|
||||
global _codex_oauth_context_cache
|
||||
now = time.time()
|
||||
if (
|
||||
_codex_oauth_context_cache
|
||||
and now - _codex_oauth_context_cache_time < _CODEX_OAUTH_CONTEXT_CACHE_TTL
|
||||
):
|
||||
return _codex_oauth_context_cache
|
||||
cache_key = _codex_oauth_token_fingerprint(access_token)
|
||||
cached = _codex_oauth_context_cache.get(cache_key)
|
||||
if cached is not None:
|
||||
cached_models, cached_at = cached
|
||||
if now - cached_at < _CODEX_OAUTH_CONTEXT_CACHE_TTL:
|
||||
return cached_models, False
|
||||
|
||||
headers = {"Authorization": f"Bearer {access_token}"}
|
||||
acct_id = _extract_chatgpt_account_id(access_token)
|
||||
if acct_id:
|
||||
headers["ChatGPT-Account-Id"] = acct_id
|
||||
|
||||
try:
|
||||
_ensure_requests()
|
||||
resp = requests.get(
|
||||
"https://chatgpt.com/backend-api/codex/models?client_version=1.0.0",
|
||||
headers={"Authorization": f"Bearer {access_token}"},
|
||||
headers=headers,
|
||||
timeout=(5, 10),
|
||||
verify=_resolve_requests_verify(),
|
||||
)
|
||||
@@ -1938,11 +2106,11 @@ def _fetch_codex_oauth_context_lengths(access_token: str) -> Dict[str, int]:
|
||||
"Codex /models probe returned HTTP %s; falling back to hardcoded defaults",
|
||||
resp.status_code,
|
||||
)
|
||||
return {}
|
||||
return {}, False
|
||||
data = resp.json()
|
||||
except Exception as exc:
|
||||
logger.debug("Codex /models probe failed: %s", exc)
|
||||
return {}
|
||||
return {}, False
|
||||
|
||||
entries = data.get("models", []) if isinstance(data, dict) else []
|
||||
result: Dict[str, int] = {}
|
||||
@@ -1955,32 +2123,50 @@ def _fetch_codex_oauth_context_lengths(access_token: str) -> Dict[str, int]:
|
||||
result[slug.strip()] = ctx
|
||||
|
||||
if result:
|
||||
_codex_oauth_context_cache = result
|
||||
_codex_oauth_context_cache_time = now
|
||||
_codex_oauth_context_cache[cache_key] = (result, now)
|
||||
return result, True
|
||||
|
||||
|
||||
def _fetch_codex_oauth_context_lengths(access_token: str) -> Dict[str, int]:
|
||||
"""Probe the ChatGPT Codex /models endpoint for per-slug context windows.
|
||||
|
||||
Codex OAuth imposes its own context limits that differ from the direct
|
||||
OpenAI API (e.g. gpt-5.5 is 1.05M on the API, 272K on Codex). The
|
||||
`context_window` field in each model entry is the authoritative source.
|
||||
|
||||
Returns a ``{slug: context_window}`` dict. Empty on failure.
|
||||
"""
|
||||
result, _fresh = _fetch_codex_oauth_context_lengths_with_source(access_token)
|
||||
return result
|
||||
|
||||
|
||||
def _resolve_codex_oauth_context_length(
|
||||
def _resolve_codex_oauth_context_length_with_source(
|
||||
model: str, access_token: str = ""
|
||||
) -> Optional[int]:
|
||||
) -> Tuple[Optional[int], str]:
|
||||
"""Resolve a Codex OAuth model's real context window.
|
||||
|
||||
Prefers a live probe of chatgpt.com/backend-api/codex/models (when we
|
||||
have a bearer token), then falls back to ``_CODEX_OAUTH_CONTEXT_FALLBACK``.
|
||||
|
||||
Returns ``(context_length, source)`` where source is ``"live"`` for a
|
||||
value returned by a fresh authenticated endpoint probe, ``"memory"`` for
|
||||
a same-token in-process catalogue hit, or ``"fallback"`` for the static
|
||||
conservative table. Only ``"live"`` is eligible for persistent writes.
|
||||
"""
|
||||
model_bare = _strip_provider_prefix(model).strip()
|
||||
if not model_bare:
|
||||
return None
|
||||
return None, ""
|
||||
|
||||
if access_token:
|
||||
live = _fetch_codex_oauth_context_lengths(access_token)
|
||||
live, fresh_probe = _fetch_codex_oauth_context_lengths_with_source(access_token)
|
||||
live_source = "live" if fresh_probe else "memory"
|
||||
if model_bare in live:
|
||||
return live[model_bare]
|
||||
return live[model_bare], live_source
|
||||
# Case-insensitive match in case casing drifts
|
||||
model_lower = model_bare.lower()
|
||||
for slug, ctx in live.items():
|
||||
if slug.lower() == model_lower:
|
||||
return ctx
|
||||
return ctx, live_source
|
||||
|
||||
# Fallback: longest-key-first substring match over hardcoded defaults.
|
||||
model_lower = model_bare.lower()
|
||||
@@ -1988,9 +2174,19 @@ def _resolve_codex_oauth_context_length(
|
||||
_CODEX_OAUTH_CONTEXT_FALLBACK.items(), key=lambda x: len(x[0]), reverse=True
|
||||
):
|
||||
if slug in model_lower:
|
||||
return ctx
|
||||
return ctx, "fallback"
|
||||
|
||||
return None
|
||||
return None, ""
|
||||
|
||||
|
||||
def _resolve_codex_oauth_context_length(
|
||||
model: str, access_token: str = ""
|
||||
) -> Optional[int]:
|
||||
"""Resolve a Codex OAuth model's context length (compatibility wrapper)."""
|
||||
context_length, _source = _resolve_codex_oauth_context_length_with_source(
|
||||
model, access_token=access_token,
|
||||
)
|
||||
return context_length
|
||||
|
||||
|
||||
def _resolve_nous_context_length(
|
||||
@@ -2080,9 +2276,9 @@ def get_model_context_length(
|
||||
Resolution order:
|
||||
0. Explicit config override (model.context_length or custom_providers per-model)
|
||||
0c. Endpoint-scoped metadata for models validated on one multiplexed endpoint
|
||||
1. Persistent cache (previously discovered via probing). Nous URLs
|
||||
bypass the cache here so step 5b can always reconcile against
|
||||
the authoritative portal /v1/models response.
|
||||
1. Persistent cache (previously discovered via probing). Nous URLs,
|
||||
LM Studio, and Codex OAuth bypass the cache here so their provider
|
||||
metadata can be reconciled against the authoritative live source.
|
||||
1b. AWS Bedrock static table (must precede custom-endpoint probe)
|
||||
2. Active endpoint metadata (/models for explicit custom endpoints)
|
||||
3. Local server query (for local endpoints)
|
||||
@@ -2112,11 +2308,18 @@ def get_model_context_length(
|
||||
# acting context, so they're ignored here.
|
||||
if (provider or "").strip().lower() == "moa":
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
from hermes_cli.config import (
|
||||
get_compatible_custom_providers,
|
||||
load_config,
|
||||
)
|
||||
from hermes_cli.moa_config import resolve_moa_preset
|
||||
from hermes_cli.runtime_provider import resolve_runtime_provider
|
||||
|
||||
preset = resolve_moa_preset(load_config().get("moa") or {}, model)
|
||||
config = load_config()
|
||||
effective_custom_providers = custom_providers
|
||||
if effective_custom_providers is None:
|
||||
effective_custom_providers = get_compatible_custom_providers(config)
|
||||
preset = resolve_moa_preset(config.get("moa") or {}, model)
|
||||
agg = preset.get("aggregator") or {}
|
||||
agg_provider = str(agg.get("provider") or "").strip()
|
||||
agg_model = str(agg.get("model") or "").strip()
|
||||
@@ -2126,7 +2329,8 @@ def get_model_context_length(
|
||||
agg_model,
|
||||
base_url=rt.get("base_url", "") or "",
|
||||
api_key=rt.get("api_key", "") or "",
|
||||
provider=agg_provider,
|
||||
provider=rt.get("provider") or agg_provider,
|
||||
custom_providers=effective_custom_providers,
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("MoA aggregator context-length resolution failed", exc_info=True)
|
||||
@@ -2172,28 +2376,23 @@ def get_model_context_length(
|
||||
if endpoint_context is not None:
|
||||
return endpoint_context
|
||||
|
||||
is_bedrock_context = provider == "bedrock" or (
|
||||
base_url
|
||||
and base_url_hostname(base_url).startswith("bedrock-runtime.")
|
||||
and base_url_host_matches(base_url, "amazonaws.com")
|
||||
)
|
||||
|
||||
# 1. Check persistent cache (model+provider)
|
||||
# LM Studio is excluded — its loaded context length is transient (the
|
||||
# user can reload the model with a different context_length at any time
|
||||
# via /api/v1/models/load), so a stale cached value would mask reloads.
|
||||
# Codex OAuth is excluded because the authenticated /models catalogue is
|
||||
# account-specific and a fallback must never suppress later revalidation.
|
||||
if base_url and not _skip_persistent_context_cache(base_url, provider):
|
||||
cached = get_cached_context_length(model, base_url)
|
||||
if cached is not None:
|
||||
# Invalidate stale Codex OAuth cache entries: pre-PR #14935 builds
|
||||
# resolved gpt-5.x to the direct-API value (e.g. 1.05M) via
|
||||
# models.dev and persisted it. Codex OAuth caps at 272K for every
|
||||
# slug, so any cached Codex entry at or above 400K is a leftover
|
||||
# from the old resolution path. Drop it and fall through to the
|
||||
# live /models probe in step 5 below.
|
||||
if provider == "openai-codex" and cached >= 400_000:
|
||||
logger.info(
|
||||
"Dropping stale Codex cache entry %s@%s -> %s (pre-fix value); "
|
||||
"re-resolving via live /models probe",
|
||||
model, base_url, f"{cached:,}",
|
||||
)
|
||||
_invalidate_cached_context_length(model, base_url)
|
||||
# Invalidate stale 32k cache entries for Kimi-family models.
|
||||
elif cached <= 32768 and _model_name_suggests_kimi(model):
|
||||
if cached <= 32768 and _model_name_suggests_kimi(model):
|
||||
logger.info(
|
||||
"Dropping stale Kimi cache entry %s@%s -> %s (OpenRouter underreport); "
|
||||
"re-resolving via hardcoded defaults",
|
||||
@@ -2240,6 +2439,30 @@ def get_model_context_length(
|
||||
model, base_url,
|
||||
)
|
||||
# Fall through; step 5b reconciles and overwrites if portal responds.
|
||||
# Invalidate stale Bedrock entries seeded before the Claude 4.6+
|
||||
# long-context table was corrected to 1M. The static table is a
|
||||
# FLOOR, not an override: probe-derived cache entries (step 1b)
|
||||
# may legitimately exceed the table (real window read from
|
||||
# Bedrock's length-validation error), so only under-reporting
|
||||
# entries are dropped — never a cached value above the table.
|
||||
elif is_bedrock_context:
|
||||
try:
|
||||
from agent.bedrock_adapter import get_bedrock_context_length
|
||||
bedrock_ctx = get_bedrock_context_length(model)
|
||||
if cached < bedrock_ctx:
|
||||
logger.info(
|
||||
"Dropping stale Bedrock cache entry %s@%s -> %s; "
|
||||
"using static Bedrock table value %s",
|
||||
model,
|
||||
base_url,
|
||||
f"{cached:,}",
|
||||
f"{bedrock_ctx:,}",
|
||||
)
|
||||
_invalidate_cached_context_length(model, base_url)
|
||||
return bedrock_ctx
|
||||
except ImportError:
|
||||
pass
|
||||
return cached
|
||||
else:
|
||||
if is_local_endpoint(base_url):
|
||||
return _reconcile_local_cached_context_length(
|
||||
@@ -2250,22 +2473,50 @@ def get_model_context_length(
|
||||
# 1b. AWS Bedrock — use static context length table.
|
||||
# Bedrock's ListFoundationModels API doesn't expose context window sizes,
|
||||
# so we maintain a curated table in bedrock_adapter.py that reflects
|
||||
# AWS-imposed limits (e.g. 200K for Claude models vs 1M on the native
|
||||
# Anthropic API). This must run BEFORE the custom-endpoint probe at
|
||||
# Bedrock-hosted model limits (e.g. older Claude 4 at 200K; Claude
|
||||
# Opus/Sonnet 4.6+ at 1M). This must run BEFORE the custom-endpoint probe at
|
||||
# step 2 — bedrock-runtime.<region>.amazonaws.com is not in
|
||||
# _URL_TO_PROVIDER, so it would otherwise be treated as a custom endpoint,
|
||||
# fail the /models probe (Bedrock doesn't expose that shape), and fall
|
||||
# back to the 128K default before reaching the original step 4b branch.
|
||||
if provider == "bedrock" or (
|
||||
base_url
|
||||
and base_url_hostname(base_url).startswith("bedrock-runtime.")
|
||||
and base_url_host_matches(base_url, "amazonaws.com")
|
||||
):
|
||||
if is_bedrock_context:
|
||||
try:
|
||||
from agent.bedrock_adapter import get_bedrock_context_length
|
||||
return get_bedrock_context_length(model)
|
||||
from agent.bedrock_adapter import (
|
||||
get_bedrock_context_length,
|
||||
resolve_bedrock_region,
|
||||
)
|
||||
except ImportError:
|
||||
pass # boto3 not installed — fall through to generic resolution
|
||||
else:
|
||||
# Bedrock does not expose the context window via any metadata API,
|
||||
# so get_bedrock_context_length() probes the live endpoint (one
|
||||
# fast, pre-inference length rejection) to read the real window.
|
||||
# Cache the probe result per model so we pay that cost once, not
|
||||
# every turn — keyed by base_url when present, else a synthetic
|
||||
# bedrock:// key so display/offline paths share the entry.
|
||||
cache_key_url = base_url or "bedrock://"
|
||||
cached = get_cached_context_length(model, cache_key_url)
|
||||
if cached is not None:
|
||||
return cached
|
||||
# Resolve region from the base_url host first, then the standard
|
||||
# AWS region chain. An empty region disables probing (table only).
|
||||
region = ""
|
||||
if base_url:
|
||||
_m = re.search(r"bedrock-runtime\.([a-z0-9-]+)\.", base_url)
|
||||
if _m:
|
||||
region = _m.group(1)
|
||||
if not region:
|
||||
try:
|
||||
region = resolve_bedrock_region()
|
||||
except Exception:
|
||||
region = ""
|
||||
ctx = get_bedrock_context_length(model, region=region, probe=bool(region))
|
||||
if ctx and region:
|
||||
# Only persist probe-derived values (region present); a pure
|
||||
# table fallback shouldn't poison the cache against a later
|
||||
# successful probe.
|
||||
save_context_length(model, cache_key_url, ctx)
|
||||
return ctx
|
||||
|
||||
if provider == "novita" or (base_url and base_url_host_matches(base_url, "api.novita.ai")):
|
||||
ctx = _resolve_endpoint_context_length(model, base_url or "https://api.novita.ai/openai/v1", api_key=api_key)
|
||||
@@ -2284,21 +2535,27 @@ def get_model_context_length(
|
||||
if context_length is not None:
|
||||
return context_length
|
||||
if not _is_known_provider_base_url(base_url):
|
||||
# 2b. Ollama native /api/show — any URL might be an Ollama server
|
||||
# (local, cloud, or custom hosting). Non-Ollama servers return
|
||||
# 404/405 quickly. Fall through on failure.
|
||||
ctx = _query_ollama_api_show(model, base_url, api_key=api_key)
|
||||
if ctx is not None:
|
||||
if not _skip_persistent_context_cache(base_url, provider):
|
||||
save_context_length(model, base_url, ctx)
|
||||
return ctx
|
||||
# 3. Try querying local server directly
|
||||
# For local endpoints, run the probe that respects configured
|
||||
# Modelfile context values first. _query_local_context_length
|
||||
# prefers num_ctx from Modelfile, while _query_ollama_api_show
|
||||
# returns the GGUF training max first which can be larger and
|
||||
# would create a false-safe window for compression (#63122).
|
||||
# Non-local endpoints preserve the existing GGUF-first behavior.
|
||||
if is_local_endpoint(base_url):
|
||||
local_ctx = _query_local_context_length(model, base_url, api_key=api_key)
|
||||
if local_ctx and local_ctx > 0:
|
||||
if not _skip_persistent_context_cache(base_url, provider):
|
||||
_maybe_cache_local_context_length(model, base_url, local_ctx)
|
||||
return local_ctx
|
||||
# 2b. Ollama native /api/show — non-local endpoints preserve
|
||||
# the existing generic /api/show GGUF-first behavior.
|
||||
# Non-Ollama servers return 404/405 quickly.
|
||||
ctx = _query_ollama_api_show(model, base_url, api_key=api_key)
|
||||
if ctx is not None:
|
||||
if not _skip_persistent_context_cache(base_url, provider):
|
||||
save_context_length(model, base_url, ctx)
|
||||
return ctx
|
||||
# 3. Probe-down fallback after endpoint-specific detection failed
|
||||
logger.info(
|
||||
"Could not detect context length for model %r at %s — "
|
||||
"defaulting to %s tokens (probe-down). Set model.context_length "
|
||||
@@ -2380,9 +2637,14 @@ def get_model_context_length(
|
||||
# Codex OAuth enforces lower context limits than the direct OpenAI
|
||||
# API for the same slug (e.g. gpt-5.5 is 1.05M on the API but 272K
|
||||
# on Codex). Authoritative source is Codex's own /models endpoint.
|
||||
codex_ctx = _resolve_codex_oauth_context_length(model, access_token=api_key or "")
|
||||
codex_ctx, codex_source = _resolve_codex_oauth_context_length_with_source(
|
||||
model, access_token=api_key or "",
|
||||
)
|
||||
if codex_ctx:
|
||||
if base_url:
|
||||
# Only a successful authenticated catalogue response is safe to
|
||||
# persist. The static fallback is deliberately runtime-only so a
|
||||
# transient OAuth/network failure cannot poison future probes.
|
||||
if base_url and codex_source == "live":
|
||||
save_context_length(model, base_url, codex_ctx)
|
||||
return codex_ctx
|
||||
if effective_provider == "gmi" and base_url:
|
||||
@@ -2525,16 +2787,61 @@ async def get_model_context_length_async(
|
||||
)
|
||||
|
||||
|
||||
def _is_cjk_token_dense_char(ch: str) -> bool:
|
||||
code = ord(ch)
|
||||
return (
|
||||
0x1100 <= code <= 0x11FF # Hangul Jamo
|
||||
or 0x2E80 <= code <= 0x9FFF # CJK radicals/ideographs
|
||||
or 0xA960 <= code <= 0xA97F # Hangul Jamo Extended-A
|
||||
or 0xAC00 <= code <= 0xD7AF # Hangul Syllables
|
||||
or 0xF900 <= code <= 0xFAFF # CJK compatibility ideographs
|
||||
or 0xFF00 <= code <= 0xFFEF # Fullwidth forms / halfwidth kana
|
||||
)
|
||||
|
||||
|
||||
# Same codepoint ranges as _is_cjk_token_dense_char, as a compiled character
|
||||
# class so dense-char counting runs in C (``len(text) - len(re.sub(...))``)
|
||||
# instead of a per-char Python loop. MUST stay in sync with
|
||||
# _is_cjk_token_dense_char.
|
||||
_CJK_DENSE_RE = re.compile(
|
||||
"[\u1100-\u11ff" # Hangul Jamo
|
||||
"\u2e80-\u9fff" # CJK radicals/ideographs
|
||||
"\ua960-\ua97f" # Hangul Jamo Extended-A
|
||||
"\uac00-\ud7af" # Hangul Syllables
|
||||
"\uf900-\ufaff" # CJK compatibility ideographs
|
||||
"\uff00-\uffef]" # Fullwidth forms / halfwidth kana
|
||||
)
|
||||
|
||||
|
||||
def estimate_tokens_rough(text: str) -> int:
|
||||
"""Rough token estimate (~4 chars/token) for pre-flight checks.
|
||||
"""Rough token estimate for pre-flight checks.
|
||||
|
||||
Uses ceiling division so short texts (1-3 chars) never estimate as
|
||||
0 tokens, which would cause the compressor and pre-flight checks to
|
||||
systematically undercount when many short tool results are present.
|
||||
CJK/Hangul/Kana text is much denser than English under common LLM
|
||||
tokenizers, so count those codepoints as roughly one token each instead
|
||||
of applying the English-centric ~4 chars/token rule.
|
||||
|
||||
Perf: this runs on every message in every preflight/compaction walk,
|
||||
including MB-scale tool outputs, so the common all-ASCII case must stay
|
||||
O(1). ``str.isascii()`` is a flag check on CPython's compact unicode
|
||||
representation (no scan), and the CJK counting itself is a single
|
||||
C-level ``re.findall`` rather than a per-character Python loop.
|
||||
"""
|
||||
if not text:
|
||||
return 0
|
||||
return (len(text) + 3) // 4
|
||||
text = str(text)
|
||||
if text.isascii():
|
||||
# O(1) fast path — ASCII text cannot contain token-dense CJK chars.
|
||||
return (len(text) + 3) // 4
|
||||
dense = len(text) - len(_CJK_DENSE_RE.sub("", text))
|
||||
if not dense:
|
||||
# Non-ASCII but no CJK (accents, Cyrillic, emoji, ...): keep the
|
||||
# classic ~4 chars/token rule.
|
||||
return (len(text) + 3) // 4
|
||||
sparse = len(text) - dense
|
||||
return dense + ((sparse + 3) // 4)
|
||||
|
||||
|
||||
def estimate_messages_tokens_rough(messages: List[Dict[str, Any]]) -> int:
|
||||
@@ -2544,14 +2851,92 @@ def estimate_messages_tokens_rough(messages: List[Dict[str, Any]]) -> int:
|
||||
image — the Anthropic pricing model — instead of counting raw base64
|
||||
character length. Without this, a single ~1MB screenshot would be
|
||||
estimated at ~250K tokens and trigger premature context compression.
|
||||
|
||||
Per-message results are memoized (see ``_estimate_message_tokens_cached``)
|
||||
keyed on a deep *identity fingerprint* of the message, so re-walking a
|
||||
long history every iteration only pays for messages whose object graph
|
||||
actually changed. The memo is exact: equal fingerprints imply identical
|
||||
leaf objects and structure, hence an identical estimate.
|
||||
"""
|
||||
_IMAGE_TOKEN_COST = 1500
|
||||
total_chars = 0
|
||||
image_tokens = 0
|
||||
total = 0
|
||||
for msg in messages:
|
||||
total_chars += _estimate_message_chars(msg)
|
||||
image_tokens += _count_image_tokens(msg, _IMAGE_TOKEN_COST)
|
||||
return ((total_chars + 3) // 4) + image_tokens
|
||||
total += _estimate_message_tokens_cached(msg, _IMAGE_TOKEN_COST)
|
||||
return total
|
||||
|
||||
|
||||
# --- Per-message token-estimate memo -------------------------------------
|
||||
#
|
||||
# ``estimate_messages_tokens_rough`` is called on the full history every
|
||||
# loop iteration (conversation_loop preflight), repeatedly during compaction
|
||||
# telemetry, and inside an O(n^2) shrink loop in moa_loop. The per-message
|
||||
# helpers are pure functions of the message's value, so a memo keyed on a
|
||||
# fingerprint that uniquely determines the value is exactly equivalent.
|
||||
#
|
||||
# Fingerprint design (soundness argument):
|
||||
# * strings are fingerprinted by ``id()`` AND pinned (a strong reference is
|
||||
# stored in the cache entry). While the entry lives, that id cannot be
|
||||
# reused by another object, so id-equality implies object-equality —
|
||||
# strings are immutable, so value-equality too (no #50372-style aliasing).
|
||||
# * ints/floats/bools/None are fingerprinted by value.
|
||||
# * dicts/lists recurse structurally, preserving key order — ``str(shadow)``
|
||||
# depends on insertion order, so order is part of the key.
|
||||
# * any other type aborts the memo and falls through to a direct compute.
|
||||
# Equal fingerprints therefore imply deep-equal messages built from identical
|
||||
# immutable leaves ⇒ identical ``str(shadow)`` bytes ⇒ identical estimate.
|
||||
#
|
||||
# Because the api_messages build shallow-copies history dicts each iteration,
|
||||
# the copies share the same content strings — so unchanged history messages
|
||||
# hit the memo even though the outer dicts are fresh objects every turn.
|
||||
_MSG_TOKENS_CACHE: Dict[Any, Tuple[list, int]] = {}
|
||||
_MSG_TOKENS_CACHE_MAX = 4096
|
||||
|
||||
|
||||
def _msg_fingerprint(value: Any, pins: list) -> Any:
|
||||
if value is None or value is True or value is False:
|
||||
return value
|
||||
t = type(value)
|
||||
if t is str:
|
||||
pins.append(value)
|
||||
return ("s", id(value))
|
||||
if t is int or t is float:
|
||||
return ("n", t.__name__, value)
|
||||
if t is dict:
|
||||
return ("d", tuple(
|
||||
(_msg_fingerprint(k, pins), _msg_fingerprint(v, pins))
|
||||
for k, v in value.items()
|
||||
))
|
||||
if t is list:
|
||||
return ("l", tuple(_msg_fingerprint(v, pins) for v in value))
|
||||
if t is tuple:
|
||||
return ("t", tuple(_msg_fingerprint(v, pins) for v in value))
|
||||
raise ValueError("unfingerprintable message value")
|
||||
|
||||
|
||||
def _estimate_message_tokens_cached(msg: Any, image_cost: int) -> int:
|
||||
try:
|
||||
pins: list = []
|
||||
key = _msg_fingerprint(msg, pins)
|
||||
hash(key)
|
||||
except Exception:
|
||||
return (
|
||||
_estimate_message_tokens_without_images(msg)
|
||||
+ _count_image_tokens(msg, image_cost)
|
||||
)
|
||||
cached = _MSG_TOKENS_CACHE.get(key)
|
||||
if cached is not None:
|
||||
return cached[1]
|
||||
tokens = (
|
||||
_estimate_message_tokens_without_images(msg)
|
||||
+ _count_image_tokens(msg, image_cost)
|
||||
)
|
||||
_MSG_TOKENS_CACHE[key] = (pins, tokens)
|
||||
while len(_MSG_TOKENS_CACHE) > _MSG_TOKENS_CACHE_MAX:
|
||||
try:
|
||||
_MSG_TOKENS_CACHE.pop(next(iter(_MSG_TOKENS_CACHE)))
|
||||
except (StopIteration, KeyError, RuntimeError):
|
||||
break
|
||||
return tokens
|
||||
|
||||
|
||||
def _count_image_tokens(msg: Dict[str, Any], cost_per_image: int) -> int:
|
||||
@@ -2613,6 +2998,35 @@ def _estimate_message_chars(msg: Dict[str, Any]) -> int:
|
||||
return len(str(shadow))
|
||||
|
||||
|
||||
def _estimate_message_tokens_without_images(msg: Dict[str, Any]) -> int:
|
||||
"""Token estimate for a message shadow with image payloads stripped."""
|
||||
if not isinstance(msg, dict):
|
||||
return estimate_tokens_rough(str(msg))
|
||||
shadow: Dict[str, Any] = {}
|
||||
for k, v in msg.items():
|
||||
if k == "_anthropic_content_blocks":
|
||||
continue
|
||||
if k == "content":
|
||||
if isinstance(v, list):
|
||||
cleaned = []
|
||||
for part in v:
|
||||
if isinstance(part, dict):
|
||||
if part.get("type") in {"image", "image_url", "input_image"}:
|
||||
cleaned.append({"type": part.get("type"), "image": "[stripped]"})
|
||||
else:
|
||||
cleaned.append(part)
|
||||
else:
|
||||
cleaned.append(part)
|
||||
shadow[k] = cleaned
|
||||
elif isinstance(v, dict) and v.get("_multimodal"):
|
||||
shadow[k] = v.get("text_summary", "")
|
||||
else:
|
||||
shadow[k] = v
|
||||
else:
|
||||
shadow[k] = v
|
||||
return estimate_tokens_rough(str(shadow))
|
||||
|
||||
|
||||
def estimate_request_tokens_rough(
|
||||
messages: List[Dict[str, Any]],
|
||||
*,
|
||||
@@ -2629,7 +3043,7 @@ def estimate_request_tokens_rough(
|
||||
"""
|
||||
total = 0
|
||||
if system_prompt:
|
||||
total += (len(system_prompt) + 3) // 4
|
||||
total += estimate_tokens_rough(system_prompt)
|
||||
if messages:
|
||||
total += estimate_messages_tokens_rough(messages)
|
||||
if tools:
|
||||
|
||||
+236
-58
@@ -8,11 +8,15 @@ of 4000+ models across 109+ providers. Provides:
|
||||
(reasoning, tools, vision, PDF, audio), modalities, knowledge cutoff,
|
||||
open-weights flag, family grouping, deprecation status
|
||||
|
||||
Data resolution order (like TypeScript OpenCode):
|
||||
1. Bundled snapshot (ships with the package — offline-first)
|
||||
2. Disk cache (~/.hermes/models_dev_cache.json)
|
||||
3. Network fetch (https://models.dev/api.json)
|
||||
4. Background refresh every 60 minutes
|
||||
Data resolution order:
|
||||
1. In-memory cache (fresh, or stale served immediately while a single
|
||||
background daemon thread refreshes)
|
||||
2. Disk cache (~/.hermes/models_dev_cache.json — any age; stale data is
|
||||
served rather than blocking callers on the network)
|
||||
3. Network fetch (https://models.dev/api.json) — only when no cache
|
||||
exists at all; failed refreshes back off for 5 minutes process-wide
|
||||
Latency-sensitive callers (gateway route-identity checks) pass
|
||||
``allow_network=False`` and never touch the network.
|
||||
|
||||
Other modules should import the dataclasses and query functions from here
|
||||
rather than parsing the raw JSON themselves.
|
||||
@@ -20,6 +24,7 @@ rather than parsing the raw JSON themselves.
|
||||
|
||||
import json
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
@@ -33,10 +38,15 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
MODELS_DEV_URL = "https://models.dev/api.json"
|
||||
_MODELS_DEV_CACHE_TTL = 3600 # 1 hour in-memory
|
||||
_MODELS_DEV_RETRY_DELAY = 300 # 5 minutes after a failed refresh
|
||||
|
||||
# In-memory cache
|
||||
_models_dev_cache: Dict[str, Any] = {}
|
||||
_models_dev_cache_time: float = 0
|
||||
_models_dev_retry_after: float = 0
|
||||
_models_dev_fetch_lock = threading.Lock()
|
||||
_models_dev_refresh_lock = threading.Lock()
|
||||
_models_dev_refresh_in_flight = False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -158,6 +168,7 @@ PROVIDER_TO_MODELS_DEV: Dict[str, str] = {
|
||||
"alibaba": "alibaba",
|
||||
"qwen-oauth": "alibaba",
|
||||
"copilot": "github-copilot",
|
||||
"ai-gateway": "vercel",
|
||||
"opencode-zen": "opencode",
|
||||
"opencode-go": "opencode-go",
|
||||
"kilocode": "kilo",
|
||||
@@ -237,27 +248,157 @@ def _save_disk_cache(data: Dict[str, Any]) -> None:
|
||||
logger.debug("Failed to save models.dev disk cache: %s", e)
|
||||
|
||||
|
||||
def fetch_models_dev(force_refresh: bool = False) -> Dict[str, Any]:
|
||||
def _fetch_models_dev_from_network() -> Dict[str, Any]:
|
||||
"""Fetch the live models.dev registry without touching local caches.
|
||||
|
||||
Raises on network errors and on an empty/invalid registry payload.
|
||||
"""
|
||||
# Tuple (connect, read): a flat timeout=15 let a blackholed connect
|
||||
# stall the first-turn critical path for the full 15 s. 5 s connect
|
||||
# fails fast on unreachable hosts; 10 s read still tolerates a slow
|
||||
# registry response (matches the OpenRouter fetch convention in
|
||||
# agent/model_metadata.py).
|
||||
response = requests.get(MODELS_DEV_URL, timeout=(5, 10))
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
if not isinstance(data, dict) or not data:
|
||||
raise ValueError("models.dev returned an empty or invalid registry")
|
||||
return data
|
||||
|
||||
|
||||
def _mark_stale_cache_grace() -> None:
|
||||
"""Give stale cache data a short in-memory grace before retrying refresh.
|
||||
|
||||
Only ever moves the timestamp forward: if a background refresh completed
|
||||
between the caller's staleness check and this call, the fresh timestamp
|
||||
is preserved instead of being rewound to a 5-minute grace.
|
||||
"""
|
||||
global _models_dev_cache_time
|
||||
grace_time = time.time() - _MODELS_DEV_CACHE_TTL + _MODELS_DEV_RETRY_DELAY
|
||||
if grace_time > _models_dev_cache_time:
|
||||
_models_dev_cache_time = grace_time
|
||||
|
||||
|
||||
def _commit_registry(data: Dict[str, Any], *, where: str) -> None:
|
||||
"""Persist a freshly fetched registry: disk + in-mem + clear backoff.
|
||||
|
||||
Callers must hold ``_models_dev_fetch_lock`` so a failing refresh on one
|
||||
path can never stomp the state a succeeding refresh on the other path
|
||||
just committed (e.g. a failing background worker re-arming the backoff
|
||||
immediately after a successful ``force_refresh``).
|
||||
"""
|
||||
global _models_dev_cache, _models_dev_cache_time, _models_dev_retry_after
|
||||
_save_disk_cache(data)
|
||||
_models_dev_cache = data
|
||||
_models_dev_cache_time = time.time()
|
||||
_models_dev_retry_after = 0
|
||||
logger.debug(
|
||||
"Refreshed models.dev registry (%s): %d providers, %d total models",
|
||||
where,
|
||||
len(data),
|
||||
sum(len(p.get("models", {})) for p in data.values() if isinstance(p, dict)),
|
||||
)
|
||||
|
||||
|
||||
def _note_refresh_failure(exc: Exception, *, where: str) -> None:
|
||||
"""Record a failed refresh: arm the process-wide 5-minute backoff.
|
||||
|
||||
Callers must hold ``_models_dev_fetch_lock`` (see ``_commit_registry``).
|
||||
"""
|
||||
global _models_dev_retry_after
|
||||
_models_dev_retry_after = time.time() + _MODELS_DEV_RETRY_DELAY
|
||||
logger.debug(
|
||||
"models.dev refresh failed (%s); retry suppressed for %ds: %s",
|
||||
where,
|
||||
_MODELS_DEV_RETRY_DELAY,
|
||||
exc,
|
||||
)
|
||||
|
||||
|
||||
def _background_refresh_models_dev() -> None:
|
||||
"""Best-effort refresh after serving stale cache data."""
|
||||
global _models_dev_refresh_in_flight
|
||||
try:
|
||||
data = _fetch_models_dev_from_network()
|
||||
with _models_dev_fetch_lock:
|
||||
_commit_registry(data, where="background")
|
||||
except Exception as e:
|
||||
with _models_dev_fetch_lock:
|
||||
_note_refresh_failure(e, where="background")
|
||||
finally:
|
||||
with _models_dev_refresh_lock:
|
||||
_models_dev_refresh_in_flight = False
|
||||
|
||||
|
||||
def _start_background_refresh_models_dev() -> None:
|
||||
"""Start one daemon refresh worker if none is already running.
|
||||
|
||||
Honors the process-wide failure backoff: after a failed refresh,
|
||||
no new background worker is spawned until ``_models_dev_retry_after``.
|
||||
"""
|
||||
global _models_dev_refresh_in_flight
|
||||
if time.time() < _models_dev_retry_after:
|
||||
return
|
||||
with _models_dev_refresh_lock:
|
||||
if _models_dev_refresh_in_flight:
|
||||
return
|
||||
_models_dev_refresh_in_flight = True
|
||||
thread = threading.Thread(
|
||||
target=_background_refresh_models_dev,
|
||||
name="models-dev-refresh",
|
||||
daemon=True,
|
||||
)
|
||||
try:
|
||||
thread.start()
|
||||
except Exception as e:
|
||||
# Thread/fd exhaustion: clear the flag so refresh isn't disabled
|
||||
# for the rest of the process lifetime. Callers still get stale data.
|
||||
with _models_dev_refresh_lock:
|
||||
_models_dev_refresh_in_flight = False
|
||||
logger.debug("Failed to start models.dev refresh thread: %s", e)
|
||||
|
||||
|
||||
def fetch_models_dev(
|
||||
force_refresh: bool = False, *, allow_network: bool = True
|
||||
) -> Dict[str, Any]:
|
||||
"""Fetch models.dev registry. Cache hierarchy: in-mem → disk → network.
|
||||
|
||||
Returns the full registry dict keyed by provider ID, or empty dict on failure.
|
||||
|
||||
Cache hierarchy (when ``force_refresh=False``):
|
||||
1. In-memory cache, populated and < TTL old → return immediately.
|
||||
2. **Disk cache file < TTL old by mtime → load, populate in-mem, return.**
|
||||
No network call. Saves ~500 ms per cold-start agent construction;
|
||||
``models.dev`` only changes when providers add new models, so a
|
||||
1 hour staleness window is acceptable (same TTL as in-mem cache).
|
||||
3. Network fetch → on success, save to disk + in-mem and return.
|
||||
4. Network fails → fall back to ANY available disk cache (even stale)
|
||||
with a short 5 min in-mem grace period before retrying network.
|
||||
1. Fresh in-memory cache → return immediately.
|
||||
2. Stale in-memory cache → return immediately and refresh in a single
|
||||
background daemon thread. Callers never block on the network while
|
||||
any cache exists; ``models.dev`` only changes when providers add
|
||||
new models, so stale data is preferable to a foreground timeout.
|
||||
3. Disk cache file (any age) → load, populate in-mem, return
|
||||
immediately. Stale disk caches trigger the same background refresh.
|
||||
4. No cache at all → singleflight foreground network fetch. On
|
||||
success, save to disk + in-mem and return.
|
||||
5. Any failed refresh (foreground or background) suppresses further
|
||||
automatic refreshes for 5 minutes process-wide.
|
||||
|
||||
When ``force_refresh=True`` (used by ``hermes config refresh``, the
|
||||
\"refresh model catalog\" code path), stages 1 and 2 are skipped. The
|
||||
function always hits the network and only falls back to disk if the
|
||||
network call fails.
|
||||
\"refresh model catalog\" code path), cache fast paths and the failure
|
||||
backoff are bypassed; the function hits the network and only falls back
|
||||
to cached data if the call fails. When ``allow_network=False``, any
|
||||
memory or disk cache is returned regardless of age and no request is
|
||||
made — used by latency-sensitive paths (gateway route-identity checks)
|
||||
that must never wait on the network.
|
||||
"""
|
||||
global _models_dev_cache, _models_dev_cache_time
|
||||
global _models_dev_cache, _models_dev_cache_time, _models_dev_retry_after
|
||||
|
||||
if not allow_network:
|
||||
if _models_dev_cache:
|
||||
return _models_dev_cache
|
||||
disk_data = _load_disk_cache()
|
||||
if disk_data:
|
||||
_models_dev_cache = disk_data
|
||||
disk_age = _disk_cache_age_seconds()
|
||||
_models_dev_cache_time = (
|
||||
time.time() - disk_age if disk_age is not None else 0
|
||||
)
|
||||
return _models_dev_cache
|
||||
|
||||
# Stage 1: fresh in-memory cache wins. This is the hot path on
|
||||
# long-lived processes — no I/O, no system calls.
|
||||
@@ -268,54 +409,82 @@ def fetch_models_dev(force_refresh: bool = False) -> Dict[str, Any]:
|
||||
):
|
||||
return _models_dev_cache
|
||||
|
||||
# Stage 2: fresh-by-mtime disk cache short-circuits the network call.
|
||||
# Only kicks in on cold-start processes (in-mem cache is empty or
|
||||
# expired) and only when the user hasn't asked for a forced refresh.
|
||||
# Skipped if the disk cache file is missing, unreadable, or older
|
||||
# than _MODELS_DEV_CACHE_TTL.
|
||||
# Stage 2: stale in-memory cache is still better than blocking provider
|
||||
# resolution on a foreground network timeout. Refresh it in the background.
|
||||
if not force_refresh and _models_dev_cache:
|
||||
_mark_stale_cache_grace()
|
||||
_start_background_refresh_models_dev()
|
||||
logger.debug(
|
||||
"Using stale in-memory models.dev cache; refreshing in background"
|
||||
)
|
||||
return _models_dev_cache
|
||||
|
||||
# Stage 3: disk cache short-circuits the network call.
|
||||
# Only kicks in on cold-start processes (in-mem cache is empty) and only
|
||||
# when the user hasn't asked for a forced refresh. A stale disk cache is
|
||||
# deliberately usable: provider/model resolution should not hang just
|
||||
# because models.dev is unreachable.
|
||||
if not force_refresh:
|
||||
disk_age = _disk_cache_age_seconds()
|
||||
if disk_age is not None and disk_age < _MODELS_DEV_CACHE_TTL:
|
||||
if disk_age is not None:
|
||||
disk_data = _load_disk_cache()
|
||||
if disk_data:
|
||||
_models_dev_cache = disk_data
|
||||
# Anchor in-mem TTL to the disk file's age so we don't
|
||||
# extend an already-aging cache by another full hour.
|
||||
_models_dev_cache_time = time.time() - disk_age
|
||||
logger.debug(
|
||||
"Loaded models.dev from fresh disk cache "
|
||||
"(%d providers, age=%.0fs)", len(disk_data), disk_age,
|
||||
)
|
||||
if disk_age < _MODELS_DEV_CACHE_TTL:
|
||||
# Anchor in-mem TTL to the disk file's age so we don't
|
||||
# extend an already-aging cache by another full hour.
|
||||
_models_dev_cache_time = time.time() - disk_age
|
||||
logger.debug(
|
||||
"Loaded models.dev from fresh disk cache "
|
||||
"(%d providers, age=%.0fs)", len(disk_data), disk_age,
|
||||
)
|
||||
else:
|
||||
_mark_stale_cache_grace()
|
||||
_start_background_refresh_models_dev()
|
||||
logger.debug(
|
||||
"Using stale models.dev disk cache (age=%.0fs); "
|
||||
"refreshing in background",
|
||||
disk_age,
|
||||
)
|
||||
return _models_dev_cache
|
||||
|
||||
# Stage 3: network fetch.
|
||||
try:
|
||||
response = requests.get(MODELS_DEV_URL, timeout=15)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
if isinstance(data, dict) and data:
|
||||
_models_dev_cache = data
|
||||
_models_dev_cache_time = time.time()
|
||||
_save_disk_cache(data)
|
||||
logger.debug(
|
||||
"Fetched models.dev registry: %d providers, %d total models",
|
||||
len(data),
|
||||
sum(len(p.get("models", {})) for p in data.values() if isinstance(p, dict)),
|
||||
)
|
||||
# Failed automatic refreshes are process-wide. Avoid making every caller
|
||||
# retry the same unreachable endpoint while no usable cache exists.
|
||||
if not force_refresh and time.time() < _models_dev_retry_after:
|
||||
return _models_dev_cache
|
||||
|
||||
# Stage 4: singleflight foreground network fetch — only reached when no
|
||||
# memory or disk cache exists (or on force_refresh). Recheck state after
|
||||
# acquiring the lock because another caller may have refreshed or
|
||||
# established backoff while we waited.
|
||||
with _models_dev_fetch_lock:
|
||||
now = time.time()
|
||||
if not force_refresh:
|
||||
if _models_dev_cache:
|
||||
return _models_dev_cache
|
||||
if now < _models_dev_retry_after:
|
||||
return _models_dev_cache
|
||||
|
||||
try:
|
||||
data = _fetch_models_dev_from_network()
|
||||
_commit_registry(data, where="foreground")
|
||||
return data
|
||||
except Exception as e:
|
||||
logger.debug("Failed to fetch models.dev: %s", e)
|
||||
except Exception as e:
|
||||
_note_refresh_failure(e, where="foreground")
|
||||
|
||||
# Stage 4: network failed — fall back to whatever disk cache exists,
|
||||
# even if it's stale. Give it a short 5 min in-mem TTL so we retry
|
||||
# the network soon instead of serving stale data for a full hour.
|
||||
if not _models_dev_cache:
|
||||
_models_dev_cache = _load_disk_cache()
|
||||
if _models_dev_cache:
|
||||
_models_dev_cache_time = time.time() - _MODELS_DEV_CACHE_TTL + 300
|
||||
logger.debug("Loaded models.dev from disk cache (%d providers)", len(_models_dev_cache))
|
||||
# Stage 5: network failed — return any stale memory/disk cache. Cache
|
||||
# freshness remains expired; the retry-after timestamp controls when
|
||||
# the next automatic request is allowed.
|
||||
if not _models_dev_cache:
|
||||
_models_dev_cache = _load_disk_cache()
|
||||
_models_dev_cache_time = 0
|
||||
if _models_dev_cache:
|
||||
logger.debug(
|
||||
"Loaded stale models.dev disk cache (%d providers)",
|
||||
len(_models_dev_cache),
|
||||
)
|
||||
|
||||
return _models_dev_cache
|
||||
return _models_dev_cache
|
||||
|
||||
|
||||
def lookup_models_dev_context(provider: str, model: str) -> Optional[int]:
|
||||
@@ -671,7 +840,9 @@ def _parse_provider_info(provider_id: str, raw: Dict[str, Any]) -> ProviderInfo:
|
||||
# Provider-level queries
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def get_provider_info(provider_id: str) -> Optional[ProviderInfo]:
|
||||
def get_provider_info(
|
||||
provider_id: str, *, allow_network: bool = True
|
||||
) -> Optional[ProviderInfo]:
|
||||
"""Get full provider metadata from models.dev.
|
||||
|
||||
Accepts either a Hermes provider ID (e.g. "kilocode") or a models.dev
|
||||
@@ -680,7 +851,14 @@ def get_provider_info(provider_id: str) -> Optional[ProviderInfo]:
|
||||
# Resolve Hermes ID → models.dev ID
|
||||
mdev_id = PROVIDER_TO_MODELS_DEV.get(provider_id, provider_id)
|
||||
|
||||
data = fetch_models_dev()
|
||||
# NOTE: keep the zero-argument call on the default path. Dozens of test
|
||||
# sites monkeypatch fetch_models_dev with zero-arg lambdas; passing the
|
||||
# kwarg unconditionally would break them all (they raise TypeError).
|
||||
data = (
|
||||
fetch_models_dev()
|
||||
if allow_network
|
||||
else fetch_models_dev(allow_network=False)
|
||||
)
|
||||
raw = data.get(mdev_id)
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
"""Hermes gateway monitoring.
|
||||
|
||||
Service health monitoring plus redacted operational diagnostics for the
|
||||
gateway daemon, exported over OTLP to an operator-configured endpoint.
|
||||
|
||||
``emitter`` is the in-process event bus: producers (gateway status hooks,
|
||||
the diagnostic log handler) hand typed events to a fire-and-forget queue,
|
||||
and subscribers (the OTLP streamers) consume them off the hot path. The
|
||||
emitter never blocks or raises into gateway code (the hot-path invariant),
|
||||
and nothing is persisted locally — monitoring is an egress path, not a store.
|
||||
|
||||
Deliberately out of scope here: run/model/tool trajectory capture, usage
|
||||
analytics, and any content-bearing signal. Those planes are served by the
|
||||
NeMo Relay integration and its Hermes-owned subscribers.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from . import emitter, events
|
||||
|
||||
emit = emitter.emit
|
||||
get_emitter = emitter.get_emitter
|
||||
|
||||
__all__ = [
|
||||
"emitter",
|
||||
"events",
|
||||
"emit",
|
||||
"get_emitter",
|
||||
]
|
||||
@@ -0,0 +1,201 @@
|
||||
"""Content-free cron service-health and execution telemetry projection."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from typing import Any, Optional
|
||||
|
||||
from agent.monitoring.events import CronExecutionEvent
|
||||
from agent.monitoring.gateway_health import GatewayHealthSnapshot, GatewayMetric
|
||||
from cron.jobs import (
|
||||
_compute_grace_seconds,
|
||||
get_catch_up_occurrence_count,
|
||||
get_ticker_heartbeat_age,
|
||||
get_ticker_success_age,
|
||||
load_jobs,
|
||||
)
|
||||
from cron.scheduler import get_running_job_ids
|
||||
from hermes_time import now as _hermes_now
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
_KNOWN_STATUSES = {"claimed", "running", "completed", "failed", "unknown"}
|
||||
_KNOWN_SOURCES = {"builtin", "direct", "external"}
|
||||
_KNOWN_DELIVERY_OUTCOMES = {"delivered", "failed", "suppressed", "not_configured"}
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class CronHealthSnapshot:
|
||||
metrics: list[GatewayMetric]
|
||||
events: list[CronExecutionEvent]
|
||||
|
||||
|
||||
def _now() -> datetime:
|
||||
return _hermes_now()
|
||||
|
||||
|
||||
def _job_key(raw: Any) -> str:
|
||||
value = str(raw or "unknown").encode("utf-8", errors="replace")
|
||||
return f"sha256:{hashlib.sha256(value).hexdigest()[:24]}"
|
||||
|
||||
|
||||
def classify_cron_error(raw: Any) -> str:
|
||||
text = str(raw or "").lower()
|
||||
if (
|
||||
re.search(r"\b(?:authentication|authenticated|authenticate|authorization|authorized|authorize|unauthorized|forbidden)\b", text)
|
||||
or re.search(r"\bbearer\b", text)
|
||||
or re.search(r"\b(?:access|api|refresh) token\b", text)
|
||||
or re.search(r"\b(?:401|403)\b", text)
|
||||
):
|
||||
return "auth_failed"
|
||||
if "rate limit" in text or "429" in text or "quota" in text:
|
||||
return "rate_limited"
|
||||
if "timeout" in text or "timed out" in text:
|
||||
return "timeout"
|
||||
if any(value in text for value in ("network", "connection", "dns", "socket", "unreachable")):
|
||||
return "network_error"
|
||||
if "dispatch" in text or "executor" in text:
|
||||
return "dispatch_failed"
|
||||
if "interrupt" in text or "owner exited" in text or "restarted" in text:
|
||||
return "interrupted"
|
||||
if "empty response" in text:
|
||||
return "empty_response"
|
||||
if any(value in text for value in ("config", "missing", "invalid")):
|
||||
return "invalid_config"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _parse_time(raw: Any) -> Optional[datetime]:
|
||||
try:
|
||||
return datetime.fromisoformat(str(raw)) if raw else None
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _duration_ms(record: dict[str, Any]) -> Optional[int]:
|
||||
start = _parse_time(record.get("started_at")) or _parse_time(record.get("claimed_at"))
|
||||
finish = _parse_time(record.get("finished_at"))
|
||||
if start is None or finish is None:
|
||||
return None
|
||||
try:
|
||||
duration = int((finish - start).total_seconds() * 1000)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return max(0, duration)
|
||||
|
||||
|
||||
def project_execution_event(
|
||||
record: dict[str, Any], *, delivery_outcome: Optional[str] = None
|
||||
) -> CronExecutionEvent:
|
||||
status = str(record.get("status") or "unknown").lower()
|
||||
source = str(record.get("source") or "unknown").lower()
|
||||
if source not in _KNOWN_SOURCES and source != "unknown":
|
||||
source = "external"
|
||||
outcome = str(delivery_outcome).lower() if delivery_outcome is not None else None
|
||||
return CronExecutionEvent(
|
||||
status=status if status in _KNOWN_STATUSES else "unknown",
|
||||
job_key=_job_key(record.get("job_id")),
|
||||
source=source if source in _KNOWN_SOURCES else "unknown",
|
||||
duration_ms=_duration_ms(record),
|
||||
delivery_outcome=(
|
||||
outcome if outcome in _KNOWN_DELIVERY_OUTCOMES else None
|
||||
),
|
||||
error_class=(
|
||||
classify_cron_error(record.get("error"))
|
||||
if status in {"failed", "unknown"}
|
||||
else None
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def emit_execution_state(
|
||||
record: Optional[dict[str, Any]], *, delivery_outcome: Optional[str] = None
|
||||
) -> None:
|
||||
"""Best-effort lifecycle emit; terminal states synchronously cross the queue barrier."""
|
||||
if not record:
|
||||
return
|
||||
try:
|
||||
from agent.monitoring import emitter
|
||||
|
||||
event = project_execution_event(record, delivery_outcome=delivery_outcome)
|
||||
target = emitter.get_emitter()
|
||||
target.emit(event)
|
||||
if event.status in {"completed", "failed", "unknown"}:
|
||||
target.flush(timeout=1.0)
|
||||
except Exception:
|
||||
logger.debug("cron execution telemetry emit failed", exc_info=True)
|
||||
|
||||
|
||||
def _is_overdue(job: dict[str, Any], now: datetime) -> bool:
|
||||
if not job.get("enabled", True):
|
||||
return False
|
||||
next_run = _parse_time(job.get("next_run_at"))
|
||||
schedule = job.get("schedule")
|
||||
if next_run is None or not isinstance(schedule, dict):
|
||||
return False
|
||||
try:
|
||||
if next_run.tzinfo is None and now.tzinfo is not None:
|
||||
next_run = next_run.replace(tzinfo=now.tzinfo)
|
||||
lateness = (now - next_run).total_seconds()
|
||||
return lateness > _compute_grace_seconds(schedule)
|
||||
except (TypeError, ValueError):
|
||||
return False
|
||||
|
||||
|
||||
def build_cron_health_snapshot() -> CronHealthSnapshot:
|
||||
metrics: list[GatewayMetric] = []
|
||||
for name, reader in (
|
||||
("hermes.cron.scheduler.heartbeat_age_seconds", get_ticker_heartbeat_age),
|
||||
("hermes.cron.scheduler.last_success_age_seconds", get_ticker_success_age),
|
||||
):
|
||||
try:
|
||||
value = reader()
|
||||
if value is not None:
|
||||
metrics.append(GatewayMetric(name, max(0.0, float(value)), {}))
|
||||
except Exception:
|
||||
logger.debug("cron freshness metric unavailable", exc_info=True)
|
||||
|
||||
try:
|
||||
metrics.append(
|
||||
GatewayMetric(
|
||||
"hermes.cron.scheduler.catch_up_occurrences",
|
||||
get_catch_up_occurrence_count(),
|
||||
{},
|
||||
)
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("cron catch-up metric unavailable", exc_info=True)
|
||||
|
||||
try:
|
||||
jobs = load_jobs()
|
||||
enabled = [job for job in jobs if job.get("enabled", True)]
|
||||
metrics.append(GatewayMetric("hermes.cron.jobs.enabled", len(enabled), {}))
|
||||
metrics.append(
|
||||
GatewayMetric(
|
||||
"hermes.cron.jobs.overdue",
|
||||
sum(1 for job in enabled if _is_overdue(job, _now())),
|
||||
{},
|
||||
)
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("cron job metrics unavailable", exc_info=True)
|
||||
|
||||
try:
|
||||
metrics.append(
|
||||
GatewayMetric("hermes.cron.jobs.running", len(get_running_job_ids()), {})
|
||||
)
|
||||
except Exception:
|
||||
logger.debug("cron running-job metric unavailable", exc_info=True)
|
||||
return CronHealthSnapshot(metrics=metrics, events=[])
|
||||
|
||||
|
||||
__all__ = [
|
||||
"CronHealthSnapshot",
|
||||
"build_cron_health_snapshot",
|
||||
"classify_cron_error",
|
||||
"emit_execution_state",
|
||||
"project_execution_event",
|
||||
]
|
||||
@@ -0,0 +1,211 @@
|
||||
"""Monitoring emitter: fire-and-forget queue + background dispatcher.
|
||||
|
||||
The emitter is the single seam between producers (gateway status hooks, the
|
||||
diagnostic log handler) and consumers (the OTLP streamers). Its contract is
|
||||
the hot-path invariant:
|
||||
|
||||
``emit()`` MUST return in O(microseconds), MUST NOT block on disk/network,
|
||||
and MUST NEVER raise into the caller. A monitoring failure is logged
|
||||
locally and dropped — it can never affect the gateway or a session.
|
||||
|
||||
Mechanism:
|
||||
* ``emit(event)`` does a non-blocking ``queue.put_nowait`` wrapped in a bare
|
||||
except. On a full queue it drops the *oldest* event and counts the drop.
|
||||
* A daemon thread drains the queue and fans each batch out to subscribers
|
||||
(the OTLP metric/span/log streamers). Each subscriber is fail-isolated —
|
||||
a slow or raising subscriber never affects the hot path or its peers.
|
||||
|
||||
Nothing is persisted here. Monitoring is an egress path, not a local store;
|
||||
if no subscriber is attached, events simply age out of the ring buffer.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import queue
|
||||
import threading
|
||||
import time
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_MAX_QUEUE = 10_000 # ring-buffer depth; oldest dropped when full
|
||||
_DRAIN_BATCH = 256
|
||||
|
||||
|
||||
class MonitoringEmitter:
|
||||
"""Owns the queue, the dispatcher thread, and the subscriber list."""
|
||||
|
||||
def __init__(self, *, enabled: bool = True) -> None:
|
||||
self._enabled = enabled
|
||||
self._q: "queue.Queue[Dict[str, Any]]" = queue.Queue(maxsize=_MAX_QUEUE)
|
||||
self._dropped = 0
|
||||
self._dispatched = 0
|
||||
self._stop = threading.Event()
|
||||
self._started = False
|
||||
self._lock = threading.Lock()
|
||||
self._thread: Optional[threading.Thread] = None
|
||||
# Live subscribers (the OTLP streamers). Called from the dispatcher
|
||||
# thread, fully fail-isolated. Each subscriber is callable(batch: list[dict]).
|
||||
self._subscribers: list = []
|
||||
|
||||
# ── public API (hot path) ───────────────────────────────────────────────
|
||||
def emit(self, event: Any) -> None:
|
||||
"""Enqueue an event. Never blocks, never raises.
|
||||
|
||||
``event`` may be a dataclass with ``to_dict()`` or a plain dict.
|
||||
"""
|
||||
if not self._enabled:
|
||||
return
|
||||
try:
|
||||
payload = event.to_dict() if hasattr(event, "to_dict") else dict(event)
|
||||
payload.setdefault("ts_ns", time.time_ns())
|
||||
self._ensure_started()
|
||||
try:
|
||||
self._q.put_nowait(payload)
|
||||
except queue.Full:
|
||||
# Drop oldest to make room — bounded memory, newest-wins.
|
||||
try:
|
||||
self._q.get_nowait()
|
||||
self._q.task_done()
|
||||
self._dropped += 1
|
||||
self._q.put_nowait(payload)
|
||||
except Exception:
|
||||
self._dropped += 1
|
||||
except Exception: # the hot-path invariant: never propagate
|
||||
logger.debug("monitoring emit failed", exc_info=True)
|
||||
|
||||
# ── lifecycle ───────────────────────────────────────────────────────────
|
||||
def _ensure_started(self) -> None:
|
||||
if self._started:
|
||||
return
|
||||
with self._lock:
|
||||
if self._started:
|
||||
return
|
||||
self._thread = threading.Thread(
|
||||
target=self._run, name="hermes-monitoring-dispatch", daemon=True
|
||||
)
|
||||
self._thread.start()
|
||||
self._started = True
|
||||
|
||||
def _run(self) -> None:
|
||||
while not self._stop.is_set():
|
||||
try:
|
||||
first = self._q.get(timeout=0.5)
|
||||
except queue.Empty:
|
||||
continue
|
||||
batch = [first]
|
||||
while len(batch) < _DRAIN_BATCH:
|
||||
try:
|
||||
batch.append(self._q.get_nowait())
|
||||
except queue.Empty:
|
||||
break
|
||||
try:
|
||||
self._dispatch(batch)
|
||||
finally:
|
||||
for _ in batch:
|
||||
self._q.task_done()
|
||||
|
||||
def _dispatch(self, batch) -> None:
|
||||
# Fan-out to subscribers (OTLP streamers) — fully fail-isolated.
|
||||
for sub in list(self._subscribers):
|
||||
try:
|
||||
sub(batch)
|
||||
except Exception:
|
||||
logger.debug("monitoring subscriber failed", exc_info=True)
|
||||
self._dispatched += len(batch)
|
||||
|
||||
def subscribe(self, callback) -> None:
|
||||
"""Register a live batch subscriber (callable(batch: list[dict]))."""
|
||||
if callback not in self._subscribers:
|
||||
self._subscribers.append(callback)
|
||||
self._enabled = True
|
||||
|
||||
def unsubscribe(self, callback) -> None:
|
||||
try:
|
||||
self._subscribers.remove(callback)
|
||||
except ValueError:
|
||||
pass
|
||||
if not self._subscribers:
|
||||
self._enabled = False
|
||||
|
||||
# ── introspection / shutdown (tests, CLI) ───────────────────────────────
|
||||
def flush(self, timeout: float = 2.0) -> None:
|
||||
"""Wait boundedly for queued and in-flight batches to finish dispatch."""
|
||||
if timeout <= 0:
|
||||
return
|
||||
|
||||
finished = threading.Event()
|
||||
|
||||
def _wait_for_completion() -> None:
|
||||
self._q.join()
|
||||
finished.set()
|
||||
|
||||
waiter = threading.Thread(
|
||||
target=_wait_for_completion,
|
||||
name="hermes-monitoring-flush",
|
||||
daemon=True,
|
||||
)
|
||||
waiter.start()
|
||||
finished.wait(timeout=timeout)
|
||||
|
||||
def stats(self) -> Dict[str, int]:
|
||||
return {
|
||||
"queued": self._q.qsize(),
|
||||
"dispatched": self._dispatched,
|
||||
"dropped": self._dropped,
|
||||
"subscribers": len(self._subscribers),
|
||||
}
|
||||
|
||||
def close(self) -> None:
|
||||
self._stop.set()
|
||||
if self._thread is not None:
|
||||
self._thread.join(timeout=2.0)
|
||||
self._started = False
|
||||
|
||||
|
||||
# ── process-wide singleton ──────────────────────────────────────────────────
|
||||
_EMITTER: Optional[MonitoringEmitter] = None
|
||||
_EMITTER_LOCK = threading.Lock()
|
||||
|
||||
|
||||
def get_emitter() -> MonitoringEmitter:
|
||||
"""Return the process-wide monitoring emitter."""
|
||||
global _EMITTER
|
||||
if _EMITTER is not None:
|
||||
return _EMITTER
|
||||
with _EMITTER_LOCK:
|
||||
if _EMITTER is None:
|
||||
# Collection is opt-in. A plane exporter enables the singleton by
|
||||
# attaching its first subscriber; until then producers are no-ops.
|
||||
_EMITTER = MonitoringEmitter(enabled=False)
|
||||
return _EMITTER
|
||||
|
||||
|
||||
def emit(event: Any) -> None:
|
||||
"""Module-level convenience: emit via the singleton."""
|
||||
get_emitter().emit(event)
|
||||
|
||||
|
||||
def reset_emitter_for_tests(emitter: Optional[MonitoringEmitter] = None) -> None:
|
||||
"""Swap the singleton (tests only)."""
|
||||
global _EMITTER
|
||||
with _EMITTER_LOCK:
|
||||
if _EMITTER is not None and emitter is not _EMITTER:
|
||||
try:
|
||||
_EMITTER.close()
|
||||
except Exception:
|
||||
pass
|
||||
_EMITTER = emitter
|
||||
|
||||
|
||||
# Back-compat alias for the salvaged class name used in emozilla's tests.
|
||||
TelemetryEmitter = MonitoringEmitter
|
||||
|
||||
__all__ = [
|
||||
"MonitoringEmitter",
|
||||
"TelemetryEmitter",
|
||||
"get_emitter",
|
||||
"emit",
|
||||
"reset_emitter_for_tests",
|
||||
]
|
||||
@@ -0,0 +1,86 @@
|
||||
"""Typed gateway monitoring events.
|
||||
|
||||
Content-free service-health and redacted diagnostic events for the gateway
|
||||
daemon. These are the only event shapes the monitoring plane emits: no
|
||||
prompts, messages, tool args/results, session history, or usage analytics.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from dataclasses import dataclass, field, asdict
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
|
||||
def _now_ns() -> int:
|
||||
return time.time_ns()
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class GatewayHealthEvent:
|
||||
"""Content-free gateway health snapshot or lifecycle event."""
|
||||
|
||||
name: str
|
||||
gateway_state: Optional[str] = None
|
||||
old_state: Optional[str] = None
|
||||
new_state: Optional[str] = None
|
||||
exit_reason: Optional[str] = None
|
||||
restart_requested: Optional[bool] = None
|
||||
active_agents: int = 0
|
||||
gateway_busy: bool = False
|
||||
gateway_drainable: bool = False
|
||||
platform_count: int = 0
|
||||
fatal_platform_count: int = 0
|
||||
profile: Optional[str] = None
|
||||
install_id: Optional[str] = None
|
||||
version: Optional[str] = None
|
||||
supervision_mode: Optional[str] = None
|
||||
pid: Optional[int] = None
|
||||
ts_ns: int = field(default_factory=_now_ns)
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"event": "gateway_health", **asdict(self)}
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class GatewayDiagnosticEvent:
|
||||
"""Redacted gateway diagnostic event for operator-owned observability."""
|
||||
|
||||
name: str
|
||||
subsystem: str
|
||||
error_class: str = "unknown"
|
||||
error_code: Optional[str] = None
|
||||
platform: Optional[str] = None
|
||||
old_state: Optional[str] = None
|
||||
new_state: Optional[str] = None
|
||||
profile: Optional[str] = None
|
||||
version: Optional[str] = None
|
||||
severity: str = "warning"
|
||||
ts_ns: int = field(default_factory=_now_ns)
|
||||
source_logger: Optional[str] = None
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"event": "gateway_diagnostic", **asdict(self)}
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class CronExecutionEvent:
|
||||
"""Content-free durable cron execution lifecycle projection."""
|
||||
|
||||
status: str
|
||||
job_key: str
|
||||
source: str = "unknown"
|
||||
duration_ms: Optional[int] = None
|
||||
delivery_outcome: Optional[str] = None
|
||||
error_class: Optional[str] = None
|
||||
ts_ns: int = field(default_factory=_now_ns)
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {"event": "cron_execution", **asdict(self)}
|
||||
|
||||
|
||||
__all__ = [
|
||||
"GatewayHealthEvent",
|
||||
"GatewayDiagnosticEvent",
|
||||
"CronExecutionEvent",
|
||||
]
|
||||
@@ -0,0 +1,469 @@
|
||||
"""Gateway health and diagnostics signal producer.
|
||||
|
||||
This module keeps the plane narrow: service health monitoring plus
|
||||
redacted operational diagnostics. It reuses the existing gateway runtime-status
|
||||
contract and emits content-free metrics/events. No prompts, messages, tool args,
|
||||
session history, audit records, or product analytics belong here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from agent.monitoring.events import GatewayDiagnosticEvent, GatewayHealthEvent
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class GatewayMetric:
|
||||
name: str
|
||||
value: int | float
|
||||
attributes: Dict[str, str]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class GatewayHealthSnapshot:
|
||||
metrics: List[GatewayMetric]
|
||||
events: List[GatewayHealthEvent | GatewayDiagnosticEvent]
|
||||
|
||||
|
||||
_RUNNING_PLATFORM_STATES = {"running", "connected", "ok", "ready"}
|
||||
_FATAL_PLATFORM_STATES = {"fatal", "degraded", "error", "failed"}
|
||||
_KNOWN_GATEWAY_STATES = {
|
||||
"starting", "draining", "stopping", "stopped", "startup_failed", "unknown"
|
||||
} | _RUNNING_PLATFORM_STATES | _FATAL_PLATFORM_STATES
|
||||
_KNOWN_PLATFORM_STATES = _RUNNING_PLATFORM_STATES | _FATAL_PLATFORM_STATES | {
|
||||
"connecting", "disconnected", "disabled", "paused", "retrying", "unknown"
|
||||
}
|
||||
_SUPERVISION_MODES = {"systemd", "s6", "container", "launchd", "manual", "unknown"}
|
||||
_SOURCE_LOGGER_RE = re.compile(r"^gateway(?:\.[A-Za-z_][A-Za-z0-9_]*)*$")
|
||||
|
||||
|
||||
def _allowed_logger(name: str) -> bool:
|
||||
return name == "gateway" or name.startswith("gateway.")
|
||||
|
||||
|
||||
def source_logger_for_export(name: Any) -> Optional[str]:
|
||||
"""Return a bounded source-controlled gateway logger name for OTLP scope."""
|
||||
value = str(name or "")
|
||||
return value if len(value) <= 128 and _SOURCE_LOGGER_RE.fullmatch(value) else None
|
||||
|
||||
|
||||
def redact_gateway_message(message: Any) -> str:
|
||||
"""Redact gateway diagnostic free text for operator-owned export.
|
||||
|
||||
Single scrub path: everything goes through
|
||||
``agent.monitoring.redaction.redact_for_export`` (unconditional
|
||||
secrets + PII), then is length-bounded.
|
||||
"""
|
||||
try:
|
||||
from agent.monitoring.redaction import redact_for_export
|
||||
redacted = redact_for_export(str(message or "")) or ""
|
||||
except Exception:
|
||||
redacted = "[redaction-unavailable]"
|
||||
return redacted[:500]
|
||||
|
||||
|
||||
def classify_gateway_error(raw: Any) -> str:
|
||||
s = str(raw or "").lower()
|
||||
if any(k in s for k in ("auth", "token", "unauthorized", "forbidden", "401", "403")):
|
||||
return "auth_failed"
|
||||
if "rate" in s and "limit" in s:
|
||||
return "rate_limited"
|
||||
if "timeout" in s or "timed out" in s:
|
||||
return "timeout"
|
||||
if any(
|
||||
k in s
|
||||
for k in (
|
||||
"network",
|
||||
"connection",
|
||||
"dns",
|
||||
"socket",
|
||||
"connect call failed",
|
||||
"failed to connect",
|
||||
"cannot connect",
|
||||
"unreachable",
|
||||
"name resolution",
|
||||
)
|
||||
):
|
||||
return "network_error"
|
||||
if any(k in s for k in ("config", "missing", "invalid")):
|
||||
return "invalid_config"
|
||||
if "startup" in s:
|
||||
return "startup_failed"
|
||||
if "fatal" in s:
|
||||
return "platform_fatal"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def classify_exit_reason(
|
||||
raw: Any, *, state: Any, restart_requested: bool
|
||||
) -> Optional[str]:
|
||||
"""Reduce free-form shutdown text to a bounded operational class."""
|
||||
if restart_requested:
|
||||
return "restart_requested"
|
||||
state_name = str(state or "").lower()
|
||||
if raw is None and state_name != "startup_failed":
|
||||
return None
|
||||
classified = classify_gateway_error(raw)
|
||||
if state_name == "startup_failed":
|
||||
return classified if classified != "unknown" else "startup_failed"
|
||||
text = str(raw or "").lower()
|
||||
if "signal" in text or "sigterm" in text or "sigint" in text:
|
||||
return "signal"
|
||||
if state_name == "stopped" and any(word in text for word in ("shutdown", "stop")):
|
||||
return "planned_stop"
|
||||
return classified
|
||||
|
||||
|
||||
def _bounded_state(raw: Any, *, allowed: set[str]) -> str:
|
||||
state = str(raw or "unknown").lower()
|
||||
return state if state in allowed else "unknown"
|
||||
|
||||
|
||||
def _safe_metric_value(raw: Any, *, limit: int = 128) -> str:
|
||||
try:
|
||||
from agent.monitoring.redaction import redact_for_export
|
||||
value = redact_for_export(str(raw or "")) or "unknown"
|
||||
except Exception:
|
||||
return "unknown"
|
||||
return value[:limit]
|
||||
|
||||
|
||||
def _safe_instance_id(raw: Any) -> str:
|
||||
"""Return a stable opaque instance key without exporting the source ID."""
|
||||
value = str(raw or "unknown").encode("utf-8", errors="replace")
|
||||
return f"sha256:{hashlib.sha256(value).hexdigest()[:24]}"
|
||||
|
||||
|
||||
def subsystem_for_logger(logger_name: str) -> str:
|
||||
if logger_name == "gateway.relay" or logger_name.startswith("gateway.relay."):
|
||||
return "platform.relay"
|
||||
if logger_name.startswith("gateway.platforms."):
|
||||
parts = logger_name.split(".")
|
||||
if len(parts) >= 3 and parts[2]:
|
||||
return f"platform.{parts[2]}"
|
||||
if logger_name.startswith("gateway.platforms"):
|
||||
return "platform"
|
||||
if logger_name.startswith("gateway"):
|
||||
return "gateway"
|
||||
return "gateway"
|
||||
|
||||
|
||||
def platform_for_subsystem(subsystem: str) -> Optional[str]:
|
||||
if subsystem.startswith("platform."):
|
||||
return subsystem.split(".", 1)[1] or None
|
||||
return None
|
||||
|
||||
|
||||
def _parse_active_agents(raw: Any) -> int:
|
||||
try:
|
||||
from gateway.status import parse_active_agents
|
||||
return parse_active_agents(raw)
|
||||
except Exception:
|
||||
try:
|
||||
return max(0, int(raw))
|
||||
except (TypeError, ValueError):
|
||||
return 0
|
||||
|
||||
|
||||
def _derive_busy(gateway_running: bool, gateway_state: Any, active_agents: Any) -> bool:
|
||||
try:
|
||||
from gateway.status import derive_gateway_busy
|
||||
return derive_gateway_busy(
|
||||
gateway_running=gateway_running,
|
||||
gateway_state=gateway_state,
|
||||
active_agents=active_agents,
|
||||
)
|
||||
except Exception:
|
||||
return bool(gateway_running and gateway_state == "running" and _parse_active_agents(active_agents) > 0)
|
||||
|
||||
|
||||
def _derive_drainable(gateway_running: bool, gateway_state: Any) -> bool:
|
||||
try:
|
||||
from gateway.status import derive_gateway_drainable
|
||||
return derive_gateway_drainable(gateway_running=gateway_running, gateway_state=gateway_state)
|
||||
except Exception:
|
||||
return bool(gateway_running and gateway_state == "running")
|
||||
|
||||
|
||||
def _base_attrs(*, profile: str, install_id: str, version: str, supervision_mode: str) -> Dict[str, str]:
|
||||
mode = str(supervision_mode or "unknown").lower()
|
||||
return {
|
||||
"service.instance.id": _safe_instance_id(install_id),
|
||||
"service.version": _safe_metric_value(version, limit=64),
|
||||
"hermes.supervision_mode": mode if mode in _SUPERVISION_MODES else "unknown",
|
||||
}
|
||||
|
||||
|
||||
def _metric(name: str, value: int | float, attrs: Dict[str, str], **extra: str) -> GatewayMetric:
|
||||
out = dict(attrs)
|
||||
for key, val in extra.items():
|
||||
if val is not None:
|
||||
out[key] = _safe_metric_value(val)
|
||||
return GatewayMetric(name=name, value=value, attributes=out)
|
||||
|
||||
|
||||
def build_gateway_health_snapshot(
|
||||
runtime: Optional[dict[str, Any]],
|
||||
*,
|
||||
gateway_running: bool,
|
||||
profile: str,
|
||||
install_id: str,
|
||||
version: str,
|
||||
supervision_mode: str = "unknown",
|
||||
) -> GatewayHealthSnapshot:
|
||||
"""Convert gateway_state.json-compatible runtime state into P0 signals."""
|
||||
runtime = runtime or {}
|
||||
gateway_state = _bounded_state(
|
||||
runtime.get("gateway_state"), allowed=_KNOWN_GATEWAY_STATES
|
||||
)
|
||||
active_agents = _parse_active_agents(runtime.get("active_agents", 0))
|
||||
busy = _derive_busy(gateway_running, gateway_state, active_agents)
|
||||
drainable = _derive_drainable(gateway_running, gateway_state)
|
||||
platforms = runtime.get("platforms") if isinstance(runtime.get("platforms"), dict) else {}
|
||||
base = _base_attrs(profile=profile, install_id=install_id, version=version, supervision_mode=supervision_mode)
|
||||
|
||||
metrics: list[GatewayMetric] = [
|
||||
_metric("hermes.gateway.up", 1 if gateway_running else 0, base),
|
||||
_metric("hermes.gateway.active_agents", active_agents, base),
|
||||
_metric("hermes.gateway.busy", 1 if busy else 0, base),
|
||||
_metric("hermes.gateway.drainable", 1 if drainable else 0, base),
|
||||
_metric("hermes.gateway.restart_requested", 1 if runtime.get("restart_requested") else 0, base),
|
||||
]
|
||||
if gateway_state:
|
||||
metrics.append(_metric("hermes.gateway.state", 1, base, **{"hermes.gateway.state": str(gateway_state)}))
|
||||
|
||||
fatal_count = 0
|
||||
events: list[GatewayHealthEvent | GatewayDiagnosticEvent] = []
|
||||
for platform, pdata in platforms.items():
|
||||
pdata = pdata if isinstance(pdata, dict) else {}
|
||||
state = _bounded_state(
|
||||
pdata.get("state"), allowed=_KNOWN_PLATFORM_STATES
|
||||
)
|
||||
raw_error = pdata.get("error_code") or pdata.get("error_message")
|
||||
error_code = classify_gateway_error(raw_error)
|
||||
is_up = state in _RUNNING_PLATFORM_STATES
|
||||
is_degraded = state in _FATAL_PLATFORM_STATES
|
||||
if is_degraded:
|
||||
fatal_count += 1
|
||||
metrics.append(_metric(
|
||||
"hermes.platform.up",
|
||||
1 if is_up else 0,
|
||||
base,
|
||||
**{"hermes.platform": str(platform), "hermes.platform.state": state},
|
||||
))
|
||||
metrics.append(_metric(
|
||||
"hermes.platform.degraded",
|
||||
1 if is_degraded else 0,
|
||||
base,
|
||||
**{"hermes.platform": str(platform), "hermes.platform.state": state, "hermes.error_code": error_code},
|
||||
))
|
||||
if is_degraded:
|
||||
events.append(GatewayDiagnosticEvent(
|
||||
name="platform.fatal",
|
||||
subsystem=f"platform.{platform}",
|
||||
platform=str(platform),
|
||||
error_code=error_code,
|
||||
error_class=classify_gateway_error(error_code or pdata.get("error_message")),
|
||||
profile=profile,
|
||||
version=version,
|
||||
severity="error" if state == "fatal" else "warning",
|
||||
))
|
||||
|
||||
events.insert(0, GatewayHealthEvent(
|
||||
name="gateway.health_snapshot",
|
||||
gateway_state=str(gateway_state) if gateway_state is not None else None,
|
||||
active_agents=active_agents,
|
||||
gateway_busy=busy,
|
||||
gateway_drainable=drainable,
|
||||
platform_count=len(platforms),
|
||||
fatal_platform_count=fatal_count,
|
||||
profile=profile,
|
||||
install_id=install_id,
|
||||
version=version,
|
||||
supervision_mode=supervision_mode,
|
||||
pid=_coerce_pid(runtime.get("pid")),
|
||||
))
|
||||
return GatewayHealthSnapshot(metrics=metrics, events=events)
|
||||
|
||||
|
||||
def _safe_profile() -> str:
|
||||
try:
|
||||
from hermes_cli.profiles import get_active_profile_name
|
||||
return str(get_active_profile_name() or "default")
|
||||
except Exception:
|
||||
return "default"
|
||||
|
||||
|
||||
def _safe_version() -> str:
|
||||
try:
|
||||
from hermes_cli import __version__
|
||||
return str(__version__)
|
||||
except Exception:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def emit_runtime_status_transition(previous: Optional[dict[str, Any]], current: dict[str, Any]) -> None:
|
||||
"""Emit immediate content-free gateway events for runtime status changes.
|
||||
|
||||
Called by gateway.status.write_runtime_status after persisting the new status.
|
||||
Fully fail-open: failures never affect gateway status writes.
|
||||
"""
|
||||
try:
|
||||
from agent.monitoring import emitter
|
||||
out: list[GatewayHealthEvent | GatewayDiagnosticEvent] = []
|
||||
profile = _safe_profile()
|
||||
version = _safe_version()
|
||||
old_gateway_state = _bounded_state(
|
||||
(previous or {}).get("gateway_state"), allowed=_KNOWN_GATEWAY_STATES
|
||||
) if (previous or {}).get("gateway_state") is not None else None
|
||||
new_gateway_state = _bounded_state(
|
||||
current.get("gateway_state"), allowed=_KNOWN_GATEWAY_STATES
|
||||
) if current.get("gateway_state") is not None else None
|
||||
if old_gateway_state != new_gateway_state and new_gateway_state:
|
||||
out.append(GatewayHealthEvent(
|
||||
name="gateway.lifecycle",
|
||||
gateway_state=new_gateway_state,
|
||||
old_state=old_gateway_state,
|
||||
new_state=new_gateway_state,
|
||||
exit_reason=classify_exit_reason(
|
||||
current.get("exit_reason"),
|
||||
state=new_gateway_state,
|
||||
restart_requested=bool(current.get("restart_requested")),
|
||||
),
|
||||
restart_requested=bool(current.get("restart_requested")),
|
||||
active_agents=_parse_active_agents(current.get("active_agents", 0)),
|
||||
profile=profile,
|
||||
version=version,
|
||||
pid=_coerce_pid(current.get("pid")),
|
||||
))
|
||||
if new_gateway_state == "startup_failed":
|
||||
out.append(GatewayDiagnosticEvent(
|
||||
name="gateway.startup_failed",
|
||||
subsystem="gateway",
|
||||
error_class=classify_gateway_error(current.get("exit_reason") or "startup_failed"),
|
||||
error_code=classify_gateway_error(current.get("exit_reason") or "startup_failed"),
|
||||
profile=profile,
|
||||
version=version,
|
||||
severity="error",
|
||||
))
|
||||
if new_gateway_state == "stopped":
|
||||
out.append(GatewayHealthEvent(
|
||||
name="gateway.exit",
|
||||
gateway_state=new_gateway_state,
|
||||
old_state=old_gateway_state,
|
||||
new_state=new_gateway_state,
|
||||
exit_reason=classify_exit_reason(
|
||||
current.get("exit_reason"),
|
||||
state=new_gateway_state,
|
||||
restart_requested=bool(current.get("restart_requested")),
|
||||
),
|
||||
restart_requested=bool(current.get("restart_requested")),
|
||||
active_agents=_parse_active_agents(current.get("active_agents", 0)),
|
||||
profile=profile,
|
||||
version=version,
|
||||
pid=_coerce_pid(current.get("pid")),
|
||||
))
|
||||
|
||||
old_platforms_raw = (previous or {}).get("platforms")
|
||||
new_platforms_raw = current.get("platforms")
|
||||
old_platforms = old_platforms_raw if isinstance(old_platforms_raw, dict) else {}
|
||||
new_platforms = new_platforms_raw if isinstance(new_platforms_raw, dict) else {}
|
||||
for platform, pdata in new_platforms.items():
|
||||
pdata = pdata if isinstance(pdata, dict) else {}
|
||||
prev_raw = old_platforms.get(platform, {})
|
||||
prev = prev_raw if isinstance(prev_raw, dict) else {}
|
||||
old_state = _bounded_state(
|
||||
prev.get("state"), allowed=_KNOWN_PLATFORM_STATES
|
||||
) if prev.get("state") is not None else None
|
||||
new_state = _bounded_state(
|
||||
pdata.get("state"), allowed=_KNOWN_PLATFORM_STATES
|
||||
) if pdata.get("state") is not None else None
|
||||
if old_state == new_state or not new_state:
|
||||
continue
|
||||
error_code = classify_gateway_error(pdata.get("error_code") or pdata.get("error_message"))
|
||||
severity = "error" if new_state.lower() in {"fatal", "failed", "error"} else "warning"
|
||||
out.append(GatewayDiagnosticEvent(
|
||||
name="platform.state_change",
|
||||
subsystem=f"platform.{platform}",
|
||||
platform=str(platform),
|
||||
old_state=old_state,
|
||||
new_state=new_state,
|
||||
error_code=error_code,
|
||||
error_class=error_code,
|
||||
profile=profile,
|
||||
version=version,
|
||||
severity=severity,
|
||||
))
|
||||
if new_state.lower() in _FATAL_PLATFORM_STATES:
|
||||
out.append(GatewayDiagnosticEvent(
|
||||
name="platform.fatal",
|
||||
subsystem=f"platform.{platform}",
|
||||
platform=str(platform),
|
||||
error_code=error_code,
|
||||
error_class=error_code,
|
||||
profile=profile,
|
||||
version=version,
|
||||
severity=severity,
|
||||
))
|
||||
for ev in out:
|
||||
emitter.emit(ev)
|
||||
except Exception:
|
||||
logging.getLogger(__name__).debug("gateway runtime status transition emit failed", exc_info=True)
|
||||
|
||||
|
||||
def _coerce_pid(raw: Any) -> Optional[int]:
|
||||
try:
|
||||
pid = int(raw)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return pid if pid > 0 else None
|
||||
|
||||
|
||||
class GatewayDiagnosticLogHandler(logging.Handler):
|
||||
"""Allowlisted warning/error bridge for gateway-owned diagnostics."""
|
||||
|
||||
def __init__(self, *, profile: str = "default", version: str = "unknown") -> None:
|
||||
super().__init__(level=logging.WARNING)
|
||||
self.profile = profile
|
||||
self.version = version
|
||||
|
||||
def emit(self, record: logging.LogRecord) -> None:
|
||||
try:
|
||||
if record.levelno < logging.WARNING:
|
||||
return
|
||||
if not _allowed_logger(record.name):
|
||||
return
|
||||
subsystem = subsystem_for_logger(record.name)
|
||||
message = record.getMessage()
|
||||
error_class = classify_gateway_error(message)
|
||||
event = GatewayDiagnosticEvent(
|
||||
name=f"gateway.log.{record.levelname.lower()}",
|
||||
subsystem=subsystem,
|
||||
source_logger=source_logger_for_export(record.name),
|
||||
platform=platform_for_subsystem(subsystem),
|
||||
error_class=error_class,
|
||||
error_code=error_class,
|
||||
profile=self.profile,
|
||||
version=self.version,
|
||||
severity=record.levelname.lower(),
|
||||
)
|
||||
from agent.monitoring import emitter
|
||||
emitter.get_emitter().emit(event)
|
||||
except Exception:
|
||||
logging.getLogger(__name__).debug("gateway diagnostic emit failed", exc_info=True)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"GatewayMetric",
|
||||
"GatewayHealthSnapshot",
|
||||
"GatewayDiagnosticLogHandler",
|
||||
"build_gateway_health_snapshot",
|
||||
"classify_gateway_error",
|
||||
"source_logger_for_export",
|
||||
"redact_gateway_message",
|
||||
]
|
||||
@@ -0,0 +1,643 @@
|
||||
"""Gateway Health & Diagnostics OTLP export runtime.
|
||||
|
||||
This exporter emits operator-owned gateway service-health metrics plus
|
||||
narrow redacted diagnostic events. It is deliberately in-process and fail-open so
|
||||
it works under systemd, launchd, s6, containers, tmux, nohup, or a simple shell
|
||||
without a sidecar/watchdog dependency.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import threading
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_DEFAULT_DIAGNOSTIC_SCOPE = "hermes.gateway.diagnostics"
|
||||
|
||||
_RESOURCE_ATTRIBUTE_KEYS = frozenset({
|
||||
"service.name",
|
||||
"service.namespace",
|
||||
"service.version",
|
||||
"service.instance.id",
|
||||
"deployment.environment.name",
|
||||
"cloud.provider",
|
||||
"cloud.platform",
|
||||
"cloud.region",
|
||||
"telemetry.scope",
|
||||
})
|
||||
_DIAGNOSTIC_ATTRIBUTE_KEYS = frozenset({
|
||||
"name",
|
||||
"subsystem",
|
||||
"error_class",
|
||||
"error_code",
|
||||
"platform",
|
||||
"old_state",
|
||||
"new_state",
|
||||
"version",
|
||||
"severity",
|
||||
})
|
||||
_SAFE_RESOURCE_VALUE = re.compile(r"^[A-Za-z0-9._:/-]{1,128}$")
|
||||
|
||||
|
||||
def _redact_string(raw: Any, *, limit: int = 500) -> str:
|
||||
try:
|
||||
from agent.monitoring.redaction import redact_for_export
|
||||
return (redact_for_export(str(raw or "")) or "[redacted]")[:limit]
|
||||
except Exception:
|
||||
return "[redaction-unavailable]"
|
||||
|
||||
|
||||
def _safe_resource_attributes(raw: Any) -> Dict[str, str]:
|
||||
"""Allowlist bounded resource labels and reject values changed by redaction."""
|
||||
attrs: Dict[str, str] = {}
|
||||
if not isinstance(raw, dict):
|
||||
return attrs
|
||||
for key, value in raw.items():
|
||||
key = str(key)
|
||||
if key not in _RESOURCE_ATTRIBUTE_KEYS or value is None:
|
||||
continue
|
||||
if key == "service.instance.id":
|
||||
from agent.monitoring.gateway_health import _safe_instance_id
|
||||
attrs[key] = _safe_instance_id(value)
|
||||
continue
|
||||
text = str(value)
|
||||
if not _SAFE_RESOURCE_VALUE.fullmatch(text):
|
||||
continue
|
||||
if _redact_string(text, limit=128) != text:
|
||||
continue
|
||||
attrs[key] = text
|
||||
return attrs
|
||||
|
||||
|
||||
def _runtime_resource_attributes(
|
||||
config: Dict[str, Any], *, telemetry_scope: str
|
||||
) -> Dict[str, str]:
|
||||
"""Build the safe OTLP resource shared by metrics and diagnostic logs."""
|
||||
gh = _gateway_health_config(config)
|
||||
attrs = _safe_resource_attributes(gh.get("resource_attributes"))
|
||||
from agent.monitoring.gateway_health import _safe_instance_id
|
||||
|
||||
attrs["service.name"] = "hermes-gateway"
|
||||
attrs["service.instance.id"] = _safe_instance_id(_install_id(config))
|
||||
attrs["telemetry.scope"] = telemetry_scope
|
||||
return attrs
|
||||
|
||||
|
||||
def _diagnostic_log_attributes(event: Dict[str, Any]) -> Dict[str, Any]:
|
||||
attrs: Dict[str, Any] = {}
|
||||
for key in _DIAGNOSTIC_ATTRIBUTE_KEYS:
|
||||
value = event.get(key)
|
||||
if value is None:
|
||||
continue
|
||||
attrs[f"hermes.{key}"] = _redact_string(value) if isinstance(value, str) else value
|
||||
return attrs
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class GatewayHealthExportRuntime:
|
||||
enabled: bool
|
||||
reason: str = "disabled"
|
||||
streamer: Any = None
|
||||
metric_provider: Any = None
|
||||
log_handler: Any = None
|
||||
log_streamer: Any = None
|
||||
thread: Optional[threading.Thread] = None
|
||||
stop_event: Optional[threading.Event] = None
|
||||
|
||||
def shutdown(self) -> None:
|
||||
if self.stop_event is not None:
|
||||
self.stop_event.set()
|
||||
if self.thread is not None:
|
||||
self.thread.join(timeout=0.25)
|
||||
if self.log_handler is not None:
|
||||
try:
|
||||
logging.getLogger().removeHandler(self.log_handler)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# All producers above are now stopped. Drain queued and in-flight
|
||||
# events before detaching subscribers so the terminal lifecycle event
|
||||
# cannot race exporter shutdown. The barrier is bounded and fail-open.
|
||||
try:
|
||||
from agent.monitoring.emitter import get_emitter
|
||||
emitter = get_emitter()
|
||||
emitter.flush(timeout=1.0)
|
||||
if self.streamer is not None:
|
||||
emitter.unsubscribe(self.streamer)
|
||||
if self.log_streamer is not None:
|
||||
emitter.unsubscribe(self.log_streamer)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Network flush/close runs under one bounded daemon-thread deadline and
|
||||
# can never delay gateway teardown indefinitely.
|
||||
closeables = [
|
||||
item for item in (self.streamer, self.log_streamer, self.metric_provider)
|
||||
if item is not None
|
||||
]
|
||||
|
||||
def _close() -> None:
|
||||
for item in closeables:
|
||||
try:
|
||||
item.shutdown()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if closeables:
|
||||
worker = threading.Thread(
|
||||
target=_close,
|
||||
name="hermes-gateway-health-export-shutdown",
|
||||
daemon=True,
|
||||
)
|
||||
worker.start()
|
||||
worker.join(timeout=2.0)
|
||||
|
||||
self.streamer = None
|
||||
self.log_streamer = None
|
||||
self.metric_provider = None
|
||||
self.thread = None
|
||||
self.stop_event = None
|
||||
|
||||
|
||||
def _gateway_health_config(config: Dict[str, Any]) -> Dict[str, Any]:
|
||||
mon = (config or {}).get("monitoring") or {}
|
||||
return mon.get("gateway_health_export") or {}
|
||||
|
||||
|
||||
def _otlp_config(config: Dict[str, Any]) -> Dict[str, Any]:
|
||||
mon = (config or {}).get("monitoring") or {}
|
||||
export = mon.get("export") or {}
|
||||
return export.get("otlp") or {}
|
||||
|
||||
|
||||
def _enabled(config: Dict[str, Any]) -> bool:
|
||||
gh = _gateway_health_config(config)
|
||||
otlp = _otlp_config(config)
|
||||
return bool(gh.get("enabled") and otlp.get("enabled") and otlp.get("endpoint"))
|
||||
|
||||
|
||||
def _require_metrics_sdk(*, auto_install: bool = True, prompt: bool = False) -> Dict[str, Any]:
|
||||
if auto_install:
|
||||
try:
|
||||
from tools.lazy_deps import ensure as _lazy_ensure
|
||||
_lazy_ensure("export.otlp", prompt=prompt)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
from opentelemetry.exporter.otlp.proto.http._log_exporter import OTLPLogExporter
|
||||
from opentelemetry.exporter.otlp.proto.http.metric_exporter import OTLPMetricExporter
|
||||
from opentelemetry.metrics import Observation
|
||||
from opentelemetry.trace import INVALID_SPAN_ID, INVALID_TRACE_ID, TraceFlags
|
||||
from opentelemetry._logs import LogRecord
|
||||
from opentelemetry._logs.severity import SeverityNumber
|
||||
from opentelemetry.sdk._logs import LoggerProvider
|
||||
from opentelemetry.sdk._logs.export import BatchLogRecordProcessor
|
||||
from opentelemetry.sdk.metrics import MeterProvider
|
||||
from opentelemetry.sdk.metrics.export import PeriodicExportingMetricReader
|
||||
from opentelemetry.sdk.resources import Resource
|
||||
return {
|
||||
"OTLPLogExporter": OTLPLogExporter,
|
||||
"OTLPMetricExporter": OTLPMetricExporter,
|
||||
"Observation": Observation,
|
||||
"LogRecord": LogRecord,
|
||||
"LoggerProvider": LoggerProvider,
|
||||
"INVALID_SPAN_ID": INVALID_SPAN_ID,
|
||||
"INVALID_TRACE_ID": INVALID_TRACE_ID,
|
||||
"TraceFlags": TraceFlags,
|
||||
"SeverityNumber": SeverityNumber,
|
||||
"BatchLogRecordProcessor": BatchLogRecordProcessor,
|
||||
"MeterProvider": MeterProvider,
|
||||
"PeriodicExportingMetricReader": PeriodicExportingMetricReader,
|
||||
"Resource": Resource,
|
||||
}
|
||||
except Exception as exc:
|
||||
raise RuntimeError(f"OTLP metrics SDK unavailable: {exc}") from exc
|
||||
|
||||
|
||||
def _resolve_headers(headers_env: Optional[Dict[str, str]]) -> Dict[str, str]:
|
||||
resolved: Dict[str, str] = {}
|
||||
for header_name, env_name in (headers_env or {}).items():
|
||||
val = os.environ.get(str(env_name))
|
||||
if val:
|
||||
resolved[str(header_name)] = val
|
||||
return resolved
|
||||
|
||||
|
||||
def _metric_endpoint(endpoint: str) -> str:
|
||||
if endpoint.endswith("/v1/traces"):
|
||||
return endpoint[: -len("/v1/traces")] + "/v1/metrics"
|
||||
return endpoint
|
||||
|
||||
|
||||
def _logs_endpoint(endpoint: str) -> str:
|
||||
if endpoint.endswith("/v1/traces"):
|
||||
return endpoint[: -len("/v1/traces")] + "/v1/logs"
|
||||
if endpoint.endswith("/v1/metrics"):
|
||||
return endpoint[: -len("/v1/metrics")] + "/v1/logs"
|
||||
return endpoint
|
||||
|
||||
|
||||
def _version() -> str:
|
||||
try:
|
||||
from hermes_cli import __version__
|
||||
return str(__version__)
|
||||
except Exception:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _profile() -> str:
|
||||
try:
|
||||
from hermes_cli.profiles import get_active_profile_name
|
||||
return str(get_active_profile_name() or "default")
|
||||
except Exception:
|
||||
return "default"
|
||||
|
||||
|
||||
def _install_id(config: Dict[str, Any]) -> str:
|
||||
try:
|
||||
from agent.monitoring.policy import ensure_install_id
|
||||
return str(ensure_install_id(config))
|
||||
except Exception:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _supervision_mode() -> str:
|
||||
if os.environ.get("INVOCATION_ID"):
|
||||
return "systemd"
|
||||
if os.environ.get("S6_CMD_ARG0") or os.environ.get("S6_VERSION"):
|
||||
return "s6"
|
||||
if os.environ.get("container") or os.path.exists("/.dockerenv"):
|
||||
return "container"
|
||||
if os.environ.get("LAUNCHD_SOCKET"):
|
||||
return "launchd"
|
||||
return "manual"
|
||||
|
||||
|
||||
def _read_gateway_snapshot(config: Dict[str, Any]):
|
||||
from agent.monitoring.gateway_health import build_gateway_health_snapshot
|
||||
try:
|
||||
from gateway.status import read_runtime_status
|
||||
runtime = read_runtime_status() or {}
|
||||
except Exception:
|
||||
runtime = {}
|
||||
return build_gateway_health_snapshot(
|
||||
runtime,
|
||||
gateway_running=True,
|
||||
profile=_profile(),
|
||||
install_id=_install_id(config),
|
||||
version=_version(),
|
||||
supervision_mode=_supervision_mode(),
|
||||
)
|
||||
|
||||
|
||||
def _read_cron_snapshot():
|
||||
from agent.monitoring.cron_health import build_cron_health_snapshot
|
||||
|
||||
return build_cron_health_snapshot()
|
||||
|
||||
|
||||
def _read_background_work_count() -> int:
|
||||
"""Count live background/subagent work that ``active_agents`` does NOT include.
|
||||
|
||||
``hermes.gateway.active_agents`` counts foreground turns + in-flight cron
|
||||
jobs + API runs, but deliberately excludes backgrounded ``delegate_task``
|
||||
subagents, ``terminal(background=true)`` processes, kanban workers, and the
|
||||
runner's own background tasks (they are tracked only for the scale-to-zero
|
||||
suspend guard, ``_scale_to_zero_has_live_background_work``). Without this
|
||||
metric a peer churning through delegated subagents shows ``active_agents=0``
|
||||
on the fleet dashboard. Best-effort and content-free: a single integer,
|
||||
no job/task identity. Returns 0 if a source can't be imported.
|
||||
|
||||
Delegation is counted TASK-granular (``active_task_count``): a fan-out batch
|
||||
of N subagents contributes N, not 1, so the metric reflects real concurrent
|
||||
subagent load rather than dispatch-unit/pool-slot count. This intentionally
|
||||
differs from the async pool's capacity accounting (one batch = one slot).
|
||||
"""
|
||||
total = 0
|
||||
try:
|
||||
from tools.async_delegation import active_task_count
|
||||
|
||||
total += max(0, int(active_task_count()))
|
||||
except Exception:
|
||||
logger.debug("background-work async-delegation count failed", exc_info=True)
|
||||
try:
|
||||
from tools.process_registry import process_registry
|
||||
|
||||
total += max(0, int(process_registry.count_running()))
|
||||
except Exception:
|
||||
logger.debug("background-work process-registry count failed", exc_info=True)
|
||||
return total
|
||||
|
||||
|
||||
def _read_background_delegations_count() -> int:
|
||||
"""Count live async delegation UNITS (dispatch/pool slots).
|
||||
|
||||
Complements ``_read_background_work_count`` (which is task-granular): this
|
||||
counts each ``delegate_task`` dispatch as ONE regardless of fan-out width,
|
||||
matching the async pool's capacity accounting (a batch = one slot). Together
|
||||
the two metrics let an operator see both slot pressure
|
||||
(``background_delegations``, alert vs ``max_concurrent_children``) and real
|
||||
concurrent subagent load (``background_work``). Delegations only — it does
|
||||
not include ``terminal(background)`` / kanban work, which are already folded
|
||||
into ``background_work``. Best-effort; 0 if the source can't be imported.
|
||||
"""
|
||||
try:
|
||||
from tools.async_delegation import active_count
|
||||
|
||||
return max(0, int(active_count()))
|
||||
except Exception:
|
||||
logger.debug("background-delegations count failed", exc_info=True)
|
||||
return 0
|
||||
|
||||
|
||||
def _read_runtime_snapshot(config: Dict[str, Any]):
|
||||
gateway_snapshot = _read_gateway_snapshot(config)
|
||||
# Background/subagent work — a distinct metric from active_agents (which
|
||||
# never counts it). Appended to the gateway snapshot so it rides the same
|
||||
# base resource attributes (service.instance.id etc.).
|
||||
try:
|
||||
from agent.monitoring.gateway_health import GatewayMetric
|
||||
|
||||
base = dict(gateway_snapshot.metrics[0].attributes) if gateway_snapshot.metrics else {}
|
||||
gateway_snapshot.metrics.append(
|
||||
GatewayMetric(
|
||||
name="hermes.gateway.background_work",
|
||||
value=_read_background_work_count(),
|
||||
attributes=base,
|
||||
)
|
||||
)
|
||||
gateway_snapshot.metrics.append(
|
||||
GatewayMetric(
|
||||
name="hermes.gateway.background_delegations",
|
||||
value=_read_background_delegations_count(),
|
||||
attributes=base,
|
||||
)
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.warning(
|
||||
"background-work snapshot unavailable; metric not exported (error_type=%s)",
|
||||
type(exc).__name__,
|
||||
)
|
||||
logger.debug("background-work snapshot traceback", exc_info=True)
|
||||
try:
|
||||
cron_snapshot = _read_cron_snapshot()
|
||||
except Exception as exc:
|
||||
# Content-free visibility: cron telemetry silently dropping out is a
|
||||
# release-relevant regression, so surface it at WARNING with only the
|
||||
# exception *type* name (never the message, which could carry paths or
|
||||
# other environment detail). exc_info stays on the DEBUG record.
|
||||
logger.warning(
|
||||
"cron health snapshot unavailable; cron telemetry not exported (error_type=%s)",
|
||||
type(exc).__name__,
|
||||
)
|
||||
logger.debug("cron health snapshot traceback", exc_info=True)
|
||||
return gateway_snapshot
|
||||
gateway_snapshot.metrics.extend(cron_snapshot.metrics)
|
||||
return gateway_snapshot
|
||||
|
||||
|
||||
def _emit_snapshot_events(config: Dict[str, Any]) -> None:
|
||||
gh = _gateway_health_config(config)
|
||||
if not gh.get("diagnostic_events_enabled", True):
|
||||
return
|
||||
try:
|
||||
from agent.monitoring import emitter
|
||||
snapshot = _read_runtime_snapshot(config)
|
||||
for event in snapshot.events:
|
||||
emitter.emit(event)
|
||||
except Exception:
|
||||
logger.debug("gateway health snapshot emit failed", exc_info=True)
|
||||
|
||||
|
||||
def _start_metric_provider(config: Dict[str, Any], sdk: Dict[str, Any]) -> Any:
|
||||
gh = _gateway_health_config(config)
|
||||
if not gh.get("metrics_enabled", True):
|
||||
return None
|
||||
otlp = _otlp_config(config)
|
||||
endpoint = _metric_endpoint(str(otlp.get("endpoint")))
|
||||
headers = _resolve_headers(otlp.get("headers_env"))
|
||||
exporter = sdk["OTLPMetricExporter"](endpoint=endpoint, headers=headers or None)
|
||||
interval_ms = max(5, int(gh.get("export_interval_seconds", 60))) * 1000
|
||||
reader = sdk["PeriodicExportingMetricReader"](exporter, export_interval_millis=interval_ms)
|
||||
resource_attrs = _runtime_resource_attributes(
|
||||
config, telemetry_scope="gateway_health"
|
||||
)
|
||||
provider = sdk["MeterProvider"](
|
||||
metric_readers=[reader],
|
||||
resource=sdk["Resource"].create(resource_attrs),
|
||||
)
|
||||
meter = provider.get_meter("hermes.gateway.health")
|
||||
Observation = sdk["Observation"]
|
||||
|
||||
metric_names = [
|
||||
"hermes.gateway.up",
|
||||
"hermes.gateway.state",
|
||||
"hermes.gateway.active_agents",
|
||||
"hermes.gateway.busy",
|
||||
"hermes.gateway.drainable",
|
||||
"hermes.gateway.restart_requested",
|
||||
"hermes.gateway.background_work",
|
||||
"hermes.gateway.background_delegations",
|
||||
"hermes.platform.up",
|
||||
"hermes.platform.degraded",
|
||||
"hermes.cron.scheduler.heartbeat_age_seconds",
|
||||
"hermes.cron.scheduler.last_success_age_seconds",
|
||||
"hermes.cron.scheduler.catch_up_occurrences",
|
||||
"hermes.cron.jobs.enabled",
|
||||
"hermes.cron.jobs.running",
|
||||
"hermes.cron.jobs.overdue",
|
||||
]
|
||||
|
||||
def callback(name: str):
|
||||
def _cb(_options=None):
|
||||
try:
|
||||
snapshot = _read_runtime_snapshot(config)
|
||||
return [Observation(m.value, m.attributes) for m in snapshot.metrics if m.name == name]
|
||||
except Exception:
|
||||
logger.debug("gateway metric callback failed", exc_info=True)
|
||||
return []
|
||||
return _cb
|
||||
|
||||
for metric_name in metric_names:
|
||||
meter.create_observable_gauge(metric_name, callbacks=[callback(metric_name)])
|
||||
return provider
|
||||
|
||||
|
||||
def _severity_number(sdk: Dict[str, Any], severity: Any) -> Any:
|
||||
SeverityNumber = sdk["SeverityNumber"]
|
||||
sev = str(severity or "warning").lower()
|
||||
if sev in {"critical", "fatal"}:
|
||||
return SeverityNumber.FATAL
|
||||
if sev == "error":
|
||||
return SeverityNumber.ERROR
|
||||
if sev in {"info", "information"}:
|
||||
return SeverityNumber.INFO
|
||||
if sev == "debug":
|
||||
return SeverityNumber.DEBUG
|
||||
return SeverityNumber.WARN
|
||||
|
||||
|
||||
class GatewayDiagnosticLogStreamer:
|
||||
"""Emitter subscriber that sends gateway diagnostic events as OTLP logs."""
|
||||
|
||||
def __init__(self, config: Dict[str, Any], sdk: Dict[str, Any]):
|
||||
otlp = _otlp_config(config)
|
||||
headers = _resolve_headers(otlp.get("headers_env"))
|
||||
endpoint = _logs_endpoint(str(otlp.get("endpoint")))
|
||||
resource_attrs = _runtime_resource_attributes(
|
||||
config, telemetry_scope="gateway_diagnostics"
|
||||
)
|
||||
self._provider = sdk["LoggerProvider"](resource=sdk["Resource"].create(resource_attrs))
|
||||
self._processor = sdk["BatchLogRecordProcessor"](
|
||||
sdk["OTLPLogExporter"](endpoint=endpoint, headers=headers or None)
|
||||
)
|
||||
self._provider.add_log_record_processor(self._processor)
|
||||
self._logger = self._provider.get_logger(_DEFAULT_DIAGNOSTIC_SCOPE)
|
||||
self._LogRecord = sdk["LogRecord"]
|
||||
self._sdk = sdk
|
||||
self.exported = 0
|
||||
|
||||
def __call__(self, batch: list[Dict[str, Any]]) -> None:
|
||||
from agent.monitoring.gateway_health import source_logger_for_export
|
||||
|
||||
for ev in batch:
|
||||
if ev.get("event") != "gateway_diagnostic":
|
||||
continue
|
||||
attrs = _diagnostic_log_attributes(ev)
|
||||
# Preserve the source-controlled Python logger as the OTel
|
||||
# instrumentation scope. This adds precise code attribution without
|
||||
# turning a fluid module layout into a maintained subsystem enum.
|
||||
# Rendered messages stay out because they may contain arbitrary IDs,
|
||||
# names, paths, or configured strings. A future, separately gated
|
||||
# ``diagnostic_detail: redacted_message`` mode may add best-effort
|
||||
# free text when an observability plane defines that privacy policy.
|
||||
source_logger = source_logger_for_export(ev.get("source_logger"))
|
||||
otel_logger = (
|
||||
self._provider.get_logger(source_logger)
|
||||
if source_logger is not None
|
||||
else self._logger
|
||||
)
|
||||
body = "gateway diagnostic"
|
||||
record = self._LogRecord(
|
||||
timestamp=ev.get("ts_ns"),
|
||||
trace_id=self._sdk["INVALID_TRACE_ID"],
|
||||
span_id=self._sdk["INVALID_SPAN_ID"],
|
||||
trace_flags=self._sdk["TraceFlags"].DEFAULT,
|
||||
severity_text=str(ev.get("severity") or "warning").upper(),
|
||||
severity_number=_severity_number(self._sdk, ev.get("severity")),
|
||||
body=_redact_string(body),
|
||||
attributes=attrs,
|
||||
)
|
||||
otel_logger.emit(record)
|
||||
self.exported += 1
|
||||
|
||||
def shutdown(self) -> None:
|
||||
try:
|
||||
from agent.monitoring.emitter import get_emitter
|
||||
get_emitter().unsubscribe(self)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
self._processor.force_flush()
|
||||
self._provider.shutdown()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _start_diagnostic_log_streamer(config: Dict[str, Any], sdk: Dict[str, Any]) -> GatewayDiagnosticLogStreamer:
|
||||
from agent.monitoring.emitter import get_emitter
|
||||
streamer = GatewayDiagnosticLogStreamer(config, sdk)
|
||||
get_emitter().subscribe(streamer)
|
||||
return streamer
|
||||
|
||||
|
||||
def _start_snapshot_thread(config: Dict[str, Any], stop_event: threading.Event) -> threading.Thread:
|
||||
interval = max(5, int(_gateway_health_config(config).get("logs_export_interval_seconds", 5)))
|
||||
|
||||
def _run() -> None:
|
||||
while not stop_event.wait(interval):
|
||||
_emit_snapshot_events(config)
|
||||
|
||||
thread = threading.Thread(target=_run, name="hermes-gateway-health-export", daemon=True)
|
||||
thread.start()
|
||||
return thread
|
||||
|
||||
|
||||
def _attach_log_handler(config: Dict[str, Any]) -> Any:
|
||||
gh = _gateway_health_config(config)
|
||||
if not gh.get("diagnostic_events_enabled", True) or not gh.get("warning_error_events_enabled", True):
|
||||
return None
|
||||
from agent.monitoring.gateway_health import GatewayDiagnosticLogHandler
|
||||
handler = GatewayDiagnosticLogHandler(profile=_profile(), version=_version())
|
||||
root = logging.getLogger()
|
||||
if handler not in root.handlers:
|
||||
root.addHandler(handler)
|
||||
return handler
|
||||
|
||||
|
||||
def _gateway_health_event(ev: Dict[str, Any]) -> bool:
|
||||
return ev.get("event") in {"gateway_health", "cron_execution"}
|
||||
|
||||
|
||||
def start_gateway_health_export(config: Dict[str, Any]) -> GatewayHealthExportRuntime:
|
||||
"""Start P0 gateway health export if configured. Never raises."""
|
||||
if not _enabled(config):
|
||||
return GatewayHealthExportRuntime(enabled=False, reason="disabled")
|
||||
gh = _gateway_health_config(config)
|
||||
runtime = GatewayHealthExportRuntime(enabled=True, reason="enabled")
|
||||
sdk: Optional[Dict[str, Any]] = None
|
||||
|
||||
if gh.get("metrics_enabled", True) or gh.get("diagnostic_events_enabled", True):
|
||||
try:
|
||||
sdk = _require_metrics_sdk(prompt=False)
|
||||
except Exception:
|
||||
logger.warning(
|
||||
"monitoring.gateway_health_export.enabled but OTLP SDK is unavailable; "
|
||||
"install 'hermes-agent[otlp]'",
|
||||
exc_info=True,
|
||||
)
|
||||
return GatewayHealthExportRuntime(enabled=False, reason="otlp_unavailable")
|
||||
|
||||
if gh.get("metrics_enabled", True) and sdk is not None:
|
||||
try:
|
||||
runtime.metric_provider = _start_metric_provider(config, sdk)
|
||||
except Exception:
|
||||
logger.warning("gateway health OTLP metrics failed to start", exc_info=True)
|
||||
runtime.shutdown()
|
||||
return GatewayHealthExportRuntime(enabled=False, reason="metrics_start_failed")
|
||||
|
||||
if gh.get("diagnostic_events_enabled", True) and sdk is not None:
|
||||
try:
|
||||
from agent.monitoring import otlp_exporter
|
||||
runtime.streamer = otlp_exporter.start_streaming(config, event_filter=_gateway_health_event)
|
||||
if runtime.streamer is None:
|
||||
raise RuntimeError("gateway health span streamer did not start")
|
||||
runtime.log_streamer = _start_diagnostic_log_streamer(config, sdk)
|
||||
except Exception:
|
||||
logger.debug("gateway diagnostic OTLP export failed to start", exc_info=True)
|
||||
runtime.shutdown()
|
||||
return GatewayHealthExportRuntime(enabled=False, reason="diagnostics_start_failed")
|
||||
|
||||
try:
|
||||
runtime.log_handler = _attach_log_handler(config)
|
||||
except Exception:
|
||||
logger.debug("gateway diagnostic log handler failed to attach", exc_info=True)
|
||||
if gh.get("diagnostic_events_enabled", True):
|
||||
try:
|
||||
_emit_snapshot_events(config)
|
||||
runtime.stop_event = threading.Event()
|
||||
runtime.thread = _start_snapshot_thread(config, runtime.stop_event)
|
||||
except Exception:
|
||||
logger.debug("gateway health snapshot thread failed to start", exc_info=True)
|
||||
return runtime
|
||||
|
||||
|
||||
__all__ = [
|
||||
"GatewayHealthExportRuntime",
|
||||
"start_gateway_health_export",
|
||||
]
|
||||
@@ -0,0 +1,272 @@
|
||||
"""Export monitoring events to an OpenTelemetry Collector over OTLP/HTTP.
|
||||
|
||||
Maps gateway monitoring events to OTel spans and sends them to the endpoint
|
||||
configured under ``monitoring.export.otlp``. Lets an operator stream Hermes
|
||||
gateway health into their own observability stack (OTEL Collector, DataDog,
|
||||
and similar).
|
||||
|
||||
Notes:
|
||||
* The destination is operator-configured; this module only sends to that
|
||||
endpoint. No default destination ships.
|
||||
* ``opentelemetry-sdk`` + ``opentelemetry-exporter-otlp-proto-http`` are an
|
||||
optional extra (``pip install hermes-agent[otlp]``), imported lazily so the
|
||||
dependency is only required when OTLP export is actually used.
|
||||
* ``headers_env`` maps a header name to an environment variable name; values
|
||||
are read from the environment at export time and never logged or stored.
|
||||
* The continuous subscriber runs in the emitter's dispatcher thread and is
|
||||
fail-isolated, so an export error cannot affect the gateway.
|
||||
|
||||
Only monitoring events (gateway_health / gateway_diagnostic) exist on this
|
||||
plane; the ``event_filter`` seam is kept so future planes sharing the emitter
|
||||
cannot silently ride along on this exporter.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from typing import Any, Callable, Dict, List, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class OTLPUnavailable(RuntimeError):
|
||||
"""Raised when the optional OpenTelemetry SDK isn't installed."""
|
||||
|
||||
|
||||
def _require_sdk(*, auto_install: bool = True, prompt: bool = True):
|
||||
"""Import the OTel SDK, lazily installing it on first use if needed.
|
||||
|
||||
Routes through tools.lazy_deps (feature 'export.otlp') so a missing SDK
|
||||
triggers the standard venv install flow — same as every other optional
|
||||
backend — gated by security.allow_lazy_installs and TTY-prompted. Falls back
|
||||
to OTLPUnavailable (with a manual install hint) when the SDK can't be made
|
||||
importable (lazy installs disabled, install failed, or auto_install=False).
|
||||
|
||||
``auto_install``: attempt the lazy install when missing (default True).
|
||||
``prompt``: ask before installing when interactive (default True); pass
|
||||
False from non-interactive contexts like the continuous streamer.
|
||||
"""
|
||||
if auto_install:
|
||||
try:
|
||||
from tools.lazy_deps import ensure as _lazy_ensure
|
||||
_lazy_ensure("export.otlp", prompt=prompt)
|
||||
except ImportError:
|
||||
pass # lazy_deps unavailable — fall through to the import attempt
|
||||
except Exception:
|
||||
# FeatureUnavailable (lazy installs disabled / declined / failed) —
|
||||
# fall through; the import below raises OTLPUnavailable with the hint.
|
||||
pass
|
||||
try:
|
||||
from opentelemetry.sdk.trace import TracerProvider
|
||||
from opentelemetry.sdk.trace.export import BatchSpanProcessor
|
||||
from opentelemetry.sdk.resources import Resource
|
||||
from opentelemetry.exporter.otlp.proto.http.trace_exporter import (
|
||||
OTLPSpanExporter,
|
||||
)
|
||||
from opentelemetry.trace import SpanKind
|
||||
return {
|
||||
"TracerProvider": TracerProvider,
|
||||
"BatchSpanProcessor": BatchSpanProcessor,
|
||||
"Resource": Resource,
|
||||
"OTLPSpanExporter": OTLPSpanExporter,
|
||||
"SpanKind": SpanKind,
|
||||
}
|
||||
except Exception as e: # ImportError or partial install
|
||||
raise OTLPUnavailable(
|
||||
"OTLP export requires the optional dependency. Install with:\n"
|
||||
" pip install 'hermes-agent[otlp]'\n"
|
||||
f"(import error: {e})"
|
||||
)
|
||||
|
||||
|
||||
def _resolve_headers(headers_env: Optional[Dict[str, str]]) -> Dict[str, str]:
|
||||
"""Resolve {header_name: ENV_VAR_NAME} -> {header_name: value} from env.
|
||||
|
||||
The config stores environment variable names, not secret values; values are
|
||||
read from the environment here. Missing variables are skipped (and noted at
|
||||
debug level without the value).
|
||||
"""
|
||||
resolved: Dict[str, str] = {}
|
||||
for header_name, env_name in (headers_env or {}).items():
|
||||
val = os.environ.get(str(env_name))
|
||||
if val:
|
||||
resolved[str(header_name)] = val
|
||||
else:
|
||||
logger.debug("OTLP header %s: env var %s not set; skipping",
|
||||
header_name, env_name)
|
||||
return resolved
|
||||
|
||||
|
||||
def _otlp_config(config: Dict[str, Any]) -> Dict[str, Any]:
|
||||
mon = (config or {}).get("monitoring") or {}
|
||||
export = mon.get("export") or {}
|
||||
return export.get("otlp") or {}
|
||||
|
||||
|
||||
def build_exporter(config: Dict[str, Any]):
|
||||
"""Construct an OTLP span exporter from config. Raises OTLPUnavailable if no SDK."""
|
||||
sdk = _require_sdk()
|
||||
otlp = _otlp_config(config)
|
||||
endpoint = otlp.get("endpoint")
|
||||
if not endpoint:
|
||||
raise ValueError("monitoring.export.otlp.endpoint is not set")
|
||||
headers = _resolve_headers(otlp.get("headers_env"))
|
||||
return sdk["OTLPSpanExporter"](endpoint=endpoint, headers=headers or None)
|
||||
|
||||
|
||||
def _resource_attributes(config: Dict[str, Any]) -> Dict[str, str]:
|
||||
from agent.monitoring.gateway_health import _safe_instance_id
|
||||
from agent.monitoring.policy import ensure_install_id
|
||||
|
||||
return {
|
||||
"service.name": "hermes-gateway",
|
||||
"service.instance.id": _safe_instance_id(ensure_install_id(config)),
|
||||
"telemetry.scope": "gateway_monitoring",
|
||||
}
|
||||
|
||||
|
||||
def _make_provider(config: Dict[str, Any]):
|
||||
sdk = _require_sdk()
|
||||
resource = sdk["Resource"].create(_resource_attributes(config))
|
||||
provider = sdk["TracerProvider"](resource=resource)
|
||||
processor = sdk["BatchSpanProcessor"](build_exporter(config))
|
||||
provider.add_span_processor(processor)
|
||||
return provider, processor
|
||||
|
||||
|
||||
# ── event -> span attribute mapping ──────────────────────────────────────────
|
||||
def _span_attrs(ev: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Span attributes for a monitoring event (content-free by construction)."""
|
||||
kind = ev.get("event")
|
||||
attrs: Dict[str, Any] = {"hermes.event": kind or "unknown"}
|
||||
keep_by_kind = {
|
||||
"gateway_health": ("name", "gateway_state", "old_state", "new_state",
|
||||
"exit_reason", "restart_requested", "active_agents",
|
||||
"gateway_busy", "gateway_drainable", "platform_count",
|
||||
"fatal_platform_count", "version",
|
||||
"supervision_mode", "pid"),
|
||||
"gateway_diagnostic": ("name", "subsystem", "error_class", "error_code",
|
||||
"platform", "old_state", "new_state",
|
||||
"version", "severity"),
|
||||
"cron_execution": ("status", "job_key", "source", "duration_ms",
|
||||
"delivery_outcome", "error_class"),
|
||||
}
|
||||
for col in keep_by_kind.get(kind, ()): # type: ignore[arg-type]
|
||||
v = ev.get(col)
|
||||
if v is not None:
|
||||
if isinstance(v, str):
|
||||
try:
|
||||
from agent.monitoring.redaction import redact_for_export
|
||||
v = (redact_for_export(v) or "[redacted]")[:500]
|
||||
except Exception:
|
||||
v = "[redaction-unavailable]"
|
||||
attrs[f"hermes.{col}"] = v
|
||||
return attrs
|
||||
|
||||
|
||||
def export_batch(provider, batch: List[Dict[str, Any]]) -> int:
|
||||
"""Map a batch of events to OTel spans. Returns spans created."""
|
||||
tracer = provider.get_tracer("hermes.monitoring")
|
||||
n = 0
|
||||
for ev in batch:
|
||||
try:
|
||||
name = f"hermes.{ev.get('event', 'event')}"
|
||||
span = tracer.start_span(name, attributes=_span_attrs(ev))
|
||||
span.end()
|
||||
n += 1
|
||||
except Exception:
|
||||
logger.debug("OTLP span map failed", exc_info=True)
|
||||
return n
|
||||
|
||||
|
||||
# ── continuous streaming subscriber ─────────────────────────────────────────
|
||||
class OTLPStreamer:
|
||||
"""A live subscriber that pushes each emitter batch to OTLP as it lands.
|
||||
|
||||
Register with ``emitter.subscribe(streamer)``. Fail-isolated by the emitter.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
config: Dict[str, Any],
|
||||
*,
|
||||
event_filter: Optional[Callable[[Dict[str, Any]], bool]] = None,
|
||||
):
|
||||
self._provider, self._processor = _make_provider(config)
|
||||
self._event_filter = event_filter
|
||||
self.exported = 0
|
||||
|
||||
def __call__(self, batch: List[Dict[str, Any]]) -> None:
|
||||
if self._event_filter is not None:
|
||||
batch = [ev for ev in batch if self._event_filter(ev)]
|
||||
if not batch:
|
||||
return
|
||||
self.exported += export_batch(self._provider, batch)
|
||||
|
||||
def shutdown(self) -> None:
|
||||
try:
|
||||
from agent.monitoring.emitter import get_emitter
|
||||
get_emitter().unsubscribe(self)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
self._processor.force_flush()
|
||||
self._provider.shutdown()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def is_available() -> bool:
|
||||
"""True when the OTel SDK is already importable. Does NOT auto-install —
|
||||
this is a pure check (e.g. for status display)."""
|
||||
try:
|
||||
_require_sdk(auto_install=False)
|
||||
return True
|
||||
except OTLPUnavailable:
|
||||
return False
|
||||
|
||||
|
||||
def is_enabled(config: Dict[str, Any]) -> bool:
|
||||
otlp = _otlp_config(config)
|
||||
return bool(otlp.get("enabled") and otlp.get("endpoint"))
|
||||
|
||||
|
||||
def start_streaming(
|
||||
config: Dict[str, Any],
|
||||
*,
|
||||
event_filter: Optional[Callable[[Dict[str, Any]], bool]] = None,
|
||||
) -> Optional[OTLPStreamer]:
|
||||
"""If OTLP is enabled, attach a streamer to the singleton emitter.
|
||||
|
||||
``event_filter`` scopes the exporter to its plane, e.g. gateway-health
|
||||
export, so enabling one plane cannot silently export unrelated events.
|
||||
|
||||
Non-interactive context (startup): attempts a lazy install with prompt=False
|
||||
so a configured-but-missing SDK is installed once (gated by
|
||||
security.allow_lazy_installs), then streams. If it still can't load, logs and
|
||||
no-ops — never blocks or raises into startup.
|
||||
"""
|
||||
if not is_enabled(config):
|
||||
return None
|
||||
try:
|
||||
_require_sdk(prompt=False)
|
||||
except OTLPUnavailable:
|
||||
logger.warning("monitoring.export.otlp.enabled but the OTel SDK could not "
|
||||
"be installed/imported; install 'hermes-agent[otlp]'")
|
||||
return None
|
||||
from agent.monitoring.emitter import get_emitter
|
||||
streamer = OTLPStreamer(config, event_filter=event_filter)
|
||||
get_emitter().subscribe(streamer)
|
||||
return streamer
|
||||
|
||||
|
||||
__all__ = [
|
||||
"OTLPUnavailable",
|
||||
"OTLPStreamer",
|
||||
"build_exporter",
|
||||
"export_batch",
|
||||
"is_available",
|
||||
"is_enabled",
|
||||
"start_streaming",
|
||||
]
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Install identity for gateway monitoring.
|
||||
|
||||
The install id is a stable, resettable pseudonymous identifier attached to
|
||||
exported health signals so an operator can tell instances apart in their
|
||||
collector. It carries no account identity and can be rotated by clearing
|
||||
``monitoring.install_id`` in config.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import uuid
|
||||
from typing import Any, Dict
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def ensure_install_id(config: Dict[str, Any]) -> str:
|
||||
"""Return a stable install id, minting and persisting one when empty.
|
||||
|
||||
The id must survive gateway restarts (it becomes ``service.instance.id``
|
||||
on exported signals), so a freshly minted UUID is written back to
|
||||
config.yaml immediately. The write is fail-open: if persisting fails
|
||||
(read-only home, managed scope), the ephemeral id is still returned and
|
||||
a new one is minted next start.
|
||||
|
||||
Clearing ``monitoring.install_id`` (e.g. ``hermes config set
|
||||
monitoring.install_id ""``) rotates the id on the next gateway start.
|
||||
"""
|
||||
mon = config.get("monitoring") if isinstance(config, dict) else None
|
||||
existing = (mon or {}).get("install_id") if isinstance(mon, dict) else None
|
||||
if isinstance(existing, str) and existing.strip():
|
||||
return existing
|
||||
|
||||
minted = str(uuid.uuid4())
|
||||
try:
|
||||
from hermes_cli.config import load_config, save_config
|
||||
|
||||
fresh = load_config()
|
||||
if isinstance(fresh, dict):
|
||||
slot = fresh.setdefault("monitoring", {})
|
||||
if isinstance(slot, dict) and not str(slot.get("install_id") or "").strip():
|
||||
slot["install_id"] = minted
|
||||
save_config(fresh)
|
||||
except Exception:
|
||||
logger.debug("install_id persist failed; using ephemeral id", exc_info=True)
|
||||
# Keep the in-memory config consistent for this process either way.
|
||||
if isinstance(config, dict):
|
||||
config.setdefault("monitoring", {})
|
||||
if isinstance(config["monitoring"], dict):
|
||||
config["monitoring"]["install_id"] = minted
|
||||
return minted
|
||||
|
||||
|
||||
__all__ = [
|
||||
"ensure_install_id",
|
||||
]
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Redaction applied to monitoring data before egress.
|
||||
|
||||
One unconditional scrub, no modes, no knobs. Every string that leaves the
|
||||
process passes through ``redact_for_export``:
|
||||
|
||||
* Secrets first — wraps ``agent/redact.py::redact_sensitive_text(force=True)``
|
||||
plus bearer/token-shape patterns, and fails CLOSED: if the redactor cannot
|
||||
run, the raw string is never emitted.
|
||||
* PII second — e-mail addresses, phone numbers, and UUID-shaped identifiers
|
||||
are rewritten to ``[email]`` / ``[phone]`` / ``[id]``.
|
||||
|
||||
There is deliberately no setting to weaken this. The monitoring plane is
|
||||
content-free by design: rendered log messages are not exported, and bounded
|
||||
structured strings are still scrubbed as defense-in-depth. This redactor also
|
||||
remains available for a future, explicitly gated redacted-message detail mode.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
# ── secret shapes (belt-and-suspenders on top of agent/redact.py) ───────────
|
||||
_BEARER_RE = re.compile(r"\bBearer\s+[A-Za-z0-9._~+\-/]+=*", re.IGNORECASE)
|
||||
_TOKEN_RE = re.compile(
|
||||
r"\b(xox[baprs]-[A-Za-z0-9-]+|sk-[A-Za-z0-9_-]{8,}|gh[pousr]_[A-Za-z0-9_]{8,})\b"
|
||||
)
|
||||
_SECRET_LITERAL_RE = re.compile(r"\*{3,}")
|
||||
_BEARER_RESIDUE_RE = re.compile(r"\bBearer\s+\[[^\]]+\]", re.IGNORECASE)
|
||||
|
||||
# ── PII shapes ───────────────────────────────────────────────────────────────
|
||||
_EMAIL_RE = re.compile(r"[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}")
|
||||
# E.164-ish and common separators; conservative to avoid nuking code/IDs.
|
||||
_PHONE_RE = re.compile(
|
||||
r"(?<!\w)(?:\+?\d{1,3}[\s.\-]?)?(?:\(\d{2,4}\)[\s.\-]?)?\d{3}[\s.\-]?\d{3,4}(?:[\s.\-]?\d{2,4})?(?!\w)"
|
||||
)
|
||||
# Long opaque hex/uuid-ish user identifiers.
|
||||
_UUID_RE = re.compile(
|
||||
r"\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b"
|
||||
)
|
||||
|
||||
|
||||
def _secret_redact(text: str) -> str:
|
||||
"""Always-on secret redaction. force=True so user config can't disable it."""
|
||||
try:
|
||||
from agent.redact import redact_sensitive_text
|
||||
out = redact_sensitive_text(text, force=True)
|
||||
except Exception:
|
||||
# Fail CLOSED: if the redactor can't run, do not emit the raw string.
|
||||
return "[redaction-unavailable]"
|
||||
out = _BEARER_RE.sub("[redacted]", out)
|
||||
out = _TOKEN_RE.sub("[redacted]", out)
|
||||
out = _SECRET_LITERAL_RE.sub("[redacted]", out)
|
||||
out = _BEARER_RESIDUE_RE.sub("[redacted]", out)
|
||||
return out
|
||||
|
||||
|
||||
def redact_for_export(text: Optional[str]) -> Optional[str]:
|
||||
"""Scrub a string for egress: secrets, then PII. Unconditional."""
|
||||
if text is None:
|
||||
return None
|
||||
out = _secret_redact(str(text))
|
||||
out = _EMAIL_RE.sub("[email]", out)
|
||||
out = _UUID_RE.sub("[id]", out)
|
||||
out = _PHONE_RE.sub("[phone]", out)
|
||||
return out
|
||||
|
||||
|
||||
__all__ = [
|
||||
"redact_for_export",
|
||||
]
|
||||
@@ -15,6 +15,9 @@ and MoonshotAI/kimi-cli#1595:
|
||||
2. When ``anyOf`` is used, ``type`` must be on the ``anyOf`` children, not
|
||||
the parent. Presence of both causes "type should be defined in anyOf
|
||||
items instead of the parent schema".
|
||||
3. Every object schema must carry a ``required`` array, even an empty one.
|
||||
Standard JSON Schema allows omitting it; Moonshot 400s with
|
||||
"required must be an array".
|
||||
|
||||
The ``#/definitions/...`` → ``#/$defs/...`` rewrite for draft-07 refs is
|
||||
handled separately in ``tools/mcp_tool._normalize_mcp_input_schema`` so it
|
||||
@@ -130,9 +133,32 @@ def _repair_schema(node: Any, is_schema: bool = True) -> Any:
|
||||
else:
|
||||
repaired.pop("enum")
|
||||
|
||||
# Rule 4: object schemas must carry a `required` array, even when empty.
|
||||
if repaired.get("type") == "object":
|
||||
repaired = _ensure_required_array(repaired)
|
||||
|
||||
return repaired
|
||||
|
||||
|
||||
def _ensure_required_array(node: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Guarantee an object schema carries a ``required`` array (Moonshot rule).
|
||||
|
||||
Standard JSON Schema lets you omit ``required`` when nothing is required;
|
||||
Moonshot 400s on that ("required must be an array"). Ensure the key is a
|
||||
list. When ``properties`` is known, prune ``required`` entries that don't
|
||||
name a real property — defensive against dangling names, which Moonshot
|
||||
also rejects. Mutates and returns ``node``.
|
||||
"""
|
||||
props = node.get("properties")
|
||||
req = node.get("required")
|
||||
if isinstance(req, list):
|
||||
if isinstance(props, dict):
|
||||
node["required"] = [r for r in req if r in props]
|
||||
else:
|
||||
node["required"] = []
|
||||
return node
|
||||
|
||||
|
||||
def _fill_missing_type(node: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Infer a reasonable ``type`` if this schema node has none."""
|
||||
node_type = node.get("type")
|
||||
@@ -174,17 +200,18 @@ def sanitize_moonshot_tool_parameters(parameters: Any) -> Dict[str, Any]:
|
||||
applied. Input is not mutated.
|
||||
"""
|
||||
if not isinstance(parameters, dict):
|
||||
return {"type": "object", "properties": {}}
|
||||
return {"type": "object", "properties": {}, "required": []}
|
||||
|
||||
repaired = _repair_schema(copy.deepcopy(parameters), is_schema=True)
|
||||
if not isinstance(repaired, dict):
|
||||
return {"type": "object", "properties": {}}
|
||||
return {"type": "object", "properties": {}, "required": []}
|
||||
|
||||
# Top-level must be an object schema
|
||||
if repaired.get("type") != "object":
|
||||
repaired["type"] = "object"
|
||||
if "properties" not in repaired:
|
||||
repaired["properties"] = {}
|
||||
_ensure_required_array(repaired)
|
||||
|
||||
return repaired
|
||||
|
||||
@@ -232,6 +259,10 @@ def is_moonshot_model(model: str | None) -> bool:
|
||||
tail = bare.rsplit("/", 1)[-1]
|
||||
if tail.startswith("kimi-") or tail == "kimi":
|
||||
return True
|
||||
# Kimi Coding Plan serves K3 under the bare slug ``k3`` (plus dated /
|
||||
# suffixed variants like ``k3.1`` or ``k3-turbo``).
|
||||
if tail == "k3" or tail.startswith(("k3.", "k3-")):
|
||||
return True
|
||||
# Vendor-prefixed forms commonly used on aggregators
|
||||
if "moonshot" in bare or "/kimi" in bare or bare.startswith("kimi"):
|
||||
return True
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user