diff --git a/.github/actionlint.yaml b/.github/actionlint.yaml new file mode 100644 index 0000000000..c641c828a0 --- /dev/null +++ b/.github/actionlint.yaml @@ -0,0 +1,9 @@ +# actionlint knows only GitHub-hosted runner labels. An org admin names the +# larger runners. Each one therefore reads as "unknown runner label" and hides +# the real findings, unless this file declares it. +self-hosted-runner: + labels: + - ubuntu-latest-96-core + - ubuntu-latest-32-core + - ubuntu-latest-32-arm-core + - windows-latest-32-core diff --git a/.github/scripts/run-workspace-checks.mjs b/.github/scripts/run-workspace-checks.mjs new file mode 100644 index 0000000000..eabc479877 --- /dev/null +++ b/.github/scripts/run-workspace-checks.mjs @@ -0,0 +1,141 @@ +// Run every workspace check at the same time and report all failures. +// +// The unit of work is a CHECK, and not a workspace. A package that declares +// `check:*` sub-scripts gives one unit for each sub-script. A package with a +// plain `check` gives that. This is the same selection rule the old CI matrix +// used, so the set of commands is unchanged. Only the schedule is different. +// +// This is not `npm run --ws check`, because that command is serial and stops +// at the first workspace that fails. This runs every unit and fails at the +// end with the full list. +// +// The output of each unit goes to a buffer and prints on completion inside a +// group that collapses. Children that write to one stdout together interleave +// their lines, and a failure is then hard to read. +// +// This also runs on a laptop: `node .github/scripts/run-workspace-checks.mjs`. +// `--concurrency N` sets the limit. `--list` prints the units and exits. + +import { execFileSync, spawn } from 'node:child_process' +import { availableParallelism } from 'node:os' + +const IS_CI = Boolean(process.env.GITHUB_ACTIONS) +const NPM = process.platform === 'win32' ? 'npm.cmd' : 'npm' + +/** @returns {{pkg: string, script: string}[]} */ +function discoverUnits() { + const raw = execFileSync(NPM, ['query', '.workspace'], { + encoding: 'utf-8', + shell: process.platform === 'win32', + }) + /** @type {{location: string, scripts?: Record}[]} */ + const pkgs = JSON.parse(raw) + + /** @type {{pkg: string, script: string}[]} */ + const units = [] + for (const pkg of pkgs) { + const scripts = pkg.scripts || {} + const subs = Object.keys(scripts).filter((s) => /^check:.+$/.test(s)) + if (subs.length > 0) { + for (const script of subs) units.push({ pkg: pkg.location, script }) + } else if (scripts.check) { + units.push({ pkg: pkg.location, script: 'check' }) + } + } + return units +} + +/** @param {{pkg: string, script: string}} unit */ +function runUnit(unit) { + return new Promise((resolve) => { + const started = Date.now() + const child = spawn(NPM, ['run', '--prefix', unit.pkg, unit.script], { + // Buffer, and do not inherit. Children that share one stdout + // interleave their lines, and a failure is then hard to read. + stdio: ['ignore', 'pipe', 'pipe'], + shell: process.platform === 'win32', + }) + /** @type {Buffer[]} */ + const chunks = [] + child.stdout.on('data', (c) => chunks.push(c)) + child.stderr.on('data', (c) => chunks.push(c)) + child.on('error', (err) => { + chunks.push(Buffer.from(`failed to spawn: ${err.message}\n`)) + resolve({ unit, code: 1, output: Buffer.concat(chunks).toString('utf-8'), ms: Date.now() - started }) + }) + child.on('close', (code) => { + resolve({ + unit, + code: code ?? 1, + output: Buffer.concat(chunks).toString('utf-8'), + ms: Date.now() - started, + }) + }) + }) +} + +async function main() { + const argv = process.argv.slice(2) + const units = discoverUnits() + + if (units.length === 0) { + console.error( + '::error::No workspace package declares a check script — refusing to report green having run nothing.', + ) + process.exit(1) + } + + if (argv.includes('--list')) { + for (const u of units) console.log(`${u.pkg} :: ${u.script}`) + return + } + + const flagIdx = argv.indexOf('--concurrency') + const concurrency = Math.max( + 1, + flagIdx !== -1 ? Number(argv[flagIdx + 1]) : Math.min(units.length, availableParallelism()), + ) + + console.log(`running ${units.length} checks, up to ${concurrency} at a time:`) + for (const u of units) console.log(` ${u.pkg} :: ${u.script}`) + console.log('') + + const queue = [...units] + /** @type {{unit: {pkg: string, script: string}, code: number, output: string, ms: number}[]} */ + const results = [] + + async function worker() { + for (;;) { + const unit = queue.shift() + if (!unit) return + const res = await runUnit(unit) + results.push(res) + const label = `${res.unit.pkg} :: ${res.unit.script}` + const secs = (res.ms / 1000).toFixed(1) + const status = res.code === 0 ? 'PASS' : 'FAIL' + if (IS_CI) console.log(`::group::${status} ${label} (${secs}s)`) + else console.log(`----- ${status} ${label} (${secs}s) -----`) + process.stdout.write(res.output.endsWith('\n') ? res.output : res.output + '\n') + if (IS_CI) console.log('::endgroup::') + } + } + + await Promise.all(Array.from({ length: Math.min(concurrency, units.length) }, worker)) + + const failed = results.filter((r) => r.code !== 0) + console.log('\n=== summary ===') + for (const r of [...results].sort((a, b) => b.ms - a.ms)) { + console.log( + ` ${r.code === 0 ? 'pass' : 'FAIL'} ${(r.ms / 1000).toFixed(1).padStart(6)}s ${r.unit.pkg} :: ${r.unit.script}`, + ) + } + + if (failed.length > 0) { + for (const r of failed) console.error(`::error::${r.unit.pkg} :: ${r.unit.script} failed`) + console.error(`::error::${failed.length} of ${results.length} checks failed`) + process.exit(1) + } + console.log(`\nall ${results.length} checks passed`) +} + +await main() diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index e2660cc4a6..caf4e2d3cb 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -38,7 +38,7 @@ jobs: detect: name: Detect affected areas runs-on: ubuntu-latest - timeout-minutes: 10 + timeout-minutes: 1 outputs: python: ${{ steps.classify.outputs.python }} python_prod: ${{ steps.classify.outputs.python_prod }} @@ -61,6 +61,8 @@ jobs: id: classify uses: ./.github/actions/detect-changes with: + sparse-checkout: scripts/ci/classify_changes.py + sparse-checkout-cone-mode: false github-token: ${{ github.token }} # ───────────────────────────────────────────────────────────────────── @@ -72,8 +74,6 @@ jobs: needs: detect if: needs.detect.outputs.python == 'true' uses: ./.github/workflows/tests.yml - with: - slice_count: 12 # macOS + Windows lanes. The main `tests` lane above is Linux-only, and # the OS-marked tests it collects are skipped there by design (see the diff --git a/.github/workflows/docker.yml b/.github/workflows/docker.yml index 8c259fe872..f245708486 100644 --- a/.github/workflows/docker.yml +++ b/.github/workflows/docker.yml @@ -76,12 +76,14 @@ jobs: matrix: include: - arch: amd64 - runner: ubuntu-latest + runner: ubuntu-latest-32-core platform: linux/amd64 cache-from: type=gha,scope=docker-amd64 cache-to: type=gha,mode=max,scope=docker-amd64 + # arm64 builds on the native arm64 larger runner. A build of + # linux/arm64 on an x64 host uses emulation. - arch: arm64 - runner: ubuntu-24.04-arm + runner: ubuntu-latest-32-arm-core platform: linux/arm64 cache-from: type=gha,scope=docker-arm64 cache-to: type=gha,mode=max,scope=docker-arm64 @@ -169,7 +171,11 @@ jobs: OPENAI_API_KEY: "" NOUS_API_KEY: "" run: | - scripts/run_tests.sh tests/docker/ --file-timeout 600 + # Each of these tests drives a container, so the docker daemon sets + # the limit and not the processor. This caps the workers. The + # default from run_tests.sh is cpu_count*2, which starts 64 + # containers together on the 32-core amd64 runner. + HERMES_TEST_WORKERS=$(nproc) scripts/run_tests.sh tests/docker/ --file-timeout 600 # --------------------------------------------------------------------------- # Rebuild and push each architecture only after the unprivileged build/test @@ -184,12 +190,13 @@ jobs: matrix: include: - arch: amd64 - runner: ubuntu-latest + runner: ubuntu-latest-32-core platform: linux/amd64 cache-from: type=gha,scope=docker-amd64 cache-to: type=gha,mode=max,scope=docker-amd64 + # Native arm64 for the same reason as the build matrix above. - arch: arm64 - runner: ubuntu-24.04-arm + runner: ubuntu-latest-32-arm-core platform: linux/arm64 cache-from: type=gha,scope=docker-arm64 cache-to: type=gha,mode=max,scope=docker-arm64 diff --git a/.github/workflows/e2e-desktop.yml b/.github/workflows/e2e-desktop.yml index 2749e7c290..8692f05d7a 100644 --- a/.github/workflows/e2e-desktop.yml +++ b/.github/workflows/e2e-desktop.yml @@ -17,7 +17,10 @@ concurrency: jobs: e2e: name: Playwright E2E (Linux) - runs-on: ubuntu-latest + # This job builds the renderer and the electron bundle, then drives a real + # Electron app under xvfb. vite, tsc and the Playwright workers all scale + # with the core count. + runs-on: ubuntu-latest-32-core timeout-minutes: 20 outputs: review_status: ${{ steps.review-status.outputs.review_status }} diff --git a/.github/workflows/js-tests.yml b/.github/workflows/js-tests.yml index 9631cdd7fa..956908293c 100644 --- a/.github/workflows/js-tests.yml +++ b/.github/workflows/js-tests.yml @@ -5,87 +5,17 @@ on: workflow_call: jobs: - workspaces: - name: List npm workspaces - runs-on: ubuntu-latest - timeout-minutes: 20 - outputs: - checks: ${{ steps.set-matrix.outputs.checks }} - steps: - - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 - with: - node-version: 26 - cache: npm - - - name: grab npm 12 - run: | - # No-op once the bundled npm is already 12.x — saves ~5-15s/job and - # keeps the installed major aligned with the npm12 cache-key tag. - npm --version | grep -q '^12\.' || npm i -g npm@12 - - # ``setup-node``'s ``cache: npm`` only caches the ~/.npm tarball cache; - # every job still re-extracts the full workspace node_modules and reruns - # postinstalls (including the Electron binary fetch). Cache the installed - # tree itself, keyed on the lockfile, and skip ``npm ci`` on an exact - # hit. No restore-keys: a partial hit would leave a stale tree, so - # anything but an exact lockfile match reinstalls from scratch. - # The discovery job installs with --ignore-scripts, so its tree differs - # from the check jobs' — hence the distinct ``-noscripts`` key. - - name: Restore node_modules - id: node-modules-cache - uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4 - with: - path: | - node_modules - apps/*/node_modules - ui-tui/node_modules - ui-tui/packages/*/node_modules - tests-js/node_modules - web/node_modules - key: node-modules-noscripts-${{ runner.os }}-node26-npm12-${{ hashFiles('package-lock.json') }} - - - uses: ./.github/actions/retry - if: steps.node-modules-cache.outputs.cache-hit != 'true' - with: - command: npm ci --ignore-scripts - - id: set-matrix - run: | - node -e ' - const { execSync } = require("child_process"); - const pkgs = JSON.parse(execSync("npm query .workspace", { encoding: "utf-8" })); - if (pkgs.length === 0) { - console.error("::error::Workspace discovery produced an empty package list — refusing to emit a zero-length matrix (would skip all JS/TS checks silently)."); - process.exit(1); - } - const checks = []; - for (const pkg of pkgs) { - const scripts = pkg.scripts || {}; - const subs = Object.keys(scripts).filter(s => /^check:.+$/.test(s)); - if (subs.length > 0) { - for (const script of subs) { - checks.push({ package: pkg.location, script }); - } - } else if (scripts.check) { - checks.push({ package: pkg.location, script: "check" }); - } - } - if (checks.length === 0) { - console.error("::error::No check scripts found in any workspace package."); - process.exit(1); - } - process.stdout.write("checks=" + JSON.stringify(checks) + "\n"); - ' >> "$GITHUB_OUTPUT" - check: - name: ${{ matrix.package }} / ${{ matrix.script }} - needs: workspaces - runs-on: ubuntu-latest - timeout-minutes: 20 - strategy: - matrix: - include: ${{ fromJson(needs.workspaces.outputs.checks) }} - fail-fast: false # report all failures, not just the first one + name: JS & TS checks + # One 32-core job replaces a 14-leg matrix. The matrix spread about 612s + # of check payload over 4-core runners. It paid about 371s of repeated + # setup to do it: 14 checkouts, 14 node installs, 14 node_modules + # restores. + # + # One larger runner installs one time. vitest, tsc and eslint each size + # their own worker pool from the core count. + runs-on: ubuntu-latest-32-core + timeout-minutes: 30 steps: - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 @@ -99,12 +29,21 @@ jobs: # keeps the installed major aligned with the npm12 cache-key tag. npm --version | grep -q '^12\.' || npm i -g npm@12 - # Same rationale as the discovery job's cache above, but this ``npm ci`` - # runs WITH install scripts, so the tree includes postinstall artifacts - # (electron's postinstall unpacks its binary into node_modules/electron/ - # dist, which lives inside the cached tree — the ~/.cache/electron - # download cache is deliberately NOT cached: with npm ci skipped on hit - # it would never be read, only inflate the archive). + # The ``cache: npm`` option of ``setup-node`` caches only the ~/.npm + # tarball cache. The job then extracts the full workspace node_modules + # again and runs the postinstalls again, which includes the Electron + # binary fetch. This caches the installed tree itself, keyed on the + # lockfile, and skips ``npm ci`` on an exact hit. There are no + # restore-keys: a partial hit leaves a stale tree, so anything other + # than an exact lockfile match reinstalls from the start. + # + # This install runs WITH scripts, so the tree holds the postinstall + # artifacts. The postinstall of electron unpacks its binary into + # node_modules/electron/dist, which is inside the cached tree. + # + # The ~/.cache/electron download cache stays out of the key on purpose. + # ``npm ci`` is skipped on a hit, so nothing reads that cache. It only + # makes the archive larger. - name: Restore node_modules id: node-modules-cache uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4 @@ -122,4 +61,23 @@ jobs: if: steps.node-modules-cache.outputs.cache-hit != 'true' with: command: npm ci - - run: npm run --prefix ${{ matrix.package }} ${{ matrix.script }} + + # Every check runs at the same time. The step fails only after all of + # them finish. There are two reasons this is not ``npm run --ws check``. + # + # * ``--ws`` is serial and stops at the first workspace that fails. A + # run then reports one failure, where the matrix this replaced + # reported every failure together. + # * The unit of work is a CHECK, and not a workspace. apps/desktop is + # most of the payload, and its own ``check`` is a serial && chain. + # A spread across workspaces alone leaves that chain as the long + # pole. This expands the ``check:*`` sub-scripts of a package, so + # its lint, ui, electron and plugin suites all run together. That + # is the same selection rule the old matrix job used. + # + # Discovery is ``npm query .workspace``. A new package or a new + # ``check:*`` script needs no change here. An empty list is an error and + # not an empty run, because an empty run reports green after it checks + # nothing. + - name: Run all workspace checks + run: node .github/scripts/run-workspace-checks.mjs diff --git a/.github/workflows/nix.yml b/.github/workflows/nix.yml index 23627cbc9f..ccd47d1b09 100644 --- a/.github/workflows/nix.yml +++ b/.github/workflows/nix.yml @@ -51,8 +51,10 @@ jobs: needs: [detect] if: needs.detect.outputs.nix == 'true' # The build compiles the package and its whole dependency closure, so this - # is minutes and not seconds when the cache misses. - runs-on: ubuntu-latest + # takes minutes and not seconds when the cache misses. `nix flake check` + # builds 21 checks, and --max-jobs defaults to the core count. It uses the + # wider runner with no more configuration. + runs-on: ubuntu-latest-32-core timeout-minutes: 60 steps: - name: Checkout code diff --git a/.github/workflows/rust-tests.yml b/.github/workflows/rust-tests.yml index 0d6c179f05..5c91e1a352 100644 --- a/.github/workflows/rust-tests.yml +++ b/.github/workflows/rust-tests.yml @@ -27,7 +27,10 @@ concurrency: jobs: bootstrap-installer: name: cargo test (bootstrap installer) - runs-on: ubuntu-latest + # cargo builds codegen units and test binaries in parallel across the + # cores. This lane also builds the crate from the start when Cargo.toml + # changes. + runs-on: ubuntu-latest-32-core timeout-minutes: 30 defaults: run: diff --git a/.github/workflows/tests-os.yml b/.github/workflows/tests-os.yml index 9ac89c20f4..12719a853c 100644 --- a/.github/workflows/tests-os.yml +++ b/.github/workflows/tests-os.yml @@ -16,9 +16,9 @@ name: OS-specific tests # # Deliberately NOT sliced. The marked set is small (tens of tests, not # thousands), so one plain ``pytest`` process per OS is both faster and far -# less machinery than the LPT-sliced per-file runner the Linux lane needs. +# less machinery than the per-file parallel runner the Linux lane uses. # If either lane grows past its timeout, that is the signal to reach for -# scripts/run_tests.sh --slice here too. +# scripts/run_tests.sh here too. # # Each lane FAILS when it selects zero tests (pytest exit code 5). Without # that guard, a renamed marker or a bad selector would report a green job @@ -48,7 +48,7 @@ jobs: runner: macos-latest marker: macos_only - name: Windows-only tests - runner: windows-latest + runner: windows-latest-32-core marker: windows_only steps: - name: Checkout code diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index a79bf08563..a0dca54ad0 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -2,11 +2,6 @@ name: Tests on: workflow_call: - inputs: - slice_count: - description: Number of parallel test slices - type: number - default: 8 permissions: contents: read @@ -17,42 +12,17 @@ concurrency: cancel-in-progress: true jobs: - generate: - name: "Generate slices" - runs-on: ubuntu-latest - timeout-minutes: 10 - outputs: - matrix: ${{ steps.matrix.outputs.matrix }} - steps: - - name: Checkout code - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - - - name: Restore duration cache - uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5 - with: - path: test_durations.json - key: test-durations - # Saves use test-durations-${run_id}, so the exact key above never - # matches — without this prefix fallback the cache ALWAYS missed, - # LPT slicing ran on no data, and unbalanced slices pushed heavy - # files toward the per-file timeout under load. - restore-keys: | - test-durations- - - - name: Generate test slices - id: matrix - run: | - MATRIX=$(python3 scripts/run_tests_parallel.py --generate-slices ${{ inputs.slice_count }}) - echo "matrix=$MATRIX" >> "$GITHUB_OUTPUT" - test: - name: Run tests slice ${{ matrix.slice.index }}/${{ inputs.slice_count }} - needs: generate - runs-on: ubuntu-latest + name: Run tests + # One 96-core runner for the whole suite. There is no slicing. Slicing + # existed to spread the suite over 4-core runners. It cost a matrix job, a + # duration cache, a per-slice artifact and a merge job to do it. + # + # 96 cores clear the floor that the slowest single test file sets (about + # 82s). A second slice divides work that is already at that floor, and + # adds a second setup. + runs-on: ubuntu-latest-96-core timeout-minutes: 30 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.generate.outputs.matrix) }} steps: - name: Checkout code uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 @@ -117,73 +87,44 @@ jobs: # re-download, keeping the persisted cache small and fast to restore. run: uv cache prune --ci - - name: Run tests (slice ${{ matrix.slice.index }}/${{ inputs.slice_count }}) + - name: Run tests # Per-file isolation via scripts/run_tests.sh: each test file runs # in its own freshly-spawned `python -m pytest ` subprocess # with bounded parallelism. No xdist, no shared workers, no # module-level state leakage between files. # - # File list is pre-computed by the generate job (--generate-slices) - # which runs LPT distribution once and passes the file list to each - # matrix job via --files. Previously each job re-discovered files and - # re-ran LPT independently — redundant N times. + # No --files: the runner discovers the suite itself. The discovered + # set is identical to the list the removed matrix job used to pass in. run: | source .venv/bin/activate - scripts/run_tests.sh --files '${{ matrix.slice.files }}' + scripts/run_tests.sh env: + # This is the maximum number of test FILES that run together. + # run_tests_parallel.py starts one pytest subprocess for each file + # from a single ThreadPoolExecutor, so this value IS the limit. The + # default is cpu_count*2, which is 192 here. + # + # Measured on this runner (96-core EPYC 7763, 377GB). Whole suite, + # two repetitions for each value. See run 32549672063: + # + # workers x cores mean + # 48 0.5x 138s + # 96 1.0x 126s <- fastest + # 144 1.5x 132s + # 192 2.0x 132s + # 240 2.5x 140s + # 288 3.0x 142s + # + # One worker for each core wins. The curve is shallow: 126s to 142s + # across a 6x range. The suite has sufficient concurrency at this + # size. The remaining time is the slowest files plus the setup. + # Workers above the core count only add contention. + HERMES_TEST_WORKERS: 96 # Ensure tests don't accidentally call real APIs OPENROUTER_API_KEY: "" OPENAI_API_KEY: "" NOUS_API_KEY: "" - - name: Upload per-slice durations - # Advisory artifact (feeds slice balancing) — a transient artifact- - # service blip must not fail an otherwise-green test slice. - continue-on-error: true - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 - with: - name: test-durations-slice-${{ matrix.slice.index }} - path: test_durations.json - retention-days: 1 - - # Merge per-slice duration data into a single cache, so future runs - # (including PRs) get balanced slicing. - save-durations: - needs: test - if: needs.test.result == 'success' && github.ref == 'refs/heads/main' - runs-on: ubuntu-latest - timeout-minutes: 10 - steps: - - name: Download all slice durations - # Each slice uploads the same file name (test_durations.json). - # With merge-multiple, the parallel downloads write to one path. - # This causes two problems: a race can write two JSON documents - # into one file, and the last write erases the other slices. - # Without merge-multiple, each artifact gets its own directory. - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 - with: - pattern: test-durations-slice-* - path: durations - - - name: Merge into single durations file - run: | - python3 -c " - import json, glob, os - merged = {} - for f in glob.glob('durations/*/test_durations.json'): - with open(f) as fh: - merged.update(json.load(fh)) - with open('test_durations.json', 'w') as fh: - json.dump(merged, fh, indent=2, sort_keys=True) - print(f'Merged {len(merged)} file durations') - " - - - name: Save merged duration cache - uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5 - with: - path: test_durations.json - key: test-durations-${{ github.run_id }} - e2e: runs-on: ubuntu-latest timeout-minutes: 15 diff --git a/.github/workflows/windows-venv-e2e.yml b/.github/workflows/windows-venv-e2e.yml new file mode 100644 index 0000000000..89b3ebf56e --- /dev/null +++ b/.github/workflows/windows-venv-e2e.yml @@ -0,0 +1,61 @@ +name: Windows venv-holder live E2E + +# ON-DEMAND ONLY (fleet-update #91277, venv-holder consolidation work). +# +# Runs the live venv-holder E2E suite on a real windows-latest runner: +# spawns actual processes with realistic Hermes argv shapes and drives the +# REAL detection/classification/exemption code against the live process +# table — the coverage that cannot exist on the Linux lanes and that the +# maintainer cannot exercise locally before the work reaches main. +# +# Deliberately NOT wired to pull_request/main: it fires only on pushes to +# wine2e/** working branches, so it costs nothing on normal PRs. Delete or +# keep dormant after the venv-holder work lands. + +on: + push: + branches: + - "wine2e/**" + +permissions: + contents: read + +concurrency: + group: windows-venv-e2e-${{ github.ref }} + cancel-in-progress: true + +jobs: + venv-holder-e2e: + name: venv-holder live E2E (windows-latest) + runs-on: windows-latest + timeout-minutes: 25 + steps: + - name: Checkout code + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - name: Install uv + uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0 + with: + version: "0.9.28" + enable-cache: true + cache-dependency-glob: | + pyproject.toml + uv.lock + + - name: Set up Python 3.11 + uses: ./.github/actions/retry + with: + command: uv python install 3.11 + + - name: Install dependencies + uses: ./.github/actions/retry + with: + command: uv sync --locked --python 3.11 --extra dev + + - name: Run venv-holder live E2E + shell: bash + run: | + set -uo pipefail + uv run --no-sync python -m pytest \ + tests/hermes_cli/test_venv_holder_windows_live.py \ + -o addopts= -v -p no:cacheprovider diff --git a/AGENTS.md b/AGENTS.md index 93705624e3..e140522e96 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -920,6 +920,72 @@ plug into `agent/context_engine.py`; image-gen providers into [`hermes-example-plugins`](https://github.com/NousResearch/hermes-example-plugins) companion repo, not in this tree. +### Bot Mode (`apps/desktop/src/plugins/hermes-bots/`) + +The desktop "Bots" experience ships bundled in-tree. Each bot is a Hermes +agent **profile** with a persistent identity. Its design rests on one settled +invariant that has been regressed repeatedly, cost users real conversation +history each time, and is not open for re-litigation in a routine PR: + +**One bot = ONE canonical forever-chat, identified by NAME.** The chat's one +and only identity is **(profile, session titled exactly "Bot Chat")** — the +state DB's UNIQUE(title) index makes that pair an exact registry of at most +one row. The full lifecycle when a bot row is clicked: + +1. **Resolve the registry, every time.** Look up the profile's `Bot Chat` + session by exact title via `session.list {title, include_hidden: true}` + (indexed, window-free; hidden rows resolve because canonical chats are + always hidden; compression lineages resolve to the live tip). Row exists → + open it. That is the entire happy path. +2. **No row → create it,** titled `Bot Chat`, born hidden, kicked off with + the bot's intro. Creation adopts-before-minting: it re-runs the registry + lookup first, so a concurrent or pre-existing row is opened, never forked. + (`set_session_title` silently drops conflicting titles — returns 0 rows — + which is how the 2026-08 infinite fork loop started; adopt-before-mint is + what kills it.) + +**There is NO session-id pin.** The previous design stored a pointer in +`ui_meta['hermes-bots'].chat` and verified it per click; five hardening +waves (#88690, #90732, #90751, the #91791 revert, #92042) each guarded a new +way that pointer dangled or got stolen — rows[0] steals, `last_session` +adoptions, transient clears, drifted-title welds (a pin re-anchored onto a +cron session passed every guard). Name-as-identity removes the failure class: +a name cannot dangle, and a corrupted historical pointer simply never gets +read. Legacy `chat` keys in ui_meta are ignored and dropped from merges. + +Why recency must never win (the #91791 → #92042 lesson): canonical Bot +Chats are **unconditionally hidden** from the Sessions sidebar, so the bot +row is the ONLY door to the forever-chat. A "newest visible session wins" +preference doesn't re-order two equivalent entry points — it walls the +entire relationship off behind a row that previews one session and opens +another, and any stray draft that catches a prompt captures the row. +Side-chats started via "New chat with this agent" are not plumbing-titled, +stay visible in the Sessions sidebar, and are reachable there; they are +never the bot row's target. + +Corollaries for reviewers: + +- There is no per-bot session browser, by explicit design (removed in + #90732). Do not add one back. +- Reject any PR that reintroduces a stored session-id pointer as canonical + identity — including "as a fallback tier" or "for verification". The + registry lookup is the whole contract; pointers are how every prior + incident started. +- Reject any PR that consults recency, visibility, or "where the user left + off" for the bot row's target — reports that motivate such a change are + almost always about side-chats, and the fix belongs in the Sessions + sidebar (hide-sweep false positives), not in the bot row's target. +- The gateway reports the registry row per profile as `canonical_session` + on `profiles.list` (resolved server-side by title); roster preview, + activity signals, and the `/new`→`/compact` guard all read it, so preview + identity and click identity are the same row by construction. + +Regression tests encoding this contract: +`tests/canonical-chat-registry.test.mjs` (includes a tripwire asserting the +open path never reads or writes a stored pointer), +`tests/canonical-chat-creation.test.mjs`, `tests/hide-bot-chats.test.mjs`, +and `tests/tui_gateway/test_profiles_list_canonical_session.py`. + --- ## Skills @@ -1201,6 +1267,79 @@ Full user-facing docs: `website/docs/user-guide/features/kanban.md`. --- +## Update Pipeline (`hermes update`) + +The updater is transactional in shape (fleet-update campaign, #91277 — +Aug 2026). Every stage exists because its absence was a real field +failure; PRs that weaken a stage need to answer for the failure class it +guards: + +``` +plan → snapshot → apply → restart-per-kind → verify → report +``` + +- **Plan** (`hermes_cli/update_inventory.py`, `hermes update --plan`): + read-only inventory — install kind, all profiles, every live gateway + with supervisor + running code version. Deployment kinds are + first-class: `git` updates in place; `docker`/`nix`/`apt` are NOT + in-place-updatable and the updater reports the correct external + command instead of fighting the deployment model. +- **Snapshot** (`hermes_cli/backup.py`): pre-update quick snapshot for + EVERY profile (the code swap + fleet restart touch all of them), each + into its own `state-snapshots/`, identical file set + 1 GiB per-file + cap + keep=1. **Never add a partial/tiered snapshot set** — mixed + coverage creates torn-restore states across schema generations. Quick + snapshots are FILE-LOSS RECOVERY (the per-profile cron-jobs safety + net restores from them), NOT code-rollback insurance; `--backup` full + mode owns rollback. +- **Apply**: git pull, or the Windows ZIP fallback — which fires ONLY + when git itself failed (`_should_zip_fallback_on_update_error`, + argv-classified; a dependency-install failure must never trigger a + tree-clobbering re-download), REFUSES a dirty working tree + (`-uall`, plus a pre-swap TOCTOU re-check), and grafts the live + `apps/desktop/release/` into the staged swap (the GitHub source ZIP + has no built desktop app; without the graft the swap deletes it). +- **Restart-per-kind**: systemd and launchd restarts are FLEET-WIDE + (every `hermes-gateway*` unit / `ai.hermes.gateway*` LaunchAgent), + drain-first (SIGUSR1) with per-unit/per-label failure isolation. + Restarting only the invoking profile's service leaves siblings on + stale `sys.modules` until they crash — the largest dupe-PR cluster in + the repo's history came from that bug. +- **Verify**: gateways stamp their running `code_sha`/`code_version` + into `gateway_state.json` on every runtime-status write + (`gateway/status.py`); after the restart phase the updater compares + each live gateway against the fresh checkout and prints a fleet + version matrix. A provably-stale gateway fails the update (exit 1) — + automation must never treat a mixed-version fleet as healthy. +- **Report**: every run writes a machine-readable receipt to + `~/.hermes/logs/update_receipts/` (`latest.json` pointer; steps, + skips WITH reasons, restart outcome, plan, fleet snapshot). + Finalization is owned by the `cmd_update` command boundary — early + `sys.exit` paths (preflight refusals, fetch failures) still persist + a receipt with the real exit code. A begun-but-unwritten receipt is + a bug: the refused/failed runs are the ones receipts exist for. + +Architecture direction: process-scan-based coordination between the +updater, serve/dashboard, and the gateway is being replaced by a +gateway-owned control socket (#92091). Do not add new scan heuristics +without checking that design; scans are the fallback layer. + +### Gateway lifecycle vs. the Desktop app + +`hermes serve` (control plane, desktop-spawned child) dies with the app +— by design. The messaging gateway (`gateway run`) SURVIVES the app: the +serve backend's `/api/gateway/*` endpoints spawn it detached +(`_spawn_hermes_action` — `start_new_session` / `DETACHED_PROCESS`), so +`before-quit`'s backend SIGTERM never reaches it. Bots keep running +when the user closes the app. The known breach of this contract is the +Windows shim-unlock teardown (`taskkill /T /F` on venv-shim holders, +#85265) — it exists to let updates proceed, and its replacement is +#92091's `pause-for-update`. Do not "fix" gateway-dies-with-app reports +by re-parenting the gateway under the backend, and do not "fix" update +locks by widening the tree-kill. + +--- + ## Important Policies ### Prompt Caching Must Not Break @@ -1314,6 +1453,27 @@ automatically scope to the active profile. ## Known Pitfalls +### DO NOT infer process identity from argv substrings +The bug class behind ~10 fleet-update issues (#90778, #87594, #78089, +#76129, #91964, ...): classifying a process by `"serve" in cmdline` or +similar. `kanban --preserve-cache` contains "serve"; a flag VALUE can +equal a subcommand (`-m dashboard serve`); truncated cmdlines hide the +real subcommand. Rules: +- Use the canonical matchers: `gateway.status.looks_like_gateway_command_line` + (gateway run), `hermes_cli.update_cmd._hermes_holder_subcommand` + (top-level subcommand of any Hermes argv). Never hand-roll token scans. +- Flag sets must be DERIVED from the parser + (`_holder_value_flags()` introspects `build_top_level_parser()`), never + hand-written lists — they drift. +- Never blanket-exclude ancestors from process scans: when `/update` runs + as the gateway's child, a gateway ancestor must stay visible to the + pause machinery (#87594). Exclude interactive ancestry, carve out + gateway-shaped ancestors. +- Match on FULL cmdlines; truncate only at display time (#78089). +- Before adding any new scan heuristic, read #92091 — the gateway control + socket replaces scans as the primary coordination mechanism; scans are + the fallback layer for old/crashed processes. + ### DO NOT hardcode `~/.hermes` paths Use `get_hermes_home()` from `hermes_constants` for code paths. Use `display_hermes_home()` for user-facing print/log messages. Hardcoding `~/.hermes` breaks profiles — each profile @@ -1491,6 +1651,22 @@ in order to pass, it belongs on that OS.** When one test body walks several platforms in sequence, split it. Keep the host-native arm on the Linux lane and move the other arm into its own marked test. +**Live Windows process-topology E2E: the `wine2e` lane.** For claims about +real Windows process behavior that mocks cannot reproduce (venv-holder +scans, process-tree parentage, launcher/worker chains, detach semantics), +there is an on-demand workflow `windows-venv-e2e.yml` that runs +`tests/hermes_cli/test_venv_holder_windows_live.py` on a real +`windows-latest` runner — spawning actual processes and driving the real +detection code, no mocked psutil. It fires ONLY on pushes to `wine2e/**` +branches (inert on PRs and main; costs nothing on normal work). The proven +workflow: write probes that pin CORRECT behavior, push to a `wine2e/` +branch to reproduce the bugs live on unfixed code, build the fix, iterate +until the lane is green, then open the PR — the live receipt on the exact +head is the Windows proof reviewers ask for. Extend the live suite when +touching that subsystem; assert against the gateway ANCESTOR found by +argv, not the direct parent (the venv shim makes every spawn a +launcher/worker chain). + **Use the marker, never a bare `skipif`.** `scripts/ci/list_os_marked_tests.py` decides which files the macOS/Windows lanes import by grepping for the marker *name*, and the lane then filters with `-m `. A test gated with diff --git a/agent/agent_init.py b/agent/agent_init.py index db92f18487..9b77d5227f 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -1259,6 +1259,7 @@ def init_agent( _gr_label = " + Guardrails" if agent._bedrock_guardrail_config else "" print(f"🤖 AI Agent initialized with model: {agent.model} (AWS Bedrock, {agent._bedrock_region}{_gr_label})") else: + client_kwargs = {} if api_key and base_url: # Explicit credentials from CLI/gateway — construct directly. # The runtime provider resolver already handled auth for us. @@ -1430,6 +1431,19 @@ def init_agent( "select a provider, or run `hermes setup` for first-time " "configuration." ) + # Bedrock GPT-5.5/5.6 use Bedrock Mantle's OpenAI Responses endpoint. + # Runtime resolution uses api_key="aws-sdk" as the IAM-auth sentinel; + # attach an httpx client that SigV4-signs every OpenAI SDK request. + # No-op for non-Mantle base URLs. + try: + from agent.bedrock_adapter import configure_bedrock_openai_client_kwargs + configure_bedrock_openai_client_kwargs( + client_kwargs, + timeout=_provider_timeout, + ) + except Exception: + if agent.provider == "bedrock" and "bedrock-mantle." in str(client_kwargs.get("base_url", "")): + raise agent._client_kwargs = client_kwargs # stored for rebuilding after interrupt diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index a069629e44..b5aeae274f 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -671,7 +671,9 @@ def _is_codex_gpt54_or_gpt55(model: Optional[str], provider: Optional[str] = Non via prefix so the override tracks every 272K-capped family (5.4, 5.5, 5.6 sol/terra/luna incl. their ``-pro`` modes) without re-listing every variant. (Name kept for backward compatibility with the - ``compression.codex_gpt55_autoraise`` config key.) + ``compression.codex_gpt55_autoraise`` config key.) The exact + ``gpt-daybreak-blue-latest`` Codex slug is also a verified Sol-family + alias and receives the same autoraise. """ prov = (provider or "").strip().lower() if prov != "openai-codex": @@ -687,6 +689,7 @@ def _is_codex_gpt54_or_gpt55(model: Optional[str], provider: Optional[str] = Non or bare == "gpt-5.6" or bare.startswith("gpt-5.6-") or bare.startswith("gpt-5.6.") + or bare == "gpt-daybreak-blue-latest" ) @@ -741,7 +744,8 @@ def _compression_threshold_for_model( Per-model/route overrides: - Arcee Trinity Large Thinking → 0.75 (preserve reasoning context). - - gpt-5.4 / gpt-5.5 / gpt-5.6 on the Codex OAuth route → 0.85, because + - gpt-5.4 / gpt-5.5 / gpt-5.6 and the exact Daybreak Sol alias on the + Codex OAuth route → 0.85, because Codex caps all three families at 272K and the default 50% trigger would compact at ~136K. Gated by ``allow_codex_gpt55_autoraise`` (historical config-key name kept for backward compatibility) so the @@ -2826,8 +2830,15 @@ _paid_lane_warned: set = set() def _is_free_model(model: Optional[str]) -> bool: - """True when ``model`` is an OpenRouter free SKU (``:free`` suffix).""" - return bool(model) and str(model).strip().endswith(":free") + """True when ``model`` is a free SKU (``:free`` suffix or ``stealth/`` prefix). + + Naming-convention trust: a paid model shipped under ``stealth/`` would + silently bypass both the free_only gate and the paid-lane warning. + """ + if not model: + return False + normalized = str(model).strip() + return normalized.endswith(":free") or normalized.startswith("stealth/") def _aux_openrouter_settings() -> Tuple[bool, str]: @@ -2849,7 +2860,8 @@ def _aux_openrouter_settings() -> Tuple[bool, str]: def _warn_paid_lane_once(model: str) -> None: - """Log a WARNING the first time a non-:free OpenRouter model is engaged.""" + """Log a WARNING the first time a non-free (neither ``:free`` nor + ``stealth/``) OpenRouter model is engaged.""" if model in _paid_lane_warned: return _paid_lane_warned.add(model) @@ -6979,8 +6991,12 @@ def resolve_provider_client( default_model = "google/gemini-3-flash-preview" final_model = _normalize_resolved_model(model or default_model, provider) try: - from openai import OpenAI - client = OpenAI(api_key=token, base_url=base_url) + # Alias the import: a bare `from openai import OpenAI` here would + # make `OpenAI` function-local and shadow the module-level lazy + # proxy for every other branch of this function (breaking both the + # Bedrock Mantle branch below and patch("agent.auxiliary_client.OpenAI")). + from openai import OpenAI as _VertexOpenAI + client = _VertexOpenAI(api_key=token, base_url=base_url) except Exception as exc: logger.warning("resolve_provider_client: cannot create Vertex " "client: %s", exc) @@ -6991,17 +7007,23 @@ def resolve_provider_client( elif pconfig.auth_type == "aws_sdk": # AWS SDK providers (Bedrock) — Claude models use the Anthropic Bedrock - # SDK (prompt caching, thinking); non-Claude models use Converse API. + # SDK (prompt caching, thinking); OpenAI models (GPT-5.5/5.6) use + # Bedrock Mantle's OpenAI Responses endpoint; all other models use the + # Converse API. try: from agent.bedrock_adapter import ( has_aws_credentials, is_anthropic_bedrock_model, - resolve_bedrock_region, + resolve_bedrock_runtime_region, + is_openai_bedrock_model, + bedrock_openai_base_url, + resolve_bedrock_bearer_token, + configure_bedrock_openai_client_kwargs, ) from agent.anthropic_adapter import build_anthropic_bedrock_client except ImportError: logger.warning("resolve_provider_client: bedrock requested but " - "boto3 or anthropic SDK not installed") + "boto3, httpx/openai, or anthropic SDK not installed") return None, None if not has_aws_credentials(): @@ -7009,9 +7031,34 @@ def resolve_provider_client( "no AWS credentials found") return None, None - region = resolve_bedrock_region() + # Region must match the main runtime's resolution (bedrock.region in + # config.yaml first, then env/profile) — see review on #53880/#65076: + # a bare resolve_bedrock_region() here let auxiliary calls (compression, + # memory, vision) leave the primary runtime's configured region. + region = resolve_bedrock_runtime_region() default_model = "anthropic.claude-haiku-4-5-20251001-v1:0" - final_model = _normalize_resolved_model(model or default_model, provider) + final_model = _normalize_resolved_model(model or default_model, provider) or default_model + + if is_openai_bedrock_model(final_model): + # NOTE: no local `from openai import OpenAI` here — the module-level + # lazy proxy (see top of file) must stay visible so tests can + # patch("agent.auxiliary_client.OpenAI", ...). + bearer = resolve_bedrock_bearer_token() + mantle_base_url = bedrock_openai_base_url(region) + client_kwargs: Dict[str, Any] = { + "api_key": bearer or "aws-sdk", + "base_url": mantle_base_url, + } + configure_bedrock_openai_client_kwargs(client_kwargs) + client = OpenAI(**client_kwargs) + logger.debug("resolve_provider_client: bedrock-openai (%s, %s)", final_model, region) + if raw_codex: + return (_to_async_client(client, final_model, is_vision=is_vision) if async_mode + else (client, final_model)) + wrapped = CodexAuxiliaryClient(client, final_model) + return (_to_async_client(wrapped, final_model, is_vision=is_vision) if async_mode + else (wrapped, final_model)) + base_url = f"https://bedrock-runtime.{region}.amazonaws.com" if is_anthropic_bedrock_model(final_model): diff --git a/agent/bedrock_adapter.py b/agent/bedrock_adapter.py index 8d63323fd2..5e2c082080 100644 --- a/agent/bedrock_adapter.py +++ b/agent/bedrock_adapter.py @@ -33,6 +33,9 @@ import os import re from types import SimpleNamespace from typing import Any, Dict, List, Optional, Tuple +from urllib.parse import urlparse + +import httpx logger = logging.getLogger(__name__) @@ -57,6 +60,25 @@ except Exception: _bedrock_runtime_client_cache: Dict[str, Any] = {} _bedrock_control_client_cache: Dict[str, Any] = {} +# Bedrock-hosted OpenAI GPT-5.5 is not exposed through the native Converse +# runtime. AWS serves it from the Bedrock Mantle OpenAI-compatible Responses +# endpoint instead (https://bedrock-mantle..api.aws/openai/v1). +# Keep the allowlist intentionally narrow so OpenAI GPT-OSS models that are +# Converse-capable continue to use the native Bedrock path. +BEDROCK_OPENAI_RESPONSES_MODEL_IDS: Tuple[str, ...] = ( + "openai.gpt-5.5", + # GPT-5.6 family (GA on Bedrock 2026-07-13): Sol (frontier), Terra + # (balanced), Luna (fast/affordable). All are Mantle-only — the model + # cards list bedrock-runtime/Converse as unsupported. + # https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html + "openai.gpt-5.6-sol", + "openai.gpt-5.6-terra", + "openai.gpt-5.6-luna", +) +_BEDROCK_OPENAI_HOST_RE = re.compile( + r"^bedrock-mantle\.([a-z0-9-]+)\.api\.aws$", re.IGNORECASE +) + _MIN_BOTO3_VERSION = (1, 34, 59) @@ -133,6 +155,143 @@ def invalidate_runtime_client(region: str) -> bool: return existed +# --------------------------------------------------------------------------- +# Bedrock Mantle / OpenAI Responses support +# --------------------------------------------------------------------------- + + +def is_openai_bedrock_model(model_id: str) -> bool: + """Return True for Bedrock-hosted OpenAI models that require Mantle. + + Bedrock's GPT-OSS models are Converse-capable and intentionally do not + match this helper. The allowlist tracks models served by the OpenAI + Responses-compatible ``bedrock-mantle`` route. + """ + normalized = str(model_id or "").strip().lower() + return normalized in {m.lower() for m in BEDROCK_OPENAI_RESPONSES_MODEL_IDS} + + +def merge_bedrock_openai_model_ids(model_ids: List[str]) -> List[str]: + """Append Bedrock OpenAI Responses models to a discovered Bedrock list. + + The Bedrock control plane's ListFoundationModels/ListInferenceProfiles + discovery covers Converse models but does not enumerate Mantle-only + OpenAI Responses models. The picker needs both surfaces under AWS Bedrock. + """ + merged = list(model_ids or []) + seen = {str(m).lower() for m in merged} + for model_id in BEDROCK_OPENAI_RESPONSES_MODEL_IDS: + if model_id.lower() not in seen: + merged.append(model_id) + seen.add(model_id.lower()) + return merged + + +def bedrock_openai_base_url(region: str) -> str: + """Return Bedrock Mantle's OpenAI-compatible base URL for *region*.""" + resolved = (region or "").strip() or resolve_bedrock_runtime_region() + return f"https://bedrock-mantle.{resolved}.api.aws/openai/v1" + + +def bedrock_openai_region_from_base_url(base_url: str) -> Optional[str]: + """Extract the AWS region from a Bedrock Mantle OpenAI base URL.""" + host = urlparse(str(base_url or "")).hostname or "" + match = _BEDROCK_OPENAI_HOST_RE.match(host) + return match.group(1) if match else None + + +def is_bedrock_openai_base_url(base_url: str) -> bool: + """Return True for Bedrock Mantle OpenAI-compatible endpoints.""" + parsed = urlparse(str(base_url or "")) + host = parsed.hostname or "" + if not _BEDROCK_OPENAI_HOST_RE.match(host): + return False + # The OpenAI GPT-5.5 Bedrock route lives under /openai/v1. Accept a bare + # host too so callers can normalize before appending the path. + path = (parsed.path or "").rstrip("/").lower() + return path in {"", "/openai", "/openai/v1"} + + +def resolve_bedrock_bearer_token(env: Optional[Dict[str, str]] = None) -> str: + """Return AWS_BEARER_TOKEN_BEDROCK when Bedrock API-key auth is configured.""" + env = env if env is not None else os.environ + return (env.get("AWS_BEARER_TOKEN_BEDROCK", "") or "").strip() + + +class BedrockOpenAISigV4Auth(httpx.Auth): + """httpx auth hook that SigV4-signs Bedrock Mantle OpenAI requests.""" + + requires_request_body = True + + def __init__(self, region: str, service: str = "bedrock"): + self.region = (region or "").strip() or resolve_bedrock_runtime_region() + self.service = service + + def auth_flow(self, request): # pragma: no cover - exercised by live call + import botocore.session + from botocore.auth import SigV4Auth + from botocore.awsrequest import AWSRequest + + credentials = botocore.session.get_session().get_credentials() + if credentials is None: + raise RuntimeError( + "No AWS credentials available for Bedrock OpenAI Responses. " + "Configure AWS_ACCESS_KEY_ID/AWS_SECRET_ACCESS_KEY, AWS_PROFILE, " + "SSO, or an instance/task role." + ) + frozen = credentials.get_frozen_credentials() + # Drop the OpenAI SDK's placeholder bearer header before signing; SigV4 + # must own Authorization. Keep all other SDK headers so AWS receives + # content-type, accept, request IDs, etc. + headers = { + str(k): str(v) + for k, v in request.headers.items() + if str(k).lower() not in {"authorization", "x-amz-date", "x-amz-security-token"} + } + aws_request = AWSRequest( + method=request.method, + url=str(request.url), + data=request.content or b"", + headers=headers, + ) + SigV4Auth(frozen, self.service, self.region).add_auth(aws_request) + request.headers.update(dict(aws_request.headers.items())) + yield request + + +def build_bedrock_openai_http_client(region: str, *, timeout: Optional[float] = None): + """Build an httpx client that SigV4-signs Bedrock OpenAI requests.""" + import httpx + + kwargs: Dict[str, Any] = {"auth": BedrockOpenAISigV4Auth(region)} + if isinstance(timeout, (int, float)) and not isinstance(timeout, bool) and timeout > 0: + kwargs["timeout"] = timeout + return httpx.Client(**kwargs) + + +def configure_bedrock_openai_client_kwargs( + client_kwargs: Dict[str, Any], + *, + timeout: Optional[float] = None, +) -> Dict[str, Any]: + """Install SigV4 auth on OpenAI SDK kwargs for Bedrock Mantle. + + ``AWS_BEARER_TOKEN_BEDROCK``/explicit Bedrock API keys continue to use the + SDK's normal bearer auth. The special ``aws-sdk`` placeholder means IAM + credential-chain auth, so we attach a per-request SigV4 httpx client. + """ + base_url = str(client_kwargs.get("base_url") or "") + if not is_bedrock_openai_base_url(base_url): + return client_kwargs + api_key = client_kwargs.get("api_key") + if isinstance(api_key, str) and api_key.strip() and api_key not in {"aws-sdk", "no-key-required"}: + return client_kwargs + region = bedrock_openai_region_from_base_url(base_url) or resolve_bedrock_runtime_region() + client_kwargs["api_key"] = "aws-sdk" + client_kwargs["http_client"] = build_bedrock_openai_http_client(region, timeout=timeout) + return client_kwargs + + # --------------------------------------------------------------------------- # Stale-connection detection # --------------------------------------------------------------------------- @@ -384,6 +543,36 @@ def resolve_bedrock_region(env: Optional[Dict[str, str]] = None) -> str: return "us-east-1" +def resolve_bedrock_runtime_region(config: Optional[Dict[str, Any]] = None) -> str: + """Resolve the Bedrock region with the same priority as the main runtime. + + Priority (matches the runtime provider resolver in + ``hermes_cli/runtime_provider.py``): + 1. ``bedrock.region`` in config.yaml + 2. ``resolve_bedrock_region()`` (AWS_REGION / AWS_DEFAULT_REGION / + botocore profile / us-east-1) + + Callers that already hold a loaded config dict should pass it to avoid a + disk read; when *config* is None the config is loaded read-only. Every + non-runtime call site that constructs a Bedrock endpoint (auxiliary + client resolution, model discovery for the picker) must use this helper — + using bare ``resolve_bedrock_region()`` there lets auxiliary calls leave + the primary runtime's configured region when ``bedrock.region`` and the + ambient AWS env/profile disagree. + """ + if config is None: + try: + from hermes_cli.config import load_config_readonly + config = load_config_readonly() + except Exception: + config = {} + bedrock_cfg = (config or {}).get("bedrock") or {} + cfg_region = str(bedrock_cfg.get("region") or "").strip() + if cfg_region: + return cfg_region + return resolve_bedrock_region() + + def bedrock_model_ids_or_none() -> Optional[List[str]]: """Live-discover Bedrock model IDs for the active region. @@ -396,9 +585,9 @@ def bedrock_model_ids_or_none() -> Optional[List[str]]: ``list_authenticated_providers`` section 2, and section 3. """ try: - discovered = discover_bedrock_models(resolve_bedrock_region()) + discovered = discover_bedrock_models(resolve_bedrock_runtime_region()) if discovered: - return [m["id"] for m in discovered] + return merge_bedrock_openai_model_ids([m["id"] for m in discovered]) except Exception: pass return None @@ -1450,6 +1639,12 @@ BEDROCK_CONTEXT_LENGTHS: Dict[str, int] = { "mistral.mistral-large": 128_000, # DeepSeek "deepseek.v3": 128_000, + # OpenAI on Bedrock (Mantle/Responses route) + # https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html + "openai.gpt-5.5": 272_000, + "openai.gpt-5.6-sol": 272_000, + "openai.gpt-5.6-terra": 272_000, + "openai.gpt-5.6-luna": 272_000, } # Default for unknown Bedrock models diff --git a/agent/compaction_display.py b/agent/compaction_display.py new file mode 100644 index 0000000000..ea1e61ed77 --- /dev/null +++ b/agent/compaction_display.py @@ -0,0 +1,47 @@ +"""Client-facing projection helpers for model-only compaction carriers.""" + +from __future__ import annotations + +from typing import Any, Dict, Optional + +from agent.context_compressor import ( + ContextCompressor, + is_compaction_summary_message, +) + + +_COMPACTION_INTERNAL_FIELDS = ( + "tool_calls", + "finish_reason", + "reasoning", + "reasoning_content", + "reasoning_details", + "codex_reasoning_items", + "codex_message_items", +) + + +def project_compaction_message_for_display( + message: Dict[str, Any], +) -> Optional[Dict[str, Any]]: + """Return authentic transcript content, or ``None`` for a pure handoff. + + Model-facing recovery history retains the complete carrier. Display + projections instead remove the handoff, inherited tool state, and internal + reasoning while preserving any real prior-tail content or live user ask + embedded in the carrier. + """ + if not isinstance(message, dict): + return None + if not is_compaction_summary_message(message): + return message.copy() + + projected = ContextCompressor._strip_context_summary_handoff_message(message) + if projected is None: + return None + + projected = projected.copy() + for key in _COMPACTION_INTERNAL_FIELDS: + projected.pop(key, None) + projected.pop("display_kind", None) + return projected diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 47bd2fd4ad..67810a7aa0 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -8186,8 +8186,24 @@ def reference_handoff_would_drive_next_model_call( for index, message in enumerate(messages): if not is_compaction_summary_message(message): continue - if _handoff_carries_live_user_content(message): - # Embedded live ask — this row is not a sole-handoff driver. + merged_completed_assistant = ( + isinstance(message, dict) + and message.get("role") == "assistant" + and ContextCompressor.classify_summary_content( + message.get("content") + ) + == "merged" + and message.get("finish_reason") == "stop" + and not message.get("tool_calls") + ) + if ( + _handoff_carries_live_user_content(message) + and not merged_completed_assistant + ): + # Embedded live ask — this row is not a sole-handoff driver. A + # completed merged assistant carrier preserves the assistant's own + # prose, not a fresh user request. A carrier with pending tool_calls + # remains live regardless of an earlier completed assistant turn. continue last_driving_handoff = index diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index a2b2643bf3..90f3b5e879 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -3460,6 +3460,24 @@ def compress_context( "could not record rejected-compaction strike", exc_info=True, ) + # Restore ONLY the prune runway (same rationale as the + # rotation-failure rollback below): compress()'s successful + # tail already zeroed _proactive_prune_rearm_tokens in + # memory, but this refusal keeps the ORIGINAL transcript — + # whose cached prefix is intact. Leaving the runway at 0 + # disarms the #79640 throttle, so the very next iteration's + # proactive prune rewrites history and breaks the prompt + # cache without the required regrowth interval (#91830). + # The durable copy was never cleared (that clear only rides + # the archive_and_compact / child-row commit that never + # ran), so restoring the snapshot re-aligns memory with + # disk. + if "_proactive_prune_rearm_tokens" in _compressor_attempt_snapshot: + agent.context_compressor._proactive_prune_rearm_tokens = ( + _compressor_attempt_snapshot[ + "_proactive_prune_rearm_tokens" + ] + ) _release_lock() return messages, _existing_sp diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 3d7f07ef39..2d49501518 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -755,6 +755,10 @@ def _billing_failure_result( "failed": True, "error": summary, "failure_reason": classified.reason.value, + # The classifier's own retry verdict — carried so UI surfaces + # (agent/error_surface.py) show Retry only when a re-run can differ, + # instead of re-deriving retryability from a second taxonomy. + "failure_retryable": bool(classified.retryable), # The billing verdict may rest on an ambiguous body (#82154) — carry # that through the structured result, not just the prose. "billing_unverified": unverified, @@ -6439,6 +6443,9 @@ def run_conversation( # different exit code. ``rate_limit`` / ``billing`` here # mean "quota wall, not a task error". "failure_reason": classified.reason.value, + # The classifier's own retry verdict — UI surfaces use + # this instead of re-deriving from the reason string. + "failure_retryable": bool(classified.retryable), # True when the billing verdict rests on an ambiguous # body (#82154) — may be a content-filter rejection. "billing_unverified": _billing_unverified, diff --git a/agent/credits_tracker.py b/agent/credits_tracker.py index b47c3f274e..39c74ea58b 100644 --- a/agent/credits_tracker.py +++ b/agent/credits_tracker.py @@ -226,12 +226,15 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool: 1. The ``:free`` suffix — the canonical Nous free SKU marker (e.g. ``nvidia/nemotron-3-ultra:free``). Free by construction on the API side (spend is forced to 0 for ``:free`` ids). - 2. A peek into the in-process pricing cache in ``hermes_cli.models`` + 2. The ``stealth/`` prefix — Nous stealth-preview SKUs (e.g. + ``stealth/ox-alpha``) are free-tier but carry no ``:free`` suffix. Spend + is forced to zero server-side, so these are also free by construction. + 3. A peek into the in-process pricing cache in ``hermes_cli.models`` (populated when the model picker fetched ``/v1/models`` pricing for *base_url*). PEEK ONLY — a cache miss never triggers a fetch. This is CLI/TUI-session best-effort: gateway sessions never run the picker's pricing fetch, so suppression there rests entirely on the ``:free`` - suffix (which all Nous free SKUs carry). + suffix and ``stealth/`` prefix. Fail-open to False (the depleted notice still shows) on any error: wrongly showing the warning is recoverable noise; wrongly hiding it on a paid model @@ -241,6 +244,11 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool: return False if model.endswith(":free"): return True + # Stealth-preview SKUs are free-tier but carry no ``:free`` suffix (see + # docstring point 2). Naming-convention trust: if a PAID model ever shipped + # under ``stealth/`` this would wrongly suppress the banner on it. + if model.startswith("stealth/"): + return True if not base_url: return False try: diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 9f112c22a3..514eda3834 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -459,6 +459,59 @@ _REQUEST_VALIDATION_PATTERNS = [ "unsupported_parameter", ] +# Request parameters that Hermes sends on SOME routes only, paired with the +# providers/hosts where sending them is deliberate. +# +# When a host that is NOT in the allowed set rejects one of these fields, the +# client never put it in the body — the provider's own gateway injected it — +# so the 400 is a server-side flake rather than a deterministic request-shape +# error. See ``_is_server_injected_param_rejection`` and the branch in +# ``_classify_400``. +# +# ``prompt_cache_retention`` is only sent for api.meta.ai and bedrock-mantle +# hosts (agent/transports/codex.py::_default_prompt_cache_retention_for_request). +# The Codex OAuth backend rejects it spontaneously on requests that provably +# never carried it. +_SERVER_INJECTED_PARAM_SENDERS: Dict[str, tuple] = { + "prompt_cache_retention": ("meta", "muse", "msl", "model-api", "bedrock", "mantle"), +} + + +def _is_server_injected_param_rejection(error_msg: str, provider: str) -> bool: + """True when a 400 blames a parameter this route never sends. + + ``error_msg`` is the lowercased, concatenated message text; ``provider`` is + the lowercased provider slug. A match means the rejection cannot be + attributed to our own request shape, so the error is transient and retrying + the identical request is the correct recovery. + + Deliberately conservative: it fires only for known one-route-only + parameters AND only when the current provider is not one of the routes that + actually sends them, so a genuine client-side bad parameter (``max_tokens`` + on a GPT-5 model) still fails fast as a ``format_error``. + """ + if not error_msg: + return False + provider_slug = (provider or "").strip().lower() + for param, senders in _SERVER_INJECTED_PARAM_SENDERS.items(): + if param not in error_msg: + continue + # Require the message to actually be a rejection of that parameter, + # not an incidental mention. + if not ( + "not supported" in error_msg + or "unsupported" in error_msg + or "unknown" in error_msg + or "unrecognized" in error_msg + ): + continue + if any(sender in provider_slug for sender in senders): + # This route sends the field on purpose — a real request error. + return False + return True + return False + + # OpenRouter aggregator policy-block patterns. # # When a user's OpenRouter account privacy setting (or a per-request @@ -1310,11 +1363,16 @@ def _classify_by_status( # server_error" rule turns one bad request into a retry flood. # Detect the unambiguous request-validation signals (in either the # message text or the structured error code) and fail fast. + # + # Exception: a parameter WE never sent on this route was injected by + # the provider/proxy itself, so the rejection is not deterministic and + # the generic retryable-5xx handling is correct. Mirrors the guard in + # _classify_400 — see _is_server_injected_param_rejection. if ( any(p in error_msg for p in _REQUEST_VALIDATION_PATTERNS) or error_code.lower() in {"invalid_request_error", "unknown_parameter", "unsupported_parameter"} - ): + ) and not _is_server_injected_param_rejection(error_msg, provider): return result_fn( FailoverReason.format_error, retryable=False, @@ -1473,6 +1531,25 @@ def _classify_400( should_fallback=False, ) + # Server-injected parameter rejection: a 400 blaming a request field the + # client never sent. MUST be checked BEFORE the request-validation branch + # below, which would otherwise class it as a deterministic format_error and + # abort the turn. + # + # Observed live on the Codex OAuth backend (chatgpt.com/backend-api/codex): + # it intermittently adds ``prompt_cache_retention`` to its own upstream + # call and then rejects it, so a byte-identical request succeeds on retry + # (measured ~20% failure over n=20 on a minimal 1-message request that + # provably carried no cache parameters). Retrying is the correct and only + # recovery; failing fast burnt an entire large-context request per attempt. + if _is_server_injected_param_rejection(error_msg, provider): + return result_fn( + FailoverReason.server_error, + retryable=True, + # The request shape was fine — never route this into compression. + should_compress=False, + ) + # Request-validation errors (unsupported / unknown parameter) MUST be # checked BEFORE context_overflow. A GPT-5 model rejecting max_tokens # returns: diff --git a/agent/error_surface.py b/agent/error_surface.py new file mode 100644 index 0000000000..4016a9fb3c --- /dev/null +++ b/agent/error_surface.py @@ -0,0 +1,258 @@ +"""Structured error-surface descriptors for UI clients (Desktop/TUI). + +Maps the internal failure taxonomy (``agent.error_classifier.FailoverReason`` +values carried in turn results as ``failure_reason``, or raw exceptions from +the turn dispatcher) onto a small, stable wire descriptor: + + {"layer": , "code": , "retryable": } + +The *layer* names which part of the stack failed, so clients can say +"Provider error" / "Gateway error" instead of toasting an opaque string and +leaving the user to guess whether the model, the gateway, or the app froze: + + provider — the model/provider API rejected or failed the call + endpoint — a user-configured custom/local endpoint failed (transport) + streaming — the provider's SSE/stream connection dropped mid-turn + auth — authentication/authorization failed + billing — credits/quota wall (clients usually have a richer + billing_block descriptor; this is the fallback signal) + gateway — the local gateway/agent runtime itself errored + runtime — agent initialization / local environment failure + disk — local disk full / persistence failure + +This module is intentionally dependency-light and NEVER raises: surfacing +diagnostics must not be able to break the error path it describes. Clients +treat the descriptor as advisory — an absent or partial descriptor falls +back to today's string-sniffing behavior (older backends keep working). +""" + +from __future__ import annotations + +import logging +from typing import Any, Optional + +logger = logging.getLogger(__name__) + +# UI layers (wire values — stable contract with desktop/TUI clients). +LAYER_PROVIDER = "provider" +LAYER_ENDPOINT = "endpoint" +LAYER_STREAMING = "streaming" +LAYER_AUTH = "auth" +LAYER_BILLING = "billing" +LAYER_GATEWAY = "gateway" +LAYER_RUNTIME = "runtime" +LAYER_DISK = "disk" + +# failure_reason (FailoverReason.value) → UI layer. Reasons not listed fall +# back to LAYER_PROVIDER: every FailoverReason is produced by classifying a +# provider API call, so "the provider call failed" is the honest default. +_REASON_TO_LAYER = { + "auth": LAYER_AUTH, + "auth_permanent": LAYER_AUTH, + "billing": LAYER_BILLING, + "billing_unverified": LAYER_BILLING, +} + +# Transport-ish reasons: the failure is between us and the base_url, not a +# verdict the provider returned. On a custom/local endpoint these point at +# the user's endpoint config, so they surface as LAYER_ENDPOINT there. +_TRANSPORT_REASONS = { + "timeout", + "ssl_cert_verification", +} + +# Reasons that are deterministic for the request — a bare "Retry" repeats the +# same failure, so clients shouldn't lead with it. Fallback only: results +# from current backends carry the classifier's own verdict in +# ``failure_retryable`` and never consult this set. Kept in sync with +# ``classify_api_error``'s retryable=False verdicts. +_NON_RETRYABLE_REASONS = { + "auth", + "auth_permanent", + "billing", + "billing_unverified", + "content_policy_blocked", + "provider_policy_blocked", + "model_not_found", + "format_error", + "ssl_cert_verification", +} + +# Providers whose base_url is user-supplied rather than a known vendor — +# transport failures against these are endpoint-config problems. +_CUSTOM_ENDPOINT_PROVIDERS = { + "custom", + "local", + "llama.cpp", + "llamacpp", + "ollama", + "lmstudio", + "vllm", +} + +# Message fragments that mark a mid-stream connection drop. Deliberately +# narrow: these strings come from our own retry-exhaustion summaries and the +# OpenAI SDK's stream-abort errors. +_STREAM_DROP_FRAGMENTS = ( + "stream connection", + "peer closed connection", + "incomplete chunked read", + "connection broken", + "stream ended prematurely", + "sse", + "mid-stream", +) + +# Exception modules that indicate the failure came from an API/transport call +# (vs. a bug in our own dispatcher code, which is a gateway-layer failure). +# Covers every SDK family our provider adapters raise from: OpenAI-compatible +# (openai/httpx/httpcore), Anthropic, Bedrock (botocore/boto3), Google +# (google.*/grpc), plus raw transports (requests/aiohttp/ssl/socket/urllib). +_API_EXC_MODULE_PREFIXES = ( + "openai", + "httpx", + "httpcore", + "anthropic", + "botocore", + "boto3", + "google", + "grpc", + "requests", + "aiohttp", + "ssl", + "socket", + "urllib", +) + + +def _is_custom_endpoint(provider: Optional[str]) -> bool: + p = (provider or "").strip().lower() + return p in _CUSTOM_ENDPOINT_PROVIDERS or p.startswith("custom:") + + +def _looks_like_stream_drop(message: str) -> bool: + msg = message.lower() + return any(fragment in msg for fragment in _STREAM_DROP_FRAGMENTS) + + +def _surface( + layer: str, + code: str, + retryable: bool, + provider: str = "", + model: str = "", +) -> dict: + out = {"layer": layer, "code": code, "retryable": bool(retryable)} + # The failing session's identity, captured at classification time so + # clients report the model/provider that actually failed — not whatever + # the foreground composer points at when a button is clicked later. + if provider: + out["provider"] = provider + if model: + out["model"] = model + return out + + +def build_error_surface_from_result( + result: Any, provider: str = "", model: str = "" +) -> Optional[dict]: + """Descriptor for a returned-error turn result (``failed=True`` dicts). + + Reads the ``failure_reason`` the conversation loop already stamps + (a ``FailoverReason.value``) plus the error text, and maps them onto a + UI layer. Returns None when the result carries no failure signal. + """ + try: + if not isinstance(result, dict): + return None + error_text = str(result.get("error") or "") + reason = str(result.get("failure_reason") or "").strip() + if not error_text and not reason: + return None + + # Disk-full wins outright: the fix (free space) is unrelated to the + # provider stack, and hermes_state owns the pattern list. + try: + from hermes_state import is_disk_full_error + + if error_text and is_disk_full_error(error_text): + return _surface(LAYER_DISK, "disk_full", False, provider, model) + except Exception: # pragma: no cover - defensive import guard + pass + + if result.get("billing_block") or reason in ("billing", "billing_unverified"): + return _surface(LAYER_BILLING, reason or "billing", False, provider, model) + + if not reason: + # Failed result without a classified reason (legacy paths). + if _looks_like_stream_drop(error_text): + return _surface(LAYER_STREAMING, "stream_drop", True, provider, model) + return _surface(LAYER_PROVIDER, "unknown", True, provider, model) + + layer = _REASON_TO_LAYER.get(reason) + if layer is None: + if reason in _TRANSPORT_REASONS and _is_custom_endpoint(provider): + layer = LAYER_ENDPOINT + elif _looks_like_stream_drop(error_text): + layer = LAYER_STREAMING + else: + layer = LAYER_PROVIDER + # Prefer the classifier's own retry verdict when the result carries it + # (conversation_loop stamps ``failure_retryable`` next to + # ``failure_reason``); the reason-set fallback covers older results. + retryable = result.get("failure_retryable") + if not isinstance(retryable, bool): + retryable = reason not in _NON_RETRYABLE_REASONS + return _surface(layer, reason, retryable, provider, model) + except Exception: # pragma: no cover — never break the error path + logger.debug("error_surface: result classification failed", exc_info=True) + return None + + +def build_error_surface_from_exception( + exc: BaseException, provider: str = "", model: str = "" +) -> Optional[dict]: + """Descriptor for an exception that escaped the turn dispatcher. + + API/transport exceptions are classified through the real + ``classify_api_error`` pipeline (same taxonomy as the retry loop); + anything else is a gateway-layer failure — a bug or environment problem + in our own dispatcher, not a provider verdict. + """ + try: + message = str(exc) or type(exc).__name__ + + try: + from hermes_state import is_disk_full_error + + if is_disk_full_error(exc): + return _surface(LAYER_DISK, "disk_full", False, provider, model) + except Exception: # pragma: no cover - defensive import guard + pass + + exc_module = type(exc).__module__ or "" + api_like = exc_module.split(".")[0] in _API_EXC_MODULE_PREFIXES or hasattr( + exc, "status_code" + ) + + if not api_like or not isinstance(exc, Exception): + return _surface(LAYER_GATEWAY, type(exc).__name__, True, provider, model) + + from agent.error_classifier import classify_api_error + + classified = classify_api_error(exc, provider=provider, model=model) + reason = classified.reason.value + + synthetic = { + "error": classified.message or message, + "failure_reason": reason, + } + surface = build_error_surface_from_result( + synthetic, provider=provider, model=model + ) + if surface is not None: + surface["retryable"] = bool(classified.retryable) + return surface + except Exception: # pragma: no cover — never break the error path + logger.debug("error_surface: exception classification failed", exc_info=True) + return None diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 8a8a4bee26..bad065573a 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -501,16 +501,19 @@ DEFAULT_CONTEXT_LENGTHS = { # https://platform.minimax.io/docs/api-reference/text-chat-openai "minimax-m3": 1000000, "minimax": 204800, - # GLM — GLM-5.2 ships with a 1M context window (verified empirically: - # needle-in-a-haystack retrieval at 789K prompt tokens succeeded with - # zero errors on api.z.ai/api/coding/paas/v4). Older GLM models - # (5, 5.1, 5-turbo) are ~202K. Longest-key-first substring matching - # ensures "glm-5.2" resolves to 1M while older variants still hit the - # generic 202K fallback. + # GLM — GLM-5.2 and GLM-5.3 ship with a 1M context window. GLM-5.2 was + # verified empirically (needle-in-a-haystack retrieval at 789K prompt + # tokens succeeded with zero errors on api.z.ai/api/coding/paas/v4). + # GLM-5.3 uses the same base model (all gains are post-training) with + # 1M context / 128K max output per docs.z.ai/guides/llm/glm-5.3 + # (verified 2026-08-14). Older GLM models (5, 5.1, 5-turbo) are ~202K. + # Longest-key-first substring matching ensures "glm-5.2"/"glm-5.3" + # resolve to 1M while older variants still hit the generic 202K fallback. "glm-5.2": 1_048_576, # OpenRouter's free GLM-5.2 variant is capped at 256K (live metadata, # 2026-08-21) — longer key wins over the 1M paid entry above. "glm-5.2:free": 256_000, + "glm-5.3": 1_048_576, "glm": 202752, # xAI Grok — xAI /v1/models does not return context_length metadata, # so these hardcoded fallbacks prevent Hermes from probing-down to @@ -2395,6 +2398,7 @@ _CODEX_OAUTH_CONTEXT_FALLBACK: Dict[str, int] = { "gpt-5.6-sol": 272_000, "gpt-5.6-terra": 272_000, "gpt-5.6-luna": 272_000, + "gpt-daybreak-blue-latest": 272_000, "gpt-5.5": 272_000, "gpt-5.4": 272_000, "gpt-5.2": 272_000, @@ -2428,6 +2432,7 @@ _CODEX_OAUTH_VERIFIED_ABOVE_ADVERTISED_PREFIXES: Dict[str, int] = { } _CODEX_OAUTH_VERIFIED_ABOVE_ADVERTISED_EXACT: Dict[str, int] = { "gpt-5.4": 900_000, # verified live at 900K; gpt-5.4-mini rejected 500K — excluded + "gpt-daybreak-blue-latest": 900_000, # exact Daybreak/Sol alias verified at 911,276 } # The advertised value the verified-above table is allowed to override. diff --git a/agent/reasoning_effort.py b/agent/reasoning_effort.py index e29c0273e5..396e9fc0be 100644 --- a/agent/reasoning_effort.py +++ b/agent/reasoning_effort.py @@ -117,6 +117,13 @@ KIMI_K3_OVERRIDES: dict[str, str] = {"medium": "high", "xhigh": "max"} GLM52_EFFORTS: tuple[str, ...] = ("high", "max") GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"} +#: GLM-5.3 widens the knob to a graded low/medium/high/max scale — verified +#: live on api.z.ai/api/coding/paas/v4 (issue #91789, 2026-08-21): every +#: level accepted with monotonic reasoning-token scaling (low=4, medium=11, +#: high=98, max=125 on the probe prompt). ``xhigh`` requests the top tier. +GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max") +GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"} + #: DeepSeek V4 OpenAI-compat endpoint: low/medium/high/max; ``xhigh`` #: requests the top tier (matches the shipped profile mapping). DEEPSEEK_V4_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max") diff --git a/agent/tool_executor.py b/agent/tool_executor.py index e7bb9126db..3f0d64fbb1 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -2066,6 +2066,31 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe tool_duration = time.time() - tool_start_time if agent._should_emit_quiet_tool_messages(): agent._vprint(f" {_get_cute_tool_message_impl('todo', function_args, tool_duration, result=function_result)}") + elif function_name == "message_agent": + # Bot Mode teammate DM (tools/bot_mode_dm.py) — injected, not + # registered: only a canonical Bot Chat session carries the + # schema, and the tool re-gates on the session title itself. + def _execute(next_args: dict) -> Any: + from tools.bot_mode_dm import message_agent_tool as _message_agent_tool + return _message_agent_tool( + target=next_args.get("target", ""), + message=next_args.get("message", ""), + task_id=effective_task_id, + agent=agent, + ) + function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware( + agent, + function_name=function_name, + function_args=function_args, + effective_task_id=effective_task_id, + tool_call_id=getattr(tool_call, "id", "") or "", + execute=_execute, + scope_block=_ts_scope_block, + display_index=i, + )) + tool_duration = time.time() - tool_start_time + if agent._should_emit_quiet_tool_messages(): + agent._vprint(f" {_get_cute_tool_message_impl('message_agent', function_args, tool_duration, result=function_result)}") elif function_name == "session_search": def _execute(next_args: dict) -> Any: session_db = agent._get_session_db_for_recall() diff --git a/agent/turn_context.py b/agent/turn_context.py index e63372d4b7..ee36505521 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -754,6 +754,19 @@ def build_turn_context( active_system_prompt = agent._cached_system_prompt + # Bot Mode DM tool — injected ONLY into a bot's canonical "Bot Chat" + # session on Bot-Mode-managed installs (same gate as the protocol + # section above). The gate is stable for a session's lifetime, so the + # tool list is byte-identical every turn: prompt-cache safe. Every + # other session (CLI, gateway chats, group-room member sessions, cron, + # subagents) fails the gate and never sees the schema. + try: + from tools.bot_mode_dm import ensure_message_agent_tool + + ensure_message_agent_tool(agent) + except Exception: + logger.debug("message_agent injection skipped", exc_info=True) + # Create the DB session row now that _cached_system_prompt is populated, so # the persisted snapshot is written non-NULL on the first turn (Issue # #45499). Idempotent: _ensure_db_session() no-ops once the row exists. diff --git a/apps/desktop/electron/backend-health.test.ts b/apps/desktop/electron/backend-health.test.ts index 8d29ea1988..a23ba5a73b 100644 --- a/apps/desktop/electron/backend-health.test.ts +++ b/apps/desktop/electron/backend-health.test.ts @@ -7,7 +7,10 @@ import { isAuthRejectionError, isGatedMissingHealthError, isMissingHealthEndpointError, + isNousCloudAgentUrl, isReauthRequiredError, + isServerSideHttpError, + makeNousCloudBackendDownError, waitForHermesReady } from './backend-health' @@ -338,3 +341,186 @@ test('error-shape predicates', () => { // A gated 401 must NOT be conflated with a missing route by the 404 predicate. assert.equal(isMissingHealthEndpointError(new Error(GATE_401)), false) }) + +test('isServerSideHttpError detects 502/503/504', () => { + // 503 — server-side fault + const result503 = isServerSideHttpError(new Error('503: Service Unavailable')) + assert.ok(result503, 'should detect 503') + assert.equal(result503?.statusCode, 503) + assert.equal(result503?.detail, '503: Service Unavailable') + + // 502 + const result502 = isServerSideHttpError(new Error('502: Bad Gateway')) + assert.ok(result502, 'should detect 502') + assert.equal(result502?.statusCode, 502) + + // 504 + const result504 = isServerSideHttpError(new Error('504: Gateway Timeout')) + assert.ok(result504, 'should detect 504') + assert.equal(result504?.statusCode, 504) + + // 500 is NOT a server-side HTTP error per our definition (keeps polling) + const result500 = isServerSideHttpError(new Error('500: Internal Server Error')) + assert.equal(result500, null) + + // 401/403/404/429 are not server-side faults + assert.equal(isServerSideHttpError(new Error('401: Unauthorized')), null) + assert.equal(isServerSideHttpError(new Error('403: Forbidden')), null) + assert.equal(isServerSideHttpError(new Error('404: Not Found')), null) + assert.equal(isServerSideHttpError(new Error('429: Too Many Requests')), null) + + // Non-HTTP errors (timeouts, network failures) don't match the pattern + assert.equal(isServerSideHttpError(new Error('connect ECONNREFUSED')), null) + assert.equal(isServerSideHttpError(null), null) + assert.equal(isServerSideHttpError('503: something'), null) // not an Error +}) + +test('isNousCloudAgentUrl detects cloud agent hosts', () => { + // Positive cases + assert.equal(isNousCloudAgentUrl('https://ares-3009.agents.nousresearch.com'), true) + assert.equal(isNousCloudAgentUrl('https://ares-3009.agents.nousresearch.com/api/health'), true) + assert.equal(isNousCloudAgentUrl('http://test.agents.nousresearch.com'), true) + + // Negative cases + assert.equal(isNousCloudAgentUrl('http://127.0.0.1:9000'), false) + assert.equal(isNousCloudAgentUrl('https://gateway.example.com'), false) + assert.equal(isNousCloudAgentUrl('https://nousresearch.com'), false) + assert.equal(isNousCloudAgentUrl('not-a-url'), false) +}) + +test('waitForHermesReady surfaces actionable error for cloud agent 503', async () => { + let attempts = 0 + const currentTime = { value: 0 } + + try { + await waitForHermesReady('https://ares-3009.agents.nousresearch.com', { + fetchPublicJson: async () => { + attempts++ + // Always return 503 + throw new Error('503: Service Unavailable') + }, + fetchJson: async () => { + throw new Error('503: Service Unavailable') + }, + sleep: async () => {}, + // Advance the mock clock per poll — a frozen now() never crosses the + // deadline and the readiness loop spins forever (hung the whole vitest + // electron project for 20m in CI). + now: () => { + currentTime.value += 20 + + return currentTime.value + }, + timeoutMs: 100, + pollMs: 1 + }) + assert.fail('should have thrown') + } catch (error: any) { + assert.ok(error.message.includes('Nous Cloud agent'), `unexpected message: ${error.message}`) + assert.ok(error.message.includes('503'), `should mention status code: ${error.message}`) + assert.ok(error.message.includes('portal.nousresearch.com'), `should mention portal: ${error.message}`) + assert.ok(error.message.includes('discord.gg/NousResearch'), `should mention Discord: ${error.message}`) + assert.equal(error.isCloudBackendDown, true) + assert.equal(error.statusCode, 503) + assert.ok(attempts > 1, 'should have retried before failing') + } +}) + +test('waitForHermesReady does not cloud-wrap non-cloud 503 errors', async () => { + const currentTime = { value: 0 } + + try { + await waitForHermesReady('http://127.0.0.1:9000', { + fetchPublicJson: async () => { + throw new Error('503: Service Unavailable') + }, + fetchJson: async () => { + throw new Error('503: Service Unavailable') + }, + sleep: async () => {}, + // Same advancing clock as above — frozen now() = infinite loop. + now: () => { + currentTime.value += 20 + + return currentTime.value + }, + timeoutMs: 100, + pollMs: 1 + }) + assert.fail('should have thrown') + } catch (error: any) { + // Non-cloud URLs get the generic message + assert.ok(error.message.includes('did not become ready'), `unexpected message: ${error.message}`) + assert.equal(error.isCloudBackendDown, undefined) + } +}) + +test('isServerSideHttpError detects structured statusCode even when the message is opaque', () => { + const err = new Error('upstream unavailable') as any + err.statusCode = 503 + const result = isServerSideHttpError(err) + assert.ok(result) + assert.equal(result?.statusCode, 503) + assert.equal(result?.detail, 'upstream unavailable') + + const err502 = new Error('bad gateway') as any + err502.statusCode = 502 + assert.equal(isServerSideHttpError(err502)?.statusCode, 502) + + const err504 = new Error('gateway timeout') as any + err504.statusCode = 504 + assert.equal(isServerSideHttpError(err504)?.statusCode, 504) +}) + +test('isServerSideHttpError rejects non-Error inputs even with a 503-shaped value', () => { + // The structured path requires an actual Error (the fetch layer attaches + // statusCode to an Error instance); a bare string/null/number must not be + // misclassified by the legacy prefix fallback. + assert.equal(isServerSideHttpError('503: something'), null) + assert.equal(isServerSideHttpError({ statusCode: 503 }), null) + assert.equal(isServerSideHttpError(null), null) + assert.equal(isServerSideHttpError(503), null) +}) + +test('isServerSideHttpError structured path excludes 500/401/403/404/429 even when statusCode is attached', () => { + for (const code of [500, 401, 403, 404, 429]) { + const err = new Error(`HTTP ${code}`) as any + err.statusCode = code + assert.equal(isServerSideHttpError(err), null, `should reject statusCode ${code}`) + } +}) + +test('makeNousCloudBackendDownError produces the Cloud shape and preserves cause', () => { + const err = new Error('upstream unavailable') as any + err.statusCode = 503 + const result = makeNousCloudBackendDownError('https://ares-3009.agents.nousresearch.com', err) + assert.ok(result) + assert.equal((result as any).isCloudBackendDown, true) + assert.equal((result as any).statusCode, 503) + assert.equal((result as any).cause, err) + assert.ok(result?.message.includes('Nous Cloud agent ares-3009.agents.nousresearch.com is down')) +}) + +test('makeNousCloudBackendDownError returns null for a Cloud 401 (routes to reauth)', () => { + const err = new Error('Unauthorized') as any + err.statusCode = 401 + assert.equal(makeNousCloudBackendDownError('https://ares-3009.agents.nousresearch.com', err), null) +}) + +test('makeNousCloudBackendDownError returns null for a non-Cloud 503 (generic remote failure)', () => { + const err = new Error('Service Unavailable') as any + err.statusCode = 503 + assert.equal(makeNousCloudBackendDownError('https://gateway.example.com', err), null) + assert.equal(makeNousCloudBackendDownError('http://127.0.0.1:9000', err), null) +}) + +test('makeNousCloudBackendDownError preserves legacy string-prefix compatibility', () => { + const result = makeNousCloudBackendDownError( + 'https://ares-3009.agents.nousresearch.com', + new Error('503: Service Unavailable') + ) + + assert.ok(result) + assert.equal((result as any).isCloudBackendDown, true) + assert.equal((result as any).statusCode, 503) +}) diff --git a/apps/desktop/electron/backend-health.ts b/apps/desktop/electron/backend-health.ts index 3da6c7a80f..ff380f1390 100644 --- a/apps/desktop/electron/backend-health.ts +++ b/apps/desktop/electron/backend-health.ts @@ -38,6 +38,124 @@ export interface HermesReadyOptions { export const REMOTE_SESSION_EXPIRED_MESSAGE = 'Your remote gateway session has expired. Open Settings → Gateway and click "Sign in" again.' +/** + * True for HTTP 502/503/504 from the backend — a server-side fault, not a + * connectivity or auth issue. These keep polling in the readiness loop but, + * when they exhaust the budget, the user needs to know it is the remote + * server that is down, not their local config. + */ +export function isServerSideHttpError(error: unknown): { + statusCode: number + detail: string +} | null { + // Reject non-Error inputs, as before. The fetch layer attaches statusCode to + // an actual Error instance (err.statusCode = statusCode), so requiring an + // Error is compatible with structured detection and keeps plain strings / + // null / numbers from being misclassified by the legacy prefix. + if (!(error instanceof Error)) { + return null + } + + // Structured-first: the real fetch layer attaches err.statusCode = statusCode + // (see fetchJson). That is the strongest transport contract, so inspect it + // before falling back to the legacy "503: ..." string prefix. + if ('statusCode' in error) { + const structured = Number((error as { statusCode?: unknown }).statusCode) + + if (Number.isInteger(structured) && (structured === 502 || structured === 503 || structured === 504)) { + const detail = error.message + + return { statusCode: structured, detail } + } + } + + // Compatibility fallback: the legacy leading "503: ..." prefix. Only reached + // when no structured statusCode matched (or was absent). + const message = error.message + const match = /^(\d{3}):/.exec(message) + + if (!match) { + return null + } + + const code = parseInt(match[1], 10) + + if (code === 502 || code === 503 || code === 504) { + return { statusCode: code, detail: message } + } + + return null +} + +/** + * The one factory for the actionable Nous Cloud agent-is-down error, shared by + * both startup boundaries that can observe a server-side HTTP fault: + * + * - OAuth WS-ticket mint (buildRemoteConnection → mintGatewayWsTicket), which + * runs BEFORE the readiness loop; and + * - readiness-probe exhaustion in waitForHermesReady(). + * + * Returns null unless the backend is a *.agents.nousresearch.com host AND the + * error classifies as 502/503/504. When it matches, returns an error carrying: + * isCloudBackendDown, statusCode, detail, and the original cause. The renderer + * overlay keys on isCloudBackendDown/statusCode; main owns the classification. + */ +export function makeNousCloudBackendDownError(baseUrl: string, error: unknown): Error | null { + if (!isNousCloudAgentUrl(baseUrl)) { + return null + } + + const serverError = isServerSideHttpError(error) + + if (serverError === null) { + return null + } + + let hostname = baseUrl + + try { + hostname = new URL(baseUrl).hostname + } catch { + // baseUrl is known to parse (isNousCloudAgentUrl already did); keep the raw + // value as a last resort rather than throwing. + } + + const detail = error instanceof Error ? error.message : String(error ?? '') + + const err = new Error( + `Nous Cloud agent ${hostname} is down ` + + `(HTTP ${serverError.statusCode}: server-side fault). ` + + 'Check https://portal.nousresearch.com for backend status, ' + + 'or switch to Local mode in Settings → Gateway. ' + + 'You can also reach out on Discord at discord.gg/NousResearch ' + + 'for immediate assistance. ' + + `Original detail: ${detail}` + ) as any + + err.isCloudBackendDown = true + err.statusCode = serverError.statusCode + err.detail = detail + err.cause = error + + return err +} + +/** + * True when the backend URL points at a Nous-managed Hermes Cloud instance + * (e.g. ares-3009.agents.nousresearch.com). These are Fly.io-hosted machines + * the user cannot restart themselves — a 503 from one means the server is down + * and the recovery path is Portal/Discord/wait. + */ +export function isNousCloudAgentUrl(baseUrl: string): boolean { + try { + const host = new URL(baseUrl).hostname + + return host.endsWith('.agents.nousresearch.com') + } catch { + return false + } +} + export function isMissingHealthEndpointError(error: unknown): boolean { const message = error instanceof Error ? error.message : String(error ?? '') @@ -165,5 +283,18 @@ export async function waitForHermesReady(baseUrl: string, options: HermesReadyOp } const detail = lastError instanceof Error ? lastError.message : 'timeout' + + // When a Nous-managed cloud agent returns a server-side HTTP error + // (502/503/504), the backend server itself is down — the user cannot + // restart it and the generic "did not become ready" message is opaque. + // Surface an actionable error instead (#85335). This is the SAME factory + // buildRemoteConnection uses at the OAuth WS-ticket-mint boundary, so both + // startup paths produce the identical Cloud-down shape. + const cloudError = makeNousCloudBackendDownError(baseUrl, lastError) + + if (cloudError !== null) { + throw cloudError + } + throw new Error(`Hermes backend did not become ready: ${detail}`) } diff --git a/apps/desktop/electron/connection-config.test.ts b/apps/desktop/electron/connection-config.test.ts index 8bc40e639a..36a042c6d3 100644 --- a/apps/desktop/electron/connection-config.test.ts +++ b/apps/desktop/electron/connection-config.test.ts @@ -14,6 +14,7 @@ import assert from 'node:assert/strict' import { test } from 'vitest' +import { makeNousCloudBackendDownError } from './backend-health' import { apiRequestRegistryConnectionId, AT_COOKIE_VARIANTS, @@ -1167,3 +1168,83 @@ test('resolveTestWsUrl (oauth) requires a mintTicket function', async () => { /mintTicket function is required/ ) }) + +test('gatewayTicketFailure preserves a structured 503 statusCode as a transport failure', () => { + const source = new Error('upstream unavailable') as any + source.statusCode = 503 + + const wrapped = gatewayTicketFailure(source, 'auth message', 'transport message') + + assert.equal(wrapped.message, 'transport message') + assert.equal((wrapped as any).statusCode, 503) + assert.equal((wrapped as any).needsOauthLogin, undefined) + assert.equal((wrapped as any).cause, source) +}) + +test('gatewayTicketFailure keeps 401 and 403 as reauth with needsOauthLogin', () => { + for (const code of [401, 403]) { + const source = new Error(`HTTP ${code}`) as any + source.statusCode = code + + const wrapped = gatewayTicketFailure(source, 'auth message', 'transport message') + + assert.equal(wrapped.message, 'auth message') + assert.equal((wrapped as any).needsOauthLogin, true) + assert.equal((wrapped as any).statusCode, code) + assert.equal((wrapped as any).cause, source) + } +}) + +test('gatewayTicketFailure only copies an integer statusCode, not a message prefix', () => { + // A legacy "503: ..." message carries no structured statusCode; the Cloud + // classifier (makeNousCloudBackendDownError) handles the prefix at the mint + // boundary. The wrapper must not invent an integer from the message. + const source = new Error('503: Service Unavailable') as any + + const wrapped = gatewayTicketFailure(source, 'auth message', 'transport message') + + assert.equal((wrapped as any).statusCode, undefined) + assert.equal((wrapped as any).needsOauthLogin, undefined) +}) + +// OAuth integration regression (#85373): the WS-ticket mint boundary runs +// BEFORE waitForHermesReady. This mirrors main.ts buildRemoteConnection's +// catch — classify a Nous Cloud server fault via the shared factory, else +// fall through to gatewayTicketFailure. Proves the production composition: +// 1. Cloud + OAuth ticket mint + 503 -> actionable Cloud-down error +// 2. Cloud + OAuth ticket mint + 401 -> reauth (never Cloud-down) +test('OAuth ticket-mint 503 surfaces the Cloud-down error (startup boundary)', () => { + const baseUrl = 'https://ares-3009.agents.nousresearch.com' + const ticketErr = new Error('upstream unavailable') as any + ticketErr.statusCode = 503 + + // The exact production sequence from main.ts. + const cloudError = makeNousCloudBackendDownError(baseUrl, ticketErr) + + if (cloudError !== null) { + assert.equal((cloudError as any).isCloudBackendDown, true) + assert.equal((cloudError as any).statusCode, 503) + assert.ok(cloudError.message.includes('Nous Cloud agent ares-3009.agents.nousresearch.com is down')) + + return + } + + const wrapped = gatewayTicketFailure(ticketErr, 'auth', 'transport') + + assert.fail(`expected Cloud-down classification, got wrapper: ${wrapped.message}`) +}) + +test('OAuth ticket-mint 401 stays on the reauth path (never Cloud-down)', () => { + const baseUrl = 'https://ares-3009.agents.nousresearch.com' + const ticketErr = new Error('Unauthorized') as any + ticketErr.statusCode = 401 + + const cloudError = makeNousCloudBackendDownError(baseUrl, ticketErr) + assert.equal(cloudError, null, 'a 401 must not become a Cloud-down error') + + const wrapped = gatewayTicketFailure(ticketErr, 'auth message', 'transport message') + + assert.equal(wrapped.message, 'auth message') + assert.equal((wrapped as any).needsOauthLogin, true) + assert.equal((wrapped as any).statusCode, 401) +}) diff --git a/apps/desktop/electron/connection-config.ts b/apps/desktop/electron/connection-config.ts index 1934fbec0d..10cdd275b1 100644 --- a/apps/desktop/electron/connection-config.ts +++ b/apps/desktop/electron/connection-config.ts @@ -134,6 +134,18 @@ function gatewayTicketFailure(error, authMessage, transportMessage) { ;(err as any).needsOauthLogin = true } + // Preserve structured HTTP context when the source error carried an integer + // statusCode (the fetch layer attaches err.statusCode). Downstream Cloud + // classification (isServerSideHttpError / makeNousCloudBackendDownError) and + // the renderer overlay depend on it surviving the ticket-error wrapper. Auth + // semantics are unchanged: 401/403 route to reauth, 5xx stays a transport + // failure, everything else keeps current behavior. + const sourceStatus = Number(error && typeof error === 'object' ? (error as any).statusCode : NaN) + + if (Number.isInteger(sourceStatus)) { + ;(err as any).statusCode = sourceStatus + } + err.cause = error return err diff --git a/apps/desktop/electron/fs-ipc.ts b/apps/desktop/electron/fs-ipc.ts index ff25db2013..61a3cb7db2 100644 --- a/apps/desktop/electron/fs-ipc.ts +++ b/apps/desktop/electron/fs-ipc.ts @@ -98,6 +98,12 @@ export function registerFsIpc({ ipcMain.handle('hermes:fs:desktopPluginsRoot', async () => localPluginsRoot('desktop-plugins')) + // The LOCAL logs root (`/logs`, profile-aware) — the error + // card's "Open Logs" action reveals agent.log/gateway.log without the user + // knowing where HERMES_HOME lives. Same Electron-local resolution as the + // plugin roots: valid in every connection mode, created on demand. + ipcMain.handle('hermes:fs:logsRoot', async () => localPluginsRoot('logs')) + // The LOCAL agent-plugin root (`/plugins`), same Electron-local // resolution as above. This is the desktop half of a UNIFIED plugin package: // an agent plugin may ship `desktop/plugin.js` alongside its Python code (the diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index adf7338ef5..e3087bdcb9 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -35,7 +35,7 @@ import { stopBackendChild as stopBackendChildImpl, stopBackendTreesForUpdate } f import { dashboardFallbackArgs, sourceDeclaresServe } from './backend-command' import { createBackendConnectionState } from './backend-connection-state' import { buildDesktopBackendEnv, hermesManagedNodePathEntries, normalizeHermesHomeRoot } from './backend-env' -import { isReauthRequiredError, waitForHermesReady } from './backend-health' +import { isReauthRequiredError, makeNousCloudBackendDownError, waitForHermesReady } from './backend-health' import { backendCommandMatches, createBackendOwnership, createBackendShutdownCoordinator } from './backend-ownership' import { canImportHermesCli, @@ -1371,11 +1371,13 @@ let nativeThemeListenerInstalled = false let bootProgressState = { error: null, fakeMode: BOOT_FAKE_MODE, + isCloudBackendDown: false, message: 'Waiting to start Hermes backend', phase: 'idle', progress: 0, retryable: false, running: false, + statusCode: null, timestamp: Date.now() } @@ -8979,6 +8981,19 @@ async function buildRemoteConnection( try { ticket = await mintGatewayWsTicket(baseUrl, remoteHeaders) } catch (error) { + // For a Nous-managed Cloud agent, a 502/503/504 from the WS-ticket mint + // means the backend server itself is down — the actionable Cloud-down + // error. This boundary runs BEFORE the readiness loop, so without this + // the ticket wrapper below would swallow the server-fault classification + // and the renderer would never see isCloudBackendDown. Preserve the + // existing 401/403 reauth and generic transport behavior for everything + // else (#85335). + const cloudError = makeNousCloudBackendDownError(baseUrl, error) + + if (cloudError !== null) { + throw cloudError + } + throw gatewayTicketFailure( error, 'Your remote gateway session has expired. Open Settings → Gateway and click "Sign in" again.', @@ -10892,6 +10907,18 @@ async function startHermes() { const message = error instanceof Error ? error.message : String(error) const hostKeyChanged = isHostKeyChangedBootFailure(error) + // Carry structured Cloud-down metadata through the boot-progress / IPC + // boundary when present, so the renderer overlay can key on it rather than + // re-classifying the message string. main owns classification; the renderer + // only consumes the structured result (#85335). + const isCloudBackendDown = Boolean(error && typeof error === 'object' && (error as any).isCloudBackendDown === true) + + const statusCode = Number( + error && typeof error === 'object' && Number.isInteger((error as any).statusCode) + ? (error as any).statusCode + : NaN + ) + // Only latch LOCAL boot failures. A remote failure (lapsed session / mint // timeout / host briefly unreachable across sleep) is transient and has no // child 'exit' handler to clear the cache — latching it would wedge the app @@ -10920,6 +10947,7 @@ async function startHermes() { updateBootProgress( { error: message, + isCloudBackendDown: isCloudBackendDown || undefined, message: `Desktop boot failed: ${message}`, phase: 'backend.error', // Renderer contract for the self-heal loop (#82679): a transient @@ -10933,7 +10961,8 @@ async function startHermes() { isReauth: isReauthRequiredError(error), isHostKeyChanged: hostKeyChanged }), - running: false + running: false, + statusCode: Number.isInteger(statusCode) ? statusCode : undefined }, { allowDecrease: true } ) diff --git a/apps/desktop/electron/preload.ts b/apps/desktop/electron/preload.ts index 7d8dac57af..7eac06bd52 100644 --- a/apps/desktop/electron/preload.ts +++ b/apps/desktop/electron/preload.ts @@ -269,6 +269,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', { revealPath: targetPath => ipcRenderer.invoke('hermes:fs:reveal', targetPath), openDir: dirPath => ipcRenderer.invoke('hermes:fs:openDir', dirPath), desktopPluginsRoot: () => ipcRenderer.invoke('hermes:fs:desktopPluginsRoot'), + logsRoot: () => ipcRenderer.invoke('hermes:fs:logsRoot'), agentPluginsRoot: () => ipcRenderer.invoke('hermes:fs:agentPluginsRoot'), renamePath: (targetPath, newName) => ipcRenderer.invoke('hermes:fs:rename', targetPath, newName), writeTextFile: (filePath, content) => ipcRenderer.invoke('hermes:fs:writeText', filePath, content), diff --git a/apps/desktop/package.json b/apps/desktop/package.json index ef1a9f6d16..73a71ed35a 100644 --- a/apps/desktop/package.json +++ b/apps/desktop/package.json @@ -66,14 +66,12 @@ "test:find-in-page-native": "electron electron/find-in-page-native-fixture", "test": "vitest run", "preview": "node scripts/assert-root-install.mjs && vite preview --host 127.0.0.1 --port 4174", + "check:test:ui": "npm run test:ui", "check:test:desktop:platforms": "npm run test:desktop:platforms", - "check:test:plugins": "node --test src/plugins/*/tests/*.test.mjs", - "check:test:ui:shard-1of3": "node scripts/run-ui-shard.mjs", - "check:test:ui:shard-2of3": "node scripts/run-ui-shard.mjs", - "check:test:ui:shard-3of3": "node scripts/run-ui-shard.mjs", "check:test:desktop:all": "npm run test:desktop:all", + "check:test:plugins": "node --test src/plugins/*/tests/*.test.mjs", "check:lint": "npm run typecheck && npm run lint", - "check": "npm run check:lint && npm run test:ui && npm run test:desktop:platforms && npm run test:desktop:all", + "check": "npm run check:lint && npm run test:ui && npm run test:desktop:platforms && npm run test:desktop:all && npm run check:test:plugins", "test:e2e": "npm run build && playwright test e2e/", "test:e2e:visual": "npm run build && WLR_BACKENDS=headless WLR_NO_HARDWARE_CURSORS=1 cage -- npx playwright test e2e/ --reporter=list", "test:e2e:update-snapshots": "npm run build && WLR_BACKENDS=headless WLR_NO_HARDWARE_CURSORS=1 cage -- npx playwright test e2e/ --reporter=list --update-snapshots", diff --git a/apps/desktop/scripts/run-ui-shard.mjs b/apps/desktop/scripts/run-ui-shard.mjs deleted file mode 100644 index 8f29f16f62..0000000000 --- a/apps/desktop/scripts/run-ui-shard.mjs +++ /dev/null @@ -1,61 +0,0 @@ -// Runs one shard of the UI vitest suite, deriving the shard index/count from -// the npm script NAME (npm_lifecycle_event), so the name and the flag can -// never disagree. A copy-paste slip like "check:test:ui:shard-2of3" running -// --shard=1/3 would silently skip a third of the suite while CI stays green; -// deriving from the name makes that impossible. -// -// It also validates that this package.json declares exactly the shard family -// 1..M for a single M, so a partial 3→4 migration (adding shard-4of4 without -// updating the siblings) fails loudly instead of dropping coverage. -import { spawnSync } from 'node:child_process' -import { readFileSync } from 'node:fs' -import { dirname, join } from 'node:path' -import { fileURLToPath } from 'node:url' - -const SHARD_RE = /^check:test:ui:shard-(\d+)of(\d+)$/ - -const scriptName = process.env.npm_lifecycle_event ?? '' -const match = scriptName.match(SHARD_RE) -if (!match) { - console.error( - `run-ui-shard: must be invoked via an npm script named check:test:ui:shard-of (got ${JSON.stringify(scriptName)})`, - ) - process.exit(1) -} -const [, indexRaw, countRaw] = match -const index = Number(indexRaw) -const count = Number(countRaw) -if (!(index >= 1 && index <= count)) { - console.error(`run-ui-shard: shard index ${index} out of range 1..${count}`) - process.exit(1) -} - -// The whole family must be exactly 1..M of one M — otherwise a rename or a -// partial count bump leaves a silently untested slice of the suite. -const pkgDir = dirname(dirname(fileURLToPath(import.meta.url))) -const pkg = JSON.parse(readFileSync(join(pkgDir, 'package.json'), 'utf8')) -const family = Object.keys(pkg.scripts ?? {}) - .map((name) => name.match(SHARD_RE)) - .filter(Boolean) -const counts = new Set(family.map((m) => Number(m[2]))) -const indices = family.map((m) => Number(m[1])).sort((a, b) => a - b) -const expected = Array.from({ length: count }, (_, i) => i + 1) -if (counts.size !== 1 || indices.length !== count || indices.some((v, i) => v !== expected[i])) { - console.error( - `run-ui-shard: shard scripts must form exactly 1..M for a single M; found indices [${indices}] with counts {${[...counts]}}`, - ) - process.exit(1) -} - -// Delegate through test:ui so the vitest command stays single-sourced. -// npm resolves to npm.cmd on Windows, which needs a shell (same handling as -// test-desktop.mjs and stage-native-deps.mjs). -const result = spawnSync( - 'npm', - ['run', 'test:ui', '--', `--shard=${index}/${count}`, ...process.argv.slice(2)], - { stdio: 'inherit', cwd: pkgDir, shell: process.platform === 'win32' }, -) -if (result.error) { - console.error(`run-ui-shard: ${result.error.message}`) -} -process.exit(result.status ?? 1) diff --git a/apps/desktop/src/app/chat/pane-mirror.ts b/apps/desktop/src/app/chat/pane-mirror.ts index 916b64500c..3a6995e1b4 100644 --- a/apps/desktop/src/app/chat/pane-mirror.ts +++ b/apps/desktop/src/app/chat/pane-mirror.ts @@ -9,7 +9,6 @@ import type { ReadableAtom } from 'nanostores' import type { ReactElement, ReactNode, PointerEvent as ReactPointerEvent } from 'react' -import type { DoubleTapContext } from '@/components/pane-shell/tree/renderer/drag-session' import { registerPaneCloser, removeTreePane, treePanesWithPrefix } from '@/components/pane-shell/tree/store' import { registry } from '@/contrib/registry' import type { TileDock } from '@/store/session-states' @@ -44,12 +43,7 @@ export interface PaneMirror { tabWrap?: (key: string, tab: ReactElement) => ReactNode /** Override the tile's TAB drag (session drop language: stack/split/link). * Returns whether it took the drag (see PaneChrome.tabDrag). */ - tabDrag?: ( - key: string, - event: ReactPointerEvent, - onTap: () => void, - double?: DoubleTapContext - ) => boolean + tabDrag?: (key: string, event: ReactPointerEvent, onTap: () => void) => boolean /** Wired as the pane's closer (tab Close). */ close: (key: string) => void } @@ -89,11 +83,10 @@ export function paneMirror(cfg: PaneMirror): () => void { minWidth: cfg.minWidth, // Every mirrored tile is a full workspace surface docked beside main — // and closeable, which is what keeps its tab when it lands in a zone of - // its own (see lone-header.ts). + // its own (see strip-visibility.ts). placement: 'main', tabDrag: cfg.tabDrag - ? (event: ReactPointerEvent, onTap: () => void, double?: DoubleTapContext) => - cfg.tabDrag!(key, event, onTap, double) + ? (event: ReactPointerEvent, onTap: () => void) => cfg.tabDrag!(key, event, onTap) : undefined, // returns boolean (handled) — see PaneChrome.tabDrag tabWrap: cfg.tabWrap ? (tab: ReactElement) => cfg.tabWrap!(key, tab) : undefined }, diff --git a/apps/desktop/src/app/chat/session-drag.ts b/apps/desktop/src/app/chat/session-drag.ts index 8c2c5fbe2b..5aebe63323 100644 --- a/apps/desktop/src/app/chat/session-drag.ts +++ b/apps/desktop/src/app/chat/session-drag.ts @@ -30,7 +30,6 @@ import type { PointerEvent as ReactPointerEvent } from 'react' import { queryAllVisible } from '@/components/pane-shell/pane-visibility' import { findGroup } from '@/components/pane-shell/tree/model' import { - type DoubleTapContext, rectContains, slotBefore, snapshotStrips, @@ -95,15 +94,15 @@ function tileZoneHost(groupId: string): { chat: boolean; pane: string } | null { /** * Begin dragging a session — a sidebar row OR a tile's own tab (same drop * language either way: stack, split, or composer link). Sub-threshold releases - * stay ordinary clicks, so `opts.onTap` (activate the tile) and `opts.double` - * (hide the tab bar) ride the tab's gestures; Esc aborts instantly. A stack/ - * split commits through `openSessionTile`, which OPENS a new tile from a sidebar - * row and MOVES the existing one when its tab is the drag source. + * stay ordinary clicks, so `opts.onTap` (activate the tile) rides the tab's + * gesture; Esc aborts instantly. A stack/split commits through + * `openSessionTile`, which OPENS a new tile from a sidebar row and MOVES the + * existing one when its tab is the drag source. */ export function startSessionDrag( payload: SessionDragPayload, e: ReactPointerEvent, - opts?: { double?: DoubleTapContext; onTap?: () => void } + opts?: { onTap?: () => void } ) { let zones: EngineZone[] = [] let strips: StripSnapshot[] = [] @@ -124,7 +123,6 @@ export function startSessionDrag( const restoreOpacity = source?.style.opacity ?? '' startDragSession(e, { - double: opts?.double, ghost: { label: sessionLabel(payload) }, onTap: opts?.onTap, diff --git a/apps/desktop/src/app/chat/session-tile.tsx b/apps/desktop/src/app/chat/session-tile.tsx index 88432e9b54..fda0fd3ef8 100644 --- a/apps/desktop/src/app/chat/session-tile.tsx +++ b/apps/desktop/src/app/chat/session-tile.tsx @@ -27,7 +27,7 @@ import { ModelMenuPanel } from '@/app/shell/model-menu-panel' import { formatRefValue } from '@/components/assistant-ui/directive-text' import { CenteredThreadSpinner } from '@/components/assistant-ui/thread/status' import { findGroupOfPane } from '@/components/pane-shell/tree/model' -import { $layoutTree, closeTreePane, moveTreePane, setTreeGroupHeaderHidden } from '@/components/pane-shell/tree/store' +import { $layoutTree, closeTreePane, moveTreePane, setTreeGroupTabStrip } from '@/components/pane-shell/tree/store' import { Button } from '@/components/ui/button' import { ConfirmDialog } from '@/components/ui/confirm-dialog' import { transcribeAudio } from '@/hermes' @@ -560,7 +560,7 @@ export function WorkspaceTabMenu({ children }: { children: React.ReactElement }) const group = tree ? findGroupOfPane(tree, 'workspace') : null if (group) { - setTreeGroupHeaderHidden(group.id, true) + setTreeGroupTabStrip(group.id, 'never') } } @@ -616,9 +616,9 @@ export const watchSessionTiles = paneMirror({ ), // A tile's tab drags like a sidebar row — stack / split / drop-to-link — with - // its tap (activate) + double-tap (hide bar) preserved. Always takes the drag. - tabDrag: (storedSessionId, event, onTap, double) => { - startSessionDrag(tileDragPayload(storedSessionId), event, { double, onTap }) + // its tap (activate) preserved. Always takes the drag. + tabDrag: (storedSessionId, event, onTap) => { + startSessionDrag(tileDragPayload(storedSessionId), event, { onTap }) return true }, diff --git a/apps/desktop/src/app/context-menu/app-context-menu.tsx b/apps/desktop/src/app/context-menu/app-context-menu.tsx index 17ca30fa38..afe4ce62ab 100644 --- a/apps/desktop/src/app/context-menu/app-context-menu.tsx +++ b/apps/desktop/src/app/context-menu/app-context-menu.tsx @@ -4,6 +4,7 @@ import { useEffect } from 'react' import { useNavigate } from 'react-router' import { terminalMenuHandleFor } from '@/app/right-sidebar/terminal/terminal-context-menu' +import { toggleTargetZoneTabStrip } from '@/components/pane-shell/tree/store' import { Codicon } from '@/components/ui/codicon' import { HERMES_CONTEXT_MENU_TRIGGER_ATTR } from '@/components/ui/context-menu' import { writeClipboardText } from '@/components/ui/copy-button' @@ -566,6 +567,15 @@ function shellSections({ navigate, t }: ShellVerbs): ReactNode[][] { label={t.keybinds.actions['view.toggleStatusbar']} onSelect={toggleStatusbarVisible} />, + // The pointer-only way back to a hidden tab strip: right-clicking the + // shell reaches this menu from anywhere, including a zone that has no + // chrome left to right-click. + void toggleTargetZoneTabStrip()} + />, { // The main tab drags like a session tile — drop it on a composer to link the // chat, on a zone/edge to stack/split. Defers (`false`) to the generic pane // move when there's no loaded session to carry. -const workspaceTabDrag = (event: ReactPointerEvent, onTap: () => void, double?: DoubleTapContext) => { +const workspaceTabDrag = (event: ReactPointerEvent, onTap: () => void) => { const payload = workspaceDragPayload() if (!payload) { return false } - startSessionDrag(payload, event, { double, onTap }) + startSessionDrag(payload, event, { onTap }) return true } @@ -163,7 +164,6 @@ registry.registerMany([ collapsible: true, dock: { pane: 'workspace', pos: 'left' }, revealAliases: ['chat-sidebar'], - showCloseButton: false, // Standing chrome: no close gestures at all — the tab is shown/hidden // (zone menu Show/Hide rows + the auto-registered ⌘K toggle below). hideOnly: true, @@ -324,6 +324,18 @@ registry.registerMany([ get: () => $statusbarVisible.get(), set: enabled => $statusbarVisible.set(enabled) }), + paletteToggle({ + id: 'view.toggleTabStrip', + label: 'Toggle tabs', + action: 'view.toggleTabStrip', + icon: PanelTop, + keywords: ['tab strip', 'tab bar', 'tabs', 'header', 'zone', 'hide', 'show', 'chrome'], + // On-screen truth for the zone the verbs target, not a stored flag: a zone + // on auto has no stored value, and the row must read as "what pressing + // this does to what I can see". + get: () => Boolean(targetZoneTabStripVisible()), + set: () => void toggleTargetZoneTabStrip() + }), // The keybind panel's non-titlebar door (the keyboard icon is gone). { id: 'keybinds.panel', diff --git a/apps/desktop/src/app/contrib/surfaces.tsx b/apps/desktop/src/app/contrib/surfaces.tsx index d2108a751f..5353035c09 100644 --- a/apps/desktop/src/app/contrib/surfaces.tsx +++ b/apps/desktop/src/app/contrib/surfaces.tsx @@ -147,11 +147,13 @@ export const ChatRoutesSurface = memo(function ChatRoutesSurface({ /> ) - // FULL-PAGE views (not chat) mark the zone body `data-zone-no-header`: a - // page is not a tab-able surface, so the zone's double-click header toggle - // stands down while one is showing (see onZoneDoubleClick). + // FULL-PAGE views (not chat): a page is not a tab-able surface, so the zone's + // tab strip stands down while one is showing. That is `paneChrome.headerVeto` + // on the contribution, not a DOM marker — the `data-zone-no-header` attribute + // that used to ride this wrapper gated a body double-click toggle that no + // longer exists, and nothing has read it since. const page = (view: ReactNode) => ( -
+
{view}
) diff --git a/apps/desktop/src/app/contrib/wiring.tsx b/apps/desktop/src/app/contrib/wiring.tsx index 1b418400ea..67eb63c822 100644 --- a/apps/desktop/src/app/contrib/wiring.tsx +++ b/apps/desktop/src/app/contrib/wiring.tsx @@ -25,6 +25,7 @@ import { DesktopOnboardingOverlay } from '@/components/onboarding' import { $newSessionTabAction, registerPaneCloser } from '@/components/pane-shell/tree/store' import { FloatingPet } from '@/components/pet/floating-pet' import { RemoteDisplayBanner } from '@/components/remote-display-banner' +import { SendDiagnosticsHost } from '@/components/send-diagnostics-dialog' import { emitGatewayEvent } from '@/contrib/events' import { getLatestSessionMessages } from '@/hermes' import { type ChatMessage, chatMessageText, preserveLocalAssistantErrors, toChatMessages } from '@/lib/chat-messages' @@ -1181,6 +1182,10 @@ export function ContribWiring({ children }: { children: ReactNode }) { {/* Backs confirm() from @/store/confirm — renders only while one is open. */} + {/* Send Diagnostics consent/upload dialog — driven by $sendDiagnostics + (error card action); renders nothing until requested. */} + + {/* Petdex floating mascot — renders nothing unless installed + enabled. Never in the HUD: that window is the chat bar and nothing else. */} {!isHudWindow() && } diff --git a/apps/desktop/src/app/hooks/use-keybinds.ts b/apps/desktop/src/app/hooks/use-keybinds.ts index caa0f9d7bc..b4a525acf4 100644 --- a/apps/desktop/src/app/hooks/use-keybinds.ts +++ b/apps/desktop/src/app/hooks/use-keybinds.ts @@ -11,7 +11,8 @@ import { cycleTreeTabInFocusedZone, isPaneVisible, layoutHasRootSide, - togglePaneVisible + togglePaneVisible, + toggleTargetZoneTabStrip } from '@/components/pane-shell/tree/store' import { onReleaseTypingFocus } from '@/components/ui/keyboard-first' import { findBarClaimsCombo } from '@/lib/find-in-page' @@ -245,6 +246,7 @@ export function useKeybinds(deps: KeybindRuntimeDeps): void { layoutHasRootSide('right') ? toggleFileBrowserOpen() : togglePaneVisible('terminal'), 'view.toggleReview': toggleReview, 'view.toggleStatusbar': toggleStatusbarVisible, + 'view.toggleTabStrip': () => void toggleTargetZoneTabStrip(), 'view.showFiles': showFiles, 'view.showBrowser': openBrowserTab, 'view.toggleHud': () => toggleHud(hudTargetSessionId()), diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts index 5efba27561..43be63e190 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts @@ -4,6 +4,7 @@ import { burstVibeHearts } from '@/components/chat/vibe-hearts' import { translateNow } from '@/i18n' import { coerceGatewayText, coerceThinkingText } from '@/lib/chat-runtime' import { playCompletionSound } from '@/lib/completion-sound' +import { parseErrorSurface } from '@/lib/error-surface' import { triggerHaptic } from '@/lib/haptics' import { billingCtaLabel, clearBillingBlock, runBillingRecovery, setBillingBlock } from '@/store/billing-block' import { clearClarifyRequest } from '@/store/clarify' @@ -335,13 +336,15 @@ export function handleMessageStreamEvent(ctx: GatewayEventContext): boolean { const finalText = coerceGatewayText(payload?.text) || coerceGatewayText(payload?.rendered) // Terminal error frames (status "error") carry the failure in - // structured fields: `error` is the message, and `partial` marks - // `text` as streamed output to keep rather than the error string. + // structured fields: `error` is the message, `partial` marks + // `text` as streamed output to keep rather than the error string, and + // `error_surface` (newer gateways) names the failing layer for the card. const failure = payload?.status === 'error' ? { error: coerceGatewayText(payload.error).trim() || finalText || 'Hermes reported an error', - partial: Boolean(payload.partial) + partial: Boolean(payload.partial), + surface: parseErrorSurface(payload.error_surface) } : undefined diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts index b7eeae2a17..36cec7c71d 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts @@ -17,6 +17,7 @@ import { sealOpenToolParts, upsertToolPart } from '@/lib/chat-messages' +import type { ErrorSurface } from '@/lib/error-surface' import { dedupeGeneratedImageEchoesInParts, generatedImageEchoSources, @@ -561,7 +562,7 @@ export function useMessageStream({ sessionId: string, text: string, responsePreviewed?: boolean, - failure?: { error: string; partial: boolean }, + failure?: { error: string; partial: boolean; surface?: ErrorSurface | null }, occurredAt = Date.now() / 1000 ) => { let shouldHydrate = false @@ -616,7 +617,8 @@ export function useMessageStream({ parts: completeOpenTimelineParts(message.parts, occurredAt), pending: false, interim: false, - ...(durationS !== undefined ? { durationS } : {}) + ...(durationS !== undefined ? { durationS } : {}), + ...(completionError && failure?.surface ? { errorSurface: failure.surface } : {}) } if (completionError && !keepFailedPartialText) { @@ -641,7 +643,8 @@ export function useMessageStream({ completedAt: occurredAt, branchGroupId: state.pendingBranchGroup ?? undefined, ...(durationS !== undefined ? { durationS } : {}), - ...(completionError && { error: completionError }) + ...(completionError && { error: completionError }), + ...(completionError && failure?.surface ? { errorSurface: failure.surface } : {}) }) const prev = state.messages diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/terminal-error-frame.test.tsx b/apps/desktop/src/app/session/hooks/use-message-stream/terminal-error-frame.test.tsx index 303cf10a2e..8ecefe8e09 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/terminal-error-frame.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-message-stream/terminal-error-frame.test.tsx @@ -78,4 +78,38 @@ describe('terminal error message.complete frames', () => { const bubble = lastAssistant() expect(bubble?.error).toBe('Error: something broke') }) + + it('attaches the structured error_surface descriptor to the failed bubble', async () => { + mountStream() + await start() + await delta('…') + + await completeWithError({ + text: 'Error: rate limited', + error: 'rate limited', + error_surface: { layer: 'provider', code: 'rate_limit', retryable: true }, + recoverable: true + }) + + const bubble = lastAssistant() + expect(bubble?.error).toBe('rate limited') + expect(bubble?.errorSurface).toEqual({ layer: 'provider', code: 'rate_limit', retryable: true }) + }) + + it('ignores a garbled error_surface payload (older/foreign backends)', async () => { + mountStream() + await start() + await delta('…') + + await completeWithError({ + text: 'Error: kaput', + error: 'kaput', + error_surface: { layer: 'not-a-layer', code: 42 }, + recoverable: true + }) + + const bubble = lastAssistant() + expect(bubble?.error).toBe('kaput') + expect(bubble?.errorSurface).toBeUndefined() + }) }) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts index 6474b93431..79c02e844c 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts @@ -3,6 +3,7 @@ import { getSession } from '@/hermes' import { assistantTextPart, type ChatMessage, chatMessageText, textPart } from '@/lib/chat-messages' import { normalizePersonalityValue } from '@/lib/chat-runtime' import { embeddedImageUrls, textWithoutEmbeddedImages } from '@/lib/embedded-images' +import { parseErrorSurface } from '@/lib/error-surface' import { reconcileApprovalModeForProfile } from '@/store/approval-mode' import { requestDesktopOnboardingForCredentialWarning } from '@/store/onboarding' import { $activeGatewayProfile, $profiles, normalizeProfileKey } from '@/store/profile' @@ -146,6 +147,9 @@ const COMPARED_FIELDS = [ 'role', 'pending', 'error', + // Structured failure layer — drives the error card's title and action row, + // so a change (e.g. resume replay attaching the descriptor) must repaint. + 'errorSurface', 'hidden', 'branchGroupId', 'interim', @@ -254,6 +258,11 @@ export function chatMessagesEquivalent(a: ChatMessage, b: ChatMessage): boolean a.role !== b.role || a.pending !== b.pending || a.error !== b.error || + // Structural compare — the descriptor arrives as a fresh object per + // resume/replay, so identity comparison would repaint forever. + (a.errorSurface?.layer ?? null) !== (b.errorSurface?.layer ?? null) || + (a.errorSurface?.code ?? null) !== (b.errorSurface?.code ?? null) || + (a.errorSurface?.retryable ?? null) !== (b.errorSurface?.retryable ?? null) || a.hidden !== b.hidden || a.branchGroupId !== b.branchGroupId || a.timestamp !== b.timestamp || @@ -742,6 +751,7 @@ export function appendLiveSessionProjection(messages: ChatMessage[], projection: // the terminal frame may have been lost to a disconnect) — surface the // failure on the projected row instead of rendering the partial as healthy. const inflightError = projection.inflight?.error?.trim() ?? '' + const inflightErrorSurface = parseErrorSurface(projection.inflight?.error_surface) const queuedUser = projection.queued?.user?.trim() ?? '' if ( @@ -904,7 +914,8 @@ export function appendLiveSessionProjection(messages: ChatMessage[], projection: role: 'assistant', parts: inflightAssistant ? [assistantTextPart(inflightAssistant)] : [], pending: inflightStreaming, - ...(inflightError ? { error: inflightError } : {}) + ...(inflightError ? { error: inflightError } : {}), + ...(inflightError && inflightErrorSurface ? { errorSurface: inflightErrorSurface } : {}) }) } diff --git a/apps/desktop/src/app/settings/appearance-settings.tsx b/apps/desktop/src/app/settings/appearance-settings.tsx index b677ae069f..bd31f043fd 100644 --- a/apps/desktop/src/app/settings/appearance-settings.tsx +++ b/apps/desktop/src/app/settings/appearance-settings.tsx @@ -21,6 +21,7 @@ import { $activeGatewayProfile, $profiles, normalizeProfileKey } from '@/store/p import { $reactionsEnabled, setReactionsEnabled } from '@/store/reactions-enabled' import { $reasoningCollapsedByDefault, setReasoningCollapsedByDefault } from '@/store/reasoning-disclosure' import { $sessionListDensity, type SessionListDensity, setSessionListDensity } from '@/store/session-list-density' +import { $tabStripDefault, setTabStripDefault, type TabStripDefault } from '@/store/tabstrip-prefs' import { $toolViewMode, setToolViewMode } from '@/store/tool-view' import { $translucency, @@ -345,6 +346,7 @@ export function AppearanceSettings() { const toolViewMode = useStore($toolViewMode) const reasoningCollapsedByDefault = useStore($reasoningCollapsedByDefault) const sessionListDensity = useStore($sessionListDensity) + const tabStripDefault = useStore($tabStripDefault) const zoomPercent = useStore($zoomPercent) const embedMode = useStore($embedMode) const embedAllowed = useStore($embedAllowed) @@ -423,6 +425,12 @@ export function AppearanceSettings() { { id: 'detailed', label: a.sessionDensityDetailed } ] as const satisfies readonly { id: SessionListDensity; label: string }[] + const tabStripOptions = [ + { id: 'auto', label: a.tabStripAuto }, + { id: 'always', label: a.tabStripAlways }, + { id: 'never', label: a.tabStripNever } + ] as const satisfies readonly { id: TabStripDefault; label: string }[] + const embedOptions = [ { id: 'ask', label: a.embedsAsk }, { id: 'always', label: a.embedsAlways }, @@ -583,6 +591,21 @@ export function AppearanceSettings() { title={a.sessionDensityTitle} /> + { + triggerHaptic('selection') + setTabStripDefault(id) + }} + options={tabStripOptions} + value={tabStripDefault} + /> + } + description={a.tabStripDesc} + title={a.tabStripTitle} + /> + {/* Linux has neither half of this setting (see TRANSLUCENCY_SUPPORTED), so the row is absent there rather than offering a dead lever. */} {TRANSLUCENCY_SUPPORTED && ( diff --git a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx index f2cfdc09c1..31e3b5790f 100644 --- a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx @@ -8,8 +8,10 @@ import { } from '@assistant-ui/react' import { useStore } from '@nanostores/react' import { type FC, type ReactNode, useCallback, useMemo, useState } from 'react' +import { useInRouterContext, useNavigate } from 'react-router' import { useSessionView } from '@/app/chat/session-view' +import { SETTINGS_ROUTE } from '@/app/routes' import { ChangedFilesCard } from '@/components/assistant-ui/thread/changed-files-card' import { contentHasVisibleText, @@ -28,14 +30,26 @@ import { PreviewAttachment } from '@/components/chat/preview-attachment' import { Codicon } from '@/components/ui/codicon' import { CopyButton } from '@/components/ui/copy-button' import { useI18n } from '@/i18n' +import { type ErrorSurface, formatErrorDiagnostics } from '@/lib/error-surface' import { triggerHaptic } from '@/lib/haptics' -import { AudioLines, GitForkIcon, Loader2Icon, RefreshCwIcon, SmilePlusIcon, VolumeXIcon, XIcon } from '@/lib/icons' +import { + AudioLines, + GitForkIcon, + Loader2Icon, + RefreshCwIcon, + SmilePlusIcon, + Upload, + VolumeXIcon, + XIcon +} from '@/lib/icons' import { extractPreviewTargets } from '@/lib/preview-targets' import { markAssistantIdSpoken } from '@/lib/spoken-reply' import { useEnterAnimation } from '@/lib/use-enter-animation' import { cn } from '@/lib/utils' import { playSpeechText, stopVoicePlayback } from '@/lib/voice-playback' import { notifyError } from '@/store/notifications' +import { requestSendDiagnostics } from '@/store/send-diagnostics' +import { $connection, $currentModel } from '@/store/session' import { $voicePlayback } from '@/store/voice-playback' // Stable empty identity for the settled-parts selector — a fresh [] per render @@ -224,20 +238,26 @@ const AssistantMessageBody: FC - - {onDismissError && ( - onDismissError(messageId)} - side="top" - tooltip={t.assistant.thread.dismissError} - > - - - )} +
+
+ + +
+ {onDismissError && ( + onDismissError(messageId)} + side="top" + tooltip={t.assistant.thread.dismissError} + > + + + )} +
+
@@ -433,6 +453,131 @@ const StreamingMarker: FC = () => { ) } +// ── Layered error card pieces ──────────────────────────────────────────── +// +// The gateway stamps failed turns with a structured {layer, code, retryable} +// descriptor (metadata.custom.errorSurface — see agent/error_surface.py). +// These leaves render the layer label + recovery actions. Older backends +// never send the descriptor: the label falls back to a generic title and the +// action row still offers Retry / Open Logs / Copy error details, so nothing +// regresses on version skew. + +const ErrorLayerLabel: FC = () => { + const { t } = useI18n() + const surface = useAuiState(s => s.message.metadata?.custom?.errorSurface as ErrorSurface | undefined) + + const labels = t.assistant.thread.errorLayers + const label = (surface && labels[surface.layer]) || labels.generic + + return
{label}
+} + +// Isolated because useNavigate() THROWS outside a (bare test +// harnesses, embedded panes render threads router-free). The parent gates +// this child's mount on useInRouterContext(), which is safe anywhere. +const SwitchProviderAction: FC<{ label: string }> = ({ label }) => { + const navigate = useNavigate() + + return ( + + ) +} + +const ErrorRecoveryActions: FC = () => { + const { t } = useI18n() + const copy = t.assistant.thread + const surface = useAuiState(s => s.message.metadata?.custom?.errorSurface as ErrorSurface | undefined) + + const errorText = useAuiState(s => { + const status = s.message.status as { error?: unknown; type?: string } | undefined + + return status?.type === 'incomplete' && typeof status.error === 'string' ? status.error : '' + }) + + // useNavigate() would throw here when no Router is above us; the deep-link + // child mounts only when one is (see SwitchProviderAction). + const inRouter = useInRouterContext() + const model = useStore($currentModel) + const connection = useStore($connection) + + // Open Logs reveals the LOCAL Electron profile's HERMES_HOME/logs. On a + // remote/cloud connection the failed turn's gateway+agent logs live on the + // remote box — the local folder only holds Desktop-side transport logs, so + // the label says "Open Desktop logs" there instead of implying it opens the + // runtime's logs. + const remoteConnection = connection?.mode === 'remote' + + // Retry = assistant-ui reload (same wiring as the footer's refresh action): + // re-runs the failed turn's prompt in place. Suppressed when the classifier + // says the failure is deterministic (retrying reproduces it). + const retryable = !surface || surface.retryable + + // Switch Provider deep-links Settings → Models for the layers where the fix + // is provider/endpoint/auth config, not a retry. + const showSwitchProvider = surface != null && ['auth', 'billing', 'endpoint', 'provider'].includes(surface.layer) + + const openLogs = useCallback(async () => { + try { + const root = await window.hermesDesktop?.logsRoot?.() + + if (!root) { + notifyError(new Error('logs root unavailable'), copy.errorOpenLogsFailed) + + return + } + + const result = await window.hermesDesktop?.openDir?.(root) + + if (result && !result.ok) { + notifyError(new Error(result.error || 'open failed'), copy.errorOpenLogsFailed) + } + } catch (error) { + notifyError(error, copy.errorOpenLogsFailed) + } + }, [copy.errorOpenLogsFailed]) + + const diagnosticsText = useCallback( + () => + formatErrorDiagnostics({ + errorText, + model: model || undefined, + surface + }), + [errorText, model, surface] + ) + + return ( +
+ {retryable && ( + + + + )} + {showSwitchProvider && inRouter && } + {window.hermesDesktop?.logsRoot && ( + + )} + + +
+ ) +} + const AssistantActionBar: FC = ({ messageId, getMessageText, onBranchInNewChat }) => { const { t } = useI18n() const copy = t.assistant.thread diff --git a/apps/desktop/src/components/boot-failure-overlay.test.tsx b/apps/desktop/src/components/boot-failure-overlay.test.tsx index e86f987a63..db03d6d046 100644 --- a/apps/desktop/src/components/boot-failure-overlay.test.tsx +++ b/apps/desktop/src/components/boot-failure-overlay.test.tsx @@ -98,4 +98,41 @@ describe('BootFailureOverlay', () => { restore() } }) + + it('shows the Nous Cloud down recovery when the backend flags isCloudBackendDown', async () => { + const restore = stubDesktop(remoteToken) + $desktopBoot.set({ + error: 'Nous Cloud agent ares-3009.agents.nousresearch.com is down (HTTP 503: server-side fault).', + fakeMode: false, + isCloudBackendDown: true, + message: 'boot failed', + phase: 'renderer.error', + progress: 40, + running: false, + statusCode: 503, + timestamp: Date.now(), + visible: true + }) + + try { + render() + // Cloud-specific title + actionable recovery instead of the generic + // remote-failure copy. + expect(await screen.findByText(/Nous Cloud agent is down/i)).toBeTruthy() + // Portal and Discord are dedicated action buttons (localized labels + // can't drift the URLs, which live in code). + expect(screen.getByRole('button', { name: /check portal status/i })).toBeTruthy() + expect(screen.getByRole('button', { name: /get help on discord/i })).toBeTruthy() + // Cloud-down is a remote failure: local-only Repair is dropped; the + // actionable paths are Gateway settings + Use local gateway. + expect(screen.queryByRole('button', { name: /repair/i })).toBeNull() + expect(screen.getByRole('button', { name: /gateway settings/i })).toBeTruthy() + expect(screen.getByRole('button', { name: /use local gateway/i })).toBeTruthy() + // The electron-built error message (portal / local mode / Discord) is + // still surfaced in the error box. + expect(screen.getByText(/ares-3009\.agents\.nousresearch\.com/i)).toBeTruthy() + } finally { + restore() + } + }) }) diff --git a/apps/desktop/src/components/boot-failure-overlay.tsx b/apps/desktop/src/components/boot-failure-overlay.tsx index f07366f289..2d71eca10c 100644 --- a/apps/desktop/src/components/boot-failure-overlay.tsx +++ b/apps/desktop/src/components/boot-failure-overlay.tsx @@ -7,7 +7,8 @@ import { Loader } from '@/components/ui/loader' import { LogView } from '@/components/ui/log-view' import type { DesktopConnectionConfig } from '@/global' import { useI18n } from '@/i18n' -import { ChevronLeft, FileText, Loader2, LogIn, RefreshCw, SlidersHorizontal, Wrench } from '@/lib/icons' +import { openExternalLink } from '@/lib/external-link' +import { ChevronLeft, ExternalLink, FileText, Loader2, LogIn, RefreshCw, SlidersHorizontal, Wrench } from '@/lib/icons' import { $desktopBoot } from '@/store/boot' import { notify, notifyError } from '@/store/notifications' import { $desktopOnboarding } from '@/store/onboarding' @@ -247,6 +248,11 @@ export function BootFailureOverlay() { let actions: RecoveryAction[] let hint: string + // The electron boot path flags a Nous Cloud backend-down (502/503/504) with + // the structured isCloudBackendDown/statusCode it carries through boot + // progress. When set, the recovery screen leads with the cloud-specific + // guidance instead of the generic remote-failure copy (#85335). + const cloudDown = Boolean(boot.isCloudBackendDown) if (remoteReauth) { actions = [ @@ -261,6 +267,31 @@ export function BootFailureOverlay() { localAction ] hint = copy.remoteSignInHint(label) + } else if (cloudDown) { + // A Nous Cloud agent is down — the user cannot restart the managed + // instance and Repair is local-only. Lead with the paths that actually + // resolve it: check the portal (status/instance controls), switch to the + // local gateway, retry, or get support on Discord. Portal/Discord are + // buttons (not URLs buried in the hint prose) so localized hints can't + // drift the links. + actions = [ + { + key: 'portal', + label: copy.cloudDownCheckPortal, + onClick: () => openExternalLink('https://portal.nousresearch.com'), + icon: + }, + localAction, + { ...retryAction, variant: 'secondary' }, + { + key: 'discord', + label: copy.cloudDownDiscord, + onClick: () => openExternalLink('https://discord.gg/NousResearch'), + variant: 'ghost' + }, + { ...settingsAction, variant: 'ghost' } + ] + hint = copy.cloudDownHint } else if (remoteFailure) { actions = [settingsAction, { ...retryAction, variant: 'secondary' }, localAction] hint = copy.remoteFailureHint @@ -284,7 +315,12 @@ export function BootFailureOverlay() { if (view === 'connect') { return ( -
+
{/* Subtle back affordance (projects/overlay idiom): muted → foreground on hover, no divider. */} @@ -307,16 +343,21 @@ export function BootFailureOverlay() { } return ( -
+

- {remoteReauth ? copy.remoteTitle : copy.title} + {remoteReauth ? copy.remoteTitle : cloudDown ? copy.cloudDownTitle : copy.title}

- {remoteReauth ? copy.remoteDescription : copy.description} + {remoteReauth ? copy.remoteDescription : cloudDown ? copy.cloudDownDescription : copy.description}

diff --git a/apps/desktop/src/components/error-boundary.tsx b/apps/desktop/src/components/error-boundary.tsx index 6ec1e4ca20..2d6593b629 100644 --- a/apps/desktop/src/components/error-boundary.tsx +++ b/apps/desktop/src/components/error-boundary.tsx @@ -143,7 +143,12 @@ function RootErrorFallback({ error, reset }: ErrorBoundaryFallbackProps) { const { t } = useI18n() return ( -
+
- {copy.free} + + {typeof price.discount_percent === 'number' ? ( + + -{price.discount_percent}% + + ) : null} + + {copy.free} + ) } diff --git a/apps/desktop/src/components/onboarding/index.tsx b/apps/desktop/src/components/onboarding/index.tsx index e75d128000..521a1a229c 100644 --- a/apps/desktop/src/components/onboarding/index.tsx +++ b/apps/desktop/src/components/onboarding/index.tsx @@ -308,6 +308,10 @@ export function DesktopOnboardingOverlay({ bare && leaving ? '[transition-delay:660ms]' : '', leaving ? 'pointer-events-none opacity-0' : 'opacity-100' )} + // Masks the whole app until onboarding finishes — must stay filled under + // window glass or the shell shows through. Contract: + // `[data-glass-opaque]` in styles.css. + data-glass-opaque="" >