Merge origin/main into feat/hermes-relay-shared-metrics

Signed-off-by: Alex Fournier <afournier@nvidia.com>
This commit is contained in:
Alex Fournier
2026-07-22 08:55:25 -07:00
480 changed files with 65880 additions and 3400 deletions
+42 -1
View File
@@ -414,6 +414,25 @@ compression:
# compaction doesn't fire with half the window still free; set above 0.75 to override.
threshold: 0.50
# Per-model threshold overrides: keys are substring-matched against the model
# name (longest match wins). Useful when some models need different compaction
# points — e.g. a 1M-context model can compress later (0.30) while a 128K
# model needs to compress earlier (0.60). The small-context floor (75% for
# <512K models) still applies on top of per-model overrides.
# model_thresholds:
# "glm-5.2": 0.40
# "claude-sonnet": 0.35
# "gpt-5": 0.30
# Optional absolute token cap for the compression trigger (default: null = disabled).
# When set, compression fires at the LOWER of the ratio-based threshold and this
# absolute token count — first-fires-wins. It never fires later than this count
# regardless of which model is active (useful when switching between models with
# very different context windows). Clamped to the model's context length at
# apply-time, so a cap above the window is a no-op (ratio-based threshold wins).
# Survives model switches and fallback activations.
# threshold_tokens: 200000
# Existing Codex gpt-5.5 behavior: raise Hermes' compaction trigger to 85%
# for the ChatGPT Codex OAuth route. Set false to opt back down to threshold.
codex_gpt55_autoraise: true
@@ -429,6 +448,12 @@ compression:
# compression of older turns.
protect_last_n: 20
# Compression retry rounds before a turn gives up with "max compression
# attempts reached" (default: 3, same as the previous hardcoded value).
# Raise (e.g. 6) for tool-schema-heavy sessions where 3 rounds cannot bring
# the request estimate under the threshold. Validated >= 1, hard cap 10.
max_attempts: 3
# Codex app-server (codex CLI runtime) thread-compaction mode. The codex
# agent owns the real thread context on this runtime, so Hermes' summarizer
# cannot shrink it — compaction goes through the app server instead.
@@ -1500,7 +1525,10 @@ updates:
# access_token_env: BWS_ACCESS_TOKEN # bootstrap token, sourced from .env
# project_id: "" # UUID of the BSM project to sync
# server_url: "" # "" = US Cloud; EU/self-hosted URL otherwise
# cache_ttl_seconds: 300 # 0 disables caching
# cache_ttl_seconds: 300 # 0 disables fresh caching
# encrypted_cache: # optional encrypted stale fallback
# enabled: false
# max_stale_seconds: 0 # 0 disables stale fallback
# override_existing: true # BSM values win over existing env
# auto_install: true # lazy-download bws into ~/.hermes/bin
#
@@ -1517,3 +1545,16 @@ updates:
# binary_path: "" # "" = resolve op via PATH; else absolute path
# cache_ttl_seconds: 300 # 0 disables BOTH cache layers
# override_existing: true # resolved values win over existing env
#
# # ---- Command helper (any CLI vault) --------------------------------------
# # Run a user-configured helper that prints KEY=VALUE lines on stdout —
# # works with any secret store that has a CLI: keepassxc-cli, secret-tool,
# # pass, gpg, or a script that cats a tmpfs env file. Composes with the
# # sources above (enable any combination). POSIX-only (needs /bin/sh).
# # The helper must be fast and NON-interactive (hard timeout, 1 MiB cap);
# # its stderr is discarded so diagnostics can't leak secret material.
# command:
# enabled: false
# command: "cat /run/user/1000/hermes-secrets.env"
# helper_timeout_seconds: 3
# override_existing: false # .env/shell win by default