"""Default configuration data for Hermes Agent: DEFAULT_CONFIG and OPTIONAL_ENV_VARS. Pure-data leaf module — must not import from hermes_cli.config. Comments are the user-facing docs of config.yaml. """ def _aux(timeout, *, reasoning_effort=True, **extra): """Standard auxiliary-task model block (see DEFAULT_CONFIG["auxiliary"]). reasoning_effort=False omits that key (MoA blocks configure depth per slot); ``extra`` keys are appended after the standard ones. """ d = {"provider": "auto", "model": "", "base_url": "", "api_key": "", "timeout": timeout, "extra_body": {}} if reasoning_effort: d["reasoning_effort"] = "" d.update(extra) return d DEFAULT_CONFIG = { "model": "", "providers": {}, "fallback_providers": [], "credential_pool_strategies": {}, "toolsets": ["hermes-cli"], # journal_mode: SQLite journal mode for every Hermes DB. "wal" default; use "delete" on # weak-fsync/shared filesystems where WAL is not crash-safe (macOS virtiofs, NFS, SMB). "database": { "journal_mode": "wal", # WAL sizing pragmas (ints). None = SQLite defaults (autocheckpoint 1000 pages, no limit). "wal_autocheckpoint": None, "journal_size_limit": None, }, # Soft fd limit for long-running server processes; clamped to OS hard limit. 0/false/null = off. "runtime": {"nofile_soft_limit": 4096}, # Global active chat session cap across CLI, TUI/dashboard, and messaging. None/0 = unbounded. "max_concurrent_sessions": None, # Soft LRU cap on in-memory TUI/desktop/dashboard sessions. Above it the gateway evicts the # least-recently-active DETACHED sessions (no live client); reopening re-resumes from disk. # 0/null disables. "max_live_sessions": 16, "session": { # Per-terminal `hermes -c`: each CLI session writes a breadcrumb under # $HERMES_HOME/terminal-sessions/, so bare -c/--continue resumes THIS # terminal's session (tmux/kitty/wezterm pane, tty). false = resume globally most-recent. "terminal_continue": True, }, "agent": { # Turn cap. null = unlimited (default; caps caused silent mid-task truncation). Positive int # caps; "none"/"unlimited"/"inf"/0/-1 also mean unlimited (resolve_turn_limit). "max_turns": None, # Optional one-time model-visible checkpoint warning before a finite turn cap is exhausted. # null = off; set a ratio strictly between 0 and 1 (for example, 0.75). "budget_warning_ratio": None, # Wall-clock budget (seconds) per run. null = off. When set: one-time wrap-up notice at 80% # elapsed; implicit provider stale timeouts capped to remaining budget. CLI equivalent: # `hermes chat --run-budget N`. "run_budget_seconds": None, # Gateway inactivity timeout (seconds). Only fires when the agent is completely idle — not # while calling tools or receiving API responses. 0 = unlimited. "gateway_timeout": 1800, # Max seconds an alias routing key waits for the active turn holding the same session lease; # on expiry the message is rejected with a resend notice. Keep short: Telegram dispatches # sequentially, so a waiter delays unrelated topics. Non-positive -> 5s. "gateway_turn_lease_timeout": 5, # Per-session AIAgent cache in the gateway. Each entry keeps a warm prompt prefix AND the # full transcript: too small re-pays uncached prompts, too large fills the heap. "agent_cache": { "max_size": 128, # LRU entry cap "idle_ttl_secs": 3600, # evict agents idle this long # Anonymous-RSS budget (MB) above which LRU transcripts are shed (reloaded from disk # next turn). "auto" = derive from cgroup memory limit (or total RAM); number = # explicit; 0/off = disable the pass. "memory_high_mb": "auto", # Max sessions shed per pass (teardown bursts can't stall the gateway) and the number of # most-recently-used sessions the pass never touches. "max_evictions_per_pass": 16, "protect_recent": 8, }, # Force-interrupt budget (seconds) once gateway stop()/drain has begun (SIGTERM, and the # final phase of in-band restart). 0 = interrupt immediately. Keep under systemd # TimeoutStopSec or risk SIGKILL mid-cleanup; for /restart prefer restart_after_turn_timeout # so turns finish BEFORE stop(). "restart_drain_timeout": 0, # Cron-only floor under the stop()/drain wait (seconds). Interrupted chat turns resume on # the next message, but an interrupted cron run is recorded as a permanent failure, so it # must not inherit restart_drain_timeout's 0. Clamped to the shutdown-watchdog leash minus # teardown headroom (~50s unless TimeoutStopSec is raised). 0 = opt out. # A chat turn interrupted by a restart is announced to the user and resumed on their next message; # an interrupted cron run is written to jobs.json as a permanent failure that nobody is waiting on, # so it must not inherit restart_drain_timeout's 0 (#82161). "cron_drain_timeout": 30, # In-band restart (/restart, SIGUSR1): refuse new work, then wait up to this many seconds # for in-flight agents/cron/api runs to finish before stop(). 0 = enter stop() at once. 30 # min is a safety valve for wedged agents, not a target; raise for long unattended turns. # Default 30 min is a safety valve for wedged agents, not a target latency — an interactive `hermes # gateway restart` must never block for hours on a turn that wedged (#79133). "restart_after_turn_timeout": 1800, # Max seconds a submitted prompt waits for the deferred agent build (MCP discovery, model # metadata, skills scan) before failing visibly. The prompt is delivered as soon as the # build completes (progress notice past 30s), so this only fires on a hung build. Raise for # many slow/unreachable MCP servers. # See #63078. "build_wait_timeout": 600, # Hermes-level retry attempts for API errors (connection drops, timeouts, 5xx) wrapping the # whole call; the OpenAI SDK also retries transient errors (max_retries=2). Set 1 for fast # failover to fallback providers; raise to tolerate longer provider hiccups. "api_max_retries": 3, # Empty-response retry guard. Empty retries re-send the full input at full price; this stops # re-billing deterministic empties (unsignaled refusals, zero output tokens) while failing # open on ambiguous evidence (missing usage, any tokens, model/provider change). "empty_response_guard": { "enabled": True, # False = legacy fixed 3 retries unconditionally # When one empty attempt's estimated input cost >= this USD, the streak's retry budget # drops from 3 to 1. Unknown pricing / missing usage leaves it untouched. "cost_threshold_usd": 0.25, }, # Fast mode: "" / "normal" (off), "fast" (always), "auto" (first fast_auto_seconds of every # turn), "cold" (first turn of a session only). "service_tier": "", "fast_auto_seconds": 60, # System-prompt guidance telling the model to call tools instead of describing actions. # "auto" = gpt/codex models; true/false = force for all models; or a list of model-name # substrings (e.g. ["gpt", "codex", "gemini", "qwen"]). "tool_use_enforcement": "auto", # Execution-discipline prompt block (tool persistence, tools for arithmetic/system facts, # read-back after external writes, count reconciliation, literal identifiers, # verification-gated completion). Chosen once per session by model name (byte-stable). # "auto" = gpt/codex/grok/deepseek/kimi/qwen/glm/minimax/mimo/mistral; true/false = force; # or a list of model-name substrings. "execution_guidance": "auto", # When the model narrates an action ("I'll go check the logs...") but emits no tool call, # inject a "continue now, execute the tools" nudge and loop (max 2 nudges/turn). Corrective # sibling of tool_use_enforcement. "auto" = codex_responses api_mode only; true = all # api_modes (fixes Gemini/Claude "stops after stating intent"); false = never; or a list of # model-name substrings. "intent_ack_continuation": "auto", # Anti-stall guards: (1) identical-call loop breaker appends a notice when the same tool is # called 3+ times with identical args AND results (never blocks; pollers like `process` # exempt); (2) continue-intent extension of empty-response recovery re-prompts once when the # model says it will continue but takes no action. False disables both. "stall_guards": True, # "Finish the job" prompt block for all models: don't stop at a stub, never fabricate output # when the real path is blocked. ~80 cached tokens. False disables. "task_completion_guidance": True, # Prompt block for all models steering independent tool calls (reads, searches, fetches, # read-only commands) into one batched turn; the runtime already runs them concurrently. ~70 # cached tokens. False disables. "parallel_tool_call_guidance": True, # Toolchain probe: surfaces Python/pip/uv/PEP-668 state in the system prompt only when # something non-default is detected (no pip module, pip/python mismatch, PEP 668 without # uv); zero tokens when clean. Skipped for docker/modal/ssh backends (own probe). "environment_probe": True, # Bot Mode teammate-messaging protocol section (silent unless desktop Bot Mode manages it). "bot_mode_protocol": True, # Embedder-supplied text appended to the system prompt's environment-hints block, so a host # wrapping Hermes (sandbox runner, managed platform) can describe proxy/credential/ mount # layout without editing SOUL.md. Env HERMES_ENVIRONMENT_HINT overrides it. "environment_hint": "", # Coding posture: on interactive coding surfaces (CLI, TUI, desktop, ACP) in a code # workspace, add a coding brief + live git/workspace snapshot to the system prompt # (agent/coding_context.py). "auto" = prompt-only when interactive AND cwd is a code # workspace (toolsets untouched, messaging platforms unaffected); "focus" = auto + collapse # toolset to the lean coding set (+ enabled MCP servers) + demote non-coding skill # categories to names-only (explicit opt-in); "on" = force everywhere; "off" = disable. "coding_context": "auto", # Standing operator instructions (string or list) appended to the coding brief as an extra # stable system block — project-wide workflow rules, e.g. "Don't run tsc/lint until I # approve." Cache-safe: takes effect next session. "coding_instructions": "", # When verify-on-stop finds edits without fresh verification evidence, add guidance for # creative UI work (no broad tsc/lint/test before visual approval) and clean-diff # expectations. false = keep the evidence nudge terse. "verify_guidance": True, # Max consecutive `pre_verify` "continue" nudges per turn (hooks can't trap the loop). "max_verify_nudges": 3, # Verification closure: after code edits in a workspace, refuse a final answer until fresh # verification evidence exists or the agent explains why it can't check (bounded loop, # passive ledger). False (default) because the nudges proved more noise than signal; true = # force on everywhere; "auto" = on for interactive coding surfaces and programmatic callers, # off for messaging surfaces. Doc/markdown/skill-only edits never fire. "verify_on_stop": False, # Inactivity warning (seconds), once per run before gateway_timeout; no interrupt. 0 = off. "gateway_timeout_warning": 900, # Max seconds the gateway blocks an agent awaiting a clarify-tool reply; then it unblocks # with "[user did not respond within Xm]". CLI clarify blocks indefinitely and ignores this. # 1h because users step away and a shorter value evicted the entry mid-think so a later # button tap hit a dead entry. Lower it to free the running-agent guard sooner. # Maximum time (seconds) the gateway will block an agent waiting for a clarify-tool response from # the user. Tradeoff: a higher value holds the gateway's running-agent guard longer for a genuinely # abandoned prompt — lower it if a single session must free up the guard sooner. See #32762. "clarify_timeout": 3600, # "Still working" status interval (seconds); 0 = off. Lower = faster feedback, more noise; # 180 catches spinning weak-model runs before users /restart. "gateway_notify_interval": 180, # Session stall watchdog (seconds): RECOVERY notifier for an in-process AIAgent with an # adapter-queued follow-up while its activity clock is stale — NOT a general stall detector # (ignores startup restore, build sentinels, leases, debounce, other processes; scan cadence # per AIAgent). Notify-only: tells the user to try /new. Distinct from gateway_timeout # (kills the turn) and gateway_notify_interval. 0 = disable. # See #76354. "session_stall_timeout": 300, # Transcript-sanitiser heal escalation: after this many pre-send heal passes within a # 10-minute window, log one ERROR and queue a ONE-TIME out-of-band notice pointing at /debug # share or `hermes doctor` (status channel only; prompt cache untouched). 0 = no escalation # (per-window WARNINGs still fire). # See #96870. "sanitizer_heal_escalation_threshold": 3, # Seconds of continuous reconnect failure before a platform gets needs_attention flagged in # gateway status (`hermes status` / fleet monitoring). Retries never stop — a signal, not a # circuit breaker. 0 = disable. "reconnect_attention_after": 7200, # Freshness window (seconds) for the auto-continue note. After a crash/restart mid-run the # next user message gets "[System note: your previous turn was interrupted...]" prepended; # only when the last persisted transcript row is younger than this, so stale markers don't # revive an unrelated old task. Covers gateway_timeout (1800) plus slack. 0 = always inject. "gateway_auto_continue_freshness": 3600, # Max seconds the gateway waits for boot auto-resume turns before releasing the # startup-restore inbound gate (all inbound is QUEUED while shut, so one long resumed turn # would leave every channel unanswered). On timeout the gate opens and the resume keeps # running in the background; duplicate-agent protection is unaffected because the resume # slot is claimed synchronously first. 0 = wait forever. "gateway_startup_restore_drain_timeout": 30, # Max seconds the boot turn-machinery warm-up (run_agent import graph, tool schemas + # availability probes, context-file tier) may hold the inbound gate shut, so an early # message isn't served with a skeleton system prompt. On timeout the gate opens and warm-up # finishes in the background. 0 = disable warm-up (lazy init). "gateway_startup_warmup_timeout": 20, # Stale-stream ceiling (seconds) for local providers (Ollama, oMLX, llama-cpp). Applied when # the base stale timeout is at its 180s default and a local endpoint is detected, so a # wedged local server eventually trips the detector instead of hanging forever. Env # HERMES_LOCAL_STREAM_STALE_TIMEOUT overrides. "local_stream_stale_timeout": 900, # How user-attached images reach the main model (gateway, TUI, CLI /attach). "auto" = native # when the model reports supports_vision=True AND auxiliary.vision.provider is not # explicitly set, else text; "native" = always attach (non-vision models error at the # provider or get a last-chance text fallback); "text" = always pre-analyze with # vision_analyze and prepend the description. vision_analyze stays a tool regardless. "image_input_mode": "auto", "disabled_toolsets": [], # Model name (any reasonable spelling) -> effort level; overrides agent.reasoning_effort # when the current model matches. Edit in config.yaml (no CLI support: dots in keys). "reasoning_overrides": {}, # Preserve assistant `reasoning_content` on history replay. Echo families (DeepSeek, # Kimi/Moonshot, Xiaomi MiMo) are auto-detected by provider name/base-URL host; custom # providers and OpenAI-compatible gateways proxying them are not. Set `reasoning_echo: true` # on a `model:` entry or a `fallback_providers:` entry to opt in per provider. Default # false: strict providers (Mistral, Groq, Cerebras) reject the field. "reasoning_echo": False, # Turn liveness watchdog: a turn with no observable progress for `timeout_s` seconds is # logged, force-interrupted so the UI can retry, and its lease stops renewing so stale-turn # cleanup can reclaim the session even if the interrupt can't unwind a wedged frame. # timeout_s <= 0 disables; poll_s = sampling interval. Invalid values (NaN, Inf, # non-positive poll) warn and fall back to defaults. See agent/turn_liveness.py. "turn_liveness": {"timeout_s": 600.0, "poll_s": 15.0}, }, "terminal": { "backend": "local", "modal_mode": "auto", # Remote-backend connection-class failures (SSH host unreachable, Docker daemon down): # "warn" = structured degraded tool result with reason + retry hint; "fail" = raise error + # traceback. "degraded_mode": "warn", "cwd": ".", # Use current directory # Root for terminal session temp files (background logs/pid/exit files, code-exec # sandboxes). Empty = TMPDIR/TMP/TEMP if set, else HERMES_HOME/cache/terminal (auto-pruned # after 72h) — NOT tmpfs /tmp, which is RAM-capped and fills under load. Must be an existing # absolute POSIX path; user-set paths are never auto-pruned. "temp_dir": "", # CSS font-family for the desktop app's xterm.js terminal (e.g. "'CaskaydiaCoveNerdFont', # monospace"). Empty = built-in default ("'JetBrains Mono', 'Cascadia Code', 'SF Mono', # Menlo, Consolas, monospace"). Lets users use a Nerd Font without patching the app. "font_family": "", "timeout": 180, # Seconds between SIGTERM and escalated SIGKILL for host process trees (browser daemons). 0 # = SIGTERM only. "daemon_term_grace_seconds": 2.0, # Max seconds a one-shot CLI run (-q/-Q/-z) lingers for tracked notify_on_complete # background processes to finish. The dying parent owns their stdout pipes, so exiting # immediately kills the delivery (e.g. Bot Mode handoff replies via message_agent / # bot_relay). Plain background processes without notify_on_complete are never waited on. 0 # disables. # Bounded linger (seconds) for one-shot CLI runs (-q/-Q/-z) that exit while background processes # spawned with notify_on_complete=true are still running. See #90879. "oneshot_completion_wait_seconds": 600.0, # Env vars passed into sandboxed terminal/execute_code (skill-declared # required_environment_variables pass through automatically). "env_passthrough": [], # HOME for host tool subprocesses: "auto" = host keeps the real OS-user HOME, containers use # HERMES_HOME/home; "real" = force real HOME; "profile" = force HERMES_HOME/home when it # exists (strict per-profile isolation). "home_mode": "auto", # Extra files sourced in the login shell when building the per-session env snapshot — for # nvm/pyenv/asdf/PATH entries registered by files a bash login shell skips (~/.bashrc, # ~/.zshrc, ~/.zprofile). Supports ~ and ${VAR}; missing files skipped. When empty and the # shell is bash, ~/.profile, ~/.bash_profile, ~/.bashrc are auto-sourced in that order (see # auto_source_bashrc). "shell_init_files": [], # Source ~/.profile, ~/.bash_profile, ~/.bashrc in the snapshot login shell to capture PATH # additions, functions, and aliases that `bash -l -c` misses (bash skips bashrc when # non-interactive; Debian/Ubuntu ~/.bashrc short-circuits). ~/.profile and ~/.bash_profile # go first because n/nvm/asdf write PATH exports there without an interactivity guard. Turn # off if an rc file misbehaves when sourced non-interactively (exits on TTY check). "auto_source_bashrc": True, "docker_image": "nikolaik/python-nodejs:python3.11-nodejs20", "docker_forward_env": [], # Exact key-value env pairs set inside Docker containers (unlike docker_forward_env, which # reads host values) — useful under systemd without the user's shell env. Example: # {"SSH_AUTH_SOCK": "/run/user/1000/ssh-agent.sock"} "docker_env": {}, "singularity_image": "docker://nikolaik/python-nodejs:python3.11-nodejs20", "modal_image": "nikolaik/python-nodejs:python3.11-nodejs20", "daytona_image": "nikolaik/python-nodejs:python3.11-nodejs20", "vercel_runtime": "node24", # vercel_sandbox backend only: node24 | node22 | python3.13 # Container limits (docker, singularity, modal, daytona, vercel_sandbox; not local/ssh). "container_cpu": 1, "container_memory": 5120, # MB (default 5GB) "container_disk": 51200, # MB (default 50GB) "container_persistent": True, # Persist filesystem across sessions # Docker volume mounts, "host_path:container_path" (docker -v syntax), e.g. # ["/home/user/.hermes/cache/documents:/output"]. For gateway MEDIA delivery, write to # /output/... inside Docker and emit the host-visible path in MEDIA:, not the container one. "docker_volumes": [], "docker_mount_cwd_to_workspace": False, # mount host cwd at /workspace (weakens isolation) "docker_network": True, # false = --network=none, no network access from commands "docker_extra_args": [], # Extra flags passed verbatim to docker run # /dev/shm size for the Docker sandbox. Docker's 64 MB default silently breaks # Chromium/Playwright and PyTorch DataLoader workers; tmpfs is lazily allocated so the # higher ceiling is free until used. "" or "0" = omit the flag (Docker default). "docker_shm_size": "1g", # Run the container as the host uid:gid (`--user`) so files written to bind mounts # (docker_volumes, persistent workspace, mounted cwd) are owned by you, not root. Off by # default for images whose entrypoints must start as root (e.g. the bundled Hermes image, # which drops to `hermes` via s6-setuidgid). When on, SETUID/SETGID caps are omitted. "docker_run_as_host_user": False, # Snap-packaged Docker under AppArmor (Ubuntu cloud images; LP#1908448) refuses to exec # anything under `--init` or `--security-opt no-new-privileges` ("operation not # permitted"). True drops those two flags; every other hardening stays. See #9730. "docker_snap_compat": False, # Trusted profiles sharing one Docker container identity; empty = per-profile boundary. "docker_shared_container_key": "", # Keep a long-lived bash shell across execute() calls so cwd/env/shell variables survive. # Applies to non-local backends (SSH); local is opt-in via TERMINAL_LOCAL_PERSISTENT env. "persistent_shell": True, }, "web": { "backend": "", # shared fallback — applies to both search and extract "search_backend": "", # per-capability override for web_search (e.g. "searxng") "extract_backend": "", # per-capability override for web_extract (e.g. "native") # per-page char budget for web_extract; larger pages truncate, full text kept in cache/web "extract_char_limit": 15000, # Keyless free-tier ring: with NO web backend configured or keyed, web_search/web_extract # rotate round-robin across exa, parallel, firecrawl, keenable public free tiers, failing # over on rate limits. Never pre-empts a configured/keyed backend. false = disable. "keyless_fallback": True, # One-shot rescue: when the chosen/keyed backend fails a call, THAT call retries once on the # keyless ring; the next call tries the chosen backend again (no sticky failover). Off when # keyless_fallback is false. "keyless_rescue": True, # Per-vendor tier for vendors with both a keyless free endpoint and a keyed paid path (exa, # parallel, firecrawl, keenable; tavily is opt-in keyless via `hermes tools`, not a ring # member). Set by the `hermes tools` picker. "free" = always anonymous endpoint even with a # key; "paid" = always keyed (missing key = error; vendor excluded from the ring); unset = # keyed when the key is present, else the ring. "provider_tier": {}, # TTL caching for web_search + web_extract: repeat searches (same query + provider) within # the TTL come from an in-process memo; repeat extracts from the cache/web store. Concurrent # identical searches coalesce into one vendor request. Only successes cached. "cache_enabled": True, "cache_ttl_minutes": 20, # Hosts always fetched live, never from the extract cache (staging deploys, tunnel URLs, # preview builds). Entries match exactly, as "*.wildcard", or as a domain suffix # ("mysite.dev" also covers "preview.mysite.dev"). localhost/private IPs always exempt. "cache_exempt_hosts": [], }, "browser": { # "" = Browser Use mode when the browser-use CLI (or uvx) is available, else built-in tools # (Camofox setups always keep built-in tools: no CDP surface); "browser-use" = force one # browser_exec tool driving the Browser Use CLI over any CDP backend (local Chrome, cloud); # "off" = force the built-in browser_navigate/browser_click/... tools. "backend": "", "inactivity_timeout": 120, "command_timeout": 30, # seconds per browser command (screenshot, navigate, etc.) "snapshot_threshold": 15000, # max chars before snapshot truncate-and-store (min 1000) "record_sessions": False, # auto-record browser sessions as WebM videos # headed: visible Chromium window (local); skips per-turn cleanup, idle reaper still applies "headed": False, "allow_private_urls": False, # allow private/internal IPs (localhost, 192.168.x.x, ...) # Local browser engine for both drivers. "auto" = Chrome; "lightpanda" = faster navigation, # no screenshots (Browser Use mode spawns `lightpanda serve` per session; built-in tools # pass `--engine ` to agent-browser with Chrome fallback); "chrome" = explicit. # Ignored while a cloud provider, Camofox, cdp_url or use_real_profile is active. Also # settable via AGENT_BROWSER_ENGINE. "engine": "auto", # With a cloud provider, auto-spawn local Chromium for LAN/localhost URLs instead "auto_local_for_private_urls": True, "cdp_url": "", # persistent CDP endpoint for attaching to an existing Chromium/Chrome # Consent to browse with the user's REAL logins locally: runs on a Hermes-managed SNAPSHOT # of the ACTIVE default-Chromium profile (Local State -> profile.last_used; cookies, logins, # prefs copied and re-synced per fresh session) driven by Hermes' packaged Chromium. The # snapshot dir sidesteps Chrome 136+'s default-profile debugging block and never contends # with the running browser. Turning off deletes ~/.hermes/browser-profile/ so credentials # don't outlive consent. Chromium-family only (Chrome, Edge, Brave, Brave Origin, Chromium); # Firefox etc. fails closed. Also gates the browser_exec `local` argument (real-profile # local session even under a cloud backend). Desktop Settings -> Browser. "use_real_profile": False, # Windows only: a running Chrome/Edge/Brave locks its cookie DB, so the profile can't be # copied. When on, a locked profile still blocks and the agent ASKS first; on approval it # runs `hermes browser close-profile` (kills that profile's browser tree, unsaved tabs lost) # and retries once; still locked -> stays blocked, no auto-kill. No effect on macOS/Linux # (copy-while-running works). "real_profile_autoclose": False, # Pin WHICH source profile directory is snapshotted for real-profile browsing (e.g. "Profile # 2"). Empty = browser's last-used profile, which on multi-profile machines can hand the # agent the wrong identity. A pin naming a missing directory FAILS CLOSED. "real_profile_pin": "", # restrict_evaluate: opt-in denylist blocking sensitive JS primitives (cookies/storage/ # clipboard/network/form values) in browser_console(expression=...); allow_unsafe_evaluate # is the legacy override that bypasses that denylist entirely. "allow_unsafe_evaluate": False, "restrict_evaluate": False, # CDP supervisor: dialog + frame detection over a persistent WebSocket; active only with a # CDP-capable backend (Browserbase, or local Chrome via /browser connect). See # website/docs/developer-guide/browser-supervisor.md. "dialog_policy": "must_respond", # must_respond | auto_dismiss | auto_accept "dialog_timeout_s": 300, # safety auto-dismiss after N seconds under must_respond "camofox": { # true = send a stable profile-scoped userId so Camofox maps it to a persistent Firefox # profile; false = random ephemeral userId per session. "managed_persistence": False, # Externally managed Camofox identity, for when another app owns the visible browser. "user_id": "", "session_key": "", "adopt_existing_tab": False, # rehydrate tab_id from Camofox before creating a tab # Docker Camofox opens page URLs from inside the container: rewrite loopback page URLs # (localhost/127.0.0.1/::1) to the host alias; CAMOFOX_URL itself is unchanged. "rewrite_loopback_urls": False, "loopback_host_alias": "host.docker.internal", }, # Authenticated browser-extension controller lane: a registered extension can become the # exact controller for a session's browser_* tools (fail-closed once bound). Local API # registration also requires the API server bearer key. developer_mode gates the privileged # browser_cdp / browser_evaluate capabilities. "extension_control": {"enabled": False, "developer_mode": False}, }, # Filesystem checkpoints: snapshot the working directory once per turn (on the first # write_file/patch call); restore with /rollback. Opt-in via `hermes chat --checkpoints` or # enabled=True (most users never use /rollback). Single shared shadow store with real pruning. "checkpoints": { "enabled": False, # Max checkpoints per working directory; enforced by ref rewrite + GC of older commits. "max_snapshots": 20, # Hard ceiling on total ~/.hermes/checkpoints/ size (MB); the oldest checkpoint per project # is dropped round-robin until under the cap. 0 disables. "max_total_size_mb": 500, # Skip files larger than this (MB) when staging (datasets, model weights). 0 = no filter. "max_file_size_mb": 10, # Startup sweep (at most once per min_interval_hours): deletes projects whose last_touch is # older than retention_days, GCs the shared store, enforces max_total_size_mb, deletes # legacy-* archives older than retention_days. It NEVER deletes orphans (workdir missing on # disk) — a missing workdir may just be an unmounted volume/VPN, and an unattended sweep # must not guess. Orphans: `hermes checkpoints prune` (`--keep-orphans` to skip). "auto_prune": True, "retention_days": 7, "min_interval_hours": 24, }, # Hard cap (chars) for one auto-loaded context file (SOUL.md, AGENTS.md, CLAUDE.md, .hermes.md, # .cursorrules) before head/tail truncation. null = scale with the model's context window (floor # 20K, ceiling 500K); a positive int pins a fixed cap. Separate from read_file limits. "context_file_max_chars": None, # Seconds to wait for a single context file read before skipping it with a warning. Guards startup # against network-backed filesystems (iCloud Drive, OneDrive, NFS) that can block a cold read. "context_file_read_timeout": 5.0, # Max chars per read_file call; larger reads are rejected with offset+limit guidance. 100K chars # ≈ 25–35K tokens. "file_read_max_chars": 100_000, # Seconds the first agent build waits for background MCP discovery before snapshotting its tool # list. Returns the instant discovery completes (no MCP servers → ~0s); the bound only bites # when a server is still connecting. Turn-1 latency knob only: a server that misses it is picked # up by the between-turns refresh (agent/turn_context.py), so keep it small — a dead server adds # this much to first-response latency. "mcp_discovery_timeout": 1.5, # Same bound for single-query mode (``hermes -q/-z``). With only ONE turn there is no # between-turns refresh, so a server that misses the window is invisible for the whole session; # the larger bound lets slow cold-start servers (npx, uvx, remote HTTP) land. Reachable servers # still only wait their real handshake time. "mcp_single_query_discovery_timeout": 15.0, "mcp": { # MCP runtime behavior (distinct from mcp_servers: definitions and auxiliary.mcp). # Auto-reload MCP connections when config.yaml's mcp_servers changes (CLI watcher). Every # reload rebuilds the tool surface and INVALIDATES the provider prompt cache (next message # re-sends the full prefix) — costly on long-context models. When false the watcher still # detects the change and prints /reload-mcp guidance. "auto_reload_on_config_change": True, }, # Tool-output truncation. max_bytes: terminal_tool output cap in chars (head+tail kept; 50_000 ≈ # 12-15K tokens). max_lines: max `limit` one read_file call may request before clamping. # max_line_length: per-line cap in read_file's line-numbered view (chars). "tool_output": {"max_bytes": 50000, "max_lines": 2000, "max_line_length": 2000}, # Tool loop guardrails nudge models that repeat failed/non-progressing tool calls. Soft warnings # are always on; hard stops are opt-in so interactive sessions keep flowing. "tool_loop_guardrails": { "warnings_enabled": True, "hard_stop_enabled": False, # Unattended gateway/cron platforms hard-stop by default (nobody can /stop a model that # ignores warnings); interactive cli/tui/desktop/acp stay warning-only. "non_interactive_hard_stop_enabled": True, "warn_after": {"exact_failure": 2, "same_tool_failure": 3, "idempotent_no_progress": 2}, "hard_stop_after": { "exact_failure": 5, "same_tool_failure": 8, "idempotent_no_progress": 5 }, # Per-turn hard ceilings for runaway-prone tools; counters reset every turn, always on # regardless of the thresholds above. Dozens of searches/subagents in ONE turn is already # pathological, hence low defaults. 0 = unlimited. "loop_caps": { "max_web_searches": 50, # web_search calls per turn "max_subagents": 50, # subagents spawned per turn }, }, "compression": { "enabled": True, # checkpoint_required: fail closed before lossy compaction unless an active memory provider # confirms checkpoint API compatibility and completes the checkpoint. "checkpoint_required": False, # progress_notices: when True, routine compression progress statuses (compacting/ # preflight/pre-API/idle/retry) reach chat gateways instead of being filtered as noise. # Failure notices and manual /compress feedback are always visible. "progress_notices": False, # threshold: compress when context usage exceeds this ratio. Models with windows below 512K # are floored at 0.75 (raise-only) so compaction doesn't fire with half the window free; set # above 0.75 to override the floor. "threshold": 0.50, # threshold_tokens: absolute token cap — compression triggers at the lower of the ratio # threshold and this count. Clamped to the model's context length. "threshold_tokens": None, # "progress_notices": False, # opt-in (#52995): when True, routine compression "target_ratio": 0.20, # fraction of threshold to preserve as recent tail # tail_mode: "lean" = clamped 2.5%-of-window tail (10K floor / 25K cap) plus chunked # digests, anchor index, verbatim user messages and session_search pointers in the summary # (~3x fewer retained tokens; a few extra summarizer calls at the boundary). "legacy" = # 0.20×threshold verbatim tail (100-240K tokens on big windows). "tail_mode": "lean", "protect_last_n": 20, # minimum recent messages kept uncompressed # min_tail_user_messages: REAL (actionable) user messages guaranteed to survive in the tail. # 1 = single last-user anchor; raise (e.g. 3) when bulky tool outputs fill the tail budget. "min_tail_user_messages": 1, # max_attempts: retry rounds before a turn gives up with "max compression attempts reached". # Raise (e.g. 6) for tool-schema-heavy sessions. Validated >= 1, cap 10. "max_attempts": 3, # proactive_prune_tokens: opt-in trigger (tokens) for the deterministic no-LLM tool-result # prune, independent of `threshold` (which rarely fires on large windows, so old tool output # is re-sent every turn); e.g. 48000 reclaims early. 0 = off. Tail protected by # `protect_last_n`. Built-in compressor only. Each committed prune rewrites sent history and # breaks the prompt-cache prefix — the min_reclaim gate below keeps those breaks episodic. "proactive_prune_tokens": 0, # Prune's summarize pass only touches tool results larger than this (chars); clamped >= 200 # so a generated summary can't be re-summarized. "proactive_prune_min_result_chars": 8000, # A prune only commits when it reclaims at least this many tokens, then waits for a # trigger-sized runway to regrow before rearming. 0 = no minimum-savings gate. "proactive_prune_min_reclaim_tokens": 4096, # micro_compact: opt-in — after each turn fold the oldest un-absorbed exchange into a # rolling summary, amortizing compression cost. Off by default because every pass rewrites # sent history and breaks the prompt-cache prefix EVERY turn; enable only if the amortized # stall beats the cached-prefix discount. See website/docs/developer-guide/micro-compaction.md. "micro_compact": False, # Cadence: run a pass every Nth completed turn (1 = one cache break per turn, 5 = a fifth of # the breaks). Clamped >= 1; ignored unless micro_compact is true. "micro_compact_every_n_turns": 1, # Once the rolling summary exceeds this many tokens, the next pass re-summarizes it. "micro_compact_defrag_threshold_tokens": 2000, # Gateway session-hygiene force-compress threshold, by message count. "hygiene_hard_message_limit": 5000, # Max seconds the gateway waits for pre-agent hygiene compression WITHOUT forward progress. # Inactivity budget: a slow model still streaming tokens extends the wait. "hygiene_timeout_seconds": 30, # Absolute cap on the hygiene wait even while tokens are moving (bounds a trickle stream). # Clamped >= hygiene_timeout_seconds. "hygiene_total_ceiling_seconds": 600, "hygiene_failure_cooldown_seconds": 300, # skip repeated failed hygiene attempts # Max seconds an ARRIVING user turn is held while a streaming hygiene summary finishes; # bounds user-visible latency (keep under chat idle timeouts, Telegram ~30s). On expiry the # turn proceeds uncompressed; the detached worker keeps its watermark-fenced commit, so the # summary is adopted at the next safe boundary. "hygiene_max_turn_hold_seconds": 10, # Inactivity budget for in-agent compress_context (loop, /compress, preflight); same # progress-aware semantics as hygiene_timeout_seconds. 0 = disable the owned wrapper # (callers passing commit_fence, e.g. gateway hygiene, never use it). "context_timeout_seconds": 120, # Absolute cap on the *pre-commit* compress_context wait (summary/stream phase) even while # tokens move. Clamped >= context_timeout_seconds when that is > 0. A started SessionDB # commit is never abandoned: past the ceiling it is logged (WARNING, then ERROR) and # surfaced on the warning channel while the host keeps waiting. "context_total_ceiling_seconds": 600, # Non-system head messages always kept verbatim, in ADDITION to the (always protected) # system prompt. 0 = pin nothing but system prompt + summary + tail. "protect_first_n": 3, # When True, auto-compression whose summary fails (aux error / non-JSON / timeout) aborts # instead of dropping the middle with a "summary unavailable" placeholder; the session # freezes at its size until /compress (bypasses the cooldown) or /new. "abort_on_summary_failure": False, # (Historical key name.) When True, gpt-5.4/5.5/5.6 and gpt-6 Astra (any slug containing # "astra" without "900k") on the ChatGPT Codex OAuth route raise their compaction trigger to # 85%: Codex hard-caps them at a 272K window, so the global 50% would compact at ~136K. False = global `threshold`. Only that route; the same models via # OpenAI direct, OpenRouter or Copilot keep the global value. "codex_gpt55_autoraise": True, # Show the one-time autoraise banner; False keeps the autoraise, hides the notice. "codex_gpt55_autoraise_notice": True, # Codex app-server thread compaction mode. The codex agent owns the thread context, so # Hermes' summarizer cannot shrink it. native = codex decides; hermes = Hermes' threshold # triggers thread/compact/start; off = never auto-trigger. "codex_app_server_auto": "native", # Opt in to OpenAI server-side compaction on the Responses API. Only gpt-5.6-family on # api.openai.com or the Codex backend; local compression stays as fallback. "codex_responses_native": False, # Absolute server compaction trigger (input tokens). None follows the local trigger with a # safety margin; explicit values only clamp downward so the server goes first. "codex_responses_compact_threshold": None, # in_place: compaction rewrites the message list and system prompt WITHOUT rotating the # session id (no parent_session_id chain, no `name #N` renumbering), avoiding the # session-rotation bug cluster. Pre-compaction turns are soft-archived under the same id # (active=0, compacted=1) — still session_search-able. False = legacy rotating-compaction # path. "in_place": True, # Per-model threshold overrides: keys substring-match the model name (longest wins), values # replace the global `threshold`, e.g. {"glm-5.2": 0.40}. Prefix a key with ":" to # scope it to one route ({"openai-codex:astra": 0.85} leaves Astra on OpenRouter/Nous at the # global value). The <512K floor (0.75) still applies raise-only on top. "model_thresholds": {}, # Opt-in idle compaction (0 = off): a session resuming after this many idle seconds compacts # up front, before the first reply. Time-based complement to `threshold`; skipped when # already at/below threshold × target_ratio; honors the same cooldown/ anti-thrash/lock # guards. Example: 1800 = 30 min. "idle_compact_after_seconds": 0, }, # Anthropic prompt caching (Claude via OpenRouter or native API). cache_ttl: "5m" | "1h"; other # non-falsy values are ignored; falsy (false, null, "off", "disabled", "no", "none") disables # caching. "prompt_caching": {"cache_ttl": "5m"}, # OpenRouter settings. response_cache: X-OpenRouter-Cache header — identical requests return # cached responses at zero billing; independent of Anthropic prompt caching. response_cache_ttl: # seconds (1-86400), only used when response_cache is on. min_coding_score (0.0-1.0): # pareto-code router knob, applied only when model.model is "openrouter/pareto-code"; higher = # stronger/pricier coders, 0.65 = mid-tier, "" = let OpenRouter pick the strongest. Docs: # openrouter.ai/docs/guides/routing/routers/pareto-router "openrouter": {"response_cache": True, "response_cache_ttl": 300, "min_coding_score": 0.65}, "bedrock": { # AWS Bedrock; only used when model.provider is "bedrock". "region": "", # empty = AWS_REGION env var → us-east-1 "discovery": { "enabled": True, # auto-discover models via ListFoundationModels "provider_filter": [], # restrict to these providers, e.g. ["anthropic", "amazon"] "refresh_interval": 3600, # cache discovery results (seconds) }, # Bedrock Guardrails: create one in the console, then set ID and version. # https://docs.aws.amazon.com/bedrock/latest/userguide/guardrails.html "guardrail": { "guardrail_identifier": "", # e.g. "abc123def456" "guardrail_version": "", # e.g. "1" or "DRAFT" "stream_processing_mode": "async", # "sync" | "async" "trace": "disabled", # "enabled" | "disabled" | "enabled_full" }, }, # Auxiliary model config — provider/model per side task. provider "auto" = auto-detect; # empty model = provider's default aux model; all tasks fall back to # openrouter:google/gemini-3-flash-preview when the configured provider is unavailable. # extra_body is forwarded verbatim as request body fields for that task, e.g. OpenRouter # routing prefs / Pareto Code floor: # auxiliary: # compression: # extra_body: # provider: {order: [anthropic, google], sort: throughput} # or price | latency # plugins: [{id: pareto-router, min_coding_score: 0.5}] # Each task is independent — main-agent provider_routing and openrouter.min_coding_score # do NOT propagate to aux calls by design. "auxiliary": { # Same-provider retries for a transient blip (reset/timeout/5xx/408) on ANY aux call before # falling back; clamped [0,6]. Matters for pinned calls (MoA advisors) where provider # fallback is not meaningful recovery. "transient_retries": 2, # When true, the auto-chain's OpenRouter step is skipped unless the fallback model ends in # ":free" — a PAID lane is never used for background aux traffic even with # OPENROUTER_API_KEY set. "free_only": False, # Override the auto-chain's OpenRouter fallback model (default google/gemini-3.6-flash, # PAID). Pair e.g. "nvidia/nemotron-3-ultra-550b-a55b:free" with free_only: true. A one-time # WARNING is logged whenever a non-":free" model is engaged. "openrouter_model": "", # Endpoints that reject NON-streaming chat (HTTP 400): aux calls are sent with stream=True # and aggregated. Case-insensitive URL substrings; copilot.tencent.com is always # stream-only. "stream_only_base_urls": [], # Per-task blocks share one shape (_aux): provider "auto" = inherit the main model; base_url # overrides provider; api_key falls back to OPENAI_API_KEY; reasoning_effort: # none|minimal|low|medium|high|xhigh|max|ultra ("" = provider default); extra_body = # OpenAI-compatible request fields. Vision: download_timeout = image HTTP download (s). "vision": _aux(120, download_timeout=30), # web_extract and session_search no longer use an aux LLM; leftover blocks in user config # are ignored. Compression: raise timeout for local models. "compression": _aux(120), "skills_hub": _aux(30), "approval": _aux(30), # classifier — a fast/cheap model is recommended # /review reviewer: a full subagent on the async delegation rail, credentials resolved like # delegation.provider pins. "auto" + "" = main agent's model. api_mode forces transport: # chat_completions | anthropic_messages | codex_responses. "review": {"provider": "auto", "model": "", "base_url": "", "api_key": "", "api_mode": ""}, "mcp": _aux(30), # prefer_fast_model opts in to the provider fast tier; auto otherwise = main model. "title_generation": { "enabled": True, # Note: session_search no longer uses an auxiliary LLM (PR #27590 — single-shape tool returns DB # content directly). The old ``auxiliary.session_search.*`` block was removed here. Existing # values in user config.yaml files are harmless leftovers and ignored. "provider": "auto", "model": "", "prefer_fast_model": False, "base_url": "", "api_key": "", "timeout": 30, "extra_body": {}, "reasoning_effort": "", "language": "", }, "memory_query_rewrite": _aux(8, reasoning_effort=False), "tts_audio_tags": _aux(30), # Kanban: triage_specifier expands a Triage one-liner into a spec (cheap model OK); # kanban_decomposer emits a JSON graph of child tasks (more tokens). "triage_specifier": _aux(120), "kanban_decomposer": _aux(180), "profile_describer": _aux(60), # 1-2 sentence profile blurb; short, cheap "goal_judge": _aux(60), # /goal satisfaction + contract drafting; JSON calls # Curator skill-usage review can take minutes on reasoning models (umbrellas over hundreds # of skills); route cheaper via `hermes model` → auxiliary → Curator. "curator": _aux(600), "monitor": _aux(60), # important-mail 0-10 scorer; high-volume, small model fine # Post-turn self-improvement fork (save memory / patch skill). "auto" = main model replaying # the full conversation (warm cache); other models replay a compact digest (~3-5x cheaper). # enabled=false skips auto spawns (/refine still works). max_input_tokens caps the SUM of # replayed input tokens over the review loop (iterations capped at 16); the loop stops # before crossing it. <= 0 = unlimited. "background_review": {"enabled": True, **_aux(120), "max_input_tokens": 600000}, # No reasoning_effort on MoA blocks by design — configured PER SLOT in the preset # (moa.presets..reference_models[].reasoning_effort / aggregator.reasoning_effort). "moa_reference": _aux(900, reasoning_effort=False), "moa_aggregator": _aux(900, reasoning_effort=False), }, "display": { "compact": False, "personality": "", "resume_display": "full", # Recap tuning for /resume and startup resume. "resume_exchanges": 10, # max user+assistant pairs to show "resume_max_user_chars": 300, # truncate user message text "resume_max_assistant_chars": 200, # truncate non-last assistant text "resume_max_assistant_lines": 3, # truncate non-last assistant lines # Skip tool-call-only assistant entries in the recap so it isn't dominated by `[2 tool # calls: ...]` lines; False shows them inline. "resume_skip_tool_only": True, "busy_input_mode": "interrupt", # interrupt | queue | steer # steer mode: false hides only the "Steered into current run" bubble; steering itself still # happens. "busy_steer_ack_enabled": True, # Classic CLI multiline beyond Alt+Enter: Ctrl+J newline, trailing backslash+Enter # continues, Shift+Enter reported distinctly. False restores the c-j submit fallback for # POSIX PTYs whose plain Enter arrives as LF. "cli_multiline_shortcuts": True, # Interface bare `hermes`/`hermes chat` launches: "cli" (prompt_toolkit REPL) | "tui" (Ink). # Flags win: `--cli` forces the REPL, `--tui` / HERMES_TUI=1 forces the TUI. "interface": "cli", # `hermes --tui` auto-resumes the most recent human-facing session (like `hermes -c`). # HERMES_TUI_RESUME= always wins. "tui_auto_resume_recent": False, # Desktop reopens the last chat/page on cold start (also in Settings → Appearance). "resume_last_session": True, # One-time TUI hint ("subagents working · /agents to watch live") on first delegation. "tui_agents_nudge": True, "bell_on_complete": False, "bell_on_prompt": False, # bell when a blocking prompt opens (clarify/approval/sudo) # Stream reasoning live before the response; otherwise thinking models show only a spinner # for tens of seconds. "show_reasoning": True, # Post-response "Reasoning" recap collapses to 10 lines; true prints it all (live streaming # is always full). "reasoning_full": False, # Background self-improvement notices in chat: "off" (review still runs) | "on" (generic "💾 # Memory updated") | "verbose" (content preview). Per-platform via # display.platforms..memory_notifications. "memory_notifications": "on", # Gateway notices when a terminal(background=true) process finishes: "concise" (one line; # failures append an output tail) | "all" (running updates + final raw output) | "result" # (final raw only) | "error" (raw only on non-zero exit) | "off". "background_process_notifications": "concise", "streaming": False, "timestamps": False, # message timestamps (CLI labels, TUI rows, desktop transcript) "timestamp_format": "%H:%M", # strftime format, e.g. "%b-%d %H:%M" "final_response_markdown": "strip", # render | strip | raw # Preserve recent classic-CLI output across Ctrl+L, /redraw and resize clears; disable if an # emulator misbehaves with replayed scrollback. "persistent_output": True, "persistent_output_max_lines": 200, # Also clear terminal scrollback on classic-CLI full redraw/resize recovery; enable when a # terminal/tmux stack stamps stale prompt chrome into scrollback. "cli_rebuild_scrollback_on_redraw": False, # Print a one-line summary of resolved modal prompts (approval/clarify) to scrollback. "persist_prompts": True, "inline_diffs": True, # inline diff previews for write_file/patch/skill_manage # Append a one-line advisory to the final response when a write_file/patch failed this turn # and was never superseded by a successful write to the same path (catches "half the # parallel patches failed, model claims success"). "file_mutation_verifier": True, # Nous credits status-bar notices (usage bands, grant-spent, depleted/restored). False mutes # them; balance data and /usage keep working. "credits_notices": True, # Append a one-line explanation when a turn ends with no usable reply (empty after retries, # truncated stream, pending tool result, iteration/budget limit) instead of the bare # "(empty)" sentinel. "turn_completion_explainer": True, "show_cost": False, # $ cost in the status bar "battery": False, # battery read-out first in status bar; no-op w/o battery # Focus view (/focus): display-only. Pins tool_progress to "off", reports per-turn # hidden-line count, pins a "focus" status segment. focus_saved_tool_progress holds the mode # /focus off restores. Never affects what the model sees (focus_view.py). "focus_view": False, "focus_saved_tool_progress": "all", "skin": "default", # UI language for static messages (approval prompts, some gateway slash replies); not agent # responses/logs/tool outputs. en, zh, ja, de, es, fr, tr, uk; unknown → en. "language": "en", # TUI busy indicator: kaomoji | emoji | unicode (braille) | ascii. `/indicator