Merge remote-tracking branch 'origin/main' into feat/keyless-tavily-firecrawl-failover

This commit is contained in:
Teknium
2026-08-20 00:18:29 -07:00
66 changed files with 4342 additions and 219 deletions
+166
View File
@@ -350,6 +350,147 @@ _SUMMARY_END_MARKER = (
_MERGED_PRIOR_CONTEXT_HEADER = "[PRIOR CONTEXT — for reference only; not a new message]"
_MERGED_SUMMARY_DELIMITER = "[END OF PRIOR CONTEXT — COMPACTION SUMMARY BELOW]"
_SALVAGE_SUMMARY_MAX_CHARS = 8_000
_SALVAGE_KEEP_RECENT_TOOLS = 2
def _looks_like_compaction_summary(msg: Dict[str, Any], content: str) -> bool:
# Only cap a standalone handoff. Merged carriers preserve a real tail ask
# in the same content string; truncating those could delete live user text.
if not content.rstrip().endswith(_SUMMARY_END_MARKER):
return False
if content.startswith(_MERGED_PRIOR_CONTEXT_HEADER):
return False
# Content heuristics alone must never authorize mutating a live turn.
# Compressor-generated summaries carry this private marker; ordinary
# user input — and live assistant replies or kept tool bodies that
# merely quote a summary header/marker — do not. Tool messages are
# handled exclusively by the stub/keep-recent pass, never the cap.
if msg.get("role") == "tool":
return False
if (
msg.get("role") in ("user", "assistant")
and not msg.get(COMPRESSED_SUMMARY_METADATA_KEY)
):
return False
head = content[:280]
return (
bool(msg.get(COMPRESSED_SUMMARY_METADATA_KEY))
or "CONTEXT COMPACTION" in head
or "[CONTEXT COMPACTION]" in head
or "Conversation Summary" in head
)
def _salvage_reduce_todo_snapshot(out: List[Dict[str, Any]]) -> None:
"""Last-resort shrink: reduce or drop the synthetic todo snapshot.
The snapshot is the only in-transcript todo re-injection at a compaction
boundary, and since 7a16840add the pruned-skill reload notice is coupled
into the same string — so it is only touched when the cheaper shrink ops
could not get under budget. When the snapshot carries a reload notice,
keep just the notice (the coupling must survive salvage); otherwise drop
the row entirely.
"""
from agent.conversation_compression import _PRUNED_SKILL_RELOAD_NOTICE_HEADER
for i in range(len(out) - 1, -1, -1):
msg = out[i]
if not isinstance(msg, dict):
continue
if msg.get("_todo_snapshot_synthetic") and msg.get("role") == "user":
content = msg.get("content")
notice_idx = (
content.find(_PRUNED_SKILL_RELOAD_NOTICE_HEADER)
if isinstance(content, str)
else -1
)
if isinstance(content, str) and notice_idx >= 0:
msg["content"] = content[notice_idx:]
else:
del out[i]
return
def salvage_grown_transcript(
original: List[Dict[str, Any]],
candidate: List[Dict[str, Any]],
budget: Optional[int] = None,
) -> Optional[List[Dict[str, Any]]]:
"""Mechanically shrink a compression candidate, or return ``None``.
Already-compacted middles can be summarized slightly larger while retained
tool bodies, stale reasoning, or a synthetic todo snapshot tip the final
candidate over the input size. Work on copies and admit the salvage only
when the same rough estimator proves it is strictly smaller than the input.
Shrink order is cheapest-information-loss first: stale reasoning keys and
codex replay sidecars, then old tool bodies, then an oversized summary cap.
The synthetic todo snapshot (which carries the pruned-skill reload notice,
see ``_salvage_reduce_todo_snapshot``) is only reduced as a LAST resort
when everything else still leaves the candidate at or over budget.
"""
if not candidate or not original:
return None
if budget is None:
budget = estimate_messages_tokens_rough(original)
if budget <= 0:
return None
out: List[Dict[str, Any]] = []
tool_indices: List[int] = []
last_assistant_idx = -1
for msg in candidate:
if not isinstance(msg, dict):
out.append(msg)
continue
copied = dict(msg)
out.append(copied)
role = copied.get("role")
if role == "tool":
tool_indices.append(len(out) - 1)
elif role == "assistant":
last_assistant_idx = len(out) - 1
salvage_reasoning_keys = _NEWEST_TURN_ONLY_BUDGET_KEYS + ("reasoning_details",)
keep_tools = set(tool_indices[-_SALVAGE_KEEP_RECENT_TOOLS:])
for index, msg in enumerate(out):
if not isinstance(msg, dict):
continue
if msg.get("role") == "assistant" and index != last_assistant_idx:
for key in salvage_reasoning_keys:
msg.pop(key, None)
if msg.get("role") == "tool" and index not in keep_tools:
content = msg.get("content")
if isinstance(content, str) and len(content) > _PRUNE_MIN_CHARS:
msg["content"] = _PRUNED_TOOL_PLACEHOLDER
content = msg.get("content")
if (
isinstance(content, str)
and len(content) > _SALVAGE_SUMMARY_MAX_CHARS
and _looks_like_compaction_summary(msg, content)
):
msg["content"] = (
content[:_SALVAGE_SUMMARY_MAX_CHARS].rstrip()
+ "\n…[summary truncated so compaction can shrink]\n\n"
+ _SUMMARY_END_MARKER
)
# Heavier codex replay sidecars (encrypted reasoning blobs) — reuse the
# proven prune with its last-user-turn safety boundary (#71058).
_prune_stale_reasoning_replay(out)
if estimate_messages_tokens_rough(out) >= budget:
_salvage_reduce_todo_snapshot(out)
if not any(
isinstance(message, dict) and message.get("role") == "user"
for message in out
):
return None
if estimate_messages_tokens_rough(out) < budget:
return out
return None
# Handoff prefixes that shipped in earlier releases. A summary persisted under
# one of these can be inherited into a resumed lineage (#35344); when it is
# re-normalized on re-compaction we must strip the OLD prefix too, otherwise the
@@ -2373,6 +2514,31 @@ class ContextCompressor(ContextEngine):
self._ineffective_compression_count = count
self._persist_ineffective_compression_count()
def record_rejected_compaction(self) -> None:
"""Record one compaction whose result was REJECTED before committing.
The anti-growth guard in the commit layer (conversation_compression)
discards a candidate that would grow the transcript and keeps the
original. Without recording the attempt, the anti-thrash breaker
never sees a strike, so automatic compression retries the SAME
unchanged transcript on every turn — same summary request, same
refusal, same user-facing warning (#88568). This counts one
ineffective strike (persisted, so the normal >= 2 latch and its
recovery window apply) WITHOUT arming post-compaction real-usage
verification — nothing was committed, so there is no new compaction
to verify — and without touching the fallback-summary streak (no
summary was accepted).
"""
self._record_ineffective_compression_verdict(
self._ineffective_compression_count + 1
)
if not self.quiet_mode:
logger.warning(
"Compaction rejected before commit (would grow the "
"transcript); ineffective_compression_count=%d",
self._ineffective_compression_count,
)
def record_completed_compaction(
self, *, used_fallback: bool = False, feasibility_skip: bool = False,
) -> None:
+35
View File
@@ -3386,6 +3386,27 @@ def compress_context(
# transcript stays untouched and durable.
_rough_in = estimate_messages_tokens_rough(messages)
_rough_out = estimate_messages_tokens_rough(compressed)
if _rough_out > _rough_in:
# Todo refresh and user-turn anchoring happen after the
# compressor's own size check, so they can tip a break-even
# candidate over. Give it one mechanical salvage pass.
from agent.context_compressor import salvage_grown_transcript
_salvaged = salvage_grown_transcript(
messages, compressed, budget=_rough_in
)
if _salvaged is not None:
_salv_est = estimate_messages_tokens_rough(_salvaged)
if _salv_est < _rough_in:
logger.info(
"Compression salvage recovered a shrinking "
"transcript (session=%s, ~%s -> ~%s tokens)",
agent.session_id or "none",
f"{_rough_in:,}",
f"{_salv_est:,}",
)
compressed = _salvaged
_rough_out = _salv_est
if _rough_out > _rough_in:
logger.warning(
"Compression refused: compressed transcript would be "
@@ -3425,6 +3446,20 @@ def compress_context(
split_status="aborted",
failure_class="would_grow",
)
# Record the rejected attempt as an ineffective
# compaction strike so the anti-thrash breaker latches
# after the normal threshold. Without this, the unchanged
# transcript stays over the compression threshold and
# automatic compression retries the identical summary
# request on every turn (#88568). Manual /compress keeps
# bypassing the latch (force=True skips the guards).
try:
agent.context_compressor.record_rejected_compaction()
except Exception:
logger.debug(
"could not record rejected-compaction strike",
exc_info=True,
)
_release_lock()
return messages, _existing_sp
+48 -1
View File
@@ -2765,6 +2765,33 @@ def run_conversation(
request_pressure_tokens,
int(getattr(_compressor, "threshold_tokens", 0) or 0),
)
elif not agent.compression_enabled and len(messages) > 1:
# Uncompressed session guard (#89297): compression is disabled, so
# nothing shrinks a growing session. Reuse the unconditionally
# computed request estimate (zero marginal cost — this site runs
# before every provider request, covering turn-start AND mid-turn
# tool-result growth) and surface a deduped, actionable warning
# when the request exceeds the model context window. The dedup is
# re-armed by the turn-context preflight once the session is back
# under the window (manual /compress works with compression
# disabled), so the guard warns again on a later re-overflow.
# context_compressor always exists (agent_init constructs it even
# when compression is disabled) and its context_length property
# hard-floors at a positive default — no metadata re-resolution
# needed here.
_ctx_len = getattr(
getattr(agent, "context_compressor", None), "context_length", None
)
if (
isinstance(_ctx_len, int)
and _ctx_len > 0
and request_pressure_tokens > _ctx_len
):
_warn_fn = getattr(
agent, "_warn_uncompressed_context_overflow", None
)
if callable(_warn_fn):
_warn_fn(request_pressure_tokens, _ctx_len)
# Thinking spinner for quiet mode (animated during API call)
thinking_spinner = None
@@ -5304,6 +5331,21 @@ def run_conversation(
FailoverReason.billing,
FailoverReason.upstream_rate_limit,
}
# Relay-wrapped output-cap errors: some gateways wrap an
# upstream "[400]: max_tokens (...) exceeds model's maximum
# output tokens (...)" as HTTP 429, which classifies as
# rate_limit. The failure is a deterministic request-shape
# problem — falling back to another provider (or burning
# generic retries) can't fix it, but the output-cap clamp
# below can, in one retry (#72281). Parse once here; the
# result gates both the eager-fallback exemption and the
# widened is_context_length_error entry, and is reused as
# available_out inside the handler.
_wrapped_output_cap_budget = (
parse_available_output_tokens_from_error(error_msg)
if classified.reason == FailoverReason.rate_limit
else None
)
_is_transport_failure = classified.reason in {
FailoverReason.timeout,
FailoverReason.overloaded,
@@ -5321,7 +5363,7 @@ def run_conversation(
if _is_zai_coding_overload:
max_retries = max(max_retries, zai_coding_overload_retry_ceiling())
_should_fallback = (
is_rate_limited
(is_rate_limited and _wrapped_output_cap_budget is None)
or (_is_transport_failure and retry_count >= 2)
)
if _should_fallback and agent._fallback_index < len(agent._fallback_chain):
@@ -5614,6 +5656,11 @@ def run_conversation(
# server disconnect + large session pattern (#2153).
is_context_length_error = (
classified.reason == FailoverReason.context_overflow
# Relay-wrapped output-cap 429s (parsed once above, where
# the eager-fallback exemption is gated) route into the
# output-cap clamp below instead of provider failover or
# generic retries (#72281).
or _wrapped_output_cap_budget is not None
)
if is_context_length_error:
+37
View File
@@ -1659,10 +1659,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]:
# The input itself fits — this is purely an output-cap error, so reduce
# max_tokens and retry; do NOT compress.
"range of max_tokens should be" in error_lower
) or (
# OpenAI-compatible relays may reject a request whose output cap exceeds
# the model's separate completion-token limit, e.g.
# "max_tokens (98304) exceeds model's maximum output tokens (65536)"
# This is independent of the input context window.
"exceeds model" in error_lower
and "maximum output tokens" in error_lower
)
if not is_output_cap_error:
return None
# Generic model-output-cap form:
# "max_tokens (98304) exceeds model's maximum output tokens (65536)"
_m_max_output = re.search(
r'exceeds model(?:\'s)? maximum output tokens\s*\(?\s*(\d+)\s*\)?',
error_lower,
)
if _m_max_output:
_cap = int(_m_max_output.group(1))
if _cap >= 1:
return _cap
# DashScope / Alibaba range form: "Range of max_tokens should be [1, 65536]".
# The upper bound is the available output cap.
_m_range = re.search(
@@ -1726,11 +1744,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]:
# Available output = window - input. When the input alone is at or over
# the window this stays None, so the caller correctly falls through to
# compression instead of futilely shrinking the output cap.
#
# Caveat: when max_tokens is the BINDING constraint, vLLM does not report
# the real prompt size at all. It back-computes a lower bound from the
# constraint itself -- "at least N input tokens" where
# N == window + 1 - requested_output -- so window - N is always exactly
# requested_output - 1. Subtracting the caller's safety margin then walks
# the cap down ~65 tokens per retry while the reported input walks up by
# the same amount, burning every compression attempt without ever fitting.
# Detect that degenerate case and halve the requested cap instead: it
# carries the same guarantee (strictly below what was rejected) and
# converges in one or two retries.
_m_vllm_input = re.search(
r'prompt contains (?:at least )?(\d+)\s*input tokens', error_lower
)
if _m_ctx_tok and _m_vllm_input:
_available = int(_m_ctx_tok.group(1)) - int(_m_vllm_input.group(1))
_m_requested_out = re.search(r'requested (\d+)\s*output tokens', error_lower)
if 'at least' in error_lower and _m_requested_out:
_requested_out = int(_m_requested_out.group(1))
if _available >= _requested_out - 1:
# The budget is derived from the constraint, not measured.
return max(1, _requested_out // 2)
if _available >= 1:
return _available
@@ -1782,6 +1817,8 @@ def is_output_cap_error(error_msg: str) -> bool:
or "should be" in error_lower # generic "max_tokens should be <= N"
or "less than or equal" in error_lower
or "must be" in error_lower
or ("exceeds model" in error_lower
and "maximum output tokens" in error_lower)
)
if not output_cap_signal:
return False
+23
View File
@@ -48,12 +48,31 @@ from tools.thread_context import propagate_context_to_thread
from tools.tool_result_storage import (
maybe_persist_tool_result,
enforce_turn_budget,
extract_persisted_path,
)
from tools.budget_config import BudgetConfig, DEFAULT_BUDGET, budget_for_context_window
logger = logging.getLogger(__name__)
def _record_persisted_path_for_stub(agent, tool_call_id: str, function_result) -> None:
"""Tell the stall guards where a persisted result's full content lives.
When a large result is spilled to disk (<persisted-output> preview), a
later result-reference stub pointing at that first occurrence must carry
the spillover file path so the reference can't dangle. Best-effort: never
lets bookkeeping break tool execution.
"""
try:
if not isinstance(function_result, str):
return
path = extract_persisted_path(function_result)
if path:
agent._tool_guardrails.record_persisted_result(tool_call_id, path)
except Exception as exc:
logger.debug("persisted-path record for result stub failed: %s", exc)
def _ensure_file_checkpoint(
agent,
function_name: str,
@@ -1733,6 +1752,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
function_args,
function_result,
failed=is_error,
tool_call_id=getattr(tc, "id", "") or "",
)
if is_error:
@@ -1767,6 +1787,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
env=get_active_env(effective_task_id),
config=_tool_budget,
) if not _is_multimodal_tool_result(function_result) else function_result
_record_persisted_path_for_stub(agent, tc.id, function_result)
subdir_hints = agent._subdirectory_hints.check_tool_call(name, args)
if subdir_hints:
@@ -2572,6 +2593,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
function_args,
function_result,
failed=_is_error_result,
tool_call_id=getattr(tool_call, "id", "") or "",
)
result_preview = function_result if agent.verbose_logging else (
function_result[:200] if len(function_result) > 200 else function_result
@@ -2610,6 +2632,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
env=get_active_env(effective_task_id),
config=_tool_budget,
) if not _is_multimodal_tool_result(function_result) else function_result
_record_persisted_path_for_stub(agent, tool_call.id, function_result)
# Discover subdirectory context files from tool arguments
subdir_hints = agent._subdirectory_hints.check_tool_call(function_name, function_args)
+151 -26
View File
@@ -85,6 +85,19 @@ _STALL_GUARD_REPEATABLE_SUFFIXES = (
# catching the observed re-issue loops (3x/4x identical calls in eval traces).
STALL_GUARD_IDENTICAL_CALL_THRESHOLD = 3
# Result-reference stubbing (agent.stall_guards): from the 2nd consecutive
# identical call whose FRESH result is byte-identical to the previous one,
# the duplicate payload is replaced in context by a short reference stub.
# Results under this size aren't worth stubbing (the stub itself plus the
# lost locality outweigh the savings), and error results are never stubbed
# (the model must see every fresh error verbatim).
IDENTICAL_RESULT_STUB_MIN_CHARS = 512
# How much of the canonical args JSON the stub carries so the model still
# knows WHAT the referenced call was even if context compression later
# evicts the referenced result (cheap dangling-reference mitigation).
_RESULT_STUB_ARGS_PREVIEW_CHARS = 120
def is_stall_guard_repeatable(tool_name: str) -> bool:
"""Whether a tool is exempt from the identical-call loop notice."""
@@ -206,6 +219,21 @@ class LoopCapConfig:
)
@dataclass(frozen=True)
class IdenticalCallObservation:
"""Outcome of observing one completed tool call for the stall guards.
``notice`` is the identical-call loop-breaker notice (appended after the
result). ``stub`` is the result-reference replacement for a byte-identical
duplicate result (replaces the result content). Both may be set on the
same call (3rd+ identical call): the stub replaces the payload and the
notice is appended after it.
"""
notice: str | None = None
stub: str | None = None
@dataclass(frozen=True)
class ToolCallSignature:
"""Stable, non-reversible identity for a tool name plus canonical args."""
@@ -320,9 +348,21 @@ class ToolCallGuardrailController:
# results were also identical. Any different call — or a different
# result — resets the streak, so legitimate re-reads after edits and
# varied polling are never flagged. Per-turn, like everything else here.
# NOTE: open PR #85352 (patrykkopycinski) tracks no-progress loops
# ACROSS turns via a detection window — a different mechanism from
# this per-turn consecutive streak. Coordinate future work there.
self._identical_streak_sig: ToolCallSignature | None = None
self._identical_streak_result_hash: str = ""
self._identical_streak_count: int = 0
# tool_call_id of the FIRST call in the current streak, so a
# result-reference stub can point at the message that carries the
# full payload.
self._identical_streak_first_call_id: str = ""
# tool_call_id -> spillover file path for results that were persisted
# out of context (persisted-output preview). Lets a reference stub
# carry the file path so the reference can't dangle when the first
# occurrence entered context as a preview.
self._persisted_result_paths: dict[str, str] = {}
# Per-turn runaway-loop cap counters. Reset every turn (this method
# runs at the start of each run_conversation), so the caps bound a
# single agent loop rather than accumulating across the session.
@@ -493,44 +533,129 @@ class ToolCallGuardrailController:
) -> str | None:
"""Track consecutive identical calls; return a loop-breaker notice or None.
Fires the compact notice when the SAME tool is called with identical
canonical arguments AND returns an identical result for the
``STALL_GUARD_IDENTICAL_CALL_THRESHOLD``-th (and every subsequent)
consecutive time within the turn. Purely observational — never blocks
the call. Allowlisted pollers (``is_stall_guard_repeatable``) are
exempt, and any intervening different call or changed result resets
the streak. Callers append the returned notice to the tool RESULT at
construction time, which is cache-safe: tool results are append-only
and never mutate already-sent context.
Back-compat wrapper around :meth:`observe_call` for callers that only
care about the loop-breaker notice.
"""
if is_stall_guard_repeatable(tool_name):
# Don't let a poller streak carry over into the next tool either.
self._identical_streak_sig = None
self._identical_streak_count = 0
return None
return self.observe_call(tool_name, args, result).notice
def observe_call(
self,
tool_name: str,
args: Mapping[str, Any] | None,
result: str | None,
*,
tool_call_id: str = "",
failed: bool = False,
) -> "IdenticalCallObservation":
"""Track consecutive identical calls; return notice + dedupe stub info.
Two independent outputs from the same consecutive-streak tracker:
- ``notice``: the compact loop-breaker notice, fired when the SAME
tool is called with identical canonical arguments AND returns an
identical result for the ``STALL_GUARD_IDENTICAL_CALL_THRESHOLD``-th
(and every subsequent) consecutive time within the turn. Purely
observational — never blocks the call. Allowlisted pollers
(``is_stall_guard_repeatable``) are exempt from the NOTICE.
- ``stub``: a short reference replacement for the CURRENT result,
produced from the 2nd consecutive identical call whose fresh result
is byte-identical to the previous one. The tool still executed —
only the context representation is deduplicated, so polling
semantics are preserved (a changed result flows through whole and
resets the streak). Pollers are NOT exempt from stubbing: for a
poller, an identical result means nothing changed, which is exactly
when the stub saves the most context and loses nothing. Results
under ``IDENTICAL_RESULT_STUB_MIN_CHARS`` and failed/error results
are never stubbed, and only plain-string results are considered.
Any intervening different call or changed result resets the streak.
Callers substitute/append at tool RESULT construction time, which is
cache-safe: tool results are append-only and never mutate
already-sent context.
"""
is_plain_str = isinstance(result, str)
signature = ToolCallSignature.from_call(tool_name, _coerce_args(args))
result_hash = _result_hash(result)
result_hash = _result_hash(result) if is_plain_str else ""
if (
self._identical_streak_sig == signature
is_plain_str
and self._identical_streak_sig == signature
and self._identical_streak_result_hash == result_hash
):
self._identical_streak_count += 1
else:
self._identical_streak_sig = signature
# New streak (or non-string result, which never forms a streak —
# multimodal content lists pass through untouched).
self._identical_streak_sig = signature if is_plain_str else None
self._identical_streak_result_hash = result_hash
self._identical_streak_count = 1
self._identical_streak_count = 1 if is_plain_str else 0
self._identical_streak_first_call_id = tool_call_id or ""
count = self._identical_streak_count
if count < STALL_GUARD_IDENTICAL_CALL_THRESHOLD:
return None
ordinal = f"{count}{'th' if 11 <= count % 100 <= 13 else {1: 'st', 2: 'nd', 3: 'rd'}.get(count % 10, 'th')}"
return (
f"[hermes note: this is the {ordinal} consecutive identical call to "
f"{tool_name} with identical arguments returning the same result. "
"Do not repeat it — change arguments, use a different tool, or "
"proceed with what you have.]"
notice = None
if (
not is_stall_guard_repeatable(tool_name)
and count >= STALL_GUARD_IDENTICAL_CALL_THRESHOLD
):
ordinal = f"{count}{'th' if 11 <= count % 100 <= 13 else {1: 'st', 2: 'nd', 3: 'rd'}.get(count % 10, 'th')}"
notice = (
f"[hermes note: this is the {ordinal} consecutive identical call to "
f"{tool_name} with identical arguments returning the same result. "
"Do not repeat it — change arguments, use a different tool, or "
"proceed with what you have.]"
)
stub = None
if (
is_plain_str
and count >= 2
and not failed
and len(result) >= IDENTICAL_RESULT_STUB_MIN_CHARS
):
stub = self._build_result_reference_stub(tool_name, args)
return IdenticalCallObservation(notice=notice, stub=stub)
def record_persisted_result(self, tool_call_id: str, file_path: str) -> None:
"""Remember the spillover path a persisted result was saved to.
When the first occurrence of a result entered context as a
persisted-output preview, a later reference stub must carry the
spillover file path so the reference can't dangle.
"""
if tool_call_id and file_path:
self._persisted_result_paths[tool_call_id] = file_path
def _build_result_reference_stub(
self, tool_name: str, args: Mapping[str, Any] | None
) -> str:
"""Build the reference stub replacing a byte-identical duplicate result.
Carries the tool name + a canonical-args preview so that even if
context compression later evicts the referenced result, the model
still knows WHAT the call was (cheap dangling-reference mitigation).
"""
try:
args_preview = canonical_tool_args(_coerce_args(args))
except TypeError:
args_preview = "{}"
if len(args_preview) > _RESULT_STUB_ARGS_PREVIEW_CHARS:
args_preview = args_preview[:_RESULT_STUB_ARGS_PREVIEW_CHARS] + "…"
first_id = self._identical_streak_first_call_id
ref = f" (tool_call_id {first_id})" if first_id else ""
stub = (
f"[hermes note: this result is byte-identical to the {tool_name} "
f"result earlier this turn{ref}. Refer to that result; it has not "
f"changed. Args: {args_preview}]"
)
spill_path = self._persisted_result_paths.get(first_id) if first_id else None
if spill_path:
stub += (
f"\n[The referenced result was persisted to: {spill_path} — "
"page through it with read_file if you need the full content.]"
)
return stub
def _check_loop_cap(
self,
+51
View File
@@ -1163,6 +1163,57 @@ def build_turn_context(
agent._last_content_with_tools = None
agent._last_content_tools_all_housekeeping = False
agent._mute_post_response = False
elif not agent.compression_enabled:
# Uncompressed session guard (#89297): when compression is explicitly
# disabled, sessions can grow past the model's context window across
# hundreds of messages with nothing to shrink them. The warning itself
# fires from the conversation loop's pre-API site, which reuses the
# unconditionally computed request estimate at zero marginal cost and
# covers both turn-start and mid-turn growth (every provider request
# passes through it). Here we only RE-ARM the dedup once the session
# is back under the window, so the guard can warn again after the
# user compacts (/compress with force=True works with compression
# disabled) and the context later regrows past the limit.
_ctx_len = getattr(
getattr(agent, "context_compressor", None), "context_length", None
)
if isinstance(_ctx_len, int) and _ctx_len > 0:
_raw_chars = 0
for _m in messages:
if not isinstance(_m, dict):
continue
_c = _m.get("content")
if isinstance(_c, str):
_raw_chars += len(_c)
elif _c:
# Non-string, non-empty content (multimodal part lists,
# dict payloads) defeats a char count — force the real
# estimate by treating it as over-gate. None/"" (routine
# assistant tool-call rows) contribute nothing.
_raw_chars = _ctx_len + 1
break
# Cheap gate: a session whose raw text is under ~1/4 of the
# window (4 chars/token upper bound) cannot be over it — skip
# the estimator. Non-string (multimodal) content defeats a char
# count, so any such message forces the real estimate.
if _raw_chars <= _ctx_len:
_clear_warn = getattr(
agent, "_clear_context_overflow_warn", None
)
if callable(_clear_warn):
_clear_warn()
else:
_uncompressed_tokens = estimate_request_tokens_rough(
messages,
system_prompt=active_system_prompt or "",
tools=agent.tools or None,
)
if _uncompressed_tokens <= _ctx_len:
_clear_warn = getattr(
agent, "_clear_context_overflow_warn", None
)
if callable(_clear_warn):
_clear_warn()
if _preflight_compressed:
# Compression rebuilt the list (tail messages are fresh compaction
+13 -3
View File
@@ -256,6 +256,7 @@ import {
import { missingRendererAssets } from './renderer-bundle'
import { attachRendererConsoleCapture, formatRendererBoundaryReport } from './renderer-log'
import {
buildInstanceWindowUrl,
buildSessionWindowUrl,
chatWindowWebPreferences,
createSessionWindowRegistry,
@@ -10870,8 +10871,10 @@ function createSessionWindow(sessionId, { watch = false } = {}) {
// Additional full "instance" windows — peers of the primary that render the
// COMPLETE app (sidebar, routing, its own draft) against the shared backend, so
// a user can run multiple GUI windows at once (⌘⇧N / the "New Window" palette
// command). Unlike the compact session windows they carry no `?win` flag. The
// primary mainWindow stays the notification / deep-link / pet-overlay anchor and
// command). Unlike the compact session windows they carry no `?win` flag; a
// separate `peer=1` marker prevents them from replaying app-launch source
// restoration after joining that shared backend. The primary mainWindow stays
// the notification / deep-link / pet-overlay anchor and
// is NOT tracked here. The set holds a strong reference so an open peer isn't
// garbage-collected, and drops it on close.
const instanceWindows = new Set<any>()
@@ -10951,7 +10954,14 @@ function createInstanceWindow() {
})
attachRendererConsoleCapture(win, 'instance', rememberLog)
loadWindowUrl(win, DEV_SERVER || pathToFileURL(resolveRendererIndex()).toString(), 'Instance window')
loadWindowUrl(
win,
buildInstanceWindowUrl({
devServer: DEV_SERVER,
rendererIndexPath: DEV_SERVER ? undefined : resolveRendererIndex()
}),
'Instance window'
)
return win
}
@@ -3,6 +3,7 @@ import assert from 'node:assert/strict'
import { test } from 'vitest'
import {
buildInstanceWindowUrl,
buildSessionWindowUrl,
chatWindowWebPreferences,
createSessionWindowRegistry,
@@ -88,6 +89,19 @@ test('buildSessionWindowUrl adds the watch flag for spectator windows, before th
assert.equal(url, 'http://localhost:5173/?win=secondary&watch=1#/abc')
})
test('buildInstanceWindowUrl marks a full peer without selecting a specialized renderer', () => {
const url = buildInstanceWindowUrl({ devServer: 'http://localhost:5173/' })
assert.equal(url, 'http://localhost:5173/?peer=1')
assert.ok(!url.includes('win='))
})
test('buildInstanceWindowUrl marks a packaged full peer', () => {
const url = buildInstanceWindowUrl({ rendererIndexPath: '/opt/app/index.html' })
assert.match(url, /^file:\/\/.*index\.html\?peer=1$/)
})
test('instanceWindowBounds cascades a new window off its source bounds', () => {
const bounds = instanceWindowBounds({ x: 100, y: 120, width: 1400, height: 900 }, { width: 1, height: 1 })
+18
View File
@@ -77,6 +77,23 @@ function buildSessionWindowUrl(sessionId: string, { devServer, rendererIndexPath
return `${pathToFileURL(rendererIndexPath).toString()}${query}${route}`
}
// Full peer windows render the ordinary app shell, so they deliberately do
// not use the `win` query parameter that selects a specialized renderer. The
// separate marker lets the renderer distinguish a peer from the one primary
// app window: app-launch source restoration belongs to the primary only, while
// a peer keeps the already-running backend it joined during boot.
function buildInstanceWindowUrl({ devServer, rendererIndexPath }: any = {}) {
const query = '?peer=1'
if (devServer) {
const base = devServer.endsWith('/') ? devServer.slice(0, -1) : devServer
return `${base}/${query}`
}
return `${pathToFileURL(rendererIndexPath).toString()}${query}`
}
// Full "instance" windows (⌘⇧N / the "New Window" command) open a complete app
// peer, not a compact chat. Cascade each one off its source window's bounds so a
// new window doesn't land exactly on top of the one it was spawned from. Pure so
@@ -160,6 +177,7 @@ function createSessionWindowRegistry() {
}
export {
buildInstanceWindowUrl,
buildSessionWindowUrl,
chatWindowWebPreferences,
createSessionWindowRegistry,
@@ -22,6 +22,24 @@ export const COMPOSER_STACK_BREAKPOINT_PX = 320
// chevron frees is spent keeping the row single for another stretch.
export const COMPOSER_COMPACT_PILL_PX = 560
// The ladder keeps going below the stack breakpoint — a pane can be far
// narrower than even the stacked controls row. Both rungs are budgeted
// against that row's real cost: menu ~24 + surface padding 16 + the cluster
// (~190; ~218 mid-turn with the queue button).
//
// At 260 the three voice toggles fold into the one menu HUD mode already
// uses, clearing the mid-turn worst case with margin. Each stage sits clear
// of the floor below it rather than arriving the instant the previous one
// gives out — the mistake COMPOSER_COMPACT_PILL_PX documents.
export const COMPOSER_FOLD_VOICE_PX = 260
// Type and send, nothing else. A pane can be dragged to MIN_PANE_PX (80), and
// even with voice folded the row still costs ~150, so the last rung drops the
// pill AND the voice menu. Both stay reachable — the model by hotkey and the
// full picker, dictation from any wider pane — and Send fits with room to
// spare at any width the layout tree allows (~74 all-in).
export const COMPOSER_MINIMAL_PX = 180
// A single editor line is ~28px (--composer-input-min-height 1.625rem + 0.5rem
// vertical padding). Anything taller means the text wrapped to a second line,
// which is when the composer should expand to the stacked layout.
@@ -98,6 +98,37 @@ describe('HUD mode', () => {
})
})
// A tile can be narrower than the controls cost, and the row is inside an
// overflow-hidden surface — so anything that doesn't fold gets clipped off the
// right edge, send button first. The ladder keeps going past `stacked`: voice
// folds into the same menu the HUD uses, then the model pill drops. Send is
// the last thing standing.
describe('narrow tiles', () => {
it('folds the voice controls into one menu without entering HUD mode', () => {
renderControls({ foldVoice: true })
expect(screen.getByLabelText('Voice')).toBeTruthy()
expect(screen.queryByLabelText('Voice dictation')).toBeNull()
expect(screen.queryByLabelText('Read replies aloud')).toBeNull()
// Folding is a width decision, not the HUD: no exit affordance appears.
expect(screen.queryByLabelText('Exit HUD mode')).toBeNull()
})
it('keeps Send at the tightest width, with everything else dropped', () => {
renderControls({ foldVoice: true, minimal: true })
expect(screen.getByLabelText('Send')).toBeTruthy()
expect(screen.queryByLabelText('Voice')).toBeNull()
})
it('keeps Stop reachable mid-turn at the tightest width', () => {
renderControls({ busy: true, busyAction: 'stop', foldVoice: true, hasComposerPayload: false, minimal: true })
expect(screen.getByLabelText('Stop')).toBeTruthy()
})
})
describe('ComposerControls shortcut tooltips', () => {
it('shows Enter for Send', async () => {
renderControls()
+33 -20
View File
@@ -39,7 +39,9 @@ export function ComposerControls({
compactModelPill = false,
conversation,
disabled,
foldVoice = false,
hasComposerPayload,
minimal = false,
state,
voiceStatus,
onDictate,
@@ -53,7 +55,9 @@ export function ComposerControls({
compactModelPill?: boolean
conversation: ConversationProps
disabled: boolean
foldVoice?: boolean
hasComposerPayload: boolean
minimal?: boolean
state: ChatBarState
voiceStatus: VoiceStatus
onDictate: () => void
@@ -73,29 +77,38 @@ export function ComposerControls({
// only when the composer is empty and a turn is running.
const showStop = busy && !hasComposerPayload
const showQueueButton = busyAction !== 'stop' && hasComposerPayload
// The HUD is a Spotlight bar a few hundred pixels wide, so the four separate
// voice toggles fold into one menu there and leave the row to the input. A
// narrow tile hits the same wall from the other direction and folds for the
// same reason — same controls, same state, different budget. Below that
// even the menu goes: at `minimal` the row is the send button and nothing
// else, which is the one thing that must survive every width.
const foldedVoice = hudMode || foldVoice
const voiceControls = foldedVoice ? (
<VoiceMenu
autoSpeak={autoSpeak}
disabled={disabled}
onDictate={onDictate}
onStartConversation={conversation.onStart}
onToggleAutoSpeak={onToggleAutoSpeak}
state={state}
voiceStatus={voiceStatus}
/>
) : (
<>
<DictationButton disabled={disabled} onToggle={onDictate} state={state.voice} status={voiceStatus} />
<AutoSpeakButton active={autoSpeak} disabled={disabled} onToggle={onToggleAutoSpeak} />
<WakeWordButton disabled={disabled} />
</>
)
return (
<div className="ml-auto flex shrink-0 items-center gap-(--composer-control-gap)">
<ModelPill compact={compactModelPill} disabled={disabled} model={state.model} />
{/* The HUD is a Spotlight bar a few hundred pixels wide, so the four
separate voice toggles fold into one menu there and leave the row to
the input. The docked composer has the width and keeps them inline —
same controls, same state, different budget. */}
{hudMode ? (
<VoiceMenu
autoSpeak={autoSpeak}
disabled={disabled}
onDictate={onDictate}
onStartConversation={conversation.onStart}
onToggleAutoSpeak={onToggleAutoSpeak}
state={state}
voiceStatus={voiceStatus}
/>
) : (
<div className="ml-auto flex min-w-0 shrink items-center gap-(--composer-control-gap)">
{minimal ? null : (
<>
<DictationButton disabled={disabled} onToggle={onDictate} state={state.voice} status={voiceStatus} />
<AutoSpeakButton active={autoSpeak} disabled={disabled} onToggle={onToggleAutoSpeak} />
<WakeWordButton disabled={disabled} />
<ModelPill compact={compactModelPill} disabled={disabled} model={state.model} />
{voiceControls}
</>
)}
{showQueueButton ? (
@@ -10,7 +10,13 @@ import {
} from '@/app/chat/surface-vars'
import { useResizeObserver } from '@/hooks/use-resize-observer'
import { COMPOSER_COMPACT_PILL_PX, COMPOSER_SINGLE_LINE_MAX_PX, COMPOSER_STACK_BREAKPOINT_PX } from '../composer-utils'
import {
COMPOSER_COMPACT_PILL_PX,
COMPOSER_FOLD_VOICE_PX,
COMPOSER_MINIMAL_PX,
COMPOSER_SINGLE_LINE_MAX_PX,
COMPOSER_STACK_BREAKPOINT_PX
} from '../composer-utils'
interface UseComposerMetricsArgs {
composerDockRef: RefObject<HTMLDivElement | null>
@@ -20,13 +26,36 @@ interface UseComposerMetricsArgs {
poppedOut: boolean
}
/** Every width-driven collapse stage, resolved from the composer's own width. */
export interface ComposerFit {
compactPill: boolean
foldVoice: boolean
minimal: boolean
tight: boolean
}
const ROOMY: ComposerFit = { compactPill: false, foldVoice: false, minimal: false, tight: false }
const fitForWidth = (width: number): ComposerFit => ({
compactPill: width < COMPOSER_COMPACT_PILL_PX,
foldVoice: width < COMPOSER_FOLD_VOICE_PX,
minimal: width < COMPOSER_MINIMAL_PX,
tight: width < COMPOSER_STACK_BREAKPOINT_PX
})
const sameFit = (a: ComposerFit, b: ComposerFit) =>
a.compactPill === b.compactPill && a.foldVoice === b.foldVoice && a.minimal === b.minimal && a.tight === b.tight
interface UseComposerMetricsResult extends ComposerFit {
stacked: boolean
}
/**
* Owns the composer's *sizing* engine: the stacked-vs-inline layout decision
* and the measured-height CSS vars the thread reads for bottom clearance. All
* work is edge-gated — the ResizeObserver only fires on real size changes, the
* height vars are 8px-bucketed so per-keystroke growth never invalidates the
* tree's computed style, and `tight` only flips when it crosses the breakpoint.
* Returns `stacked` (the only value the render needs).
* tree's computed style, and the fit only re-renders when it crosses a stage.
*/
export function useComposerMetrics({
composerDockRef,
@@ -34,14 +63,9 @@ export function useComposerMetrics({
composerSurfaceRef,
editorRef,
poppedOut
}: UseComposerMetricsArgs): {
compactPill: boolean
stacked: boolean
} {
}: UseComposerMetricsArgs): UseComposerMetricsResult {
const [expanded, setExpanded] = useState(false)
const [tight, setTight] = useState(false)
// Wider than `tight`: the pill goes icon-only before the row has to stack.
const [compactPill, setCompactPill] = useState(false)
const [fit, setFit] = useState<ComposerFit>(ROOMY)
// Edge signals, not the live text: these only re-render when emptiness / the
// presence of a non-trailing newline actually flips, so typing within a line
@@ -84,8 +108,7 @@ export function useComposerMetrics({
// until a wrap or row change actually happens.
const lastBucketedHeightRef = useRef(0)
const lastBucketedSurfaceHeightRef = useRef(0)
const lastTightRef = useRef<boolean | null>(null)
const lastCompactPillRef = useRef<boolean | null>(null)
const lastFitRef = useRef(ROOMY)
// Mirrored into a ref so `syncComposerMetrics` stays referentially stable —
// it's the shared ResizeObserver's handler, and a new identity every render
// would re-register the observation.
@@ -121,18 +144,11 @@ export function useComposerMetrics({
const surfaceHeight = composerSurfaceRef.current?.getBoundingClientRect().height
if (width > 0) {
const nextTight = width < COMPOSER_STACK_BREAKPOINT_PX
const nextFit = fitForWidth(width)
if (nextTight !== lastTightRef.current) {
lastTightRef.current = nextTight
setTight(nextTight)
}
const nextCompactPill = width < COMPOSER_COMPACT_PILL_PX
if (nextCompactPill !== lastCompactPillRef.current) {
lastCompactPillRef.current = nextCompactPill
setCompactPill(nextCompactPill)
if (!sameFit(nextFit, lastFitRef.current)) {
lastFitRef.current = nextFit
setFit(nextFit)
}
}
@@ -189,7 +205,7 @@ export function useComposerMetrics({
}
}, [composerRef])
// Both decisions come from the composer's OWN measured width, never the
// Every decision comes from the composer's OWN measured width, never the
// viewport's. There used to be a `(max-width: 30rem)` media query in here as
// well, and it quietly outranked everything: any window under 480px stacked
// the row AND compacted the pill in the same instant, regardless of how much
@@ -199,7 +215,14 @@ export function useComposerMetrics({
// stack) by 160px. The ResizeObserver knows the real width; the viewport is
// not a proxy for it.
//
// The pill still compacts whenever the row stacks, so the controls row can't
// over-run once it has the width to itself.
return { compactPill: compactPill || tight, stacked: expanded || tight }
// The ladder is monotonic: each stage implies the ones above it, so the pill
// is always compact by the time the row stacks, and the voice controls are
// always folded before minimal drops them.
return {
compactPill: fit.compactPill || fit.tight,
foldVoice: fit.foldVoice || fit.minimal,
minimal: fit.minimal,
stacked: expanded || fit.tight,
tight: fit.tight
}
}
+11 -3
View File
@@ -321,7 +321,7 @@ export function ChatBar({
return onCancel()
}, [activeQueueSessionKeyRef, onCancel])
const { compactPill, stacked } = useComposerMetrics({
const { compactPill, foldVoice, minimal, stacked } = useComposerMetrics({
composerDockRef,
composerRef,
composerSurfaceRef,
@@ -984,7 +984,9 @@ export function ChatBar({
status: conversation.status
}}
disabled={disabled}
foldVoice={foldVoice}
hasComposerPayload={hasComposerPayload}
minimal={minimal}
onDictate={dictate}
onQueue={queueDraft}
onToggleAutoSpeak={handleToggleAutoSpeak}
@@ -1242,7 +1244,13 @@ export function ChatBar({
{hudMode && busy && <span aria-hidden className="arc-border arc-composer" />}
<div
className={cn(
'group/composer-surface relative z-4 isolate grid grid-rows-[auto_1fr] overflow-hidden rounded-[inherit] border border-[color-mix(in_srgb,var(--dt-composer-ring)_calc(18%*var(--composer-ring-strength)),var(--dt-input))]',
// grid-cols-[minmax(0,1fr)]: the implicit `auto` column sized
// itself to its items' min-content, so a status row whose
// content out-measured a narrow pane silently widened the
// track past the surface — and every `w-full` child (the fade,
// the input/controls row) laid out against that phantom width
// and got clipped by overflow-hidden, send button first.
'group/composer-surface relative z-4 isolate grid grid-cols-[minmax(0,1fr)] grid-rows-[auto_1fr] overflow-hidden rounded-[inherit] border border-[color-mix(in_srgb,var(--dt-composer-ring)_calc(18%*var(--composer-ring-strength)),var(--dt-input))]',
COMPOSER_DROP_FADE_CLASS,
dragActive && COMPOSER_DROP_ACTIVE_CLASS
)}
@@ -1322,7 +1330,7 @@ export function ChatBar({
<ContribSlot area={COMPOSER_AREAS.leading} />
</div>
<div className="min-w-0 [grid-area:input]">{input}</div>
<div className="flex items-center justify-end gap-(--composer-control-gap) [grid-area:controls]">
<div className="flex min-w-0 items-center justify-end gap-(--composer-control-gap) [grid-area:controls]">
<ContribSlot area={COMPOSER_AREAS.actions} />
{controls}
</div>
@@ -18,8 +18,11 @@ import { onComposerModelMenuRequest } from './focus'
import { useComposerScope } from './scope'
import type { ChatBarState } from './types'
// `shrink` (not `shrink-0`) with a truncating label: the pill is the one
// control in the row that can give width back continuously, so it absorbs the
// squeeze between collapse stages instead of pushing Send past the edge.
const PILL = cn(
'h-(--composer-control-size) max-w-40 shrink-0 gap-1 rounded-md px-2 text-xs font-normal',
'h-(--composer-control-size) min-w-0 max-w-40 shrink gap-1 rounded-md px-2 text-xs font-normal',
'text-(--ui-text-tertiary) hover:bg-(--chrome-action-hover) hover:text-foreground'
)
@@ -40,6 +40,11 @@ vi.mock('@/store/boot', () => ({
})
}))
vi.mock('@/store/windows', () => ({
isAuxiliaryWindow: vi.fn(() => false),
isPeerInstanceWindow: vi.fn(() => false)
}))
vi.mock('@/i18n', () => ({
useI18n: () => ({
t: {
@@ -65,6 +70,7 @@ vi.mock('@/i18n', () => ({
const connectionStore = await import('@/store/connections')
const bootStore = await import('@/store/boot')
const windowStore = await import('@/store/windows')
const $activeConnectionId = connectionStore.$activeConnectionId as ReturnType<typeof atom<null | string>>
const $connectionsRegistry = connectionStore.$connectionsRegistry
const $desktopBoot = bootStore.$desktopBoot
@@ -72,6 +78,8 @@ const $pendingConnectionId = connectionStore.$pendingConnectionId
const initializeConnectionsRegistry = vi.mocked(connectionStore.initializeConnectionsRegistry)
const refreshConnectionsRegistry = vi.mocked(connectionStore.refreshConnectionsRegistry)
const selectConnection = vi.mocked(connectionStore.selectConnection)
const isAuxiliaryWindow = vi.mocked(windowStore.isAuxiliaryWindow)
const isPeerInstanceWindow = vi.mocked(windowStore.isPeerInstanceWindow)
const onConnect = vi.fn()
const connection = (id: string, label: string, kind: 'local' | 'remote' = 'remote') => ({
@@ -106,6 +114,8 @@ afterEach(() => {
})
$pendingConnectionId.set(null)
$findInPage.set({ active: false, query: '', matchOrdinal: 0, matchCount: 0 })
isAuxiliaryWindow.mockReturnValue(false)
isPeerInstanceWindow.mockReturnValue(false)
})
describe('ConnectionSwitcher', () => {
@@ -127,6 +137,38 @@ describe('ConnectionSwitcher', () => {
await waitFor(() => expect(initializeConnectionsRegistry).toHaveBeenCalledTimes(1))
})
it('keeps a full peer on the shared backend instead of replaying app-launch source restoration', async () => {
isPeerInstanceWindow.mockReturnValue(true)
$desktopBoot.set({
...$desktopBoot.get(),
phase: 'renderer.ready',
progress: 100,
running: false,
visible: false
})
render(<ConnectionSwitcher onConnect={onConnect} />)
await waitFor(() => expect(refreshConnectionsRegistry).toHaveBeenCalledTimes(1))
expect(initializeConnectionsRegistry).not.toHaveBeenCalled()
})
it('keeps a secondary session window from replaying app-launch source restoration', async () => {
isAuxiliaryWindow.mockReturnValue(true)
$desktopBoot.set({
...$desktopBoot.get(),
phase: 'renderer.ready',
progress: 100,
running: false,
visible: false
})
render(<ConnectionSwitcher onConnect={onConnect} />)
await waitFor(() => expect(refreshConnectionsRegistry).toHaveBeenCalledTimes(1))
expect(initializeConnectionsRegistry).not.toHaveBeenCalled()
})
it('adds no source chrome for a local-only setup', () => {
$connectionsRegistry.set(registry([connection('local', 'This device', 'local')]))
render(<ConnectionSwitcher onConnect={onConnect} />)
@@ -36,6 +36,7 @@ import {
} from '@/store/connections'
import { closeFindBar } from '@/store/find-in-page'
import { notifyError } from '@/store/notifications'
import { isAuxiliaryWindow, isPeerInstanceWindow } from '@/store/windows'
export function ConnectionSwitcher({ compact = false, onConnect }: { compact?: boolean; onConnect: () => void }) {
const { t } = useI18n()
@@ -64,8 +65,10 @@ export function ConnectionSwitcher({ compact = false, onConnect }: { compact?: b
// The primary boot owns its initial config/session fetches. Restoring a
// different source before those settle lets a late primary response repaint
// the sidebar under the new source label. Switch only after boot completes,
// then the normal source reset/refetch remains the final writer.
if (!boot.running) {
// then the normal source reset/refetch remains the final writer. Peer and
// auxiliary windows already boot into their intended runtime; replaying the
// primary window's app-launch preference would move them away from it.
if (!boot.running && !isAuxiliaryWindow() && !isPeerInstanceWindow()) {
void initializeConnectionsRegistry().catch(() => undefined)
}
}, [boot.running])
@@ -150,6 +150,10 @@ export function NarrowOverlays() {
? 'left-0 border-r border-(--ui-stroke-secondary)'
: 'right-0 border-l border-(--ui-stroke-secondary)'
)}
// Floats OVER the layout, so under glass its surface must mask the
// panes beneath it — a see-through overlay reads as text bleeding
// through text. Contract: `[data-glass-opaque]` in styles.css.
data-glass-opaque=""
onMouseLeave={() => setReveal(current => (current?.pinned ? current : null))}
// Match the pane's docked width (sessions ~237px, files its rail
// width) instead of a fat fixed 20rem — capped for tiny screens.
+113 -35
View File
@@ -2969,6 +2969,53 @@ function botHandle(name, bot) {
return (name || '').trim().toLowerCase() === 'default' ? 'hermes' : name
}
/** Taggable @-forms derived from a bot's friendly names — the core profile
* display name (`hermes profile rename`) and the Bot Mode title. Free text
* reduces to the mention charset two ways: slugified ("Research Buddy" →
* research-buddy, the form autocomplete inserts) and collapsed
* (researchbuddy). Reserved tokens are dropped so a bot renamed "Hermes"
* can never hijack the primary profile's @hermes alias. */
function mentionNameForms(value) {
const name = String(value || '').trim().toLowerCase()
if (!name) {
return []
}
const slug = name.replace(/[^a-z0-9_-]+/g, '-').replace(/^-+|-+$/g, '')
const collapsed = name.replace(/[^a-z0-9_-]+/g, '')
return [...new Set([slug, collapsed])].filter(
form => /^[a-z0-9][a-z0-9_-]*$/.test(form) && !['all', 'everyone', 'user', 'default', 'hermes'].includes(form)
)
}
/** Every friendly (renameable) name a roster row carries: the Bot Mode title
* (server-synced via ui_meta, locally stored, or persisted on a durable
* group descriptor) and the core profile display_name — in displayName's
* precedence order. Remote rows never borrow local meta (two `default`s
* must not share a title). */
function botFriendlyNames(bot) {
const localTitle = !bot?.remoteSource && typeof $botMeta !== 'undefined' ? $botMeta.get()?.[bot?.name]?.title : null
return [bot?.ui_meta?.['hermes-bots']?.title, localTitle, bot?.title, bot?.display_name]
}
/** The tag autocomplete inserts for a bot: the renamed (friendly) slug when
* the user gave the bot a real name, otherwise the profile @handle. The
* resolvers accept both, so older muscle memory keeps working. */
function botMentionTag(bot) {
for (const friendly of botFriendlyNames(bot)) {
const forms = mentionNameForms(friendly)
if (forms.length) {
return forms[0]
}
}
return botHandle(bot?.name, bot)
}
function isActiveRosterBot(bot, active) {
const activeName = String(active?.name || 'default').trim() || 'default'
const activeId = String(active?.connectionId || '').trim()
@@ -3007,6 +3054,15 @@ function resolveRosterMentions(text, roster, active = {}) {
forms.add(String(bot.handle).toLowerCase())
}
// Renamed bots are taggable by their friendly names too — the core
// profile display_name and the Bot Mode title (issue: renaming a bot
// didn't change what you @-tag it with).
for (const friendly of botFriendlyNames(bot)) {
for (const form of mentionNameForms(friendly)) {
forms.add(form)
}
}
for (const form of forms) {
if (!form) {
continue
@@ -3697,15 +3753,24 @@ function groupChatMemberBots(group, roster, metaByName) {
* source's row may become remote after a connection switch, so retaining it
* here is what keeps the same room intact across machines. */
function durableGroupChatMembers(bots) {
return (bots || []).map(bot => ({
name: bot.name,
handle: bot.handle || bot.name,
connectionId: bot.connectionId,
connectionKind: bot.connectionKind,
connectionLabel: bot.connectionLabel,
remoteSource: true,
sourceScoped: true
}))
return (bots || []).map(bot => {
// Keep the friendly identity on the stored descriptor: after a
// connection switch the live roster row may be gone, and renamed-tag
// mentions must still resolve against the persisted member.
const title = String(botRosterMeta(bot, $botMeta.get())?.title || bot.ui_meta?.['hermes-bots']?.title || bot.title || '').trim()
return {
name: bot.name,
handle: bot.handle || bot.name,
...(title ? { title } : {}),
...(bot.display_name ? { display_name: bot.display_name } : {}),
connectionId: bot.connectionId,
connectionKind: bot.connectionKind,
connectionLabel: bot.connectionLabel,
remoteSource: true,
sourceScoped: true
}
})
}
/** Existing group names, alphabetical — feeds the Manage-groups dialog. */
@@ -3774,6 +3839,15 @@ function parseGroupChatMentions(text, members) {
: [])
])
// Renamed members answer to their friendly names too (profile
// display_name and Bot Mode title), in slugged and collapsed forms —
// the same tags the roster autocomplete inserts.
for (const friendly of botFriendlyNames(member)) {
for (const form of mentionNameForms(friendly)) {
forms.add(form)
}
}
for (const form of forms) {
if (form) {
handles.set(form, groupMemberKey(member))
@@ -8658,14 +8732,26 @@ function GroupMentionInput({ members, onChange, value, ...inputProps }) {
for (const member of members) {
const handle = String(member.handle || botHandle(member.name, member) || '').trim()
const display = displayName(member, botRosterMeta(member, allMeta))
// Renamed members complete on their friendly tag; parser resolves both.
const tag = String(botMentionTag(member) || handle).trim()
if (!handle || (token.query && !handle.toLowerCase().startsWith(token.query))) {
if (!tag) {
continue
}
if (
token.query &&
!tag.toLowerCase().startsWith(token.query) &&
!(handle && handle.toLowerCase().startsWith(token.query)) &&
!display.toLowerCase().startsWith(token.query)
) {
continue
}
options.push({
handle,
meta: displayName(member, botRosterMeta(member, allMeta))
handle: tag,
meta: display
})
}
}
@@ -10109,17 +10195,25 @@ export default {
}
const handle = botHandle(profile.name, profile)
const display = displayName(profile, $botMeta.get()[profile.name])
// Renamed bots complete on their friendly name — the tag is the
// renamed slug when one exists, the profile handle otherwise.
const tag = botMentionTag(profile)
if (q && !handle.toLowerCase().startsWith(q)) {
if (
q &&
!tag.toLowerCase().startsWith(q) &&
!handle.toLowerCase().startsWith(q) &&
!display.toLowerCase().startsWith(q)
) {
continue
}
const display = displayName(profile, $botMeta.get()[profile.name])
const source = profile.connectionLabel ? ` · ${profile.connectionLabel}` : ''
items.push({
insert: `@${handle}`,
display: `@${handle}`,
insert: `@${tag}`,
display: `@${tag}`,
meta: `Bot · ${display}${source}`
})
}
@@ -10397,30 +10491,14 @@ export default {
let mentionedBots = roster ? resolveRosterMentions(text, roster, live) : []
if (!roster) {
let names = []
try {
const res = await host.request('profiles.list', { include_sessions: false })
names = (res?.profiles ?? []).map(p => p.name)
// Same resolver as the cached path — renamed bots (display_name
// / ui_meta title) stay taggable when the roster cache is cold.
mentionedBots = resolveRosterMentions(text, res?.profiles ?? [], live).map(bot => ({ ...bot, remoteSource: false }))
} catch {
return draft
}
const prose = text.replace(/```[\s\S]*?```/g, ' ').replace(/`[^`\n]*`/g, ' ')
const mentioned = []
for (const match of prose.matchAll(/(^|\s)@([a-z0-9][a-z0-9_-]*)/gi)) {
let name = match[2].toLowerCase()
if (name === 'hermes' && !names.includes('hermes') && names.includes('default')) {
name = 'default'
}
if (names.includes(name) && name !== live.name && !mentioned.includes(name)) {
mentioned.push(name)
}
}
mentionedBots = mentioned.map(name => ({ name }))
}
if (!mentionedBots.length) {
@@ -0,0 +1,188 @@
import assert from 'node:assert/strict'
import { readFileSync } from 'node:fs'
import test from 'node:test'
import vm from 'node:vm'
// Renamed bots stay taggable (Discord report, Aug 2026): renaming a bot —
// Bot Mode title or `hermes profile rename` display_name — must update what
// the user can @-tag it with, in the mention middleware, group rooms, and
// the composer autocomplete. Old profile handles keep resolving.
const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8')
function runtime({ meta } = {}) {
const context = {
console,
setTimeout,
clearTimeout,
Date,
URL,
atom: initial => {
let value = initial
return { get: () => value, set: next => (value = next), listen: () => () => undefined }
},
host: {
request: async () => ({}),
requestProfile: async () => ({}),
state: {
profile: { get: () => 'default', listen: () => undefined },
connectionId: { get: () => 'local', listen: () => undefined },
gateway: { listen: () => undefined }
}
},
document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } }
}
const code = source
.replace(/^import\s+\*\s+as\s+sdk\s+from '@hermes\/plugin-sdk'\r?\n/m, '')
.replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '')
.replace(/^const \{ McpTab, ToolsetConfigPanel \} = sdk\r?\n/m, '')
.replace(/^import .* from 'react'\r?\n/m, '')
.replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '')
.replace('export default {', 'globalThis.plugin = {')
.concat(
'\nglobalThis.__x = { mentionNameForms, botFriendlyNames, botMentionTag, resolveRosterMentions, parseGroupChatMentions, groupMemberKey, durableGroupChatMembers, $botMeta };\n'
)
vm.runInNewContext(code, context, { filename: 'plugin.js' })
if (meta) {
context.__x.$botMeta.set(meta)
}
return context.__x
}
test('mentionNameForms: slugged + collapsed, reserved tokens dropped', () => {
const { mentionNameForms } = runtime()
assert.deepEqual(JSON.parse(JSON.stringify(mentionNameForms('Research Buddy'))), ['research-buddy', 'researchbuddy'])
assert.deepEqual(JSON.parse(JSON.stringify(mentionNameForms('Ops'))), ['ops'])
// A bot renamed "Hermes" or "everyone" cannot hijack reserved tags.
assert.equal(mentionNameForms('Hermes').length, 0)
assert.equal(mentionNameForms('@everyone').length, 0)
assert.equal(mentionNameForms('').length, 0)
})
test('botMentionTag: renamed slug wins, profile handle is the fallback', () => {
const { botMentionTag } = runtime({ meta: { writer: { title: 'Research Buddy' } } })
assert.equal(botMentionTag({ name: 'writer' }), 'research-buddy')
assert.equal(botMentionTag({ name: 'ops' }), 'ops')
assert.equal(botMentionTag({ name: 'default' }), 'hermes')
// display_name (hermes profile rename) drives the tag too.
assert.equal(botMentionTag({ name: 'scout', display_name: 'Deal Finder' }), 'deal-finder')
})
test('resolveRosterMentions: renamed bots resolve by friendly tag AND old handle', () => {
const { resolveRosterMentions } = runtime({ meta: { writer: { title: 'Research Buddy' } } })
const roster = [{ name: 'default' }, { name: 'writer' }, { name: 'scout', display_name: 'Deal Finder' }]
const live = { name: 'default', connectionId: 'local' }
const byTitle = resolveRosterMentions('hey @research-buddy check this', roster, live)
assert.deepEqual(JSON.parse(JSON.stringify(byTitle.map(b => b.name))), ['writer'])
const byDisplayName = resolveRosterMentions('ping @dealfinder please', roster, live)
assert.deepEqual(JSON.parse(JSON.stringify(byDisplayName.map(b => b.name))), ['scout'])
// The profile name keeps working after a rename.
const byName = resolveRosterMentions('hey @writer', roster, live)
assert.deepEqual(JSON.parse(JSON.stringify(byName.map(b => b.name))), ['writer'])
})
test('parseGroupChatMentions: display_name and ui_meta title forms address a member', () => {
const { parseGroupChatMentions, groupMemberKey } = runtime()
const members = [
{ name: 'research', title: '' },
{ name: 'builder', title: '', display_name: 'Site Smith' },
{ name: 'ops', title: '', ui_meta: { 'hermes-bots': { title: 'Night Watch' } } }
]
const bySlug = parseGroupChatMentions('@site-smith and @night-watch please', members)
assert.equal(bySlug.mentioned.has(groupMemberKey(members[1])), true)
assert.equal(bySlug.mentioned.has(groupMemberKey(members[2])), true)
assert.equal(bySlug.mentioned.has(groupMemberKey(members[0])), false)
const collapsed = parseGroupChatMentions('@sitesmith take a look', members)
assert.equal(collapsed.mentioned.has(groupMemberKey(members[1])), true)
})
test('durableGroupChatMembers persists the friendly identity for renamed-tag mentions', () => {
const { durableGroupChatMembers, $botMeta } = runtime()
$botMeta.set({ writer: { title: 'Research Buddy' } })
const [stored] = durableGroupChatMembers([{ name: 'writer', connectionId: 'local' }])
assert.equal(stored.title, 'Research Buddy')
const [remote] = durableGroupChatMembers([
{ name: 'scout', display_name: 'Deal Finder', connectionId: 'mac-mini', remoteSource: true }
])
assert.equal(remote.display_name, 'Deal Finder')
})
test('composer autocomplete offers the renamed tag and matches on the display name', () => {
const registrations = []
const atom = value => {
let current = value
return { get: () => current, set: next => (current = next), listen: () => () => undefined }
}
const jsx = (type, props = {}) => ({ type, props })
const roster = { profiles: [{ name: 'default' }, { name: 'writer' }, { name: 'ops' }] }
const context = {
atom,
jsx,
jsxs: jsx,
useQuery: () => ({}),
useValue: v => (v?.get ? v.get() : v),
useState: v => [v, () => undefined],
setTimeout,
clearTimeout,
performance,
window: {
setTimeout,
clearTimeout,
performance,
requestAnimationFrame: () => 0,
cancelAnimationFrame: () => undefined,
addEventListener: () => undefined,
removeEventListener: () => undefined
},
document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } },
queryClient: { getQueryData: () => roster, invalidateQueries: () => undefined, setQueryData: () => undefined },
host: {
state: { profile: { get: () => 'default', listen: () => () => undefined }, gateway: { get: () => null, listen: () => () => undefined } },
request: async () => ({}),
onEvent: () => () => undefined
},
COMPOSER_AREAS: { middleware: 'composer.middleware', atCompletions: 'composer.atCompletions' },
sdk: new Proxy({}, { get: () => undefined })
}
const code = source
.replace(/^import \* as sdk from '@hermes\/plugin-sdk'\r?\n/m, '')
.replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '')
.replace(/^import .* from 'react'\r?\n/m, '')
.replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '')
.replace('export default {', 'globalThis.plugin = {')
.concat('\nglobalThis.__botMeta = $botMeta;')
vm.runInNewContext(code, context)
context.__botMeta.set({ writer: { title: 'Research Buddy' } })
try {
context.plugin.register({
register: c => registrations.push(c),
storage: { get: async () => undefined, set: async () => undefined }
})
} catch {
/* registration walks UI surfaces the stub doesn't fully model */
}
const provide = registrations.find(c => c.area === 'composer.atCompletions').data.provide
// The renamed bot completes under its friendly prefix and inserts the tag.
const renamed = provide('rese')
assert.deepEqual(JSON.parse(JSON.stringify(renamed.map(i => i.insert))), ['@research-buddy'])
// Old profile-handle muscle memory still finds it (insert is the new tag).
const byHandle = provide('wri')
assert.deepEqual(JSON.parse(JSON.stringify(byHandle.map(i => i.insert))), ['@research-buddy'])
// Un-renamed bots keep their plain handle.
const plain = provide('op')
assert.deepEqual(JSON.parse(JSON.stringify(plain.map(i => i.insert))), ['@ops'])
})
+16 -1
View File
@@ -1,6 +1,12 @@
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { canOpenNewWindow, canOpenSessionWindow, openNewWindow, openSessionInNewWindow } from './windows'
import {
canOpenNewWindow,
canOpenSessionWindow,
isPeerInstanceWindow,
openNewWindow,
openSessionInNewWindow
} from './windows'
const desktopWindow = window as unknown as { hermesDesktop?: Window['hermesDesktop'] }
const initialHermesDesktop = desktopWindow.hermesDesktop
@@ -50,6 +56,15 @@ describe('canOpenSessionWindow', () => {
})
})
describe('isPeerInstanceWindow', () => {
it('recognizes only the full peer marker', () => {
expect(isPeerInstanceWindow('?peer=1')).toBe(true)
expect(isPeerInstanceWindow('?peer=0')).toBe(false)
expect(isPeerInstanceWindow('?win=secondary')).toBe(false)
expect(isPeerInstanceWindow('')).toBe(false)
})
})
describe('openSessionInNewWindow', () => {
it('no-ops without a session id', async () => {
const open = vi.fn().mockResolvedValue({ ok: true })
+12
View File
@@ -83,6 +83,18 @@ export function isWatchWindow(): boolean {
// keystroke into N prompts, and a HUD is the last place to paint onboarding.
export const isAuxiliaryWindow = (): boolean => isSecondaryWindow() || isHudWindow()
// A full peer window renders the ordinary app shell against the backend that
// Electron already has running. It is not an auxiliary/specialized renderer,
// but it must not replay the primary window's app-launch source restoration
// after boot and silently re-home itself to another registered gateway.
export function isPeerInstanceWindow(search = typeof window === 'undefined' ? '' : window.location.search): boolean {
try {
return new URLSearchParams(search).get('peer') === '1'
} catch {
return false
}
}
// The profile a helper window (the HUD) was asked to boot against, carried in
// the query string by the main process (see hudUrl). The HUD is a full app
// renderer that otherwise adopts the PRIMARY backend's profile — wrong the
-5
View File
@@ -833,11 +833,6 @@ memory:
# every N user turns. Set to 0 to disable. Only active when memory is enabled.
nudge_interval: 10 # Nudge every 10 user turns (0 = disabled)
# Memory flush: give the agent one turn to save memories before context is
# lost (compression, /new, /reset, exit). Set to 0 to disable.
# For exit/reset, only fires if the session had at least this many user turns.
flush_min_turns: 6 # Min user turns to trigger flush on exit/reset (0 = disabled)
# =============================================================================
# Session Reset Policy (Messaging Platforms)
# =============================================================================
+133 -7
View File
@@ -1866,9 +1866,14 @@ def _setup_worktree(repo_root: str = None, sync_base: bool = True,
"-c", "checkout.thresholdForParallelism=100",
]
try:
# 120s, not 30: on a multi-agent box the ~10k-file materialization
# contends with sibling sessions' checkouts/fetches/Electron dev
# builds for the same disk — measured 113s wall at near-zero CPU
# under load vs 1.2s idle (Aug 2026). A too-tight timeout kills a
# legitimately slow create and wastes the work already done.
result = subprocess.run(
["git", *_wt_add_cfg, "worktree", "add", str(wt_path), "-b", branch_name, base_ref],
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=30, cwd=repo_root,
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=120, cwd=repo_root,
)
if result.returncode != 0:
# If branching from the resolved remote ref failed for any reason
@@ -1883,7 +1888,7 @@ def _setup_worktree(repo_root: str = None, sync_base: bool = True,
base_ref, base_label = "HEAD", "HEAD (fallback — remote base failed)"
result = subprocess.run(
["git", "worktree", "add", str(wt_path), "-b", branch_name, base_ref],
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=30, cwd=repo_root,
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=120, cwd=repo_root,
)
if result.returncode != 0:
_cleanup_failed_worktree_add(repo_root, wt_path, branch_name)
@@ -2292,6 +2297,69 @@ def _worktree_commits_all_merged_upstream(
return False
def _worktree_branch_pr_merged(
worktree_path: str,
timeout: int = 15,
cache: Optional[Dict[str, bool]] = None,
) -> bool:
"""Return whether the worktree branch's PR is MERGED on GitHub.
Escape hatch for the case ``git cherry`` cannot catch: a rebase-merge that
altered the diff (conflict resolution against a moved base, follow-up
commits added during salvage/CI-fix) changes the patch-id, so the local
commits are no longer patch-equivalent to anything upstream even though
the PR merged. Those trees survive the cherry check forever (Aug 2026:
12 of 22 "unpushed" trees on a loaded box had MERGED PRs).
GitHub's PR state is the authoritative merge signal, so a clean tree
whose branch has a MERGED PR is reaped. Verdicts are memoized keyed on
``(branch, head_sha)`` — MERGED is monotonic, so a True verdict is cached
permanently; False is never cached (the PR may merge later without new
local commits, which would leave the key unchanged).
Fails SAFE toward False (preserve): no gh binary, offline, rate-limited,
detached HEAD, or any parse failure keeps the tree.
"""
import subprocess
try:
head = subprocess.run(
["git", "rev-parse", "--abbrev-ref", "HEAD"],
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, cwd=worktree_path,
)
if head.returncode != 0:
return False
branch = head.stdout.strip()
if not branch or branch == "HEAD": # detached — no PR to look up
return False
cache_key = None
if cache is not None:
sha = subprocess.run(
["git", "rev-parse", "HEAD"],
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, cwd=worktree_path,
)
if sha.returncode == 0 and sha.stdout.strip():
cache_key = f"pr-merged:{branch}:{sha.stdout.strip()}"
if cache.get(cache_key) is True:
return True
result = subprocess.run(
["gh", "pr", "list", "--head", branch, "--state", "merged",
"--json", "number", "--limit", "1"],
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, cwd=worktree_path,
)
if result.returncode != 0:
return False
prs = json.loads(result.stdout or "[]")
merged = isinstance(prs, list) and len(prs) > 0
if merged and cache is not None and cache_key is not None:
cache[cache_key] = True
return merged
except Exception:
return False
def _worktree_lock_is_live(repo_root: str, worktree_path: str, timeout: int = 10):
"""Classify a worktree's git lock as live, dead, or absent.
@@ -2652,6 +2720,14 @@ def _prune_stale_worktrees(repo_root: str, max_age_hours: int = 24) -> None:
merged = _worktree_commits_all_merged_upstream(
str(entry), timeout=30, cache=snapshot
)
if not merged:
# Rebase-merge escape hatch: conflict resolution or follow-up
# commits change the patch-id, so cherry misses them — but
# GitHub knows the PR merged. Authoritative and cheap (~0.3s,
# memoized on (branch, head_sha) so it's paid once per tree).
merged = _worktree_branch_pr_merged(
str(entry), timeout=15, cache=snapshot
)
with cache_lock:
merge_cache.update(snapshot)
if not merged:
@@ -2741,12 +2817,31 @@ def _prune_stale_worktrees(repo_root: str, max_age_hours: int = 24) -> None:
if preserved_stale:
logger.warning(
"Preserving %d worktree(s) older than 7 days with unmerged work "
"(push or remove them to reclaim disk): %s",
"(run `hermes worktree prune` to review and reclaim): %s",
len(preserved_stale), ", ".join(sorted(preserved_stale)),
)
_prune_orphaned_branches(repo_root)
# Escalation notice: the startup pass is deliberately conservative, so
# installs accumulate preserved trees it can never reclaim. Once the
# footprint is clearly a problem (many trees or multi-GB), say so once
# per launch and name the attended reclaim command — silence here is how
# boxes reach 15GB+ of .worktrees/ without anyone noticing.
try:
from hermes_cli.worktree_gc import worktrees_summary
count, size_mb = worktrees_summary(repo_root)
if count >= 10 or (size_mb or 0) >= 5120:
size_txt = f"{size_mb / 1024:.1f}GB" if size_mb else "unknown size"
logger.warning(
".worktrees/ holds %d tree(s) (%s) — run `hermes worktree list` "
"to audit and `hermes worktree prune` to reclaim safely.",
count, size_txt,
)
except Exception:
pass
def _prune_orphaned_branches(repo_root: str) -> None:
"""Delete local ``hermes/hermes-*`` and ``pr-*`` branches with no worktree.
@@ -4094,12 +4189,36 @@ _TERMINAL_INPUT_MODE_RESET_SEQ = (
"\x1b[0m" # reset text attributes
"\x1b[?25h" # ensure cursor visible
)
_EXTENDED_ENTER_KEYS_SEQ = "\x1b[>1u\x1b[>4;2m"
_KITTY_KEYBOARD_PUSH_SEQ = "\x1b[>1u"
_MODIFY_OTHER_KEYS_SEQ = "\x1b[>4;2m"
_EXTENDED_ENTER_KEYS_SEQ = _KITTY_KEYBOARD_PUSH_SEQ + _MODIFY_OTHER_KEYS_SEQ
_BACKSLASH_LINE_CONTINUATION_RE = re.compile(r"\\[ \t]*$")
def _is_ghostty_terminal(env: Optional[Mapping[str, str]] = None) -> bool:
"""Whether the terminal is Ghostty (either detection path).
Ghostty must be pushed ONLY modifyOtherKeys, not the Kitty keyboard
protocol: its Kitty disambiguate-mode implementation strips the Alt
modifier from the Backspace key, so Option+Backspace arrives as bare
\\x7f instead of the CSI-u form ``\\x1b[127;3u`` the protocol calls for
(upstream Ghostty bug), breaking backward-kill-word (#87630
regression). Ghostty implements modifyOtherKeys correctly (it then
emits ``\\x1b[27;3;127~``, which the alias table also maps).
Matches exactly the two conditions that admit Ghostty through
``_terminal_supports_extended_enter_keys``.
"""
if env is None:
env = os.environ
return (
(env.get("TERM_PROGRAM") or "").strip() == "ghostty"
or (env.get("TERM") or "").strip().lower() == "xterm-ghostty"
)
def _terminal_supports_extended_enter_keys(env: Optional[Mapping[str, str]] = None) -> bool:
"""Whether it is safe/useful to request modified Enter key reporting.
@@ -4131,7 +4250,9 @@ def _enable_extended_enter_keys(output=None, env: Optional[Mapping[str, str]] =
Writes the Kitty keyboard protocol push (CSI >1u, disambiguate mode) AND
xterm modifyOtherKeys level 2 (CSI >4;2m), mirroring the Ink TUI —
terminals honor whichever protocol they implement. Both are needed:
terminals honor whichever protocol they implement (except Ghostty, which
gets only modifyOtherKeys; see the Ghostty exception below). Both are
needed:
kitty-the-terminal removed modifyOtherKeys support entirely (it only
speaks its own protocol), while tmux/VS Code only accept modifyOtherKeys.
@@ -4148,20 +4269,25 @@ def _enable_extended_enter_keys(output=None, env: Optional[Mapping[str, str]] =
Ctrl+C, which is handled by prompt_toolkit's ``c-c`` binding (raw mode
clears ISIG, so the kernel INTR path was never in play for the CLI).
Ghostty exception: pushes only modifyOtherKeys — see
``_is_ghostty_terminal`` for the full rationale (#87630).
The exit reset sequence pops/resets both modes, so this is safe across
normal exits, Ctrl+C, and SIGTERM cleanup.
"""
if not _terminal_supports_extended_enter_keys(env):
return False
# Ghostty exception: only modifyOtherKeys — see _is_ghostty_terminal.
seq = _MODIFY_OTHER_KEYS_SEQ if _is_ghostty_terminal(env) else _EXTENDED_ENTER_KEYS_SEQ
try:
target = output
if target is not None and hasattr(target, "write_raw"):
target.write_raw(_EXTENDED_ENTER_KEYS_SEQ)
target.write_raw(seq)
target.flush()
return True
stream = sys.stdout
if stream is not None and stream.isatty():
stream.write(_EXTENDED_ENTER_KEYS_SEQ)
stream.write(seq)
stream.flush()
return True
except Exception:
+1
View File
@@ -0,0 +1 @@
acgh213
@@ -0,0 +1 @@
grahfmusic
@@ -0,0 +1,2 @@
Dhruv7201
# PR #89923 salvage (vLLM output-cap convergence)
+2
View File
@@ -0,0 +1,2 @@
ekinnee
# PR #72283 salvage (maximum output tokens parsing)
+52
View File
@@ -95,6 +95,9 @@ class RelayAdapter(BasePlatformAdapter):
# feedback off SendResult — see send()). Consumed by the gateway's
# semantic thread-rename lane; bounded like the sibling caches.
self._auto_thread_by_chat: Dict[str, Tuple[str, str]] = {}
# Bounded FIFO seen-set for inbound replay dedupe (finding #3);
# dict preserves insertion order, giving cheap oldest-first eviction.
self._seen_inbound: Dict[str, None] = {}
# chat_id -> draft_id of the currently OPEN native draft stream
# (NS-658 live cards). Armed by send_draft on a successful frame;
# consumed by send() to convert the turn-final delivery into the
@@ -892,6 +895,24 @@ class RelayAdapter(BasePlatformAdapter):
async def _on_inbound(self, event) -> None:
"""Bridge a connector-delivered MessageEvent into the normal adapter path."""
# Inbound replay dedupe (live-canary finding #3, Alice staging): the
# relay leg is at-least-once — on WS re-handshake the connector
# replays its durable per-instance buffer, and a long multi-tool turn
# (60-100s) straddling a quiet socket drop gets its ORIGINAL inbound
# replayed after the turn completes, re-running the whole turn (user
# saw the final answer 2-5x). Platform message identity (chat_id +
# message_id/ts) is stable across replays, so a bounded seen-set
# drops them. Consumer-side idempotency; no wire change.
dedupe_key = self._inbound_dedupe_key(event)
if dedupe_key is not None:
if dedupe_key in self._seen_inbound:
logger.info(
"relay inbound dropped as replay (dedupe key=%s)", dedupe_key
)
return
self._seen_inbound[dedupe_key] = None
while len(self._seen_inbound) > self._SEEN_INBOUND_MAX:
self._seen_inbound.pop(next(iter(self._seen_inbound)))
self._capture_scope(event)
self._stamp_slack_session_thread(event)
# Phase 3: a structured prompt answer resolves its waiting primitive
@@ -903,6 +924,37 @@ class RelayAdapter(BasePlatformAdapter):
await self._localize_inbound_media(event)
await self.handle_message(event)
_SEEN_INBOUND_MAX = 512
def _inbound_dedupe_key(self, event) -> Optional[str]:
"""Stable replay identity: (platform, chat, platform message id).
Chat identity lives on ``event.source`` (MessageEvent has no top-level
``chat_id``), and this adapter can front SEVERAL platforms over one
relay socket (Phase 1.5 multiplex), so the underlying platform joins
the key — two platforms' numeric chat/message ids must never collide
into one identity.
Returns None when the event carries no platform message id (synthetic
events, some prompt responses) — those never dedupe, fail-open by
design: dropping a real user message is strictly worse than rerunning
one, so only dedupe when identity is certain.
"""
source = getattr(event, "source", None)
message_id = getattr(event, "message_id", None)
chat_id = getattr(source, "chat_id", None)
if not message_id or not chat_id:
return None
# Normalize the platform component: production wire decoding always
# yields a Platform enum (unknowns canonicalize to Platform.RELAY),
# but alternate constructors may carry the plain string. Use the
# enum's value when present, the string itself otherwise — both
# spellings of one platform must produce ONE key, and two different
# string platforms must not collapse into the same empty component.
raw_platform = getattr(source, "platform", None)
platform = getattr(raw_platform, "value", raw_platform) or ""
return f"{platform}:{chat_id}:{message_id}"
def _relay_slack_extra(self) -> Dict[str, Any]:
"""The Slack-behavior subset of the RELAY platform config.
+124 -43
View File
@@ -497,10 +497,26 @@ class WebSocketRelayTransport:
# normal fast backoff, not the dormant cadence.
self._dormant = False
headers = self._upgrade_headers()
# WAN-friendly keepalive: customer gateways cross WAN paths to the
# connector; the websockets library default (ping_interval=20,
# ping_timeout=20) gives the peer only a 20s pong deadline, which
# produces spurious `1011 keepalive ping timeout` closes under
# transient latency / event-loop stalls (Coatue incident 2026-08-18).
# ping_timeout=60 tolerates such stalls while still detecting a dead
# link within ~90s worst case (30s interval + 60s pong deadline).
if headers:
self._ws = await websockets.connect(self._url, additional_headers=headers) # type: ignore[union-attr]
self._ws = await websockets.connect( # type: ignore[union-attr]
self._url,
additional_headers=headers,
ping_interval=30,
ping_timeout=60,
)
else:
self._ws = await websockets.connect(self._url) # type: ignore[union-attr]
self._ws = await websockets.connect( # type: ignore[union-attr]
self._url,
ping_interval=30,
ping_timeout=60,
)
self._reader = asyncio.create_task(self._read_loop(), name="relay-ws-reader")
# Send one hello PER fronted identity (Phase 1.5 Shape A). The connector
# accumulates them into its advertised set (the first sets the session
@@ -791,8 +807,10 @@ class WebSocketRelayTransport:
bot_id = self._bot_id_for(platform)
if bot_id:
frame["botId"] = bot_id
frame_sent = False
try:
await self._send(frame)
frame_sent = True
return await asyncio.wait_for(fut, timeout=self._outbound_timeout_s)
except asyncio.TimeoutError:
# AMBIGUOUS by contract (PR 85796 review): the frame reached the
@@ -807,6 +825,28 @@ class WebSocketRelayTransport:
"error": "relay outbound timed out",
"ambiguous": True,
}
except Exception as exc: # noqa: BLE001 - a dead socket is a failed send, not a raise
# No `is None` check can close the window where the socket dies
# BETWEEN the liveness guard above and the actual write — the
# reader's finally hasn't cleared _ws yet, so _send raises
# ConnectionClosed straight into callers whose contract is a
# result dict (RelayAdapter.send consumes it with no try).
# Report it like every other failed send. CancelledError is a
# BaseException, so cancellation still propagates.
#
# Ambiguity contract (PR 85796): a raise from the WRITE means the
# frame never reached the wire — definite non-delivery, no flag.
# A failure surfaced by the FUTURE (e.g. disconnect() failing
# pending mid-flight) means the frame WAS sent and only the
# outcome is unknown — mark it ambiguous like the timeout above.
logger.debug("relay %s send failed", frame_type, exc_info=True)
result: Dict[str, Any] = {
"success": False,
"error": f"relay send failed: {exc}",
}
if frame_sent:
result["ambiguous"] = True
return result
finally:
self._pending.pop(request_id, None)
@@ -817,50 +857,91 @@ class WebSocketRelayTransport:
await self._ws.send(json.dumps(frame) + "\n")
async def _read_loop(self) -> None:
assert self._ws is not None
# Bind the socket this reader serves: the finally below must only
# clear _ws if it still points at THIS socket (a supervisor re-dial
# may have already installed a fresh one by the time we unwind).
ws = self._ws
buf = ""
try:
async for chunk in self._ws:
buf += chunk if isinstance(chunk, str) else chunk.decode("utf-8")
# Newline-delimited frames; keep any trailing partial line.
*lines, buf = buf.split("\n")
for line in lines:
if line.strip():
await self._handle_frame(line)
except asyncio.CancelledError:
raise
except Exception as exc: # noqa: BLE001 - log + let the task end; reconnection handled below
# Phase 7 Unit 7d-B: detect a 4401 (unauthorized) close. After a prior
# successful handshake this is a REVOCATION (opt-out / deprovision) —
# the per-gateway secret is gone, so reconnecting is futile. Latch a
# terminal "auth revoked" state and DON'T re-dial. Before any
# successful handshake a 4401 stays retryable (cold-start race).
if self._close_code_of(exc) == _RELAY_UNAUTHORIZED_CLOSE_CODE and self._handshake_succeeded:
self._auth_revoked = True
if not self._closing:
logger.warning(
"relay ws closed 4401 (unauthorized) after a successful handshake — "
"treating as a revoked relay credential (opt-out); not reconnecting"
if ws is None:
# Scheduled without a socket (a lifecycle bug, not a normal
# path). The old `assert` here escaped BEFORE the finally
# existed to fail pending futures — the one exit that could
# still strand waiters for the full outbound timeout. Fall
# through to the finally instead; it settles them all.
logger.error("relay ws read loop started with no socket")
return
try:
async for chunk in self._ws:
buf += chunk if isinstance(chunk, str) else chunk.decode("utf-8")
# Newline-delimited frames; keep any trailing partial line.
*lines, buf = buf.split("\n")
for line in lines:
if line.strip():
await self._handle_frame(line)
except asyncio.CancelledError:
raise
except Exception as exc: # noqa: BLE001 - log + let the task end; reconnection handled below
# Phase 7 Unit 7d-B: detect a 4401 (unauthorized) close. After a prior
# successful handshake this is a REVOCATION (opt-out / deprovision) —
# the per-gateway secret is gone, so reconnecting is futile. Latch a
# terminal "auth revoked" state and DON'T re-dial. Before any
# successful handshake a 4401 stays retryable (cold-start race).
if self._close_code_of(exc) == _RELAY_UNAUTHORIZED_CLOSE_CODE and self._handshake_succeeded:
self._auth_revoked = True
if not self._closing:
logger.warning(
"relay ws closed 4401 (unauthorized) after a successful handshake — "
"treating as a revoked relay credential (opt-out); not reconnecting"
)
elif not self._closing:
logger.warning("relay ws read loop ended: %s", exc)
# Phase 5 §5.3: the socket closed. If reconnect is enabled and this was
# NOT a deliberate disconnect(), kick the reconnect supervisor so the
# gateway re-dials + re-handshakes (which triggers the connector's
# buffered-flip drain on the new handshake). Self-scheduling: the reader
# ends here, the supervisor re-dials and starts a fresh reader.
# Phase 7 Unit 7d-B: a revoked credential (terminal 4401) is the one case
# we deliberately do NOT reconnect — the secret is dead until the
# instance is recreated, so spinning would just reproduce the failure.
if (
self._reconnect
and not self._closing
and not self._auth_revoked
and (self._supervisor is None or self._supervisor.done())
):
self._supervisor = asyncio.create_task(
self._reconnect_loop(), name="relay-ws-reconnect"
)
finally:
# The socket this reader served is dead. Drop the handle (identity-
# guarded: a re-dial that already installed a FRESH socket must not
# be clobbered) so every `self._ws is None` liveness check — send,
# _request_response, go_idle, go_dormant — reports "not connected"
# for the whole outage. Without this, _ws kept pointing at the dead
# socket on every reader exit that arms NO supervisor (terminal
# 4401 revocation, reconnect=False transports), and a send there
# registered a future nothing could resolve: a full
# _outbound_timeout_s (~30s) wedge — including the revocation
# path's own fatal-error notification. disconnect() owns the
# handle during deliberate teardown, so leave it alone then.
if self._ws is ws and not self._closing:
self._ws = None
# The reader is the ONLY thing that can resolve a pending
# outbound_result future — once it exits (socket dropped, error,
# or cancellation cleanup) every in-flight _request_response waiter
# is unresolvable and would otherwise block the full
# _outbound_timeout_s (~30s) on a dead socket (Coatue incident
# 2026-08-18: stuck sends after a 1011 keepalive close). Fail them
# NOW with the dict shape callers expect (never an exception on
# the outbound path). list() snapshot: set_result wakes waiters
# whose finally-pop would otherwise mutate the dict mid-iteration.
for _rid, fut in list(self._pending.items()):
if not fut.done():
fut.set_result(
{"success": False, "error": "relay transport connection lost"}
)
elif not self._closing:
logger.warning("relay ws read loop ended: %s", exc)
# Phase 5 §5.3: the socket closed. If reconnect is enabled and this was
# NOT a deliberate disconnect(), kick the reconnect supervisor so the
# gateway re-dials + re-handshakes (which triggers the connector's
# buffered-flip drain on the new handshake). Self-scheduling: the reader
# ends here, the supervisor re-dials and starts a fresh reader.
# Phase 7 Unit 7d-B: a revoked credential (terminal 4401) is the one case
# we deliberately do NOT reconnect — the secret is dead until the
# instance is recreated, so spinning would just reproduce the failure.
if (
self._reconnect
and not self._closing
and not self._auth_revoked
and (self._supervisor is None or self._supervisor.done())
):
self._supervisor = asyncio.create_task(
self._reconnect_loop(), name="relay-ws-reconnect"
)
self._pending.clear()
@staticmethod
def _close_code_of(exc: BaseException) -> Optional[int]:
+49 -4
View File
@@ -1219,18 +1219,23 @@ class CLICommandsMixin:
self._handle_resume_command(f"/resume {arg}")
def _handle_worktree_command(self, cmd_original: str) -> None:
"""Handle /worktree — inspect or create isolated git worktrees.
"""Handle /worktree — inspect, create, or reclaim isolated git worktrees.
Syntax:
/worktree — show the active worktree (if any)
/worktree new [name] — create a worktree and move this session into it
/worktree list — list worktrees under the repo's .worktrees/
/worktree — show the active worktree (if any)
/worktree new [name] — create a worktree and move this session into it
/worktree list — list worktrees under the repo's .worktrees/
/worktree prune [--dry-run] — reclaim safe trees + merged branches
Inspired by Copilot CLI's ``/worktree new``: start isolated work in a
fresh worktree without leaving the session. Creating one retargets the
terminal/file tools (``TERMINAL_CWD`` + process cwd) at the new tree;
the launcher's exit cleanup applies (kept only when it has unpushed
commits, same as ``hermes -w``).
``prune`` is the same attended reclaim as ``hermes worktree prune``
(hermes_cli/worktree_gc.py): never deletes tracked changes, unique
unpushed commits, or in-use trees; archives untracked-only scratch.
"""
import subprocess
@@ -1250,10 +1255,50 @@ class CLICommandsMixin:
print(" No active worktree for this session.")
if repo_root:
print(" /worktree new [name] — create one and move this session into it")
print(" /worktree prune — reclaim stale trees and merged branches")
else:
print(" (not inside a git repository)")
return
if sub in {"prune", "gc", "clean"}:
if not repo_root:
print(" Not inside a git repository.")
return
rest = parts[2].strip().lower() if len(parts) > 2 else ""
dry_run = "--dry-run" in rest or "-n" in rest.split()
from hermes_cli import worktree_gc
active = _cli._active_worktree
tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False)
if active:
# Never reap the tree this very session is sitting in, even
# if a concurrent audit would judge it clean+merged.
active_path = str(active.get("path") or "")
tree_records = [
record for record in tree_records
if record.path != active_path
]
actions = worktree_gc.reclaim_worktrees(
repo_root, dry_run=dry_run, records=tree_records
)
actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run)
if actions:
for line in actions:
print(f" {line}")
print(f" {len(actions)} action(s) {'planned' if dry_run else 'done'}.")
else:
print(" Nothing to reclaim — remaining trees/branches carry real work.")
kept = [
record for record in tree_records
if record.verdict == "keep"
and "kanban" not in record.reason and "in use" not in record.reason
]
if kept:
print(f" Preserved {len(kept)} tree(s) with real work:")
for record in kept:
print(f" {record.name}: {record.reason}")
return
if sub in {"list", "ls"}:
if not repo_root:
print(" Not inside a git repository.")
+3 -3
View File
@@ -167,9 +167,9 @@ COMMAND_REGISTRY: list[CommandDef] = [
args_hint="<platform>", cli_only=True),
CommandDef("branch", "Branch the current session (explore a different path)", "Session",
aliases=("fork",), args_hint="[name]"),
CommandDef("worktree", "Show, list, or create isolated git worktrees for this session", "Session",
cli_only=True, args_hint="[new [name]|list]",
subcommands=("new", "list")),
CommandDef("worktree", "Show, list, create, or prune isolated git worktrees", "Session",
cli_only=True, args_hint="[new [name]|list|prune [--dry-run]]",
subcommands=("new", "list", "prune")),
CommandDef("compress", "Compress conversation context (add 'here [N]' to keep recent N turns; --preview shows what would happen)", "Session",
aliases=("compact",), args_hint="[here [N] | focus topic | --preview|--dry-run]"),
CommandDef("rollback", "List or restore filesystem checkpoints (restores keep your hand-edits; --all overrides)", "Session",
+4
View File
@@ -1868,6 +1868,10 @@ DEFAULT_CONFIG = {
"write_approval": False,
"memory_char_limit": 2200, # ~800 tokens at 2.75 chars/token
"user_char_limit": 1375, # ~500 tokens at 2.75 chars/token
# Periodic built-in memory review. External providers with automatic
# turn/session extraction can set this to 0 and keep the small local
# store reserved for explicit high-frequency operational facts.
"nudge_interval": 10,
# External memory provider plugin (empty = built-in only).
# Set to a provider name to activate: "openviking", "mem0",
# "hindsight", "holographic", "retaindb", "byterover".
+52 -1
View File
@@ -11644,7 +11644,7 @@ _BUILTIN_SUBCOMMANDS = frozenset(
"resume",
"send", "sessions", "setup",
"skin", "skills", "slack", "status", "sync", "tools", "uninstall", "update",
"version", "webhook", "whatsapp", "whatsapp-cloud", "chat", "secrets", "security",
"version", "webhook", "whatsapp", "whatsapp-cloud", "worktree", "chat", "secrets", "security",
"verify",
# Help-ish invocations — plugin commands not being listed in
# top-level --help is an acceptable trade-off for skipping an
@@ -12514,6 +12514,57 @@ def main():
)
fallback_parser.set_defaults(func=cmd_fallback)
# =========================================================================
# worktree command — audit/reclaim accumulated git worktrees + branches
# =========================================================================
worktree_parser = subparsers.add_parser(
"worktree",
help="Audit and reclaim accumulated git worktrees and merged branches",
description=(
"Attended reclaim for the .worktrees/ directory hermes -w sessions "
"accumulate. Never deletes uncommitted tracked changes, unique "
"unpushed commits, or in-use trees; untracked-only scratch is "
"archived to ~/.hermes/archive/worktree-prune/ before removal. See: "
"https://hermes-agent.nousresearch.com/docs/user-guide/cli#worktree-cleanup"
),
)
worktree_subparsers = worktree_parser.add_subparsers(dest="worktree_action")
worktree_list = worktree_subparsers.add_parser(
"list",
aliases=["ls", "audit"],
help="Classify every tree: age, size, verdict, reason (default action)",
)
worktree_list.add_argument("--repo", help="Repo root (default: current repo)")
worktree_prune = worktree_subparsers.add_parser(
"prune",
help="Remove safe trees and delete fully-merged local branches",
)
worktree_prune.add_argument("--repo", help="Repo root (default: current repo)")
worktree_prune.add_argument(
"--dry-run", action="store_true",
help="Show the plan without changing anything",
)
worktree_prune.add_argument(
"--trees-only", action="store_true",
help="Only remove worktrees; leave local branches alone",
)
worktree_prune.add_argument(
"--branches-only", action="store_true",
help="Only delete merged local branches; leave worktrees alone",
)
def _dispatch_worktree(_args):
from hermes_cli.worktree_cmd import cmd_worktree
# argparse aliases set dest to the literal typed string ("ls"/"audit").
action = getattr(_args, "worktree_action", None)
if action in ("ls", "audit"):
_args.worktree_action = "list"
return cmd_worktree(_args)
worktree_parser.set_defaults(func=_dispatch_worktree)
# =========================================================================
# secrets command — external secret managers (Bitwarden, 1Password)
# =========================================================================
+3 -2
View File
@@ -490,10 +490,11 @@ def cmd_status(args) -> None:
user_mark = "enabled ✓" if user_profile_enabled else "disabled ✗"
# Check if the memory tool is enabled for the CLI platform via the
# canonical resolver (handles composite toolsets like hermes-cli).
# canonical resolver and respects the check_fn gate when both stores are disabled.
from hermes_cli.tools_config import _get_platform_tools
from tools.memory_tool import check_memory_requirements
cli_tools = _get_platform_tools(config, "cli", include_default_mcp_servers=False)
memory_tool_enabled = "memory" in cli_tools
memory_tool_enabled = ("memory" in cli_tools) and check_memory_requirements()
tool_mark = "enabled ✓" if memory_tool_enabled else "disabled ✗"
print("\nMemory status\n" + "─" * 40)
+125 -19
View File
@@ -11,6 +11,24 @@ can be unit-tested without importing the whole CLI runtime.
from __future__ import annotations
# kitty CSI-u ORs lock-key state into the modifier parameter of every key
# event while a lock is on: CapsLock=64, NumLock=128, both=192 (#88221,
# #89651). Every fixed-modifier CSI-u (and legacy CSI-tilde / CSI-letter)
# registration therefore needs lock-offset twins, or those events leak into
# the prompt as literal text. The xterm modifyOtherKeys ``ESC[27;N;CP~``
# encoding never carries lock bits, so it never gets the twins.
_LOCK_BIT_OFFSETS = (0, 64, 128, 192)
def _lock_variants(modifier: int) -> tuple[int, ...]:
"""Return ``modifier`` plus its CapsLock/NumLock/both twins."""
return tuple(modifier + off for off in _LOCK_BIT_OFFSETS)
def _lock_twins(modifier: int) -> tuple[int, ...]:
"""Return only the lock twins of ``modifier`` (never the base value)."""
return tuple(modifier + off for off in _LOCK_BIT_OFFSETS[1:])
def _clear_vt100_prefix_cache() -> None:
"""Drop prompt_toolkit's memoized "is this a prefix of a longer match?"
@@ -37,6 +55,7 @@ def install_shift_enter_alias() -> int:
Sequences mapped:
- "\\x1b[13;2u" — Kitty keyboard protocol / CSI-u, modifier=2 (Shift)
(plus its CapsLock/NumLock lock twins via ``_lock_variants``)
- "\\x1b[27;2;13~" — xterm modifyOtherKeys=2, modifier=2 (Shift)
- "\\x1b[27;2;13u" — alternate ordering some emitters use
@@ -62,7 +81,9 @@ def install_shift_enter_alias() -> int:
alt_enter = (Keys.Escape, Keys.ControlM)
changed = 0
for seq in ("\x1b[13;2u", "\x1b[27;2;13~", "\x1b[27;2;13u"):
seqs = [f"\x1b[13;{m}u" for m in _lock_variants(2)]
seqs += ["\x1b[27;2;13~", "\x1b[27;2;13u"]
for seq in seqs:
if ANSI_SEQUENCES.get(seq) != alt_enter:
ANSI_SEQUENCES[seq] = alt_enter
changed += 1
@@ -78,10 +99,13 @@ def install_ctrl_enter_alias() -> int:
Sequences mapped:
- "\\x1b[13;5u" — Kitty keyboard protocol / CSI-u, modifier=5 (Ctrl)
(plus its CapsLock/NumLock lock twins via ``_lock_variants``)
- "\\x1b[27;5;13~" — xterm modifyOtherKeys=2, modifier=5 (Ctrl)
- "\\x1b[27;5;13u" — alternate ordering some emitters use
Stock prompt_toolkit doesn't map any of these. Without this alias,
Stock prompt_toolkit maps only the tilde form ``\\x1b[27;5;13~`` (to
plain ``Keys.ControlM``, which this deliberately overwrites — same
bug-fix rationale as install_shift_enter_alias). Without this alias,
Kitty/mintty/xterm-with-modifyOtherKeys users over SSH never get a
Ctrl+Enter newline — the keystroke arrives as a raw CSI sequence that
falls through to the default character-insert handler. See #22379.
@@ -96,7 +120,9 @@ def install_ctrl_enter_alias() -> int:
alt_enter = (Keys.Escape, Keys.ControlM)
changed = 0
for seq in ("\x1b[13;5u", "\x1b[27;5;13~", "\x1b[27;5;13u"):
seqs = [f"\x1b[13;{m}u" for m in _lock_variants(5)]
seqs += ["\x1b[27;5;13~", "\x1b[27;5;13u"]
for seq in seqs:
if ANSI_SEQUENCES.get(seq) != alt_enter:
ANSI_SEQUENCES[seq] = alt_enter
changed += 1
@@ -116,7 +142,8 @@ def install_cmd_backspace_alias() -> int:
literal insertion.
Cmd+Backspace → ``Keys.ControlU`` (kill backward to start of line).
Codepoint 127 with modifier 9 (super) / 10 (super+shift):
Codepoint 127 with modifier 9 (super) / 10 (super+shift), each with
its CapsLock/NumLock lock twins via ``_lock_variants``:
- ``\\x1b[127;9u`` / ``\\x1b[127;10u`` — Kitty CSI-u
- ``\\x1b[27;9;127~`` — xterm modifyOtherKeys
@@ -133,13 +160,12 @@ def install_cmd_backspace_alias() -> int:
except Exception:
return 0
aliases = {
"\x1b[127;9u": Keys.ControlU,
"\x1b[127;10u": Keys.ControlU,
"\x1b[27;9;127~": Keys.ControlU,
"\x1b[3;9~": Keys.ControlK,
"\x1b[3;10~": Keys.ControlK,
}
aliases: dict[str, object] = {}
for base in (9, 10): # super / super+shift
for mod in _lock_variants(base):
aliases[f"\x1b[127;{mod}u"] = Keys.ControlU
aliases[f"\x1b[3;{mod}~"] = Keys.ControlK
aliases["\x1b[27;9;127~"] = Keys.ControlU
changed = 0
for seq, key in aliases.items():
if ANSI_SEQUENCES.get(seq) != key:
@@ -181,6 +207,11 @@ def install_modify_other_keys_aliases() -> int:
Ctrl+Alt+Shift=8): normalized onto the same targets — Ctrl-bearing
combos behave as the Ctrl key (Alt adds an ``Escape`` prefix),
matching how dte/kakoune normalize these protocols.
* **Lock-bit variants**: every CSI-u mapping above is also installed
with the CapsLock (64) and NumLock (128) bits ORed into the modifier
parameter — kitty/ghostty include them while a lock is on, and
without the variants every key combo dies with the lock enabled
(``ESC[99;133u`` instead of ``ESC[99;5u``, #89651).
* **Esc key**: ``ESC[27u`` / ``ESC[27;<mod>u`` (Kitty disambiguate mode
reports Esc this way, #56684) → ``Keys.Escape``.
* **Modified Enter/Tab/Backspace/Space**: Alt+Enter → the Alt+Enter
@@ -244,15 +275,26 @@ def install_modify_other_keys_aliases() -> int:
changed = 0
# Kitty CSI-u encodes CapsLock/NumLock state as extra modifier bits
# (caps=64, num=128) ORed into the parameter: with NumLock on, Ctrl+C
# arrives as ESC[99;133u (5 + 128) instead of ESC[99;5u. Terminals
# that report these bits (kitty, ghostty) break every key combo while
# a lock is on (#89651) unless the lock variants are mapped too. The
# xterm modifyOtherKeys encoding never carries the lock bits, so only
# the CSI-u form needs them.
def _install_paired(modifier: int, mapping: dict) -> None:
"""Install both modifyOtherKeys (ESC[27;N;CP~) and CSI-u (ESC[CP;Nu)
mappings for the given modifier and codepoint→key mapping."""
mappings for the given modifier and codepoint→key mapping.
The tilde form is skipped for modifier 1 ("no modifier") — xterm
never emits modifier-1 tilde sequences.
"""
nonlocal changed
for codepoint, key_val in mapping.items():
for seq in (
f"\x1b[27;{modifier};{codepoint}~",
f"\x1b[{codepoint};{modifier}u",
):
seqs = [] if modifier == 1 else [f"\x1b[27;{modifier};{codepoint}~"]
for mod in _lock_variants(modifier):
seqs.append(f"\x1b[{codepoint};{mod}u")
for seq in seqs:
if seq not in ANSI_SEQUENCES:
ANSI_SEQUENCES[seq] = key_val
changed += 1
@@ -319,9 +361,16 @@ def install_modify_other_keys_aliases() -> int:
# Disambiguate mode reports the Esc key as CSI-u so it is
# distinguishable from the ESC byte that starts escape sequences
# (#56684 — previously leaked "[27u" as literal text into the prompt).
# Modifiers run to 16 because kitty reports Cmd as the super bit
# (mod 9+) — same reason install_cmd_backspace_alias maps 9/10.
for seq in ["\x1b[27u"] + [f"\x1b[27;{m}u" for m in range(2, 17)]:
# Modifiers run from 1 to 16: kitty reports Cmd as the super bit
# (mod 9+) — same reason install_cmd_backspace_alias maps 9/10 — and
# the lock-bit variants of the modifier-less form (1+64/128/192) are
# how a lone Esc keypress arrives with a lock on. Lock bits (caps/num)
# get the same variant treatment as _install_paired.
for seq in ["\x1b[27u"] + [
f"\x1b[27;{mod}u"
for m in range(1, 17)
for mod in _lock_variants(m)
]:
if seq not in ANSI_SEQUENCES:
ANSI_SEQUENCES[seq] = Keys.Escape
changed += 1
@@ -345,6 +394,57 @@ def install_modify_other_keys_aliases() -> int:
# matching Ink TUI + Desktop (#78285)
})
# -- Unmodified keys with a lock bit set (kitty modifier 1 = "none") --
# With a lock on, kitty stamps the lock bit onto keys pressed with NO
# real modifier too, so plain Backspace arrives as ESC[127;129u
# (1 + 128) rather than \x7f. _install_paired(1, ...) registers the
# bare mod-1 spelling and its lock twins. Only keys kitty CSI-u-encodes
# on their own are listed; plain text characters are still delivered
# as UTF-8, lock bits or not.
_install_paired(1, {
9: Keys.ControlI, # Tab
13: Keys.ControlM, # Enter
32: " ", # Space
127: Keys.ControlH, # Backspace
})
# -- Lock-key modifier bits (NumLock=128, CapsLock=64) on the legacy
# CSI-letter / CSI-tilde forms kitty keeps using under the disambiguate
# push: kitty encodes lock state into the modifier parameter, so a
# plain Down with NumLock on arrives as ESC[1;129B (NumLock), ESC[1;65B
# (CapsLock) or ESC[1;193B (both) instead of the legacy ESC[B — and a
# modified one shifts the same way (Alt+Left → ESC[1;131D). Those fall
# through the parser and leak as literal text ("[1;129B") in the input
# line. Derive the lock twins from whatever the table already maps for
# the base modifier (stock prompt_toolkit entries included), so every
# modifier the terminal can report keeps working under a lock.
for m in range(1, 17):
# CSI-letter navigation: Up/Down/Right/Left/End/Home + F1-F4
for trailer in "ABCDFHPQRS":
base_seq = f"\x1b[1;{m}{trailer}" if m > 1 else f"\x1b[{trailer}"
key = ANSI_SEQUENCES.get(base_seq)
if key is None and m == 1:
# Plain F1-F4 live in the table as SS3 (ESC O P) forms.
key = ANSI_SEQUENCES.get(f"\x1bO{trailer}")
if key is None:
continue
for mod in _lock_twins(m):
seq = f"\x1b[1;{mod}{trailer}"
if seq not in ANSI_SEQUENCES:
ANSI_SEQUENCES[seq] = key
changed += 1
# CSI-tilde navigation: Insert/Delete/PageUp/PageDown/Home/End
for num in (1, 2, 3, 4, 5, 6, 7, 8):
base_seq = f"\x1b[{num};{m}~" if m > 1 else f"\x1b[{num}~"
key = ANSI_SEQUENCES.get(base_seq)
if key is None:
continue
for mod in _lock_twins(m):
seq = f"\x1b[{num};{mod}~"
if seq not in ANSI_SEQUENCES:
ANSI_SEQUENCES[seq] = key
changed += 1
# -- Kitty functional keys (Private Use Area codepoints) ----
# kitty emits these CSI-u encodings even in LEGACY mode for keys that
# have no legacy encoding, so unmapped they leak as literal text in any
@@ -379,6 +479,12 @@ def install_modify_other_keys_aliases() -> int:
if seq not in ANSI_SEQUENCES:
ANSI_SEQUENCES[seq] = key_val
changed += 1
# Lock twins: with a lock on these arrive as ESC[<code>;129u etc.
for mod in _lock_twins(1):
seq = f"\x1b[{code};{mod}u"
if seq not in ANSI_SEQUENCES:
ANSI_SEQUENCES[seq] = key_val
changed += 1
# New longer sequences can flip "is this a prefix of a longer match?"
# answers the VT100 parser already cached — drop the cache so parsers
+99
View File
@@ -0,0 +1,99 @@
"""``hermes worktree`` — audit and reclaim accumulated git worktrees/branches.
Attended counterpart of the silent startup pruner (see
``hermes_cli/worktree_gc.py`` for the policy and shared invariants). Usage:
hermes worktree list # audit: verdict + reason per tree
hermes worktree prune # reap safe trees + merged branches
hermes worktree prune --dry-run # show the plan, change nothing
hermes worktree prune --trees-only / --branches-only
"""
from __future__ import annotations
from typing import Optional
def _repo_root() -> Optional[str]:
import cli as _cli
return _cli._git_repo_root()
def _fmt_size(size_mb: Optional[int]) -> str:
if size_mb is None:
return "?"
if size_mb >= 1024:
return f"{size_mb / 1024:.1f}G"
return f"{size_mb}M"
def cmd_worktree(args) -> int:
from hermes_cli import worktree_gc
repo_root = getattr(args, "repo", None) or _repo_root()
if not repo_root:
print("Not inside a git repository (or pass --repo <path>).")
return 1
action = getattr(args, "worktree_action", None) or "list"
if action == "list":
records = worktree_gc.audit_worktrees(repo_root)
if not records:
print("No worktrees under .worktrees/ — nothing to reclaim.")
return 0
total_mb = sum(r.size_mb or 0 for r in records)
reapable_mb = sum(
r.size_mb or 0 for r in records if r.verdict.startswith("reap")
)
print(f"{'TREE':32} {'AGE':>6} {'SIZE':>6} {'VERDICT':13} REASON")
for r in sorted(records, key=lambda x: -(x.size_mb or 0)):
print(
f"{r.name[:32]:32} {r.age_days:>5.1f}d {_fmt_size(r.size_mb):>6} "
f"{r.verdict:13} {r.reason}"
)
print(
f"\n{len(records)} tree(s), {_fmt_size(total_mb)} total — "
f"{_fmt_size(reapable_mb)} reclaimable now via `hermes worktree prune`."
)
branch_records = worktree_gc.audit_branches(repo_root)
deletable = [b for b in branch_records if b.verdict == "delete"]
if deletable:
print(
f"{len(deletable)} local branch(es) fully merged/patch-equivalent "
f"upstream would also be deleted."
)
return 0
if action == "prune":
dry_run = bool(getattr(args, "dry_run", False))
trees_only = bool(getattr(args, "trees_only", False))
branches_only = bool(getattr(args, "branches_only", False))
actions: list = []
if not branches_only:
tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False)
actions += worktree_gc.reclaim_worktrees(
repo_root, dry_run=dry_run, records=tree_records
)
kept = [r for r in tree_records if r.verdict == "keep"
and "kanban" not in r.reason and "in use" not in r.reason]
if kept:
print(f"Preserved {len(kept)} tree(s) with real work:")
for r in kept:
print(f" {r.name}: {r.reason}")
if not trees_only:
actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run)
if actions:
for line in actions:
print(f" {line}")
verb = "planned" if dry_run else "done"
print(f"{len(actions)} action(s) {verb}.")
else:
print("Nothing to reclaim — all trees/branches carry real work or are in use.")
return 0
print(f"Unknown worktree action: {action}")
return 1
+432
View File
@@ -0,0 +1,432 @@
"""On-demand worktree + branch reclaim (``hermes worktree`` / ``/worktree prune``).
The startup pruner in ``cli._prune_stale_worktrees`` is deliberately
conservative and silent: it runs before the banner on every ``hermes -w``
launch, so it only reaps clean, fully-merged scratch trees past an age tier
and preserves everything else. That policy is correct for an unattended
startup path — but it means real installs accumulate two kinds of debris the
startup pass can never touch:
- **Preserved trees** whose only "dirt" is untracked scratch (PR body drafts,
logs) on an otherwise merged branch — preserved forever by the dirty guard.
- **Orphaned local branches** beyond the two auto-generated prefixes the
startup pass deletes (``hermes/hermes-*``, ``pr-*``): salvage lanes, port
branches, feature branches whose PRs merged months ago. Multi-agent boxes
reach hundreds.
This module is the *attended* counterpart: an explicit, loud, dry-run-first
reclaim the user invokes, so it can be more thorough while staying just as
safe. Invariants shared with the startup pruner (never violated here either):
- tracked modifications are NEVER deleted, at any age, in any mode;
- unique unpushed commits are NEVER deleted (``git cherry`` patch-equivalence
decides "unique"; shallow repos are deepened bloblessly first so the
verdict is trustworthy);
- live-locked trees (owning pid alive) are never touched;
- a branch is deleted only after its worktree removal succeeded — a failed
removal must not orphan reachable commits;
- untracked-only dirt is ARCHIVED to ``~/.hermes/archive/worktree-prune/``
before its tree is reaped, never destroyed.
Classification primitives are imported from ``cli`` so the two paths can
never drift apart on what "dirty", "unpushed", or "merged" means.
"""
from __future__ import annotations
import logging
import os
import re
import shutil
import subprocess
import time
from dataclasses import dataclass, field
from pathlib import Path
from typing import List, Optional
logger = logging.getLogger(__name__)
# Branches never considered for deletion, in any mode.
_PROTECTED_BRANCHES = {"main", "master", "develop", "dev", "trunk"}
# Trees owned by another lifecycle (kanban dispatcher gc) — never touched.
_KANBAN_RE = re.compile(r"^t_[0-9a-f]+$")
# Bounded cherry probe: a branch this far ahead of upstream is a stale-base
# lane, not merged scratch; checking it is expensive and it stays preserved.
_MAX_CHERRY_AHEAD = 50
@dataclass
class TreeRecord:
name: str
path: str
branch: str
age_days: float
size_mb: Optional[int]
verdict: str # reap | reap-archive | keep
reason: str
untracked: List[str] = field(default_factory=list)
@dataclass
class BranchRecord:
name: str
verdict: str # delete | keep
reason: str
def _git(args: list, cwd: str, timeout: int = 15) -> subprocess.CompletedProcess:
"""Run git, translating timeouts into a nonzero returncode.
Every verdict in this module fails safe toward "keep" on a nonzero
returncode, so a hung/slow git call (large repos make ``git cherry``
genuinely slow) must degrade to keep — never crash the whole audit
(live-verified failure on a 746MB .git: TimeoutExpired escaped and
aborted the branch audit mid-list).
"""
try:
return subprocess.run(
["git", *args],
capture_output=True, text=True, encoding="utf-8",
errors="replace", timeout=timeout, cwd=cwd,
)
except subprocess.TimeoutExpired:
return subprocess.CompletedProcess(
args=["git", *args], returncode=124,
stdout="", stderr=f"timeout after {timeout}s",
)
def _tree_size_mb(path: Path) -> Optional[int]:
"""Cheap directory size via ``du -sm`` — best-effort, None on failure."""
try:
result = subprocess.run(
["du", "-sm", str(path)],
capture_output=True, text=True, encoding="utf-8",
errors="replace", timeout=30,
)
if result.returncode == 0 and result.stdout.strip():
return int(result.stdout.split()[0])
except Exception:
pass
return None
def _dirty_split(path: str) -> tuple[bool, List[str]]:
"""Return (has_tracked_modifications, untracked_paths).
``git status --porcelain`` counts untracked scratch equally with real
edits; the reclaim policy treats them very differently (tracked = real
work, untracked = archivable scratch), so split here.
"""
try:
result = _git(["status", "--porcelain"], cwd=path, timeout=10)
if result.returncode != 0:
return True, [] # fail safe: treat as real work
tracked = False
untracked: List[str] = []
for line in result.stdout.splitlines():
if not line.strip():
continue
if line.startswith("??"):
untracked.append(line[3:].strip())
else:
tracked = True
return tracked, untracked
except Exception:
return True, []
def _archive_untracked(tree: Path, untracked: List[str]) -> Optional[Path]:
"""Copy untracked files out of a doomed tree. Returns the archive dir.
Never destroys: on any copy failure the caller must treat the tree as
keep. Costs almost nothing and removes the "did I just delete
something?" question.
"""
stamp = time.strftime("%Y%m%d-%H%M%S")
dest = (
Path.home() / ".hermes" / "archive" / "worktree-prune"
/ f"{tree.name}-{stamp}"
)
try:
for rel in untracked:
src = tree / rel
if not src.exists() or src.is_symlink():
continue
target = dest / rel
target.parent.mkdir(parents=True, exist_ok=True)
if src.is_dir():
shutil.copytree(src, target, dirs_exist_ok=True)
else:
shutil.copy2(src, target)
return dest if dest.exists() else None
except Exception as exc:
logger.warning("Could not archive untracked files from %s: %s", tree, exc)
return None
def audit_worktrees(repo_root: str, *, with_sizes: bool = True) -> List[TreeRecord]:
"""Classify every tree under ``.worktrees/`` without mutating anything."""
import cli as _cli # lazy: cli.py is heavy
worktrees_dir = Path(repo_root) / ".worktrees"
if not worktrees_dir.exists():
return []
if _cli._repo_is_shallow(repo_root):
_cli._deepen_shallow_repo(repo_root)
merge_cache = _cli._load_worktree_merge_cache()
cache_size_before = len(merge_cache)
now = time.time()
records: List[TreeRecord] = []
for entry in sorted(worktrees_dir.iterdir()):
if not entry.is_dir():
continue
try:
age_days = (now - entry.stat().st_mtime) / 86400.0
except Exception:
continue
size_mb = _tree_size_mb(entry) if with_sizes else None
try:
branch_result = _git(["branch", "--show-current"], cwd=str(entry), timeout=5)
branch = branch_result.stdout.strip()
except Exception:
branch = ""
def rec(verdict: str, reason: str, untracked: Optional[List[str]] = None):
records.append(TreeRecord(
name=entry.name, path=str(entry), branch=branch,
age_days=age_days, size_mb=size_mb,
verdict=verdict, reason=reason,
untracked=untracked or [],
))
if _KANBAN_RE.match(entry.name):
rec("keep", "kanban task tree (owned by kanban gc)")
continue
lock_state = _cli._worktree_lock_is_live(repo_root, str(entry), timeout=5)
if lock_state == "live":
rec("keep", "in use by a running hermes session")
continue
tracked_dirty, untracked = _dirty_split(str(entry))
if tracked_dirty:
rec("keep", "uncommitted tracked changes (real work)")
continue
if _cli._worktree_has_unpushed_commits(str(entry), timeout=5):
merged = _cli._worktree_commits_all_merged_upstream(
str(entry), timeout=30, cache=merge_cache,
max_ahead=_MAX_CHERRY_AHEAD,
)
if not merged:
rec("keep", "unpushed commits not found upstream")
continue
if untracked:
rec("reap-archive",
f"merged/pushed; {len(untracked)} untracked file(s) will be archived",
untracked)
else:
rec("reap", "clean and fully merged/pushed")
if len(merge_cache) != cache_size_before:
_cli._save_worktree_merge_cache(merge_cache)
return records
def reclaim_worktrees(
repo_root: str,
*,
dry_run: bool = False,
records: Optional[List[TreeRecord]] = None,
) -> List[str]:
"""Remove every reap-verdict tree from a frozen audit list.
Operates ONLY on the provided (or freshly computed) audit records — never
re-globs inside the destructive loop, so trees created by concurrent
sessions after the audit are out of scope by construction.
"""
if records is None:
records = audit_worktrees(repo_root, with_sizes=False)
actions: List[str] = []
for record in records:
if record.verdict not in {"reap", "reap-archive"}:
continue
if dry_run:
actions.append(f"would remove {record.name} ({record.reason})")
continue
entry = Path(record.path)
if record.verdict == "reap-archive" and record.untracked:
archive = _archive_untracked(entry, record.untracked)
if archive is None:
actions.append(f"kept {record.name} (archive of untracked files failed)")
continue
actions.append(f"archived {len(record.untracked)} untracked file(s) → {archive}")
# Dead-pid locks must be unlocked or `remove --force` refuses.
try:
_git(["worktree", "unlock", record.path], cwd=repo_root, timeout=10)
except Exception:
pass
try:
remove_result = _git(
["worktree", "remove", record.path, "--force"],
cwd=repo_root, timeout=30,
)
if remove_result.returncode != 0:
actions.append(
f"failed to remove {record.name}: {remove_result.stderr.strip()}"
)
continue
if record.branch and record.branch not in _PROTECTED_BRANCHES:
_git(["branch", "-D", record.branch], cwd=repo_root, timeout=10)
actions.append(f"removed {record.name}")
except Exception as exc:
actions.append(f"failed to remove {record.name}: {exc}")
if not dry_run:
try:
_git(["worktree", "prune"], cwd=repo_root, timeout=15)
except Exception:
pass
return actions
def audit_branches(repo_root: str) -> List[BranchRecord]:
"""Classify local branches: safe to delete when their content is on
upstream (fully merged OR every commit patch-equivalent via ``git
cherry``) and they are not checked out anywhere.
Generalizes the startup pass's prefix list (``hermes/hermes-*``/``pr-*``)
to EVERY local branch, because deletion is gated on content reachability
rather than name: a branch whose commits are all upstream loses nothing
when its ref goes. Branch names checked out in any worktree, protected
names, and branches with unique commits are kept.
"""
import cli as _cli
if _cli._repo_is_shallow(repo_root):
_cli._deepen_shallow_repo(repo_root)
upstream = None
for candidate in ("origin/HEAD", "origin/main", "origin/master"):
probe = _git(["rev-parse", "--verify", "--quiet", candidate], cwd=repo_root, timeout=5)
if probe.returncode == 0:
upstream = candidate
break
if upstream is None:
return []
result = _git(["branch", "--format=%(refname:short)"], cwd=repo_root, timeout=10)
if result.returncode != 0:
return []
branches = [b.strip() for b in result.stdout.splitlines() if b.strip()]
active: set = set()
wt = _git(["worktree", "list", "--porcelain"], cwd=repo_root, timeout=10)
for line in wt.stdout.splitlines():
if line.startswith("branch refs/heads/"):
active.add(line.split("branch refs/heads/", 1)[-1].strip())
merged_result = _git(["branch", "--merged", upstream, "--format=%(refname:short)"],
cwd=repo_root, timeout=15)
merged = {b.strip() for b in merged_result.stdout.splitlines() if b.strip()}
def _classify_branch(branch: str) -> BranchRecord:
if branch in _PROTECTED_BRANCHES or branch in active:
return BranchRecord(branch, "keep", "protected or checked out")
if branch in merged:
return BranchRecord(branch, "delete", "fully merged into " + upstream)
# Rebase merges rewrite SHAs, so --merged misses them; cherry
# patch-equivalence catches the dominant leak. Bounded: a branch
# far ahead is a stale-base lane, keep it.
ahead = _git(["rev-list", "--count", f"{upstream}..{branch}"], cwd=repo_root, timeout=10)
try:
ahead_count = int(ahead.stdout.strip() or "0")
except ValueError:
ahead_count = _MAX_CHERRY_AHEAD + 1
if ahead_count == 0:
return BranchRecord(branch, "delete", "no commits beyond " + upstream)
if ahead_count > _MAX_CHERRY_AHEAD:
return BranchRecord(branch, "keep", f"{ahead_count} commits ahead (stale-base lane)")
cherry = _git(["cherry", upstream, branch], cwd=repo_root, timeout=30)
if cherry.returncode != 0:
return BranchRecord(branch, "keep", "could not verify (git cherry failed)")
lines = [ln for ln in cherry.stdout.splitlines() if ln.strip()]
if lines and all(ln.startswith("-") for ln in lines):
return BranchRecord(branch, "delete", "all commits patch-equivalent upstream")
unique = sum(1 for ln in lines if ln.startswith("+"))
return BranchRecord(branch, "keep", f"{unique} unique commit(s) not upstream")
# Read-only classification — parallel, like the tree audit (a busy
# multi-agent box carries hundreds of local branches; serial cherry
# probes at ~0.2-1s each make the audit minutes long).
import concurrent.futures
workers = max(1, min(8, (os.cpu_count() or 4), len(branches)))
if workers > 1:
try:
with concurrent.futures.ThreadPoolExecutor(
max_workers=workers, thread_name_prefix="hermes-branch-gc"
) as pool:
return list(pool.map(_classify_branch, branches))
except Exception:
pass
return [_classify_branch(b) for b in branches]
def reclaim_branches(
repo_root: str,
*,
dry_run: bool = False,
records: Optional[List[BranchRecord]] = None,
) -> List[str]:
"""Delete every delete-verdict branch from a frozen audit list."""
if records is None:
records = audit_branches(repo_root)
actions: List[str] = []
for record in records:
if record.verdict != "delete":
continue
if dry_run:
actions.append(f"would delete branch {record.name} ({record.reason})")
continue
result = _git(["branch", "-D", record.name], cwd=repo_root, timeout=10)
if result.returncode == 0:
actions.append(f"deleted branch {record.name}")
else:
actions.append(f"failed to delete {record.name}: {result.stderr.strip()}")
return actions
def worktrees_summary(repo_root: str) -> tuple[int, Optional[int]]:
"""(tree_count, total_size_mb) for the escalation notice. Size is
best-effort with a hard timeout so the startup path never stalls."""
worktrees_dir = Path(repo_root) / ".worktrees"
if not worktrees_dir.exists():
return 0, None
try:
count = sum(1 for e in worktrees_dir.iterdir() if e.is_dir())
except Exception:
return 0, None
size_mb: Optional[int] = None
try:
result = subprocess.run(
["du", "-sm", str(worktrees_dir)],
capture_output=True, text=True, encoding="utf-8",
errors="replace", timeout=20,
)
if result.returncode == 0 and result.stdout.strip():
size_mb = int(result.stdout.split()[0])
except Exception:
pass
return count, size_mb
+41 -4
View File
@@ -1042,6 +1042,26 @@ class AIAgent:
)
)
def _warn_uncompressed_context_overflow(
self, preflight_tokens: int, context_length: int
) -> None:
"""Surface a deduped warning when uncompressed context exceeds model limit.
When compression is explicitly disabled (compression.enabled: false), long
sessions can grow past the model context window with no compression to shrink
them (#89297). Surface an actionable warning so the user knows to run /compact
or enable compression.
"""
_warn_key = ("uncompressed_ctx_overflow", context_length)
if getattr(self, "_last_ctx_overflow_warn", None) != _warn_key:
self._last_ctx_overflow_warn = _warn_key
self._emit_warning(
f"⚠️ Session context (~{preflight_tokens:,} tokens) exceeds the model "
f"context window (~{context_length:,} tokens) with compression disabled "
f"(compression.enabled: false). Use /compact to compress history or "
f"enable compression in config.yaml."
)
def _clear_context_overflow_warn(self) -> None:
"""Reset the dedup state for the blocked-overflow warning.
@@ -8219,6 +8239,7 @@ class AIAgent:
function_result: str,
*,
failed: bool,
tool_call_id: str = "",
) -> str:
decision = self._tool_guardrails.after_call(
tool_name,
@@ -8226,20 +8247,36 @@ class AIAgent:
function_result,
failed=failed,
)
# Identical-call loop breaker (agent.stall_guards): notice-only, no
# Identical-call stall guards (agent.stall_guards): notice-only, no
# blocking. Observed on the RAW result (before the loop-warning suffix
# below, whose embedded count changes per call and would defeat
# result-identity matching). Appended here — at result construction,
# result-identity matching). Applied here — at result construction,
# before the tool message is built — so it is cache-safe (tool results
# are append-only; nothing already sent to the provider is mutated).
stall_notice = None
result_stub = None
if self._stall_guards_enabled():
try:
stall_notice = self._tool_guardrails.observe_identical_call(
tool_name, function_args, function_result,
observation = self._tool_guardrails.observe_call(
tool_name,
function_args,
function_result if isinstance(function_result, str) else None,
tool_call_id=tool_call_id,
failed=failed,
)
stall_notice = observation.notice
result_stub = observation.stub
except Exception as exc:
logger.debug("stall-guard identical-call observation failed: %s", exc)
# Result-reference stubbing: a 2nd+ consecutive identical call whose
# FRESH result is byte-identical enters context as a short reference
# stub instead of the duplicate payload. The tool still executed —
# this is not a cache; a changed result flows through whole. Only
# plain-string results are stubbed (multimodal content lists pass
# through untouched), and the current message keeps its role and
# tool_call_id — only the content is replaced.
if result_stub and isinstance(function_result, str):
function_result = result_stub
if decision.action in {"warn", "halt"}:
function_result = append_toolguard_guidance(function_result, decision)
if decision.should_halt:
@@ -203,3 +203,47 @@ class TestMinimumMessagesBranch:
assert len(out) == len(msgs), "nothing should have been compressed"
assert cc._last_compression_made_progress is False
assert cc._ineffective_compression_count == before + 1
class TestRejectedCompactionStrike:
"""#88568 — a would-grow refusal must count as an ineffective strike.
The anti-growth guard correctly keeps the original transcript, but the
rejection used to leave ``_ineffective_compression_count`` untouched, so
the breaker never latched and automatic compression retried the SAME
unchanged transcript on every turn.
"""
def test_rejected_compaction_increments_strike(self):
cc = _compressor(threshold_tokens=1)
assert cc._ineffective_compression_count == 0
cc.record_rejected_compaction()
assert cc._ineffective_compression_count == 1
def test_two_rejections_stop_further_automatic_compression(self):
cc = _compressor(threshold_tokens=1)
cc.record_rejected_compaction()
cc.record_rejected_compaction()
# The latch consumers key on the counter itself (>= 2 blocks);
# pin the counter and the recovery-clock arming side effect.
assert cc._ineffective_compression_count >= 2
def test_rejection_does_not_arm_real_usage_verification(self):
"""Nothing was committed, so the next response must not be scored
against the pre-rejection transcript (that verdict belongs to
committed compactions only)."""
cc = _compressor(threshold_tokens=1)
cc.record_rejected_compaction()
assert cc._verify_compaction_cleared_threshold is False
def test_rejection_leaves_fallback_streak_untouched(self):
cc = _compressor(threshold_tokens=1)
cc._fallback_compression_streak = 1
cc.record_rejected_compaction()
assert cc._fallback_compression_streak == 1
@@ -0,0 +1,181 @@
"""Mechanical salvage for compression candidates that would grow."""
from agent.context_compressor import (
COMPRESSED_SUMMARY_METADATA_KEY,
_SUMMARY_END_MARKER,
salvage_grown_transcript,
)
from agent.model_metadata import estimate_messages_tokens_rough
def test_salvage_stubs_old_tools_and_keeps_todo_when_stubbing_suffices():
"""Tool stubbing alone gets under budget → the todo snapshot survives.
The snapshot is the only in-transcript todo re-injection at the boundary
(and may carry the pruned-skill reload notice), so it is last-resort only.
"""
original = [
{"role": "user", "content": "go"},
{"role": "assistant", "content": "ok"},
{"role": "tool", "tool_call_id": "a", "content": "A" * 4000},
{"role": "tool", "tool_call_id": "b", "content": "B" * 4000},
{"role": "tool", "tool_call_id": "c", "content": "keep-latest"},
]
grown = original + [
{
"role": "user",
"content": "Current todos:\n- [ ] x",
"_todo_snapshot_synthetic": True,
}
]
assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original)
out = salvage_grown_transcript(original, grown)
assert out is not None
assert estimate_messages_tokens_rough(out) < estimate_messages_tokens_rough(original)
assert any(m.get("_todo_snapshot_synthetic") for m in out)
tools = [m["content"] for m in out if m.get("role") == "tool"]
assert tools[-1] == "keep-latest"
assert any("cleared to save context space" in t for t in tools)
def test_salvage_drops_todo_only_as_last_resort():
"""When cheaper ops cannot get under budget, the snapshot is dropped."""
original = [
{"role": "user", "content": "please do the thing " + ("o" * 600)},
{"role": "assistant", "content": "ok"},
]
grown = [
{"role": "user", "content": "summary of the ask"},
{"role": "assistant", "content": "ok"},
{
"role": "user",
"content": "Current todos:\n- [ ] " + ("t" * 800),
"_todo_snapshot_synthetic": True,
},
]
assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original)
out = salvage_grown_transcript(original, grown)
assert out is not None
assert estimate_messages_tokens_rough(out) < estimate_messages_tokens_rough(original)
assert not any(m.get("_todo_snapshot_synthetic") for m in out)
def test_salvage_last_resort_preserves_pruned_skill_reload_notice():
"""7a16840add couples the reload notice into the snapshot — it survives."""
from agent.conversation_compression import _PRUNED_SKILL_RELOAD_NOTICE_HEADER
notice = (
f"{_PRUNED_SKILL_RELOAD_NOTICE_HEADER}\n"
"Reload with skill_view(name='example-skill') before acting."
)
original = [
{"role": "user", "content": "please do the thing " + ("o" * 3000)},
{"role": "assistant", "content": "ok"},
]
grown = [
{"role": "user", "content": "summary of the ask"},
{"role": "assistant", "content": "ok"},
{
"role": "user",
"content": "Current todos:\n- [ ] " + ("t" * 4000) + f"\n\n{notice}",
"_todo_snapshot_synthetic": True,
},
]
assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original)
out = salvage_grown_transcript(original, grown)
assert out is not None
assert estimate_messages_tokens_rough(out) < estimate_messages_tokens_rough(original)
snapshot_rows = [m for m in out if m.get("_todo_snapshot_synthetic")]
assert len(snapshot_rows) == 1
assert snapshot_rows[0]["content"].startswith(_PRUNED_SKILL_RELOAD_NOTICE_HEADER)
assert "Current todos" not in snapshot_rows[0]["content"]
def test_salvage_returns_none_when_nothing_can_shrink():
original = [{"role": "user", "content": "tiny"}]
huge = [{"role": "user", "content": "X" * 200_000}]
assert salvage_grown_transcript(original, huge) is None
def test_salvage_caps_oversized_summary():
original = [
{"role": "user", "content": "ask " + ("o" * 12_000)},
{"role": "assistant", "content": "short reply"},
]
grown = [
{
"role": "user",
"content": (
"[CONTEXT COMPACTION] "
+ ("S" * 20_000)
+ "\n\n"
+ _SUMMARY_END_MARKER
),
COMPRESSED_SUMMARY_METADATA_KEY: True,
},
{"role": "assistant", "content": "short reply"},
]
assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original)
out = salvage_grown_transcript(original, grown)
assert out is not None
assert len(out[0]["content"]) < 12_000
assert "truncated so compaction can shrink" in out[0]["content"]
assert out[0]["content"].endswith(_SUMMARY_END_MARKER)
def test_salvage_never_truncates_merged_summary_with_live_user_tail():
original = [{"role": "user", "content": "O" * 12_000}]
merged = [
{
"role": "user",
"content": (
"[CONTEXT COMPACTION] "
+ ("S" * 20_000)
+ "\n\n"
+ _SUMMARY_END_MARKER
+ "\n\nLIVE USER REQUEST"
),
COMPRESSED_SUMMARY_METADATA_KEY: True,
}
]
assert salvage_grown_transcript(original, merged) is None
assert merged[0]["content"].endswith("LIVE USER REQUEST")
def test_salvage_does_not_cap_plain_user_text_quoting_summary_marker():
original = [{"role": "user", "content": "O" * 20_000}]
quoted = "ordinary user text " + ("Q" * 12_000) + "\n\n" + _SUMMARY_END_MARKER
candidate = [{"role": "user", "content": quoted}]
out = salvage_grown_transcript(original, candidate)
assert out is not None
assert out[0]["content"] == quoted
assert "truncated so compaction can shrink" not in out[0]["content"]
def test_salvage_never_caps_unmarked_summary_shaped_live_user_text():
original = [{"role": "user", "content": "O" * 20_000}]
live_user_text = (
"[CONTEXT COMPACTION] "
+ ("U" * 12_000)
+ "\n\n"
+ _SUMMARY_END_MARKER
)
candidate = [{"role": "user", "content": live_user_text}]
out = salvage_grown_transcript(original, candidate)
assert out is not None
assert out[0]["content"] == live_user_text
assert "truncated so compaction can shrink" not in out[0]["content"]
+169 -2
View File
@@ -17,6 +17,7 @@ These assert behavior contracts, not message snapshots.
from agent.agent_runtime_helpers import trailing_continue_intent
from agent.tool_guardrails import (
IDENTICAL_RESULT_STUB_MIN_CHARS,
STALL_GUARD_IDENTICAL_CALL_THRESHOLD,
STALL_GUARD_REPEATABLE_TOOLS,
ToolCallGuardrailController,
@@ -142,8 +143,10 @@ def _fake_agent(stall_guards=True):
lambda decision: AIAgent._set_tool_guardrail_halt(agent, decision)
)
agent._append = (
lambda name, args, result: AIAgent._append_guardrail_observation(
agent, name, args, result, failed=False
lambda name, args, result, failed=False, tool_call_id="": (
AIAgent._append_guardrail_observation(
agent, name, args, result, failed=failed, tool_call_id=tool_call_id
)
)
)
return agent
@@ -181,6 +184,170 @@ def test_notice_streak_keys_on_raw_result_not_annotated_result():
assert "hermes note" in outs[2]
# ── result-reference stubbing (byte-identical duplicate results) ──────────
_BIG = "x" * IDENTICAL_RESULT_STUB_MIN_CHARS # exactly at the stub threshold
def test_stub_on_second_identical_call_first_full():
agent = _fake_agent()
args = {"query": "hermes"}
r1 = agent._append("web_search", args, _BIG, tool_call_id="call_1")
r2 = agent._append("web_search", args, _BIG, tool_call_id="call_2")
assert r1 == _BIG # first occurrence always enters context whole
assert r2 != _BIG
assert "byte-identical" in r2
assert "web_search" in r2
assert "call_1" in r2 # references the FIRST occurrence in the streak
assert len(r2) < len(_BIG)
def test_no_stub_when_fresh_result_differs():
agent = _fake_agent()
args = {"query": "hermes"}
agent._append("web_search", args, _BIG, tool_call_id="call_1")
changed = "y" + _BIG
r2 = agent._append("web_search", args, changed, tool_call_id="call_2")
assert r2 == changed # changed result flows through whole
def test_changed_result_resets_streak_then_stub_references_new_first():
agent = _fake_agent()
args = {"id": "job"}
agent._append("web_search", args, _BIG, tool_call_id="a")
changed = _BIG + "done"
r2 = agent._append("web_search", args, changed, tool_call_id="b")
assert r2 == changed
r3 = agent._append("web_search", args, changed, tool_call_id="c")
assert "byte-identical" in r3
assert "tool_call_id b" in r3 # new streak's first occurrence, not 'a'
def test_no_stub_below_min_chars():
agent = _fake_agent()
small = "x" * (IDENTICAL_RESULT_STUB_MIN_CHARS - 1)
args = {"query": "hermes"}
agent._append("web_search", args, small, tool_call_id="c1")
r2 = agent._append("web_search", args, small, tool_call_id="c2")
assert "byte-identical" not in r2
assert r2.startswith(small) # full payload kept (pre-existing warning suffix allowed)
def test_no_stub_for_error_results():
agent = _fake_agent()
err = "Error executing tool: " + _BIG
args = {"command": "boom"}
agent._append("terminal", args, err, failed=True, tool_call_id="c1")
r2 = agent._append("terminal", args, err, failed=True, tool_call_id="c2")
assert "byte-identical" not in r2
assert r2.startswith(err) # models must see fresh errors whole
def test_pollers_get_stub_but_never_loop_notice():
agent = _fake_agent()
args = {"id": "job1"}
results = [
agent._append("bfl_flux3_get_result", args, _BIG, tool_call_id=f"c{i}")
for i in range(4)
]
assert results[0] == _BIG
for r in results[1:]:
assert "byte-identical" in r # stubbed: unchanged poll saves context
assert "consecutive identical call" not in r # notice stays exempt
def test_third_identical_call_gets_stub_plus_loop_notice():
agent = _fake_agent()
args = {"query": "hermes"}
agent._append("web_search", args, _BIG, tool_call_id="c1")
agent._append("web_search", args, _BIG, tool_call_id="c2")
r3 = agent._append("web_search", args, _BIG, tool_call_id="c3")
assert "byte-identical" in r3 # stub replaces the payload
assert "3rd consecutive identical call" in r3 # notice appended after it
assert r3.index("byte-identical") < r3.index("3rd consecutive")
def test_stub_carries_spillover_path_when_first_result_persisted():
agent = _fake_agent()
args = {"query": "big"}
agent._tool_guardrails.record_persisted_result(
"c1", "/home/u/.hermes/cache/spillover/c1.txt"
)
agent._append("web_search", args, _BIG, tool_call_id="c1")
r2 = agent._append("web_search", args, _BIG, tool_call_id="c2")
assert "/home/u/.hermes/cache/spillover/c1.txt" in r2
def test_stub_includes_args_summary_for_compression_safety():
agent = _fake_agent()
args = {"query": "hermes result stubbing", "limit": 5}
agent._append("web_search", args, _BIG, tool_call_id="c1")
r2 = agent._append("web_search", args, _BIG, tool_call_id="c2")
# Canonical-args preview so the model knows WHAT the call was even if
# the referenced message is later evicted by compression.
assert "hermes result stubbing" in r2
def test_stub_args_summary_truncated_to_120_chars():
c = ToolCallGuardrailController()
args = {"query": "q" * 500}
assert c.observe_call("web_search", args, _BIG, tool_call_id="c1").stub is None
stub = c.observe_call("web_search", args, _BIG, tool_call_id="c2").stub
assert stub is not None
args_part = stub.split("Args: ", 1)[1]
assert len(args_part) < 200 # ~120-char preview + ellipsis + closer
def test_config_off_disables_stub():
agent = _fake_agent(stall_guards=False)
args = {"query": "hermes"}
agent._append("web_search", args, _BIG, tool_call_id="c1")
r2 = agent._append("web_search", args, _BIG, tool_call_id="c2")
assert "byte-identical" not in r2
assert r2.startswith(_BIG)
def test_streak_reset_by_different_call_means_next_identical_is_full():
agent = _fake_agent()
args = {"query": "hermes"}
agent._append("web_search", args, _BIG, tool_call_id="c1")
agent._append("read_file", {"path": "/a"}, "other", tool_call_id="c2")
r3 = agent._append("web_search", args, _BIG, tool_call_id="c3")
assert "byte-identical" not in r3
assert r3.startswith(_BIG) # fresh streak — first occurrence full again
def test_multimodal_content_never_stubbed_and_breaks_streak():
c = ToolCallGuardrailController()
args = {"path": "/img.png"}
assert c.observe_call("vision", args, _BIG, tool_call_id="c1").stub is None
# Non-string (multimodal) results never form or extend a streak.
obs = c.observe_call("vision", args, None, tool_call_id="c2")
assert obs.stub is None
assert c.observe_call("vision", args, _BIG, tool_call_id="c3").stub is None
def test_observe_identical_call_backcompat_notice_still_fires():
c = ToolCallGuardrailController()
for _ in range(STALL_GUARD_IDENTICAL_CALL_THRESHOLD - 1):
assert c.observe_identical_call("web_search", {"q": 1}, "r") is None
assert c.observe_identical_call("web_search", {"q": 1}, "r") is not None
def test_extract_persisted_path_round_trip():
# The stub's spillover reference is parsed from the <persisted-output>
# block that maybe_persist_tool_result builds — assert the round trip.
from tools.tool_result_storage import (
_build_persisted_message,
extract_persisted_path,
)
block = _build_persisted_message("preview", True, 50_000, "/tmp/spill/x.txt")
assert extract_persisted_path(block) == "/tmp/spill/x.txt"
assert extract_persisted_path("plain result") is None
# ── said-continue-but-stopped detector ─────────────────────────────────────
@@ -0,0 +1,183 @@
"""Uncompressed context overflow guardrail (#89297).
When compression is explicitly disabled (``compression.enabled: false``),
sessions can grow past the model context window with nothing to shrink them.
The conversation loop's pre-API site warns (deduped, actionable); the
turn-context preflight re-arms the dedup once the session is back under the
window so a later re-overflow warns again.
The fake binds the PRODUCTION ``_warn_uncompressed_context_overflow`` /
``_clear_context_overflow_warn`` methods so their dedup logic is actually
under test (not a reimplementation).
"""
from __future__ import annotations
import types
from unittest.mock import MagicMock
from agent.turn_context import TurnContext, build_turn_context # noqa: F401
from run_agent import AIAgent
from tests.agent.test_turn_context import _FakeAgent, _build
class _FakeUncompressedAgent(_FakeAgent):
"""Agent stub with compression disabled, bound to the REAL warn methods."""
# Production methods under test — bound from AIAgent so the dedup key
# handling and message text cannot silently drift from what ships.
_warn_uncompressed_context_overflow = (
AIAgent._warn_uncompressed_context_overflow
)
_clear_context_overflow_warn = AIAgent._clear_context_overflow_warn
def __init__(self, model="deepseek-v4-flash", context_length=10_000):
super().__init__()
self.model = model
self.provider = "deepseek"
self.compression_enabled = False
self.context_compressor = types.SimpleNamespace(
protect_first_n=2,
protect_last_n=2,
context_length=context_length,
threshold_tokens=int(context_length * 0.75),
last_prompt_tokens=-1,
)
def _oversized_history(n_turns: int = 10) -> list:
large_turn = "Large context content " * 500 # ~2,500 tokens each
history = []
for i in range(n_turns):
history.append({"role": "user", "content": f"Turn {i}: {large_turn}"})
history.append({"role": "assistant", "content": f"Reply {i}: {large_turn}"})
return history
def test_production_warn_emits_once_and_dedups():
"""The real method warns once, then dedups identical overflows."""
agent = _FakeUncompressedAgent(context_length=10_000)
agent._emit_warning = MagicMock()
agent._warn_uncompressed_context_overflow(15_000, 10_000)
agent._warn_uncompressed_context_overflow(16_000, 10_000)
agent._emit_warning.assert_called_once()
msg = agent._emit_warning.call_args[0][0]
assert "exceeds the model context window" in msg
assert "compression.enabled: false" in msg
assert "10,000 tokens" in msg
def test_clear_rearms_the_warning():
"""After _clear_context_overflow_warn (session back under the window),
a later re-overflow warns again."""
agent = _FakeUncompressedAgent(context_length=10_000)
agent._emit_warning = MagicMock()
agent._warn_uncompressed_context_overflow(15_000, 10_000)
agent._clear_context_overflow_warn()
agent._warn_uncompressed_context_overflow(15_500, 10_000)
assert agent._emit_warning.call_count == 2
def test_uncompressed_session_within_limits_emits_no_warning():
agent = _FakeUncompressedAgent(context_length=128_000)
agent._emit_warning = MagicMock()
history = [
{"role": "user", "content": "hello"},
{"role": "assistant", "content": "hi there"},
]
tctx = _build(agent, conversation_history=history)
assert isinstance(tctx, TurnContext)
agent._emit_warning.assert_not_called()
def test_preflight_rearm_clears_dedup_when_back_under_window():
"""The turn-context preflight re-arms the dedup once the session fits
again (e.g. after a manual /compress), so growth past the window later
warns a second time."""
agent = _FakeUncompressedAgent(context_length=128_000)
agent._emit_warning = MagicMock()
# Simulate a previously fired warning.
agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 128_000)
history = [
{"role": "user", "content": "hello"},
{"role": "assistant", "content": "hi there"},
]
tctx = _build(agent, conversation_history=history)
assert isinstance(tctx, TurnContext)
# Preflight cleared the dedup — the next overflow warns again.
assert agent._last_ctx_overflow_warn is None
agent._warn_uncompressed_context_overflow(200_000, 128_000)
agent._emit_warning.assert_called_once()
def test_preflight_does_not_rearm_while_still_over_window():
"""While the session is still over the window, the dedup must survive
the preflight (no per-turn warn spam)."""
agent = _FakeUncompressedAgent(context_length=10_000)
agent._emit_warning = MagicMock()
agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 10_000)
tctx = _build(agent, conversation_history=_oversized_history())
assert isinstance(tctx, TurnContext)
assert agent._last_ctx_overflow_warn == ("uncompressed_ctx_overflow", 10_000)
agent._emit_warning.assert_not_called()
def test_multimodal_content_forces_real_estimate_in_rearm_gate():
"""List (multimodal) content defeats a char count; the pre-check must
treat it as over-gate so the real estimator decides. A tiny multimodal
session is still under the window, so the dedup is re-armed."""
agent = _FakeUncompressedAgent(context_length=128_000)
agent._emit_warning = MagicMock()
agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 128_000)
history = [
{
"role": "user",
"content": [
{"type": "text", "text": "look at this"},
{"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}},
],
},
{"role": "assistant", "content": "looking"},
]
tctx = _build(agent, conversation_history=history)
assert isinstance(tctx, TurnContext)
assert agent._last_ctx_overflow_warn is None
def test_none_content_tool_call_rows_do_not_defeat_cheap_gate():
"""Assistant tool-call rows routinely carry content=None; they must
count as zero chars (NOT force the estimator) so the cheap gate keeps
its value in ordinary tool-using sessions. Regression for the salvage
follow-up's first draft, where `None` hit the over-gate branch."""
from unittest.mock import patch as _patch
agent = _FakeUncompressedAgent(context_length=128_000)
agent._emit_warning = MagicMock()
agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 128_000)
history = [
{"role": "user", "content": "run the tool"},
{"role": "assistant", "content": None,
"tool_calls": [{"id": "c1", "function": {"name": "t", "arguments": "{}"}}]},
{"role": "tool", "tool_call_id": "c1", "content": "small result"},
{"role": "assistant", "content": "done"},
]
with _patch(
"agent.turn_context.estimate_request_tokens_rough"
) as mock_est:
tctx = _build(agent, conversation_history=history)
assert isinstance(tctx, TurnContext)
# Cheap gate decided (tiny session, under window): estimator never ran,
# and the dedup was still re-armed via the raw-chars branch.
mock_est.assert_not_called()
assert agent._last_ctx_overflow_warn is None
+69
View File
@@ -153,6 +153,64 @@ def test_unknown_terminal_does_not_enable_extended_enter_keys():
assert cli_mod._terminal_supports_extended_enter_keys({"TERM_PROGRAM": "unknown"}) is False
# ---------------------------------------------------------------------------
# Ghostty: must push ONLY modifyOtherKeys, not the Kitty keyboard protocol —
# see cli._is_ghostty_terminal for the full rationale (#87630).
# ---------------------------------------------------------------------------
class _FakeOutput:
"""Minimal output object with write_raw + flush for _enable_extended_enter_keys."""
def __init__(self):
self.written = b""
def write_raw(self, data):
self.written += data.encode() if isinstance(data, str) else data
def flush(self):
pass
def test_ghostty_uses_modify_other_keys_only():
"""Ghostty must NOT push the Kitty keyboard protocol (CSI >1u)."""
import cli as cli_mod
out = _FakeOutput()
result = cli_mod._enable_extended_enter_keys(
output=out,
env={"TERM_PROGRAM": "ghostty", "TERM": "xterm-ghostty"},
)
assert result is True
# Must contain modifyOtherKeys push ...
assert b"\x1b[>4;2m" in out.written
# ... but NOT the Kitty protocol push.
assert b"\x1b[>1u" not in out.written
def test_ghostty_via_term_var_uses_modify_other_keys_only():
"""xterm-ghostty TERM (without TERM_PROGRAM) also skips Kitty protocol."""
import cli as cli_mod
out = _FakeOutput()
result = cli_mod._enable_extended_enter_keys(
output=out,
env={"TERM": "xterm-ghostty"},
)
assert result is True
assert b"\x1b[>4;2m" in out.written
assert b"\x1b[>1u" not in out.written
def test_non_ghostty_terminals_still_push_kitty_protocol():
"""iTerm2 and others still get the full dual-protocol push."""
import cli as cli_mod
out = _FakeOutput()
result = cli_mod._enable_extended_enter_keys(
output=out,
env={"TERM_PROGRAM": "iTerm.app", "TERM": "xterm-256color"},
)
assert result is True
assert b"\x1b[>1u" in out.written
assert b"\x1b[>4;2m" in out.written
@pytest.mark.linux_only
def test_proc_version_microsoft_marker_preserves_newline():
"""WSL detection via /proc when env vars are scrubbed (sudo etc.).
@@ -177,3 +235,14 @@ def test_proc_version_microsoft_marker_preserves_newline():
# ---------------------------------------------------------------------------
# install_ctrl_enter_alias() — ANSI sequence mappings for enhanced terminals
# ---------------------------------------------------------------------------
def test_is_ghostty_terminal_detection_paths():
"""_is_ghostty_terminal matches exactly the two allowlist conditions."""
import cli as cli_mod
assert cli_mod._is_ghostty_terminal({"TERM_PROGRAM": "ghostty"}) is True
assert cli_mod._is_ghostty_terminal({"TERM": "xterm-ghostty"}) is True
assert cli_mod._is_ghostty_terminal({"TERM": "XTERM-GHOSTTY"}) is True
assert cli_mod._is_ghostty_terminal({"TERM_PROGRAM": "iTerm.app"}) is False
assert cli_mod._is_ghostty_terminal({}) is False
+127
View File
@@ -433,3 +433,130 @@ def test_cmd_backspace_alias_not_clobbered():
install_cmd_backspace_alias()
install_modify_other_keys_aliases()
assert _parse("\x1b[127;9u") == [Keys.ControlU]
# ---------------------------------------------------------------------------
# Lock-bit variants (#89651): kitty/ghostty OR the CapsLock (64) / NumLock
# (128) state into the CSI-u modifier parameter, so with a lock enabled
# every combo arrives shifted (ESC[99;133u instead of ESC[99;5u) and died
# as literal text without these aliases.
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("letter", CTRL_LETTERS)
def test_ctrl_letter_with_numlock_parses_as_raw_byte(letter):
"""Ctrl+<letter> with NumLock on (modifier + 128) must parse identically
to the raw control byte — the exact garbage from #89651 ([127;133u)."""
raw_byte = chr(ord(letter) - ord('a') + 1)
raw_result = _parse(raw_byte)
numlock_seq = f"\x1b[{ord(letter)};133u" # 5 + 128
assert _parse(numlock_seq) == raw_result, (
f"NumLock Ctrl+{letter} ({numlock_seq!r}) should parse identically "
f"to raw {raw_byte!r}"
)
@pytest.mark.parametrize("letter", ["a", "c", "z"])
def test_ctrl_letter_with_capslock_parses_as_raw_byte(letter):
raw_byte = chr(ord(letter) - ord('a') + 1)
capslock_seq = f"\x1b[{ord(letter)};69u" # 5 + 64
assert _parse(capslock_seq) == _parse(raw_byte)
def test_ctrl_c_with_both_locks_parses_as_raw_byte():
"""Ctrl+C with CapsLock and NumLock both on (5 + 64 + 128 = 197)."""
assert _parse("\x1b[99;197u") == _parse("\x03")
def test_alt_letter_with_numlock_keeps_escape_prefix():
assert _parse("\x1b[97;131u") == [Keys.Escape, "a"] # 3 + 128
def test_shift_letter_with_capslock_types_uppercase():
assert _parse("\x1b[97;66u") == ["A"] # 2 + 64
def test_esc_key_with_numlock_is_escape():
assert _parse("\x1b[27;129u") == [Keys.Escape] # 1 + 128
assert _parse("\x1b[27;133u") == [Keys.Escape] # 5 + 128
def test_ctrl_backspace_with_numlock_is_backward_kill_word():
"""The exact sequence from the #89651 report ([127;133u)."""
assert _parse("\x1b[127;133u") == [Keys.Escape, Keys.ControlH]
def test_modify_other_keys_tilde_form_has_no_lock_variants():
"""The xterm modifyOtherKeys encoding never carries lock bits, so no
+64/+128 variants of the ESC[27;N;CP~ form may be installed."""
for seq in ("\x1b[27;69;99~", "\x1b[27;133;99~", "\x1b[27;197;99~"):
assert seq not in ANSI_SEQUENCES
# ---------------------------------------------------------------------------
# Follow-up widening: lock twins on the alias installers, legacy CSI-letter /
# CSI-tilde navigation, unmodified CSI-u keys, and PUA functional keys.
# ---------------------------------------------------------------------------
def test_lock_bits_on_legacy_cursor_keys_map_to_plain_keys():
"""kitty stamps lock bits onto legacy CSI-letter arrows too:
plain Down + NumLock = ESC[1;129B, CapsLock = ESC[1;65B, both = 193."""
for mod, key in ((129, Keys.Down), (65, Keys.Down), (193, Keys.Down)):
assert _parse(f"\x1b[1;{mod}B") == [key]
assert _parse("\x1b[1;130B") == [Keys.ShiftDown] # Shift + NumLock
assert _parse("\x1b[1;131D") == [Keys.Escape, Keys.Left] # Alt + NumLock
assert _parse("\x1b[1;133D") == [Keys.ControlLeft] # Ctrl + NumLock
def test_lock_bits_on_tilde_navigation_keys():
"""Delete/PageUp/etc. carry the modifier in CSI-tilde form."""
assert _parse("\x1b[3;129~") == [Keys.Delete]
assert _parse("\x1b[3;69~") == [Keys.ControlDelete] # Ctrl+Delete + Caps
assert _parse("\x1b[5;193~") == [Keys.PageUp] # both locks
def test_lock_bits_on_plain_f1_through_f4():
"""Plain F1-F4 base mappings are SS3 (ESC O P); their CSI lock twins
must still resolve (ESC[1;129P etc.)."""
base = _parse("\x1bOP")
assert _parse("\x1b[1;129P") == base
assert _parse("\x1b[1;65P") == base
def test_lock_bits_on_unmodified_csi_u_keys():
"""Tab/Enter/Space/Backspace with only a lock held (modifier 1+lock)."""
assert _parse("\x1b[9;65u") == _parse("\t")
assert _parse("\x1b[13;193u") == _parse("\r")
assert _parse("\x1b[32;129u") == _parse(" ")
assert _parse("\x1b[127;129u") == _parse("\x7f")
def test_lock_bits_on_pua_functional_keys():
"""Kitty PUA functional keys (keypad, F13+) keep working under locks —
NumLock especially matters because it gates the keypad itself."""
assert _parse("\x1b[57399;129u") == ["0"] # KP_0 + NumLock
assert _parse("\x1b[57376;129u") == [Keys.F13] # F13 + NumLock
assert _parse("\x1b[57427;129u") == [Keys.Ignore] # KP_BEGIN + NumLock
def test_lock_bits_on_shift_enter_and_ctrl_enter_aliases():
from hermes_cli.pt_input_extras import (
install_ctrl_enter_alias,
install_shift_enter_alias,
)
install_shift_enter_alias()
install_ctrl_enter_alias()
newline = _parse("\x1b\r")
assert _parse("\x1b[13;130u") == newline # Shift+Enter + NumLock
assert _parse("\x1b[13;66u") == newline # Shift+Enter + CapsLock
assert _parse("\x1b[13;133u") == newline # Ctrl+Enter + NumLock
def test_lock_bits_on_cmd_backspace_alias():
from hermes_cli.pt_input_extras import install_cmd_backspace_alias
install_cmd_backspace_alias()
assert _parse("\x1b[127;137u") == [Keys.ControlU] # Cmd+Backspace + NumLock
assert _parse("\x1b[127;73u") == [Keys.ControlU] # Cmd+Backspace + Caps
assert _parse("\x1b[3;137~") == [Keys.ControlK] # Cmd+FwdDel + NumLock
+102
View File
@@ -1340,3 +1340,105 @@ class TestShallowCloneDeepening:
assert wt.exists(), (
"genuinely unpushed commit must survive even after deepening"
)
class TestPrMergedEscapeHatch:
"""Rebase-merged PRs whose diff changed during salvage defeat ``git
cherry`` (patch-id mismatch), so the pruner asks GitHub whether the
branch's PR is MERGED. These tests stub the ``gh`` binary on PATH — the
contract is about how the pruner consumes the answer, not about GitHub.
Contract:
- gh reports a merged PR + tree is clean -> reaped
- gh reports no merged PR -> preserved
- gh missing/failing -> preserved (fail safe)
- dirty tree -> never reaped regardless of gh
"""
_age = staticmethod(TestWorktreeLockReaping._age)
@staticmethod
def _mk_diverged(repo, name, age_h=100):
"""Worktree with a commit NOT patch-equivalent to anything upstream."""
p = repo / ".worktrees" / name
(repo / ".worktrees").mkdir(exist_ok=True)
subprocess.run(
["git", "worktree", "add", str(p), "-b", f"hermes/{name}", "HEAD"],
cwd=repo, capture_output=True,
)
(p / "salvaged.txt").write_text("diff that was reworked during salvage\n")
subprocess.run(["git", "add", "salvaged.txt"], cwd=p, capture_output=True)
subprocess.run(["git", "commit", "-m", "salvaged work"], cwd=p, capture_output=True)
TestPrMergedEscapeHatch._age(p, age_h)
return p
@staticmethod
def _stub_gh(tmp_path, monkeypatch, stdout='[{"number": 1}]', exit_code=0):
gh = tmp_path / "bin" / "gh"
gh.parent.mkdir(parents=True, exist_ok=True)
gh.write_text(f"#!/bin/sh\nprintf '%s' '{stdout}'\nexit {exit_code}\n")
gh.chmod(0o755)
monkeypatch.setenv("PATH", f"{gh.parent}:{os.environ['PATH']}")
def test_merged_pr_tree_is_reaped(self, git_repo, tmp_path, monkeypatch):
import cli
wt = self._mk_diverged(git_repo, "hermes-rebase-merged")
assert cli._worktree_commits_all_merged_upstream(str(wt)) is False, (
"precondition: cherry must NOT consider this merged — the PR "
"check is the only thing that can reap it"
)
self._stub_gh(tmp_path, monkeypatch)
cli._prune_stale_worktrees(str(git_repo))
assert not wt.exists(), (
"clean tree whose branch has a MERGED PR is merged work — reap it"
)
def test_no_merged_pr_preserved(self, git_repo, tmp_path, monkeypatch):
import cli
wt = self._mk_diverged(git_repo, "hermes-pr-open")
self._stub_gh(tmp_path, monkeypatch, stdout="[]")
cli._prune_stale_worktrees(str(git_repo))
assert wt.exists(), "no merged PR -> still unpushed work, preserve"
def test_gh_failure_fails_safe(self, git_repo, tmp_path, monkeypatch):
import cli
wt = self._mk_diverged(git_repo, "hermes-gh-down")
self._stub_gh(tmp_path, monkeypatch, stdout="", exit_code=1)
cli._prune_stale_worktrees(str(git_repo))
assert wt.exists(), "gh failure must preserve the tree (fail safe)"
def test_dirty_tree_never_reaped_even_with_merged_pr(
self, git_repo, tmp_path, monkeypatch
):
import cli
wt = self._mk_diverged(git_repo, "hermes-dirty-merged")
(wt / "uncommitted.txt").write_text("in-flight\n")
self._age(wt, 100)
self._stub_gh(tmp_path, monkeypatch)
cli._prune_stale_worktrees(str(git_repo))
assert wt.exists(), "dirty guard outranks the PR-merged verdict"
def test_merged_verdict_memoized_by_branch_and_head(
self, git_repo, tmp_path, monkeypatch
):
import cli
wt = self._mk_diverged(git_repo, "hermes-memo")
self._stub_gh(tmp_path, monkeypatch)
cache: dict = {}
assert cli._worktree_branch_pr_merged(str(wt), cache=cache) is True
keys = [k for k in cache if k.startswith("pr-merged:")]
assert len(keys) == 1 and cache[keys[0]] is True
# Break gh: a cached True verdict must not re-consult it.
self._stub_gh(tmp_path, monkeypatch, stdout="", exit_code=1)
assert cli._worktree_branch_pr_merged(str(wt), cache=cache) is True
def test_negative_verdict_not_cached(self, git_repo, tmp_path, monkeypatch):
import cli
wt = self._mk_diverged(git_repo, "hermes-nocache-neg")
self._stub_gh(tmp_path, monkeypatch, stdout="[]")
cache: dict = {}
assert cli._worktree_branch_pr_merged(str(wt), cache=cache) is False
assert not [k for k in cache if k.startswith("pr-merged:")], (
"False must not be memoized — the PR can merge later with the "
"same (branch, head) key"
)
@@ -0,0 +1,231 @@
"""Inbound replay dedupe on the relay adapter (transplanted from the
live-cards branch for the rc.4 relay-fixes train).
Live-canary finding #3 (Alice, staging): the relay inbound leg is
at-least-once. On WS re-handshake the connector replays its durable
per-instance buffer; a long multi-tool turn straddling a quiet socket drop
got its ORIGINAL inbound replayed after the turn finished, re-running the
entire turn — the user saw the final answer posted 2-5x. Platform message
identity (chat_id + message_id/ts) is stable across replays, so a bounded
seen-set drops them. Fail-open: events without a message_id never dedupe
(dropping a real message is strictly worse than rerunning one).
"""
from __future__ import annotations
import asyncio
import pytest
from gateway.config import Platform, PlatformConfig
from gateway.platforms.base import MessageEvent, SessionSource
from gateway.relay.adapter import RelayAdapter
from gateway.relay.descriptor import CONTRACT_VERSION, CapabilityDescriptor
from tests.gateway.relay.stub_connector import StubConnector
def make_desc(**kw) -> CapabilityDescriptor:
base = dict(
contract_version=CONTRACT_VERSION,
platform="slack",
label="Slack",
max_message_length=39000,
supports_draft_streaming=True,
supports_edit=True,
supports_threads=True,
markdown_dialect="slack",
len_unit="chars",
emoji="\U0001f4ac",
platform_hint="",
pii_safe=False,
supported_ops=("send", "edit", "typing"),
)
base.update(kw)
return CapabilityDescriptor(**base)
def _connected_adapter(**desc_kw):
desc = make_desc(**desc_kw)
stub = StubConnector(desc)
adapter = RelayAdapter(PlatformConfig(), desc, transport=stub)
return adapter, stub
@pytest.fixture()
def loop():
loop = asyncio.new_event_loop()
asyncio.set_event_loop(loop)
yield loop
loop.close()
def _record(bucket, event):
async def _coro():
bucket.append(event)
return _coro()
async def _false_coro():
return False
async def _none_coro():
return None
class TestInboundReplayDedupe:
"""Finding #3 (live canary): connector replay of the original inbound
after a WS re-handshake must not re-run the turn."""
def _event(self, message_id="1700.100", chat_id="C1", text="hi"):
# A REAL MessageEvent, shaped exactly as _event_from_wire produces it:
# chat identity lives on event.source, NOT as a top-level attribute.
# (The first version of these tests used a SimpleNamespace with a
# top-level chat_id — a shape no production code path produces — and
# green-lit a dedupe key that read the wrong field.)
source = SessionSource(
platform=Platform.SLACK,
chat_id=chat_id,
chat_type="channel",
user_id="U1",
message_id=message_id,
)
return MessageEvent(text=text, source=source, message_id=message_id)
def _tap(self, adapter, handled):
adapter.handle_message = lambda e: _record(handled, e)
adapter._consume_prompt_response = lambda e: _false_coro()
adapter._localize_inbound_media = lambda e: _none_coro()
def test_replayed_inbound_dropped(self, loop):
adapter, _ = _connected_adapter()
handled = []
self._tap(adapter, handled)
e = self._event()
loop.run_until_complete(adapter._on_inbound(e))
loop.run_until_complete(adapter._on_inbound(e)) # replay
assert len(handled) == 1
def test_distinct_messages_both_handled(self, loop):
adapter, _ = _connected_adapter()
handled = []
self._tap(adapter, handled)
loop.run_until_complete(adapter._on_inbound(self._event("1700.100")))
loop.run_until_complete(adapter._on_inbound(self._event("1700.200")))
assert len(handled) == 2
def test_missing_message_id_fails_open(self, loop):
adapter, _ = _connected_adapter()
handled = []
self._tap(adapter, handled)
e = self._event(message_id=None)
loop.run_until_complete(adapter._on_inbound(e))
loop.run_until_complete(adapter._on_inbound(e))
assert len(handled) == 2 # never dedupe without identity
def test_seen_set_bounded(self, loop):
adapter, _ = _connected_adapter()
adapter.handle_message = lambda e: _none_coro()
adapter._consume_prompt_response = lambda e: _false_coro()
adapter._localize_inbound_media = lambda e: _none_coro()
for i in range(600):
loop.run_until_complete(adapter._on_inbound(self._event(f"ts.{i}")))
assert len(adapter._seen_inbound) <= adapter._SEEN_INBOUND_MAX
class TestWireLevelReplayDedupe:
"""The full production inbound path: a connector wire frame decoded by
_event_from_wire, then dispatched through RelayAdapter._on_inbound.
This is the layer the hand-built-event tests above cannot vouch for: the
dedupe key must work on the exact object shape the wire decoder emits.
The original dedupe commit shipped green on hand-built events while being
a no-op on decoded ones — this class exists so that can't recur.
"""
WIRE = {
"text": "hi",
"message_type": "text",
"message_id": "1700.100",
"source": {
"platform": "slack",
"chat_id": "C1",
"chat_type": "channel",
"user_id": "U1",
"message_id": "1700.100",
},
}
def _tap(self, adapter, handled):
adapter.handle_message = lambda e: _record(handled, e)
adapter._consume_prompt_response = lambda e: _false_coro()
adapter._localize_inbound_media = lambda e: _none_coro()
def _decode(self, **overrides):
from gateway.relay.ws_transport import _event_from_wire
raw = {**self.WIRE, **overrides}
if "source" in overrides:
raw["source"] = {**self.WIRE["source"], **overrides["source"]}
return _event_from_wire(raw)
def test_decoded_event_yields_a_dedupe_key(self):
adapter, _ = _connected_adapter()
key = adapter._inbound_dedupe_key(self._decode())
assert key is not None, (
"the wire decoder's event shape must produce a dedupe key — "
"None here means the dedupe is fail-open for ALL production "
"traffic (the original ship-broken state)"
)
def test_replayed_wire_frame_dropped(self, loop):
adapter, _ = _connected_adapter()
handled = []
self._tap(adapter, handled)
loop.run_until_complete(adapter._on_inbound(self._decode()))
# The connector re-delivers the SAME frame on re-handshake; the
# decoder builds a fresh object each time, so identity must come
# from the key, not object identity.
loop.run_until_complete(adapter._on_inbound(self._decode()))
assert len(handled) == 1
def test_same_ids_on_different_platforms_not_conflated(self, loop):
# Phase 1.5 multiplex: one adapter fronts several platforms. Numeric
# chat/message ids can collide across platforms; both must dispatch.
adapter, _ = _connected_adapter()
handled = []
self._tap(adapter, handled)
loop.run_until_complete(adapter._on_inbound(self._decode()))
loop.run_until_complete(
adapter._on_inbound(self._decode(source={"platform": "discord"}))
)
assert len(handled) == 2
class TestDedupeKeyPlatformNormalization:
"""The platform component of the key must be spelling-invariant: a
Platform enum and its plain-string form are ONE platform (one key), and
two different string platforms must never collapse into a shared empty
component. Production wire decoding always yields the enum; alternate
event constructors may carry the string."""
def _key(self, platform):
adapter, _ = _connected_adapter()
source = SessionSource(
platform=platform, chat_id="C1", chat_type="channel", message_id="m1"
)
event = MessageEvent(text="hi", source=source, message_id="m1")
return adapter._inbound_dedupe_key(event)
def test_enum_and_string_spellings_produce_one_key(self):
assert self._key(Platform.SLACK) == self._key("slack")
def test_distinct_string_platforms_stay_distinct(self):
assert self._key("slack") != self._key("discord")
def test_missing_platform_still_yields_a_key(self):
# Fail-open on identity is reserved for missing message/chat ids;
# a missing platform alone must not disable dedupe.
key = self._key(None)
assert key is not None
assert key.startswith(":")
@@ -0,0 +1,373 @@
"""Regression tests for the relay WS transport hardening fix.
Coatue incident 2026-08-18: WAN latency / event-loop stalls tripped the
websockets library's default 20s pong deadline, closing customer-gateway
sockets with `1011 keepalive ping timeout`. On top of the spurious close,
every in-flight outbound then hung for the full _outbound_timeout_s (~30s)
because only disconnect() failed pending futures — an unexpected socket drop
left them stranded — and sends issued while the reconnect supervisor was
backing off registered futures no reader could ever resolve.
Three hardening changes under test:
1. _read_loop fails all in-flight _pending futures on ANY exit path with
the dict shape callers expect ({"success": False, ...}).
2. _request_response fails fast while the reconnect supervisor is
mid-redial (live supervisor task = the redial window).
3. connect() passes explicit WAN-friendly keepalive tuning
(ping_interval=30, ping_timeout=60) to websockets.connect().
"""
from __future__ import annotations
import asyncio
import pytest
import gateway.relay.ws_transport as ws_transport_mod
from gateway.relay.ws_transport import WebSocketRelayTransport, WEBSOCKETS_AVAILABLE
pytestmark = pytest.mark.skipif(not WEBSOCKETS_AVAILABLE, reason="websockets not installed")
if WEBSOCKETS_AVAILABLE:
from websockets.exceptions import ConnectionClosedError
class _DroppingWS:
"""Fake socket: accepts sends, then the read loop dies mid-iteration —
the shape of an unexpected close (e.g. 1011 keepalive ping timeout)."""
def __init__(self, close_code: int | None = None):
self.sent: list[str] = []
# Reader blocks here until the test releases it, so the outbound
# future is registered BEFORE the "socket" drops.
self.drop = asyncio.Event()
self._close_code = close_code
async def send(self, data):
self.sent.append(data)
def __aiter__(self):
return self
async def __anext__(self):
await self.drop.wait()
if self._close_code is not None:
from websockets.frames import Close
raise ConnectionClosedError(Close(self._close_code, ""), None)
raise ConnectionClosedError(None, None)
async def close(self):
pass
@pytest.mark.asyncio
async def test_read_loop_exit_fails_pending_futures_promptly():
"""When the socket drops unexpectedly, in-flight _request_response callers
must get {"success": False, ...} promptly — not block ~30s on a future
only the (now dead) reader could have resolved."""
t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=30.0)
fake = _DroppingWS()
t._ws = fake
t._reader = asyncio.create_task(t._read_loop())
send_task = asyncio.create_task(t.send_outbound({"op": "send_message", "text": "hi"}))
# Let the outbound frame go out and its future register in _pending.
for _ in range(50):
if t._pending:
break
await asyncio.sleep(0.01)
assert t._pending, "outbound future never registered"
# Drop the socket: the read loop exits on ConnectionClosedError.
fake.drop.set()
result = await asyncio.wait_for(send_task, timeout=2.0)
assert result == {"success": False, "error": "relay transport connection lost"}
assert t._pending == {}
await t._reader
@pytest.mark.asyncio
async def test_send_during_redial_window_fails_fast():
"""While the reconnect supervisor is backing off after a drop, a send must
return an error dict immediately (no RuntimeError, no 30s timeout on an
unresolvable future). Drives the REAL sequence — reader exit arms the
supervisor and clears _ws — rather than hand-crafting a stale-_ws state
the transport can no longer reach."""
t = WebSocketRelayTransport(
"ws://unused",
"discord",
"bot1",
reconnect=True,
reconnect_backoff_s=60.0, # park the supervisor in backoff
outbound_timeout_s=30.0,
)
fake = _DroppingWS()
t._ws = fake
await _run_reader_to_exit(t, fake)
supervisor = t._supervisor
try:
assert supervisor is not None and not supervisor.done(), (
"reader exit must arm the reconnect supervisor"
)
result = await asyncio.wait_for(
t.send_outbound({"op": "send_message", "text": "hi"}), timeout=1.0
)
assert result["success"] is False
assert t._pending == {}
finally:
if supervisor is not None:
supervisor.cancel()
try:
await supervisor
except asyncio.CancelledError:
pass
@pytest.mark.asyncio
async def test_send_allowed_once_redial_installs_fresh_socket(monkeypatch):
"""The moment _dial_and_start() installs a fresh socket and its reader,
the transport is genuinely usable — even though the supervisor task has
not finished unwinding (it is still awaiting the hello sends). A send in
that window must be ACCEPTED, not rejected as 'reconnecting': gating
sends on supervisor state rejected real traffic on a live socket."""
class _LiveWS:
def __init__(self):
self.sent: list[str] = []
self.hello_seen = asyncio.Event()
self.release = asyncio.Event()
async def send(self, data):
self.sent.append(data)
if '"hello"' in data:
# Inside _dial_and_start, AFTER _ws and the reader are
# installed. Park here to hold the window open.
self.hello_seen.set()
await self.release.wait()
def __aiter__(self):
return self
async def __anext__(self):
await asyncio.sleep(3600)
async def close(self):
pass
live = _LiveWS()
async def _fake_connect(url, **kwargs):
return live
monkeypatch.setattr(ws_transport_mod.websockets, "connect", _fake_connect)
t = WebSocketRelayTransport(
"ws://unused",
"discord",
"bot1",
reconnect=True,
reconnect_backoff_s=0.01,
outbound_timeout_s=5.0,
)
# Arm the supervisor exactly as the reader's fall-through does.
t._supervisor = asyncio.create_task(t._reconnect_loop())
await asyncio.wait_for(live.hello_seen.wait(), timeout=2.0)
try:
assert t._ws is live and not t._supervisor.done()
send_task = asyncio.create_task(
t.send_outbound({"op": "send_message", "text": "hi"})
)
# The send must reach the live socket (registered + frame written),
# not fail fast: wait for the outbound frame to land.
for _ in range(100):
if any('"outbound"' in s for s in live.sent):
break
await asyncio.sleep(0.01)
assert any('"outbound"' in s for s in live.sent), (
"send was rejected during the post-dial window despite a live "
"socket and running reader"
)
# Resolve it via the reader path shape: answer directly.
rid = next(iter(t._pending))
t._pending[rid].set_result({"success": True})
assert (await asyncio.wait_for(send_task, timeout=2.0)) == {"success": True}
finally:
live.release.set()
await asyncio.wait_for(t._supervisor, timeout=2.0)
if t._reader is not None:
t._reader.cancel()
try:
await t._reader
except asyncio.CancelledError:
pass
@pytest.mark.asyncio
async def test_connect_passes_wan_keepalive_tuning(monkeypatch):
"""connect() must pass ping_interval=30 / ping_timeout=60 explicitly —
the library defaults (20/20) caused spurious 1011 keepalive closes over
WAN paths (Coatue 2026-08-18). Both call sites (with/without auth
headers) are exercised."""
captured: list[dict] = []
class _IdleWS:
async def send(self, data):
pass
def __aiter__(self):
return self
async def __anext__(self):
await asyncio.sleep(3600)
async def close(self):
pass
async def _fake_connect(url, **kwargs):
captured.append(kwargs)
return _IdleWS()
monkeypatch.setattr(ws_transport_mod.websockets, "connect", _fake_connect)
# Site 1: no upgrade secret -> the headerless connect() call.
t = WebSocketRelayTransport("ws://unused", "discord", "bot1")
await t.connect()
await t.disconnect(budget_s=0)
# Site 2: secret + gateway_id -> the additional_headers connect() call.
t2 = WebSocketRelayTransport(
"ws://unused", "discord", "bot1", gateway_id="gw-1", upgrade_secret="s3cret"
)
await t2.connect()
await t2.disconnect(budget_s=0)
assert len(captured) == 2
no_header_kwargs, header_kwargs = captured
assert "additional_headers" not in no_header_kwargs
assert "additional_headers" in header_kwargs
for kwargs in captured:
assert kwargs.get("ping_interval") == 30
assert kwargs.get("ping_timeout") == 60
async def _run_reader_to_exit(t: WebSocketRelayTransport, fake: _DroppingWS) -> None:
"""Start the reader on ``fake``, drop the socket, and wait for the reader
to fully unwind — the state every post-drop assertion depends on."""
t._reader = asyncio.create_task(t._read_loop())
await asyncio.sleep(0)
fake.drop.set()
await t._reader
@pytest.mark.asyncio
async def test_read_loop_without_socket_still_fails_pending():
"""If the reader is ever scheduled with no socket (lifecycle bug), it must
still settle in-flight waiters on its way out — the old `assert` escaped
before the fail-pending cleanup and left them to the full 30s timeout."""
t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=30.0)
loop = asyncio.get_running_loop()
fut: asyncio.Future = loop.create_future()
t._pending["rid"] = fut
t._ws = None
await t._read_loop() # must not raise
assert fut.done()
assert fut.result() == {"success": False, "error": "relay transport connection lost"}
assert t._pending == {}
@pytest.mark.asyncio
async def test_send_after_terminal_4401_revocation_fails_fast():
"""A terminal 4401 revocation deliberately arms NO reconnect supervisor,
so the reader's exit is the LAST liveness transition this transport will
ever make. If _ws still points at the dead socket afterwards, the
revocation path's own fatal-error notification send wedges for the full
_outbound_timeout_s. The reader must leave _ws cleared so the
not-connected guard answers instantly."""
t = WebSocketRelayTransport(
"ws://unused", "discord", "bot1", reconnect=True, outbound_timeout_s=30.0
)
fake = _DroppingWS(close_code=4401)
t._ws = fake
t._handshake_succeeded = True # prior handshake -> 4401 is a revocation
await _run_reader_to_exit(t, fake)
assert t._auth_revoked is True
assert t._supervisor is None # revocation must not re-dial
assert t._ws is None, "dead socket handle must not survive the reader"
result = await asyncio.wait_for(
t.send_outbound({"op": "send_message", "text": "hi"}), timeout=2.0
)
assert result["success"] is False
assert t._pending == {}
@pytest.mark.asyncio
async def test_send_after_drop_with_reconnect_disabled_fails_fast():
"""reconnect=False transports never arm a supervisor either — the same
stranded-_ws wedge as the revocation path, reachable by configuration."""
t = WebSocketRelayTransport(
"ws://unused", "discord", "bot1", reconnect=False, outbound_timeout_s=30.0
)
fake = _DroppingWS()
t._ws = fake
await _run_reader_to_exit(t, fake)
assert t._ws is None, "dead socket handle must not survive the reader"
result = await asyncio.wait_for(
t.send_outbound({"op": "send_message", "text": "hi"}), timeout=2.0
)
assert result["success"] is False
assert t._pending == {}
@pytest.mark.asyncio
async def test_send_raising_socket_returns_error_dict():
"""The socket can die BETWEEN the `_ws is None` liveness guard and the
actual write (the reader's finally hasn't cleared the handle yet). The
write then raises ConnectionClosed — but send_outbound's contract is a
result dict, and RelayAdapter.send consumes it with no try. The raise
must be converted to {"success": False, ...}, with no future left in
_pending."""
class _RaisingWS:
"""Send raises (already dead); the reader hasn't noticed yet."""
def __init__(self):
self.reader_release = asyncio.Event()
async def send(self, data):
raise ConnectionClosedError(None, None)
def __aiter__(self):
return self
async def __anext__(self):
await self.reader_release.wait()
raise ConnectionClosedError(None, None)
async def close(self):
pass
t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=5.0)
fake = _RaisingWS()
t._ws = fake
t._reader = asyncio.create_task(t._read_loop())
await asyncio.sleep(0)
result = await asyncio.wait_for(
t.send_outbound({"op": "send_message", "text": "hi"}), timeout=2.0
)
assert result["success"] is False
assert "relay send failed" in result["error"]
assert t._pending == {}
fake.reader_release.set()
await t._reader
+35
View File
@@ -96,3 +96,38 @@ def test_install_dependencies_force_reinstalls_versioned_specs(tmp_path, monkeyp
assert installed, "force=True must reach the install step"
assert any("mem0ai>=2.0.10,<3" in specs for specs in installed)
def test_cmd_status_memory_tool_gate_disabled(capsys, monkeypatch):
"""When both memory stores are disabled, Memory status reports memory tool as disabled."""
_cfg = {"memory": {"memory_enabled": False, "user_profile_enabled": False}}
monkeypatch.setattr("hermes_cli.config.load_config", lambda: _cfg)
# check_memory_requirements() reads the readonly loader, not load_config.
monkeypatch.setattr(
"hermes_cli.config.load_config_readonly", lambda: _cfg, raising=False
)
monkeypatch.setattr(memory_setup, "_get_available_providers", lambda: [])
memory_setup.cmd_status(SimpleNamespace())
captured = capsys.readouterr().out
assert "Memory tool: disabled ✗" in captured
assert "Memory injection: disabled ✗" in captured
assert "User profile: disabled ✗" in captured
def test_cmd_status_memory_tool_gate_enabled(capsys, monkeypatch):
"""When at least one memory store is enabled, Memory status reports memory tool as enabled."""
_cfg = {"memory": {"memory_enabled": True, "user_profile_enabled": False}}
monkeypatch.setattr("hermes_cli.config.load_config", lambda: _cfg)
monkeypatch.setattr(
"hermes_cli.config.load_config_readonly", lambda: _cfg, raising=False
)
monkeypatch.setattr(memory_setup, "_get_available_providers", lambda: [])
memory_setup.cmd_status(SimpleNamespace())
captured = capsys.readouterr().out
assert "Memory tool: enabled ✓" in captured
assert "Memory injection: enabled ✓" in captured
assert "User profile: disabled ✗" in captured
@@ -111,6 +111,13 @@ class TestConfigYamlRouting:
assert "not a recognized config key" not in capsys.readouterr().out
assert "script_timeout_seconds: 600" in _read_config(_isolated_hermes_home)
def test_memory_nudge_interval_is_recognized(self, _isolated_hermes_home, capsys):
"""The documented background-memory review interval is runtime config."""
set_config_value("memory.nudge_interval", "0")
assert "not a recognized config key" not in capsys.readouterr().out
assert "nudge_interval: 0" in _read_config(_isolated_hermes_home)
def test_terminal_docker_cwd_mount_flag_goes_to_config_and_env(self, _isolated_hermes_home):
set_config_value("terminal.docker_mount_cwd_to_workspace", "true")
config = _read_config(_isolated_hermes_home)
+243
View File
@@ -0,0 +1,243 @@
"""Behavior contracts for hermes_cli.worktree_gc (attended reclaim).
Each guard gets its own contract against a REAL git repo fixture (no mocks —
the entire value of these tests is exercising actual git verdicts):
- clean + fully merged tree → reap
- untracked-only dirt → reap-archive (files archived, then removed)
- tracked modifications → keep, any age
- unique unpushed commits → keep
- patch-equivalent commits (rebase/squash-merge leak) → reap
- live-locked tree → keep
- kanban t_<hex> tree → keep (owned by kanban gc)
- branch GC: merged branch deleted, unique-commit branch kept,
checked-out branch kept, protected names kept
- reclaim operates ONLY on the frozen audit list (concurrent-session trap)
"""
import os
import subprocess
from pathlib import Path
import pytest
from hermes_cli import worktree_gc
def _git(args, cwd, env=None):
e = dict(os.environ)
e.update({
"GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t",
"GIT_COMMITTER_NAME": "t", "GIT_COMMITTER_EMAIL": "t@t",
})
if env:
e.update(env)
result = subprocess.run(
["git", *args], capture_output=True, text=True, cwd=str(cwd), env=e,
)
assert result.returncode == 0, f"git {args} failed: {result.stderr}"
return result.stdout.strip()
@pytest.fixture
def repo(tmp_path, monkeypatch):
"""origin (bare) + clone with .worktrees/, HOME redirected for archives."""
monkeypatch.setenv("HOME", str(tmp_path / "home"))
(tmp_path / "home").mkdir()
origin = tmp_path / "origin.git"
origin.mkdir()
_git(["init", "--bare", "-b", "main"], origin)
clone = tmp_path / "repo"
_git(["clone", str(origin), str(clone)], tmp_path)
(clone / "README.md").write_text("hello\n")
_git(["add", "."], clone)
_git(["commit", "-m", "init"], clone)
_git(["push", "origin", "main"], clone)
# origin/HEAD so upstream resolution works like a real clone.
_git(["remote", "set-head", "origin", "main"], clone)
(clone / ".worktrees").mkdir()
return clone
def _add_worktree(repo_path, name, branch=None):
tree = repo_path / ".worktrees" / name
branch = branch or f"hermes/{name}"
_git(["worktree", "add", str(tree), "-b", branch], repo_path)
return tree, branch
def _verdict(records, name):
match = [record for record in records if record.name == name]
assert match, f"no record for {name}"
return match[0]
class TestAuditVerdicts:
def test_clean_merged_tree_reaps(self, repo):
_add_worktree(repo, "hermes-clean")
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
assert _verdict(records, "hermes-clean").verdict == "reap"
def test_tracked_modifications_keep(self, repo):
tree, _ = _add_worktree(repo, "hermes-dirty")
(tree / "README.md").write_text("edited\n")
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
record = _verdict(records, "hermes-dirty")
assert record.verdict == "keep"
assert "tracked" in record.reason
def test_untracked_only_is_reap_archive(self, repo):
tree, _ = _add_worktree(repo, "hermes-scratch")
(tree / "PR_BODY_DRAFT.md").write_text("draft\n")
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
record = _verdict(records, "hermes-scratch")
assert record.verdict == "reap-archive"
assert record.untracked == ["PR_BODY_DRAFT.md"]
def test_unique_unpushed_commits_keep(self, repo):
tree, _ = _add_worktree(repo, "hermes-work")
(tree / "new.py").write_text("x = 1\n")
_git(["add", "."], tree)
_git(["commit", "-m", "unique work"], tree)
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
record = _verdict(records, "hermes-work")
assert record.verdict == "keep"
assert "unpushed" in record.reason
def test_patch_equivalent_commits_reap(self, repo):
"""The squash/rebase-merge leak: local commit unreachable from any
remote ref but patch-equivalent to an upstream commit → merged work."""
tree, _ = _add_worktree(repo, "hermes-merged")
(tree / "feat.py").write_text("y = 2\n")
_git(["add", "."], tree)
_git(["commit", "-m", "feat"], tree)
sha = _git(["rev-parse", "HEAD"], tree)
# "Merge" it to main with a DIFFERENT committer so the cherry-pick
# produces a distinct sha (same-second identical-committer cherry
# picks can produce the identical sha — pitfall from the skill).
_git(["cherry-pick", sha], repo,
env={"GIT_COMMITTER_NAME": "other", "GIT_COMMITTER_EMAIL": "o@o"})
_git(["push", "origin", "main"], repo)
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
assert _verdict(records, "hermes-merged").verdict == "reap"
def test_live_locked_tree_keeps(self, repo):
tree, _ = _add_worktree(repo, "hermes-live")
_git(["worktree", "lock", str(tree),
"--reason", f"hermes pid={os.getpid()}"], repo)
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
record = _verdict(records, "hermes-live")
assert record.verdict == "keep"
assert "in use" in record.reason
def test_kanban_tree_untouched(self, repo):
_add_worktree(repo, "t_deadbeef", branch="kanban/t_deadbeef")
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
record = _verdict(records, "t_deadbeef")
assert record.verdict == "keep"
assert "kanban" in record.reason
class TestReclaim:
def test_reap_removes_tree_and_branch(self, repo):
tree, branch = _add_worktree(repo, "hermes-clean")
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
actions = worktree_gc.reclaim_worktrees(str(repo), records=records)
assert any("removed hermes-clean" in a for a in actions)
assert not tree.exists()
probe = subprocess.run(
["git", "rev-parse", "--verify", "--quiet", branch],
capture_output=True, text=True, cwd=str(repo),
)
assert probe.returncode != 0, "branch should be gone with its tree"
def test_untracked_files_archived_before_removal(self, repo):
tree, _ = _add_worktree(repo, "hermes-scratch")
(tree / "NOTES.md").write_text("important scribbles\n")
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
worktree_gc.reclaim_worktrees(str(repo), records=records)
assert not tree.exists()
archive_root = Path.home() / ".hermes" / "archive" / "worktree-prune"
archived = list(archive_root.rglob("NOTES.md"))
assert archived, "untracked file must be archived, not destroyed"
assert archived[0].read_text() == "important scribbles\n"
def test_dry_run_changes_nothing(self, repo):
tree, _ = _add_worktree(repo, "hermes-clean")
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
actions = worktree_gc.reclaim_worktrees(
str(repo), dry_run=True, records=records
)
assert any("would remove" in a for a in actions)
assert tree.exists()
def test_frozen_list_ignores_trees_created_after_audit(self, repo):
"""Concurrent-session trap: a tree created between audit and reclaim
must be out of scope by construction."""
_add_worktree(repo, "hermes-old")
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
late_tree, _ = _add_worktree(repo, "hermes-late")
worktree_gc.reclaim_worktrees(str(repo), records=records)
assert late_tree.exists(), "tree created after the audit must survive"
def test_dead_locked_tree_is_unlocked_and_reaped(self, repo):
tree, _ = _add_worktree(repo, "hermes-zombie")
_git(["worktree", "lock", str(tree),
"--reason", "hermes pid=999999999"], repo)
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
assert _verdict(records, "hermes-zombie").verdict == "reap"
worktree_gc.reclaim_worktrees(str(repo), records=records)
assert not tree.exists()
class TestBranchGC:
def test_merged_branch_deleted_any_name(self, repo):
"""Branch GC is content-gated, not name-gated: any fully-merged local
branch is safe to delete regardless of prefix."""
_git(["branch", "salv-12345", "main"], repo)
_git(["branch", "feat/some-old-thing", "main"], repo)
records = worktree_gc.audit_branches(str(repo))
by_name = {record.name: record for record in records}
assert by_name["salv-12345"].verdict == "delete"
assert by_name["feat/some-old-thing"].verdict == "delete"
worktree_gc.reclaim_branches(str(repo), records=records)
out = _git(["branch", "--format=%(refname:short)"], repo)
assert "salv-12345" not in out
assert "feat/some-old-thing" not in out
def test_unique_commit_branch_kept(self, repo):
_git(["checkout", "-b", "feat/real-work"], repo)
(repo / "wip.py").write_text("z = 3\n")
_git(["add", "."], repo)
_git(["commit", "-m", "wip"], repo)
_git(["checkout", "main"], repo)
records = worktree_gc.audit_branches(str(repo))
by_name = {record.name: record for record in records}
assert by_name["feat/real-work"].verdict == "keep"
assert "unique" in by_name["feat/real-work"].reason
def test_patch_equivalent_branch_deleted(self, repo):
"""Rebase-merged PR branch: SHAs differ from main but every commit is
patch-equivalent — the dominant branch leak."""
_git(["checkout", "-b", "fix/landed"], repo)
(repo / "fix.py").write_text("a = 4\n")
_git(["add", "."], repo)
_git(["commit", "-m", "fix"], repo)
sha = _git(["rev-parse", "HEAD"], repo)
_git(["checkout", "main"], repo)
_git(["cherry-pick", sha], repo,
env={"GIT_COMMITTER_NAME": "other", "GIT_COMMITTER_EMAIL": "o@o"})
_git(["push", "origin", "main"], repo)
records = worktree_gc.audit_branches(str(repo))
by_name = {record.name: record for record in records}
assert by_name["fix/landed"].verdict == "delete"
assert "patch-equivalent" in by_name["fix/landed"].reason
def test_checked_out_and_protected_kept(self, repo):
_tree, branch = _add_worktree(repo, "hermes-active")
records = worktree_gc.audit_branches(str(repo))
by_name = {record.name: record for record in records}
assert by_name["main"].verdict == "keep"
assert by_name[branch].verdict == "keep"
@@ -72,7 +72,12 @@ def _tool_call(i: int):
return SimpleNamespace(
id=f"call_{i}",
type="function",
function=SimpleNamespace(name="web_search", arguments='{"query": "x"}'),
# Vary the query per call: a real marathon turn issues distinct
# lookups, and identical (args, result) pairs are now legitimately
# deduped into reference stubs by the stall-guard subsystem —
# zero-variance args here would deflate the very context pressure
# this test exists to exercise.
function=SimpleNamespace(name="web_search", arguments=f'{{"query": "x{i}"}}'),
)
@@ -293,6 +293,62 @@ class TestInPlaceAntiGrowthGuard:
assert agent.session_id == sid
assert db.get_session(sid)["end_reason"] is None
def test_in_place_salvages_near_break_even_growth(self):
"""Fat retained tool output + todo state should be salvaged and committed."""
from hermes_state import SessionDB
from agent.conversation_compression import compress_context
from agent.model_metadata import estimate_messages_tokens_rough
with tempfile.TemporaryDirectory() as tmp:
db = SessionDB(db_path=Path(tmp) / "t.db")
sid = "20260819_salvage"
_seed(db, sid, "salvage")
agent = _make_agent(db, sid, in_place=True)
agent._last_flushed_db_idx = 5
tool_calls = [
{"id": call_id, "function": {"name": "terminal", "arguments": "{}"}}
for call_id in ("c1", "c2", "c3")
]
original = [
{"role": "user", "content": "do the work"},
{"role": "assistant", "content": "calling tools", "tool_calls": tool_calls},
{"role": "tool", "tool_call_id": "c1", "content": "OLD1 " + ("x" * 8000)},
{"role": "tool", "tool_call_id": "c2", "content": "OLD2 " + ("y" * 8000)},
{"role": "tool", "tool_call_id": "c3", "content": "keep-me " + ("z" * 100)},
{"role": "user", "content": "thanks"},
]
grown = [
{"role": "user", "content": "[CONTEXT COMPACTION] " + ("S" * 200)},
{"role": "assistant", "content": "calling tools", "tool_calls": tool_calls},
{"role": "tool", "tool_call_id": "c1", "content": "OLD1 " + ("x" * 8000)},
{"role": "tool", "tool_call_id": "c2", "content": "OLD2 " + ("y" * 8000)},
{"role": "tool", "tool_call_id": "c3", "content": "keep-me " + ("z" * 100)},
{
"role": "user",
"content": "Current todos:\n- [ ] leftover",
"_todo_snapshot_synthetic": True,
},
]
assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original)
compressor = getattr(agent, "context_compressor")
compressor.compress = (
lambda messages, current_tokens=None, focus_topic=None, force=False: grown
)
compressed, _sp = compress_context(
agent, original, approx_tokens=100_000, system_message="sys"
)
assert getattr(agent, "_last_compaction_in_place") is True
assert estimate_messages_tokens_rough(compressed) < estimate_messages_tokens_rough(original)
tool_bodies = [m.get("content") for m in compressed if m.get("role") == "tool"]
assert any(isinstance(body, str) and body.startswith("keep-me") for body in tool_bodies)
assert any("cleared to save context space" in (body or "") for body in tool_bodies)
# Tool stubbing alone got under budget, so the todo snapshot (the
# only in-transcript todo re-injection) survives the salvage.
assert any(m.get("_todo_snapshot_synthetic") for m in compressed)
def test_in_place_still_commits_shrinking_compression(self):
"""The guard must not block legitimate compressions — a result SMALLER
than the input still commits in place (regression net for #83339)."""
+112
View File
@@ -4535,6 +4535,118 @@ class TestRunConversation:
assert agent.context_compressor.context_length == 200_000
mock_compress.assert_called_once()
def test_output_cap_retry_before_generic_retry_exhaustion(self, agent):
"""Provider max-output-cap 400s clamp via the output-cap handler, not
the generic retry loop ("failed after 3 retries").
"""
self._setup_agent(agent)
agent.api_mode = "chat_completions"
agent.provider = "deepseek"
agent.base_url = "https://api.deepseek.com/v1"
agent.model = "deepseek-v4-flash"
agent.max_tokens = 98_304
agent.compression_enabled = True
agent.context_compressor.context_length = 200_000
agent.context_compressor.should_compress = MagicMock(return_value=False)
error_msg = (
"[400]: max_tokens (98304) exceeds model's maximum output tokens "
"(65536) for model deepseek-v4-flash "
"(ref: 7735422e-9cb4-4075-a779-dfecb3204a0e)"
)
exc = Exception(error_msg)
exc.status_code = 400
exc.code = 400
ok_resp = _mock_response(content="done", finish_reason="stop")
agent.client.chat.completions.create.side_effect = [exc, ok_resp]
mock_compress = MagicMock(return_value=(
[{"role": "user", "content": "hello"}],
"You are helpful.",
))
with (
patch.object(agent, "_persist_session"),
patch.object(agent, "_save_trajectory"),
patch.object(agent, "_cleanup_task_resources"),
patch.object(agent.context_compressor, "update_model"),
patch.object(agent, "_compress_context", mock_compress),
):
result = agent.run_conversation("hello")
assert len(agent.client.chat.completions.create.call_args_list) == 2
second_call = agent.client.chat.completions.create.call_args_list[1].kwargs
assert result["completed"] is True
assert second_call["max_tokens"] <= 65_472
assert agent.context_compressor.context_length == 200_000
def test_output_cap_retry_when_gateway_wraps_error_as_rate_limit(self, agent):
"""Some relays wrap the upstream max-output 400 as HTTP 429. The
parseable output cap must still route into the output-cap handler
instead of burning generic rate-limit retries (#72281).
"""
self._run_wrapped_429_output_cap(agent, fallback_chain=[])
def test_wrapped_output_cap_429_not_consumed_by_eager_fallback(self, agent):
"""With a NON-EMPTY fallback chain, the eager rate-limit fallback must
NOT consume the wrapped output-cap 429 — the failure is a deterministic
request-shape problem the clamp fixes in one retry; switching provider
burns a fallback slot for nothing (#72281 ordering guard).
"""
self._run_wrapped_429_output_cap(
agent,
fallback_chain=[{"provider": "openrouter", "model": "anthropic/claude-sonnet-4"}],
)
def _run_wrapped_429_output_cap(self, agent, *, fallback_chain):
self._setup_agent(agent)
agent.api_mode = "chat_completions"
agent.provider = "custom"
agent.base_url = "http://192.168.1.254:20128/v1"
agent.model = "deepseekv4flash"
agent.max_tokens = 98_304
agent.compression_enabled = True
agent._fallback_chain = fallback_chain
agent._fallback_index = 0
agent.context_compressor.context_length = 200_000
agent.context_compressor.should_compress = MagicMock(return_value=False)
error_msg = (
"Error code: 429 - {'error': {'message': \"[400]: max_tokens "
"(98304) exceeds model's maximum output tokens (65536) for model "
"deepseek-v4-flash (ref: 37bde60f-44e7-44e2-b995-4af17fba6d6b)\", "
"'type': 'rate_limit_error', 'code': 'rate_limit_exceeded'}}"
)
exc = Exception(error_msg)
exc.status_code = 429
exc.code = "rate_limit_exceeded"
ok_resp = _mock_response(content="done", finish_reason="stop")
agent.client.chat.completions.create.side_effect = [exc, ok_resp]
mock_compress = MagicMock(return_value=(
[{"role": "user", "content": "hello"}],
"You are helpful.",
))
with (
patch.object(agent, "_persist_session"),
patch.object(agent, "_save_trajectory"),
patch.object(agent, "_cleanup_task_resources"),
patch.object(agent.context_compressor, "update_model"),
patch.object(agent, "_compress_context", mock_compress),
):
result = agent.run_conversation("hello")
assert len(agent.client.chat.completions.create.call_args_list) == 2
second_call = agent.client.chat.completions.create.call_args_list[1].kwargs
assert result["completed"] is True
assert second_call["max_tokens"] <= 65_472
assert agent.context_compressor.context_length == 200_000
# The clamp, not provider failover, must have recovered: no fallback
# slot consumed and the model unchanged.
assert agent._fallback_index == 0
assert agent.model == "deepseekv4flash"
def test_output_cap_retry_with_large_api_only_content(self, agent):
"""When a large system prompt makes api_messages huge while persisted
messages stay tiny, the retry cap must still respect provider
+62 -3
View File
@@ -68,6 +68,22 @@ class TestParseDashScopeOutputCap:
assert parse_available_output_tokens_from_error(msg) == 32768
class TestParseMaximumOutputTokensCap:
"""Some OpenAI-compatible relays report the model's separate output cap."""
def test_parenthesized_max_output_cap(self):
msg = (
"API call failed after 3 retries: [400]: max_tokens (98304) "
"exceeds model's maximum output tokens (65536)"
)
assert parse_available_output_tokens_from_error(msg) == 65536
def test_parenthesized_max_output_cap_is_output_cap(self):
assert is_output_cap_error(
"max_tokens (98304) exceeds model's maximum output tokens (65536)"
) is True
class TestIsOutputCapError:
"""`is_output_cap_error` is the broader yes/no gate that keeps an
output-cap 400 out of the compression death-loop even when we can't parse
@@ -121,10 +137,28 @@ class TestParseVllmTokenBasedOutputCap:
"output tokens."
)
def test_vllm_token_based_format(self):
# available output = 131072 - 65537 = 65535
assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 65535
# Verbatim vLLM response where the input is MEASURED, not back-computed:
# window - input != requested - 1, so the reported figure is real.
_VLLM_MSG_REAL_INPUT = (
"This model's maximum context length is 131072 tokens. However, you "
"requested 65536 output tokens and your prompt contains 100000 "
"input tokens, for a total of 165536 tokens. Please reduce the length "
"of the input prompt or the number of requested output tokens."
)
def test_vllm_token_based_format(self):
# The reported input is a LOWER BOUND that vLLM back-computes from the
# constraint (65537 == 131072 + 1 - 65536), so window - input is just
# requested - 1 and carries no information about the real prompt.
# Halve the requested cap instead so the retry actually converges.
assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 32768
def test_vllm_measured_input_is_trusted(self):
# When the input is measured rather than derived, use it as-is.
# available output = 131072 - 100000 = 31072
assert parse_available_output_tokens_from_error(
self._VLLM_MSG_REAL_INPUT
) == 31072
def test_vllm_retry_fits_inside_window(self):
# The retried cap plus the reported input must fit in the window.
@@ -132,3 +166,28 @@ class TestParseVllmTokenBasedOutputCap:
assert available is not None
assert available + 65537 <= 131072
def test_vllm_retry_converges(self):
"""The retry sequence must reach a working cap in a few attempts.
Regression test for the 65-tokens-per-retry crawl: with a 102400
window and a real prompt of ~37000 tokens, retrying from a 65536 cap
used to produce 65471 -> 65406 -> 65341 and exhaust the compression
budget without ever fitting.
"""
window, real_input, cap = 102400, 37000, 65536
for _ in range(5):
if real_input + cap <= window:
break
# vLLM's message when max_tokens is the binding constraint.
msg = (
f"This model's maximum context length is {window} tokens. "
f"However, you requested {cap} output tokens and your prompt "
f"contains at least {window + 1 - cap} input tokens, for a "
f"total of at least {window + 1} tokens."
)
available = parse_available_output_tokens_from_error(msg)
assert available is not None
assert available < cap, "each retry must lower the cap"
cap = available
assert real_input + cap <= window, f"did not converge: cap={cap}"
+16
View File
@@ -295,6 +295,22 @@ def _build_persisted_message(
return msg
_PERSISTED_PATH_RE = re.compile(r"^Full output saved to: (.+)$", re.MULTILINE)
def extract_persisted_path(content: str) -> str | None:
"""Return the file path from a <persisted-output> replacement block.
Used by the result-reference stubbing guard (agent/tool_guardrails.py) so
a stub referencing a persisted first occurrence can carry the spillover
path instead of dangling. Returns None for non-persisted content.
"""
if not isinstance(content, str) or PERSISTED_OUTPUT_TAG not in content:
return None
match = _PERSISTED_PATH_RE.search(content)
return match.group(1).strip() if match else None
def maybe_persist_tool_result(
content: str,
tool_name: str,
@@ -28,6 +28,7 @@ import {
setTerminalBackgroundHex,
setTerminalForegroundHex,
setXtversionName,
skipKittyKeyboardProtocol,
supportsExtendedKeys
} from '../terminal.js'
import {
@@ -333,9 +334,13 @@ export default class App extends PureComponent<Props, State> {
// distinguishable from ctrl+<letter>. We write both the kitty stack
// push (CSI >1u) and xterm modifyOtherKeys level 2 (CSI >4;2m) —
// terminals honor whichever they implement (tmux only accepts the
// latter).
// latter). Ghostty gets only modifyOtherKeys — its kitty
// disambiguate mode strips Alt from Backspace (see
// skipKittyKeyboardProtocol).
if (supportsExtendedKeys()) {
this.props.stdout.write(ENABLE_KITTY_KEYBOARD)
if (!skipKittyKeyboardProtocol()) {
this.props.stdout.write(ENABLE_KITTY_KEYBOARD)
}
this.props.stdout.write(ENABLE_MODIFY_OTHER_KEYS)
}
+11 -2
View File
@@ -77,6 +77,7 @@ import {
} from './selection.js'
import {
needsAltScreenResizeScrollbackClear,
skipKittyKeyboardProtocol,
supportsExtendedKeys,
SYNC_OUTPUT_SUPPORTED,
type Terminal,
@@ -733,7 +734,11 @@ export default class Ink {
// without the pop we'd accumulate depth on each editor round-trip).
this.options.stdout.write(
'\x1b[?1004h' +
(supportsExtendedKeys() ? DISABLE_KITTY_KEYBOARD + ENABLE_KITTY_KEYBOARD + ENABLE_MODIFY_OTHER_KEYS : '')
(supportsExtendedKeys()
? DISABLE_KITTY_KEYBOARD +
(skipKittyKeyboardProtocol() ? '' : ENABLE_KITTY_KEYBOARD) +
ENABLE_MODIFY_OTHER_KEYS
: '')
)
}
onRender() {
@@ -1472,7 +1477,11 @@ export default class Ink {
// Pop-before-push keeps Kitty stack depth at 1 instead of accumulating
// on each call.
if (supportsExtendedKeys()) {
this.options.stdout.write(DISABLE_KITTY_KEYBOARD + ENABLE_KITTY_KEYBOARD + ENABLE_MODIFY_OTHER_KEYS)
this.options.stdout.write(
DISABLE_KITTY_KEYBOARD +
(skipKittyKeyboardProtocol() ? '' : ENABLE_KITTY_KEYBOARD) +
ENABLE_MODIFY_OTHER_KEYS
)
}
if (!this.altScreenActive) {
@@ -74,3 +74,36 @@ describe('writeDiffToTerminal sync-marker gating (#66490 main-screen gap)', () =
expect(out).toContain('frame')
})
})
describe('skipKittyKeyboardProtocol', () => {
// Ghostty's kitty disambiguate mode strips the Alt modifier from
// Backspace (Option+Backspace arrives as bare \x7f instead of CSI-u
// \x1b[127;3u), so Ghostty must get only the modifyOtherKeys push.
// Mirrors cli.py's _GHOSTTY_EXTENDED_ENTER_KEYS_SEQ exception.
it('skips the kitty protocol push for ghostty', async () => {
const { env } = await import('../utils/env.js')
const { skipKittyKeyboardProtocol } = await import('./terminal.js')
const saved = env.terminal
try {
env.terminal = 'ghostty'
expect(skipKittyKeyboardProtocol()).toBe(true)
} finally {
env.terminal = saved
}
})
it.each(['iTerm.app', 'kitty', 'WezTerm', 'tmux', 'windows-terminal', 'vscode'])(
'keeps the dual push for %s',
async terminal => {
const { env } = await import('../utils/env.js')
const { skipKittyKeyboardProtocol } = await import('./terminal.js')
const saved = env.terminal
try {
env.terminal = terminal
expect(skipKittyKeyboardProtocol()).toBe(false)
} finally {
env.terminal = saved
}
}
)
})
@@ -326,6 +326,16 @@ export function supportsExtendedKeys(): boolean {
return EXTENDED_KEYS_TERMINALS.includes(env.terminal ?? '')
}
/** True when the Kitty keyboard protocol push (CSI >1u) must be skipped for
* this terminal even though extended keys are supported. Ghostty's Kitty
* disambiguate-mode implementation strips the Alt modifier from Backspace —
* Option+Backspace arrives as bare \x7f instead of CSI-u \x1b[127;3u —
* breaking backward-kill-word. Ghostty implements modifyOtherKeys correctly,
* so it gets only that push (mirrors cli.py's Ghostty exception). */
export function skipKittyKeyboardProtocol(): boolean {
return env.terminal === 'ghostty'
}
/** True if the terminal scrolls the viewport when it receives cursor-up
* sequences that reach above the visible area. On Windows, conhost's
* SetConsoleCursorPosition follows the cursor into scrollback
+1
View File
@@ -94,6 +94,7 @@ Groups are standalone rows in the same activity-ordered roster as Bot DMs. A Bot
Bots message each other with attribution, and you can hand work off from any chat:
- **@mentions** — type `@researcher have a look at this` in any chat and the active Bot hands the message off, waits for the reply, and reports back. Mention names are validated against the live roster, so an email address or an unknown `@` passes through untouched.
- **Renamed Bots keep their tags in sync** — give a Bot a friendly name (the pencil in its chat header, or `hermes profile rename`) and it becomes taggable by that name: a Bot titled *Research Buddy* answers to `@research-buddy` (and `@researchbuddy`), in regular chats and in group rooms alike. The composer's `@` autocomplete offers the renamed tag and also matches when you type the old profile name, which keeps resolving too.
- **@mentions across machines** — mentioning a Bot that lives on another registered connection (use its `@name-device` handle when names collide) delivers over the Connections registry in the background: the active Bot stays on this device, the desktop routes the message to the recipient's machine, and the reply is relayed back attributed to that agent. Your window's gateway never switches.
- **Direct messages** — a Bot reaches a teammate's Bot Chat through the standard CLI: it writes the message to a temp file (opening with the `Message from 🤖 <sender> (@<sender>):` prefix), then runs `hermes -p <bot> chat --in ~ -c "Bot Chat" --create-if-missing -Q --query-file <file>`. The file transport means nothing is shell-interpreted — quotes, `$(...)`, and backticks in the message arrive verbatim. The receiving Bot sees the message the next time it runs and knows how to reply, because the messaging protocol is part of its Bot Chat system prompt.
+37
View File
@@ -58,6 +58,43 @@ hermes -w # Interactive mode in worktree
hermes -w -z "Fix issue #123" # Single query in worktree
```
### Worktree cleanup
`hermes -w` sessions create disposable worktrees under `<repo>/.worktrees/`.
A conservative pruner runs automatically at startup (it only removes clean,
fully-merged scratch trees past an age threshold), but preserved trees and
merged local branches still accumulate on busy machines. Reclaim them
explicitly:
```bash
hermes worktree list # audit: age, size, verdict, reason per tree
hermes worktree prune # remove safe trees + delete merged branches
hermes worktree prune --dry-run # show the plan without changing anything
hermes worktree prune --trees-only # leave local branches alone
hermes worktree prune --branches-only # leave worktrees alone
```
Inside a session, `/worktree prune [--dry-run]` does the same (and never
touches the tree the session is running in).
Safety guarantees (all modes, any age):
- Uncommitted **tracked** changes are never deleted.
- **Unique unpushed commits** are never deleted — commits that were
rebase/squash-merged upstream are detected via `git cherry`
patch-equivalence and count as merged, which is what lets the dominant
"merged PR, tree preserved forever" leak finally reclaim.
- Trees **in use by a running hermes session** are never touched.
- **Untracked-only scratch** (PR body drafts, notes) is archived to
`~/.hermes/archive/worktree-prune/` before its tree is removed — never
destroyed.
- Branch deletion is content-gated, not name-gated: any local branch whose
commits are all on upstream is safe to delete; branches with unique work,
checked-out branches, and `main`/`master`/`develop` are always kept.
When `.worktrees/` grows past 10 trees or 5 GB, startup prints a one-line
notice pointing at these commands.
### Plugin management
The `hermes plugins` commands manage native Hermes plugins and portable Agent
+2
View File
@@ -1752,6 +1752,8 @@ agent:
stall_guards: false
```
The same gate also enables **result-reference stubbing**: when a re-issued identical tool call returns a byte-identical fresh result, the duplicate payload enters context as a short reference stub pointing at the earlier result (tool name, `tool_call_id`, an args summary, and — if the first result was persisted to disk — its spillover path) instead of repeating the full output. The tool still executes every time, so polling semantics are preserved: a changed result always flows through whole. Results under 512 characters, error results, and multimodal results are never stubbed, and pollers *are* stubbed (an unchanged poll is exactly the case where the duplicate payload carries no information).
## TTS Configuration
```yaml
@@ -284,6 +284,23 @@ the model substitutes the platform's idiomatic shortcut and app name):
During all of this, your cursor stays wherever you left it and the email
app never comes to front.
## Receiving the actual screenshot
Screenshots taken during computer control are normally internal — they exist
so the model can see the screen, and the agent replies in text. But every
image capture also saves a bounded, shareable copy under Hermes' image cache
and reports its path, so on attachment-capable surfaces (Telegram, Discord,
Desktop, and other gateway platforms) you can simply ask:
> *"Send me a screenshot of my screen."*
and the agent delivers the real image as a native attachment, not just a
description. On the CLI there is no attachment channel, so the agent gives
you the saved file's path instead.
Only the 20 most recent capture files are kept, and screenshots are never
sent automatically — only when you ask for one.
## Provider compatibility
| Provider | Vision? | Works? | Notes |
+1 -1
View File
@@ -266,7 +266,7 @@ first, set `memory.write_approval: true`. It's a simple on/off gate applied to
| `false` (default) | Write freely — the gate is off (the pre-gate behaviour). |
| `true` | Require approval before anything is saved. In the interactive CLI, foreground writes prompt you inline (entries are small enough to read in full). Everywhere else — messaging platforms, scripts, and the background self-improvement review — writes are **staged** for review with `/memory pending`. |
> To turn memory off entirely (not just gate it), set `memory_enabled: false`.
> To turn memory off entirely (not just gate it), set both `memory_enabled: false` and `user_profile_enabled: false`. When both built-in stores are disabled, the built-in `memory` tool is automatically hidden.
Review staged writes from the CLI or any messaging platform: