Merge remote-tracking branch 'origin/main' into feat/keyless-tavily-firecrawl-failover
This commit is contained in:
@@ -350,6 +350,147 @@ _SUMMARY_END_MARKER = (
|
||||
_MERGED_PRIOR_CONTEXT_HEADER = "[PRIOR CONTEXT — for reference only; not a new message]"
|
||||
_MERGED_SUMMARY_DELIMITER = "[END OF PRIOR CONTEXT — COMPACTION SUMMARY BELOW]"
|
||||
|
||||
_SALVAGE_SUMMARY_MAX_CHARS = 8_000
|
||||
_SALVAGE_KEEP_RECENT_TOOLS = 2
|
||||
|
||||
|
||||
def _looks_like_compaction_summary(msg: Dict[str, Any], content: str) -> bool:
|
||||
# Only cap a standalone handoff. Merged carriers preserve a real tail ask
|
||||
# in the same content string; truncating those could delete live user text.
|
||||
if not content.rstrip().endswith(_SUMMARY_END_MARKER):
|
||||
return False
|
||||
if content.startswith(_MERGED_PRIOR_CONTEXT_HEADER):
|
||||
return False
|
||||
# Content heuristics alone must never authorize mutating a live turn.
|
||||
# Compressor-generated summaries carry this private marker; ordinary
|
||||
# user input — and live assistant replies or kept tool bodies that
|
||||
# merely quote a summary header/marker — do not. Tool messages are
|
||||
# handled exclusively by the stub/keep-recent pass, never the cap.
|
||||
if msg.get("role") == "tool":
|
||||
return False
|
||||
if (
|
||||
msg.get("role") in ("user", "assistant")
|
||||
and not msg.get(COMPRESSED_SUMMARY_METADATA_KEY)
|
||||
):
|
||||
return False
|
||||
head = content[:280]
|
||||
return (
|
||||
bool(msg.get(COMPRESSED_SUMMARY_METADATA_KEY))
|
||||
or "CONTEXT COMPACTION" in head
|
||||
or "[CONTEXT COMPACTION]" in head
|
||||
or "Conversation Summary" in head
|
||||
)
|
||||
|
||||
|
||||
def _salvage_reduce_todo_snapshot(out: List[Dict[str, Any]]) -> None:
|
||||
"""Last-resort shrink: reduce or drop the synthetic todo snapshot.
|
||||
|
||||
The snapshot is the only in-transcript todo re-injection at a compaction
|
||||
boundary, and since 7a16840add the pruned-skill reload notice is coupled
|
||||
into the same string — so it is only touched when the cheaper shrink ops
|
||||
could not get under budget. When the snapshot carries a reload notice,
|
||||
keep just the notice (the coupling must survive salvage); otherwise drop
|
||||
the row entirely.
|
||||
"""
|
||||
from agent.conversation_compression import _PRUNED_SKILL_RELOAD_NOTICE_HEADER
|
||||
|
||||
for i in range(len(out) - 1, -1, -1):
|
||||
msg = out[i]
|
||||
if not isinstance(msg, dict):
|
||||
continue
|
||||
if msg.get("_todo_snapshot_synthetic") and msg.get("role") == "user":
|
||||
content = msg.get("content")
|
||||
notice_idx = (
|
||||
content.find(_PRUNED_SKILL_RELOAD_NOTICE_HEADER)
|
||||
if isinstance(content, str)
|
||||
else -1
|
||||
)
|
||||
if isinstance(content, str) and notice_idx >= 0:
|
||||
msg["content"] = content[notice_idx:]
|
||||
else:
|
||||
del out[i]
|
||||
return
|
||||
|
||||
|
||||
def salvage_grown_transcript(
|
||||
original: List[Dict[str, Any]],
|
||||
candidate: List[Dict[str, Any]],
|
||||
budget: Optional[int] = None,
|
||||
) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Mechanically shrink a compression candidate, or return ``None``.
|
||||
|
||||
Already-compacted middles can be summarized slightly larger while retained
|
||||
tool bodies, stale reasoning, or a synthetic todo snapshot tip the final
|
||||
candidate over the input size. Work on copies and admit the salvage only
|
||||
when the same rough estimator proves it is strictly smaller than the input.
|
||||
|
||||
Shrink order is cheapest-information-loss first: stale reasoning keys and
|
||||
codex replay sidecars, then old tool bodies, then an oversized summary cap.
|
||||
The synthetic todo snapshot (which carries the pruned-skill reload notice,
|
||||
see ``_salvage_reduce_todo_snapshot``) is only reduced as a LAST resort
|
||||
when everything else still leaves the candidate at or over budget.
|
||||
"""
|
||||
if not candidate or not original:
|
||||
return None
|
||||
if budget is None:
|
||||
budget = estimate_messages_tokens_rough(original)
|
||||
if budget <= 0:
|
||||
return None
|
||||
|
||||
out: List[Dict[str, Any]] = []
|
||||
tool_indices: List[int] = []
|
||||
last_assistant_idx = -1
|
||||
for msg in candidate:
|
||||
if not isinstance(msg, dict):
|
||||
out.append(msg)
|
||||
continue
|
||||
copied = dict(msg)
|
||||
out.append(copied)
|
||||
role = copied.get("role")
|
||||
if role == "tool":
|
||||
tool_indices.append(len(out) - 1)
|
||||
elif role == "assistant":
|
||||
last_assistant_idx = len(out) - 1
|
||||
|
||||
salvage_reasoning_keys = _NEWEST_TURN_ONLY_BUDGET_KEYS + ("reasoning_details",)
|
||||
keep_tools = set(tool_indices[-_SALVAGE_KEEP_RECENT_TOOLS:])
|
||||
for index, msg in enumerate(out):
|
||||
if not isinstance(msg, dict):
|
||||
continue
|
||||
if msg.get("role") == "assistant" and index != last_assistant_idx:
|
||||
for key in salvage_reasoning_keys:
|
||||
msg.pop(key, None)
|
||||
if msg.get("role") == "tool" and index not in keep_tools:
|
||||
content = msg.get("content")
|
||||
if isinstance(content, str) and len(content) > _PRUNE_MIN_CHARS:
|
||||
msg["content"] = _PRUNED_TOOL_PLACEHOLDER
|
||||
content = msg.get("content")
|
||||
if (
|
||||
isinstance(content, str)
|
||||
and len(content) > _SALVAGE_SUMMARY_MAX_CHARS
|
||||
and _looks_like_compaction_summary(msg, content)
|
||||
):
|
||||
msg["content"] = (
|
||||
content[:_SALVAGE_SUMMARY_MAX_CHARS].rstrip()
|
||||
+ "\n…[summary truncated so compaction can shrink]\n\n"
|
||||
+ _SUMMARY_END_MARKER
|
||||
)
|
||||
# Heavier codex replay sidecars (encrypted reasoning blobs) — reuse the
|
||||
# proven prune with its last-user-turn safety boundary (#71058).
|
||||
_prune_stale_reasoning_replay(out)
|
||||
|
||||
if estimate_messages_tokens_rough(out) >= budget:
|
||||
_salvage_reduce_todo_snapshot(out)
|
||||
|
||||
if not any(
|
||||
isinstance(message, dict) and message.get("role") == "user"
|
||||
for message in out
|
||||
):
|
||||
return None
|
||||
if estimate_messages_tokens_rough(out) < budget:
|
||||
return out
|
||||
return None
|
||||
|
||||
# Handoff prefixes that shipped in earlier releases. A summary persisted under
|
||||
# one of these can be inherited into a resumed lineage (#35344); when it is
|
||||
# re-normalized on re-compaction we must strip the OLD prefix too, otherwise the
|
||||
@@ -2373,6 +2514,31 @@ class ContextCompressor(ContextEngine):
|
||||
self._ineffective_compression_count = count
|
||||
self._persist_ineffective_compression_count()
|
||||
|
||||
def record_rejected_compaction(self) -> None:
|
||||
"""Record one compaction whose result was REJECTED before committing.
|
||||
|
||||
The anti-growth guard in the commit layer (conversation_compression)
|
||||
discards a candidate that would grow the transcript and keeps the
|
||||
original. Without recording the attempt, the anti-thrash breaker
|
||||
never sees a strike, so automatic compression retries the SAME
|
||||
unchanged transcript on every turn — same summary request, same
|
||||
refusal, same user-facing warning (#88568). This counts one
|
||||
ineffective strike (persisted, so the normal >= 2 latch and its
|
||||
recovery window apply) WITHOUT arming post-compaction real-usage
|
||||
verification — nothing was committed, so there is no new compaction
|
||||
to verify — and without touching the fallback-summary streak (no
|
||||
summary was accepted).
|
||||
"""
|
||||
self._record_ineffective_compression_verdict(
|
||||
self._ineffective_compression_count + 1
|
||||
)
|
||||
if not self.quiet_mode:
|
||||
logger.warning(
|
||||
"Compaction rejected before commit (would grow the "
|
||||
"transcript); ineffective_compression_count=%d",
|
||||
self._ineffective_compression_count,
|
||||
)
|
||||
|
||||
def record_completed_compaction(
|
||||
self, *, used_fallback: bool = False, feasibility_skip: bool = False,
|
||||
) -> None:
|
||||
|
||||
@@ -3386,6 +3386,27 @@ def compress_context(
|
||||
# transcript stays untouched and durable.
|
||||
_rough_in = estimate_messages_tokens_rough(messages)
|
||||
_rough_out = estimate_messages_tokens_rough(compressed)
|
||||
if _rough_out > _rough_in:
|
||||
# Todo refresh and user-turn anchoring happen after the
|
||||
# compressor's own size check, so they can tip a break-even
|
||||
# candidate over. Give it one mechanical salvage pass.
|
||||
from agent.context_compressor import salvage_grown_transcript
|
||||
|
||||
_salvaged = salvage_grown_transcript(
|
||||
messages, compressed, budget=_rough_in
|
||||
)
|
||||
if _salvaged is not None:
|
||||
_salv_est = estimate_messages_tokens_rough(_salvaged)
|
||||
if _salv_est < _rough_in:
|
||||
logger.info(
|
||||
"Compression salvage recovered a shrinking "
|
||||
"transcript (session=%s, ~%s -> ~%s tokens)",
|
||||
agent.session_id or "none",
|
||||
f"{_rough_in:,}",
|
||||
f"{_salv_est:,}",
|
||||
)
|
||||
compressed = _salvaged
|
||||
_rough_out = _salv_est
|
||||
if _rough_out > _rough_in:
|
||||
logger.warning(
|
||||
"Compression refused: compressed transcript would be "
|
||||
@@ -3425,6 +3446,20 @@ def compress_context(
|
||||
split_status="aborted",
|
||||
failure_class="would_grow",
|
||||
)
|
||||
# Record the rejected attempt as an ineffective
|
||||
# compaction strike so the anti-thrash breaker latches
|
||||
# after the normal threshold. Without this, the unchanged
|
||||
# transcript stays over the compression threshold and
|
||||
# automatic compression retries the identical summary
|
||||
# request on every turn (#88568). Manual /compress keeps
|
||||
# bypassing the latch (force=True skips the guards).
|
||||
try:
|
||||
agent.context_compressor.record_rejected_compaction()
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"could not record rejected-compaction strike",
|
||||
exc_info=True,
|
||||
)
|
||||
_release_lock()
|
||||
return messages, _existing_sp
|
||||
|
||||
|
||||
@@ -2765,7 +2765,34 @@ def run_conversation(
|
||||
request_pressure_tokens,
|
||||
int(getattr(_compressor, "threshold_tokens", 0) or 0),
|
||||
)
|
||||
|
||||
elif not agent.compression_enabled and len(messages) > 1:
|
||||
# Uncompressed session guard (#89297): compression is disabled, so
|
||||
# nothing shrinks a growing session. Reuse the unconditionally
|
||||
# computed request estimate (zero marginal cost — this site runs
|
||||
# before every provider request, covering turn-start AND mid-turn
|
||||
# tool-result growth) and surface a deduped, actionable warning
|
||||
# when the request exceeds the model context window. The dedup is
|
||||
# re-armed by the turn-context preflight once the session is back
|
||||
# under the window (manual /compress works with compression
|
||||
# disabled), so the guard warns again on a later re-overflow.
|
||||
# context_compressor always exists (agent_init constructs it even
|
||||
# when compression is disabled) and its context_length property
|
||||
# hard-floors at a positive default — no metadata re-resolution
|
||||
# needed here.
|
||||
_ctx_len = getattr(
|
||||
getattr(agent, "context_compressor", None), "context_length", None
|
||||
)
|
||||
if (
|
||||
isinstance(_ctx_len, int)
|
||||
and _ctx_len > 0
|
||||
and request_pressure_tokens > _ctx_len
|
||||
):
|
||||
_warn_fn = getattr(
|
||||
agent, "_warn_uncompressed_context_overflow", None
|
||||
)
|
||||
if callable(_warn_fn):
|
||||
_warn_fn(request_pressure_tokens, _ctx_len)
|
||||
|
||||
# Thinking spinner for quiet mode (animated during API call)
|
||||
thinking_spinner = None
|
||||
|
||||
@@ -5304,6 +5331,21 @@ def run_conversation(
|
||||
FailoverReason.billing,
|
||||
FailoverReason.upstream_rate_limit,
|
||||
}
|
||||
# Relay-wrapped output-cap errors: some gateways wrap an
|
||||
# upstream "[400]: max_tokens (...) exceeds model's maximum
|
||||
# output tokens (...)" as HTTP 429, which classifies as
|
||||
# rate_limit. The failure is a deterministic request-shape
|
||||
# problem — falling back to another provider (or burning
|
||||
# generic retries) can't fix it, but the output-cap clamp
|
||||
# below can, in one retry (#72281). Parse once here; the
|
||||
# result gates both the eager-fallback exemption and the
|
||||
# widened is_context_length_error entry, and is reused as
|
||||
# available_out inside the handler.
|
||||
_wrapped_output_cap_budget = (
|
||||
parse_available_output_tokens_from_error(error_msg)
|
||||
if classified.reason == FailoverReason.rate_limit
|
||||
else None
|
||||
)
|
||||
_is_transport_failure = classified.reason in {
|
||||
FailoverReason.timeout,
|
||||
FailoverReason.overloaded,
|
||||
@@ -5321,7 +5363,7 @@ def run_conversation(
|
||||
if _is_zai_coding_overload:
|
||||
max_retries = max(max_retries, zai_coding_overload_retry_ceiling())
|
||||
_should_fallback = (
|
||||
is_rate_limited
|
||||
(is_rate_limited and _wrapped_output_cap_budget is None)
|
||||
or (_is_transport_failure and retry_count >= 2)
|
||||
)
|
||||
if _should_fallback and agent._fallback_index < len(agent._fallback_chain):
|
||||
@@ -5614,6 +5656,11 @@ def run_conversation(
|
||||
# server disconnect + large session pattern (#2153).
|
||||
is_context_length_error = (
|
||||
classified.reason == FailoverReason.context_overflow
|
||||
# Relay-wrapped output-cap 429s (parsed once above, where
|
||||
# the eager-fallback exemption is gated) route into the
|
||||
# output-cap clamp below instead of provider failover or
|
||||
# generic retries (#72281).
|
||||
or _wrapped_output_cap_budget is not None
|
||||
)
|
||||
|
||||
if is_context_length_error:
|
||||
|
||||
@@ -1659,10 +1659,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]:
|
||||
# The input itself fits — this is purely an output-cap error, so reduce
|
||||
# max_tokens and retry; do NOT compress.
|
||||
"range of max_tokens should be" in error_lower
|
||||
) or (
|
||||
# OpenAI-compatible relays may reject a request whose output cap exceeds
|
||||
# the model's separate completion-token limit, e.g.
|
||||
# "max_tokens (98304) exceeds model's maximum output tokens (65536)"
|
||||
# This is independent of the input context window.
|
||||
"exceeds model" in error_lower
|
||||
and "maximum output tokens" in error_lower
|
||||
)
|
||||
if not is_output_cap_error:
|
||||
return None
|
||||
|
||||
# Generic model-output-cap form:
|
||||
# "max_tokens (98304) exceeds model's maximum output tokens (65536)"
|
||||
_m_max_output = re.search(
|
||||
r'exceeds model(?:\'s)? maximum output tokens\s*\(?\s*(\d+)\s*\)?',
|
||||
error_lower,
|
||||
)
|
||||
if _m_max_output:
|
||||
_cap = int(_m_max_output.group(1))
|
||||
if _cap >= 1:
|
||||
return _cap
|
||||
|
||||
# DashScope / Alibaba range form: "Range of max_tokens should be [1, 65536]".
|
||||
# The upper bound is the available output cap.
|
||||
_m_range = re.search(
|
||||
@@ -1726,11 +1744,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]:
|
||||
# Available output = window - input. When the input alone is at or over
|
||||
# the window this stays None, so the caller correctly falls through to
|
||||
# compression instead of futilely shrinking the output cap.
|
||||
#
|
||||
# Caveat: when max_tokens is the BINDING constraint, vLLM does not report
|
||||
# the real prompt size at all. It back-computes a lower bound from the
|
||||
# constraint itself -- "at least N input tokens" where
|
||||
# N == window + 1 - requested_output -- so window - N is always exactly
|
||||
# requested_output - 1. Subtracting the caller's safety margin then walks
|
||||
# the cap down ~65 tokens per retry while the reported input walks up by
|
||||
# the same amount, burning every compression attempt without ever fitting.
|
||||
# Detect that degenerate case and halve the requested cap instead: it
|
||||
# carries the same guarantee (strictly below what was rejected) and
|
||||
# converges in one or two retries.
|
||||
_m_vllm_input = re.search(
|
||||
r'prompt contains (?:at least )?(\d+)\s*input tokens', error_lower
|
||||
)
|
||||
if _m_ctx_tok and _m_vllm_input:
|
||||
_available = int(_m_ctx_tok.group(1)) - int(_m_vllm_input.group(1))
|
||||
_m_requested_out = re.search(r'requested (\d+)\s*output tokens', error_lower)
|
||||
if 'at least' in error_lower and _m_requested_out:
|
||||
_requested_out = int(_m_requested_out.group(1))
|
||||
if _available >= _requested_out - 1:
|
||||
# The budget is derived from the constraint, not measured.
|
||||
return max(1, _requested_out // 2)
|
||||
if _available >= 1:
|
||||
return _available
|
||||
|
||||
@@ -1782,6 +1817,8 @@ def is_output_cap_error(error_msg: str) -> bool:
|
||||
or "should be" in error_lower # generic "max_tokens should be <= N"
|
||||
or "less than or equal" in error_lower
|
||||
or "must be" in error_lower
|
||||
or ("exceeds model" in error_lower
|
||||
and "maximum output tokens" in error_lower)
|
||||
)
|
||||
if not output_cap_signal:
|
||||
return False
|
||||
|
||||
@@ -48,12 +48,31 @@ from tools.thread_context import propagate_context_to_thread
|
||||
from tools.tool_result_storage import (
|
||||
maybe_persist_tool_result,
|
||||
enforce_turn_budget,
|
||||
extract_persisted_path,
|
||||
)
|
||||
from tools.budget_config import BudgetConfig, DEFAULT_BUDGET, budget_for_context_window
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _record_persisted_path_for_stub(agent, tool_call_id: str, function_result) -> None:
|
||||
"""Tell the stall guards where a persisted result's full content lives.
|
||||
|
||||
When a large result is spilled to disk (<persisted-output> preview), a
|
||||
later result-reference stub pointing at that first occurrence must carry
|
||||
the spillover file path so the reference can't dangle. Best-effort: never
|
||||
lets bookkeeping break tool execution.
|
||||
"""
|
||||
try:
|
||||
if not isinstance(function_result, str):
|
||||
return
|
||||
path = extract_persisted_path(function_result)
|
||||
if path:
|
||||
agent._tool_guardrails.record_persisted_result(tool_call_id, path)
|
||||
except Exception as exc:
|
||||
logger.debug("persisted-path record for result stub failed: %s", exc)
|
||||
|
||||
|
||||
def _ensure_file_checkpoint(
|
||||
agent,
|
||||
function_name: str,
|
||||
@@ -1733,6 +1752,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
|
||||
function_args,
|
||||
function_result,
|
||||
failed=is_error,
|
||||
tool_call_id=getattr(tc, "id", "") or "",
|
||||
)
|
||||
|
||||
if is_error:
|
||||
@@ -1767,6 +1787,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
|
||||
env=get_active_env(effective_task_id),
|
||||
config=_tool_budget,
|
||||
) if not _is_multimodal_tool_result(function_result) else function_result
|
||||
_record_persisted_path_for_stub(agent, tc.id, function_result)
|
||||
|
||||
subdir_hints = agent._subdirectory_hints.check_tool_call(name, args)
|
||||
if subdir_hints:
|
||||
@@ -2572,6 +2593,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
|
||||
function_args,
|
||||
function_result,
|
||||
failed=_is_error_result,
|
||||
tool_call_id=getattr(tool_call, "id", "") or "",
|
||||
)
|
||||
result_preview = function_result if agent.verbose_logging else (
|
||||
function_result[:200] if len(function_result) > 200 else function_result
|
||||
@@ -2610,6 +2632,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
|
||||
env=get_active_env(effective_task_id),
|
||||
config=_tool_budget,
|
||||
) if not _is_multimodal_tool_result(function_result) else function_result
|
||||
_record_persisted_path_for_stub(agent, tool_call.id, function_result)
|
||||
|
||||
# Discover subdirectory context files from tool arguments
|
||||
subdir_hints = agent._subdirectory_hints.check_tool_call(function_name, function_args)
|
||||
|
||||
+151
-26
@@ -85,6 +85,19 @@ _STALL_GUARD_REPEATABLE_SUFFIXES = (
|
||||
# catching the observed re-issue loops (3x/4x identical calls in eval traces).
|
||||
STALL_GUARD_IDENTICAL_CALL_THRESHOLD = 3
|
||||
|
||||
# Result-reference stubbing (agent.stall_guards): from the 2nd consecutive
|
||||
# identical call whose FRESH result is byte-identical to the previous one,
|
||||
# the duplicate payload is replaced in context by a short reference stub.
|
||||
# Results under this size aren't worth stubbing (the stub itself plus the
|
||||
# lost locality outweigh the savings), and error results are never stubbed
|
||||
# (the model must see every fresh error verbatim).
|
||||
IDENTICAL_RESULT_STUB_MIN_CHARS = 512
|
||||
|
||||
# How much of the canonical args JSON the stub carries so the model still
|
||||
# knows WHAT the referenced call was even if context compression later
|
||||
# evicts the referenced result (cheap dangling-reference mitigation).
|
||||
_RESULT_STUB_ARGS_PREVIEW_CHARS = 120
|
||||
|
||||
|
||||
def is_stall_guard_repeatable(tool_name: str) -> bool:
|
||||
"""Whether a tool is exempt from the identical-call loop notice."""
|
||||
@@ -206,6 +219,21 @@ class LoopCapConfig:
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class IdenticalCallObservation:
|
||||
"""Outcome of observing one completed tool call for the stall guards.
|
||||
|
||||
``notice`` is the identical-call loop-breaker notice (appended after the
|
||||
result). ``stub`` is the result-reference replacement for a byte-identical
|
||||
duplicate result (replaces the result content). Both may be set on the
|
||||
same call (3rd+ identical call): the stub replaces the payload and the
|
||||
notice is appended after it.
|
||||
"""
|
||||
|
||||
notice: str | None = None
|
||||
stub: str | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ToolCallSignature:
|
||||
"""Stable, non-reversible identity for a tool name plus canonical args."""
|
||||
@@ -320,9 +348,21 @@ class ToolCallGuardrailController:
|
||||
# results were also identical. Any different call — or a different
|
||||
# result — resets the streak, so legitimate re-reads after edits and
|
||||
# varied polling are never flagged. Per-turn, like everything else here.
|
||||
# NOTE: open PR #85352 (patrykkopycinski) tracks no-progress loops
|
||||
# ACROSS turns via a detection window — a different mechanism from
|
||||
# this per-turn consecutive streak. Coordinate future work there.
|
||||
self._identical_streak_sig: ToolCallSignature | None = None
|
||||
self._identical_streak_result_hash: str = ""
|
||||
self._identical_streak_count: int = 0
|
||||
# tool_call_id of the FIRST call in the current streak, so a
|
||||
# result-reference stub can point at the message that carries the
|
||||
# full payload.
|
||||
self._identical_streak_first_call_id: str = ""
|
||||
# tool_call_id -> spillover file path for results that were persisted
|
||||
# out of context (persisted-output preview). Lets a reference stub
|
||||
# carry the file path so the reference can't dangle when the first
|
||||
# occurrence entered context as a preview.
|
||||
self._persisted_result_paths: dict[str, str] = {}
|
||||
# Per-turn runaway-loop cap counters. Reset every turn (this method
|
||||
# runs at the start of each run_conversation), so the caps bound a
|
||||
# single agent loop rather than accumulating across the session.
|
||||
@@ -493,44 +533,129 @@ class ToolCallGuardrailController:
|
||||
) -> str | None:
|
||||
"""Track consecutive identical calls; return a loop-breaker notice or None.
|
||||
|
||||
Fires the compact notice when the SAME tool is called with identical
|
||||
canonical arguments AND returns an identical result for the
|
||||
``STALL_GUARD_IDENTICAL_CALL_THRESHOLD``-th (and every subsequent)
|
||||
consecutive time within the turn. Purely observational — never blocks
|
||||
the call. Allowlisted pollers (``is_stall_guard_repeatable``) are
|
||||
exempt, and any intervening different call or changed result resets
|
||||
the streak. Callers append the returned notice to the tool RESULT at
|
||||
construction time, which is cache-safe: tool results are append-only
|
||||
and never mutate already-sent context.
|
||||
Back-compat wrapper around :meth:`observe_call` for callers that only
|
||||
care about the loop-breaker notice.
|
||||
"""
|
||||
if is_stall_guard_repeatable(tool_name):
|
||||
# Don't let a poller streak carry over into the next tool either.
|
||||
self._identical_streak_sig = None
|
||||
self._identical_streak_count = 0
|
||||
return None
|
||||
return self.observe_call(tool_name, args, result).notice
|
||||
|
||||
def observe_call(
|
||||
self,
|
||||
tool_name: str,
|
||||
args: Mapping[str, Any] | None,
|
||||
result: str | None,
|
||||
*,
|
||||
tool_call_id: str = "",
|
||||
failed: bool = False,
|
||||
) -> "IdenticalCallObservation":
|
||||
"""Track consecutive identical calls; return notice + dedupe stub info.
|
||||
|
||||
Two independent outputs from the same consecutive-streak tracker:
|
||||
|
||||
- ``notice``: the compact loop-breaker notice, fired when the SAME
|
||||
tool is called with identical canonical arguments AND returns an
|
||||
identical result for the ``STALL_GUARD_IDENTICAL_CALL_THRESHOLD``-th
|
||||
(and every subsequent) consecutive time within the turn. Purely
|
||||
observational — never blocks the call. Allowlisted pollers
|
||||
(``is_stall_guard_repeatable``) are exempt from the NOTICE.
|
||||
- ``stub``: a short reference replacement for the CURRENT result,
|
||||
produced from the 2nd consecutive identical call whose fresh result
|
||||
is byte-identical to the previous one. The tool still executed —
|
||||
only the context representation is deduplicated, so polling
|
||||
semantics are preserved (a changed result flows through whole and
|
||||
resets the streak). Pollers are NOT exempt from stubbing: for a
|
||||
poller, an identical result means nothing changed, which is exactly
|
||||
when the stub saves the most context and loses nothing. Results
|
||||
under ``IDENTICAL_RESULT_STUB_MIN_CHARS`` and failed/error results
|
||||
are never stubbed, and only plain-string results are considered.
|
||||
|
||||
Any intervening different call or changed result resets the streak.
|
||||
Callers substitute/append at tool RESULT construction time, which is
|
||||
cache-safe: tool results are append-only and never mutate
|
||||
already-sent context.
|
||||
"""
|
||||
is_plain_str = isinstance(result, str)
|
||||
signature = ToolCallSignature.from_call(tool_name, _coerce_args(args))
|
||||
result_hash = _result_hash(result)
|
||||
result_hash = _result_hash(result) if is_plain_str else ""
|
||||
|
||||
if (
|
||||
self._identical_streak_sig == signature
|
||||
is_plain_str
|
||||
and self._identical_streak_sig == signature
|
||||
and self._identical_streak_result_hash == result_hash
|
||||
):
|
||||
self._identical_streak_count += 1
|
||||
else:
|
||||
self._identical_streak_sig = signature
|
||||
# New streak (or non-string result, which never forms a streak —
|
||||
# multimodal content lists pass through untouched).
|
||||
self._identical_streak_sig = signature if is_plain_str else None
|
||||
self._identical_streak_result_hash = result_hash
|
||||
self._identical_streak_count = 1
|
||||
self._identical_streak_count = 1 if is_plain_str else 0
|
||||
self._identical_streak_first_call_id = tool_call_id or ""
|
||||
|
||||
count = self._identical_streak_count
|
||||
if count < STALL_GUARD_IDENTICAL_CALL_THRESHOLD:
|
||||
return None
|
||||
ordinal = f"{count}{'th' if 11 <= count % 100 <= 13 else {1: 'st', 2: 'nd', 3: 'rd'}.get(count % 10, 'th')}"
|
||||
return (
|
||||
f"[hermes note: this is the {ordinal} consecutive identical call to "
|
||||
f"{tool_name} with identical arguments returning the same result. "
|
||||
"Do not repeat it — change arguments, use a different tool, or "
|
||||
"proceed with what you have.]"
|
||||
|
||||
notice = None
|
||||
if (
|
||||
not is_stall_guard_repeatable(tool_name)
|
||||
and count >= STALL_GUARD_IDENTICAL_CALL_THRESHOLD
|
||||
):
|
||||
ordinal = f"{count}{'th' if 11 <= count % 100 <= 13 else {1: 'st', 2: 'nd', 3: 'rd'}.get(count % 10, 'th')}"
|
||||
notice = (
|
||||
f"[hermes note: this is the {ordinal} consecutive identical call to "
|
||||
f"{tool_name} with identical arguments returning the same result. "
|
||||
"Do not repeat it — change arguments, use a different tool, or "
|
||||
"proceed with what you have.]"
|
||||
)
|
||||
|
||||
stub = None
|
||||
if (
|
||||
is_plain_str
|
||||
and count >= 2
|
||||
and not failed
|
||||
and len(result) >= IDENTICAL_RESULT_STUB_MIN_CHARS
|
||||
):
|
||||
stub = self._build_result_reference_stub(tool_name, args)
|
||||
|
||||
return IdenticalCallObservation(notice=notice, stub=stub)
|
||||
|
||||
def record_persisted_result(self, tool_call_id: str, file_path: str) -> None:
|
||||
"""Remember the spillover path a persisted result was saved to.
|
||||
|
||||
When the first occurrence of a result entered context as a
|
||||
persisted-output preview, a later reference stub must carry the
|
||||
spillover file path so the reference can't dangle.
|
||||
"""
|
||||
if tool_call_id and file_path:
|
||||
self._persisted_result_paths[tool_call_id] = file_path
|
||||
|
||||
def _build_result_reference_stub(
|
||||
self, tool_name: str, args: Mapping[str, Any] | None
|
||||
) -> str:
|
||||
"""Build the reference stub replacing a byte-identical duplicate result.
|
||||
|
||||
Carries the tool name + a canonical-args preview so that even if
|
||||
context compression later evicts the referenced result, the model
|
||||
still knows WHAT the call was (cheap dangling-reference mitigation).
|
||||
"""
|
||||
try:
|
||||
args_preview = canonical_tool_args(_coerce_args(args))
|
||||
except TypeError:
|
||||
args_preview = "{}"
|
||||
if len(args_preview) > _RESULT_STUB_ARGS_PREVIEW_CHARS:
|
||||
args_preview = args_preview[:_RESULT_STUB_ARGS_PREVIEW_CHARS] + "…"
|
||||
first_id = self._identical_streak_first_call_id
|
||||
ref = f" (tool_call_id {first_id})" if first_id else ""
|
||||
stub = (
|
||||
f"[hermes note: this result is byte-identical to the {tool_name} "
|
||||
f"result earlier this turn{ref}. Refer to that result; it has not "
|
||||
f"changed. Args: {args_preview}]"
|
||||
)
|
||||
spill_path = self._persisted_result_paths.get(first_id) if first_id else None
|
||||
if spill_path:
|
||||
stub += (
|
||||
f"\n[The referenced result was persisted to: {spill_path} — "
|
||||
"page through it with read_file if you need the full content.]"
|
||||
)
|
||||
return stub
|
||||
|
||||
def _check_loop_cap(
|
||||
self,
|
||||
|
||||
@@ -1163,6 +1163,57 @@ def build_turn_context(
|
||||
agent._last_content_with_tools = None
|
||||
agent._last_content_tools_all_housekeeping = False
|
||||
agent._mute_post_response = False
|
||||
elif not agent.compression_enabled:
|
||||
# Uncompressed session guard (#89297): when compression is explicitly
|
||||
# disabled, sessions can grow past the model's context window across
|
||||
# hundreds of messages with nothing to shrink them. The warning itself
|
||||
# fires from the conversation loop's pre-API site, which reuses the
|
||||
# unconditionally computed request estimate at zero marginal cost and
|
||||
# covers both turn-start and mid-turn growth (every provider request
|
||||
# passes through it). Here we only RE-ARM the dedup once the session
|
||||
# is back under the window, so the guard can warn again after the
|
||||
# user compacts (/compress with force=True works with compression
|
||||
# disabled) and the context later regrows past the limit.
|
||||
_ctx_len = getattr(
|
||||
getattr(agent, "context_compressor", None), "context_length", None
|
||||
)
|
||||
if isinstance(_ctx_len, int) and _ctx_len > 0:
|
||||
_raw_chars = 0
|
||||
for _m in messages:
|
||||
if not isinstance(_m, dict):
|
||||
continue
|
||||
_c = _m.get("content")
|
||||
if isinstance(_c, str):
|
||||
_raw_chars += len(_c)
|
||||
elif _c:
|
||||
# Non-string, non-empty content (multimodal part lists,
|
||||
# dict payloads) defeats a char count — force the real
|
||||
# estimate by treating it as over-gate. None/"" (routine
|
||||
# assistant tool-call rows) contribute nothing.
|
||||
_raw_chars = _ctx_len + 1
|
||||
break
|
||||
# Cheap gate: a session whose raw text is under ~1/4 of the
|
||||
# window (4 chars/token upper bound) cannot be over it — skip
|
||||
# the estimator. Non-string (multimodal) content defeats a char
|
||||
# count, so any such message forces the real estimate.
|
||||
if _raw_chars <= _ctx_len:
|
||||
_clear_warn = getattr(
|
||||
agent, "_clear_context_overflow_warn", None
|
||||
)
|
||||
if callable(_clear_warn):
|
||||
_clear_warn()
|
||||
else:
|
||||
_uncompressed_tokens = estimate_request_tokens_rough(
|
||||
messages,
|
||||
system_prompt=active_system_prompt or "",
|
||||
tools=agent.tools or None,
|
||||
)
|
||||
if _uncompressed_tokens <= _ctx_len:
|
||||
_clear_warn = getattr(
|
||||
agent, "_clear_context_overflow_warn", None
|
||||
)
|
||||
if callable(_clear_warn):
|
||||
_clear_warn()
|
||||
|
||||
if _preflight_compressed:
|
||||
# Compression rebuilt the list (tail messages are fresh compaction
|
||||
|
||||
@@ -256,6 +256,7 @@ import {
|
||||
import { missingRendererAssets } from './renderer-bundle'
|
||||
import { attachRendererConsoleCapture, formatRendererBoundaryReport } from './renderer-log'
|
||||
import {
|
||||
buildInstanceWindowUrl,
|
||||
buildSessionWindowUrl,
|
||||
chatWindowWebPreferences,
|
||||
createSessionWindowRegistry,
|
||||
@@ -10870,8 +10871,10 @@ function createSessionWindow(sessionId, { watch = false } = {}) {
|
||||
// Additional full "instance" windows — peers of the primary that render the
|
||||
// COMPLETE app (sidebar, routing, its own draft) against the shared backend, so
|
||||
// a user can run multiple GUI windows at once (⌘⇧N / the "New Window" palette
|
||||
// command). Unlike the compact session windows they carry no `?win` flag. The
|
||||
// primary mainWindow stays the notification / deep-link / pet-overlay anchor and
|
||||
// command). Unlike the compact session windows they carry no `?win` flag; a
|
||||
// separate `peer=1` marker prevents them from replaying app-launch source
|
||||
// restoration after joining that shared backend. The primary mainWindow stays
|
||||
// the notification / deep-link / pet-overlay anchor and
|
||||
// is NOT tracked here. The set holds a strong reference so an open peer isn't
|
||||
// garbage-collected, and drops it on close.
|
||||
const instanceWindows = new Set<any>()
|
||||
@@ -10951,7 +10954,14 @@ function createInstanceWindow() {
|
||||
})
|
||||
|
||||
attachRendererConsoleCapture(win, 'instance', rememberLog)
|
||||
loadWindowUrl(win, DEV_SERVER || pathToFileURL(resolveRendererIndex()).toString(), 'Instance window')
|
||||
loadWindowUrl(
|
||||
win,
|
||||
buildInstanceWindowUrl({
|
||||
devServer: DEV_SERVER,
|
||||
rendererIndexPath: DEV_SERVER ? undefined : resolveRendererIndex()
|
||||
}),
|
||||
'Instance window'
|
||||
)
|
||||
|
||||
return win
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ import assert from 'node:assert/strict'
|
||||
import { test } from 'vitest'
|
||||
|
||||
import {
|
||||
buildInstanceWindowUrl,
|
||||
buildSessionWindowUrl,
|
||||
chatWindowWebPreferences,
|
||||
createSessionWindowRegistry,
|
||||
@@ -88,6 +89,19 @@ test('buildSessionWindowUrl adds the watch flag for spectator windows, before th
|
||||
assert.equal(url, 'http://localhost:5173/?win=secondary&watch=1#/abc')
|
||||
})
|
||||
|
||||
test('buildInstanceWindowUrl marks a full peer without selecting a specialized renderer', () => {
|
||||
const url = buildInstanceWindowUrl({ devServer: 'http://localhost:5173/' })
|
||||
|
||||
assert.equal(url, 'http://localhost:5173/?peer=1')
|
||||
assert.ok(!url.includes('win='))
|
||||
})
|
||||
|
||||
test('buildInstanceWindowUrl marks a packaged full peer', () => {
|
||||
const url = buildInstanceWindowUrl({ rendererIndexPath: '/opt/app/index.html' })
|
||||
|
||||
assert.match(url, /^file:\/\/.*index\.html\?peer=1$/)
|
||||
})
|
||||
|
||||
test('instanceWindowBounds cascades a new window off its source bounds', () => {
|
||||
const bounds = instanceWindowBounds({ x: 100, y: 120, width: 1400, height: 900 }, { width: 1, height: 1 })
|
||||
|
||||
|
||||
@@ -77,6 +77,23 @@ function buildSessionWindowUrl(sessionId: string, { devServer, rendererIndexPath
|
||||
return `${pathToFileURL(rendererIndexPath).toString()}${query}${route}`
|
||||
}
|
||||
|
||||
// Full peer windows render the ordinary app shell, so they deliberately do
|
||||
// not use the `win` query parameter that selects a specialized renderer. The
|
||||
// separate marker lets the renderer distinguish a peer from the one primary
|
||||
// app window: app-launch source restoration belongs to the primary only, while
|
||||
// a peer keeps the already-running backend it joined during boot.
|
||||
function buildInstanceWindowUrl({ devServer, rendererIndexPath }: any = {}) {
|
||||
const query = '?peer=1'
|
||||
|
||||
if (devServer) {
|
||||
const base = devServer.endsWith('/') ? devServer.slice(0, -1) : devServer
|
||||
|
||||
return `${base}/${query}`
|
||||
}
|
||||
|
||||
return `${pathToFileURL(rendererIndexPath).toString()}${query}`
|
||||
}
|
||||
|
||||
// Full "instance" windows (⌘⇧N / the "New Window" command) open a complete app
|
||||
// peer, not a compact chat. Cascade each one off its source window's bounds so a
|
||||
// new window doesn't land exactly on top of the one it was spawned from. Pure so
|
||||
@@ -160,6 +177,7 @@ function createSessionWindowRegistry() {
|
||||
}
|
||||
|
||||
export {
|
||||
buildInstanceWindowUrl,
|
||||
buildSessionWindowUrl,
|
||||
chatWindowWebPreferences,
|
||||
createSessionWindowRegistry,
|
||||
|
||||
@@ -22,6 +22,24 @@ export const COMPOSER_STACK_BREAKPOINT_PX = 320
|
||||
// chevron frees is spent keeping the row single for another stretch.
|
||||
export const COMPOSER_COMPACT_PILL_PX = 560
|
||||
|
||||
// The ladder keeps going below the stack breakpoint — a pane can be far
|
||||
// narrower than even the stacked controls row. Both rungs are budgeted
|
||||
// against that row's real cost: menu ~24 + surface padding 16 + the cluster
|
||||
// (~190; ~218 mid-turn with the queue button).
|
||||
//
|
||||
// At 260 the three voice toggles fold into the one menu HUD mode already
|
||||
// uses, clearing the mid-turn worst case with margin. Each stage sits clear
|
||||
// of the floor below it rather than arriving the instant the previous one
|
||||
// gives out — the mistake COMPOSER_COMPACT_PILL_PX documents.
|
||||
export const COMPOSER_FOLD_VOICE_PX = 260
|
||||
|
||||
// Type and send, nothing else. A pane can be dragged to MIN_PANE_PX (80), and
|
||||
// even with voice folded the row still costs ~150, so the last rung drops the
|
||||
// pill AND the voice menu. Both stay reachable — the model by hotkey and the
|
||||
// full picker, dictation from any wider pane — and Send fits with room to
|
||||
// spare at any width the layout tree allows (~74 all-in).
|
||||
export const COMPOSER_MINIMAL_PX = 180
|
||||
|
||||
// A single editor line is ~28px (--composer-input-min-height 1.625rem + 0.5rem
|
||||
// vertical padding). Anything taller means the text wrapped to a second line,
|
||||
// which is when the composer should expand to the stacked layout.
|
||||
|
||||
@@ -98,6 +98,37 @@ describe('HUD mode', () => {
|
||||
})
|
||||
})
|
||||
|
||||
// A tile can be narrower than the controls cost, and the row is inside an
|
||||
// overflow-hidden surface — so anything that doesn't fold gets clipped off the
|
||||
// right edge, send button first. The ladder keeps going past `stacked`: voice
|
||||
// folds into the same menu the HUD uses, then the model pill drops. Send is
|
||||
// the last thing standing.
|
||||
describe('narrow tiles', () => {
|
||||
it('folds the voice controls into one menu without entering HUD mode', () => {
|
||||
renderControls({ foldVoice: true })
|
||||
|
||||
expect(screen.getByLabelText('Voice')).toBeTruthy()
|
||||
expect(screen.queryByLabelText('Voice dictation')).toBeNull()
|
||||
expect(screen.queryByLabelText('Read replies aloud')).toBeNull()
|
||||
|
||||
// Folding is a width decision, not the HUD: no exit affordance appears.
|
||||
expect(screen.queryByLabelText('Exit HUD mode')).toBeNull()
|
||||
})
|
||||
|
||||
it('keeps Send at the tightest width, with everything else dropped', () => {
|
||||
renderControls({ foldVoice: true, minimal: true })
|
||||
|
||||
expect(screen.getByLabelText('Send')).toBeTruthy()
|
||||
expect(screen.queryByLabelText('Voice')).toBeNull()
|
||||
})
|
||||
|
||||
it('keeps Stop reachable mid-turn at the tightest width', () => {
|
||||
renderControls({ busy: true, busyAction: 'stop', foldVoice: true, hasComposerPayload: false, minimal: true })
|
||||
|
||||
expect(screen.getByLabelText('Stop')).toBeTruthy()
|
||||
})
|
||||
})
|
||||
|
||||
describe('ComposerControls shortcut tooltips', () => {
|
||||
it('shows Enter for Send', async () => {
|
||||
renderControls()
|
||||
|
||||
@@ -39,7 +39,9 @@ export function ComposerControls({
|
||||
compactModelPill = false,
|
||||
conversation,
|
||||
disabled,
|
||||
foldVoice = false,
|
||||
hasComposerPayload,
|
||||
minimal = false,
|
||||
state,
|
||||
voiceStatus,
|
||||
onDictate,
|
||||
@@ -53,7 +55,9 @@ export function ComposerControls({
|
||||
compactModelPill?: boolean
|
||||
conversation: ConversationProps
|
||||
disabled: boolean
|
||||
foldVoice?: boolean
|
||||
hasComposerPayload: boolean
|
||||
minimal?: boolean
|
||||
state: ChatBarState
|
||||
voiceStatus: VoiceStatus
|
||||
onDictate: () => void
|
||||
@@ -73,29 +77,38 @@ export function ComposerControls({
|
||||
// only when the composer is empty and a turn is running.
|
||||
const showStop = busy && !hasComposerPayload
|
||||
const showQueueButton = busyAction !== 'stop' && hasComposerPayload
|
||||
// The HUD is a Spotlight bar a few hundred pixels wide, so the four separate
|
||||
// voice toggles fold into one menu there and leave the row to the input. A
|
||||
// narrow tile hits the same wall from the other direction and folds for the
|
||||
// same reason — same controls, same state, different budget. Below that
|
||||
// even the menu goes: at `minimal` the row is the send button and nothing
|
||||
// else, which is the one thing that must survive every width.
|
||||
const foldedVoice = hudMode || foldVoice
|
||||
|
||||
const voiceControls = foldedVoice ? (
|
||||
<VoiceMenu
|
||||
autoSpeak={autoSpeak}
|
||||
disabled={disabled}
|
||||
onDictate={onDictate}
|
||||
onStartConversation={conversation.onStart}
|
||||
onToggleAutoSpeak={onToggleAutoSpeak}
|
||||
state={state}
|
||||
voiceStatus={voiceStatus}
|
||||
/>
|
||||
) : (
|
||||
<>
|
||||
<DictationButton disabled={disabled} onToggle={onDictate} state={state.voice} status={voiceStatus} />
|
||||
<AutoSpeakButton active={autoSpeak} disabled={disabled} onToggle={onToggleAutoSpeak} />
|
||||
<WakeWordButton disabled={disabled} />
|
||||
</>
|
||||
)
|
||||
|
||||
return (
|
||||
<div className="ml-auto flex shrink-0 items-center gap-(--composer-control-gap)">
|
||||
<ModelPill compact={compactModelPill} disabled={disabled} model={state.model} />
|
||||
{/* The HUD is a Spotlight bar a few hundred pixels wide, so the four
|
||||
separate voice toggles fold into one menu there and leave the row to
|
||||
the input. The docked composer has the width and keeps them inline —
|
||||
same controls, same state, different budget. */}
|
||||
{hudMode ? (
|
||||
<VoiceMenu
|
||||
autoSpeak={autoSpeak}
|
||||
disabled={disabled}
|
||||
onDictate={onDictate}
|
||||
onStartConversation={conversation.onStart}
|
||||
onToggleAutoSpeak={onToggleAutoSpeak}
|
||||
state={state}
|
||||
voiceStatus={voiceStatus}
|
||||
/>
|
||||
) : (
|
||||
<div className="ml-auto flex min-w-0 shrink items-center gap-(--composer-control-gap)">
|
||||
{minimal ? null : (
|
||||
<>
|
||||
<DictationButton disabled={disabled} onToggle={onDictate} state={state.voice} status={voiceStatus} />
|
||||
<AutoSpeakButton active={autoSpeak} disabled={disabled} onToggle={onToggleAutoSpeak} />
|
||||
<WakeWordButton disabled={disabled} />
|
||||
<ModelPill compact={compactModelPill} disabled={disabled} model={state.model} />
|
||||
{voiceControls}
|
||||
</>
|
||||
)}
|
||||
{showQueueButton ? (
|
||||
|
||||
@@ -10,7 +10,13 @@ import {
|
||||
} from '@/app/chat/surface-vars'
|
||||
import { useResizeObserver } from '@/hooks/use-resize-observer'
|
||||
|
||||
import { COMPOSER_COMPACT_PILL_PX, COMPOSER_SINGLE_LINE_MAX_PX, COMPOSER_STACK_BREAKPOINT_PX } from '../composer-utils'
|
||||
import {
|
||||
COMPOSER_COMPACT_PILL_PX,
|
||||
COMPOSER_FOLD_VOICE_PX,
|
||||
COMPOSER_MINIMAL_PX,
|
||||
COMPOSER_SINGLE_LINE_MAX_PX,
|
||||
COMPOSER_STACK_BREAKPOINT_PX
|
||||
} from '../composer-utils'
|
||||
|
||||
interface UseComposerMetricsArgs {
|
||||
composerDockRef: RefObject<HTMLDivElement | null>
|
||||
@@ -20,13 +26,36 @@ interface UseComposerMetricsArgs {
|
||||
poppedOut: boolean
|
||||
}
|
||||
|
||||
/** Every width-driven collapse stage, resolved from the composer's own width. */
|
||||
export interface ComposerFit {
|
||||
compactPill: boolean
|
||||
foldVoice: boolean
|
||||
minimal: boolean
|
||||
tight: boolean
|
||||
}
|
||||
|
||||
const ROOMY: ComposerFit = { compactPill: false, foldVoice: false, minimal: false, tight: false }
|
||||
|
||||
const fitForWidth = (width: number): ComposerFit => ({
|
||||
compactPill: width < COMPOSER_COMPACT_PILL_PX,
|
||||
foldVoice: width < COMPOSER_FOLD_VOICE_PX,
|
||||
minimal: width < COMPOSER_MINIMAL_PX,
|
||||
tight: width < COMPOSER_STACK_BREAKPOINT_PX
|
||||
})
|
||||
|
||||
const sameFit = (a: ComposerFit, b: ComposerFit) =>
|
||||
a.compactPill === b.compactPill && a.foldVoice === b.foldVoice && a.minimal === b.minimal && a.tight === b.tight
|
||||
|
||||
interface UseComposerMetricsResult extends ComposerFit {
|
||||
stacked: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* Owns the composer's *sizing* engine: the stacked-vs-inline layout decision
|
||||
* and the measured-height CSS vars the thread reads for bottom clearance. All
|
||||
* work is edge-gated — the ResizeObserver only fires on real size changes, the
|
||||
* height vars are 8px-bucketed so per-keystroke growth never invalidates the
|
||||
* tree's computed style, and `tight` only flips when it crosses the breakpoint.
|
||||
* Returns `stacked` (the only value the render needs).
|
||||
* tree's computed style, and the fit only re-renders when it crosses a stage.
|
||||
*/
|
||||
export function useComposerMetrics({
|
||||
composerDockRef,
|
||||
@@ -34,14 +63,9 @@ export function useComposerMetrics({
|
||||
composerSurfaceRef,
|
||||
editorRef,
|
||||
poppedOut
|
||||
}: UseComposerMetricsArgs): {
|
||||
compactPill: boolean
|
||||
stacked: boolean
|
||||
} {
|
||||
}: UseComposerMetricsArgs): UseComposerMetricsResult {
|
||||
const [expanded, setExpanded] = useState(false)
|
||||
const [tight, setTight] = useState(false)
|
||||
// Wider than `tight`: the pill goes icon-only before the row has to stack.
|
||||
const [compactPill, setCompactPill] = useState(false)
|
||||
const [fit, setFit] = useState<ComposerFit>(ROOMY)
|
||||
|
||||
// Edge signals, not the live text: these only re-render when emptiness / the
|
||||
// presence of a non-trailing newline actually flips, so typing within a line
|
||||
@@ -84,8 +108,7 @@ export function useComposerMetrics({
|
||||
// until a wrap or row change actually happens.
|
||||
const lastBucketedHeightRef = useRef(0)
|
||||
const lastBucketedSurfaceHeightRef = useRef(0)
|
||||
const lastTightRef = useRef<boolean | null>(null)
|
||||
const lastCompactPillRef = useRef<boolean | null>(null)
|
||||
const lastFitRef = useRef(ROOMY)
|
||||
// Mirrored into a ref so `syncComposerMetrics` stays referentially stable —
|
||||
// it's the shared ResizeObserver's handler, and a new identity every render
|
||||
// would re-register the observation.
|
||||
@@ -121,18 +144,11 @@ export function useComposerMetrics({
|
||||
const surfaceHeight = composerSurfaceRef.current?.getBoundingClientRect().height
|
||||
|
||||
if (width > 0) {
|
||||
const nextTight = width < COMPOSER_STACK_BREAKPOINT_PX
|
||||
const nextFit = fitForWidth(width)
|
||||
|
||||
if (nextTight !== lastTightRef.current) {
|
||||
lastTightRef.current = nextTight
|
||||
setTight(nextTight)
|
||||
}
|
||||
|
||||
const nextCompactPill = width < COMPOSER_COMPACT_PILL_PX
|
||||
|
||||
if (nextCompactPill !== lastCompactPillRef.current) {
|
||||
lastCompactPillRef.current = nextCompactPill
|
||||
setCompactPill(nextCompactPill)
|
||||
if (!sameFit(nextFit, lastFitRef.current)) {
|
||||
lastFitRef.current = nextFit
|
||||
setFit(nextFit)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -189,7 +205,7 @@ export function useComposerMetrics({
|
||||
}
|
||||
}, [composerRef])
|
||||
|
||||
// Both decisions come from the composer's OWN measured width, never the
|
||||
// Every decision comes from the composer's OWN measured width, never the
|
||||
// viewport's. There used to be a `(max-width: 30rem)` media query in here as
|
||||
// well, and it quietly outranked everything: any window under 480px stacked
|
||||
// the row AND compacted the pill in the same instant, regardless of how much
|
||||
@@ -199,7 +215,14 @@ export function useComposerMetrics({
|
||||
// stack) by 160px. The ResizeObserver knows the real width; the viewport is
|
||||
// not a proxy for it.
|
||||
//
|
||||
// The pill still compacts whenever the row stacks, so the controls row can't
|
||||
// over-run once it has the width to itself.
|
||||
return { compactPill: compactPill || tight, stacked: expanded || tight }
|
||||
// The ladder is monotonic: each stage implies the ones above it, so the pill
|
||||
// is always compact by the time the row stacks, and the voice controls are
|
||||
// always folded before minimal drops them.
|
||||
return {
|
||||
compactPill: fit.compactPill || fit.tight,
|
||||
foldVoice: fit.foldVoice || fit.minimal,
|
||||
minimal: fit.minimal,
|
||||
stacked: expanded || fit.tight,
|
||||
tight: fit.tight
|
||||
}
|
||||
}
|
||||
|
||||
@@ -321,7 +321,7 @@ export function ChatBar({
|
||||
return onCancel()
|
||||
}, [activeQueueSessionKeyRef, onCancel])
|
||||
|
||||
const { compactPill, stacked } = useComposerMetrics({
|
||||
const { compactPill, foldVoice, minimal, stacked } = useComposerMetrics({
|
||||
composerDockRef,
|
||||
composerRef,
|
||||
composerSurfaceRef,
|
||||
@@ -984,7 +984,9 @@ export function ChatBar({
|
||||
status: conversation.status
|
||||
}}
|
||||
disabled={disabled}
|
||||
foldVoice={foldVoice}
|
||||
hasComposerPayload={hasComposerPayload}
|
||||
minimal={minimal}
|
||||
onDictate={dictate}
|
||||
onQueue={queueDraft}
|
||||
onToggleAutoSpeak={handleToggleAutoSpeak}
|
||||
@@ -1242,7 +1244,13 @@ export function ChatBar({
|
||||
{hudMode && busy && <span aria-hidden className="arc-border arc-composer" />}
|
||||
<div
|
||||
className={cn(
|
||||
'group/composer-surface relative z-4 isolate grid grid-rows-[auto_1fr] overflow-hidden rounded-[inherit] border border-[color-mix(in_srgb,var(--dt-composer-ring)_calc(18%*var(--composer-ring-strength)),var(--dt-input))]',
|
||||
// grid-cols-[minmax(0,1fr)]: the implicit `auto` column sized
|
||||
// itself to its items' min-content, so a status row whose
|
||||
// content out-measured a narrow pane silently widened the
|
||||
// track past the surface — and every `w-full` child (the fade,
|
||||
// the input/controls row) laid out against that phantom width
|
||||
// and got clipped by overflow-hidden, send button first.
|
||||
'group/composer-surface relative z-4 isolate grid grid-cols-[minmax(0,1fr)] grid-rows-[auto_1fr] overflow-hidden rounded-[inherit] border border-[color-mix(in_srgb,var(--dt-composer-ring)_calc(18%*var(--composer-ring-strength)),var(--dt-input))]',
|
||||
COMPOSER_DROP_FADE_CLASS,
|
||||
dragActive && COMPOSER_DROP_ACTIVE_CLASS
|
||||
)}
|
||||
@@ -1322,7 +1330,7 @@ export function ChatBar({
|
||||
<ContribSlot area={COMPOSER_AREAS.leading} />
|
||||
</div>
|
||||
<div className="min-w-0 [grid-area:input]">{input}</div>
|
||||
<div className="flex items-center justify-end gap-(--composer-control-gap) [grid-area:controls]">
|
||||
<div className="flex min-w-0 items-center justify-end gap-(--composer-control-gap) [grid-area:controls]">
|
||||
<ContribSlot area={COMPOSER_AREAS.actions} />
|
||||
{controls}
|
||||
</div>
|
||||
|
||||
@@ -18,8 +18,11 @@ import { onComposerModelMenuRequest } from './focus'
|
||||
import { useComposerScope } from './scope'
|
||||
import type { ChatBarState } from './types'
|
||||
|
||||
// `shrink` (not `shrink-0`) with a truncating label: the pill is the one
|
||||
// control in the row that can give width back continuously, so it absorbs the
|
||||
// squeeze between collapse stages instead of pushing Send past the edge.
|
||||
const PILL = cn(
|
||||
'h-(--composer-control-size) max-w-40 shrink-0 gap-1 rounded-md px-2 text-xs font-normal',
|
||||
'h-(--composer-control-size) min-w-0 max-w-40 shrink gap-1 rounded-md px-2 text-xs font-normal',
|
||||
'text-(--ui-text-tertiary) hover:bg-(--chrome-action-hover) hover:text-foreground'
|
||||
)
|
||||
|
||||
|
||||
@@ -40,6 +40,11 @@ vi.mock('@/store/boot', () => ({
|
||||
})
|
||||
}))
|
||||
|
||||
vi.mock('@/store/windows', () => ({
|
||||
isAuxiliaryWindow: vi.fn(() => false),
|
||||
isPeerInstanceWindow: vi.fn(() => false)
|
||||
}))
|
||||
|
||||
vi.mock('@/i18n', () => ({
|
||||
useI18n: () => ({
|
||||
t: {
|
||||
@@ -65,6 +70,7 @@ vi.mock('@/i18n', () => ({
|
||||
|
||||
const connectionStore = await import('@/store/connections')
|
||||
const bootStore = await import('@/store/boot')
|
||||
const windowStore = await import('@/store/windows')
|
||||
const $activeConnectionId = connectionStore.$activeConnectionId as ReturnType<typeof atom<null | string>>
|
||||
const $connectionsRegistry = connectionStore.$connectionsRegistry
|
||||
const $desktopBoot = bootStore.$desktopBoot
|
||||
@@ -72,6 +78,8 @@ const $pendingConnectionId = connectionStore.$pendingConnectionId
|
||||
const initializeConnectionsRegistry = vi.mocked(connectionStore.initializeConnectionsRegistry)
|
||||
const refreshConnectionsRegistry = vi.mocked(connectionStore.refreshConnectionsRegistry)
|
||||
const selectConnection = vi.mocked(connectionStore.selectConnection)
|
||||
const isAuxiliaryWindow = vi.mocked(windowStore.isAuxiliaryWindow)
|
||||
const isPeerInstanceWindow = vi.mocked(windowStore.isPeerInstanceWindow)
|
||||
const onConnect = vi.fn()
|
||||
|
||||
const connection = (id: string, label: string, kind: 'local' | 'remote' = 'remote') => ({
|
||||
@@ -106,6 +114,8 @@ afterEach(() => {
|
||||
})
|
||||
$pendingConnectionId.set(null)
|
||||
$findInPage.set({ active: false, query: '', matchOrdinal: 0, matchCount: 0 })
|
||||
isAuxiliaryWindow.mockReturnValue(false)
|
||||
isPeerInstanceWindow.mockReturnValue(false)
|
||||
})
|
||||
|
||||
describe('ConnectionSwitcher', () => {
|
||||
@@ -127,6 +137,38 @@ describe('ConnectionSwitcher', () => {
|
||||
await waitFor(() => expect(initializeConnectionsRegistry).toHaveBeenCalledTimes(1))
|
||||
})
|
||||
|
||||
it('keeps a full peer on the shared backend instead of replaying app-launch source restoration', async () => {
|
||||
isPeerInstanceWindow.mockReturnValue(true)
|
||||
$desktopBoot.set({
|
||||
...$desktopBoot.get(),
|
||||
phase: 'renderer.ready',
|
||||
progress: 100,
|
||||
running: false,
|
||||
visible: false
|
||||
})
|
||||
|
||||
render(<ConnectionSwitcher onConnect={onConnect} />)
|
||||
|
||||
await waitFor(() => expect(refreshConnectionsRegistry).toHaveBeenCalledTimes(1))
|
||||
expect(initializeConnectionsRegistry).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('keeps a secondary session window from replaying app-launch source restoration', async () => {
|
||||
isAuxiliaryWindow.mockReturnValue(true)
|
||||
$desktopBoot.set({
|
||||
...$desktopBoot.get(),
|
||||
phase: 'renderer.ready',
|
||||
progress: 100,
|
||||
running: false,
|
||||
visible: false
|
||||
})
|
||||
|
||||
render(<ConnectionSwitcher onConnect={onConnect} />)
|
||||
|
||||
await waitFor(() => expect(refreshConnectionsRegistry).toHaveBeenCalledTimes(1))
|
||||
expect(initializeConnectionsRegistry).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('adds no source chrome for a local-only setup', () => {
|
||||
$connectionsRegistry.set(registry([connection('local', 'This device', 'local')]))
|
||||
render(<ConnectionSwitcher onConnect={onConnect} />)
|
||||
|
||||
@@ -36,6 +36,7 @@ import {
|
||||
} from '@/store/connections'
|
||||
import { closeFindBar } from '@/store/find-in-page'
|
||||
import { notifyError } from '@/store/notifications'
|
||||
import { isAuxiliaryWindow, isPeerInstanceWindow } from '@/store/windows'
|
||||
|
||||
export function ConnectionSwitcher({ compact = false, onConnect }: { compact?: boolean; onConnect: () => void }) {
|
||||
const { t } = useI18n()
|
||||
@@ -64,8 +65,10 @@ export function ConnectionSwitcher({ compact = false, onConnect }: { compact?: b
|
||||
// The primary boot owns its initial config/session fetches. Restoring a
|
||||
// different source before those settle lets a late primary response repaint
|
||||
// the sidebar under the new source label. Switch only after boot completes,
|
||||
// then the normal source reset/refetch remains the final writer.
|
||||
if (!boot.running) {
|
||||
// then the normal source reset/refetch remains the final writer. Peer and
|
||||
// auxiliary windows already boot into their intended runtime; replaying the
|
||||
// primary window's app-launch preference would move them away from it.
|
||||
if (!boot.running && !isAuxiliaryWindow() && !isPeerInstanceWindow()) {
|
||||
void initializeConnectionsRegistry().catch(() => undefined)
|
||||
}
|
||||
}, [boot.running])
|
||||
|
||||
@@ -150,6 +150,10 @@ export function NarrowOverlays() {
|
||||
? 'left-0 border-r border-(--ui-stroke-secondary)'
|
||||
: 'right-0 border-l border-(--ui-stroke-secondary)'
|
||||
)}
|
||||
// Floats OVER the layout, so under glass its surface must mask the
|
||||
// panes beneath it — a see-through overlay reads as text bleeding
|
||||
// through text. Contract: `[data-glass-opaque]` in styles.css.
|
||||
data-glass-opaque=""
|
||||
onMouseLeave={() => setReveal(current => (current?.pinned ? current : null))}
|
||||
// Match the pane's docked width (sessions ~237px, files its rail
|
||||
// width) instead of a fat fixed 20rem — capped for tiny screens.
|
||||
|
||||
@@ -2969,6 +2969,53 @@ function botHandle(name, bot) {
|
||||
return (name || '').trim().toLowerCase() === 'default' ? 'hermes' : name
|
||||
}
|
||||
|
||||
/** Taggable @-forms derived from a bot's friendly names — the core profile
|
||||
* display name (`hermes profile rename`) and the Bot Mode title. Free text
|
||||
* reduces to the mention charset two ways: slugified ("Research Buddy" →
|
||||
* research-buddy, the form autocomplete inserts) and collapsed
|
||||
* (researchbuddy). Reserved tokens are dropped so a bot renamed "Hermes"
|
||||
* can never hijack the primary profile's @hermes alias. */
|
||||
function mentionNameForms(value) {
|
||||
const name = String(value || '').trim().toLowerCase()
|
||||
|
||||
if (!name) {
|
||||
return []
|
||||
}
|
||||
|
||||
const slug = name.replace(/[^a-z0-9_-]+/g, '-').replace(/^-+|-+$/g, '')
|
||||
const collapsed = name.replace(/[^a-z0-9_-]+/g, '')
|
||||
|
||||
return [...new Set([slug, collapsed])].filter(
|
||||
form => /^[a-z0-9][a-z0-9_-]*$/.test(form) && !['all', 'everyone', 'user', 'default', 'hermes'].includes(form)
|
||||
)
|
||||
}
|
||||
|
||||
/** Every friendly (renameable) name a roster row carries: the Bot Mode title
|
||||
* (server-synced via ui_meta, locally stored, or persisted on a durable
|
||||
* group descriptor) and the core profile display_name — in displayName's
|
||||
* precedence order. Remote rows never borrow local meta (two `default`s
|
||||
* must not share a title). */
|
||||
function botFriendlyNames(bot) {
|
||||
const localTitle = !bot?.remoteSource && typeof $botMeta !== 'undefined' ? $botMeta.get()?.[bot?.name]?.title : null
|
||||
|
||||
return [bot?.ui_meta?.['hermes-bots']?.title, localTitle, bot?.title, bot?.display_name]
|
||||
}
|
||||
|
||||
/** The tag autocomplete inserts for a bot: the renamed (friendly) slug when
|
||||
* the user gave the bot a real name, otherwise the profile @handle. The
|
||||
* resolvers accept both, so older muscle memory keeps working. */
|
||||
function botMentionTag(bot) {
|
||||
for (const friendly of botFriendlyNames(bot)) {
|
||||
const forms = mentionNameForms(friendly)
|
||||
|
||||
if (forms.length) {
|
||||
return forms[0]
|
||||
}
|
||||
}
|
||||
|
||||
return botHandle(bot?.name, bot)
|
||||
}
|
||||
|
||||
function isActiveRosterBot(bot, active) {
|
||||
const activeName = String(active?.name || 'default').trim() || 'default'
|
||||
const activeId = String(active?.connectionId || '').trim()
|
||||
@@ -3007,6 +3054,15 @@ function resolveRosterMentions(text, roster, active = {}) {
|
||||
forms.add(String(bot.handle).toLowerCase())
|
||||
}
|
||||
|
||||
// Renamed bots are taggable by their friendly names too — the core
|
||||
// profile display_name and the Bot Mode title (issue: renaming a bot
|
||||
// didn't change what you @-tag it with).
|
||||
for (const friendly of botFriendlyNames(bot)) {
|
||||
for (const form of mentionNameForms(friendly)) {
|
||||
forms.add(form)
|
||||
}
|
||||
}
|
||||
|
||||
for (const form of forms) {
|
||||
if (!form) {
|
||||
continue
|
||||
@@ -3697,15 +3753,24 @@ function groupChatMemberBots(group, roster, metaByName) {
|
||||
* source's row may become remote after a connection switch, so retaining it
|
||||
* here is what keeps the same room intact across machines. */
|
||||
function durableGroupChatMembers(bots) {
|
||||
return (bots || []).map(bot => ({
|
||||
name: bot.name,
|
||||
handle: bot.handle || bot.name,
|
||||
connectionId: bot.connectionId,
|
||||
connectionKind: bot.connectionKind,
|
||||
connectionLabel: bot.connectionLabel,
|
||||
remoteSource: true,
|
||||
sourceScoped: true
|
||||
}))
|
||||
return (bots || []).map(bot => {
|
||||
// Keep the friendly identity on the stored descriptor: after a
|
||||
// connection switch the live roster row may be gone, and renamed-tag
|
||||
// mentions must still resolve against the persisted member.
|
||||
const title = String(botRosterMeta(bot, $botMeta.get())?.title || bot.ui_meta?.['hermes-bots']?.title || bot.title || '').trim()
|
||||
|
||||
return {
|
||||
name: bot.name,
|
||||
handle: bot.handle || bot.name,
|
||||
...(title ? { title } : {}),
|
||||
...(bot.display_name ? { display_name: bot.display_name } : {}),
|
||||
connectionId: bot.connectionId,
|
||||
connectionKind: bot.connectionKind,
|
||||
connectionLabel: bot.connectionLabel,
|
||||
remoteSource: true,
|
||||
sourceScoped: true
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
/** Existing group names, alphabetical — feeds the Manage-groups dialog. */
|
||||
@@ -3774,6 +3839,15 @@ function parseGroupChatMentions(text, members) {
|
||||
: [])
|
||||
])
|
||||
|
||||
// Renamed members answer to their friendly names too (profile
|
||||
// display_name and Bot Mode title), in slugged and collapsed forms —
|
||||
// the same tags the roster autocomplete inserts.
|
||||
for (const friendly of botFriendlyNames(member)) {
|
||||
for (const form of mentionNameForms(friendly)) {
|
||||
forms.add(form)
|
||||
}
|
||||
}
|
||||
|
||||
for (const form of forms) {
|
||||
if (form) {
|
||||
handles.set(form, groupMemberKey(member))
|
||||
@@ -8658,14 +8732,26 @@ function GroupMentionInput({ members, onChange, value, ...inputProps }) {
|
||||
|
||||
for (const member of members) {
|
||||
const handle = String(member.handle || botHandle(member.name, member) || '').trim()
|
||||
const display = displayName(member, botRosterMeta(member, allMeta))
|
||||
// Renamed members complete on their friendly tag; parser resolves both.
|
||||
const tag = String(botMentionTag(member) || handle).trim()
|
||||
|
||||
if (!handle || (token.query && !handle.toLowerCase().startsWith(token.query))) {
|
||||
if (!tag) {
|
||||
continue
|
||||
}
|
||||
|
||||
if (
|
||||
token.query &&
|
||||
!tag.toLowerCase().startsWith(token.query) &&
|
||||
!(handle && handle.toLowerCase().startsWith(token.query)) &&
|
||||
!display.toLowerCase().startsWith(token.query)
|
||||
) {
|
||||
continue
|
||||
}
|
||||
|
||||
options.push({
|
||||
handle,
|
||||
meta: displayName(member, botRosterMeta(member, allMeta))
|
||||
handle: tag,
|
||||
meta: display
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -10109,17 +10195,25 @@ export default {
|
||||
}
|
||||
|
||||
const handle = botHandle(profile.name, profile)
|
||||
const display = displayName(profile, $botMeta.get()[profile.name])
|
||||
// Renamed bots complete on their friendly name — the tag is the
|
||||
// renamed slug when one exists, the profile handle otherwise.
|
||||
const tag = botMentionTag(profile)
|
||||
|
||||
if (q && !handle.toLowerCase().startsWith(q)) {
|
||||
if (
|
||||
q &&
|
||||
!tag.toLowerCase().startsWith(q) &&
|
||||
!handle.toLowerCase().startsWith(q) &&
|
||||
!display.toLowerCase().startsWith(q)
|
||||
) {
|
||||
continue
|
||||
}
|
||||
|
||||
const display = displayName(profile, $botMeta.get()[profile.name])
|
||||
const source = profile.connectionLabel ? ` · ${profile.connectionLabel}` : ''
|
||||
|
||||
items.push({
|
||||
insert: `@${handle}`,
|
||||
display: `@${handle}`,
|
||||
insert: `@${tag}`,
|
||||
display: `@${tag}`,
|
||||
meta: `Bot · ${display}${source}`
|
||||
})
|
||||
}
|
||||
@@ -10397,30 +10491,14 @@ export default {
|
||||
let mentionedBots = roster ? resolveRosterMentions(text, roster, live) : []
|
||||
|
||||
if (!roster) {
|
||||
let names = []
|
||||
try {
|
||||
const res = await host.request('profiles.list', { include_sessions: false })
|
||||
names = (res?.profiles ?? []).map(p => p.name)
|
||||
// Same resolver as the cached path — renamed bots (display_name
|
||||
// / ui_meta title) stay taggable when the roster cache is cold.
|
||||
mentionedBots = resolveRosterMentions(text, res?.profiles ?? [], live).map(bot => ({ ...bot, remoteSource: false }))
|
||||
} catch {
|
||||
return draft
|
||||
}
|
||||
|
||||
const prose = text.replace(/```[\s\S]*?```/g, ' ').replace(/`[^`\n]*`/g, ' ')
|
||||
const mentioned = []
|
||||
|
||||
for (const match of prose.matchAll(/(^|\s)@([a-z0-9][a-z0-9_-]*)/gi)) {
|
||||
let name = match[2].toLowerCase()
|
||||
|
||||
if (name === 'hermes' && !names.includes('hermes') && names.includes('default')) {
|
||||
name = 'default'
|
||||
}
|
||||
|
||||
if (names.includes(name) && name !== live.name && !mentioned.includes(name)) {
|
||||
mentioned.push(name)
|
||||
}
|
||||
}
|
||||
|
||||
mentionedBots = mentioned.map(name => ({ name }))
|
||||
}
|
||||
|
||||
if (!mentionedBots.length) {
|
||||
|
||||
@@ -0,0 +1,188 @@
|
||||
import assert from 'node:assert/strict'
|
||||
import { readFileSync } from 'node:fs'
|
||||
import test from 'node:test'
|
||||
import vm from 'node:vm'
|
||||
|
||||
// Renamed bots stay taggable (Discord report, Aug 2026): renaming a bot —
|
||||
// Bot Mode title or `hermes profile rename` display_name — must update what
|
||||
// the user can @-tag it with, in the mention middleware, group rooms, and
|
||||
// the composer autocomplete. Old profile handles keep resolving.
|
||||
|
||||
const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8')
|
||||
|
||||
function runtime({ meta } = {}) {
|
||||
const context = {
|
||||
console,
|
||||
setTimeout,
|
||||
clearTimeout,
|
||||
Date,
|
||||
URL,
|
||||
atom: initial => {
|
||||
let value = initial
|
||||
return { get: () => value, set: next => (value = next), listen: () => () => undefined }
|
||||
},
|
||||
host: {
|
||||
request: async () => ({}),
|
||||
requestProfile: async () => ({}),
|
||||
state: {
|
||||
profile: { get: () => 'default', listen: () => undefined },
|
||||
connectionId: { get: () => 'local', listen: () => undefined },
|
||||
gateway: { listen: () => undefined }
|
||||
}
|
||||
},
|
||||
document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } }
|
||||
}
|
||||
const code = source
|
||||
.replace(/^import\s+\*\s+as\s+sdk\s+from '@hermes\/plugin-sdk'\r?\n/m, '')
|
||||
.replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '')
|
||||
.replace(/^const \{ McpTab, ToolsetConfigPanel \} = sdk\r?\n/m, '')
|
||||
.replace(/^import .* from 'react'\r?\n/m, '')
|
||||
.replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '')
|
||||
.replace('export default {', 'globalThis.plugin = {')
|
||||
.concat(
|
||||
'\nglobalThis.__x = { mentionNameForms, botFriendlyNames, botMentionTag, resolveRosterMentions, parseGroupChatMentions, groupMemberKey, durableGroupChatMembers, $botMeta };\n'
|
||||
)
|
||||
vm.runInNewContext(code, context, { filename: 'plugin.js' })
|
||||
if (meta) {
|
||||
context.__x.$botMeta.set(meta)
|
||||
}
|
||||
return context.__x
|
||||
}
|
||||
|
||||
test('mentionNameForms: slugged + collapsed, reserved tokens dropped', () => {
|
||||
const { mentionNameForms } = runtime()
|
||||
|
||||
assert.deepEqual(JSON.parse(JSON.stringify(mentionNameForms('Research Buddy'))), ['research-buddy', 'researchbuddy'])
|
||||
assert.deepEqual(JSON.parse(JSON.stringify(mentionNameForms('Ops'))), ['ops'])
|
||||
// A bot renamed "Hermes" or "everyone" cannot hijack reserved tags.
|
||||
assert.equal(mentionNameForms('Hermes').length, 0)
|
||||
assert.equal(mentionNameForms('@everyone').length, 0)
|
||||
assert.equal(mentionNameForms('').length, 0)
|
||||
})
|
||||
|
||||
test('botMentionTag: renamed slug wins, profile handle is the fallback', () => {
|
||||
const { botMentionTag } = runtime({ meta: { writer: { title: 'Research Buddy' } } })
|
||||
|
||||
assert.equal(botMentionTag({ name: 'writer' }), 'research-buddy')
|
||||
assert.equal(botMentionTag({ name: 'ops' }), 'ops')
|
||||
assert.equal(botMentionTag({ name: 'default' }), 'hermes')
|
||||
// display_name (hermes profile rename) drives the tag too.
|
||||
assert.equal(botMentionTag({ name: 'scout', display_name: 'Deal Finder' }), 'deal-finder')
|
||||
})
|
||||
|
||||
test('resolveRosterMentions: renamed bots resolve by friendly tag AND old handle', () => {
|
||||
const { resolveRosterMentions } = runtime({ meta: { writer: { title: 'Research Buddy' } } })
|
||||
const roster = [{ name: 'default' }, { name: 'writer' }, { name: 'scout', display_name: 'Deal Finder' }]
|
||||
const live = { name: 'default', connectionId: 'local' }
|
||||
|
||||
const byTitle = resolveRosterMentions('hey @research-buddy check this', roster, live)
|
||||
assert.deepEqual(JSON.parse(JSON.stringify(byTitle.map(b => b.name))), ['writer'])
|
||||
|
||||
const byDisplayName = resolveRosterMentions('ping @dealfinder please', roster, live)
|
||||
assert.deepEqual(JSON.parse(JSON.stringify(byDisplayName.map(b => b.name))), ['scout'])
|
||||
|
||||
// The profile name keeps working after a rename.
|
||||
const byName = resolveRosterMentions('hey @writer', roster, live)
|
||||
assert.deepEqual(JSON.parse(JSON.stringify(byName.map(b => b.name))), ['writer'])
|
||||
})
|
||||
|
||||
test('parseGroupChatMentions: display_name and ui_meta title forms address a member', () => {
|
||||
const { parseGroupChatMentions, groupMemberKey } = runtime()
|
||||
const members = [
|
||||
{ name: 'research', title: '' },
|
||||
{ name: 'builder', title: '', display_name: 'Site Smith' },
|
||||
{ name: 'ops', title: '', ui_meta: { 'hermes-bots': { title: 'Night Watch' } } }
|
||||
]
|
||||
|
||||
const bySlug = parseGroupChatMentions('@site-smith and @night-watch please', members)
|
||||
assert.equal(bySlug.mentioned.has(groupMemberKey(members[1])), true)
|
||||
assert.equal(bySlug.mentioned.has(groupMemberKey(members[2])), true)
|
||||
assert.equal(bySlug.mentioned.has(groupMemberKey(members[0])), false)
|
||||
|
||||
const collapsed = parseGroupChatMentions('@sitesmith take a look', members)
|
||||
assert.equal(collapsed.mentioned.has(groupMemberKey(members[1])), true)
|
||||
})
|
||||
|
||||
test('durableGroupChatMembers persists the friendly identity for renamed-tag mentions', () => {
|
||||
const { durableGroupChatMembers, $botMeta } = runtime()
|
||||
$botMeta.set({ writer: { title: 'Research Buddy' } })
|
||||
|
||||
const [stored] = durableGroupChatMembers([{ name: 'writer', connectionId: 'local' }])
|
||||
assert.equal(stored.title, 'Research Buddy')
|
||||
|
||||
const [remote] = durableGroupChatMembers([
|
||||
{ name: 'scout', display_name: 'Deal Finder', connectionId: 'mac-mini', remoteSource: true }
|
||||
])
|
||||
assert.equal(remote.display_name, 'Deal Finder')
|
||||
})
|
||||
|
||||
test('composer autocomplete offers the renamed tag and matches on the display name', () => {
|
||||
const registrations = []
|
||||
const atom = value => {
|
||||
let current = value
|
||||
return { get: () => current, set: next => (current = next), listen: () => () => undefined }
|
||||
}
|
||||
const jsx = (type, props = {}) => ({ type, props })
|
||||
const roster = { profiles: [{ name: 'default' }, { name: 'writer' }, { name: 'ops' }] }
|
||||
const context = {
|
||||
atom,
|
||||
jsx,
|
||||
jsxs: jsx,
|
||||
useQuery: () => ({}),
|
||||
useValue: v => (v?.get ? v.get() : v),
|
||||
useState: v => [v, () => undefined],
|
||||
setTimeout,
|
||||
clearTimeout,
|
||||
performance,
|
||||
window: {
|
||||
setTimeout,
|
||||
clearTimeout,
|
||||
performance,
|
||||
requestAnimationFrame: () => 0,
|
||||
cancelAnimationFrame: () => undefined,
|
||||
addEventListener: () => undefined,
|
||||
removeEventListener: () => undefined
|
||||
},
|
||||
document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } },
|
||||
queryClient: { getQueryData: () => roster, invalidateQueries: () => undefined, setQueryData: () => undefined },
|
||||
host: {
|
||||
state: { profile: { get: () => 'default', listen: () => () => undefined }, gateway: { get: () => null, listen: () => () => undefined } },
|
||||
request: async () => ({}),
|
||||
onEvent: () => () => undefined
|
||||
},
|
||||
COMPOSER_AREAS: { middleware: 'composer.middleware', atCompletions: 'composer.atCompletions' },
|
||||
sdk: new Proxy({}, { get: () => undefined })
|
||||
}
|
||||
const code = source
|
||||
.replace(/^import \* as sdk from '@hermes\/plugin-sdk'\r?\n/m, '')
|
||||
.replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '')
|
||||
.replace(/^import .* from 'react'\r?\n/m, '')
|
||||
.replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '')
|
||||
.replace('export default {', 'globalThis.plugin = {')
|
||||
.concat('\nglobalThis.__botMeta = $botMeta;')
|
||||
vm.runInNewContext(code, context)
|
||||
context.__botMeta.set({ writer: { title: 'Research Buddy' } })
|
||||
|
||||
try {
|
||||
context.plugin.register({
|
||||
register: c => registrations.push(c),
|
||||
storage: { get: async () => undefined, set: async () => undefined }
|
||||
})
|
||||
} catch {
|
||||
/* registration walks UI surfaces the stub doesn't fully model */
|
||||
}
|
||||
|
||||
const provide = registrations.find(c => c.area === 'composer.atCompletions').data.provide
|
||||
|
||||
// The renamed bot completes under its friendly prefix and inserts the tag.
|
||||
const renamed = provide('rese')
|
||||
assert.deepEqual(JSON.parse(JSON.stringify(renamed.map(i => i.insert))), ['@research-buddy'])
|
||||
|
||||
// Old profile-handle muscle memory still finds it (insert is the new tag).
|
||||
const byHandle = provide('wri')
|
||||
assert.deepEqual(JSON.parse(JSON.stringify(byHandle.map(i => i.insert))), ['@research-buddy'])
|
||||
|
||||
// Un-renamed bots keep their plain handle.
|
||||
const plain = provide('op')
|
||||
assert.deepEqual(JSON.parse(JSON.stringify(plain.map(i => i.insert))), ['@ops'])
|
||||
})
|
||||
@@ -1,6 +1,12 @@
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { canOpenNewWindow, canOpenSessionWindow, openNewWindow, openSessionInNewWindow } from './windows'
|
||||
import {
|
||||
canOpenNewWindow,
|
||||
canOpenSessionWindow,
|
||||
isPeerInstanceWindow,
|
||||
openNewWindow,
|
||||
openSessionInNewWindow
|
||||
} from './windows'
|
||||
|
||||
const desktopWindow = window as unknown as { hermesDesktop?: Window['hermesDesktop'] }
|
||||
const initialHermesDesktop = desktopWindow.hermesDesktop
|
||||
@@ -50,6 +56,15 @@ describe('canOpenSessionWindow', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('isPeerInstanceWindow', () => {
|
||||
it('recognizes only the full peer marker', () => {
|
||||
expect(isPeerInstanceWindow('?peer=1')).toBe(true)
|
||||
expect(isPeerInstanceWindow('?peer=0')).toBe(false)
|
||||
expect(isPeerInstanceWindow('?win=secondary')).toBe(false)
|
||||
expect(isPeerInstanceWindow('')).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
describe('openSessionInNewWindow', () => {
|
||||
it('no-ops without a session id', async () => {
|
||||
const open = vi.fn().mockResolvedValue({ ok: true })
|
||||
|
||||
@@ -83,6 +83,18 @@ export function isWatchWindow(): boolean {
|
||||
// keystroke into N prompts, and a HUD is the last place to paint onboarding.
|
||||
export const isAuxiliaryWindow = (): boolean => isSecondaryWindow() || isHudWindow()
|
||||
|
||||
// A full peer window renders the ordinary app shell against the backend that
|
||||
// Electron already has running. It is not an auxiliary/specialized renderer,
|
||||
// but it must not replay the primary window's app-launch source restoration
|
||||
// after boot and silently re-home itself to another registered gateway.
|
||||
export function isPeerInstanceWindow(search = typeof window === 'undefined' ? '' : window.location.search): boolean {
|
||||
try {
|
||||
return new URLSearchParams(search).get('peer') === '1'
|
||||
} catch {
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// The profile a helper window (the HUD) was asked to boot against, carried in
|
||||
// the query string by the main process (see hudUrl). The HUD is a full app
|
||||
// renderer that otherwise adopts the PRIMARY backend's profile — wrong the
|
||||
|
||||
@@ -833,11 +833,6 @@ memory:
|
||||
# every N user turns. Set to 0 to disable. Only active when memory is enabled.
|
||||
nudge_interval: 10 # Nudge every 10 user turns (0 = disabled)
|
||||
|
||||
# Memory flush: give the agent one turn to save memories before context is
|
||||
# lost (compression, /new, /reset, exit). Set to 0 to disable.
|
||||
# For exit/reset, only fires if the session had at least this many user turns.
|
||||
flush_min_turns: 6 # Min user turns to trigger flush on exit/reset (0 = disabled)
|
||||
|
||||
# =============================================================================
|
||||
# Session Reset Policy (Messaging Platforms)
|
||||
# =============================================================================
|
||||
|
||||
@@ -1866,9 +1866,14 @@ def _setup_worktree(repo_root: str = None, sync_base: bool = True,
|
||||
"-c", "checkout.thresholdForParallelism=100",
|
||||
]
|
||||
try:
|
||||
# 120s, not 30: on a multi-agent box the ~10k-file materialization
|
||||
# contends with sibling sessions' checkouts/fetches/Electron dev
|
||||
# builds for the same disk — measured 113s wall at near-zero CPU
|
||||
# under load vs 1.2s idle (Aug 2026). A too-tight timeout kills a
|
||||
# legitimately slow create and wastes the work already done.
|
||||
result = subprocess.run(
|
||||
["git", *_wt_add_cfg, "worktree", "add", str(wt_path), "-b", branch_name, base_ref],
|
||||
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=30, cwd=repo_root,
|
||||
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=120, cwd=repo_root,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
# If branching from the resolved remote ref failed for any reason
|
||||
@@ -1883,7 +1888,7 @@ def _setup_worktree(repo_root: str = None, sync_base: bool = True,
|
||||
base_ref, base_label = "HEAD", "HEAD (fallback — remote base failed)"
|
||||
result = subprocess.run(
|
||||
["git", "worktree", "add", str(wt_path), "-b", branch_name, base_ref],
|
||||
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=30, cwd=repo_root,
|
||||
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=120, cwd=repo_root,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
_cleanup_failed_worktree_add(repo_root, wt_path, branch_name)
|
||||
@@ -2292,6 +2297,69 @@ def _worktree_commits_all_merged_upstream(
|
||||
return False
|
||||
|
||||
|
||||
def _worktree_branch_pr_merged(
|
||||
worktree_path: str,
|
||||
timeout: int = 15,
|
||||
cache: Optional[Dict[str, bool]] = None,
|
||||
) -> bool:
|
||||
"""Return whether the worktree branch's PR is MERGED on GitHub.
|
||||
|
||||
Escape hatch for the case ``git cherry`` cannot catch: a rebase-merge that
|
||||
altered the diff (conflict resolution against a moved base, follow-up
|
||||
commits added during salvage/CI-fix) changes the patch-id, so the local
|
||||
commits are no longer patch-equivalent to anything upstream even though
|
||||
the PR merged. Those trees survive the cherry check forever (Aug 2026:
|
||||
12 of 22 "unpushed" trees on a loaded box had MERGED PRs).
|
||||
|
||||
GitHub's PR state is the authoritative merge signal, so a clean tree
|
||||
whose branch has a MERGED PR is reaped. Verdicts are memoized keyed on
|
||||
``(branch, head_sha)`` — MERGED is monotonic, so a True verdict is cached
|
||||
permanently; False is never cached (the PR may merge later without new
|
||||
local commits, which would leave the key unchanged).
|
||||
|
||||
Fails SAFE toward False (preserve): no gh binary, offline, rate-limited,
|
||||
detached HEAD, or any parse failure keeps the tree.
|
||||
"""
|
||||
import subprocess
|
||||
|
||||
try:
|
||||
head = subprocess.run(
|
||||
["git", "rev-parse", "--abbrev-ref", "HEAD"],
|
||||
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, cwd=worktree_path,
|
||||
)
|
||||
if head.returncode != 0:
|
||||
return False
|
||||
branch = head.stdout.strip()
|
||||
if not branch or branch == "HEAD": # detached — no PR to look up
|
||||
return False
|
||||
|
||||
cache_key = None
|
||||
if cache is not None:
|
||||
sha = subprocess.run(
|
||||
["git", "rev-parse", "HEAD"],
|
||||
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, cwd=worktree_path,
|
||||
)
|
||||
if sha.returncode == 0 and sha.stdout.strip():
|
||||
cache_key = f"pr-merged:{branch}:{sha.stdout.strip()}"
|
||||
if cache.get(cache_key) is True:
|
||||
return True
|
||||
|
||||
result = subprocess.run(
|
||||
["gh", "pr", "list", "--head", branch, "--state", "merged",
|
||||
"--json", "number", "--limit", "1"],
|
||||
capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, cwd=worktree_path,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
return False
|
||||
prs = json.loads(result.stdout or "[]")
|
||||
merged = isinstance(prs, list) and len(prs) > 0
|
||||
if merged and cache is not None and cache_key is not None:
|
||||
cache[cache_key] = True
|
||||
return merged
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _worktree_lock_is_live(repo_root: str, worktree_path: str, timeout: int = 10):
|
||||
"""Classify a worktree's git lock as live, dead, or absent.
|
||||
|
||||
@@ -2652,6 +2720,14 @@ def _prune_stale_worktrees(repo_root: str, max_age_hours: int = 24) -> None:
|
||||
merged = _worktree_commits_all_merged_upstream(
|
||||
str(entry), timeout=30, cache=snapshot
|
||||
)
|
||||
if not merged:
|
||||
# Rebase-merge escape hatch: conflict resolution or follow-up
|
||||
# commits change the patch-id, so cherry misses them — but
|
||||
# GitHub knows the PR merged. Authoritative and cheap (~0.3s,
|
||||
# memoized on (branch, head_sha) so it's paid once per tree).
|
||||
merged = _worktree_branch_pr_merged(
|
||||
str(entry), timeout=15, cache=snapshot
|
||||
)
|
||||
with cache_lock:
|
||||
merge_cache.update(snapshot)
|
||||
if not merged:
|
||||
@@ -2741,12 +2817,31 @@ def _prune_stale_worktrees(repo_root: str, max_age_hours: int = 24) -> None:
|
||||
if preserved_stale:
|
||||
logger.warning(
|
||||
"Preserving %d worktree(s) older than 7 days with unmerged work "
|
||||
"(push or remove them to reclaim disk): %s",
|
||||
"(run `hermes worktree prune` to review and reclaim): %s",
|
||||
len(preserved_stale), ", ".join(sorted(preserved_stale)),
|
||||
)
|
||||
|
||||
_prune_orphaned_branches(repo_root)
|
||||
|
||||
# Escalation notice: the startup pass is deliberately conservative, so
|
||||
# installs accumulate preserved trees it can never reclaim. Once the
|
||||
# footprint is clearly a problem (many trees or multi-GB), say so once
|
||||
# per launch and name the attended reclaim command — silence here is how
|
||||
# boxes reach 15GB+ of .worktrees/ without anyone noticing.
|
||||
try:
|
||||
from hermes_cli.worktree_gc import worktrees_summary
|
||||
|
||||
count, size_mb = worktrees_summary(repo_root)
|
||||
if count >= 10 or (size_mb or 0) >= 5120:
|
||||
size_txt = f"{size_mb / 1024:.1f}GB" if size_mb else "unknown size"
|
||||
logger.warning(
|
||||
".worktrees/ holds %d tree(s) (%s) — run `hermes worktree list` "
|
||||
"to audit and `hermes worktree prune` to reclaim safely.",
|
||||
count, size_txt,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _prune_orphaned_branches(repo_root: str) -> None:
|
||||
"""Delete local ``hermes/hermes-*`` and ``pr-*`` branches with no worktree.
|
||||
@@ -4094,12 +4189,36 @@ _TERMINAL_INPUT_MODE_RESET_SEQ = (
|
||||
"\x1b[0m" # reset text attributes
|
||||
"\x1b[?25h" # ensure cursor visible
|
||||
)
|
||||
_EXTENDED_ENTER_KEYS_SEQ = "\x1b[>1u\x1b[>4;2m"
|
||||
_KITTY_KEYBOARD_PUSH_SEQ = "\x1b[>1u"
|
||||
_MODIFY_OTHER_KEYS_SEQ = "\x1b[>4;2m"
|
||||
_EXTENDED_ENTER_KEYS_SEQ = _KITTY_KEYBOARD_PUSH_SEQ + _MODIFY_OTHER_KEYS_SEQ
|
||||
|
||||
|
||||
_BACKSLASH_LINE_CONTINUATION_RE = re.compile(r"\\[ \t]*$")
|
||||
|
||||
|
||||
def _is_ghostty_terminal(env: Optional[Mapping[str, str]] = None) -> bool:
|
||||
"""Whether the terminal is Ghostty (either detection path).
|
||||
|
||||
Ghostty must be pushed ONLY modifyOtherKeys, not the Kitty keyboard
|
||||
protocol: its Kitty disambiguate-mode implementation strips the Alt
|
||||
modifier from the Backspace key, so Option+Backspace arrives as bare
|
||||
\\x7f instead of the CSI-u form ``\\x1b[127;3u`` the protocol calls for
|
||||
(upstream Ghostty bug), breaking backward-kill-word (#87630
|
||||
regression). Ghostty implements modifyOtherKeys correctly (it then
|
||||
emits ``\\x1b[27;3;127~``, which the alias table also maps).
|
||||
|
||||
Matches exactly the two conditions that admit Ghostty through
|
||||
``_terminal_supports_extended_enter_keys``.
|
||||
"""
|
||||
if env is None:
|
||||
env = os.environ
|
||||
return (
|
||||
(env.get("TERM_PROGRAM") or "").strip() == "ghostty"
|
||||
or (env.get("TERM") or "").strip().lower() == "xterm-ghostty"
|
||||
)
|
||||
|
||||
|
||||
def _terminal_supports_extended_enter_keys(env: Optional[Mapping[str, str]] = None) -> bool:
|
||||
"""Whether it is safe/useful to request modified Enter key reporting.
|
||||
|
||||
@@ -4131,7 +4250,9 @@ def _enable_extended_enter_keys(output=None, env: Optional[Mapping[str, str]] =
|
||||
|
||||
Writes the Kitty keyboard protocol push (CSI >1u, disambiguate mode) AND
|
||||
xterm modifyOtherKeys level 2 (CSI >4;2m), mirroring the Ink TUI —
|
||||
terminals honor whichever protocol they implement. Both are needed:
|
||||
terminals honor whichever protocol they implement (except Ghostty, which
|
||||
gets only modifyOtherKeys; see the Ghostty exception below). Both are
|
||||
needed:
|
||||
kitty-the-terminal removed modifyOtherKeys support entirely (it only
|
||||
speaks its own protocol), while tmux/VS Code only accept modifyOtherKeys.
|
||||
|
||||
@@ -4148,20 +4269,25 @@ def _enable_extended_enter_keys(output=None, env: Optional[Mapping[str, str]] =
|
||||
Ctrl+C, which is handled by prompt_toolkit's ``c-c`` binding (raw mode
|
||||
clears ISIG, so the kernel INTR path was never in play for the CLI).
|
||||
|
||||
Ghostty exception: pushes only modifyOtherKeys — see
|
||||
``_is_ghostty_terminal`` for the full rationale (#87630).
|
||||
|
||||
The exit reset sequence pops/resets both modes, so this is safe across
|
||||
normal exits, Ctrl+C, and SIGTERM cleanup.
|
||||
"""
|
||||
if not _terminal_supports_extended_enter_keys(env):
|
||||
return False
|
||||
# Ghostty exception: only modifyOtherKeys — see _is_ghostty_terminal.
|
||||
seq = _MODIFY_OTHER_KEYS_SEQ if _is_ghostty_terminal(env) else _EXTENDED_ENTER_KEYS_SEQ
|
||||
try:
|
||||
target = output
|
||||
if target is not None and hasattr(target, "write_raw"):
|
||||
target.write_raw(_EXTENDED_ENTER_KEYS_SEQ)
|
||||
target.write_raw(seq)
|
||||
target.flush()
|
||||
return True
|
||||
stream = sys.stdout
|
||||
if stream is not None and stream.isatty():
|
||||
stream.write(_EXTENDED_ENTER_KEYS_SEQ)
|
||||
stream.write(seq)
|
||||
stream.flush()
|
||||
return True
|
||||
except Exception:
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
acgh213
|
||||
@@ -0,0 +1 @@
|
||||
grahfmusic
|
||||
@@ -0,0 +1,2 @@
|
||||
Dhruv7201
|
||||
# PR #89923 salvage (vLLM output-cap convergence)
|
||||
@@ -0,0 +1,2 @@
|
||||
ekinnee
|
||||
# PR #72283 salvage (maximum output tokens parsing)
|
||||
@@ -95,6 +95,9 @@ class RelayAdapter(BasePlatformAdapter):
|
||||
# feedback off SendResult — see send()). Consumed by the gateway's
|
||||
# semantic thread-rename lane; bounded like the sibling caches.
|
||||
self._auto_thread_by_chat: Dict[str, Tuple[str, str]] = {}
|
||||
# Bounded FIFO seen-set for inbound replay dedupe (finding #3);
|
||||
# dict preserves insertion order, giving cheap oldest-first eviction.
|
||||
self._seen_inbound: Dict[str, None] = {}
|
||||
# chat_id -> draft_id of the currently OPEN native draft stream
|
||||
# (NS-658 live cards). Armed by send_draft on a successful frame;
|
||||
# consumed by send() to convert the turn-final delivery into the
|
||||
@@ -892,6 +895,24 @@ class RelayAdapter(BasePlatformAdapter):
|
||||
|
||||
async def _on_inbound(self, event) -> None:
|
||||
"""Bridge a connector-delivered MessageEvent into the normal adapter path."""
|
||||
# Inbound replay dedupe (live-canary finding #3, Alice staging): the
|
||||
# relay leg is at-least-once — on WS re-handshake the connector
|
||||
# replays its durable per-instance buffer, and a long multi-tool turn
|
||||
# (60-100s) straddling a quiet socket drop gets its ORIGINAL inbound
|
||||
# replayed after the turn completes, re-running the whole turn (user
|
||||
# saw the final answer 2-5x). Platform message identity (chat_id +
|
||||
# message_id/ts) is stable across replays, so a bounded seen-set
|
||||
# drops them. Consumer-side idempotency; no wire change.
|
||||
dedupe_key = self._inbound_dedupe_key(event)
|
||||
if dedupe_key is not None:
|
||||
if dedupe_key in self._seen_inbound:
|
||||
logger.info(
|
||||
"relay inbound dropped as replay (dedupe key=%s)", dedupe_key
|
||||
)
|
||||
return
|
||||
self._seen_inbound[dedupe_key] = None
|
||||
while len(self._seen_inbound) > self._SEEN_INBOUND_MAX:
|
||||
self._seen_inbound.pop(next(iter(self._seen_inbound)))
|
||||
self._capture_scope(event)
|
||||
self._stamp_slack_session_thread(event)
|
||||
# Phase 3: a structured prompt answer resolves its waiting primitive
|
||||
@@ -903,6 +924,37 @@ class RelayAdapter(BasePlatformAdapter):
|
||||
await self._localize_inbound_media(event)
|
||||
await self.handle_message(event)
|
||||
|
||||
_SEEN_INBOUND_MAX = 512
|
||||
|
||||
def _inbound_dedupe_key(self, event) -> Optional[str]:
|
||||
"""Stable replay identity: (platform, chat, platform message id).
|
||||
|
||||
Chat identity lives on ``event.source`` (MessageEvent has no top-level
|
||||
``chat_id``), and this adapter can front SEVERAL platforms over one
|
||||
relay socket (Phase 1.5 multiplex), so the underlying platform joins
|
||||
the key — two platforms' numeric chat/message ids must never collide
|
||||
into one identity.
|
||||
|
||||
Returns None when the event carries no platform message id (synthetic
|
||||
events, some prompt responses) — those never dedupe, fail-open by
|
||||
design: dropping a real user message is strictly worse than rerunning
|
||||
one, so only dedupe when identity is certain.
|
||||
"""
|
||||
source = getattr(event, "source", None)
|
||||
message_id = getattr(event, "message_id", None)
|
||||
chat_id = getattr(source, "chat_id", None)
|
||||
if not message_id or not chat_id:
|
||||
return None
|
||||
# Normalize the platform component: production wire decoding always
|
||||
# yields a Platform enum (unknowns canonicalize to Platform.RELAY),
|
||||
# but alternate constructors may carry the plain string. Use the
|
||||
# enum's value when present, the string itself otherwise — both
|
||||
# spellings of one platform must produce ONE key, and two different
|
||||
# string platforms must not collapse into the same empty component.
|
||||
raw_platform = getattr(source, "platform", None)
|
||||
platform = getattr(raw_platform, "value", raw_platform) or ""
|
||||
return f"{platform}:{chat_id}:{message_id}"
|
||||
|
||||
def _relay_slack_extra(self) -> Dict[str, Any]:
|
||||
"""The Slack-behavior subset of the RELAY platform config.
|
||||
|
||||
|
||||
+124
-43
@@ -497,10 +497,26 @@ class WebSocketRelayTransport:
|
||||
# normal fast backoff, not the dormant cadence.
|
||||
self._dormant = False
|
||||
headers = self._upgrade_headers()
|
||||
# WAN-friendly keepalive: customer gateways cross WAN paths to the
|
||||
# connector; the websockets library default (ping_interval=20,
|
||||
# ping_timeout=20) gives the peer only a 20s pong deadline, which
|
||||
# produces spurious `1011 keepalive ping timeout` closes under
|
||||
# transient latency / event-loop stalls (Coatue incident 2026-08-18).
|
||||
# ping_timeout=60 tolerates such stalls while still detecting a dead
|
||||
# link within ~90s worst case (30s interval + 60s pong deadline).
|
||||
if headers:
|
||||
self._ws = await websockets.connect(self._url, additional_headers=headers) # type: ignore[union-attr]
|
||||
self._ws = await websockets.connect( # type: ignore[union-attr]
|
||||
self._url,
|
||||
additional_headers=headers,
|
||||
ping_interval=30,
|
||||
ping_timeout=60,
|
||||
)
|
||||
else:
|
||||
self._ws = await websockets.connect(self._url) # type: ignore[union-attr]
|
||||
self._ws = await websockets.connect( # type: ignore[union-attr]
|
||||
self._url,
|
||||
ping_interval=30,
|
||||
ping_timeout=60,
|
||||
)
|
||||
self._reader = asyncio.create_task(self._read_loop(), name="relay-ws-reader")
|
||||
# Send one hello PER fronted identity (Phase 1.5 Shape A). The connector
|
||||
# accumulates them into its advertised set (the first sets the session
|
||||
@@ -791,8 +807,10 @@ class WebSocketRelayTransport:
|
||||
bot_id = self._bot_id_for(platform)
|
||||
if bot_id:
|
||||
frame["botId"] = bot_id
|
||||
frame_sent = False
|
||||
try:
|
||||
await self._send(frame)
|
||||
frame_sent = True
|
||||
return await asyncio.wait_for(fut, timeout=self._outbound_timeout_s)
|
||||
except asyncio.TimeoutError:
|
||||
# AMBIGUOUS by contract (PR 85796 review): the frame reached the
|
||||
@@ -807,6 +825,28 @@ class WebSocketRelayTransport:
|
||||
"error": "relay outbound timed out",
|
||||
"ambiguous": True,
|
||||
}
|
||||
except Exception as exc: # noqa: BLE001 - a dead socket is a failed send, not a raise
|
||||
# No `is None` check can close the window where the socket dies
|
||||
# BETWEEN the liveness guard above and the actual write — the
|
||||
# reader's finally hasn't cleared _ws yet, so _send raises
|
||||
# ConnectionClosed straight into callers whose contract is a
|
||||
# result dict (RelayAdapter.send consumes it with no try).
|
||||
# Report it like every other failed send. CancelledError is a
|
||||
# BaseException, so cancellation still propagates.
|
||||
#
|
||||
# Ambiguity contract (PR 85796): a raise from the WRITE means the
|
||||
# frame never reached the wire — definite non-delivery, no flag.
|
||||
# A failure surfaced by the FUTURE (e.g. disconnect() failing
|
||||
# pending mid-flight) means the frame WAS sent and only the
|
||||
# outcome is unknown — mark it ambiguous like the timeout above.
|
||||
logger.debug("relay %s send failed", frame_type, exc_info=True)
|
||||
result: Dict[str, Any] = {
|
||||
"success": False,
|
||||
"error": f"relay send failed: {exc}",
|
||||
}
|
||||
if frame_sent:
|
||||
result["ambiguous"] = True
|
||||
return result
|
||||
finally:
|
||||
self._pending.pop(request_id, None)
|
||||
|
||||
@@ -817,50 +857,91 @@ class WebSocketRelayTransport:
|
||||
await self._ws.send(json.dumps(frame) + "\n")
|
||||
|
||||
async def _read_loop(self) -> None:
|
||||
assert self._ws is not None
|
||||
# Bind the socket this reader serves: the finally below must only
|
||||
# clear _ws if it still points at THIS socket (a supervisor re-dial
|
||||
# may have already installed a fresh one by the time we unwind).
|
||||
ws = self._ws
|
||||
buf = ""
|
||||
try:
|
||||
async for chunk in self._ws:
|
||||
buf += chunk if isinstance(chunk, str) else chunk.decode("utf-8")
|
||||
# Newline-delimited frames; keep any trailing partial line.
|
||||
*lines, buf = buf.split("\n")
|
||||
for line in lines:
|
||||
if line.strip():
|
||||
await self._handle_frame(line)
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception as exc: # noqa: BLE001 - log + let the task end; reconnection handled below
|
||||
# Phase 7 Unit 7d-B: detect a 4401 (unauthorized) close. After a prior
|
||||
# successful handshake this is a REVOCATION (opt-out / deprovision) —
|
||||
# the per-gateway secret is gone, so reconnecting is futile. Latch a
|
||||
# terminal "auth revoked" state and DON'T re-dial. Before any
|
||||
# successful handshake a 4401 stays retryable (cold-start race).
|
||||
if self._close_code_of(exc) == _RELAY_UNAUTHORIZED_CLOSE_CODE and self._handshake_succeeded:
|
||||
self._auth_revoked = True
|
||||
if not self._closing:
|
||||
logger.warning(
|
||||
"relay ws closed 4401 (unauthorized) after a successful handshake — "
|
||||
"treating as a revoked relay credential (opt-out); not reconnecting"
|
||||
if ws is None:
|
||||
# Scheduled without a socket (a lifecycle bug, not a normal
|
||||
# path). The old `assert` here escaped BEFORE the finally
|
||||
# existed to fail pending futures — the one exit that could
|
||||
# still strand waiters for the full outbound timeout. Fall
|
||||
# through to the finally instead; it settles them all.
|
||||
logger.error("relay ws read loop started with no socket")
|
||||
return
|
||||
try:
|
||||
async for chunk in self._ws:
|
||||
buf += chunk if isinstance(chunk, str) else chunk.decode("utf-8")
|
||||
# Newline-delimited frames; keep any trailing partial line.
|
||||
*lines, buf = buf.split("\n")
|
||||
for line in lines:
|
||||
if line.strip():
|
||||
await self._handle_frame(line)
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception as exc: # noqa: BLE001 - log + let the task end; reconnection handled below
|
||||
# Phase 7 Unit 7d-B: detect a 4401 (unauthorized) close. After a prior
|
||||
# successful handshake this is a REVOCATION (opt-out / deprovision) —
|
||||
# the per-gateway secret is gone, so reconnecting is futile. Latch a
|
||||
# terminal "auth revoked" state and DON'T re-dial. Before any
|
||||
# successful handshake a 4401 stays retryable (cold-start race).
|
||||
if self._close_code_of(exc) == _RELAY_UNAUTHORIZED_CLOSE_CODE and self._handshake_succeeded:
|
||||
self._auth_revoked = True
|
||||
if not self._closing:
|
||||
logger.warning(
|
||||
"relay ws closed 4401 (unauthorized) after a successful handshake — "
|
||||
"treating as a revoked relay credential (opt-out); not reconnecting"
|
||||
)
|
||||
elif not self._closing:
|
||||
logger.warning("relay ws read loop ended: %s", exc)
|
||||
# Phase 5 §5.3: the socket closed. If reconnect is enabled and this was
|
||||
# NOT a deliberate disconnect(), kick the reconnect supervisor so the
|
||||
# gateway re-dials + re-handshakes (which triggers the connector's
|
||||
# buffered-flip drain on the new handshake). Self-scheduling: the reader
|
||||
# ends here, the supervisor re-dials and starts a fresh reader.
|
||||
# Phase 7 Unit 7d-B: a revoked credential (terminal 4401) is the one case
|
||||
# we deliberately do NOT reconnect — the secret is dead until the
|
||||
# instance is recreated, so spinning would just reproduce the failure.
|
||||
if (
|
||||
self._reconnect
|
||||
and not self._closing
|
||||
and not self._auth_revoked
|
||||
and (self._supervisor is None or self._supervisor.done())
|
||||
):
|
||||
self._supervisor = asyncio.create_task(
|
||||
self._reconnect_loop(), name="relay-ws-reconnect"
|
||||
)
|
||||
finally:
|
||||
# The socket this reader served is dead. Drop the handle (identity-
|
||||
# guarded: a re-dial that already installed a FRESH socket must not
|
||||
# be clobbered) so every `self._ws is None` liveness check — send,
|
||||
# _request_response, go_idle, go_dormant — reports "not connected"
|
||||
# for the whole outage. Without this, _ws kept pointing at the dead
|
||||
# socket on every reader exit that arms NO supervisor (terminal
|
||||
# 4401 revocation, reconnect=False transports), and a send there
|
||||
# registered a future nothing could resolve: a full
|
||||
# _outbound_timeout_s (~30s) wedge — including the revocation
|
||||
# path's own fatal-error notification. disconnect() owns the
|
||||
# handle during deliberate teardown, so leave it alone then.
|
||||
if self._ws is ws and not self._closing:
|
||||
self._ws = None
|
||||
# The reader is the ONLY thing that can resolve a pending
|
||||
# outbound_result future — once it exits (socket dropped, error,
|
||||
# or cancellation cleanup) every in-flight _request_response waiter
|
||||
# is unresolvable and would otherwise block the full
|
||||
# _outbound_timeout_s (~30s) on a dead socket (Coatue incident
|
||||
# 2026-08-18: stuck sends after a 1011 keepalive close). Fail them
|
||||
# NOW with the dict shape callers expect (never an exception on
|
||||
# the outbound path). list() snapshot: set_result wakes waiters
|
||||
# whose finally-pop would otherwise mutate the dict mid-iteration.
|
||||
for _rid, fut in list(self._pending.items()):
|
||||
if not fut.done():
|
||||
fut.set_result(
|
||||
{"success": False, "error": "relay transport connection lost"}
|
||||
)
|
||||
elif not self._closing:
|
||||
logger.warning("relay ws read loop ended: %s", exc)
|
||||
# Phase 5 §5.3: the socket closed. If reconnect is enabled and this was
|
||||
# NOT a deliberate disconnect(), kick the reconnect supervisor so the
|
||||
# gateway re-dials + re-handshakes (which triggers the connector's
|
||||
# buffered-flip drain on the new handshake). Self-scheduling: the reader
|
||||
# ends here, the supervisor re-dials and starts a fresh reader.
|
||||
# Phase 7 Unit 7d-B: a revoked credential (terminal 4401) is the one case
|
||||
# we deliberately do NOT reconnect — the secret is dead until the
|
||||
# instance is recreated, so spinning would just reproduce the failure.
|
||||
if (
|
||||
self._reconnect
|
||||
and not self._closing
|
||||
and not self._auth_revoked
|
||||
and (self._supervisor is None or self._supervisor.done())
|
||||
):
|
||||
self._supervisor = asyncio.create_task(
|
||||
self._reconnect_loop(), name="relay-ws-reconnect"
|
||||
)
|
||||
self._pending.clear()
|
||||
|
||||
@staticmethod
|
||||
def _close_code_of(exc: BaseException) -> Optional[int]:
|
||||
|
||||
@@ -1219,18 +1219,23 @@ class CLICommandsMixin:
|
||||
self._handle_resume_command(f"/resume {arg}")
|
||||
|
||||
def _handle_worktree_command(self, cmd_original: str) -> None:
|
||||
"""Handle /worktree — inspect or create isolated git worktrees.
|
||||
"""Handle /worktree — inspect, create, or reclaim isolated git worktrees.
|
||||
|
||||
Syntax:
|
||||
/worktree — show the active worktree (if any)
|
||||
/worktree new [name] — create a worktree and move this session into it
|
||||
/worktree list — list worktrees under the repo's .worktrees/
|
||||
/worktree — show the active worktree (if any)
|
||||
/worktree new [name] — create a worktree and move this session into it
|
||||
/worktree list — list worktrees under the repo's .worktrees/
|
||||
/worktree prune [--dry-run] — reclaim safe trees + merged branches
|
||||
|
||||
Inspired by Copilot CLI's ``/worktree new``: start isolated work in a
|
||||
fresh worktree without leaving the session. Creating one retargets the
|
||||
terminal/file tools (``TERMINAL_CWD`` + process cwd) at the new tree;
|
||||
the launcher's exit cleanup applies (kept only when it has unpushed
|
||||
commits, same as ``hermes -w``).
|
||||
|
||||
``prune`` is the same attended reclaim as ``hermes worktree prune``
|
||||
(hermes_cli/worktree_gc.py): never deletes tracked changes, unique
|
||||
unpushed commits, or in-use trees; archives untracked-only scratch.
|
||||
"""
|
||||
import subprocess
|
||||
|
||||
@@ -1250,10 +1255,50 @@ class CLICommandsMixin:
|
||||
print(" No active worktree for this session.")
|
||||
if repo_root:
|
||||
print(" /worktree new [name] — create one and move this session into it")
|
||||
print(" /worktree prune — reclaim stale trees and merged branches")
|
||||
else:
|
||||
print(" (not inside a git repository)")
|
||||
return
|
||||
|
||||
if sub in {"prune", "gc", "clean"}:
|
||||
if not repo_root:
|
||||
print(" Not inside a git repository.")
|
||||
return
|
||||
rest = parts[2].strip().lower() if len(parts) > 2 else ""
|
||||
dry_run = "--dry-run" in rest or "-n" in rest.split()
|
||||
from hermes_cli import worktree_gc
|
||||
|
||||
active = _cli._active_worktree
|
||||
tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False)
|
||||
if active:
|
||||
# Never reap the tree this very session is sitting in, even
|
||||
# if a concurrent audit would judge it clean+merged.
|
||||
active_path = str(active.get("path") or "")
|
||||
tree_records = [
|
||||
record for record in tree_records
|
||||
if record.path != active_path
|
||||
]
|
||||
actions = worktree_gc.reclaim_worktrees(
|
||||
repo_root, dry_run=dry_run, records=tree_records
|
||||
)
|
||||
actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run)
|
||||
if actions:
|
||||
for line in actions:
|
||||
print(f" {line}")
|
||||
print(f" {len(actions)} action(s) {'planned' if dry_run else 'done'}.")
|
||||
else:
|
||||
print(" Nothing to reclaim — remaining trees/branches carry real work.")
|
||||
kept = [
|
||||
record for record in tree_records
|
||||
if record.verdict == "keep"
|
||||
and "kanban" not in record.reason and "in use" not in record.reason
|
||||
]
|
||||
if kept:
|
||||
print(f" Preserved {len(kept)} tree(s) with real work:")
|
||||
for record in kept:
|
||||
print(f" {record.name}: {record.reason}")
|
||||
return
|
||||
|
||||
if sub in {"list", "ls"}:
|
||||
if not repo_root:
|
||||
print(" Not inside a git repository.")
|
||||
|
||||
@@ -167,9 +167,9 @@ COMMAND_REGISTRY: list[CommandDef] = [
|
||||
args_hint="<platform>", cli_only=True),
|
||||
CommandDef("branch", "Branch the current session (explore a different path)", "Session",
|
||||
aliases=("fork",), args_hint="[name]"),
|
||||
CommandDef("worktree", "Show, list, or create isolated git worktrees for this session", "Session",
|
||||
cli_only=True, args_hint="[new [name]|list]",
|
||||
subcommands=("new", "list")),
|
||||
CommandDef("worktree", "Show, list, create, or prune isolated git worktrees", "Session",
|
||||
cli_only=True, args_hint="[new [name]|list|prune [--dry-run]]",
|
||||
subcommands=("new", "list", "prune")),
|
||||
CommandDef("compress", "Compress conversation context (add 'here [N]' to keep recent N turns; --preview shows what would happen)", "Session",
|
||||
aliases=("compact",), args_hint="[here [N] | focus topic | --preview|--dry-run]"),
|
||||
CommandDef("rollback", "List or restore filesystem checkpoints (restores keep your hand-edits; --all overrides)", "Session",
|
||||
|
||||
@@ -1868,6 +1868,10 @@ DEFAULT_CONFIG = {
|
||||
"write_approval": False,
|
||||
"memory_char_limit": 2200, # ~800 tokens at 2.75 chars/token
|
||||
"user_char_limit": 1375, # ~500 tokens at 2.75 chars/token
|
||||
# Periodic built-in memory review. External providers with automatic
|
||||
# turn/session extraction can set this to 0 and keep the small local
|
||||
# store reserved for explicit high-frequency operational facts.
|
||||
"nudge_interval": 10,
|
||||
# External memory provider plugin (empty = built-in only).
|
||||
# Set to a provider name to activate: "openviking", "mem0",
|
||||
# "hindsight", "holographic", "retaindb", "byterover".
|
||||
|
||||
+52
-1
@@ -11644,7 +11644,7 @@ _BUILTIN_SUBCOMMANDS = frozenset(
|
||||
"resume",
|
||||
"send", "sessions", "setup",
|
||||
"skin", "skills", "slack", "status", "sync", "tools", "uninstall", "update",
|
||||
"version", "webhook", "whatsapp", "whatsapp-cloud", "chat", "secrets", "security",
|
||||
"version", "webhook", "whatsapp", "whatsapp-cloud", "worktree", "chat", "secrets", "security",
|
||||
"verify",
|
||||
# Help-ish invocations — plugin commands not being listed in
|
||||
# top-level --help is an acceptable trade-off for skipping an
|
||||
@@ -12514,6 +12514,57 @@ def main():
|
||||
)
|
||||
fallback_parser.set_defaults(func=cmd_fallback)
|
||||
|
||||
# =========================================================================
|
||||
# worktree command — audit/reclaim accumulated git worktrees + branches
|
||||
# =========================================================================
|
||||
worktree_parser = subparsers.add_parser(
|
||||
"worktree",
|
||||
help="Audit and reclaim accumulated git worktrees and merged branches",
|
||||
description=(
|
||||
"Attended reclaim for the .worktrees/ directory hermes -w sessions "
|
||||
"accumulate. Never deletes uncommitted tracked changes, unique "
|
||||
"unpushed commits, or in-use trees; untracked-only scratch is "
|
||||
"archived to ~/.hermes/archive/worktree-prune/ before removal. See: "
|
||||
"https://hermes-agent.nousresearch.com/docs/user-guide/cli#worktree-cleanup"
|
||||
),
|
||||
)
|
||||
worktree_subparsers = worktree_parser.add_subparsers(dest="worktree_action")
|
||||
worktree_list = worktree_subparsers.add_parser(
|
||||
"list",
|
||||
aliases=["ls", "audit"],
|
||||
help="Classify every tree: age, size, verdict, reason (default action)",
|
||||
)
|
||||
worktree_list.add_argument("--repo", help="Repo root (default: current repo)")
|
||||
worktree_prune = worktree_subparsers.add_parser(
|
||||
"prune",
|
||||
help="Remove safe trees and delete fully-merged local branches",
|
||||
)
|
||||
worktree_prune.add_argument("--repo", help="Repo root (default: current repo)")
|
||||
worktree_prune.add_argument(
|
||||
"--dry-run", action="store_true",
|
||||
help="Show the plan without changing anything",
|
||||
)
|
||||
worktree_prune.add_argument(
|
||||
"--trees-only", action="store_true",
|
||||
help="Only remove worktrees; leave local branches alone",
|
||||
)
|
||||
worktree_prune.add_argument(
|
||||
"--branches-only", action="store_true",
|
||||
help="Only delete merged local branches; leave worktrees alone",
|
||||
)
|
||||
|
||||
def _dispatch_worktree(_args):
|
||||
from hermes_cli.worktree_cmd import cmd_worktree
|
||||
|
||||
# argparse aliases set dest to the literal typed string ("ls"/"audit").
|
||||
action = getattr(_args, "worktree_action", None)
|
||||
if action in ("ls", "audit"):
|
||||
_args.worktree_action = "list"
|
||||
return cmd_worktree(_args)
|
||||
|
||||
worktree_parser.set_defaults(func=_dispatch_worktree)
|
||||
|
||||
|
||||
# =========================================================================
|
||||
# secrets command — external secret managers (Bitwarden, 1Password)
|
||||
# =========================================================================
|
||||
|
||||
@@ -490,10 +490,11 @@ def cmd_status(args) -> None:
|
||||
user_mark = "enabled ✓" if user_profile_enabled else "disabled ✗"
|
||||
|
||||
# Check if the memory tool is enabled for the CLI platform via the
|
||||
# canonical resolver (handles composite toolsets like hermes-cli).
|
||||
# canonical resolver and respects the check_fn gate when both stores are disabled.
|
||||
from hermes_cli.tools_config import _get_platform_tools
|
||||
from tools.memory_tool import check_memory_requirements
|
||||
cli_tools = _get_platform_tools(config, "cli", include_default_mcp_servers=False)
|
||||
memory_tool_enabled = "memory" in cli_tools
|
||||
memory_tool_enabled = ("memory" in cli_tools) and check_memory_requirements()
|
||||
tool_mark = "enabled ✓" if memory_tool_enabled else "disabled ✗"
|
||||
|
||||
print("\nMemory status\n" + "─" * 40)
|
||||
|
||||
+125
-19
@@ -11,6 +11,24 @@ can be unit-tested without importing the whole CLI runtime.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
# kitty CSI-u ORs lock-key state into the modifier parameter of every key
|
||||
# event while a lock is on: CapsLock=64, NumLock=128, both=192 (#88221,
|
||||
# #89651). Every fixed-modifier CSI-u (and legacy CSI-tilde / CSI-letter)
|
||||
# registration therefore needs lock-offset twins, or those events leak into
|
||||
# the prompt as literal text. The xterm modifyOtherKeys ``ESC[27;N;CP~``
|
||||
# encoding never carries lock bits, so it never gets the twins.
|
||||
_LOCK_BIT_OFFSETS = (0, 64, 128, 192)
|
||||
|
||||
|
||||
def _lock_variants(modifier: int) -> tuple[int, ...]:
|
||||
"""Return ``modifier`` plus its CapsLock/NumLock/both twins."""
|
||||
return tuple(modifier + off for off in _LOCK_BIT_OFFSETS)
|
||||
|
||||
|
||||
def _lock_twins(modifier: int) -> tuple[int, ...]:
|
||||
"""Return only the lock twins of ``modifier`` (never the base value)."""
|
||||
return tuple(modifier + off for off in _LOCK_BIT_OFFSETS[1:])
|
||||
|
||||
|
||||
def _clear_vt100_prefix_cache() -> None:
|
||||
"""Drop prompt_toolkit's memoized "is this a prefix of a longer match?"
|
||||
@@ -37,6 +55,7 @@ def install_shift_enter_alias() -> int:
|
||||
|
||||
Sequences mapped:
|
||||
- "\\x1b[13;2u" — Kitty keyboard protocol / CSI-u, modifier=2 (Shift)
|
||||
(plus its CapsLock/NumLock lock twins via ``_lock_variants``)
|
||||
- "\\x1b[27;2;13~" — xterm modifyOtherKeys=2, modifier=2 (Shift)
|
||||
- "\\x1b[27;2;13u" — alternate ordering some emitters use
|
||||
|
||||
@@ -62,7 +81,9 @@ def install_shift_enter_alias() -> int:
|
||||
|
||||
alt_enter = (Keys.Escape, Keys.ControlM)
|
||||
changed = 0
|
||||
for seq in ("\x1b[13;2u", "\x1b[27;2;13~", "\x1b[27;2;13u"):
|
||||
seqs = [f"\x1b[13;{m}u" for m in _lock_variants(2)]
|
||||
seqs += ["\x1b[27;2;13~", "\x1b[27;2;13u"]
|
||||
for seq in seqs:
|
||||
if ANSI_SEQUENCES.get(seq) != alt_enter:
|
||||
ANSI_SEQUENCES[seq] = alt_enter
|
||||
changed += 1
|
||||
@@ -78,10 +99,13 @@ def install_ctrl_enter_alias() -> int:
|
||||
|
||||
Sequences mapped:
|
||||
- "\\x1b[13;5u" — Kitty keyboard protocol / CSI-u, modifier=5 (Ctrl)
|
||||
(plus its CapsLock/NumLock lock twins via ``_lock_variants``)
|
||||
- "\\x1b[27;5;13~" — xterm modifyOtherKeys=2, modifier=5 (Ctrl)
|
||||
- "\\x1b[27;5;13u" — alternate ordering some emitters use
|
||||
|
||||
Stock prompt_toolkit doesn't map any of these. Without this alias,
|
||||
Stock prompt_toolkit maps only the tilde form ``\\x1b[27;5;13~`` (to
|
||||
plain ``Keys.ControlM``, which this deliberately overwrites — same
|
||||
bug-fix rationale as install_shift_enter_alias). Without this alias,
|
||||
Kitty/mintty/xterm-with-modifyOtherKeys users over SSH never get a
|
||||
Ctrl+Enter newline — the keystroke arrives as a raw CSI sequence that
|
||||
falls through to the default character-insert handler. See #22379.
|
||||
@@ -96,7 +120,9 @@ def install_ctrl_enter_alias() -> int:
|
||||
|
||||
alt_enter = (Keys.Escape, Keys.ControlM)
|
||||
changed = 0
|
||||
for seq in ("\x1b[13;5u", "\x1b[27;5;13~", "\x1b[27;5;13u"):
|
||||
seqs = [f"\x1b[13;{m}u" for m in _lock_variants(5)]
|
||||
seqs += ["\x1b[27;5;13~", "\x1b[27;5;13u"]
|
||||
for seq in seqs:
|
||||
if ANSI_SEQUENCES.get(seq) != alt_enter:
|
||||
ANSI_SEQUENCES[seq] = alt_enter
|
||||
changed += 1
|
||||
@@ -116,7 +142,8 @@ def install_cmd_backspace_alias() -> int:
|
||||
literal insertion.
|
||||
|
||||
Cmd+Backspace → ``Keys.ControlU`` (kill backward to start of line).
|
||||
Codepoint 127 with modifier 9 (super) / 10 (super+shift):
|
||||
Codepoint 127 with modifier 9 (super) / 10 (super+shift), each with
|
||||
its CapsLock/NumLock lock twins via ``_lock_variants``:
|
||||
- ``\\x1b[127;9u`` / ``\\x1b[127;10u`` — Kitty CSI-u
|
||||
- ``\\x1b[27;9;127~`` — xterm modifyOtherKeys
|
||||
|
||||
@@ -133,13 +160,12 @@ def install_cmd_backspace_alias() -> int:
|
||||
except Exception:
|
||||
return 0
|
||||
|
||||
aliases = {
|
||||
"\x1b[127;9u": Keys.ControlU,
|
||||
"\x1b[127;10u": Keys.ControlU,
|
||||
"\x1b[27;9;127~": Keys.ControlU,
|
||||
"\x1b[3;9~": Keys.ControlK,
|
||||
"\x1b[3;10~": Keys.ControlK,
|
||||
}
|
||||
aliases: dict[str, object] = {}
|
||||
for base in (9, 10): # super / super+shift
|
||||
for mod in _lock_variants(base):
|
||||
aliases[f"\x1b[127;{mod}u"] = Keys.ControlU
|
||||
aliases[f"\x1b[3;{mod}~"] = Keys.ControlK
|
||||
aliases["\x1b[27;9;127~"] = Keys.ControlU
|
||||
changed = 0
|
||||
for seq, key in aliases.items():
|
||||
if ANSI_SEQUENCES.get(seq) != key:
|
||||
@@ -181,6 +207,11 @@ def install_modify_other_keys_aliases() -> int:
|
||||
Ctrl+Alt+Shift=8): normalized onto the same targets — Ctrl-bearing
|
||||
combos behave as the Ctrl key (Alt adds an ``Escape`` prefix),
|
||||
matching how dte/kakoune normalize these protocols.
|
||||
* **Lock-bit variants**: every CSI-u mapping above is also installed
|
||||
with the CapsLock (64) and NumLock (128) bits ORed into the modifier
|
||||
parameter — kitty/ghostty include them while a lock is on, and
|
||||
without the variants every key combo dies with the lock enabled
|
||||
(``ESC[99;133u`` instead of ``ESC[99;5u``, #89651).
|
||||
* **Esc key**: ``ESC[27u`` / ``ESC[27;<mod>u`` (Kitty disambiguate mode
|
||||
reports Esc this way, #56684) → ``Keys.Escape``.
|
||||
* **Modified Enter/Tab/Backspace/Space**: Alt+Enter → the Alt+Enter
|
||||
@@ -244,15 +275,26 @@ def install_modify_other_keys_aliases() -> int:
|
||||
|
||||
changed = 0
|
||||
|
||||
# Kitty CSI-u encodes CapsLock/NumLock state as extra modifier bits
|
||||
# (caps=64, num=128) ORed into the parameter: with NumLock on, Ctrl+C
|
||||
# arrives as ESC[99;133u (5 + 128) instead of ESC[99;5u. Terminals
|
||||
# that report these bits (kitty, ghostty) break every key combo while
|
||||
# a lock is on (#89651) unless the lock variants are mapped too. The
|
||||
# xterm modifyOtherKeys encoding never carries the lock bits, so only
|
||||
# the CSI-u form needs them.
|
||||
def _install_paired(modifier: int, mapping: dict) -> None:
|
||||
"""Install both modifyOtherKeys (ESC[27;N;CP~) and CSI-u (ESC[CP;Nu)
|
||||
mappings for the given modifier and codepoint→key mapping."""
|
||||
mappings for the given modifier and codepoint→key mapping.
|
||||
|
||||
The tilde form is skipped for modifier 1 ("no modifier") — xterm
|
||||
never emits modifier-1 tilde sequences.
|
||||
"""
|
||||
nonlocal changed
|
||||
for codepoint, key_val in mapping.items():
|
||||
for seq in (
|
||||
f"\x1b[27;{modifier};{codepoint}~",
|
||||
f"\x1b[{codepoint};{modifier}u",
|
||||
):
|
||||
seqs = [] if modifier == 1 else [f"\x1b[27;{modifier};{codepoint}~"]
|
||||
for mod in _lock_variants(modifier):
|
||||
seqs.append(f"\x1b[{codepoint};{mod}u")
|
||||
for seq in seqs:
|
||||
if seq not in ANSI_SEQUENCES:
|
||||
ANSI_SEQUENCES[seq] = key_val
|
||||
changed += 1
|
||||
@@ -319,9 +361,16 @@ def install_modify_other_keys_aliases() -> int:
|
||||
# Disambiguate mode reports the Esc key as CSI-u so it is
|
||||
# distinguishable from the ESC byte that starts escape sequences
|
||||
# (#56684 — previously leaked "[27u" as literal text into the prompt).
|
||||
# Modifiers run to 16 because kitty reports Cmd as the super bit
|
||||
# (mod 9+) — same reason install_cmd_backspace_alias maps 9/10.
|
||||
for seq in ["\x1b[27u"] + [f"\x1b[27;{m}u" for m in range(2, 17)]:
|
||||
# Modifiers run from 1 to 16: kitty reports Cmd as the super bit
|
||||
# (mod 9+) — same reason install_cmd_backspace_alias maps 9/10 — and
|
||||
# the lock-bit variants of the modifier-less form (1+64/128/192) are
|
||||
# how a lone Esc keypress arrives with a lock on. Lock bits (caps/num)
|
||||
# get the same variant treatment as _install_paired.
|
||||
for seq in ["\x1b[27u"] + [
|
||||
f"\x1b[27;{mod}u"
|
||||
for m in range(1, 17)
|
||||
for mod in _lock_variants(m)
|
||||
]:
|
||||
if seq not in ANSI_SEQUENCES:
|
||||
ANSI_SEQUENCES[seq] = Keys.Escape
|
||||
changed += 1
|
||||
@@ -345,6 +394,57 @@ def install_modify_other_keys_aliases() -> int:
|
||||
# matching Ink TUI + Desktop (#78285)
|
||||
})
|
||||
|
||||
# -- Unmodified keys with a lock bit set (kitty modifier 1 = "none") --
|
||||
# With a lock on, kitty stamps the lock bit onto keys pressed with NO
|
||||
# real modifier too, so plain Backspace arrives as ESC[127;129u
|
||||
# (1 + 128) rather than \x7f. _install_paired(1, ...) registers the
|
||||
# bare mod-1 spelling and its lock twins. Only keys kitty CSI-u-encodes
|
||||
# on their own are listed; plain text characters are still delivered
|
||||
# as UTF-8, lock bits or not.
|
||||
_install_paired(1, {
|
||||
9: Keys.ControlI, # Tab
|
||||
13: Keys.ControlM, # Enter
|
||||
32: " ", # Space
|
||||
127: Keys.ControlH, # Backspace
|
||||
})
|
||||
|
||||
# -- Lock-key modifier bits (NumLock=128, CapsLock=64) on the legacy
|
||||
# CSI-letter / CSI-tilde forms kitty keeps using under the disambiguate
|
||||
# push: kitty encodes lock state into the modifier parameter, so a
|
||||
# plain Down with NumLock on arrives as ESC[1;129B (NumLock), ESC[1;65B
|
||||
# (CapsLock) or ESC[1;193B (both) instead of the legacy ESC[B — and a
|
||||
# modified one shifts the same way (Alt+Left → ESC[1;131D). Those fall
|
||||
# through the parser and leak as literal text ("[1;129B") in the input
|
||||
# line. Derive the lock twins from whatever the table already maps for
|
||||
# the base modifier (stock prompt_toolkit entries included), so every
|
||||
# modifier the terminal can report keeps working under a lock.
|
||||
for m in range(1, 17):
|
||||
# CSI-letter navigation: Up/Down/Right/Left/End/Home + F1-F4
|
||||
for trailer in "ABCDFHPQRS":
|
||||
base_seq = f"\x1b[1;{m}{trailer}" if m > 1 else f"\x1b[{trailer}"
|
||||
key = ANSI_SEQUENCES.get(base_seq)
|
||||
if key is None and m == 1:
|
||||
# Plain F1-F4 live in the table as SS3 (ESC O P) forms.
|
||||
key = ANSI_SEQUENCES.get(f"\x1bO{trailer}")
|
||||
if key is None:
|
||||
continue
|
||||
for mod in _lock_twins(m):
|
||||
seq = f"\x1b[1;{mod}{trailer}"
|
||||
if seq not in ANSI_SEQUENCES:
|
||||
ANSI_SEQUENCES[seq] = key
|
||||
changed += 1
|
||||
# CSI-tilde navigation: Insert/Delete/PageUp/PageDown/Home/End
|
||||
for num in (1, 2, 3, 4, 5, 6, 7, 8):
|
||||
base_seq = f"\x1b[{num};{m}~" if m > 1 else f"\x1b[{num}~"
|
||||
key = ANSI_SEQUENCES.get(base_seq)
|
||||
if key is None:
|
||||
continue
|
||||
for mod in _lock_twins(m):
|
||||
seq = f"\x1b[{num};{mod}~"
|
||||
if seq not in ANSI_SEQUENCES:
|
||||
ANSI_SEQUENCES[seq] = key
|
||||
changed += 1
|
||||
|
||||
# -- Kitty functional keys (Private Use Area codepoints) ----
|
||||
# kitty emits these CSI-u encodings even in LEGACY mode for keys that
|
||||
# have no legacy encoding, so unmapped they leak as literal text in any
|
||||
@@ -379,6 +479,12 @@ def install_modify_other_keys_aliases() -> int:
|
||||
if seq not in ANSI_SEQUENCES:
|
||||
ANSI_SEQUENCES[seq] = key_val
|
||||
changed += 1
|
||||
# Lock twins: with a lock on these arrive as ESC[<code>;129u etc.
|
||||
for mod in _lock_twins(1):
|
||||
seq = f"\x1b[{code};{mod}u"
|
||||
if seq not in ANSI_SEQUENCES:
|
||||
ANSI_SEQUENCES[seq] = key_val
|
||||
changed += 1
|
||||
|
||||
# New longer sequences can flip "is this a prefix of a longer match?"
|
||||
# answers the VT100 parser already cached — drop the cache so parsers
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
"""``hermes worktree`` — audit and reclaim accumulated git worktrees/branches.
|
||||
|
||||
Attended counterpart of the silent startup pruner (see
|
||||
``hermes_cli/worktree_gc.py`` for the policy and shared invariants). Usage:
|
||||
|
||||
hermes worktree list # audit: verdict + reason per tree
|
||||
hermes worktree prune # reap safe trees + merged branches
|
||||
hermes worktree prune --dry-run # show the plan, change nothing
|
||||
hermes worktree prune --trees-only / --branches-only
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Optional
|
||||
|
||||
|
||||
def _repo_root() -> Optional[str]:
|
||||
import cli as _cli
|
||||
|
||||
return _cli._git_repo_root()
|
||||
|
||||
|
||||
def _fmt_size(size_mb: Optional[int]) -> str:
|
||||
if size_mb is None:
|
||||
return "?"
|
||||
if size_mb >= 1024:
|
||||
return f"{size_mb / 1024:.1f}G"
|
||||
return f"{size_mb}M"
|
||||
|
||||
|
||||
def cmd_worktree(args) -> int:
|
||||
from hermes_cli import worktree_gc
|
||||
|
||||
repo_root = getattr(args, "repo", None) or _repo_root()
|
||||
if not repo_root:
|
||||
print("Not inside a git repository (or pass --repo <path>).")
|
||||
return 1
|
||||
|
||||
action = getattr(args, "worktree_action", None) or "list"
|
||||
|
||||
if action == "list":
|
||||
records = worktree_gc.audit_worktrees(repo_root)
|
||||
if not records:
|
||||
print("No worktrees under .worktrees/ — nothing to reclaim.")
|
||||
return 0
|
||||
total_mb = sum(r.size_mb or 0 for r in records)
|
||||
reapable_mb = sum(
|
||||
r.size_mb or 0 for r in records if r.verdict.startswith("reap")
|
||||
)
|
||||
print(f"{'TREE':32} {'AGE':>6} {'SIZE':>6} {'VERDICT':13} REASON")
|
||||
for r in sorted(records, key=lambda x: -(x.size_mb or 0)):
|
||||
print(
|
||||
f"{r.name[:32]:32} {r.age_days:>5.1f}d {_fmt_size(r.size_mb):>6} "
|
||||
f"{r.verdict:13} {r.reason}"
|
||||
)
|
||||
print(
|
||||
f"\n{len(records)} tree(s), {_fmt_size(total_mb)} total — "
|
||||
f"{_fmt_size(reapable_mb)} reclaimable now via `hermes worktree prune`."
|
||||
)
|
||||
branch_records = worktree_gc.audit_branches(repo_root)
|
||||
deletable = [b for b in branch_records if b.verdict == "delete"]
|
||||
if deletable:
|
||||
print(
|
||||
f"{len(deletable)} local branch(es) fully merged/patch-equivalent "
|
||||
f"upstream would also be deleted."
|
||||
)
|
||||
return 0
|
||||
|
||||
if action == "prune":
|
||||
dry_run = bool(getattr(args, "dry_run", False))
|
||||
trees_only = bool(getattr(args, "trees_only", False))
|
||||
branches_only = bool(getattr(args, "branches_only", False))
|
||||
|
||||
actions: list = []
|
||||
if not branches_only:
|
||||
tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False)
|
||||
actions += worktree_gc.reclaim_worktrees(
|
||||
repo_root, dry_run=dry_run, records=tree_records
|
||||
)
|
||||
kept = [r for r in tree_records if r.verdict == "keep"
|
||||
and "kanban" not in r.reason and "in use" not in r.reason]
|
||||
if kept:
|
||||
print(f"Preserved {len(kept)} tree(s) with real work:")
|
||||
for r in kept:
|
||||
print(f" {r.name}: {r.reason}")
|
||||
if not trees_only:
|
||||
actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run)
|
||||
|
||||
if actions:
|
||||
for line in actions:
|
||||
print(f" {line}")
|
||||
verb = "planned" if dry_run else "done"
|
||||
print(f"{len(actions)} action(s) {verb}.")
|
||||
else:
|
||||
print("Nothing to reclaim — all trees/branches carry real work or are in use.")
|
||||
return 0
|
||||
|
||||
print(f"Unknown worktree action: {action}")
|
||||
return 1
|
||||
@@ -0,0 +1,432 @@
|
||||
"""On-demand worktree + branch reclaim (``hermes worktree`` / ``/worktree prune``).
|
||||
|
||||
The startup pruner in ``cli._prune_stale_worktrees`` is deliberately
|
||||
conservative and silent: it runs before the banner on every ``hermes -w``
|
||||
launch, so it only reaps clean, fully-merged scratch trees past an age tier
|
||||
and preserves everything else. That policy is correct for an unattended
|
||||
startup path — but it means real installs accumulate two kinds of debris the
|
||||
startup pass can never touch:
|
||||
|
||||
- **Preserved trees** whose only "dirt" is untracked scratch (PR body drafts,
|
||||
logs) on an otherwise merged branch — preserved forever by the dirty guard.
|
||||
- **Orphaned local branches** beyond the two auto-generated prefixes the
|
||||
startup pass deletes (``hermes/hermes-*``, ``pr-*``): salvage lanes, port
|
||||
branches, feature branches whose PRs merged months ago. Multi-agent boxes
|
||||
reach hundreds.
|
||||
|
||||
This module is the *attended* counterpart: an explicit, loud, dry-run-first
|
||||
reclaim the user invokes, so it can be more thorough while staying just as
|
||||
safe. Invariants shared with the startup pruner (never violated here either):
|
||||
|
||||
- tracked modifications are NEVER deleted, at any age, in any mode;
|
||||
- unique unpushed commits are NEVER deleted (``git cherry`` patch-equivalence
|
||||
decides "unique"; shallow repos are deepened bloblessly first so the
|
||||
verdict is trustworthy);
|
||||
- live-locked trees (owning pid alive) are never touched;
|
||||
- a branch is deleted only after its worktree removal succeeded — a failed
|
||||
removal must not orphan reachable commits;
|
||||
- untracked-only dirt is ARCHIVED to ``~/.hermes/archive/worktree-prune/``
|
||||
before its tree is reaped, never destroyed.
|
||||
|
||||
Classification primitives are imported from ``cli`` so the two paths can
|
||||
never drift apart on what "dirty", "unpushed", or "merged" means.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import List, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Branches never considered for deletion, in any mode.
|
||||
_PROTECTED_BRANCHES = {"main", "master", "develop", "dev", "trunk"}
|
||||
|
||||
# Trees owned by another lifecycle (kanban dispatcher gc) — never touched.
|
||||
_KANBAN_RE = re.compile(r"^t_[0-9a-f]+$")
|
||||
|
||||
# Bounded cherry probe: a branch this far ahead of upstream is a stale-base
|
||||
# lane, not merged scratch; checking it is expensive and it stays preserved.
|
||||
_MAX_CHERRY_AHEAD = 50
|
||||
|
||||
|
||||
@dataclass
|
||||
class TreeRecord:
|
||||
name: str
|
||||
path: str
|
||||
branch: str
|
||||
age_days: float
|
||||
size_mb: Optional[int]
|
||||
verdict: str # reap | reap-archive | keep
|
||||
reason: str
|
||||
untracked: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
@dataclass
|
||||
class BranchRecord:
|
||||
name: str
|
||||
verdict: str # delete | keep
|
||||
reason: str
|
||||
|
||||
|
||||
def _git(args: list, cwd: str, timeout: int = 15) -> subprocess.CompletedProcess:
|
||||
"""Run git, translating timeouts into a nonzero returncode.
|
||||
|
||||
Every verdict in this module fails safe toward "keep" on a nonzero
|
||||
returncode, so a hung/slow git call (large repos make ``git cherry``
|
||||
genuinely slow) must degrade to keep — never crash the whole audit
|
||||
(live-verified failure on a 746MB .git: TimeoutExpired escaped and
|
||||
aborted the branch audit mid-list).
|
||||
"""
|
||||
try:
|
||||
return subprocess.run(
|
||||
["git", *args],
|
||||
capture_output=True, text=True, encoding="utf-8",
|
||||
errors="replace", timeout=timeout, cwd=cwd,
|
||||
)
|
||||
except subprocess.TimeoutExpired:
|
||||
return subprocess.CompletedProcess(
|
||||
args=["git", *args], returncode=124,
|
||||
stdout="", stderr=f"timeout after {timeout}s",
|
||||
)
|
||||
|
||||
|
||||
def _tree_size_mb(path: Path) -> Optional[int]:
|
||||
"""Cheap directory size via ``du -sm`` — best-effort, None on failure."""
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["du", "-sm", str(path)],
|
||||
capture_output=True, text=True, encoding="utf-8",
|
||||
errors="replace", timeout=30,
|
||||
)
|
||||
if result.returncode == 0 and result.stdout.strip():
|
||||
return int(result.stdout.split()[0])
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def _dirty_split(path: str) -> tuple[bool, List[str]]:
|
||||
"""Return (has_tracked_modifications, untracked_paths).
|
||||
|
||||
``git status --porcelain`` counts untracked scratch equally with real
|
||||
edits; the reclaim policy treats them very differently (tracked = real
|
||||
work, untracked = archivable scratch), so split here.
|
||||
"""
|
||||
try:
|
||||
result = _git(["status", "--porcelain"], cwd=path, timeout=10)
|
||||
if result.returncode != 0:
|
||||
return True, [] # fail safe: treat as real work
|
||||
tracked = False
|
||||
untracked: List[str] = []
|
||||
for line in result.stdout.splitlines():
|
||||
if not line.strip():
|
||||
continue
|
||||
if line.startswith("??"):
|
||||
untracked.append(line[3:].strip())
|
||||
else:
|
||||
tracked = True
|
||||
return tracked, untracked
|
||||
except Exception:
|
||||
return True, []
|
||||
|
||||
|
||||
def _archive_untracked(tree: Path, untracked: List[str]) -> Optional[Path]:
|
||||
"""Copy untracked files out of a doomed tree. Returns the archive dir.
|
||||
|
||||
Never destroys: on any copy failure the caller must treat the tree as
|
||||
keep. Costs almost nothing and removes the "did I just delete
|
||||
something?" question.
|
||||
"""
|
||||
stamp = time.strftime("%Y%m%d-%H%M%S")
|
||||
dest = (
|
||||
Path.home() / ".hermes" / "archive" / "worktree-prune"
|
||||
/ f"{tree.name}-{stamp}"
|
||||
)
|
||||
try:
|
||||
for rel in untracked:
|
||||
src = tree / rel
|
||||
if not src.exists() or src.is_symlink():
|
||||
continue
|
||||
target = dest / rel
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
if src.is_dir():
|
||||
shutil.copytree(src, target, dirs_exist_ok=True)
|
||||
else:
|
||||
shutil.copy2(src, target)
|
||||
return dest if dest.exists() else None
|
||||
except Exception as exc:
|
||||
logger.warning("Could not archive untracked files from %s: %s", tree, exc)
|
||||
return None
|
||||
|
||||
|
||||
def audit_worktrees(repo_root: str, *, with_sizes: bool = True) -> List[TreeRecord]:
|
||||
"""Classify every tree under ``.worktrees/`` without mutating anything."""
|
||||
import cli as _cli # lazy: cli.py is heavy
|
||||
|
||||
worktrees_dir = Path(repo_root) / ".worktrees"
|
||||
if not worktrees_dir.exists():
|
||||
return []
|
||||
|
||||
if _cli._repo_is_shallow(repo_root):
|
||||
_cli._deepen_shallow_repo(repo_root)
|
||||
|
||||
merge_cache = _cli._load_worktree_merge_cache()
|
||||
cache_size_before = len(merge_cache)
|
||||
|
||||
now = time.time()
|
||||
records: List[TreeRecord] = []
|
||||
for entry in sorted(worktrees_dir.iterdir()):
|
||||
if not entry.is_dir():
|
||||
continue
|
||||
try:
|
||||
age_days = (now - entry.stat().st_mtime) / 86400.0
|
||||
except Exception:
|
||||
continue
|
||||
size_mb = _tree_size_mb(entry) if with_sizes else None
|
||||
|
||||
try:
|
||||
branch_result = _git(["branch", "--show-current"], cwd=str(entry), timeout=5)
|
||||
branch = branch_result.stdout.strip()
|
||||
except Exception:
|
||||
branch = ""
|
||||
|
||||
def rec(verdict: str, reason: str, untracked: Optional[List[str]] = None):
|
||||
records.append(TreeRecord(
|
||||
name=entry.name, path=str(entry), branch=branch,
|
||||
age_days=age_days, size_mb=size_mb,
|
||||
verdict=verdict, reason=reason,
|
||||
untracked=untracked or [],
|
||||
))
|
||||
|
||||
if _KANBAN_RE.match(entry.name):
|
||||
rec("keep", "kanban task tree (owned by kanban gc)")
|
||||
continue
|
||||
|
||||
lock_state = _cli._worktree_lock_is_live(repo_root, str(entry), timeout=5)
|
||||
if lock_state == "live":
|
||||
rec("keep", "in use by a running hermes session")
|
||||
continue
|
||||
|
||||
tracked_dirty, untracked = _dirty_split(str(entry))
|
||||
if tracked_dirty:
|
||||
rec("keep", "uncommitted tracked changes (real work)")
|
||||
continue
|
||||
|
||||
if _cli._worktree_has_unpushed_commits(str(entry), timeout=5):
|
||||
merged = _cli._worktree_commits_all_merged_upstream(
|
||||
str(entry), timeout=30, cache=merge_cache,
|
||||
max_ahead=_MAX_CHERRY_AHEAD,
|
||||
)
|
||||
if not merged:
|
||||
rec("keep", "unpushed commits not found upstream")
|
||||
continue
|
||||
|
||||
if untracked:
|
||||
rec("reap-archive",
|
||||
f"merged/pushed; {len(untracked)} untracked file(s) will be archived",
|
||||
untracked)
|
||||
else:
|
||||
rec("reap", "clean and fully merged/pushed")
|
||||
|
||||
if len(merge_cache) != cache_size_before:
|
||||
_cli._save_worktree_merge_cache(merge_cache)
|
||||
return records
|
||||
|
||||
|
||||
def reclaim_worktrees(
|
||||
repo_root: str,
|
||||
*,
|
||||
dry_run: bool = False,
|
||||
records: Optional[List[TreeRecord]] = None,
|
||||
) -> List[str]:
|
||||
"""Remove every reap-verdict tree from a frozen audit list.
|
||||
|
||||
Operates ONLY on the provided (or freshly computed) audit records — never
|
||||
re-globs inside the destructive loop, so trees created by concurrent
|
||||
sessions after the audit are out of scope by construction.
|
||||
"""
|
||||
if records is None:
|
||||
records = audit_worktrees(repo_root, with_sizes=False)
|
||||
actions: List[str] = []
|
||||
for record in records:
|
||||
if record.verdict not in {"reap", "reap-archive"}:
|
||||
continue
|
||||
if dry_run:
|
||||
actions.append(f"would remove {record.name} ({record.reason})")
|
||||
continue
|
||||
|
||||
entry = Path(record.path)
|
||||
if record.verdict == "reap-archive" and record.untracked:
|
||||
archive = _archive_untracked(entry, record.untracked)
|
||||
if archive is None:
|
||||
actions.append(f"kept {record.name} (archive of untracked files failed)")
|
||||
continue
|
||||
actions.append(f"archived {len(record.untracked)} untracked file(s) → {archive}")
|
||||
|
||||
# Dead-pid locks must be unlocked or `remove --force` refuses.
|
||||
try:
|
||||
_git(["worktree", "unlock", record.path], cwd=repo_root, timeout=10)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
try:
|
||||
remove_result = _git(
|
||||
["worktree", "remove", record.path, "--force"],
|
||||
cwd=repo_root, timeout=30,
|
||||
)
|
||||
if remove_result.returncode != 0:
|
||||
actions.append(
|
||||
f"failed to remove {record.name}: {remove_result.stderr.strip()}"
|
||||
)
|
||||
continue
|
||||
if record.branch and record.branch not in _PROTECTED_BRANCHES:
|
||||
_git(["branch", "-D", record.branch], cwd=repo_root, timeout=10)
|
||||
actions.append(f"removed {record.name}")
|
||||
except Exception as exc:
|
||||
actions.append(f"failed to remove {record.name}: {exc}")
|
||||
|
||||
if not dry_run:
|
||||
try:
|
||||
_git(["worktree", "prune"], cwd=repo_root, timeout=15)
|
||||
except Exception:
|
||||
pass
|
||||
return actions
|
||||
|
||||
|
||||
def audit_branches(repo_root: str) -> List[BranchRecord]:
|
||||
"""Classify local branches: safe to delete when their content is on
|
||||
upstream (fully merged OR every commit patch-equivalent via ``git
|
||||
cherry``) and they are not checked out anywhere.
|
||||
|
||||
Generalizes the startup pass's prefix list (``hermes/hermes-*``/``pr-*``)
|
||||
to EVERY local branch, because deletion is gated on content reachability
|
||||
rather than name: a branch whose commits are all upstream loses nothing
|
||||
when its ref goes. Branch names checked out in any worktree, protected
|
||||
names, and branches with unique commits are kept.
|
||||
"""
|
||||
import cli as _cli
|
||||
|
||||
if _cli._repo_is_shallow(repo_root):
|
||||
_cli._deepen_shallow_repo(repo_root)
|
||||
|
||||
upstream = None
|
||||
for candidate in ("origin/HEAD", "origin/main", "origin/master"):
|
||||
probe = _git(["rev-parse", "--verify", "--quiet", candidate], cwd=repo_root, timeout=5)
|
||||
if probe.returncode == 0:
|
||||
upstream = candidate
|
||||
break
|
||||
if upstream is None:
|
||||
return []
|
||||
|
||||
result = _git(["branch", "--format=%(refname:short)"], cwd=repo_root, timeout=10)
|
||||
if result.returncode != 0:
|
||||
return []
|
||||
branches = [b.strip() for b in result.stdout.splitlines() if b.strip()]
|
||||
|
||||
active: set = set()
|
||||
wt = _git(["worktree", "list", "--porcelain"], cwd=repo_root, timeout=10)
|
||||
for line in wt.stdout.splitlines():
|
||||
if line.startswith("branch refs/heads/"):
|
||||
active.add(line.split("branch refs/heads/", 1)[-1].strip())
|
||||
|
||||
merged_result = _git(["branch", "--merged", upstream, "--format=%(refname:short)"],
|
||||
cwd=repo_root, timeout=15)
|
||||
merged = {b.strip() for b in merged_result.stdout.splitlines() if b.strip()}
|
||||
|
||||
def _classify_branch(branch: str) -> BranchRecord:
|
||||
if branch in _PROTECTED_BRANCHES or branch in active:
|
||||
return BranchRecord(branch, "keep", "protected or checked out")
|
||||
if branch in merged:
|
||||
return BranchRecord(branch, "delete", "fully merged into " + upstream)
|
||||
# Rebase merges rewrite SHAs, so --merged misses them; cherry
|
||||
# patch-equivalence catches the dominant leak. Bounded: a branch
|
||||
# far ahead is a stale-base lane, keep it.
|
||||
ahead = _git(["rev-list", "--count", f"{upstream}..{branch}"], cwd=repo_root, timeout=10)
|
||||
try:
|
||||
ahead_count = int(ahead.stdout.strip() or "0")
|
||||
except ValueError:
|
||||
ahead_count = _MAX_CHERRY_AHEAD + 1
|
||||
if ahead_count == 0:
|
||||
return BranchRecord(branch, "delete", "no commits beyond " + upstream)
|
||||
if ahead_count > _MAX_CHERRY_AHEAD:
|
||||
return BranchRecord(branch, "keep", f"{ahead_count} commits ahead (stale-base lane)")
|
||||
cherry = _git(["cherry", upstream, branch], cwd=repo_root, timeout=30)
|
||||
if cherry.returncode != 0:
|
||||
return BranchRecord(branch, "keep", "could not verify (git cherry failed)")
|
||||
lines = [ln for ln in cherry.stdout.splitlines() if ln.strip()]
|
||||
if lines and all(ln.startswith("-") for ln in lines):
|
||||
return BranchRecord(branch, "delete", "all commits patch-equivalent upstream")
|
||||
unique = sum(1 for ln in lines if ln.startswith("+"))
|
||||
return BranchRecord(branch, "keep", f"{unique} unique commit(s) not upstream")
|
||||
|
||||
# Read-only classification — parallel, like the tree audit (a busy
|
||||
# multi-agent box carries hundreds of local branches; serial cherry
|
||||
# probes at ~0.2-1s each make the audit minutes long).
|
||||
import concurrent.futures
|
||||
|
||||
workers = max(1, min(8, (os.cpu_count() or 4), len(branches)))
|
||||
if workers > 1:
|
||||
try:
|
||||
with concurrent.futures.ThreadPoolExecutor(
|
||||
max_workers=workers, thread_name_prefix="hermes-branch-gc"
|
||||
) as pool:
|
||||
return list(pool.map(_classify_branch, branches))
|
||||
except Exception:
|
||||
pass
|
||||
return [_classify_branch(b) for b in branches]
|
||||
|
||||
|
||||
def reclaim_branches(
|
||||
repo_root: str,
|
||||
*,
|
||||
dry_run: bool = False,
|
||||
records: Optional[List[BranchRecord]] = None,
|
||||
) -> List[str]:
|
||||
"""Delete every delete-verdict branch from a frozen audit list."""
|
||||
if records is None:
|
||||
records = audit_branches(repo_root)
|
||||
actions: List[str] = []
|
||||
for record in records:
|
||||
if record.verdict != "delete":
|
||||
continue
|
||||
if dry_run:
|
||||
actions.append(f"would delete branch {record.name} ({record.reason})")
|
||||
continue
|
||||
result = _git(["branch", "-D", record.name], cwd=repo_root, timeout=10)
|
||||
if result.returncode == 0:
|
||||
actions.append(f"deleted branch {record.name}")
|
||||
else:
|
||||
actions.append(f"failed to delete {record.name}: {result.stderr.strip()}")
|
||||
return actions
|
||||
|
||||
|
||||
def worktrees_summary(repo_root: str) -> tuple[int, Optional[int]]:
|
||||
"""(tree_count, total_size_mb) for the escalation notice. Size is
|
||||
best-effort with a hard timeout so the startup path never stalls."""
|
||||
worktrees_dir = Path(repo_root) / ".worktrees"
|
||||
if not worktrees_dir.exists():
|
||||
return 0, None
|
||||
try:
|
||||
count = sum(1 for e in worktrees_dir.iterdir() if e.is_dir())
|
||||
except Exception:
|
||||
return 0, None
|
||||
size_mb: Optional[int] = None
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["du", "-sm", str(worktrees_dir)],
|
||||
capture_output=True, text=True, encoding="utf-8",
|
||||
errors="replace", timeout=20,
|
||||
)
|
||||
if result.returncode == 0 and result.stdout.strip():
|
||||
size_mb = int(result.stdout.split()[0])
|
||||
except Exception:
|
||||
pass
|
||||
return count, size_mb
|
||||
+41
-4
@@ -1042,6 +1042,26 @@ class AIAgent:
|
||||
)
|
||||
)
|
||||
|
||||
def _warn_uncompressed_context_overflow(
|
||||
self, preflight_tokens: int, context_length: int
|
||||
) -> None:
|
||||
"""Surface a deduped warning when uncompressed context exceeds model limit.
|
||||
|
||||
When compression is explicitly disabled (compression.enabled: false), long
|
||||
sessions can grow past the model context window with no compression to shrink
|
||||
them (#89297). Surface an actionable warning so the user knows to run /compact
|
||||
or enable compression.
|
||||
"""
|
||||
_warn_key = ("uncompressed_ctx_overflow", context_length)
|
||||
if getattr(self, "_last_ctx_overflow_warn", None) != _warn_key:
|
||||
self._last_ctx_overflow_warn = _warn_key
|
||||
self._emit_warning(
|
||||
f"⚠️ Session context (~{preflight_tokens:,} tokens) exceeds the model "
|
||||
f"context window (~{context_length:,} tokens) with compression disabled "
|
||||
f"(compression.enabled: false). Use /compact to compress history or "
|
||||
f"enable compression in config.yaml."
|
||||
)
|
||||
|
||||
def _clear_context_overflow_warn(self) -> None:
|
||||
"""Reset the dedup state for the blocked-overflow warning.
|
||||
|
||||
@@ -8219,6 +8239,7 @@ class AIAgent:
|
||||
function_result: str,
|
||||
*,
|
||||
failed: bool,
|
||||
tool_call_id: str = "",
|
||||
) -> str:
|
||||
decision = self._tool_guardrails.after_call(
|
||||
tool_name,
|
||||
@@ -8226,20 +8247,36 @@ class AIAgent:
|
||||
function_result,
|
||||
failed=failed,
|
||||
)
|
||||
# Identical-call loop breaker (agent.stall_guards): notice-only, no
|
||||
# Identical-call stall guards (agent.stall_guards): notice-only, no
|
||||
# blocking. Observed on the RAW result (before the loop-warning suffix
|
||||
# below, whose embedded count changes per call and would defeat
|
||||
# result-identity matching). Appended here — at result construction,
|
||||
# result-identity matching). Applied here — at result construction,
|
||||
# before the tool message is built — so it is cache-safe (tool results
|
||||
# are append-only; nothing already sent to the provider is mutated).
|
||||
stall_notice = None
|
||||
result_stub = None
|
||||
if self._stall_guards_enabled():
|
||||
try:
|
||||
stall_notice = self._tool_guardrails.observe_identical_call(
|
||||
tool_name, function_args, function_result,
|
||||
observation = self._tool_guardrails.observe_call(
|
||||
tool_name,
|
||||
function_args,
|
||||
function_result if isinstance(function_result, str) else None,
|
||||
tool_call_id=tool_call_id,
|
||||
failed=failed,
|
||||
)
|
||||
stall_notice = observation.notice
|
||||
result_stub = observation.stub
|
||||
except Exception as exc:
|
||||
logger.debug("stall-guard identical-call observation failed: %s", exc)
|
||||
# Result-reference stubbing: a 2nd+ consecutive identical call whose
|
||||
# FRESH result is byte-identical enters context as a short reference
|
||||
# stub instead of the duplicate payload. The tool still executed —
|
||||
# this is not a cache; a changed result flows through whole. Only
|
||||
# plain-string results are stubbed (multimodal content lists pass
|
||||
# through untouched), and the current message keeps its role and
|
||||
# tool_call_id — only the content is replaced.
|
||||
if result_stub and isinstance(function_result, str):
|
||||
function_result = result_stub
|
||||
if decision.action in {"warn", "halt"}:
|
||||
function_result = append_toolguard_guidance(function_result, decision)
|
||||
if decision.should_halt:
|
||||
|
||||
@@ -203,3 +203,47 @@ class TestMinimumMessagesBranch:
|
||||
assert len(out) == len(msgs), "nothing should have been compressed"
|
||||
assert cc._last_compression_made_progress is False
|
||||
assert cc._ineffective_compression_count == before + 1
|
||||
|
||||
|
||||
class TestRejectedCompactionStrike:
|
||||
"""#88568 — a would-grow refusal must count as an ineffective strike.
|
||||
|
||||
The anti-growth guard correctly keeps the original transcript, but the
|
||||
rejection used to leave ``_ineffective_compression_count`` untouched, so
|
||||
the breaker never latched and automatic compression retried the SAME
|
||||
unchanged transcript on every turn.
|
||||
"""
|
||||
|
||||
def test_rejected_compaction_increments_strike(self):
|
||||
cc = _compressor(threshold_tokens=1)
|
||||
assert cc._ineffective_compression_count == 0
|
||||
|
||||
cc.record_rejected_compaction()
|
||||
|
||||
assert cc._ineffective_compression_count == 1
|
||||
|
||||
def test_two_rejections_stop_further_automatic_compression(self):
|
||||
cc = _compressor(threshold_tokens=1)
|
||||
cc.record_rejected_compaction()
|
||||
cc.record_rejected_compaction()
|
||||
|
||||
# The latch consumers key on the counter itself (>= 2 blocks);
|
||||
# pin the counter and the recovery-clock arming side effect.
|
||||
assert cc._ineffective_compression_count >= 2
|
||||
|
||||
def test_rejection_does_not_arm_real_usage_verification(self):
|
||||
"""Nothing was committed, so the next response must not be scored
|
||||
against the pre-rejection transcript (that verdict belongs to
|
||||
committed compactions only)."""
|
||||
cc = _compressor(threshold_tokens=1)
|
||||
cc.record_rejected_compaction()
|
||||
|
||||
assert cc._verify_compaction_cleared_threshold is False
|
||||
|
||||
def test_rejection_leaves_fallback_streak_untouched(self):
|
||||
cc = _compressor(threshold_tokens=1)
|
||||
cc._fallback_compression_streak = 1
|
||||
|
||||
cc.record_rejected_compaction()
|
||||
|
||||
assert cc._fallback_compression_streak == 1
|
||||
|
||||
@@ -0,0 +1,181 @@
|
||||
"""Mechanical salvage for compression candidates that would grow."""
|
||||
|
||||
from agent.context_compressor import (
|
||||
COMPRESSED_SUMMARY_METADATA_KEY,
|
||||
_SUMMARY_END_MARKER,
|
||||
salvage_grown_transcript,
|
||||
)
|
||||
from agent.model_metadata import estimate_messages_tokens_rough
|
||||
|
||||
|
||||
def test_salvage_stubs_old_tools_and_keeps_todo_when_stubbing_suffices():
|
||||
"""Tool stubbing alone gets under budget → the todo snapshot survives.
|
||||
|
||||
The snapshot is the only in-transcript todo re-injection at the boundary
|
||||
(and may carry the pruned-skill reload notice), so it is last-resort only.
|
||||
"""
|
||||
original = [
|
||||
{"role": "user", "content": "go"},
|
||||
{"role": "assistant", "content": "ok"},
|
||||
{"role": "tool", "tool_call_id": "a", "content": "A" * 4000},
|
||||
{"role": "tool", "tool_call_id": "b", "content": "B" * 4000},
|
||||
{"role": "tool", "tool_call_id": "c", "content": "keep-latest"},
|
||||
]
|
||||
grown = original + [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Current todos:\n- [ ] x",
|
||||
"_todo_snapshot_synthetic": True,
|
||||
}
|
||||
]
|
||||
assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original)
|
||||
|
||||
out = salvage_grown_transcript(original, grown)
|
||||
|
||||
assert out is not None
|
||||
assert estimate_messages_tokens_rough(out) < estimate_messages_tokens_rough(original)
|
||||
assert any(m.get("_todo_snapshot_synthetic") for m in out)
|
||||
tools = [m["content"] for m in out if m.get("role") == "tool"]
|
||||
assert tools[-1] == "keep-latest"
|
||||
assert any("cleared to save context space" in t for t in tools)
|
||||
|
||||
|
||||
def test_salvage_drops_todo_only_as_last_resort():
|
||||
"""When cheaper ops cannot get under budget, the snapshot is dropped."""
|
||||
original = [
|
||||
{"role": "user", "content": "please do the thing " + ("o" * 600)},
|
||||
{"role": "assistant", "content": "ok"},
|
||||
]
|
||||
grown = [
|
||||
{"role": "user", "content": "summary of the ask"},
|
||||
{"role": "assistant", "content": "ok"},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Current todos:\n- [ ] " + ("t" * 800),
|
||||
"_todo_snapshot_synthetic": True,
|
||||
},
|
||||
]
|
||||
assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original)
|
||||
|
||||
out = salvage_grown_transcript(original, grown)
|
||||
|
||||
assert out is not None
|
||||
assert estimate_messages_tokens_rough(out) < estimate_messages_tokens_rough(original)
|
||||
assert not any(m.get("_todo_snapshot_synthetic") for m in out)
|
||||
|
||||
|
||||
def test_salvage_last_resort_preserves_pruned_skill_reload_notice():
|
||||
"""7a16840add couples the reload notice into the snapshot — it survives."""
|
||||
from agent.conversation_compression import _PRUNED_SKILL_RELOAD_NOTICE_HEADER
|
||||
|
||||
notice = (
|
||||
f"{_PRUNED_SKILL_RELOAD_NOTICE_HEADER}\n"
|
||||
"Reload with skill_view(name='example-skill') before acting."
|
||||
)
|
||||
original = [
|
||||
{"role": "user", "content": "please do the thing " + ("o" * 3000)},
|
||||
{"role": "assistant", "content": "ok"},
|
||||
]
|
||||
grown = [
|
||||
{"role": "user", "content": "summary of the ask"},
|
||||
{"role": "assistant", "content": "ok"},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Current todos:\n- [ ] " + ("t" * 4000) + f"\n\n{notice}",
|
||||
"_todo_snapshot_synthetic": True,
|
||||
},
|
||||
]
|
||||
assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original)
|
||||
|
||||
out = salvage_grown_transcript(original, grown)
|
||||
|
||||
assert out is not None
|
||||
assert estimate_messages_tokens_rough(out) < estimate_messages_tokens_rough(original)
|
||||
snapshot_rows = [m for m in out if m.get("_todo_snapshot_synthetic")]
|
||||
assert len(snapshot_rows) == 1
|
||||
assert snapshot_rows[0]["content"].startswith(_PRUNED_SKILL_RELOAD_NOTICE_HEADER)
|
||||
assert "Current todos" not in snapshot_rows[0]["content"]
|
||||
|
||||
|
||||
def test_salvage_returns_none_when_nothing_can_shrink():
|
||||
original = [{"role": "user", "content": "tiny"}]
|
||||
huge = [{"role": "user", "content": "X" * 200_000}]
|
||||
|
||||
assert salvage_grown_transcript(original, huge) is None
|
||||
|
||||
|
||||
def test_salvage_caps_oversized_summary():
|
||||
original = [
|
||||
{"role": "user", "content": "ask " + ("o" * 12_000)},
|
||||
{"role": "assistant", "content": "short reply"},
|
||||
]
|
||||
grown = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": (
|
||||
"[CONTEXT COMPACTION] "
|
||||
+ ("S" * 20_000)
|
||||
+ "\n\n"
|
||||
+ _SUMMARY_END_MARKER
|
||||
),
|
||||
COMPRESSED_SUMMARY_METADATA_KEY: True,
|
||||
},
|
||||
{"role": "assistant", "content": "short reply"},
|
||||
]
|
||||
assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original)
|
||||
|
||||
out = salvage_grown_transcript(original, grown)
|
||||
|
||||
assert out is not None
|
||||
assert len(out[0]["content"]) < 12_000
|
||||
assert "truncated so compaction can shrink" in out[0]["content"]
|
||||
assert out[0]["content"].endswith(_SUMMARY_END_MARKER)
|
||||
|
||||
|
||||
def test_salvage_never_truncates_merged_summary_with_live_user_tail():
|
||||
original = [{"role": "user", "content": "O" * 12_000}]
|
||||
merged = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": (
|
||||
"[CONTEXT COMPACTION] "
|
||||
+ ("S" * 20_000)
|
||||
+ "\n\n"
|
||||
+ _SUMMARY_END_MARKER
|
||||
+ "\n\nLIVE USER REQUEST"
|
||||
),
|
||||
COMPRESSED_SUMMARY_METADATA_KEY: True,
|
||||
}
|
||||
]
|
||||
|
||||
assert salvage_grown_transcript(original, merged) is None
|
||||
assert merged[0]["content"].endswith("LIVE USER REQUEST")
|
||||
|
||||
|
||||
def test_salvage_does_not_cap_plain_user_text_quoting_summary_marker():
|
||||
original = [{"role": "user", "content": "O" * 20_000}]
|
||||
quoted = "ordinary user text " + ("Q" * 12_000) + "\n\n" + _SUMMARY_END_MARKER
|
||||
candidate = [{"role": "user", "content": quoted}]
|
||||
|
||||
out = salvage_grown_transcript(original, candidate)
|
||||
|
||||
assert out is not None
|
||||
assert out[0]["content"] == quoted
|
||||
assert "truncated so compaction can shrink" not in out[0]["content"]
|
||||
|
||||
|
||||
def test_salvage_never_caps_unmarked_summary_shaped_live_user_text():
|
||||
original = [{"role": "user", "content": "O" * 20_000}]
|
||||
live_user_text = (
|
||||
"[CONTEXT COMPACTION] "
|
||||
+ ("U" * 12_000)
|
||||
+ "\n\n"
|
||||
+ _SUMMARY_END_MARKER
|
||||
)
|
||||
candidate = [{"role": "user", "content": live_user_text}]
|
||||
|
||||
out = salvage_grown_transcript(original, candidate)
|
||||
|
||||
assert out is not None
|
||||
assert out[0]["content"] == live_user_text
|
||||
assert "truncated so compaction can shrink" not in out[0]["content"]
|
||||
@@ -17,6 +17,7 @@ These assert behavior contracts, not message snapshots.
|
||||
|
||||
from agent.agent_runtime_helpers import trailing_continue_intent
|
||||
from agent.tool_guardrails import (
|
||||
IDENTICAL_RESULT_STUB_MIN_CHARS,
|
||||
STALL_GUARD_IDENTICAL_CALL_THRESHOLD,
|
||||
STALL_GUARD_REPEATABLE_TOOLS,
|
||||
ToolCallGuardrailController,
|
||||
@@ -142,8 +143,10 @@ def _fake_agent(stall_guards=True):
|
||||
lambda decision: AIAgent._set_tool_guardrail_halt(agent, decision)
|
||||
)
|
||||
agent._append = (
|
||||
lambda name, args, result: AIAgent._append_guardrail_observation(
|
||||
agent, name, args, result, failed=False
|
||||
lambda name, args, result, failed=False, tool_call_id="": (
|
||||
AIAgent._append_guardrail_observation(
|
||||
agent, name, args, result, failed=failed, tool_call_id=tool_call_id
|
||||
)
|
||||
)
|
||||
)
|
||||
return agent
|
||||
@@ -181,6 +184,170 @@ def test_notice_streak_keys_on_raw_result_not_annotated_result():
|
||||
assert "hermes note" in outs[2]
|
||||
|
||||
|
||||
# ── result-reference stubbing (byte-identical duplicate results) ──────────
|
||||
|
||||
|
||||
_BIG = "x" * IDENTICAL_RESULT_STUB_MIN_CHARS # exactly at the stub threshold
|
||||
|
||||
|
||||
def test_stub_on_second_identical_call_first_full():
|
||||
agent = _fake_agent()
|
||||
args = {"query": "hermes"}
|
||||
r1 = agent._append("web_search", args, _BIG, tool_call_id="call_1")
|
||||
r2 = agent._append("web_search", args, _BIG, tool_call_id="call_2")
|
||||
assert r1 == _BIG # first occurrence always enters context whole
|
||||
assert r2 != _BIG
|
||||
assert "byte-identical" in r2
|
||||
assert "web_search" in r2
|
||||
assert "call_1" in r2 # references the FIRST occurrence in the streak
|
||||
assert len(r2) < len(_BIG)
|
||||
|
||||
|
||||
def test_no_stub_when_fresh_result_differs():
|
||||
agent = _fake_agent()
|
||||
args = {"query": "hermes"}
|
||||
agent._append("web_search", args, _BIG, tool_call_id="call_1")
|
||||
changed = "y" + _BIG
|
||||
r2 = agent._append("web_search", args, changed, tool_call_id="call_2")
|
||||
assert r2 == changed # changed result flows through whole
|
||||
|
||||
|
||||
def test_changed_result_resets_streak_then_stub_references_new_first():
|
||||
agent = _fake_agent()
|
||||
args = {"id": "job"}
|
||||
agent._append("web_search", args, _BIG, tool_call_id="a")
|
||||
changed = _BIG + "done"
|
||||
r2 = agent._append("web_search", args, changed, tool_call_id="b")
|
||||
assert r2 == changed
|
||||
r3 = agent._append("web_search", args, changed, tool_call_id="c")
|
||||
assert "byte-identical" in r3
|
||||
assert "tool_call_id b" in r3 # new streak's first occurrence, not 'a'
|
||||
|
||||
|
||||
def test_no_stub_below_min_chars():
|
||||
agent = _fake_agent()
|
||||
small = "x" * (IDENTICAL_RESULT_STUB_MIN_CHARS - 1)
|
||||
args = {"query": "hermes"}
|
||||
agent._append("web_search", args, small, tool_call_id="c1")
|
||||
r2 = agent._append("web_search", args, small, tool_call_id="c2")
|
||||
assert "byte-identical" not in r2
|
||||
assert r2.startswith(small) # full payload kept (pre-existing warning suffix allowed)
|
||||
|
||||
|
||||
def test_no_stub_for_error_results():
|
||||
agent = _fake_agent()
|
||||
err = "Error executing tool: " + _BIG
|
||||
args = {"command": "boom"}
|
||||
agent._append("terminal", args, err, failed=True, tool_call_id="c1")
|
||||
r2 = agent._append("terminal", args, err, failed=True, tool_call_id="c2")
|
||||
assert "byte-identical" not in r2
|
||||
assert r2.startswith(err) # models must see fresh errors whole
|
||||
|
||||
|
||||
def test_pollers_get_stub_but_never_loop_notice():
|
||||
agent = _fake_agent()
|
||||
args = {"id": "job1"}
|
||||
results = [
|
||||
agent._append("bfl_flux3_get_result", args, _BIG, tool_call_id=f"c{i}")
|
||||
for i in range(4)
|
||||
]
|
||||
assert results[0] == _BIG
|
||||
for r in results[1:]:
|
||||
assert "byte-identical" in r # stubbed: unchanged poll saves context
|
||||
assert "consecutive identical call" not in r # notice stays exempt
|
||||
|
||||
|
||||
def test_third_identical_call_gets_stub_plus_loop_notice():
|
||||
agent = _fake_agent()
|
||||
args = {"query": "hermes"}
|
||||
agent._append("web_search", args, _BIG, tool_call_id="c1")
|
||||
agent._append("web_search", args, _BIG, tool_call_id="c2")
|
||||
r3 = agent._append("web_search", args, _BIG, tool_call_id="c3")
|
||||
assert "byte-identical" in r3 # stub replaces the payload
|
||||
assert "3rd consecutive identical call" in r3 # notice appended after it
|
||||
assert r3.index("byte-identical") < r3.index("3rd consecutive")
|
||||
|
||||
|
||||
def test_stub_carries_spillover_path_when_first_result_persisted():
|
||||
agent = _fake_agent()
|
||||
args = {"query": "big"}
|
||||
agent._tool_guardrails.record_persisted_result(
|
||||
"c1", "/home/u/.hermes/cache/spillover/c1.txt"
|
||||
)
|
||||
agent._append("web_search", args, _BIG, tool_call_id="c1")
|
||||
r2 = agent._append("web_search", args, _BIG, tool_call_id="c2")
|
||||
assert "/home/u/.hermes/cache/spillover/c1.txt" in r2
|
||||
|
||||
|
||||
def test_stub_includes_args_summary_for_compression_safety():
|
||||
agent = _fake_agent()
|
||||
args = {"query": "hermes result stubbing", "limit": 5}
|
||||
agent._append("web_search", args, _BIG, tool_call_id="c1")
|
||||
r2 = agent._append("web_search", args, _BIG, tool_call_id="c2")
|
||||
# Canonical-args preview so the model knows WHAT the call was even if
|
||||
# the referenced message is later evicted by compression.
|
||||
assert "hermes result stubbing" in r2
|
||||
|
||||
|
||||
def test_stub_args_summary_truncated_to_120_chars():
|
||||
c = ToolCallGuardrailController()
|
||||
args = {"query": "q" * 500}
|
||||
assert c.observe_call("web_search", args, _BIG, tool_call_id="c1").stub is None
|
||||
stub = c.observe_call("web_search", args, _BIG, tool_call_id="c2").stub
|
||||
assert stub is not None
|
||||
args_part = stub.split("Args: ", 1)[1]
|
||||
assert len(args_part) < 200 # ~120-char preview + ellipsis + closer
|
||||
|
||||
|
||||
def test_config_off_disables_stub():
|
||||
agent = _fake_agent(stall_guards=False)
|
||||
args = {"query": "hermes"}
|
||||
agent._append("web_search", args, _BIG, tool_call_id="c1")
|
||||
r2 = agent._append("web_search", args, _BIG, tool_call_id="c2")
|
||||
assert "byte-identical" not in r2
|
||||
assert r2.startswith(_BIG)
|
||||
|
||||
|
||||
def test_streak_reset_by_different_call_means_next_identical_is_full():
|
||||
agent = _fake_agent()
|
||||
args = {"query": "hermes"}
|
||||
agent._append("web_search", args, _BIG, tool_call_id="c1")
|
||||
agent._append("read_file", {"path": "/a"}, "other", tool_call_id="c2")
|
||||
r3 = agent._append("web_search", args, _BIG, tool_call_id="c3")
|
||||
assert "byte-identical" not in r3
|
||||
assert r3.startswith(_BIG) # fresh streak — first occurrence full again
|
||||
|
||||
|
||||
def test_multimodal_content_never_stubbed_and_breaks_streak():
|
||||
c = ToolCallGuardrailController()
|
||||
args = {"path": "/img.png"}
|
||||
assert c.observe_call("vision", args, _BIG, tool_call_id="c1").stub is None
|
||||
# Non-string (multimodal) results never form or extend a streak.
|
||||
obs = c.observe_call("vision", args, None, tool_call_id="c2")
|
||||
assert obs.stub is None
|
||||
assert c.observe_call("vision", args, _BIG, tool_call_id="c3").stub is None
|
||||
|
||||
|
||||
def test_observe_identical_call_backcompat_notice_still_fires():
|
||||
c = ToolCallGuardrailController()
|
||||
for _ in range(STALL_GUARD_IDENTICAL_CALL_THRESHOLD - 1):
|
||||
assert c.observe_identical_call("web_search", {"q": 1}, "r") is None
|
||||
assert c.observe_identical_call("web_search", {"q": 1}, "r") is not None
|
||||
|
||||
|
||||
def test_extract_persisted_path_round_trip():
|
||||
# The stub's spillover reference is parsed from the <persisted-output>
|
||||
# block that maybe_persist_tool_result builds — assert the round trip.
|
||||
from tools.tool_result_storage import (
|
||||
_build_persisted_message,
|
||||
extract_persisted_path,
|
||||
)
|
||||
|
||||
block = _build_persisted_message("preview", True, 50_000, "/tmp/spill/x.txt")
|
||||
assert extract_persisted_path(block) == "/tmp/spill/x.txt"
|
||||
assert extract_persisted_path("plain result") is None
|
||||
|
||||
|
||||
# ── said-continue-but-stopped detector ─────────────────────────────────────
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,183 @@
|
||||
"""Uncompressed context overflow guardrail (#89297).
|
||||
|
||||
When compression is explicitly disabled (``compression.enabled: false``),
|
||||
sessions can grow past the model context window with nothing to shrink them.
|
||||
The conversation loop's pre-API site warns (deduped, actionable); the
|
||||
turn-context preflight re-arms the dedup once the session is back under the
|
||||
window so a later re-overflow warns again.
|
||||
|
||||
The fake binds the PRODUCTION ``_warn_uncompressed_context_overflow`` /
|
||||
``_clear_context_overflow_warn`` methods so their dedup logic is actually
|
||||
under test (not a reimplementation).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import types
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from agent.turn_context import TurnContext, build_turn_context # noqa: F401
|
||||
from run_agent import AIAgent
|
||||
from tests.agent.test_turn_context import _FakeAgent, _build
|
||||
|
||||
|
||||
class _FakeUncompressedAgent(_FakeAgent):
|
||||
"""Agent stub with compression disabled, bound to the REAL warn methods."""
|
||||
|
||||
# Production methods under test — bound from AIAgent so the dedup key
|
||||
# handling and message text cannot silently drift from what ships.
|
||||
_warn_uncompressed_context_overflow = (
|
||||
AIAgent._warn_uncompressed_context_overflow
|
||||
)
|
||||
_clear_context_overflow_warn = AIAgent._clear_context_overflow_warn
|
||||
|
||||
def __init__(self, model="deepseek-v4-flash", context_length=10_000):
|
||||
super().__init__()
|
||||
self.model = model
|
||||
self.provider = "deepseek"
|
||||
self.compression_enabled = False
|
||||
self.context_compressor = types.SimpleNamespace(
|
||||
protect_first_n=2,
|
||||
protect_last_n=2,
|
||||
context_length=context_length,
|
||||
threshold_tokens=int(context_length * 0.75),
|
||||
last_prompt_tokens=-1,
|
||||
)
|
||||
|
||||
|
||||
def _oversized_history(n_turns: int = 10) -> list:
|
||||
large_turn = "Large context content " * 500 # ~2,500 tokens each
|
||||
history = []
|
||||
for i in range(n_turns):
|
||||
history.append({"role": "user", "content": f"Turn {i}: {large_turn}"})
|
||||
history.append({"role": "assistant", "content": f"Reply {i}: {large_turn}"})
|
||||
return history
|
||||
|
||||
|
||||
def test_production_warn_emits_once_and_dedups():
|
||||
"""The real method warns once, then dedups identical overflows."""
|
||||
agent = _FakeUncompressedAgent(context_length=10_000)
|
||||
agent._emit_warning = MagicMock()
|
||||
|
||||
agent._warn_uncompressed_context_overflow(15_000, 10_000)
|
||||
agent._warn_uncompressed_context_overflow(16_000, 10_000)
|
||||
|
||||
agent._emit_warning.assert_called_once()
|
||||
msg = agent._emit_warning.call_args[0][0]
|
||||
assert "exceeds the model context window" in msg
|
||||
assert "compression.enabled: false" in msg
|
||||
assert "10,000 tokens" in msg
|
||||
|
||||
|
||||
def test_clear_rearms_the_warning():
|
||||
"""After _clear_context_overflow_warn (session back under the window),
|
||||
a later re-overflow warns again."""
|
||||
agent = _FakeUncompressedAgent(context_length=10_000)
|
||||
agent._emit_warning = MagicMock()
|
||||
|
||||
agent._warn_uncompressed_context_overflow(15_000, 10_000)
|
||||
agent._clear_context_overflow_warn()
|
||||
agent._warn_uncompressed_context_overflow(15_500, 10_000)
|
||||
|
||||
assert agent._emit_warning.call_count == 2
|
||||
|
||||
|
||||
def test_uncompressed_session_within_limits_emits_no_warning():
|
||||
agent = _FakeUncompressedAgent(context_length=128_000)
|
||||
agent._emit_warning = MagicMock()
|
||||
history = [
|
||||
{"role": "user", "content": "hello"},
|
||||
{"role": "assistant", "content": "hi there"},
|
||||
]
|
||||
tctx = _build(agent, conversation_history=history)
|
||||
assert isinstance(tctx, TurnContext)
|
||||
agent._emit_warning.assert_not_called()
|
||||
|
||||
|
||||
def test_preflight_rearm_clears_dedup_when_back_under_window():
|
||||
"""The turn-context preflight re-arms the dedup once the session fits
|
||||
again (e.g. after a manual /compress), so growth past the window later
|
||||
warns a second time."""
|
||||
agent = _FakeUncompressedAgent(context_length=128_000)
|
||||
agent._emit_warning = MagicMock()
|
||||
# Simulate a previously fired warning.
|
||||
agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 128_000)
|
||||
|
||||
history = [
|
||||
{"role": "user", "content": "hello"},
|
||||
{"role": "assistant", "content": "hi there"},
|
||||
]
|
||||
tctx = _build(agent, conversation_history=history)
|
||||
assert isinstance(tctx, TurnContext)
|
||||
|
||||
# Preflight cleared the dedup — the next overflow warns again.
|
||||
assert agent._last_ctx_overflow_warn is None
|
||||
agent._warn_uncompressed_context_overflow(200_000, 128_000)
|
||||
agent._emit_warning.assert_called_once()
|
||||
|
||||
|
||||
def test_preflight_does_not_rearm_while_still_over_window():
|
||||
"""While the session is still over the window, the dedup must survive
|
||||
the preflight (no per-turn warn spam)."""
|
||||
agent = _FakeUncompressedAgent(context_length=10_000)
|
||||
agent._emit_warning = MagicMock()
|
||||
agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 10_000)
|
||||
|
||||
tctx = _build(agent, conversation_history=_oversized_history())
|
||||
assert isinstance(tctx, TurnContext)
|
||||
|
||||
assert agent._last_ctx_overflow_warn == ("uncompressed_ctx_overflow", 10_000)
|
||||
agent._emit_warning.assert_not_called()
|
||||
|
||||
|
||||
def test_multimodal_content_forces_real_estimate_in_rearm_gate():
|
||||
"""List (multimodal) content defeats a char count; the pre-check must
|
||||
treat it as over-gate so the real estimator decides. A tiny multimodal
|
||||
session is still under the window, so the dedup is re-armed."""
|
||||
agent = _FakeUncompressedAgent(context_length=128_000)
|
||||
agent._emit_warning = MagicMock()
|
||||
agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 128_000)
|
||||
|
||||
history = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "look at this"},
|
||||
{"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}},
|
||||
],
|
||||
},
|
||||
{"role": "assistant", "content": "looking"},
|
||||
]
|
||||
tctx = _build(agent, conversation_history=history)
|
||||
assert isinstance(tctx, TurnContext)
|
||||
assert agent._last_ctx_overflow_warn is None
|
||||
|
||||
|
||||
def test_none_content_tool_call_rows_do_not_defeat_cheap_gate():
|
||||
"""Assistant tool-call rows routinely carry content=None; they must
|
||||
count as zero chars (NOT force the estimator) so the cheap gate keeps
|
||||
its value in ordinary tool-using sessions. Regression for the salvage
|
||||
follow-up's first draft, where `None` hit the over-gate branch."""
|
||||
from unittest.mock import patch as _patch
|
||||
|
||||
agent = _FakeUncompressedAgent(context_length=128_000)
|
||||
agent._emit_warning = MagicMock()
|
||||
agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 128_000)
|
||||
|
||||
history = [
|
||||
{"role": "user", "content": "run the tool"},
|
||||
{"role": "assistant", "content": None,
|
||||
"tool_calls": [{"id": "c1", "function": {"name": "t", "arguments": "{}"}}]},
|
||||
{"role": "tool", "tool_call_id": "c1", "content": "small result"},
|
||||
{"role": "assistant", "content": "done"},
|
||||
]
|
||||
with _patch(
|
||||
"agent.turn_context.estimate_request_tokens_rough"
|
||||
) as mock_est:
|
||||
tctx = _build(agent, conversation_history=history)
|
||||
|
||||
assert isinstance(tctx, TurnContext)
|
||||
# Cheap gate decided (tiny session, under window): estimator never ran,
|
||||
# and the dedup was still re-armed via the raw-chars branch.
|
||||
mock_est.assert_not_called()
|
||||
assert agent._last_ctx_overflow_warn is None
|
||||
@@ -153,6 +153,64 @@ def test_unknown_terminal_does_not_enable_extended_enter_keys():
|
||||
assert cli_mod._terminal_supports_extended_enter_keys({"TERM_PROGRAM": "unknown"}) is False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Ghostty: must push ONLY modifyOtherKeys, not the Kitty keyboard protocol —
|
||||
# see cli._is_ghostty_terminal for the full rationale (#87630).
|
||||
# ---------------------------------------------------------------------------
|
||||
class _FakeOutput:
|
||||
"""Minimal output object with write_raw + flush for _enable_extended_enter_keys."""
|
||||
def __init__(self):
|
||||
self.written = b""
|
||||
def write_raw(self, data):
|
||||
self.written += data.encode() if isinstance(data, str) else data
|
||||
def flush(self):
|
||||
pass
|
||||
|
||||
|
||||
def test_ghostty_uses_modify_other_keys_only():
|
||||
"""Ghostty must NOT push the Kitty keyboard protocol (CSI >1u)."""
|
||||
import cli as cli_mod
|
||||
|
||||
out = _FakeOutput()
|
||||
result = cli_mod._enable_extended_enter_keys(
|
||||
output=out,
|
||||
env={"TERM_PROGRAM": "ghostty", "TERM": "xterm-ghostty"},
|
||||
)
|
||||
assert result is True
|
||||
# Must contain modifyOtherKeys push ...
|
||||
assert b"\x1b[>4;2m" in out.written
|
||||
# ... but NOT the Kitty protocol push.
|
||||
assert b"\x1b[>1u" not in out.written
|
||||
|
||||
|
||||
def test_ghostty_via_term_var_uses_modify_other_keys_only():
|
||||
"""xterm-ghostty TERM (without TERM_PROGRAM) also skips Kitty protocol."""
|
||||
import cli as cli_mod
|
||||
|
||||
out = _FakeOutput()
|
||||
result = cli_mod._enable_extended_enter_keys(
|
||||
output=out,
|
||||
env={"TERM": "xterm-ghostty"},
|
||||
)
|
||||
assert result is True
|
||||
assert b"\x1b[>4;2m" in out.written
|
||||
assert b"\x1b[>1u" not in out.written
|
||||
|
||||
|
||||
def test_non_ghostty_terminals_still_push_kitty_protocol():
|
||||
"""iTerm2 and others still get the full dual-protocol push."""
|
||||
import cli as cli_mod
|
||||
|
||||
out = _FakeOutput()
|
||||
result = cli_mod._enable_extended_enter_keys(
|
||||
output=out,
|
||||
env={"TERM_PROGRAM": "iTerm.app", "TERM": "xterm-256color"},
|
||||
)
|
||||
assert result is True
|
||||
assert b"\x1b[>1u" in out.written
|
||||
assert b"\x1b[>4;2m" in out.written
|
||||
|
||||
|
||||
@pytest.mark.linux_only
|
||||
def test_proc_version_microsoft_marker_preserves_newline():
|
||||
"""WSL detection via /proc when env vars are scrubbed (sudo etc.).
|
||||
@@ -177,3 +235,14 @@ def test_proc_version_microsoft_marker_preserves_newline():
|
||||
# ---------------------------------------------------------------------------
|
||||
# install_ctrl_enter_alias() — ANSI sequence mappings for enhanced terminals
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_is_ghostty_terminal_detection_paths():
|
||||
"""_is_ghostty_terminal matches exactly the two allowlist conditions."""
|
||||
import cli as cli_mod
|
||||
|
||||
assert cli_mod._is_ghostty_terminal({"TERM_PROGRAM": "ghostty"}) is True
|
||||
assert cli_mod._is_ghostty_terminal({"TERM": "xterm-ghostty"}) is True
|
||||
assert cli_mod._is_ghostty_terminal({"TERM": "XTERM-GHOSTTY"}) is True
|
||||
assert cli_mod._is_ghostty_terminal({"TERM_PROGRAM": "iTerm.app"}) is False
|
||||
assert cli_mod._is_ghostty_terminal({}) is False
|
||||
|
||||
@@ -433,3 +433,130 @@ def test_cmd_backspace_alias_not_clobbered():
|
||||
install_cmd_backspace_alias()
|
||||
install_modify_other_keys_aliases()
|
||||
assert _parse("\x1b[127;9u") == [Keys.ControlU]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Lock-bit variants (#89651): kitty/ghostty OR the CapsLock (64) / NumLock
|
||||
# (128) state into the CSI-u modifier parameter, so with a lock enabled
|
||||
# every combo arrives shifted (ESC[99;133u instead of ESC[99;5u) and died
|
||||
# as literal text without these aliases.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize("letter", CTRL_LETTERS)
|
||||
def test_ctrl_letter_with_numlock_parses_as_raw_byte(letter):
|
||||
"""Ctrl+<letter> with NumLock on (modifier + 128) must parse identically
|
||||
to the raw control byte — the exact garbage from #89651 ([127;133u)."""
|
||||
raw_byte = chr(ord(letter) - ord('a') + 1)
|
||||
raw_result = _parse(raw_byte)
|
||||
|
||||
numlock_seq = f"\x1b[{ord(letter)};133u" # 5 + 128
|
||||
assert _parse(numlock_seq) == raw_result, (
|
||||
f"NumLock Ctrl+{letter} ({numlock_seq!r}) should parse identically "
|
||||
f"to raw {raw_byte!r}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("letter", ["a", "c", "z"])
|
||||
def test_ctrl_letter_with_capslock_parses_as_raw_byte(letter):
|
||||
raw_byte = chr(ord(letter) - ord('a') + 1)
|
||||
capslock_seq = f"\x1b[{ord(letter)};69u" # 5 + 64
|
||||
assert _parse(capslock_seq) == _parse(raw_byte)
|
||||
|
||||
|
||||
def test_ctrl_c_with_both_locks_parses_as_raw_byte():
|
||||
"""Ctrl+C with CapsLock and NumLock both on (5 + 64 + 128 = 197)."""
|
||||
assert _parse("\x1b[99;197u") == _parse("\x03")
|
||||
|
||||
|
||||
def test_alt_letter_with_numlock_keeps_escape_prefix():
|
||||
assert _parse("\x1b[97;131u") == [Keys.Escape, "a"] # 3 + 128
|
||||
|
||||
|
||||
def test_shift_letter_with_capslock_types_uppercase():
|
||||
assert _parse("\x1b[97;66u") == ["A"] # 2 + 64
|
||||
|
||||
|
||||
def test_esc_key_with_numlock_is_escape():
|
||||
assert _parse("\x1b[27;129u") == [Keys.Escape] # 1 + 128
|
||||
assert _parse("\x1b[27;133u") == [Keys.Escape] # 5 + 128
|
||||
|
||||
|
||||
def test_ctrl_backspace_with_numlock_is_backward_kill_word():
|
||||
"""The exact sequence from the #89651 report ([127;133u)."""
|
||||
assert _parse("\x1b[127;133u") == [Keys.Escape, Keys.ControlH]
|
||||
|
||||
|
||||
def test_modify_other_keys_tilde_form_has_no_lock_variants():
|
||||
"""The xterm modifyOtherKeys encoding never carries lock bits, so no
|
||||
+64/+128 variants of the ESC[27;N;CP~ form may be installed."""
|
||||
for seq in ("\x1b[27;69;99~", "\x1b[27;133;99~", "\x1b[27;197;99~"):
|
||||
assert seq not in ANSI_SEQUENCES
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Follow-up widening: lock twins on the alias installers, legacy CSI-letter /
|
||||
# CSI-tilde navigation, unmodified CSI-u keys, and PUA functional keys.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_lock_bits_on_legacy_cursor_keys_map_to_plain_keys():
|
||||
"""kitty stamps lock bits onto legacy CSI-letter arrows too:
|
||||
plain Down + NumLock = ESC[1;129B, CapsLock = ESC[1;65B, both = 193."""
|
||||
for mod, key in ((129, Keys.Down), (65, Keys.Down), (193, Keys.Down)):
|
||||
assert _parse(f"\x1b[1;{mod}B") == [key]
|
||||
assert _parse("\x1b[1;130B") == [Keys.ShiftDown] # Shift + NumLock
|
||||
assert _parse("\x1b[1;131D") == [Keys.Escape, Keys.Left] # Alt + NumLock
|
||||
assert _parse("\x1b[1;133D") == [Keys.ControlLeft] # Ctrl + NumLock
|
||||
|
||||
|
||||
def test_lock_bits_on_tilde_navigation_keys():
|
||||
"""Delete/PageUp/etc. carry the modifier in CSI-tilde form."""
|
||||
assert _parse("\x1b[3;129~") == [Keys.Delete]
|
||||
assert _parse("\x1b[3;69~") == [Keys.ControlDelete] # Ctrl+Delete + Caps
|
||||
assert _parse("\x1b[5;193~") == [Keys.PageUp] # both locks
|
||||
|
||||
|
||||
def test_lock_bits_on_plain_f1_through_f4():
|
||||
"""Plain F1-F4 base mappings are SS3 (ESC O P); their CSI lock twins
|
||||
must still resolve (ESC[1;129P etc.)."""
|
||||
base = _parse("\x1bOP")
|
||||
assert _parse("\x1b[1;129P") == base
|
||||
assert _parse("\x1b[1;65P") == base
|
||||
|
||||
|
||||
def test_lock_bits_on_unmodified_csi_u_keys():
|
||||
"""Tab/Enter/Space/Backspace with only a lock held (modifier 1+lock)."""
|
||||
assert _parse("\x1b[9;65u") == _parse("\t")
|
||||
assert _parse("\x1b[13;193u") == _parse("\r")
|
||||
assert _parse("\x1b[32;129u") == _parse(" ")
|
||||
assert _parse("\x1b[127;129u") == _parse("\x7f")
|
||||
|
||||
|
||||
def test_lock_bits_on_pua_functional_keys():
|
||||
"""Kitty PUA functional keys (keypad, F13+) keep working under locks —
|
||||
NumLock especially matters because it gates the keypad itself."""
|
||||
assert _parse("\x1b[57399;129u") == ["0"] # KP_0 + NumLock
|
||||
assert _parse("\x1b[57376;129u") == [Keys.F13] # F13 + NumLock
|
||||
assert _parse("\x1b[57427;129u") == [Keys.Ignore] # KP_BEGIN + NumLock
|
||||
|
||||
|
||||
def test_lock_bits_on_shift_enter_and_ctrl_enter_aliases():
|
||||
from hermes_cli.pt_input_extras import (
|
||||
install_ctrl_enter_alias,
|
||||
install_shift_enter_alias,
|
||||
)
|
||||
install_shift_enter_alias()
|
||||
install_ctrl_enter_alias()
|
||||
newline = _parse("\x1b\r")
|
||||
assert _parse("\x1b[13;130u") == newline # Shift+Enter + NumLock
|
||||
assert _parse("\x1b[13;66u") == newline # Shift+Enter + CapsLock
|
||||
assert _parse("\x1b[13;133u") == newline # Ctrl+Enter + NumLock
|
||||
|
||||
|
||||
def test_lock_bits_on_cmd_backspace_alias():
|
||||
from hermes_cli.pt_input_extras import install_cmd_backspace_alias
|
||||
install_cmd_backspace_alias()
|
||||
assert _parse("\x1b[127;137u") == [Keys.ControlU] # Cmd+Backspace + NumLock
|
||||
assert _parse("\x1b[127;73u") == [Keys.ControlU] # Cmd+Backspace + Caps
|
||||
assert _parse("\x1b[3;137~") == [Keys.ControlK] # Cmd+FwdDel + NumLock
|
||||
|
||||
@@ -1340,3 +1340,105 @@ class TestShallowCloneDeepening:
|
||||
assert wt.exists(), (
|
||||
"genuinely unpushed commit must survive even after deepening"
|
||||
)
|
||||
|
||||
|
||||
class TestPrMergedEscapeHatch:
|
||||
"""Rebase-merged PRs whose diff changed during salvage defeat ``git
|
||||
cherry`` (patch-id mismatch), so the pruner asks GitHub whether the
|
||||
branch's PR is MERGED. These tests stub the ``gh`` binary on PATH — the
|
||||
contract is about how the pruner consumes the answer, not about GitHub.
|
||||
|
||||
Contract:
|
||||
- gh reports a merged PR + tree is clean -> reaped
|
||||
- gh reports no merged PR -> preserved
|
||||
- gh missing/failing -> preserved (fail safe)
|
||||
- dirty tree -> never reaped regardless of gh
|
||||
"""
|
||||
|
||||
_age = staticmethod(TestWorktreeLockReaping._age)
|
||||
|
||||
@staticmethod
|
||||
def _mk_diverged(repo, name, age_h=100):
|
||||
"""Worktree with a commit NOT patch-equivalent to anything upstream."""
|
||||
p = repo / ".worktrees" / name
|
||||
(repo / ".worktrees").mkdir(exist_ok=True)
|
||||
subprocess.run(
|
||||
["git", "worktree", "add", str(p), "-b", f"hermes/{name}", "HEAD"],
|
||||
cwd=repo, capture_output=True,
|
||||
)
|
||||
(p / "salvaged.txt").write_text("diff that was reworked during salvage\n")
|
||||
subprocess.run(["git", "add", "salvaged.txt"], cwd=p, capture_output=True)
|
||||
subprocess.run(["git", "commit", "-m", "salvaged work"], cwd=p, capture_output=True)
|
||||
TestPrMergedEscapeHatch._age(p, age_h)
|
||||
return p
|
||||
|
||||
@staticmethod
|
||||
def _stub_gh(tmp_path, monkeypatch, stdout='[{"number": 1}]', exit_code=0):
|
||||
gh = tmp_path / "bin" / "gh"
|
||||
gh.parent.mkdir(parents=True, exist_ok=True)
|
||||
gh.write_text(f"#!/bin/sh\nprintf '%s' '{stdout}'\nexit {exit_code}\n")
|
||||
gh.chmod(0o755)
|
||||
monkeypatch.setenv("PATH", f"{gh.parent}:{os.environ['PATH']}")
|
||||
|
||||
def test_merged_pr_tree_is_reaped(self, git_repo, tmp_path, monkeypatch):
|
||||
import cli
|
||||
wt = self._mk_diverged(git_repo, "hermes-rebase-merged")
|
||||
assert cli._worktree_commits_all_merged_upstream(str(wt)) is False, (
|
||||
"precondition: cherry must NOT consider this merged — the PR "
|
||||
"check is the only thing that can reap it"
|
||||
)
|
||||
self._stub_gh(tmp_path, monkeypatch)
|
||||
cli._prune_stale_worktrees(str(git_repo))
|
||||
assert not wt.exists(), (
|
||||
"clean tree whose branch has a MERGED PR is merged work — reap it"
|
||||
)
|
||||
|
||||
def test_no_merged_pr_preserved(self, git_repo, tmp_path, monkeypatch):
|
||||
import cli
|
||||
wt = self._mk_diverged(git_repo, "hermes-pr-open")
|
||||
self._stub_gh(tmp_path, monkeypatch, stdout="[]")
|
||||
cli._prune_stale_worktrees(str(git_repo))
|
||||
assert wt.exists(), "no merged PR -> still unpushed work, preserve"
|
||||
|
||||
def test_gh_failure_fails_safe(self, git_repo, tmp_path, monkeypatch):
|
||||
import cli
|
||||
wt = self._mk_diverged(git_repo, "hermes-gh-down")
|
||||
self._stub_gh(tmp_path, monkeypatch, stdout="", exit_code=1)
|
||||
cli._prune_stale_worktrees(str(git_repo))
|
||||
assert wt.exists(), "gh failure must preserve the tree (fail safe)"
|
||||
|
||||
def test_dirty_tree_never_reaped_even_with_merged_pr(
|
||||
self, git_repo, tmp_path, monkeypatch
|
||||
):
|
||||
import cli
|
||||
wt = self._mk_diverged(git_repo, "hermes-dirty-merged")
|
||||
(wt / "uncommitted.txt").write_text("in-flight\n")
|
||||
self._age(wt, 100)
|
||||
self._stub_gh(tmp_path, monkeypatch)
|
||||
cli._prune_stale_worktrees(str(git_repo))
|
||||
assert wt.exists(), "dirty guard outranks the PR-merged verdict"
|
||||
|
||||
def test_merged_verdict_memoized_by_branch_and_head(
|
||||
self, git_repo, tmp_path, monkeypatch
|
||||
):
|
||||
import cli
|
||||
wt = self._mk_diverged(git_repo, "hermes-memo")
|
||||
self._stub_gh(tmp_path, monkeypatch)
|
||||
cache: dict = {}
|
||||
assert cli._worktree_branch_pr_merged(str(wt), cache=cache) is True
|
||||
keys = [k for k in cache if k.startswith("pr-merged:")]
|
||||
assert len(keys) == 1 and cache[keys[0]] is True
|
||||
# Break gh: a cached True verdict must not re-consult it.
|
||||
self._stub_gh(tmp_path, monkeypatch, stdout="", exit_code=1)
|
||||
assert cli._worktree_branch_pr_merged(str(wt), cache=cache) is True
|
||||
|
||||
def test_negative_verdict_not_cached(self, git_repo, tmp_path, monkeypatch):
|
||||
import cli
|
||||
wt = self._mk_diverged(git_repo, "hermes-nocache-neg")
|
||||
self._stub_gh(tmp_path, monkeypatch, stdout="[]")
|
||||
cache: dict = {}
|
||||
assert cli._worktree_branch_pr_merged(str(wt), cache=cache) is False
|
||||
assert not [k for k in cache if k.startswith("pr-merged:")], (
|
||||
"False must not be memoized — the PR can merge later with the "
|
||||
"same (branch, head) key"
|
||||
)
|
||||
|
||||
@@ -0,0 +1,231 @@
|
||||
"""Inbound replay dedupe on the relay adapter (transplanted from the
|
||||
live-cards branch for the rc.4 relay-fixes train).
|
||||
|
||||
Live-canary finding #3 (Alice, staging): the relay inbound leg is
|
||||
at-least-once. On WS re-handshake the connector replays its durable
|
||||
per-instance buffer; a long multi-tool turn straddling a quiet socket drop
|
||||
got its ORIGINAL inbound replayed after the turn finished, re-running the
|
||||
entire turn — the user saw the final answer posted 2-5x. Platform message
|
||||
identity (chat_id + message_id/ts) is stable across replays, so a bounded
|
||||
seen-set drops them. Fail-open: events without a message_id never dedupe
|
||||
(dropping a real message is strictly worse than rerunning one).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import Platform, PlatformConfig
|
||||
from gateway.platforms.base import MessageEvent, SessionSource
|
||||
from gateway.relay.adapter import RelayAdapter
|
||||
from gateway.relay.descriptor import CONTRACT_VERSION, CapabilityDescriptor
|
||||
from tests.gateway.relay.stub_connector import StubConnector
|
||||
|
||||
|
||||
def make_desc(**kw) -> CapabilityDescriptor:
|
||||
base = dict(
|
||||
contract_version=CONTRACT_VERSION,
|
||||
platform="slack",
|
||||
label="Slack",
|
||||
max_message_length=39000,
|
||||
supports_draft_streaming=True,
|
||||
supports_edit=True,
|
||||
supports_threads=True,
|
||||
markdown_dialect="slack",
|
||||
len_unit="chars",
|
||||
emoji="\U0001f4ac",
|
||||
platform_hint="",
|
||||
pii_safe=False,
|
||||
supported_ops=("send", "edit", "typing"),
|
||||
)
|
||||
base.update(kw)
|
||||
return CapabilityDescriptor(**base)
|
||||
|
||||
|
||||
def _connected_adapter(**desc_kw):
|
||||
desc = make_desc(**desc_kw)
|
||||
stub = StubConnector(desc)
|
||||
adapter = RelayAdapter(PlatformConfig(), desc, transport=stub)
|
||||
return adapter, stub
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def loop():
|
||||
loop = asyncio.new_event_loop()
|
||||
asyncio.set_event_loop(loop)
|
||||
yield loop
|
||||
loop.close()
|
||||
|
||||
|
||||
def _record(bucket, event):
|
||||
async def _coro():
|
||||
bucket.append(event)
|
||||
return _coro()
|
||||
|
||||
|
||||
async def _false_coro():
|
||||
return False
|
||||
|
||||
|
||||
async def _none_coro():
|
||||
return None
|
||||
|
||||
|
||||
class TestInboundReplayDedupe:
|
||||
"""Finding #3 (live canary): connector replay of the original inbound
|
||||
after a WS re-handshake must not re-run the turn."""
|
||||
|
||||
def _event(self, message_id="1700.100", chat_id="C1", text="hi"):
|
||||
# A REAL MessageEvent, shaped exactly as _event_from_wire produces it:
|
||||
# chat identity lives on event.source, NOT as a top-level attribute.
|
||||
# (The first version of these tests used a SimpleNamespace with a
|
||||
# top-level chat_id — a shape no production code path produces — and
|
||||
# green-lit a dedupe key that read the wrong field.)
|
||||
source = SessionSource(
|
||||
platform=Platform.SLACK,
|
||||
chat_id=chat_id,
|
||||
chat_type="channel",
|
||||
user_id="U1",
|
||||
message_id=message_id,
|
||||
)
|
||||
return MessageEvent(text=text, source=source, message_id=message_id)
|
||||
|
||||
def _tap(self, adapter, handled):
|
||||
adapter.handle_message = lambda e: _record(handled, e)
|
||||
adapter._consume_prompt_response = lambda e: _false_coro()
|
||||
adapter._localize_inbound_media = lambda e: _none_coro()
|
||||
|
||||
def test_replayed_inbound_dropped(self, loop):
|
||||
adapter, _ = _connected_adapter()
|
||||
handled = []
|
||||
self._tap(adapter, handled)
|
||||
e = self._event()
|
||||
loop.run_until_complete(adapter._on_inbound(e))
|
||||
loop.run_until_complete(adapter._on_inbound(e)) # replay
|
||||
assert len(handled) == 1
|
||||
|
||||
def test_distinct_messages_both_handled(self, loop):
|
||||
adapter, _ = _connected_adapter()
|
||||
handled = []
|
||||
self._tap(adapter, handled)
|
||||
loop.run_until_complete(adapter._on_inbound(self._event("1700.100")))
|
||||
loop.run_until_complete(adapter._on_inbound(self._event("1700.200")))
|
||||
assert len(handled) == 2
|
||||
|
||||
def test_missing_message_id_fails_open(self, loop):
|
||||
adapter, _ = _connected_adapter()
|
||||
handled = []
|
||||
self._tap(adapter, handled)
|
||||
e = self._event(message_id=None)
|
||||
loop.run_until_complete(adapter._on_inbound(e))
|
||||
loop.run_until_complete(adapter._on_inbound(e))
|
||||
assert len(handled) == 2 # never dedupe without identity
|
||||
|
||||
def test_seen_set_bounded(self, loop):
|
||||
adapter, _ = _connected_adapter()
|
||||
adapter.handle_message = lambda e: _none_coro()
|
||||
adapter._consume_prompt_response = lambda e: _false_coro()
|
||||
adapter._localize_inbound_media = lambda e: _none_coro()
|
||||
for i in range(600):
|
||||
loop.run_until_complete(adapter._on_inbound(self._event(f"ts.{i}")))
|
||||
assert len(adapter._seen_inbound) <= adapter._SEEN_INBOUND_MAX
|
||||
|
||||
|
||||
class TestWireLevelReplayDedupe:
|
||||
"""The full production inbound path: a connector wire frame decoded by
|
||||
_event_from_wire, then dispatched through RelayAdapter._on_inbound.
|
||||
|
||||
This is the layer the hand-built-event tests above cannot vouch for: the
|
||||
dedupe key must work on the exact object shape the wire decoder emits.
|
||||
The original dedupe commit shipped green on hand-built events while being
|
||||
a no-op on decoded ones — this class exists so that can't recur.
|
||||
"""
|
||||
|
||||
WIRE = {
|
||||
"text": "hi",
|
||||
"message_type": "text",
|
||||
"message_id": "1700.100",
|
||||
"source": {
|
||||
"platform": "slack",
|
||||
"chat_id": "C1",
|
||||
"chat_type": "channel",
|
||||
"user_id": "U1",
|
||||
"message_id": "1700.100",
|
||||
},
|
||||
}
|
||||
|
||||
def _tap(self, adapter, handled):
|
||||
adapter.handle_message = lambda e: _record(handled, e)
|
||||
adapter._consume_prompt_response = lambda e: _false_coro()
|
||||
adapter._localize_inbound_media = lambda e: _none_coro()
|
||||
|
||||
def _decode(self, **overrides):
|
||||
from gateway.relay.ws_transport import _event_from_wire
|
||||
|
||||
raw = {**self.WIRE, **overrides}
|
||||
if "source" in overrides:
|
||||
raw["source"] = {**self.WIRE["source"], **overrides["source"]}
|
||||
return _event_from_wire(raw)
|
||||
|
||||
def test_decoded_event_yields_a_dedupe_key(self):
|
||||
adapter, _ = _connected_adapter()
|
||||
key = adapter._inbound_dedupe_key(self._decode())
|
||||
assert key is not None, (
|
||||
"the wire decoder's event shape must produce a dedupe key — "
|
||||
"None here means the dedupe is fail-open for ALL production "
|
||||
"traffic (the original ship-broken state)"
|
||||
)
|
||||
|
||||
def test_replayed_wire_frame_dropped(self, loop):
|
||||
adapter, _ = _connected_adapter()
|
||||
handled = []
|
||||
self._tap(adapter, handled)
|
||||
loop.run_until_complete(adapter._on_inbound(self._decode()))
|
||||
# The connector re-delivers the SAME frame on re-handshake; the
|
||||
# decoder builds a fresh object each time, so identity must come
|
||||
# from the key, not object identity.
|
||||
loop.run_until_complete(adapter._on_inbound(self._decode()))
|
||||
assert len(handled) == 1
|
||||
|
||||
def test_same_ids_on_different_platforms_not_conflated(self, loop):
|
||||
# Phase 1.5 multiplex: one adapter fronts several platforms. Numeric
|
||||
# chat/message ids can collide across platforms; both must dispatch.
|
||||
adapter, _ = _connected_adapter()
|
||||
handled = []
|
||||
self._tap(adapter, handled)
|
||||
loop.run_until_complete(adapter._on_inbound(self._decode()))
|
||||
loop.run_until_complete(
|
||||
adapter._on_inbound(self._decode(source={"platform": "discord"}))
|
||||
)
|
||||
assert len(handled) == 2
|
||||
|
||||
|
||||
class TestDedupeKeyPlatformNormalization:
|
||||
"""The platform component of the key must be spelling-invariant: a
|
||||
Platform enum and its plain-string form are ONE platform (one key), and
|
||||
two different string platforms must never collapse into a shared empty
|
||||
component. Production wire decoding always yields the enum; alternate
|
||||
event constructors may carry the string."""
|
||||
|
||||
def _key(self, platform):
|
||||
adapter, _ = _connected_adapter()
|
||||
source = SessionSource(
|
||||
platform=platform, chat_id="C1", chat_type="channel", message_id="m1"
|
||||
)
|
||||
event = MessageEvent(text="hi", source=source, message_id="m1")
|
||||
return adapter._inbound_dedupe_key(event)
|
||||
|
||||
def test_enum_and_string_spellings_produce_one_key(self):
|
||||
assert self._key(Platform.SLACK) == self._key("slack")
|
||||
|
||||
def test_distinct_string_platforms_stay_distinct(self):
|
||||
assert self._key("slack") != self._key("discord")
|
||||
|
||||
def test_missing_platform_still_yields_a_key(self):
|
||||
# Fail-open on identity is reserved for missing message/chat ids;
|
||||
# a missing platform alone must not disable dedupe.
|
||||
key = self._key(None)
|
||||
assert key is not None
|
||||
assert key.startswith(":")
|
||||
@@ -0,0 +1,373 @@
|
||||
"""Regression tests for the relay WS transport hardening fix.
|
||||
|
||||
Coatue incident 2026-08-18: WAN latency / event-loop stalls tripped the
|
||||
websockets library's default 20s pong deadline, closing customer-gateway
|
||||
sockets with `1011 keepalive ping timeout`. On top of the spurious close,
|
||||
every in-flight outbound then hung for the full _outbound_timeout_s (~30s)
|
||||
because only disconnect() failed pending futures — an unexpected socket drop
|
||||
left them stranded — and sends issued while the reconnect supervisor was
|
||||
backing off registered futures no reader could ever resolve.
|
||||
|
||||
Three hardening changes under test:
|
||||
1. _read_loop fails all in-flight _pending futures on ANY exit path with
|
||||
the dict shape callers expect ({"success": False, ...}).
|
||||
2. _request_response fails fast while the reconnect supervisor is
|
||||
mid-redial (live supervisor task = the redial window).
|
||||
3. connect() passes explicit WAN-friendly keepalive tuning
|
||||
(ping_interval=30, ping_timeout=60) to websockets.connect().
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import pytest
|
||||
|
||||
import gateway.relay.ws_transport as ws_transport_mod
|
||||
from gateway.relay.ws_transport import WebSocketRelayTransport, WEBSOCKETS_AVAILABLE
|
||||
|
||||
pytestmark = pytest.mark.skipif(not WEBSOCKETS_AVAILABLE, reason="websockets not installed")
|
||||
|
||||
if WEBSOCKETS_AVAILABLE:
|
||||
from websockets.exceptions import ConnectionClosedError
|
||||
|
||||
|
||||
class _DroppingWS:
|
||||
"""Fake socket: accepts sends, then the read loop dies mid-iteration —
|
||||
the shape of an unexpected close (e.g. 1011 keepalive ping timeout)."""
|
||||
|
||||
def __init__(self, close_code: int | None = None):
|
||||
self.sent: list[str] = []
|
||||
# Reader blocks here until the test releases it, so the outbound
|
||||
# future is registered BEFORE the "socket" drops.
|
||||
self.drop = asyncio.Event()
|
||||
self._close_code = close_code
|
||||
|
||||
async def send(self, data):
|
||||
self.sent.append(data)
|
||||
|
||||
def __aiter__(self):
|
||||
return self
|
||||
|
||||
async def __anext__(self):
|
||||
await self.drop.wait()
|
||||
if self._close_code is not None:
|
||||
from websockets.frames import Close
|
||||
|
||||
raise ConnectionClosedError(Close(self._close_code, ""), None)
|
||||
raise ConnectionClosedError(None, None)
|
||||
|
||||
async def close(self):
|
||||
pass
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_read_loop_exit_fails_pending_futures_promptly():
|
||||
"""When the socket drops unexpectedly, in-flight _request_response callers
|
||||
must get {"success": False, ...} promptly — not block ~30s on a future
|
||||
only the (now dead) reader could have resolved."""
|
||||
t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=30.0)
|
||||
fake = _DroppingWS()
|
||||
t._ws = fake
|
||||
t._reader = asyncio.create_task(t._read_loop())
|
||||
|
||||
send_task = asyncio.create_task(t.send_outbound({"op": "send_message", "text": "hi"}))
|
||||
# Let the outbound frame go out and its future register in _pending.
|
||||
for _ in range(50):
|
||||
if t._pending:
|
||||
break
|
||||
await asyncio.sleep(0.01)
|
||||
assert t._pending, "outbound future never registered"
|
||||
|
||||
# Drop the socket: the read loop exits on ConnectionClosedError.
|
||||
fake.drop.set()
|
||||
|
||||
result = await asyncio.wait_for(send_task, timeout=2.0)
|
||||
assert result == {"success": False, "error": "relay transport connection lost"}
|
||||
assert t._pending == {}
|
||||
await t._reader
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_during_redial_window_fails_fast():
|
||||
"""While the reconnect supervisor is backing off after a drop, a send must
|
||||
return an error dict immediately (no RuntimeError, no 30s timeout on an
|
||||
unresolvable future). Drives the REAL sequence — reader exit arms the
|
||||
supervisor and clears _ws — rather than hand-crafting a stale-_ws state
|
||||
the transport can no longer reach."""
|
||||
t = WebSocketRelayTransport(
|
||||
"ws://unused",
|
||||
"discord",
|
||||
"bot1",
|
||||
reconnect=True,
|
||||
reconnect_backoff_s=60.0, # park the supervisor in backoff
|
||||
outbound_timeout_s=30.0,
|
||||
)
|
||||
fake = _DroppingWS()
|
||||
t._ws = fake
|
||||
await _run_reader_to_exit(t, fake)
|
||||
supervisor = t._supervisor
|
||||
try:
|
||||
assert supervisor is not None and not supervisor.done(), (
|
||||
"reader exit must arm the reconnect supervisor"
|
||||
)
|
||||
result = await asyncio.wait_for(
|
||||
t.send_outbound({"op": "send_message", "text": "hi"}), timeout=1.0
|
||||
)
|
||||
assert result["success"] is False
|
||||
assert t._pending == {}
|
||||
finally:
|
||||
if supervisor is not None:
|
||||
supervisor.cancel()
|
||||
try:
|
||||
await supervisor
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_allowed_once_redial_installs_fresh_socket(monkeypatch):
|
||||
"""The moment _dial_and_start() installs a fresh socket and its reader,
|
||||
the transport is genuinely usable — even though the supervisor task has
|
||||
not finished unwinding (it is still awaiting the hello sends). A send in
|
||||
that window must be ACCEPTED, not rejected as 'reconnecting': gating
|
||||
sends on supervisor state rejected real traffic on a live socket."""
|
||||
|
||||
class _LiveWS:
|
||||
def __init__(self):
|
||||
self.sent: list[str] = []
|
||||
self.hello_seen = asyncio.Event()
|
||||
self.release = asyncio.Event()
|
||||
|
||||
async def send(self, data):
|
||||
self.sent.append(data)
|
||||
if '"hello"' in data:
|
||||
# Inside _dial_and_start, AFTER _ws and the reader are
|
||||
# installed. Park here to hold the window open.
|
||||
self.hello_seen.set()
|
||||
await self.release.wait()
|
||||
|
||||
def __aiter__(self):
|
||||
return self
|
||||
|
||||
async def __anext__(self):
|
||||
await asyncio.sleep(3600)
|
||||
|
||||
async def close(self):
|
||||
pass
|
||||
|
||||
live = _LiveWS()
|
||||
|
||||
async def _fake_connect(url, **kwargs):
|
||||
return live
|
||||
|
||||
monkeypatch.setattr(ws_transport_mod.websockets, "connect", _fake_connect)
|
||||
|
||||
t = WebSocketRelayTransport(
|
||||
"ws://unused",
|
||||
"discord",
|
||||
"bot1",
|
||||
reconnect=True,
|
||||
reconnect_backoff_s=0.01,
|
||||
outbound_timeout_s=5.0,
|
||||
)
|
||||
# Arm the supervisor exactly as the reader's fall-through does.
|
||||
t._supervisor = asyncio.create_task(t._reconnect_loop())
|
||||
await asyncio.wait_for(live.hello_seen.wait(), timeout=2.0)
|
||||
try:
|
||||
assert t._ws is live and not t._supervisor.done()
|
||||
|
||||
send_task = asyncio.create_task(
|
||||
t.send_outbound({"op": "send_message", "text": "hi"})
|
||||
)
|
||||
# The send must reach the live socket (registered + frame written),
|
||||
# not fail fast: wait for the outbound frame to land.
|
||||
for _ in range(100):
|
||||
if any('"outbound"' in s for s in live.sent):
|
||||
break
|
||||
await asyncio.sleep(0.01)
|
||||
assert any('"outbound"' in s for s in live.sent), (
|
||||
"send was rejected during the post-dial window despite a live "
|
||||
"socket and running reader"
|
||||
)
|
||||
|
||||
# Resolve it via the reader path shape: answer directly.
|
||||
rid = next(iter(t._pending))
|
||||
t._pending[rid].set_result({"success": True})
|
||||
assert (await asyncio.wait_for(send_task, timeout=2.0)) == {"success": True}
|
||||
finally:
|
||||
live.release.set()
|
||||
await asyncio.wait_for(t._supervisor, timeout=2.0)
|
||||
if t._reader is not None:
|
||||
t._reader.cancel()
|
||||
try:
|
||||
await t._reader
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_connect_passes_wan_keepalive_tuning(monkeypatch):
|
||||
"""connect() must pass ping_interval=30 / ping_timeout=60 explicitly —
|
||||
the library defaults (20/20) caused spurious 1011 keepalive closes over
|
||||
WAN paths (Coatue 2026-08-18). Both call sites (with/without auth
|
||||
headers) are exercised."""
|
||||
captured: list[dict] = []
|
||||
|
||||
class _IdleWS:
|
||||
async def send(self, data):
|
||||
pass
|
||||
|
||||
def __aiter__(self):
|
||||
return self
|
||||
|
||||
async def __anext__(self):
|
||||
await asyncio.sleep(3600)
|
||||
|
||||
async def close(self):
|
||||
pass
|
||||
|
||||
async def _fake_connect(url, **kwargs):
|
||||
captured.append(kwargs)
|
||||
return _IdleWS()
|
||||
|
||||
monkeypatch.setattr(ws_transport_mod.websockets, "connect", _fake_connect)
|
||||
|
||||
# Site 1: no upgrade secret -> the headerless connect() call.
|
||||
t = WebSocketRelayTransport("ws://unused", "discord", "bot1")
|
||||
await t.connect()
|
||||
await t.disconnect(budget_s=0)
|
||||
|
||||
# Site 2: secret + gateway_id -> the additional_headers connect() call.
|
||||
t2 = WebSocketRelayTransport(
|
||||
"ws://unused", "discord", "bot1", gateway_id="gw-1", upgrade_secret="s3cret"
|
||||
)
|
||||
await t2.connect()
|
||||
await t2.disconnect(budget_s=0)
|
||||
|
||||
assert len(captured) == 2
|
||||
no_header_kwargs, header_kwargs = captured
|
||||
assert "additional_headers" not in no_header_kwargs
|
||||
assert "additional_headers" in header_kwargs
|
||||
for kwargs in captured:
|
||||
assert kwargs.get("ping_interval") == 30
|
||||
assert kwargs.get("ping_timeout") == 60
|
||||
|
||||
|
||||
async def _run_reader_to_exit(t: WebSocketRelayTransport, fake: _DroppingWS) -> None:
|
||||
"""Start the reader on ``fake``, drop the socket, and wait for the reader
|
||||
to fully unwind — the state every post-drop assertion depends on."""
|
||||
t._reader = asyncio.create_task(t._read_loop())
|
||||
await asyncio.sleep(0)
|
||||
fake.drop.set()
|
||||
await t._reader
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_read_loop_without_socket_still_fails_pending():
|
||||
"""If the reader is ever scheduled with no socket (lifecycle bug), it must
|
||||
still settle in-flight waiters on its way out — the old `assert` escaped
|
||||
before the fail-pending cleanup and left them to the full 30s timeout."""
|
||||
t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=30.0)
|
||||
loop = asyncio.get_running_loop()
|
||||
fut: asyncio.Future = loop.create_future()
|
||||
t._pending["rid"] = fut
|
||||
t._ws = None
|
||||
|
||||
await t._read_loop() # must not raise
|
||||
|
||||
assert fut.done()
|
||||
assert fut.result() == {"success": False, "error": "relay transport connection lost"}
|
||||
assert t._pending == {}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_after_terminal_4401_revocation_fails_fast():
|
||||
"""A terminal 4401 revocation deliberately arms NO reconnect supervisor,
|
||||
so the reader's exit is the LAST liveness transition this transport will
|
||||
ever make. If _ws still points at the dead socket afterwards, the
|
||||
revocation path's own fatal-error notification send wedges for the full
|
||||
_outbound_timeout_s. The reader must leave _ws cleared so the
|
||||
not-connected guard answers instantly."""
|
||||
t = WebSocketRelayTransport(
|
||||
"ws://unused", "discord", "bot1", reconnect=True, outbound_timeout_s=30.0
|
||||
)
|
||||
fake = _DroppingWS(close_code=4401)
|
||||
t._ws = fake
|
||||
t._handshake_succeeded = True # prior handshake -> 4401 is a revocation
|
||||
await _run_reader_to_exit(t, fake)
|
||||
|
||||
assert t._auth_revoked is True
|
||||
assert t._supervisor is None # revocation must not re-dial
|
||||
assert t._ws is None, "dead socket handle must not survive the reader"
|
||||
|
||||
result = await asyncio.wait_for(
|
||||
t.send_outbound({"op": "send_message", "text": "hi"}), timeout=2.0
|
||||
)
|
||||
assert result["success"] is False
|
||||
assert t._pending == {}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_after_drop_with_reconnect_disabled_fails_fast():
|
||||
"""reconnect=False transports never arm a supervisor either — the same
|
||||
stranded-_ws wedge as the revocation path, reachable by configuration."""
|
||||
t = WebSocketRelayTransport(
|
||||
"ws://unused", "discord", "bot1", reconnect=False, outbound_timeout_s=30.0
|
||||
)
|
||||
fake = _DroppingWS()
|
||||
t._ws = fake
|
||||
await _run_reader_to_exit(t, fake)
|
||||
|
||||
assert t._ws is None, "dead socket handle must not survive the reader"
|
||||
|
||||
result = await asyncio.wait_for(
|
||||
t.send_outbound({"op": "send_message", "text": "hi"}), timeout=2.0
|
||||
)
|
||||
assert result["success"] is False
|
||||
assert t._pending == {}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_raising_socket_returns_error_dict():
|
||||
"""The socket can die BETWEEN the `_ws is None` liveness guard and the
|
||||
actual write (the reader's finally hasn't cleared the handle yet). The
|
||||
write then raises ConnectionClosed — but send_outbound's contract is a
|
||||
result dict, and RelayAdapter.send consumes it with no try. The raise
|
||||
must be converted to {"success": False, ...}, with no future left in
|
||||
_pending."""
|
||||
|
||||
class _RaisingWS:
|
||||
"""Send raises (already dead); the reader hasn't noticed yet."""
|
||||
|
||||
def __init__(self):
|
||||
self.reader_release = asyncio.Event()
|
||||
|
||||
async def send(self, data):
|
||||
raise ConnectionClosedError(None, None)
|
||||
|
||||
def __aiter__(self):
|
||||
return self
|
||||
|
||||
async def __anext__(self):
|
||||
await self.reader_release.wait()
|
||||
raise ConnectionClosedError(None, None)
|
||||
|
||||
async def close(self):
|
||||
pass
|
||||
|
||||
t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=5.0)
|
||||
fake = _RaisingWS()
|
||||
t._ws = fake
|
||||
t._reader = asyncio.create_task(t._read_loop())
|
||||
await asyncio.sleep(0)
|
||||
|
||||
result = await asyncio.wait_for(
|
||||
t.send_outbound({"op": "send_message", "text": "hi"}), timeout=2.0
|
||||
)
|
||||
assert result["success"] is False
|
||||
assert "relay send failed" in result["error"]
|
||||
assert t._pending == {}
|
||||
|
||||
fake.reader_release.set()
|
||||
await t._reader
|
||||
@@ -96,3 +96,38 @@ def test_install_dependencies_force_reinstalls_versioned_specs(tmp_path, monkeyp
|
||||
|
||||
assert installed, "force=True must reach the install step"
|
||||
assert any("mem0ai>=2.0.10,<3" in specs for specs in installed)
|
||||
|
||||
|
||||
def test_cmd_status_memory_tool_gate_disabled(capsys, monkeypatch):
|
||||
"""When both memory stores are disabled, Memory status reports memory tool as disabled."""
|
||||
_cfg = {"memory": {"memory_enabled": False, "user_profile_enabled": False}}
|
||||
monkeypatch.setattr("hermes_cli.config.load_config", lambda: _cfg)
|
||||
# check_memory_requirements() reads the readonly loader, not load_config.
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.config.load_config_readonly", lambda: _cfg, raising=False
|
||||
)
|
||||
monkeypatch.setattr(memory_setup, "_get_available_providers", lambda: [])
|
||||
|
||||
memory_setup.cmd_status(SimpleNamespace())
|
||||
|
||||
captured = capsys.readouterr().out
|
||||
assert "Memory tool: disabled ✗" in captured
|
||||
assert "Memory injection: disabled ✗" in captured
|
||||
assert "User profile: disabled ✗" in captured
|
||||
|
||||
|
||||
def test_cmd_status_memory_tool_gate_enabled(capsys, monkeypatch):
|
||||
"""When at least one memory store is enabled, Memory status reports memory tool as enabled."""
|
||||
_cfg = {"memory": {"memory_enabled": True, "user_profile_enabled": False}}
|
||||
monkeypatch.setattr("hermes_cli.config.load_config", lambda: _cfg)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.config.load_config_readonly", lambda: _cfg, raising=False
|
||||
)
|
||||
monkeypatch.setattr(memory_setup, "_get_available_providers", lambda: [])
|
||||
|
||||
memory_setup.cmd_status(SimpleNamespace())
|
||||
|
||||
captured = capsys.readouterr().out
|
||||
assert "Memory tool: enabled ✓" in captured
|
||||
assert "Memory injection: enabled ✓" in captured
|
||||
assert "User profile: disabled ✗" in captured
|
||||
|
||||
@@ -111,6 +111,13 @@ class TestConfigYamlRouting:
|
||||
assert "not a recognized config key" not in capsys.readouterr().out
|
||||
assert "script_timeout_seconds: 600" in _read_config(_isolated_hermes_home)
|
||||
|
||||
def test_memory_nudge_interval_is_recognized(self, _isolated_hermes_home, capsys):
|
||||
"""The documented background-memory review interval is runtime config."""
|
||||
set_config_value("memory.nudge_interval", "0")
|
||||
|
||||
assert "not a recognized config key" not in capsys.readouterr().out
|
||||
assert "nudge_interval: 0" in _read_config(_isolated_hermes_home)
|
||||
|
||||
def test_terminal_docker_cwd_mount_flag_goes_to_config_and_env(self, _isolated_hermes_home):
|
||||
set_config_value("terminal.docker_mount_cwd_to_workspace", "true")
|
||||
config = _read_config(_isolated_hermes_home)
|
||||
|
||||
@@ -0,0 +1,243 @@
|
||||
"""Behavior contracts for hermes_cli.worktree_gc (attended reclaim).
|
||||
|
||||
Each guard gets its own contract against a REAL git repo fixture (no mocks —
|
||||
the entire value of these tests is exercising actual git verdicts):
|
||||
|
||||
- clean + fully merged tree → reap
|
||||
- untracked-only dirt → reap-archive (files archived, then removed)
|
||||
- tracked modifications → keep, any age
|
||||
- unique unpushed commits → keep
|
||||
- patch-equivalent commits (rebase/squash-merge leak) → reap
|
||||
- live-locked tree → keep
|
||||
- kanban t_<hex> tree → keep (owned by kanban gc)
|
||||
- branch GC: merged branch deleted, unique-commit branch kept,
|
||||
checked-out branch kept, protected names kept
|
||||
- reclaim operates ONLY on the frozen audit list (concurrent-session trap)
|
||||
"""
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from hermes_cli import worktree_gc
|
||||
|
||||
|
||||
def _git(args, cwd, env=None):
|
||||
e = dict(os.environ)
|
||||
e.update({
|
||||
"GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t",
|
||||
"GIT_COMMITTER_NAME": "t", "GIT_COMMITTER_EMAIL": "t@t",
|
||||
})
|
||||
if env:
|
||||
e.update(env)
|
||||
result = subprocess.run(
|
||||
["git", *args], capture_output=True, text=True, cwd=str(cwd), env=e,
|
||||
)
|
||||
assert result.returncode == 0, f"git {args} failed: {result.stderr}"
|
||||
return result.stdout.strip()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def repo(tmp_path, monkeypatch):
|
||||
"""origin (bare) + clone with .worktrees/, HOME redirected for archives."""
|
||||
monkeypatch.setenv("HOME", str(tmp_path / "home"))
|
||||
(tmp_path / "home").mkdir()
|
||||
|
||||
origin = tmp_path / "origin.git"
|
||||
origin.mkdir()
|
||||
_git(["init", "--bare", "-b", "main"], origin)
|
||||
|
||||
clone = tmp_path / "repo"
|
||||
_git(["clone", str(origin), str(clone)], tmp_path)
|
||||
(clone / "README.md").write_text("hello\n")
|
||||
_git(["add", "."], clone)
|
||||
_git(["commit", "-m", "init"], clone)
|
||||
_git(["push", "origin", "main"], clone)
|
||||
# origin/HEAD so upstream resolution works like a real clone.
|
||||
_git(["remote", "set-head", "origin", "main"], clone)
|
||||
(clone / ".worktrees").mkdir()
|
||||
return clone
|
||||
|
||||
|
||||
def _add_worktree(repo_path, name, branch=None):
|
||||
tree = repo_path / ".worktrees" / name
|
||||
branch = branch or f"hermes/{name}"
|
||||
_git(["worktree", "add", str(tree), "-b", branch], repo_path)
|
||||
return tree, branch
|
||||
|
||||
|
||||
def _verdict(records, name):
|
||||
match = [record for record in records if record.name == name]
|
||||
assert match, f"no record for {name}"
|
||||
return match[0]
|
||||
|
||||
|
||||
class TestAuditVerdicts:
|
||||
def test_clean_merged_tree_reaps(self, repo):
|
||||
_add_worktree(repo, "hermes-clean")
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
assert _verdict(records, "hermes-clean").verdict == "reap"
|
||||
|
||||
def test_tracked_modifications_keep(self, repo):
|
||||
tree, _ = _add_worktree(repo, "hermes-dirty")
|
||||
(tree / "README.md").write_text("edited\n")
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
record = _verdict(records, "hermes-dirty")
|
||||
assert record.verdict == "keep"
|
||||
assert "tracked" in record.reason
|
||||
|
||||
def test_untracked_only_is_reap_archive(self, repo):
|
||||
tree, _ = _add_worktree(repo, "hermes-scratch")
|
||||
(tree / "PR_BODY_DRAFT.md").write_text("draft\n")
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
record = _verdict(records, "hermes-scratch")
|
||||
assert record.verdict == "reap-archive"
|
||||
assert record.untracked == ["PR_BODY_DRAFT.md"]
|
||||
|
||||
def test_unique_unpushed_commits_keep(self, repo):
|
||||
tree, _ = _add_worktree(repo, "hermes-work")
|
||||
(tree / "new.py").write_text("x = 1\n")
|
||||
_git(["add", "."], tree)
|
||||
_git(["commit", "-m", "unique work"], tree)
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
record = _verdict(records, "hermes-work")
|
||||
assert record.verdict == "keep"
|
||||
assert "unpushed" in record.reason
|
||||
|
||||
def test_patch_equivalent_commits_reap(self, repo):
|
||||
"""The squash/rebase-merge leak: local commit unreachable from any
|
||||
remote ref but patch-equivalent to an upstream commit → merged work."""
|
||||
tree, _ = _add_worktree(repo, "hermes-merged")
|
||||
(tree / "feat.py").write_text("y = 2\n")
|
||||
_git(["add", "."], tree)
|
||||
_git(["commit", "-m", "feat"], tree)
|
||||
sha = _git(["rev-parse", "HEAD"], tree)
|
||||
# "Merge" it to main with a DIFFERENT committer so the cherry-pick
|
||||
# produces a distinct sha (same-second identical-committer cherry
|
||||
# picks can produce the identical sha — pitfall from the skill).
|
||||
_git(["cherry-pick", sha], repo,
|
||||
env={"GIT_COMMITTER_NAME": "other", "GIT_COMMITTER_EMAIL": "o@o"})
|
||||
_git(["push", "origin", "main"], repo)
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
assert _verdict(records, "hermes-merged").verdict == "reap"
|
||||
|
||||
def test_live_locked_tree_keeps(self, repo):
|
||||
tree, _ = _add_worktree(repo, "hermes-live")
|
||||
_git(["worktree", "lock", str(tree),
|
||||
"--reason", f"hermes pid={os.getpid()}"], repo)
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
record = _verdict(records, "hermes-live")
|
||||
assert record.verdict == "keep"
|
||||
assert "in use" in record.reason
|
||||
|
||||
def test_kanban_tree_untouched(self, repo):
|
||||
_add_worktree(repo, "t_deadbeef", branch="kanban/t_deadbeef")
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
record = _verdict(records, "t_deadbeef")
|
||||
assert record.verdict == "keep"
|
||||
assert "kanban" in record.reason
|
||||
|
||||
|
||||
class TestReclaim:
|
||||
def test_reap_removes_tree_and_branch(self, repo):
|
||||
tree, branch = _add_worktree(repo, "hermes-clean")
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
actions = worktree_gc.reclaim_worktrees(str(repo), records=records)
|
||||
assert any("removed hermes-clean" in a for a in actions)
|
||||
assert not tree.exists()
|
||||
probe = subprocess.run(
|
||||
["git", "rev-parse", "--verify", "--quiet", branch],
|
||||
capture_output=True, text=True, cwd=str(repo),
|
||||
)
|
||||
assert probe.returncode != 0, "branch should be gone with its tree"
|
||||
|
||||
def test_untracked_files_archived_before_removal(self, repo):
|
||||
tree, _ = _add_worktree(repo, "hermes-scratch")
|
||||
(tree / "NOTES.md").write_text("important scribbles\n")
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
worktree_gc.reclaim_worktrees(str(repo), records=records)
|
||||
assert not tree.exists()
|
||||
archive_root = Path.home() / ".hermes" / "archive" / "worktree-prune"
|
||||
archived = list(archive_root.rglob("NOTES.md"))
|
||||
assert archived, "untracked file must be archived, not destroyed"
|
||||
assert archived[0].read_text() == "important scribbles\n"
|
||||
|
||||
def test_dry_run_changes_nothing(self, repo):
|
||||
tree, _ = _add_worktree(repo, "hermes-clean")
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
actions = worktree_gc.reclaim_worktrees(
|
||||
str(repo), dry_run=True, records=records
|
||||
)
|
||||
assert any("would remove" in a for a in actions)
|
||||
assert tree.exists()
|
||||
|
||||
def test_frozen_list_ignores_trees_created_after_audit(self, repo):
|
||||
"""Concurrent-session trap: a tree created between audit and reclaim
|
||||
must be out of scope by construction."""
|
||||
_add_worktree(repo, "hermes-old")
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
late_tree, _ = _add_worktree(repo, "hermes-late")
|
||||
worktree_gc.reclaim_worktrees(str(repo), records=records)
|
||||
assert late_tree.exists(), "tree created after the audit must survive"
|
||||
|
||||
def test_dead_locked_tree_is_unlocked_and_reaped(self, repo):
|
||||
tree, _ = _add_worktree(repo, "hermes-zombie")
|
||||
_git(["worktree", "lock", str(tree),
|
||||
"--reason", "hermes pid=999999999"], repo)
|
||||
records = worktree_gc.audit_worktrees(str(repo), with_sizes=False)
|
||||
assert _verdict(records, "hermes-zombie").verdict == "reap"
|
||||
worktree_gc.reclaim_worktrees(str(repo), records=records)
|
||||
assert not tree.exists()
|
||||
|
||||
|
||||
class TestBranchGC:
|
||||
def test_merged_branch_deleted_any_name(self, repo):
|
||||
"""Branch GC is content-gated, not name-gated: any fully-merged local
|
||||
branch is safe to delete regardless of prefix."""
|
||||
_git(["branch", "salv-12345", "main"], repo)
|
||||
_git(["branch", "feat/some-old-thing", "main"], repo)
|
||||
records = worktree_gc.audit_branches(str(repo))
|
||||
by_name = {record.name: record for record in records}
|
||||
assert by_name["salv-12345"].verdict == "delete"
|
||||
assert by_name["feat/some-old-thing"].verdict == "delete"
|
||||
worktree_gc.reclaim_branches(str(repo), records=records)
|
||||
out = _git(["branch", "--format=%(refname:short)"], repo)
|
||||
assert "salv-12345" not in out
|
||||
assert "feat/some-old-thing" not in out
|
||||
|
||||
def test_unique_commit_branch_kept(self, repo):
|
||||
_git(["checkout", "-b", "feat/real-work"], repo)
|
||||
(repo / "wip.py").write_text("z = 3\n")
|
||||
_git(["add", "."], repo)
|
||||
_git(["commit", "-m", "wip"], repo)
|
||||
_git(["checkout", "main"], repo)
|
||||
records = worktree_gc.audit_branches(str(repo))
|
||||
by_name = {record.name: record for record in records}
|
||||
assert by_name["feat/real-work"].verdict == "keep"
|
||||
assert "unique" in by_name["feat/real-work"].reason
|
||||
|
||||
def test_patch_equivalent_branch_deleted(self, repo):
|
||||
"""Rebase-merged PR branch: SHAs differ from main but every commit is
|
||||
patch-equivalent — the dominant branch leak."""
|
||||
_git(["checkout", "-b", "fix/landed"], repo)
|
||||
(repo / "fix.py").write_text("a = 4\n")
|
||||
_git(["add", "."], repo)
|
||||
_git(["commit", "-m", "fix"], repo)
|
||||
sha = _git(["rev-parse", "HEAD"], repo)
|
||||
_git(["checkout", "main"], repo)
|
||||
_git(["cherry-pick", sha], repo,
|
||||
env={"GIT_COMMITTER_NAME": "other", "GIT_COMMITTER_EMAIL": "o@o"})
|
||||
_git(["push", "origin", "main"], repo)
|
||||
records = worktree_gc.audit_branches(str(repo))
|
||||
by_name = {record.name: record for record in records}
|
||||
assert by_name["fix/landed"].verdict == "delete"
|
||||
assert "patch-equivalent" in by_name["fix/landed"].reason
|
||||
|
||||
def test_checked_out_and_protected_kept(self, repo):
|
||||
_tree, branch = _add_worktree(repo, "hermes-active")
|
||||
records = worktree_gc.audit_branches(str(repo))
|
||||
by_name = {record.name: record for record in records}
|
||||
assert by_name["main"].verdict == "keep"
|
||||
assert by_name[branch].verdict == "keep"
|
||||
@@ -72,7 +72,12 @@ def _tool_call(i: int):
|
||||
return SimpleNamespace(
|
||||
id=f"call_{i}",
|
||||
type="function",
|
||||
function=SimpleNamespace(name="web_search", arguments='{"query": "x"}'),
|
||||
# Vary the query per call: a real marathon turn issues distinct
|
||||
# lookups, and identical (args, result) pairs are now legitimately
|
||||
# deduped into reference stubs by the stall-guard subsystem —
|
||||
# zero-variance args here would deflate the very context pressure
|
||||
# this test exists to exercise.
|
||||
function=SimpleNamespace(name="web_search", arguments=f'{{"query": "x{i}"}}'),
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -293,6 +293,62 @@ class TestInPlaceAntiGrowthGuard:
|
||||
assert agent.session_id == sid
|
||||
assert db.get_session(sid)["end_reason"] is None
|
||||
|
||||
def test_in_place_salvages_near_break_even_growth(self):
|
||||
"""Fat retained tool output + todo state should be salvaged and committed."""
|
||||
from hermes_state import SessionDB
|
||||
from agent.conversation_compression import compress_context
|
||||
from agent.model_metadata import estimate_messages_tokens_rough
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
db = SessionDB(db_path=Path(tmp) / "t.db")
|
||||
sid = "20260819_salvage"
|
||||
_seed(db, sid, "salvage")
|
||||
agent = _make_agent(db, sid, in_place=True)
|
||||
agent._last_flushed_db_idx = 5
|
||||
|
||||
tool_calls = [
|
||||
{"id": call_id, "function": {"name": "terminal", "arguments": "{}"}}
|
||||
for call_id in ("c1", "c2", "c3")
|
||||
]
|
||||
original = [
|
||||
{"role": "user", "content": "do the work"},
|
||||
{"role": "assistant", "content": "calling tools", "tool_calls": tool_calls},
|
||||
{"role": "tool", "tool_call_id": "c1", "content": "OLD1 " + ("x" * 8000)},
|
||||
{"role": "tool", "tool_call_id": "c2", "content": "OLD2 " + ("y" * 8000)},
|
||||
{"role": "tool", "tool_call_id": "c3", "content": "keep-me " + ("z" * 100)},
|
||||
{"role": "user", "content": "thanks"},
|
||||
]
|
||||
grown = [
|
||||
{"role": "user", "content": "[CONTEXT COMPACTION] " + ("S" * 200)},
|
||||
{"role": "assistant", "content": "calling tools", "tool_calls": tool_calls},
|
||||
{"role": "tool", "tool_call_id": "c1", "content": "OLD1 " + ("x" * 8000)},
|
||||
{"role": "tool", "tool_call_id": "c2", "content": "OLD2 " + ("y" * 8000)},
|
||||
{"role": "tool", "tool_call_id": "c3", "content": "keep-me " + ("z" * 100)},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Current todos:\n- [ ] leftover",
|
||||
"_todo_snapshot_synthetic": True,
|
||||
},
|
||||
]
|
||||
assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original)
|
||||
compressor = getattr(agent, "context_compressor")
|
||||
compressor.compress = (
|
||||
lambda messages, current_tokens=None, focus_topic=None, force=False: grown
|
||||
)
|
||||
|
||||
compressed, _sp = compress_context(
|
||||
agent, original, approx_tokens=100_000, system_message="sys"
|
||||
)
|
||||
|
||||
assert getattr(agent, "_last_compaction_in_place") is True
|
||||
assert estimate_messages_tokens_rough(compressed) < estimate_messages_tokens_rough(original)
|
||||
tool_bodies = [m.get("content") for m in compressed if m.get("role") == "tool"]
|
||||
assert any(isinstance(body, str) and body.startswith("keep-me") for body in tool_bodies)
|
||||
assert any("cleared to save context space" in (body or "") for body in tool_bodies)
|
||||
# Tool stubbing alone got under budget, so the todo snapshot (the
|
||||
# only in-transcript todo re-injection) survives the salvage.
|
||||
assert any(m.get("_todo_snapshot_synthetic") for m in compressed)
|
||||
|
||||
def test_in_place_still_commits_shrinking_compression(self):
|
||||
"""The guard must not block legitimate compressions — a result SMALLER
|
||||
than the input still commits in place (regression net for #83339)."""
|
||||
|
||||
@@ -4535,6 +4535,118 @@ class TestRunConversation:
|
||||
assert agent.context_compressor.context_length == 200_000
|
||||
mock_compress.assert_called_once()
|
||||
|
||||
def test_output_cap_retry_before_generic_retry_exhaustion(self, agent):
|
||||
"""Provider max-output-cap 400s clamp via the output-cap handler, not
|
||||
the generic retry loop ("failed after 3 retries").
|
||||
"""
|
||||
self._setup_agent(agent)
|
||||
agent.api_mode = "chat_completions"
|
||||
agent.provider = "deepseek"
|
||||
agent.base_url = "https://api.deepseek.com/v1"
|
||||
agent.model = "deepseek-v4-flash"
|
||||
agent.max_tokens = 98_304
|
||||
agent.compression_enabled = True
|
||||
agent.context_compressor.context_length = 200_000
|
||||
agent.context_compressor.should_compress = MagicMock(return_value=False)
|
||||
|
||||
error_msg = (
|
||||
"[400]: max_tokens (98304) exceeds model's maximum output tokens "
|
||||
"(65536) for model deepseek-v4-flash "
|
||||
"(ref: 7735422e-9cb4-4075-a779-dfecb3204a0e)"
|
||||
)
|
||||
exc = Exception(error_msg)
|
||||
exc.status_code = 400
|
||||
exc.code = 400
|
||||
|
||||
ok_resp = _mock_response(content="done", finish_reason="stop")
|
||||
agent.client.chat.completions.create.side_effect = [exc, ok_resp]
|
||||
|
||||
mock_compress = MagicMock(return_value=(
|
||||
[{"role": "user", "content": "hello"}],
|
||||
"You are helpful.",
|
||||
))
|
||||
with (
|
||||
patch.object(agent, "_persist_session"),
|
||||
patch.object(agent, "_save_trajectory"),
|
||||
patch.object(agent, "_cleanup_task_resources"),
|
||||
patch.object(agent.context_compressor, "update_model"),
|
||||
patch.object(agent, "_compress_context", mock_compress),
|
||||
):
|
||||
result = agent.run_conversation("hello")
|
||||
|
||||
assert len(agent.client.chat.completions.create.call_args_list) == 2
|
||||
second_call = agent.client.chat.completions.create.call_args_list[1].kwargs
|
||||
assert result["completed"] is True
|
||||
assert second_call["max_tokens"] <= 65_472
|
||||
assert agent.context_compressor.context_length == 200_000
|
||||
|
||||
def test_output_cap_retry_when_gateway_wraps_error_as_rate_limit(self, agent):
|
||||
"""Some relays wrap the upstream max-output 400 as HTTP 429. The
|
||||
parseable output cap must still route into the output-cap handler
|
||||
instead of burning generic rate-limit retries (#72281).
|
||||
"""
|
||||
self._run_wrapped_429_output_cap(agent, fallback_chain=[])
|
||||
|
||||
def test_wrapped_output_cap_429_not_consumed_by_eager_fallback(self, agent):
|
||||
"""With a NON-EMPTY fallback chain, the eager rate-limit fallback must
|
||||
NOT consume the wrapped output-cap 429 — the failure is a deterministic
|
||||
request-shape problem the clamp fixes in one retry; switching provider
|
||||
burns a fallback slot for nothing (#72281 ordering guard).
|
||||
"""
|
||||
self._run_wrapped_429_output_cap(
|
||||
agent,
|
||||
fallback_chain=[{"provider": "openrouter", "model": "anthropic/claude-sonnet-4"}],
|
||||
)
|
||||
|
||||
def _run_wrapped_429_output_cap(self, agent, *, fallback_chain):
|
||||
self._setup_agent(agent)
|
||||
agent.api_mode = "chat_completions"
|
||||
agent.provider = "custom"
|
||||
agent.base_url = "http://192.168.1.254:20128/v1"
|
||||
agent.model = "deepseekv4flash"
|
||||
agent.max_tokens = 98_304
|
||||
agent.compression_enabled = True
|
||||
agent._fallback_chain = fallback_chain
|
||||
agent._fallback_index = 0
|
||||
agent.context_compressor.context_length = 200_000
|
||||
agent.context_compressor.should_compress = MagicMock(return_value=False)
|
||||
|
||||
error_msg = (
|
||||
"Error code: 429 - {'error': {'message': \"[400]: max_tokens "
|
||||
"(98304) exceeds model's maximum output tokens (65536) for model "
|
||||
"deepseek-v4-flash (ref: 37bde60f-44e7-44e2-b995-4af17fba6d6b)\", "
|
||||
"'type': 'rate_limit_error', 'code': 'rate_limit_exceeded'}}"
|
||||
)
|
||||
exc = Exception(error_msg)
|
||||
exc.status_code = 429
|
||||
exc.code = "rate_limit_exceeded"
|
||||
|
||||
ok_resp = _mock_response(content="done", finish_reason="stop")
|
||||
agent.client.chat.completions.create.side_effect = [exc, ok_resp]
|
||||
|
||||
mock_compress = MagicMock(return_value=(
|
||||
[{"role": "user", "content": "hello"}],
|
||||
"You are helpful.",
|
||||
))
|
||||
with (
|
||||
patch.object(agent, "_persist_session"),
|
||||
patch.object(agent, "_save_trajectory"),
|
||||
patch.object(agent, "_cleanup_task_resources"),
|
||||
patch.object(agent.context_compressor, "update_model"),
|
||||
patch.object(agent, "_compress_context", mock_compress),
|
||||
):
|
||||
result = agent.run_conversation("hello")
|
||||
|
||||
assert len(agent.client.chat.completions.create.call_args_list) == 2
|
||||
second_call = agent.client.chat.completions.create.call_args_list[1].kwargs
|
||||
assert result["completed"] is True
|
||||
assert second_call["max_tokens"] <= 65_472
|
||||
assert agent.context_compressor.context_length == 200_000
|
||||
# The clamp, not provider failover, must have recovered: no fallback
|
||||
# slot consumed and the model unchanged.
|
||||
assert agent._fallback_index == 0
|
||||
assert agent.model == "deepseekv4flash"
|
||||
|
||||
def test_output_cap_retry_with_large_api_only_content(self, agent):
|
||||
"""When a large system prompt makes api_messages huge while persisted
|
||||
messages stay tiny, the retry cap must still respect provider
|
||||
|
||||
@@ -68,6 +68,22 @@ class TestParseDashScopeOutputCap:
|
||||
assert parse_available_output_tokens_from_error(msg) == 32768
|
||||
|
||||
|
||||
class TestParseMaximumOutputTokensCap:
|
||||
"""Some OpenAI-compatible relays report the model's separate output cap."""
|
||||
|
||||
def test_parenthesized_max_output_cap(self):
|
||||
msg = (
|
||||
"API call failed after 3 retries: [400]: max_tokens (98304) "
|
||||
"exceeds model's maximum output tokens (65536)"
|
||||
)
|
||||
assert parse_available_output_tokens_from_error(msg) == 65536
|
||||
|
||||
def test_parenthesized_max_output_cap_is_output_cap(self):
|
||||
assert is_output_cap_error(
|
||||
"max_tokens (98304) exceeds model's maximum output tokens (65536)"
|
||||
) is True
|
||||
|
||||
|
||||
class TestIsOutputCapError:
|
||||
"""`is_output_cap_error` is the broader yes/no gate that keeps an
|
||||
output-cap 400 out of the compression death-loop even when we can't parse
|
||||
@@ -121,10 +137,28 @@ class TestParseVllmTokenBasedOutputCap:
|
||||
"output tokens."
|
||||
)
|
||||
|
||||
def test_vllm_token_based_format(self):
|
||||
# available output = 131072 - 65537 = 65535
|
||||
assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 65535
|
||||
# Verbatim vLLM response where the input is MEASURED, not back-computed:
|
||||
# window - input != requested - 1, so the reported figure is real.
|
||||
_VLLM_MSG_REAL_INPUT = (
|
||||
"This model's maximum context length is 131072 tokens. However, you "
|
||||
"requested 65536 output tokens and your prompt contains 100000 "
|
||||
"input tokens, for a total of 165536 tokens. Please reduce the length "
|
||||
"of the input prompt or the number of requested output tokens."
|
||||
)
|
||||
|
||||
def test_vllm_token_based_format(self):
|
||||
# The reported input is a LOWER BOUND that vLLM back-computes from the
|
||||
# constraint (65537 == 131072 + 1 - 65536), so window - input is just
|
||||
# requested - 1 and carries no information about the real prompt.
|
||||
# Halve the requested cap instead so the retry actually converges.
|
||||
assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 32768
|
||||
|
||||
def test_vllm_measured_input_is_trusted(self):
|
||||
# When the input is measured rather than derived, use it as-is.
|
||||
# available output = 131072 - 100000 = 31072
|
||||
assert parse_available_output_tokens_from_error(
|
||||
self._VLLM_MSG_REAL_INPUT
|
||||
) == 31072
|
||||
|
||||
def test_vllm_retry_fits_inside_window(self):
|
||||
# The retried cap plus the reported input must fit in the window.
|
||||
@@ -132,3 +166,28 @@ class TestParseVllmTokenBasedOutputCap:
|
||||
assert available is not None
|
||||
assert available + 65537 <= 131072
|
||||
|
||||
def test_vllm_retry_converges(self):
|
||||
"""The retry sequence must reach a working cap in a few attempts.
|
||||
|
||||
Regression test for the 65-tokens-per-retry crawl: with a 102400
|
||||
window and a real prompt of ~37000 tokens, retrying from a 65536 cap
|
||||
used to produce 65471 -> 65406 -> 65341 and exhaust the compression
|
||||
budget without ever fitting.
|
||||
"""
|
||||
window, real_input, cap = 102400, 37000, 65536
|
||||
for _ in range(5):
|
||||
if real_input + cap <= window:
|
||||
break
|
||||
# vLLM's message when max_tokens is the binding constraint.
|
||||
msg = (
|
||||
f"This model's maximum context length is {window} tokens. "
|
||||
f"However, you requested {cap} output tokens and your prompt "
|
||||
f"contains at least {window + 1 - cap} input tokens, for a "
|
||||
f"total of at least {window + 1} tokens."
|
||||
)
|
||||
available = parse_available_output_tokens_from_error(msg)
|
||||
assert available is not None
|
||||
assert available < cap, "each retry must lower the cap"
|
||||
cap = available
|
||||
assert real_input + cap <= window, f"did not converge: cap={cap}"
|
||||
|
||||
|
||||
@@ -295,6 +295,22 @@ def _build_persisted_message(
|
||||
return msg
|
||||
|
||||
|
||||
_PERSISTED_PATH_RE = re.compile(r"^Full output saved to: (.+)$", re.MULTILINE)
|
||||
|
||||
|
||||
def extract_persisted_path(content: str) -> str | None:
|
||||
"""Return the file path from a <persisted-output> replacement block.
|
||||
|
||||
Used by the result-reference stubbing guard (agent/tool_guardrails.py) so
|
||||
a stub referencing a persisted first occurrence can carry the spillover
|
||||
path instead of dangling. Returns None for non-persisted content.
|
||||
"""
|
||||
if not isinstance(content, str) or PERSISTED_OUTPUT_TAG not in content:
|
||||
return None
|
||||
match = _PERSISTED_PATH_RE.search(content)
|
||||
return match.group(1).strip() if match else None
|
||||
|
||||
|
||||
def maybe_persist_tool_result(
|
||||
content: str,
|
||||
tool_name: str,
|
||||
|
||||
@@ -28,6 +28,7 @@ import {
|
||||
setTerminalBackgroundHex,
|
||||
setTerminalForegroundHex,
|
||||
setXtversionName,
|
||||
skipKittyKeyboardProtocol,
|
||||
supportsExtendedKeys
|
||||
} from '../terminal.js'
|
||||
import {
|
||||
@@ -333,9 +334,13 @@ export default class App extends PureComponent<Props, State> {
|
||||
// distinguishable from ctrl+<letter>. We write both the kitty stack
|
||||
// push (CSI >1u) and xterm modifyOtherKeys level 2 (CSI >4;2m) —
|
||||
// terminals honor whichever they implement (tmux only accepts the
|
||||
// latter).
|
||||
// latter). Ghostty gets only modifyOtherKeys — its kitty
|
||||
// disambiguate mode strips Alt from Backspace (see
|
||||
// skipKittyKeyboardProtocol).
|
||||
if (supportsExtendedKeys()) {
|
||||
this.props.stdout.write(ENABLE_KITTY_KEYBOARD)
|
||||
if (!skipKittyKeyboardProtocol()) {
|
||||
this.props.stdout.write(ENABLE_KITTY_KEYBOARD)
|
||||
}
|
||||
this.props.stdout.write(ENABLE_MODIFY_OTHER_KEYS)
|
||||
}
|
||||
|
||||
|
||||
@@ -77,6 +77,7 @@ import {
|
||||
} from './selection.js'
|
||||
import {
|
||||
needsAltScreenResizeScrollbackClear,
|
||||
skipKittyKeyboardProtocol,
|
||||
supportsExtendedKeys,
|
||||
SYNC_OUTPUT_SUPPORTED,
|
||||
type Terminal,
|
||||
@@ -733,7 +734,11 @@ export default class Ink {
|
||||
// without the pop we'd accumulate depth on each editor round-trip).
|
||||
this.options.stdout.write(
|
||||
'\x1b[?1004h' +
|
||||
(supportsExtendedKeys() ? DISABLE_KITTY_KEYBOARD + ENABLE_KITTY_KEYBOARD + ENABLE_MODIFY_OTHER_KEYS : '')
|
||||
(supportsExtendedKeys()
|
||||
? DISABLE_KITTY_KEYBOARD +
|
||||
(skipKittyKeyboardProtocol() ? '' : ENABLE_KITTY_KEYBOARD) +
|
||||
ENABLE_MODIFY_OTHER_KEYS
|
||||
: '')
|
||||
)
|
||||
}
|
||||
onRender() {
|
||||
@@ -1472,7 +1477,11 @@ export default class Ink {
|
||||
// Pop-before-push keeps Kitty stack depth at 1 instead of accumulating
|
||||
// on each call.
|
||||
if (supportsExtendedKeys()) {
|
||||
this.options.stdout.write(DISABLE_KITTY_KEYBOARD + ENABLE_KITTY_KEYBOARD + ENABLE_MODIFY_OTHER_KEYS)
|
||||
this.options.stdout.write(
|
||||
DISABLE_KITTY_KEYBOARD +
|
||||
(skipKittyKeyboardProtocol() ? '' : ENABLE_KITTY_KEYBOARD) +
|
||||
ENABLE_MODIFY_OTHER_KEYS
|
||||
)
|
||||
}
|
||||
|
||||
if (!this.altScreenActive) {
|
||||
|
||||
@@ -74,3 +74,36 @@ describe('writeDiffToTerminal sync-marker gating (#66490 main-screen gap)', () =
|
||||
expect(out).toContain('frame')
|
||||
})
|
||||
})
|
||||
|
||||
describe('skipKittyKeyboardProtocol', () => {
|
||||
// Ghostty's kitty disambiguate mode strips the Alt modifier from
|
||||
// Backspace (Option+Backspace arrives as bare \x7f instead of CSI-u
|
||||
// \x1b[127;3u), so Ghostty must get only the modifyOtherKeys push.
|
||||
// Mirrors cli.py's _GHOSTTY_EXTENDED_ENTER_KEYS_SEQ exception.
|
||||
it('skips the kitty protocol push for ghostty', async () => {
|
||||
const { env } = await import('../utils/env.js')
|
||||
const { skipKittyKeyboardProtocol } = await import('./terminal.js')
|
||||
const saved = env.terminal
|
||||
try {
|
||||
env.terminal = 'ghostty'
|
||||
expect(skipKittyKeyboardProtocol()).toBe(true)
|
||||
} finally {
|
||||
env.terminal = saved
|
||||
}
|
||||
})
|
||||
|
||||
it.each(['iTerm.app', 'kitty', 'WezTerm', 'tmux', 'windows-terminal', 'vscode'])(
|
||||
'keeps the dual push for %s',
|
||||
async terminal => {
|
||||
const { env } = await import('../utils/env.js')
|
||||
const { skipKittyKeyboardProtocol } = await import('./terminal.js')
|
||||
const saved = env.terminal
|
||||
try {
|
||||
env.terminal = terminal
|
||||
expect(skipKittyKeyboardProtocol()).toBe(false)
|
||||
} finally {
|
||||
env.terminal = saved
|
||||
}
|
||||
}
|
||||
)
|
||||
})
|
||||
|
||||
@@ -326,6 +326,16 @@ export function supportsExtendedKeys(): boolean {
|
||||
return EXTENDED_KEYS_TERMINALS.includes(env.terminal ?? '')
|
||||
}
|
||||
|
||||
/** True when the Kitty keyboard protocol push (CSI >1u) must be skipped for
|
||||
* this terminal even though extended keys are supported. Ghostty's Kitty
|
||||
* disambiguate-mode implementation strips the Alt modifier from Backspace —
|
||||
* Option+Backspace arrives as bare \x7f instead of CSI-u \x1b[127;3u —
|
||||
* breaking backward-kill-word. Ghostty implements modifyOtherKeys correctly,
|
||||
* so it gets only that push (mirrors cli.py's Ghostty exception). */
|
||||
export function skipKittyKeyboardProtocol(): boolean {
|
||||
return env.terminal === 'ghostty'
|
||||
}
|
||||
|
||||
/** True if the terminal scrolls the viewport when it receives cursor-up
|
||||
* sequences that reach above the visible area. On Windows, conhost's
|
||||
* SetConsoleCursorPosition follows the cursor into scrollback
|
||||
|
||||
@@ -94,6 +94,7 @@ Groups are standalone rows in the same activity-ordered roster as Bot DMs. A Bot
|
||||
Bots message each other with attribution, and you can hand work off from any chat:
|
||||
|
||||
- **@mentions** — type `@researcher have a look at this` in any chat and the active Bot hands the message off, waits for the reply, and reports back. Mention names are validated against the live roster, so an email address or an unknown `@` passes through untouched.
|
||||
- **Renamed Bots keep their tags in sync** — give a Bot a friendly name (the pencil in its chat header, or `hermes profile rename`) and it becomes taggable by that name: a Bot titled *Research Buddy* answers to `@research-buddy` (and `@researchbuddy`), in regular chats and in group rooms alike. The composer's `@` autocomplete offers the renamed tag and also matches when you type the old profile name, which keeps resolving too.
|
||||
- **@mentions across machines** — mentioning a Bot that lives on another registered connection (use its `@name-device` handle when names collide) delivers over the Connections registry in the background: the active Bot stays on this device, the desktop routes the message to the recipient's machine, and the reply is relayed back attributed to that agent. Your window's gateway never switches.
|
||||
- **Direct messages** — a Bot reaches a teammate's Bot Chat through the standard CLI: it writes the message to a temp file (opening with the `Message from 🤖 <sender> (@<sender>):` prefix), then runs `hermes -p <bot> chat --in ~ -c "Bot Chat" --create-if-missing -Q --query-file <file>`. The file transport means nothing is shell-interpreted — quotes, `$(...)`, and backticks in the message arrive verbatim. The receiving Bot sees the message the next time it runs and knows how to reply, because the messaging protocol is part of its Bot Chat system prompt.
|
||||
|
||||
|
||||
@@ -58,6 +58,43 @@ hermes -w # Interactive mode in worktree
|
||||
hermes -w -z "Fix issue #123" # Single query in worktree
|
||||
```
|
||||
|
||||
### Worktree cleanup
|
||||
|
||||
`hermes -w` sessions create disposable worktrees under `<repo>/.worktrees/`.
|
||||
A conservative pruner runs automatically at startup (it only removes clean,
|
||||
fully-merged scratch trees past an age threshold), but preserved trees and
|
||||
merged local branches still accumulate on busy machines. Reclaim them
|
||||
explicitly:
|
||||
|
||||
```bash
|
||||
hermes worktree list # audit: age, size, verdict, reason per tree
|
||||
hermes worktree prune # remove safe trees + delete merged branches
|
||||
hermes worktree prune --dry-run # show the plan without changing anything
|
||||
hermes worktree prune --trees-only # leave local branches alone
|
||||
hermes worktree prune --branches-only # leave worktrees alone
|
||||
```
|
||||
|
||||
Inside a session, `/worktree prune [--dry-run]` does the same (and never
|
||||
touches the tree the session is running in).
|
||||
|
||||
Safety guarantees (all modes, any age):
|
||||
|
||||
- Uncommitted **tracked** changes are never deleted.
|
||||
- **Unique unpushed commits** are never deleted — commits that were
|
||||
rebase/squash-merged upstream are detected via `git cherry`
|
||||
patch-equivalence and count as merged, which is what lets the dominant
|
||||
"merged PR, tree preserved forever" leak finally reclaim.
|
||||
- Trees **in use by a running hermes session** are never touched.
|
||||
- **Untracked-only scratch** (PR body drafts, notes) is archived to
|
||||
`~/.hermes/archive/worktree-prune/` before its tree is removed — never
|
||||
destroyed.
|
||||
- Branch deletion is content-gated, not name-gated: any local branch whose
|
||||
commits are all on upstream is safe to delete; branches with unique work,
|
||||
checked-out branches, and `main`/`master`/`develop` are always kept.
|
||||
|
||||
When `.worktrees/` grows past 10 trees or 5 GB, startup prints a one-line
|
||||
notice pointing at these commands.
|
||||
|
||||
### Plugin management
|
||||
|
||||
The `hermes plugins` commands manage native Hermes plugins and portable Agent
|
||||
|
||||
@@ -1752,6 +1752,8 @@ agent:
|
||||
stall_guards: false
|
||||
```
|
||||
|
||||
The same gate also enables **result-reference stubbing**: when a re-issued identical tool call returns a byte-identical fresh result, the duplicate payload enters context as a short reference stub pointing at the earlier result (tool name, `tool_call_id`, an args summary, and — if the first result was persisted to disk — its spillover path) instead of repeating the full output. The tool still executes every time, so polling semantics are preserved: a changed result always flows through whole. Results under 512 characters, error results, and multimodal results are never stubbed, and pollers *are* stubbed (an unchanged poll is exactly the case where the duplicate payload carries no information).
|
||||
|
||||
## TTS Configuration
|
||||
|
||||
```yaml
|
||||
|
||||
@@ -284,6 +284,23 @@ the model substitutes the platform's idiomatic shortcut and app name):
|
||||
During all of this, your cursor stays wherever you left it and the email
|
||||
app never comes to front.
|
||||
|
||||
## Receiving the actual screenshot
|
||||
|
||||
Screenshots taken during computer control are normally internal — they exist
|
||||
so the model can see the screen, and the agent replies in text. But every
|
||||
image capture also saves a bounded, shareable copy under Hermes' image cache
|
||||
and reports its path, so on attachment-capable surfaces (Telegram, Discord,
|
||||
Desktop, and other gateway platforms) you can simply ask:
|
||||
|
||||
> *"Send me a screenshot of my screen."*
|
||||
|
||||
and the agent delivers the real image as a native attachment, not just a
|
||||
description. On the CLI there is no attachment channel, so the agent gives
|
||||
you the saved file's path instead.
|
||||
|
||||
Only the 20 most recent capture files are kept, and screenshots are never
|
||||
sent automatically — only when you ask for one.
|
||||
|
||||
## Provider compatibility
|
||||
|
||||
| Provider | Vision? | Works? | Notes |
|
||||
|
||||
@@ -266,7 +266,7 @@ first, set `memory.write_approval: true`. It's a simple on/off gate applied to
|
||||
| `false` (default) | Write freely — the gate is off (the pre-gate behaviour). |
|
||||
| `true` | Require approval before anything is saved. In the interactive CLI, foreground writes prompt you inline (entries are small enough to read in full). Everywhere else — messaging platforms, scripts, and the background self-improvement review — writes are **staged** for review with `/memory pending`. |
|
||||
|
||||
> To turn memory off entirely (not just gate it), set `memory_enabled: false`.
|
||||
> To turn memory off entirely (not just gate it), set both `memory_enabled: false` and `user_profile_enabled: false`. When both built-in stores are disabled, the built-in `memory` tool is automatically hidden.
|
||||
|
||||
Review staged writes from the CLI or any messaging platform:
|
||||
|
||||
|
||||
Reference in New Issue
Block a user