refactor(tui_gateway): table-drive billing error kinds, share pet selection, compact group H docstrings

This commit is contained in:
Teknium
2026-09-02 23:25:59 -07:00
parent 2b538e535a
commit 052a5dec66
11 changed files with 207 additions and 273 deletions
+29 -42
View File
@@ -1,13 +1,11 @@
"""Bot-relay JSON-RPC handlers — the gateway side of cross-connection A2A.
Connections ARE the peer set: the Desktop owns every gateway socket (local, remote,
SSH, Cloud, docker) and relays between them through these four doors on EACH gateway:
``roster.sync`` (push the union roster of OTHER connections' agents so ``message_agent``
resolves them), ``outbox.drain`` (collect envelopes queued here for other connections),
``deliver`` (run a one-turn Bot Chat delivery on the TARGET gateway, return the reply),
``reply`` (write the reply/error back on the SENDER gateway for the waiter to pick up).
Storage/validation plumbing lives in ``tools/bot_relay.py``. Handlers are rebound onto
server.py's globals (method_ctx.py) and reference ``_ok``/``_err`` etc. bare.
Connections ARE the peer set: the Desktop owns every gateway socket and relays between them via
four doors on EACH gateway: ``roster.sync`` (push OTHER connections' agents so ``message_agent``
resolves them), ``outbox.drain`` (collect envelopes queued here for other connections), ``deliver``
(one-turn Bot Chat delivery on the TARGET gateway, returns the reply), ``reply`` (write the
reply/error back on the SENDER gateway for its waiter). Plumbing: ``tools/bot_relay.py``.
Handlers are rebound onto server.py's globals (method_ctx.py) and reference ``_ok``/``_err`` bare.
"""
import os
@@ -36,10 +34,10 @@ def _run_delivery(profile: str, tmp: str) -> subprocess.CompletedProcess:
@method("bot_relay.roster.sync")
def _(rid, params: dict, _root=_relay_root) -> dict:
"""Replace this gateway's view of agents on OTHER connections.
"""Replace this gateway's view of agents on OTHER connections → ``{count}`` accepted rows.
Params: ``agents`` — rows ``{profile, handle, connection_id, connection_label?, title?,
description?}``; rows failing validation are dropped, not fatal. Result: ``{count}``.
``agents``: rows ``{profile, handle, connection_id, connection_label?, title?, description?}``;
rows failing validation are dropped, not fatal.
"""
try:
from tools.bot_relay import write_remote_roster
@@ -51,7 +49,7 @@ def _(rid, params: dict, _root=_relay_root) -> dict:
@method("bot_relay.outbox.drain")
def _(rid, params: dict, _root=_relay_root) -> dict:
"""Claim every pending cross-connection envelope queued on this gateway. Result: ``{envelopes}``.
"""Claim every pending cross-connection envelope queued on this gateway → ``{envelopes}``.
Claimed envelopes move to ``claimed/`` atomically, so concurrent drains can't double-deliver.
"""
@@ -65,13 +63,9 @@ def _(rid, params: dict, _root=_relay_root) -> dict:
@method("bot_relay.deliver")
def _(rid, params: dict, _root=_relay_root, _run=_run_delivery) -> dict:
"""Deliver a relayed DM into a profile's Bot Chat ON THIS GATEWAY.
Params: ``profile`` (target on this install), ``message`` (already attribution-prefixed).
Runs the same one-turn ``hermes -p <profile> chat -c "Bot Chat"`` transport local DMs
use and returns ``{reply}``. Blocking by design (the Desktop calls it from its relay
worker; the RPC pool keeps it off the WS reader thread).
"""
"""Deliver a relayed DM (``profile``, attribution-prefixed ``message``) into a Bot Chat ON THIS
GATEWAY via the one-turn ``hermes -p <profile> chat -c "Bot Chat"`` transport local DMs use →
``{reply}``. Blocking by design (Desktop relay worker; the RPC pool keeps it off the reader)."""
import os
import subprocess
import tempfile
@@ -95,11 +89,10 @@ def _(rid, params: dict, _root=_relay_root, _run=_run_delivery) -> dict:
if resolved not in known:
return _err(rid, 4092, f"no profile '{profile}' on this gateway")
# When THIS gateway already hosts the target's Bot Chat live, the subprocess
# transport is fenced out by the single-owner lease and the payload dropped. Land
# the DM in the live session via prompt.submit — the composer's choke point, so
# role alternation, persistence and streaming behave as a typed message would.
# (Nested: needs server globals via method_ctx rebinding.)
# When THIS gateway already hosts the target's Bot Chat live, the subprocess transport is
# fenced out by the single-owner lease and the payload dropped. Land the DM in the live
# session via prompt.submit — the composer's choke point, so role alternation, persistence
# and streaming behave as a typed message would. (Nested: needs server globals via rebind.)
def _live_bot_chat_sid(profile_name: str) -> str:
from tools.bot_mode_probe import BOT_CHAT_TITLE
@@ -117,30 +110,27 @@ def _(rid, params: dict, _root=_relay_root, _run=_run_delivery) -> dict:
live_sid = _live_bot_chat_sid(resolved)
if live_sid:
# queued=True: a teammate's DM runs as the NEXT turn and never interrupts or
# steers a turn in flight (the default busy mode does); arrivals queue in order.
# queued=True: a teammate's DM runs as the NEXT turn and never interrupts or steers a
# turn in flight (the default busy mode does); arrivals queue in order.
submitted = _methods["prompt.submit"](rid, {"session_id": live_sid, "text": message, "queued": True})
if "error" in submitted:
return submitted
return _ok(
rid, {"reply": f"Delivered into @{resolved}'s open Bot Chat; the reply will appear there."}
)
reply = f"Delivered into @{resolved}'s open Bot Chat; the reply will appear there."
return _ok(rid, {"reply": reply})
fd, tmp = tempfile.mkstemp(prefix="hermes-relay-dm-", suffix=".txt", text=True)
try:
with os.fdopen(fd, "w", encoding="utf-8") as f:
f.write(message)
# Per-profile turn lock serializes with any other delivery turn into this
# profile and covers only the turn window. Worst-case hold is lock wait
# (bot_mode.turn_wait_seconds, default 120s) + the 600s turn timeout, doubled
# when the retry policy grants one re-run — callers must tolerate ~1320s.
# Per-profile turn lock serializes with any other delivery turn into this profile and
# covers only the turn window. Worst-case hold is lock wait (bot_mode.turn_wait_seconds,
# default 120s) + the 600s turn timeout, doubled on one retry — callers tolerate ~1320s.
with acquire_turn_lock(root, resolved):
proc = _run(resolved, tmp)
if proc.returncode != 0:
# Retry policy: transient classes re-run the SAME session once;
# context_overflow too — the retried turn's pre-API compaction pass
# compacts the over-threshold transcript first (no fresh session is
# ever minted). Auth/quota/config classes never retry.
# Retry policy: transient classes re-run the SAME session once; context_overflow
# too — the retried turn's pre-API compaction pass compacts the over-threshold
# transcript first (no fresh session is minted). Auth/quota/config never retry.
from tools.bot_failure_reasons import (
RETRY_NONE, classify_agent_error, retry_action)
@@ -169,11 +159,8 @@ def _(rid, params: dict, _root=_relay_root, _run=_run_delivery) -> dict:
@method("bot_relay.reply")
def _(rid, params: dict, _root=_relay_root) -> dict:
"""Write a relayed reply (or delivery error) for a sender-side waiter.
Params: ``id`` (envelope id), ``reply`` and/or ``error``, optional ``reason``
(typed failure code, see ``tools.bot_failure_reasons``).
"""
"""Write a relayed ``reply`` and/or ``error`` (+ optional typed ``reason``, see
``tools.bot_failure_reasons``) for envelope ``id`` so the sender-side waiter picks it up."""
envelope_id = str(params.get("id") or "").strip()
if not envelope_id:
return _err(rid, 4093, "id required")