"""Read-only gateway introspection commands: /status, /context, /usage, /agents, /insights, /topup. Split out of ``gateway/slash_commands.py``; bound onto ``GatewayRunner`` through ``GatewaySlashCommandsMixin``. Origin internals are imported lazily (``from gateway.slash_commands import ...``) inside the bodies to avoid the import cycle. """ from __future__ import annotations import logging import asyncio import hashlib import os import re import time from typing import Any from agent.account_usage import fetch_account_usage, render_account_usage_lines from agent.i18n import t from gateway.config import Platform from gateway.platforms.base import MessageEvent # Log-record parity with gateway/run.py and the origin module. logger = logging.getLogger("gateway.run") def _clean_str(value: Any) -> str: """Strip and return a non-empty string value, or empty string.""" return value.strip() if isinstance(value, str) and value.strip() else "" def _int_value(value: Any) -> int: """Safely coerce to int.""" try: return int(value) except (TypeError, ValueError): return 0 def _status_model_route(status_agent, persisted_route: dict, session_row: dict, session_entry): """``(model, provider, context_used, context_total)`` for /status. Order: live/cached agent route -> persisted dominant route -> SessionDB row -> gateway config (only loaded when something is still missing). """ from gateway.run import _AGENT_PENDING_SENTINEL, _load_gateway_config, _resolve_gateway_model model_name = provider_name = "" route_resolved = False context_used = context_total = 0 if status_agent is not None and status_agent is not _AGENT_PENDING_SENTINEL: live_model = _clean_str(getattr(status_agent, "model", "")) live_provider = _clean_str(getattr(status_agent, "provider", "")) if live_model and live_provider: model_name, provider_name, route_resolved = live_model, live_provider, True ctx = getattr(status_agent, "context_compressor", None) if ctx is not None: context_used = _int_value(getattr(ctx, "last_prompt_tokens", 0)) context_total = _int_value(getattr(ctx, "context_length", 0)) persisted_model = _clean_str(persisted_route.get("model")) persisted_provider = _clean_str(persisted_route.get("billing_provider")) if not route_resolved and persisted_model and persisted_provider: model_name, provider_name, route_resolved = persisted_model, persisted_provider, True if not route_resolved: model_name = _clean_str(session_row.get("model")) provider_name = _clean_str(session_row.get("billing_provider")) context_used = context_used or _int_value(getattr(session_entry, "last_prompt_tokens", 0)) user_config: dict[str, Any] = {} if not model_name or not provider_name or not context_total: try: user_config = _load_gateway_config() except Exception: user_config = {} model_cfg = user_config.get("model", {}) if isinstance(user_config, dict) else {} if not isinstance(model_cfg, dict): model_cfg = {} if not model_name: model_name = _resolve_gateway_model(user_config) if not provider_name: provider_name = _clean_str(model_cfg.get("provider")) if not context_total: configured_context = model_cfg.get("context_length") if isinstance(configured_context, int) and configured_context > 0: context_total = configured_context return model_name, provider_name, context_used, context_total def _context_compressor_lines(agent, ctx, used: int) -> list[str]: """/context full view: auto-compression threshold/headroom, compression count + last savings, and cumulative throughput (labelled as throughput, NOT context size).""" lines: list[str] = [] threshold = getattr(ctx, "threshold_tokens", 0) or 0 threshold_pct = (getattr(ctx, "threshold_percent", 0) or 0) * 100 if threshold > 0: if used >= threshold: lines.append( t("gateway.context.over_threshold", threshold=f"{threshold:,}", threshold_pct=f"{threshold_pct:.0f}") ) else: lines.append( t( "gateway.context.threshold", threshold=f"{threshold:,}", threshold_pct=f"{threshold_pct:.0f}", to_go=f"{threshold - used:,}", ) ) compressions = getattr(ctx, "compression_count", 0) or 0 lines.append(t("gateway.context.compressions", count=compressions)) if compressions: savings = getattr(ctx, "_last_compression_savings_pct", None) if savings is not None: lines.append(t("gateway.context.last_savings", savings=f"{savings:.0f}")) def _n(attr): return getattr(agent, attr, 0) or 0 lines.append("") lines.append(t("gateway.context.totals_header", calls=_n("session_api_calls"))) lines.append( t( "gateway.context.totals_line", input=f"{_n('session_input_tokens'):,}", output=f"{_n('session_output_tokens'):,}", reasoning=f"{_n('session_reasoning_tokens'):,}", ) ) lines.append(t("gateway.context.total_billed", total=f"{_n('session_total_tokens'):,}")) lines.append(t("gateway.context.throughput_note")) return lines def _agents_delegation_lines(d: dict) -> list[str]: """/agents rows for one background delegation. Live per-child activity comes from the registry's progress sampler: api calls, current tool, seconds since last activity.""" goal = " ".join(str(d.get("goal") or "").split()) if len(goal) > 70: goal = goal[:67] + "..." status = d.get("status", "?") row = f"- `{d.get('delegation_id', '?')}` · {status}" if status == "stalling": quiet = d.get("stalled_after_quiet_seconds") if quiet is not None: row += f" · no progress {quiet:.0f}s" elif d.get("seconds_since_progress", 0) >= 60: row += f" · quiet {d['seconds_since_progress']:.0f}s" if goal: row += f" · {goal}" lines = [row] for i, child in enumerate(d.get("children_activity") or []): if not isinstance(child, dict): continue tool = child.get("current_tool") doing = f"`{tool}`" if tool else "between turns" part = f" - child {i + 1}: {child.get('api_calls', '?')} api calls · {doing}" idle = child.get("seconds_since_activity") if idle is not None: part += f" · active {idle:.0f}s ago" lines.append(part) return lines def _usage_agent_stats_lines(agent) -> list[str]: """/usage session block for a live agent: rate limits, token breakdown (matches the CLI), context window and compression count.""" lines: list[str] = [] rl_state = agent.get_rate_limit_state() if rl_state and rl_state.has_data: from agent.rate_limit_tracker import format_rate_limit_compact lines.append(t("gateway.usage.rate_limits", state=format_rate_limit_compact(rl_state))) lines.append("") input_tokens = getattr(agent, "session_input_tokens", 0) or 0 output_tokens = getattr(agent, "session_output_tokens", 0) or 0 lines.append(t("gateway.usage.header_session")) lines.append(t("gateway.usage.label_model", model=agent.model)) lines.append(t("gateway.usage.label_input_tokens", count=f"{input_tokens:,}")) lines.append(t("gateway.usage.label_output_tokens", count=f"{output_tokens:,}")) lines.append(t("gateway.usage.label_total", count=f"{agent.session_total_tokens:,}")) lines.append(t("gateway.usage.label_api_calls", count=agent.session_api_calls)) ctx = agent.context_compressor _lpt = ctx.last_prompt_tokens if ctx.last_prompt_tokens > 0 else 0 if _lpt: pct = min(100, _lpt / ctx.context_length * 100) if ctx.context_length else 0 lines.append(t("gateway.usage.label_context", used=f"{_lpt:,}", total=f"{ctx.context_length:,}", pct=f"{pct:.0f}")) if ctx.compression_count: lines.append(t("gateway.usage.label_compressions", count=ctx.compression_count)) return lines class GatewayStatusCommandsMixin: """Read-only gateway introspection commands: /status, /context, /usage, /agents, /insights, /topup.""" async def _handle_status_command(self, event: MessageEvent) -> str: """Handle /status command.""" from gateway.run import _AGENT_PENDING_SENTINEL source = event.source session_entry = await self.async_session_store.get_or_create_session(source) connected_platforms = [p.value for p in self.adapters] # Check if there's an active agent. Keep the sentinel distinct: a # starting/pending run should not be treated as a fully usable agent for # model/context display, but it still occupies the session slot. session_key = session_entry.session_key agent = self._running_agents.get(session_key) is_running = agent is not None and agent is not _AGENT_PENDING_SENTINEL # Count pending /queue follow-ups (slot + overflow). adapter = self.adapters.get(source.platform) if source else None queue_depth = self._queue_depth(session_key, adapter=adapter) title, session_row, db_total_tokens, persisted_route = await self._status_session_db_facts( session_entry.session_id ) # Resolve model/context for cockpit-style status. Prefer the live or cached agent because it # carries the actual runtime route and context compressor; fall back to SessionDB metadata + # last_prompt_tokens so /status stays useful between turns without billing/account calls. status_agent = agent if is_running else self._cached_agent_for(session_key) model_name, provider_name, context_used, context_total = _status_model_route( status_agent, persisted_route, session_row, session_entry ) model_line = "" if model_name: if provider_name: model_line = t("gateway.status.model_provider", model=model_name, provider=provider_name) else: model_line = t("gateway.status.model", model=model_name) context_line = "" if context_total: pct = min(100, round((context_used / context_total) * 100)) if context_total else 0 context_line = t( "gateway.status.context", used=f"{context_used:,}", total=f"{context_total:,}", pct=f"{pct}", ) elif context_used: context_line = t("gateway.status.context_used", used=f"{context_used:,}") lines = [ t("gateway.status.header"), "", t("gateway.status.session_id", session_id=session_entry.session_id), ] if title: lines.append(t("gateway.status.title", title=title)) lines.extend([ t("gateway.status.created", timestamp=session_entry.created_at.strftime('%Y-%m-%d %H:%M')), t("gateway.status.last_activity", timestamp=session_entry.updated_at.strftime('%Y-%m-%d %H:%M')), ]) if model_line: lines.append(model_line) if context_line: lines.append(context_line) lines.extend([ t("gateway.status.tokens", tokens=f"{db_total_tokens:,}"), t("gateway.status.agent_running", state=t("gateway.status.state_yes") if is_running else t("gateway.status.state_no")), ]) if queue_depth: lines.append(t("gateway.status.queued", count=queue_depth)) if source.platform == Platform.MATRIX: scope = getattr(self.adapters.get(Platform.MATRIX), "_matrix_session_scope", os.getenv("MATRIX_SESSION_SCOPE", "auto")) thread = source.thread_id or "none" lines.extend([ "", t("gateway.status.matrix_scope_header"), t("gateway.status.matrix_scope_room", room=source.chat_name or source.chat_id), t("gateway.status.matrix_scope_room_id", room_id=source.chat_id), t("gateway.status.matrix_scope_thread", thread_id=thread), t("gateway.status.matrix_scope_mode", scope=scope), t( "gateway.status.matrix_scope_key", session_key=self._redact_matrix_session_key(session_key), ), ]) lines.extend([ "", t("gateway.status.platforms", platforms=', '.join(connected_platforms)), ]) return "\n".join(lines) async def _status_session_db_facts(self, session_id: str): """``(title, session_row, db_total_tokens, persisted_route)`` for /status; each fail-open. Token totals come from the SQLite session DB rather than the in-memory SessionStore: the agent's per-turn token deltas are persisted into sessions_db (run_agent.py), not into SessionEntry, so session_entry.total_tokens is always 0. """ title = None session_row: dict[str, Any] = {} db_total_tokens = 0 persisted_route: dict[str, Any] = {} if not self._session_db: return title, session_row, db_total_tokens, persisted_route try: title = await self._session_db.get_session_title(session_id) except Exception: title = None try: row = await self._session_db.get_session(session_id) if isinstance(row, dict): session_row = row db_total_tokens = sum( _int_value(row.get(k)) for k in ("input_tokens", "output_tokens", "cache_read_tokens", "cache_write_tokens", "reasoning_tokens") ) except Exception: db_total_tokens = 0 try: route = await self._session_db.get_dominant_session_model_route(session_id) if isinstance(route, dict): persisted_route = route except Exception: persisted_route = {} return title, session_row, db_total_tokens, persisted_route @staticmethod def _redact_matrix_session_key(session_key: str) -> str: """Return a stable Matrix session-key fingerprint for shared room status.""" text = str(session_key or "") digest = hashlib.sha256(text.encode("utf-8")).hexdigest()[:12] return f"sha256:{digest}" async def _handle_context_command(self, event: MessageEvent) -> str: """Handle /context — the dedicated context-window view. /status shows a one-line ``used / total`` summary; this command is the deep view: a usage gauge, auto-compression threshold and headroom, compression count and last savings, and cumulative throughput — the last clearly labelled as throughput, NOT context size. Resolution order: running agent, cached agent, SessionStore/SessionDB metadata, and a transcript estimate only as last resort. ``/context all`` adds per-skill/toolset listings. """ source = event.source session_key = self._session_key_for_source(source) session_entry = await self.async_session_store.get_or_create_session(source) expanded = event.get_command_args().strip().lower() in {"all", "full", "details"} # Running agent first (mid-turn), then cached agent (between turns). agent = self._resident_agent_for(session_key) has_agent = bool(agent) ctx = getattr(agent, "context_compressor", None) if has_agent else None used, context_length, model_name = await self._resolve_context_figures( agent if has_agent else None, ctx, session_entry, source ) # Gauge path: real current-context figure if used > 0 and context_length > 0: pct = min(100.0, used / context_length * 100) headroom = max(0, context_length - used) BAR_WIDTH = 24 filled = int(round(pct / 100 * BAR_WIDTH)) bar = "█" * max(0, filled) + "░" * max(0, BAR_WIDTH - filled) lines = [ t("gateway.context.header"), "", t("gateway.context.model", model=model_name or "?"), t("gateway.context.window", total=f"{context_length:,}"), t( "gateway.context.in_use", used=f"{used:,}", total=f"{context_length:,}", pct=f"{pct:.0f}", ), t("gateway.context.bar", bar=bar), t("gateway.context.headroom", headroom=f"{headroom:,}"), "", ] # Full view — compression / throughput need the live agent. if ctx is not None: lines.extend(_context_compressor_lines(agent, ctx, used)) else: lines.append(t("gateway.context.detail_after_first")) # Per-category estimated breakdown (+ optional expanded listings). Same chars/4 engine # the desktop popover and /usage use; plain text (no glyph grid — monospace isn't # guaranteed on messaging platforms). Fail-open: rendering errors never break /context. if has_agent: breakdown = await asyncio.to_thread( self._context_breakdown_block, agent, source, expanded ) if breakdown: lines.append("") lines.extend(breakdown) return "\n".join(lines) # Last resort: rough estimate from transcript history = await self.async_session_store.load_transcript(session_entry.session_id) if history: from agent.model_metadata import estimate_messages_tokens_rough msgs = [ m for m in history if m.get("role") in {"user", "assistant"} and m.get("content") ] approx = estimate_messages_tokens_rough(msgs) return "\n".join( [ t("gateway.context.header"), "", t( "gateway.context.estimated", count=f"{approx:,}", messages=len(msgs), ), t("gateway.context.detail_after_first"), ] ) return t("gateway.context.no_data") async def _resolve_context_figures(self, agent, ctx, session_entry, source): """``(used, context_length, model_name)`` for /context with cascading fallbacks. used : compressor.last_prompt_tokens -> SessionStore.last_prompt_tokens model : agent.model -> SessionDB row model window: compressor.context_length -> effective gateway model route -> model metadata """ used = context_length = 0 if ctx is not None: used = getattr(ctx, "last_prompt_tokens", 0) or 0 context_length = getattr(ctx, "context_length", 0) or 0 model_name = _clean_str(getattr(agent, "model", "")) if agent is not None else "" if not used: used = _int_value(getattr(session_entry, "last_prompt_tokens", 0)) if not model_name and self._session_db: try: row = await self._session_db.get_session(session_entry.session_id) or {} if isinstance(row, dict): model_name = _clean_str(row.get("model", "")) except Exception: model_name = "" if not context_length: try: from gateway.run import _profile_runtime_scope, _resolve_gateway_model_context def _resolve_nonresident_context(): if getattr(getattr(self, "config", None), "multiplex_profiles", False): profile_home = self._resolve_profile_home_for_source(source) with _profile_runtime_scope(profile_home): return _resolve_gateway_model_context(model_name or None) return _resolve_gateway_model_context(model_name or None) resolved = await asyncio.to_thread(_resolve_nonresident_context) model_name = model_name or resolved.model context_length = _int_value(resolved.context_length) except Exception: context_length = 0 if not context_length and model_name: try: from agent.model_metadata import get_model_context_length context_length = _int_value(await asyncio.to_thread(get_model_context_length, model_name)) except Exception: context_length = 0 return used, context_length, model_name async def _handle_agents_command(self, event: MessageEvent) -> str: """Handle /agents command - list active agents and running tasks.""" from gateway.run import _AGENT_PENDING_SENTINEL from tools.process_registry import format_uptime_short, process_registry now = time.time() current_session_key = self._session_key_for_source(event.source) running_agents: dict = getattr(self, "_running_agents", {}) or {} running_started: dict = getattr(self, "_running_agents_ts", {}) or {} agent_rows: list[dict] = [] for session_key, agent in running_agents.items(): started = float(running_started.get(session_key, now)) elapsed = max(0, int(now - started)) is_pending = agent is _AGENT_PENDING_SENTINEL agent_rows.append( { "session_key": session_key, "elapsed": elapsed, "state": t("gateway.agents.state_starting") if is_pending else t("gateway.agents.state_running"), "session_id": "" if is_pending else str(getattr(agent, "session_id", "") or ""), "model": "" if is_pending else str(getattr(agent, "model", "") or ""), } ) agent_rows.sort(key=lambda row: row["elapsed"], reverse=True) running_processes: list[dict] = [] try: running_processes = [ p for p in process_registry.list_sessions() if p.get("status") == "running" ] except Exception: running_processes = [] background_tasks = [ t for t in (getattr(self, "_background_tasks", set()) or set()) if hasattr(t, "done") and not t.done() ] lines = [ t("gateway.agents.header"), "", t("gateway.agents.active_agents", count=len(agent_rows)), ] if agent_rows: for idx, row in enumerate(agent_rows[:12], 1): current = t("gateway.agents.this_chat") if row["session_key"] == current_session_key else "" sid = f" · `{row['session_id']}`" if row["session_id"] else "" model = f" · `{row['model']}`" if row["model"] else "" lines.append( f"{idx}. `{row['session_key']}` · {row['state']} · " f"{format_uptime_short(row['elapsed'])}{sid}{model}{current}" ) if len(agent_rows) > 12: lines.append(t("gateway.agents.more", count=len(agent_rows) - 12)) lines.extend( [ "", t("gateway.agents.running_processes", count=len(running_processes)), ] ) if running_processes: for proc in running_processes[:12]: cmd = " ".join(str(proc.get("command", "")).split()) if len(cmd) > 90: cmd = cmd[:87] + "..." lines.append( f"- `{proc.get('session_id', '?')}` · " f"{format_uptime_short(int(proc.get('uptime_seconds', 0)))} · `{cmd}`" ) if len(running_processes) > 12: lines.append(t("gateway.agents.more", count=len(running_processes) - 12)) lines.extend( [ "", t("gateway.agents.async_jobs", count=len(background_tasks)), ] ) # Background (async) delegations — delegate_task(background=true). try: from tools.async_delegation import list_async_delegations delegations = [ d for d in list_async_delegations() if d.get("status") in ("running", "stalling", "finalizing") ] except Exception: delegations = [] if delegations: lines.extend(["", t("gateway.agents.background_delegations", count=len(delegations))]) for d in delegations[:12]: lines.extend(_agents_delegation_lines(d)) if len(delegations) > 12: lines.append(t("gateway.agents.more", count=len(delegations) - 12)) if ( not agent_rows and not running_processes and not background_tasks and not delegations ): lines.append("") lines.append(t("gateway.agents.none")) return "\n".join(lines) async def _handle_topup_command(self, event: MessageEvent) -> str: """Handle /topup -- show the Nous balance and hand off to the portal. Does NOT charge, confirm, or track payment — that happens in the browser; the next /topup shows the new balance. Fetched off the event loop; fail-open. """ from agent.account_usage import build_credits_view try: view = await asyncio.to_thread(build_credits_view, markdown=True) except Exception: view = None if view is None or not view.logged_in: return t("gateway.credits.not_logged_in") lines: list[str] = ["💳 **Nous balance**"] for line in view.balance_lines: if line.lstrip().startswith("📈"): continue # drop the helper's header; we print our own lines.append(line) if view.identity_line: lines.append("") lines.append(view.identity_line) if view.topup_url: lines.append("") lines.append(f"Manage billing on the portal: {view.topup_url}") lines.append("Top up and manage billing in the browser — your balance updates here after.") return "\n".join(lines) def _context_breakdown_block(self, agent, source, expanded: bool) -> list[str]: """Render the /context per-category block (plain text, no grid). Estimated (chars/4), same engine as /usage. Runs in a thread; returns [] and never raises. """ try: from agent.context_breakdown import compute_context_details, render_context_breakdown_lines payload = self._session_context_breakdown(agent, source) if not (payload.get("categories") or []): return [] details = None if expanded: try: details = compute_context_details(agent) except Exception: details = {"skills": [], "toolsets": []} return render_context_breakdown_lines(payload, details=details, grid=False) except Exception: return [] def _session_context_breakdown(self, agent, source) -> dict: """Per-category context estimate (chars/4) for *agent* over the session transcript (sync).""" from agent.context_breakdown import compute_session_context_breakdown history: list[dict] = [] try: entry = self.session_store.get_or_create_session(source) history = self.session_store.load_transcript(entry.session_id) or [] except Exception: history = [] return compute_session_context_breakdown(agent, history) def _context_breakdown_lines(self, agent, source) -> list[str]: """Render the per-category context breakdown for /usage. Estimated (chars/4). Returns [] and never raises so /usage stays robust. """ try: payload = self._session_context_breakdown(agent, source) categories = payload.get("categories") or [] if not categories: return [] total = payload.get("estimated_total") or 0 out = [t("gateway.usage.breakdown_header")] for cat in categories: tokens = int(cat.get("tokens") or 0) if tokens <= 0: continue cat_id = str(cat.get("id") or "") label = t(f"gateway.usage.breakdown_cat_{cat_id}") # Missing key → t() echoes the key back; fall back to the # English label the engine already provides. if label.endswith(f"breakdown_cat_{cat_id}"): label = str(cat.get("label") or cat_id) pct = round(tokens / total * 100) if total else 0 out.append( t("gateway.usage.breakdown_line", label=label, count=f"{tokens:,}", pct=pct) ) return out if len(out) > 1 else [] except Exception: return [] async def _handle_usage_command(self, event: MessageEvent) -> str: """Handle /usage command -- show token usage for the current session. Checks both _running_agents (mid-turn) and _agent_cache (between turns) so details are available whenever the user asks. """ source = event.source session_key = self._session_key_for_source(source) # `/usage reset [--force]` — redeem one banked Codex rate-limit reset # credit. Parsed before the display path so it never mixes with the # stats rendering below. raw_args = event.get_command_args().strip() args = [a.lower() for a in raw_args.split()] if raw_args else [] wants_reset = bool(args) and args[0] == "reset" if args and not wants_reset: return t("gateway.usage.unknown_subcommand", args=raw_args) # Running agent first (mid-turn), then cached agent (between turns). agent = self._resident_agent_for(session_key) # Resolve provider/base_url/api_key for the account-usage fetch. Prefer the live agent; fall # back to persisted billing data on the SessionDB row so `/usage` still returns account info # between turns when no agent is resident. provider = getattr(agent, "provider", None) if agent else None base_url = getattr(agent, "base_url", None) if agent else None api_key = getattr(agent, "api_key", None) if agent else None if not provider and getattr(self, "_session_db", None) is not None: provider, base_url = await self._persisted_billing_route(source) if wants_reset: normalized_provider = str(provider or "").strip().lower() if normalized_provider != "openai-codex": return t("gateway.usage.reset_wrong_provider") force = "--force" in args[1:] from agent.account_usage import redeem_codex_reset_credit result = await asyncio.to_thread( redeem_codex_reset_credit, base_url=base_url, api_key=api_key, force=force, ) return result.message # Fetch account usage off the event loop so slow provider APIs don't # block the gateway. Failures are non-fatal -- account_lines stays []. account_lines: list[str] = [] credits_lines: list[str] = [] if provider: try: account_snapshot = await asyncio.to_thread( fetch_account_usage, provider, base_url=base_url, api_key=api_key, ) except Exception: account_snapshot = None if account_snapshot: account_lines = render_account_usage_lines(account_snapshot, markdown=True) # ── Nous credits magnitudes + monthly-grant % gauge ───────────── # Shared with CLI/TUI via nous_credits_lines(); run off the event loop. Gates on "a Nous # account is logged in" — NOT the inference provider, NOT under `if provider:` — so a Nous # user inferring elsewhere still sees a balance. No recovery trigger; fail-open. try: from agent.account_usage import nous_credits_lines credits_lines = await asyncio.to_thread(nous_credits_lines, markdown=True) except Exception: credits_lines = [] # fail-open: never break /usage def _with_account_blocks(lines: list[str]) -> str: # Each block is preceded by a blank divider only when something precedes it. for block in (account_lines, credits_lines): if block: if lines: lines.append("") lines.extend(block) return "\n".join(lines) if agent and hasattr(agent, "session_total_tokens") and agent.session_api_calls > 0: lines = _usage_agent_stats_lines(agent) # Per-category context breakdown (estimated — chars/4 heuristic). Same engine the # desktop popover uses. The system prompt / tools / skills / memory slices read off the # live agent; the conversation slice is estimated from the session transcript. breakdown_lines = await asyncio.to_thread(self._context_breakdown_lines, agent, source) if breakdown_lines: lines.append("") lines.extend(breakdown_lines) return _with_account_blocks(lines) # No agent at all -- check session history for a rough count session_entry = await self.async_session_store.get_or_create_session(source) history = await self.async_session_store.load_transcript(session_entry.session_id) if history: from agent.model_metadata import estimate_messages_tokens_rough msgs = [m for m in history if m.get("role") in {"user", "assistant"} and m.get("content")] approx = estimate_messages_tokens_rough(msgs) return _with_account_blocks([ t("gateway.usage.header_session_info"), t("gateway.usage.label_messages", count=len(msgs)), t("gateway.usage.label_estimated_context", count=f"{approx:,}"), t("gateway.usage.detailed_after_first"), ]) if account_lines or credits_lines: return _with_account_blocks([]) return t("gateway.usage.no_data") async def _persisted_billing_route(self, source): """``(provider, base_url)`` from the SessionDB row / dominant route when no agent is resident.""" try: entry = await self.async_session_store.get_or_create_session(source) persisted = await self._session_db.get_session(entry.session_id) or {} route = await self._session_db.get_dominant_session_model_route(entry.session_id) persisted_route = route if isinstance(route, dict) else {} except Exception: persisted = {} persisted_route = {} if persisted_route.get("billing_provider"): return persisted_route["billing_provider"], persisted_route.get("billing_base_url") return persisted.get("billing_provider"), persisted.get("billing_base_url") async def _handle_insights_command(self, event: MessageEvent) -> str: """Handle /insights command -- show usage insights and analytics.""" args = event.get_command_args().strip() # Normalize Unicode dashes (Telegram/iOS auto-converts -- to em/en dash) args = re.sub(r'[\u2012\u2013\u2014\u2015](days|source)', r'--\1', args) days = 30 source = None # Parse simple args: /insights 7 or /insights --days 7 if args: parts = args.split() i = 0 while i < len(parts): if parts[i] == "--days" and i + 1 < len(parts): try: days = int(parts[i + 1]) except ValueError: return t("gateway.insights.invalid_days", value=parts[i + 1]) i += 2 elif parts[i] == "--source" and i + 1 < len(parts): source = parts[i + 1] i += 2 elif parts[i].isdigit(): days = int(parts[i]) i += 1 else: i += 1 try: from hermes_state import get_shared_session_db from agent.insights import InsightsEngine def _run_insights(): db = get_shared_session_db() try: engine = InsightsEngine(db) report = engine.generate(days=days, source=source) result = engine.format_gateway(report) return result finally: from hermes_state import release_or_close release_or_close(db) # Not a bare hop: ``SessionDB()`` resolves ``get_hermes_home()`` at call time, which is # a contextvar set by ``_profile_runtime_scope``; a default-executor hop starts with an # EMPTY context and would read the DEFAULT profile's state.db. return await self._run_in_executor_with_context(_run_insights) except Exception as e: logger.error("Insights command error: %s", e, exc_info=True) return t("gateway.insights.error", error=e)