"""Honcho memory plugin — MemoryProvider for Honcho AI-native memory. Cross-session user modeling with dialectic Q&A, semantic search, peer cards and persistent conclusions via the Honcho SDK. Five tools (profile, search, reasoning, context, conclude) are exposed through the MemoryProvider interface. Config chain: $HERMES_HOME/honcho.json (profile-scoped) -> ~/.honcho/config.json (legacy global) -> environment variables. """ from __future__ import annotations import json import logging import re import threading import time from typing import Any, Callable, Dict, List, Optional from agent.memory_manager import sanitize_context from agent.memory_provider import MemoryProvider, is_trivial_prompt from plugins.memory.honcho.client import spawn_context_thread from plugins.memory.honcho.dialectic import DialecticMixin from plugins.memory.honcho.tool_schemas import ( # noqa: F401 — re-exported ALL_TOOL_SCHEMAS, CONCLUDE_SCHEMA, CONTEXT_SCHEMA, PROFILE_SCHEMA, REASONING_SCHEMA, SEARCH_SCHEMA, ) from tools.registry import tool_error logger = logging.getLogger(__name__) # Gateway-internal notifications arrive through the same user-role channel as genuine # user messages; they are execution metadata and must never become durable memory. # Deliberately anchored: a human discussing one of these strings mid-message is valid input. _INTERNAL_GATEWAY_TURN_RE = re.compile( r"^\s*(?:" r"\[ASYNC (?:DELEGATION )?(?:BATCH )?COMPLETE[^\]]*\]|" r"\[CONTEXT COMPACTION[^\]]*\]|" r"\[CONTEXT SUMMARY\]:?|" r"\[PRIOR CONTEXT[^\]]*\]|" r"\[Your active task list was preserved across context compression\]|" r"\[IMPORTANT: Background process \d+ matched watch pattern[^\n]*|" r"A background fan-out of \d+ subagent\(s\) you dispatched earlier has finished\.|" r"A background subagent you dispatched earlier has finished\." r")", re.IGNORECASE, ) def _is_internal_gateway_turn(text: str) -> bool: """Return True for machine-generated gateway/delegation notifications.""" return bool(_INTERNAL_GATEWAY_TURN_RE.match(text or "")) def _cfg_usable(cfg) -> bool: """Enabled with a credential or a self-hosted URL to talk to.""" return bool(cfg.enabled and (cfg.api_key or cfg.base_url)) # Static per-mode system prompt text (prompt-cache friendly: never changes between turns). _TOOL_GUIDE = ( "Use honcho_profile for a quick factual snapshot, " "honcho_search for raw excerpts, honcho_context for raw peer context, " "honcho_reasoning for synthesized answers (pass reasoning_level " "minimal/low/medium/high/max — you pick the depth per call), " "honcho_conclude to save facts about the user." ) _PROMPT_HEADERS = { "context": ( "# Honcho Memory\nActive (context-injection mode). Relevant user context is automatically " "injected before each turn. No memory tools are available — context is managed automatically." ), "tools": ( f"# Honcho Memory\nActive (tools-only mode). {_TOOL_GUIDE} " "No automatic context injection — you must use tools to access memory." ), "hybrid": ( "# Honcho Memory\nActive (hybrid mode). Relevant context is auto-injected AND memory tools " f"are available. {_TOOL_GUIDE}" ), } # (context key, section header) for the injected base-context block, in display order. _CONTEXT_SECTIONS = ( ("summary", "Session Summary"), ("representation", "User Representation"), ("card", "User Peer Card"), ("ai_representation", "AI Self-Representation"), ("ai_card", "AI Identity Card"), ) _PREWARM_QUERY = "Summarize what you know about this user. Focus on preferences, current projects, and working style." class HonchoMemoryProvider(DialecticMixin, MemoryProvider): """Honcho AI-native memory with dialectic Q&A and persistent user modeling.""" def backup_paths(self) -> List[str]: """Whole ~/.honcho dir (peer/session config when no profile-local honcho.json exists).""" try: from .client import resolve_global_config_path return [str(resolve_global_config_path().parent)] except Exception: return [] def __init__(self, query_rewriter: Optional[Callable[[str], str]] = None): self._manager = None # HonchoSessionManager self._config = None # HonchoClientConfig self._session_key = "" self._query_rewriter = query_rewriter self._prefetch_result = "" self._prefetch_lock = threading.Lock() self._prefetch_thread: Optional[threading.Thread] = None self._sync_thread: Optional[threading.Thread] = None self._memwrite_thread: Optional[threading.Thread] = None self._recall_mode = "hybrid" # "context", "tools", or "hybrid" # Base context cache — refreshed on context_cadence, not frozen self._base_context_cache: Optional[str] = None self._base_context_lock = threading.Lock() # Recall cadence state (overwritten from config in initialize()). self._turn_count = 0 self._query_rewrite_enabled = False self._injection_frequency = "every-turn" # or "first-turn" self._context_cadence = 1 # minimum turns between context API calls self._dialectic_cadence = 1 # backwards-compat fallback; wizard writes 2 on new configs self._dialectic_depth = 1 # .chat() calls per dialectic cycle (1-3) self._dialectic_depth_levels: list[str] | None = None # per-pass reasoning levels self._reasoning_heuristic: bool = True # scale base level by query length self._reasoning_level_cap: str = "high" # ceiling for auto-selected level self._last_context_turn = self._last_dialectic_turn = -999 # Liveness state self._prefetch_thread_started_at: float = 0.0 # monotonic ts of current thread self._prefetch_result_fired_at: int = -999 # turn the pending result was fired at self._dialectic_empty_streak: int = 0 # consecutive empty returns # Tools-only mode may defer session initialization until a tool call. self._session_initialized = False self._lazy_init_kwargs: Optional[dict] = None self._lazy_init_session_id: Optional[str] = None self._init_thread: Optional[threading.Thread] = None self._init_lock = threading.Lock() # Init auth failures live here because the failed manager is discarded. self._init_auth_failure: Optional[str] = None self._init_auth_notice_emitted = False # Cron and flush contexts disable the plugin entirely. self._cron_skipped = False @property def name(self) -> str: return "honcho" def is_available(self) -> bool: """Check if Honcho is configured. No network calls.""" try: from plugins.memory.honcho.client import HonchoClientConfig return _cfg_usable(HonchoClientConfig.from_global_config()) except Exception: return False def save_config(self, values, hermes_home): """Write config to $HERMES_HOME/honcho.json (Honcho SDK native format).""" from pathlib import Path from utils import atomic_json_write from plugins.memory.honcho.client import _read_config config_path = Path(hermes_home) / "honcho.json" try: existing = _read_config(config_path) except Exception: existing = {} atomic_json_write(config_path, {**existing, **values}, mode=0o600) def get_config_schema(self): return [ {"key": "api_key", "description": "Honcho API key", "secret": True, "env_var": "HONCHO_API_KEY", "url": "https://app.honcho.dev"}, {"key": "baseUrl", "description": "Honcho base URL (for self-hosted)"}, ] def post_setup(self, hermes_home: str, config: dict) -> None: """Run the full Honcho setup wizard after provider selection.""" import types from plugins.memory.honcho.cli import cmd_setup cmd_setup(types.SimpleNamespace()) # ----- Session lifecycle ----- def initialize(self, session_id: str, **kwargs) -> None: """Configure recall settings and start (or defer) Honcho session creation.""" try: agent_context, platform = kwargs.get("agent_context", ""), kwargs.get("platform", "cli") if agent_context in {"cron", "flush"} or platform == "cron": logger.debug("Honcho skipped: cron/flush context (agent_context=%s, platform=%s)", agent_context, platform) self._cron_skipped = True return from plugins.memory.honcho.client import HonchoClientConfig, get_honcho_client # noqa: F401 — ImportError probe from plugins.memory.honcho.session import HonchoSessionManager # noqa: F401 cfg = HonchoClientConfig.from_global_config() if not _cfg_usable(cfg): logger.debug("Honcho not configured — plugin inactive") return self._config = cfg self._recall_mode = cfg.recall_mode logger.debug("Honcho recall_mode: %s", self._recall_mode) for name in ("injection_frequency", "context_cadence", "dialectic_cadence", "dialectic_depth_levels", "reasoning_heuristic"): setattr(self, f"_{name}", getattr(cfg, name)) self._query_rewrite_enabled = cfg.query_rewrite self._FIRST_TURN_BASE_TIMEOUT = cfg.first_turn_base_wait self._FIRST_TURN_DIALECTIC_CAP = cfg.first_turn_dialectic_wait self._dialectic_depth = max(1, min(cfg.dialectic_depth, 3)) if cfg.reasoning_level_cap in self._LEVEL_ORDER: self._reasoning_level_cap = cfg.reasoning_level_cap # aiPeer comes from honcho.json only; SOUL.md is persona content, not identity config. self._lazy_init_kwargs = dict(kwargs) self._lazy_init_session_id = session_id self._session_key = self._resolve_session_key(cfg, session_id, **kwargs) # Session creation can block on Honcho/DB outages, so context/hybrid startup # fails open in a background thread. Tools-only mode has an explicit contract: # init_on_session_start=False stays lazy until the first tool call, True is eager. if self._recall_mode == "tools": if cfg.init_on_session_start: self._ensure_session() else: logger.debug("Honcho tools-only mode — deferring session init until first tool call") return self._start_session_init_background(wait_timeout=0.1) except ImportError: logger.debug("honcho-ai package not installed — plugin inactive") except Exception as e: logger.warning("Honcho init failed: %s", e) self._manager = None def _resolve_session_key(self, cfg, session_id: str, **kwargs) -> str: """Resolve the Honcho session key without touching the network.""" return ( cfg.resolve_session_name( session_title=kwargs.get("session_title"), session_id=session_id, gateway_session_key=kwargs.get("gateway_session_key"), ) or session_id or "hermes-default" ) def _can_start_init(self) -> bool: return not (self._cron_skipped or self._session_initialized) and bool(self._config) and self._lazy_init_kwargs is not None def _run_session_init(self, label: str) -> bool: """Run _do_session_init with the deferred kwargs; on failure discard the manager and (for auth failures) keep the detail for the one-time notice.""" from plugins.memory.honcho.session import HonchoAuthError init_kwargs = self._lazy_init_kwargs if init_kwargs is None: # another init path already consumed the deferred kwargs return self._manager is not None try: self._do_session_init(self._config, self._lazy_init_session_id or "hermes-default", **dict(init_kwargs)) self._lazy_init_kwargs = None self._lazy_init_session_id = None if self._init_auth_failure is not None: self._init_auth_failure = None self._init_auth_notice_emitted = False return True except Exception as e: self._manager = None self._session_initialized = False detail: object = e if isinstance(e, HonchoAuthError): # Keep the auth detail so the one-time notice survives the manager discard. self._init_auth_failure = str(e) detail = "authentication rejected" logger.warning("Honcho %s session init failed: %s", label, detail) return False def _start_session_init_background(self, *, wait_timeout: float = 0.0) -> None: """Start session initialization in a daemon thread so a slow/down Honcho can't block agent construction or first prompt assembly. ``wait_timeout`` lets fast (mock) initializations finish before returning.""" if not self._can_start_init(): return with self._init_lock: if not self._can_start_init() or (self._init_thread and self._init_thread.is_alive()): return self._init_thread = spawn_context_thread( lambda: self._run_session_init("background"), name="honcho-session-init", ) self._init_thread.start() if wait_timeout > 0: self._init_thread.join(timeout=wait_timeout) def _ensure_session(self) -> bool: """Lazily initialize the Honcho session (tools-only mode). True when the manager is ready.""" if self._manager and self._session_initialized: return True if not self._can_start_init() or (self._init_thread and self._init_thread.is_alive()): return False return self._run_session_init("lazy") and self._manager is not None def _do_session_init(self, cfg, session_id: str, **kwargs) -> None: """Shared session initialization for both eager and lazy paths.""" from plugins.memory.honcho.client import get_honcho_client from plugins.memory.honcho.session import HonchoSessionManager self._manager = HonchoSessionManager( honcho=get_honcho_client(cfg), config=cfg, context_tokens=cfg.context_tokens, runtime_user_peer_name=kwargs.get("user_id") or None, runtime_user_peer_name_alt=kwargs.get("user_id_alt") or None, ) self._session_key = self._resolve_session_key(cfg, session_id, **kwargs) logger.debug("Honcho session key resolved: %s", self._session_key) # The provider is not "ready" until this method returns: background startup sets # _manager before get_or_create/migration/prewarm finish, and lifecycle hooks must # not treat that partially initialized state as usable. session = self._manager.get_or_create(self._session_key) # Per-session strategy creates a fresh Honcho session every run, so a per-run # MEMORY.md/USER.md/SOUL.md upload would flood the backend with duplicates. if cfg.session_strategy == "per-session": logger.debug( "Honcho memory file migration skipped: per-session strategy creates a fresh session per run (%s)", self._session_key, ) elif not session.messages: try: from hermes_constants import get_hermes_home self._manager.migrate_memory_files(self._session_key, str(get_hermes_home() / "memories")) logger.debug("Honcho memory file migration attempted for new session: %s", self._session_key) except Exception as e: logger.debug("Honcho memory file migration skipped: %s", e) # Generic dialectic prewarm is incompatible with latest-message query rewriting, # which needs the first substantive user message. if self._recall_mode in {"context", "hybrid"}: if self._query_rewriter is None or not self._query_rewrite_enabled: self._spawn_dialectic( _PREWARM_QUERY, thread_name="honcho-prewarm-dialectic", fired_at=0, log_label="dialectic prewarm", use_query_rewrite=False, ) logger.debug("Honcho dialectic prewarm started for session: %s", self._session_key) else: logger.debug("Honcho generic dialectic prewarm skipped: awaiting first user message") self._session_initialized = True def _session_ready(self) -> bool: """Whether the manager/session key can be used safely. Background init sets ``_manager`` before get-or-create completes, so ``_session_initialized`` is the real guard; tests/legacy construction may inject a ready manager without the flag — allow that only with no init thread in flight. """ if not self._manager or not self._session_key: return False if self._session_initialized: return True return not (self._init_thread and self._init_thread.is_alive()) def _writes_enabled(self) -> bool: """``saveMessages`` is the operator's hard write gate for every Honcho mutation path.""" return not self._cron_skipped and getattr(self._config, "save_messages", True) def _ready_or_kick_init(self) -> bool: """True when writes may proceed; otherwise (outside tools mode) start background init.""" if self._session_ready(): return True if self._recall_mode != "tools": self._start_session_init_background() return False # ----- Prompt / prefetch ----- def _format_first_turn_context(self, ctx: dict) -> str: """Format the prefetch context dict into a readable system prompt block.""" parts = [f"## {header}\n{ctx.get(key, '')}" for key, header in _CONTEXT_SECTIONS if ctx.get(key, "")] return "\n\n".join(parts) def system_prompt_block(self) -> str: """Static mode header + tool instructions (prompt-cache friendly). Live context (representation, card) is injected via prefetch().""" if self._cron_skipped or not (self._config or (self._manager and self._session_key)): return "" return _PROMPT_HEADERS.get(self._recall_mode, _PROMPT_HEADERS["hybrid"]) def _first_turn_wait(self, base: float) -> float: """Turn-1 wait budget: a short request timeout may tighten, but never expand, it.""" request_timeout = getattr(self._config, "timeout", None) if request_timeout is not None: base = min(base, max(0.0, request_timeout)) return max(0.0, base) def _fetch_base_context_layer(self, query: str, first_turn_base_deadline: float | None) -> str: """Layer 1: representation + card. The first fetch gets the remaining turn-1 budget; later turns consume the refresh queued by the previous turn.""" with self._base_context_lock: first_base_fetch = self._base_context_cache is None if first_base_fetch: self._base_context_cache = "" self._last_context_turn = self._turn_count base_context = self._base_context_cache if not self._manager: return base_context def _adopt(ctx: dict) -> str: """Cache a fresh context dict's formatted block; keep the old text if it formats empty.""" formatted = self._format_first_turn_context(ctx) if formatted: with self._base_context_lock: self._base_context_cache = formatted return formatted or base_context if not first_base_fetch: fresh_ctx = self._manager.pop_context_result(self._session_key) return _adopt(fresh_ctx) if fresh_ctx else base_context ctx_holder: dict[str, dict] = {} def _fetch_base() -> None: ctx = self._manager.get_prefetch_context(self._session_key, query or None) or {} ctx_holder["ctx"] = ctx if ctx: self._manager.set_context_result(self._session_key, ctx) bt = self._spawn_write(_fetch_base, "honcho-base-first", "Honcho first-turn base context failed: %s") base_wait = max(0.0, first_turn_base_deadline - time.monotonic()) if first_turn_base_deadline is not None else 0.0 bt.join(timeout=base_wait) ctx = ctx_holder.get("ctx") if ctx: self._manager.pop_context_result(self._session_key) return _adopt(ctx) if bt.is_alive(): logger.debug("Honcho first-turn base context still running after %.1fs — will surface on next turn", base_wait) return base_context def _first_turn_dialectic_wait(self, query: str) -> None: """Turn 1 only: reuse an in-flight prewarm or start one dialectic, then wait briefly. Unfinished work stays async and surfaces on a later turn.""" with self._prefetch_lock: prewarm_landed = bool(self._prefetch_result) if prewarm_landed and self._last_dialectic_turn == -999: self._last_dialectic_turn = self._turn_count if self._last_dialectic_turn != -999 or not query: return dia_wait = self._first_turn_wait(self._FIRST_TURN_DIALECTIC_CAP) if not self._thread_is_live(): self._spawn_dialectic( query, thread_name="honcho-prefetch-first", fired_at=self._turn_count, log_label="first-turn dialectic", ) live = self._prefetch_thread if live is not None: live.join(timeout=dia_wait) if self._prefetch_thread and self._prefetch_thread.is_alive(): logger.debug("Honcho first-turn dialectic still running after %.1fs — will surface on next turn", dia_wait) def prefetch(self, query: str, *, session_id: str = "") -> str: """Base context (representation + card, refreshed on context_cadence) plus the dialectic supplement (refreshed on dialectic_cadence), within the context budget. Empty in tools-only mode.""" if self._cron_skipped or self._recall_mode == "tools": return "" first_turn_base_deadline = ( time.monotonic() + self._first_turn_wait(self._FIRST_TURN_BASE_TIMEOUT) if self._turn_count <= 1 else None ) if not self._session_ready(): # Only turn 1 may wait for session init; later turns fail open. self._start_session_init_background() if first_turn_base_deadline is not None and self._init_thread is not None: self._init_thread.join(timeout=max(0.0, first_turn_base_deadline - time.monotonic())) if not self._session_ready(): # A failed auth init still owes the user the one-time notice. return self._pop_auth_notice() # Trivial turns start no work, but may consume a ready pending result. if self._is_trivial_prompt(query): ready = self._consume_pending_dialectic() return self._truncate_to_budget(ready) if ready else "" # One-time notice, relayed by the model, that auth is dead and memory is paused. parts = [self._pop_auth_notice()] # First-turn mode suppresses only the base layer; dialectic is independent. if not (self._injection_frequency == "first-turn" and self._turn_count > 1): parts.append(self._fetch_base_context_layer(query, first_turn_base_deadline)) self._first_turn_dialectic_wait(query) # Consume only results that are already ready; later turns never wait. parts.append(self._consume_pending_dialectic()) parts = [p for p in parts if p and p.strip()] return self._truncate_to_budget("\n\n".join(parts)) if parts else "" def _pop_auth_notice(self) -> str: """One-time model-facing notice that Honcho auth expired and memory is paused.""" # getattr (not a direct call): test fakes install minimal managers without pop_auth_notice. pop = getattr(self._manager, "pop_auth_notice", None) msg = pop() if callable(pop) else None if not isinstance(msg, str) or not msg: # Init failures discard the manager; the provider kept the detail. if self._init_auth_failure is None or self._init_auth_notice_emitted: return "" self._init_auth_notice_emitted = True msg = self._init_auth_failure return ( "[Honcho memory status] Authentication with the Honcho memory backend has expired and automatic " f"token refresh failed, so memory sync and recall are paused. Reason: {msg}\n" "Tell the user (once) that Honcho memory is paused and that running 'hermes honcho setup' " "to re-authenticate will restore it." ) def _truncate_to_budget(self, text: str) -> str: """Truncate text to the context_tokens budget (≈4 chars/token) at a word boundary.""" if not self._config or not self._config.context_tokens: return text budget_chars = self._config.context_tokens * 4 if len(text) <= budget_chars: return text truncated = text[:budget_chars] last_space = truncated.rfind(" ") if last_space > budget_chars * 0.8: truncated = truncated[:last_space] return truncated + " …" def queue_prefetch(self, query: str, *, session_id: str = "") -> None: """Fire background prefetch threads for the upcoming turn. Context and dialectic refreshes have independent cadence controls.""" if self._cron_skipped or self._recall_mode == "tools": return if not self._session_ready() or not query: self._start_session_init_background() return # Trivial prompts don't warrant either a context refresh or a dialectic call. if self._is_trivial_prompt(query): return # First-turn-only base context never needs a later refresh. context_due = self._context_cadence <= 1 or (self._turn_count - self._last_context_turn) >= self._context_cadence if self._injection_frequency != "first-turn" and context_due: self._last_context_turn = self._turn_count try: self._manager.prefetch_context(self._session_key, query) except Exception as e: logger.debug("Honcho context prefetch failed: %s", e) # Dialectic layer: a hung call older than timeout × multiplier counts as dead. if self._thread_is_live(): logger.debug("Honcho dialectic prefetch skipped: prior thread still running") return # Cadence gate, widened by the empty-streak backoff so a persistently silent # backend doesn't retry every turn forever. effective = self._effective_cadence() if (self._turn_count - self._last_dialectic_turn) < effective: logger.debug( "Honcho dialectic prefetch skipped: effective cadence %d " "(base %d, empty streak %d), turns since last: %d", effective, self._dialectic_cadence, self._dialectic_empty_streak, self._turn_count - self._last_dialectic_turn, ) return self._spawn_dialectic( query, thread_name="honcho-prefetch", fired_at=self._turn_count, log_label="prefetch", ) # Shared with the core prefetch gate so the two classifiers can never drift apart. _is_trivial_prompt = staticmethod(is_trivial_prompt) def on_turn_start(self, turn_number: int, message: str, **kwargs) -> None: """Track turn count for cadence and injection_frequency logic.""" self._turn_count = turn_number # ----- Writes ----- @staticmethod def _chunk_message(content: str, limit: int) -> list[str]: """Split content to fit the Honcho message limit, cutting at paragraph, then sentence, then word boundaries; continuation chunks get a "[continued] " prefix so Honcho's representation engine can reconstruct the full message.""" if len(content) <= limit: return [content] prefix = "[continued] " chunks = [] remaining = content first = True while remaining: effective = limit if first else limit - len(prefix) if len(remaining) <= effective: chunks.append(remaining if first else prefix + remaining) break segment = remaining[:effective] # Paragraph, then sentence (keeping ". "), then word boundary; else a hard cut. for sep in ("\n\n", ". ", " "): cut = segment.rfind(sep) if cut >= 0 and sep == ". ": cut += 2 if cut >= effective * 0.3: break else: cut = effective chunk = remaining[:cut].rstrip() remaining = remaining[cut:].lstrip() chunks.append(chunk if first else prefix + chunk) first = False return chunks def sync_turn(self, user_content: str, assistant_content: str, *, session_id: str = "") -> None: """Record the conversation turn in Honcho (non-blocking), chunking messages that exceed the Honcho API limit. Honors saveMessages: false.""" if not self._writes_enabled(): return if _is_internal_gateway_turn(user_content): logger.debug("Honcho sync skipped machine-generated gateway turn") return if not self._ready_or_kick_init(): return msg_limit = self._config.message_max_chars if self._config else 25000 clean_user_content = sanitize_context(user_content or "").strip() clean_assistant_content = sanitize_context(assistant_content or "").strip() # Skip only when the whole turn is empty: an interrupted or tool-only turn can have # an empty assistant side, and the user's message must still be persisted. if not clean_user_content and not clean_assistant_content: return def _sync(): session = self._manager.get_or_create(self._session_key) for role, content in (("user", clean_user_content), ("assistant", clean_assistant_content)): for chunk in self._chunk_message(content, msg_limit) if content else (): session.add_message(role, chunk) # save() (not _flush_session) so writeFrequency batching is honored. self._manager.save(session) if self._sync_thread and self._sync_thread.is_alive(): self._sync_thread.join(timeout=5.0) self._sync_thread = self._spawn_write(_sync, "honcho-sync", "Honcho sync_turn failed: %s") @staticmethod def _spawn_write(fn: Callable[[], None], name: str, fail_msg: str) -> threading.Thread: """Run a Honcho write off-thread; failures are debug-logged, never raised into the turn.""" def _run(): try: fn() except Exception as e: logger.debug(fail_msg, e) thread = spawn_context_thread(_run, name=name) thread.start() return thread def on_memory_write( self, action: str, target: str, content: str, metadata: Optional[Dict[str, Any]] = None, ) -> None: """Mirror built-in user-profile writes as Honcho conclusions (``metadata`` accepted for interface compatibility, not yet threaded into the conclusion payload).""" if action != "add" or target != "user" or not content: return if not self._writes_enabled() or not self._ready_or_kick_init(): return self._memwrite_thread = self._spawn_write( lambda: self._manager.create_conclusion(self._session_key, content), "honcho-memwrite", "Honcho memory mirror failed: %s", ) def on_session_end(self, messages: List[Dict[str, Any]]) -> None: """Flush all pending messages to Honcho on session end.""" if not self._writes_enabled() or not self._manager: return if not self._session_initialized and self._init_thread and self._init_thread.is_alive(): return if self._sync_thread and self._sync_thread.is_alive(): self._sync_thread.join(timeout=10.0) try: self._manager.flush_all() except Exception as e: logger.debug("Honcho session-end flush failed: %s", e) # ----- Tools ----- def get_tool_schemas(self) -> List[Dict[str, Any]]: """Tool schemas by recall_mode; context-only mode exposes no Honcho tools.""" if self._cron_skipped or self._recall_mode == "context": return [] return list(ALL_TOOL_SCHEMAS) def _empty_profile_hint(self, peer: str) -> Dict[str, Any]: """Diagnostic hint for an empty honcho_profile card, so the model can explain WHY instead of surfacing a cryptic "no facts" to the user. Likely causes, in order: observation disabled for the peer; card not accumulated yet (fresh peer / few dialectic cycles); self-hosted Honcho < 3.x without peer-card support.""" cfg = self._config reasons: List[str] = [] kind = "user" if peer == "user" else "ai" if cfg is not None and not ( getattr(cfg, f"{kind}_observe_me", True) or getattr(cfg, f"{kind}_observe_others", True) ): reasons.append(f"observation is disabled for peer '{peer}' (user_observe_me/ai_observe_me in config)") cadence, turn = self._dialectic_cadence, self._turn_count if turn < max(2, cadence): reasons.append( f"this session has only {turn} turn(s); peer cards accumulate as the dialectic " f"layer reasons over conversation history (cadence every {cadence} turn(s))" ) if not reasons: reasons.append( "peer card has no facts yet — Honcho's dialectic layer builds this over time from " "observed turns; self-hosted Honcho < 3.x does not support peer cards at all" ) return { "result": "No profile facts available yet.", "hint": ( "This is not an error. " + "; ".join(reasons) + ". Try honcho_reasoning for a synthesized answer, or honcho_search to query raw conversation excerpts." ), } def _tool_profile(self, args: dict) -> str: peer = args.get("peer", "user") card_update = args.get("card") if card_update: result = self._manager.set_peer_card(self._session_key, card_update, peer=peer) if result is None: return tool_error("Failed to update peer card.") return json.dumps({"result": f"Peer card updated ({len(result)} facts).", "card": result}) card = self._manager.get_peer_card(self._session_key, peer=peer) if not card: return json.dumps(self._empty_profile_hint(peer)) return json.dumps({"result": card}) def _tool_search(self, args: dict) -> str: query = (args.get("query") or "").strip() if not query: return tool_error("Missing required parameter: query") max_tokens = min(int(args.get("max_tokens", 800)), 2000) result = self._manager.search_context( self._session_key, query, max_tokens=max_tokens, peer=args.get("peer", "user"), ) return json.dumps({"result": result or "No relevant context found."}) def _tool_reasoning(self, args: dict) -> str: from plugins.memory.honcho.session import HonchoAuthError query = (args.get("query") or "").strip() if not query: return tool_error("Missing required parameter: query") try: result = self._manager.dialectic_query( self._session_key, query, reasoning_level=args.get("reasoning_level"), peer=args.get("peer", "user"), # Explicit reasoning bypasses the automatic-injection cap, and surfaces # timeouts/server errors as errors rather than an indistinguishable "no result". apply_injection_cap=False, raise_errors=True, ) except HonchoAuthError: raise # rendered by handle_tool_call's auth-specific handler except Exception as e: logger.warning("honcho_reasoning failed: %s", e) return tool_error( f"Honcho reasoning query failed ({e}). This is a backend error, not an empty result — " "the peer may still have relevant context. Slow dialectic calls at higher reasoning levels " "can exceed the configured timeout; consider a lower reasoning_level or raising the " "'timeout' value in honcho.json." ) # Auto-injection respects the cadence gap after an explicit call. self._last_dialectic_turn = self._turn_count return json.dumps({"result": result or "No result from Honcho."}) def _tool_context(self, args: dict) -> str: ctx = self._manager.get_session_context(self._session_key, peer=args.get("peer", "user")) if not ctx: return json.dumps({"result": "No context available yet."}) parts = [ f"## {header}\n{ctx[key]}" for key, header in (("summary", "Summary"), ("representation", "Representation"), ("card", "Card")) if ctx.get(key) ] if ctx.get("recent_messages"): msg_str = "\n".join(f" [{m['role']}] {m['content'][:200]}" for m in ctx["recent_messages"][-5:]) parts.append(f"## Recent messages\n{msg_str}") return json.dumps({"result": "\n\n".join(parts) or "No context available."}) def _tool_conclude(self, args: dict) -> str: delete_id = (args.get("delete_id") or "").strip() conclusion = args.get("conclusion", "").strip() list_mode = bool(args.get("list")) peer = args.get("peer", "user") if sum([bool(delete_id), bool(conclusion), list_mode]) != 1: return tool_error("Exactly one of conclusion, delete_id, or list must be provided.") query = (args.get("query") or "").strip() if query and not list_mode: return tool_error("query is only valid when list is true.") if list_mode: conclusions = self._manager.list_conclusions(self._session_key, query=query or None, peer=peer) return json.dumps({"conclusions": conclusions}) if delete_id: if self._manager.delete_conclusion(self._session_key, delete_id, peer=peer): return json.dumps({"result": f"Conclusion {delete_id} deleted."}) return tool_error(f"Failed to delete conclusion {delete_id}.") if self._manager.create_conclusion(self._session_key, conclusion, peer=peer): return json.dumps({"result": f"Conclusion saved for {peer}: {conclusion}"}) return tool_error("Failed to save conclusion.") _TOOL_HANDLERS = { "honcho_profile": _tool_profile, "honcho_search": _tool_search, "honcho_reasoning": _tool_reasoning, "honcho_context": _tool_context, "honcho_conclude": _tool_conclude, } def handle_tool_call(self, tool_name: str, args: dict, **kwargs) -> str: """Dispatch a Honcho tool call, lazily initializing the session in tools-only mode.""" from plugins.memory.honcho.session import HonchoAuthError if self._cron_skipped: return tool_error("Honcho is not active (cron context).") if not self._session_initialized: if self._init_thread and self._init_thread.is_alive(): return tool_error("Honcho session is still initializing; try again shortly.") if not self._ensure_session(): if self._init_auth_failure: return tool_error(f"Honcho memory authentication failed: {self._init_auth_failure}") return tool_error("Honcho session could not be initialized.") if not self._manager or not self._session_key: return tool_error("Honcho is not active for this session.") handler = self._TOOL_HANDLERS.get(tool_name) if handler is None: return tool_error(f"Unknown tool: {tool_name}") try: return handler(self, args) except HonchoAuthError as e: # Never report an auth failure as an empty result; the model would read it as "no memory". logger.error("Honcho tool %s failed: authentication rejected", tool_name) return tool_error(f"Honcho memory authentication failed: {e}") except Exception as e: logger.error("Honcho tool %s failed: %s", tool_name, e) return tool_error(f"Honcho {tool_name} failed: {e}") def shutdown(self) -> None: for t in (self._prefetch_thread, self._sync_thread, self._memwrite_thread): if t and t.is_alive(): t.join(timeout=5.0) manager = self._manager if not manager or (self._init_thread and self._init_thread.is_alive() and not self._session_initialized): return try: # saveMessages: false skips persistence, but the async-writer thread must still # be joined so daemon threads aren't left blocked in httpx I/O at interpreter exit. if getattr(self._config, "save_messages", True): manager.shutdown() # flush_all() + join the writer else: manager.stop_async_writer() except Exception: pass def register(ctx) -> None: """Register Honcho as a memory provider plugin.""" from plugins.memory.query_rewrite import rewrite_memory_query ctx.register_memory_provider( HonchoMemoryProvider(query_rewriter=rewrite_memory_query) )