refactor(memory/holographic): single extraction-pattern table, drop non-package import shims, compact schema literals and docstrings

This commit is contained in:
Teknium
2026-09-02 22:05:59 -07:00
parent 12e291b0f3
commit 58a323999c
3 changed files with 46 additions and 83 deletions
+29 -51
View File
@@ -25,10 +25,8 @@ logger = logging.getLogger(__name__)
FACT_STORE_SCHEMA = {
"name": "fact_store",
"description": (
"Deep structured memory with algebraic reasoning. "
"Use alongside the memory tool — memory for always-on context, "
"fact_store for deep recall and compositional queries.\n\n"
"ACTIONS (simple → powerful):\n"
"Deep structured memory with algebraic reasoning. Use alongside the memory tool — memory for always-on "
"context, fact_store for deep recall and compositional queries.\n\nACTIONS (simple → powerful):\n"
"• add — Store a fact the user would expect you to remember.\n"
"• search — Keyword lookup ('editor config', 'deploy process').\n"
"• probe — Entity recall: ALL facts about a person/thing.\n"
@@ -41,10 +39,7 @@ FACT_STORE_SCHEMA = {
"parameters": {
"type": "object",
"properties": {
"action": {
"type": "string",
"enum": ["add", "search", "probe", "related", "reason", "contradict", "update", "remove", "list"],
},
"action": {"type": "string", "enum": ["add", "search", "probe", "related", "reason", "contradict", "update", "remove", "list"]},
"content": {"type": "string", "description": "Fact content (required for 'add')."},
"query": {"type": "string", "description": "Search query (required for 'search')."},
"entity": {"type": "string", "description": "Entity name for 'probe'/'related'."},
@@ -62,31 +57,24 @@ FACT_STORE_SCHEMA = {
FACT_FEEDBACK_SCHEMA = {
"name": "fact_feedback",
"description": (
"Rate a fact after using it. Mark 'helpful' if accurate, 'unhelpful' if outdated. "
"This trains the memory — good facts rise, bad facts sink."
),
"description": ("Rate a fact after using it. Mark 'helpful' if accurate, 'unhelpful' if outdated. "
"This trains the memory — good facts rise, bad facts sink."),
"parameters": {
"type": "object",
"properties": {
"action": {"type": "string", "enum": ["helpful", "unhelpful"]},
"fact_id": {"type": "integer", "description": "The fact ID to rate."},
},
"properties": {"action": {"type": "string", "enum": ["helpful", "unhelpful"]},
"fact_id": {"type": "integer", "description": "The fact ID to rate."}},
"required": ["action", "fact_id"],
},
}
# Auto-extraction patterns (on_session_end): user preferences -> user_pref, decisions -> project.
_PREF_PATTERNS = [
re.compile(r'\bI\s+(?:prefer|like|love|use|want|need)\s+(.+)', re.IGNORECASE),
re.compile(r'\bmy\s+(?:favorite|preferred|default)\s+\w+\s+is\s+(.+)', re.IGNORECASE),
re.compile(r'\bI\s+(?:always|never|usually)\s+(.+)', re.IGNORECASE),
]
_DECISION_PATTERNS = [
re.compile(r'\bwe\s+(?:decided|agreed|chose)\s+(?:to\s+)?(.+)', re.IGNORECASE),
re.compile(r'\bthe\s+project\s+(?:uses|needs|requires)\s+(.+)', re.IGNORECASE),
]
_EXTRACT_CATEGORIES = ((_PREF_PATTERNS, "user_pref"), (_DECISION_PATTERNS, "project"))
# Auto-extraction (on_session_end): (patterns, category) — user preferences -> user_pref, decisions -> project.
_EXTRACT_CATEGORIES = (
([re.compile(r'\bI\s+(?:prefer|like|love|use|want|need)\s+(.+)', re.IGNORECASE),
re.compile(r'\bmy\s+(?:favorite|preferred|default)\s+\w+\s+is\s+(.+)', re.IGNORECASE),
re.compile(r'\bI\s+(?:always|never|usually)\s+(.+)', re.IGNORECASE)], "user_pref"),
([re.compile(r'\bwe\s+(?:decided|agreed|chose)\s+(?:to\s+)?(.+)', re.IGNORECASE),
re.compile(r'\bthe\s+project\s+(?:uses|needs|requires)\s+(.+)', re.IGNORECASE)], "project"),
)
def _load_plugin_config() -> dict:
@@ -144,9 +132,8 @@ class HolographicMemoryProvider(MemoryProvider):
def get_config_schema(self):
from hermes_constants import display_hermes_home
_default_db = f"{display_hermes_home()}/memory_store.db"
return [
{"key": "db_path", "description": "SQLite database path", "default": _default_db},
{"key": "db_path", "description": "SQLite database path", "default": f"{display_hermes_home()}/memory_store.db"},
{"key": "auto_extract", "description": "Auto-extract facts at session end", "default": "false", "choices": ["true", "false"]},
{"key": "default_trust", "description": "Default trust score for new facts", "default": "0.5"},
{"key": "hrr_dim", "description": "HRR vector dimensions", "default": "1024"},
@@ -160,10 +147,8 @@ class HolographicMemoryProvider(MemoryProvider):
db_path = db_path.replace("$HERMES_HOME", _hermes_home).replace("${HERMES_HOME}", _hermes_home)
hrr_dim = int(self._config.get("hrr_dim", 1024))
self._store = MemoryStore(db_path=db_path, default_trust=float(self._config.get("default_trust", 0.5)), hrr_dim=hrr_dim)
self._retriever = FactRetriever(
store=self._store, temporal_decay_half_life=int(self._config.get("temporal_decay_half_life", 0)),
hrr_weight=float(self._config.get("hrr_weight", 0.3)), hrr_dim=hrr_dim,
)
self._retriever = FactRetriever(store=self._store, hrr_dim=hrr_dim, hrr_weight=float(self._config.get("hrr_weight", 0.3)),
temporal_decay_half_life=int(self._config.get("temporal_decay_half_life", 0)))
self._session_id = session_id
def system_prompt_block(self) -> str:
@@ -173,13 +158,11 @@ class HolographicMemoryProvider(MemoryProvider):
total = self._store._conn.execute("SELECT COUNT(*) FROM facts").fetchone()[0]
except Exception:
total = 0
body = (
"Active. Empty fact store — proactively add facts the user would expect you to remember.\n"
"Use fact_store(action='add') to store durable structured facts about people, projects, preferences, decisions.\n"
) if total == 0 else (
f"Active. {total} facts stored with entity resolution and trust scoring.\n"
"Use fact_store to search, probe entities, reason across entities, or add facts.\n"
)
body = ("Active. Empty fact store — proactively add facts the user would expect you to remember.\n"
"Use fact_store(action='add') to store durable structured facts about people, projects, preferences, decisions.\n"
if total == 0 else
f"Active. {total} facts stored with entity resolution and trust scoring.\n"
"Use fact_store to search, probe entities, reason across entities, or add facts.\n")
return "# Holographic Memory\n" + body + "Use fact_feedback to rate facts after using them (trains trust scores)."
def prefetch(self, query: str, *, session_id: str = "") -> str:
@@ -221,15 +204,13 @@ class HolographicMemoryProvider(MemoryProvider):
logger.debug("Holographic memory_write mirror failed: %s", e)
def shutdown(self) -> None:
# Close on the caller's thread: leaving the shared connection (and its write lock) to GC keeps it
# alive on a long-running gateway. close() is idempotent and refcount-guarded.
# Close on the caller's thread: leaving the shared connection (+ write lock) to GC keeps it alive on a gateway.
if self._store is not None:
try:
self._store.close()
except Exception as e:
logger.debug("Holographic shutdown close() failed: %s", e)
self._store = None
self._retriever = None
self._store = self._retriever = None
# Tool handlers (self, args) -> str. KeyError from args[...] / Exception -> tool_error in handle_tool_call;
# argument coercion order (and therefore which error surfaces first) mirrors the underlying call order.
@@ -261,13 +242,10 @@ class HolographicMemoryProvider(MemoryProvider):
@staticmethod
def _harvestable_text(msg: dict):
"""User text eligible for extraction, or None. Compaction handoff summaries arrive as role="user" and
reliably match the decision patterns; never store the compactor's own output as a durable fact. A
merge-into-tail row holds genuine prior user text BEFORE _MERGED_SUMMARY_DELIMITER (prefixed with the
header) and the summary AFTER it — harvest only the pre-delimiter segment."""
# Local import: the compressor module is heavier than this plugin and only needed here.
from agent.context_compressor import _MERGED_PRIOR_CONTEXT_HEADER, _MERGED_SUMMARY_DELIMITER, is_compaction_summary_message
"""User text eligible for extraction, or None. Compaction handoff summaries arrive as role="user" and match
the decision patterns; never store the compactor's own output as a fact. A merge-into-tail row holds genuine
prior user text BEFORE _MERGED_SUMMARY_DELIMITER (after the header) and the summary AFTER — harvest only that."""
from agent.context_compressor import _MERGED_PRIOR_CONTEXT_HEADER, _MERGED_SUMMARY_DELIMITER, is_compaction_summary_message # heavy; lazy
content, pre = msg.get("content", ""), ""
if isinstance(content, str) and _MERGED_SUMMARY_DELIMITER in content:
pre = content.split(_MERGED_SUMMARY_DELIMITER, 1)[0].removeprefix(_MERGED_PRIOR_CONTEXT_HEADER).strip()
+11 -22
View File
@@ -11,28 +11,21 @@ from typing import TYPE_CHECKING
if TYPE_CHECKING:
from .store import MemoryStore
try:
from . import holographic as hrr
except ImportError:
import holographic as hrr # type: ignore[no-redef]
from . import holographic as hrr
_FACT_COLUMNS = (
"fact_id, content, category, tags, trust_score, "
"retrieval_count, helpful_count, created_at, updated_at"
)
_FACT_COLUMNS = "fact_id, content, category, tags, trust_score, retrieval_count, helpful_count, created_at, updated_at"
_ROLE_ENTITY, _ROLE_CONTENT = hrr.ROLE_ENTITY, hrr.ROLE_CONTENT
_PUNCT = ".,;:!?\"'()[]{}#@<>"
_FTS_OPERATORS = str.maketrans("", "", '"()*^:-+')
# Stopwords dropped before FTS5 OR-expansion: short English function words that
# carry no retrieval signal and force false-negative AND matches.
_FTS_STOPWORDS = frozenset("""
a about above after again all am an and any are as at be because been before being between both but by
can could did do does doing don down during each few for from further had has have having he her here hers
herself him himself his how i if in into is it its itself just me more most my myself no nor not now of off
on once only or other our ours ourselves out over own same she should so some such than that the their theirs
them themselves then there these they this those through to too under until up very was we were what when
where which while who whom why will with would you your yours yourself yourselves
""".split())
a about above after again all am an and any are as at be because been before being between both but by can could
did do does doing don down during each few for from further had has have having he her here hers herself him himself
his how i if in into is it its itself just me more most my myself no nor not now of off on once only or other our
ours ourselves out over own same she should so some such than that the their theirs them themselves then there these
they this those through to too under until up very was we were what when where which while who whom why will with
would you your yours yourself yourselves""".split())
def _shift(sim: float) -> float:
@@ -60,11 +53,9 @@ class FactRetriever:
"""Hybrid search: FTS5 candidates (limit*3) → Jaccard + HRR rerank → trust weighting →
optional temporal decay 0.5^(age_days / half_life). Returns fact dicts with 'score', sorted desc."""
candidates = self._fts_candidates(query, category, min_trust, limit * 3)
if not candidates:
return []
query_tokens = self._tokenize(query)
# Query vector is loop-invariant; encode lazily on the first candidate that carries
# an HRR vector so stores whose hrr_vector was never backfilled don't pay for it.
# Query vector is loop-invariant; encode lazily on the first candidate that carries an HRR vector
# so stores whose hrr_vector was never backfilled don't pay for it.
query_vec = None
for fact in candidates:
jaccard = self._jaccard_similarity(query_tokens, self._tokenize(fact["content"]) | self._tokenize(fact.get("tags", "")))
@@ -234,9 +225,7 @@ class FactRetriever:
return 1.0
try:
ts = datetime.fromisoformat(timestamp_str.replace("Z", "+00:00")) if isinstance(timestamp_str, str) else timestamp_str
if ts.tzinfo is None:
ts = ts.replace(tzinfo=timezone.utc)
age_days = (datetime.now(timezone.utc) - ts).total_seconds() / 86400
age_days = (datetime.now(timezone.utc) - (ts if ts.tzinfo else ts.replace(tzinfo=timezone.utc))).total_seconds() / 86400
return 1.0 if age_days < 0 else math.pow(0.5, age_days / self.half_life)
except (ValueError, TypeError):
return 1.0
+6 -10
View File
@@ -6,10 +6,7 @@ import sqlite3
import threading
from pathlib import Path
try:
from . import holographic as hrr
except ImportError:
import holographic as hrr # type: ignore[no-redef]
from . import holographic as hrr
_SCHEMA = """
CREATE TABLE IF NOT EXISTS facts (
@@ -121,8 +118,8 @@ class MemoryStore:
with MemoryStore._shared_guard:
entry = MemoryStore._shared.get(self._key)
if entry is None:
# Autocommit: a write that raises mid-method can never leave a dangling
# transaction (and its write lock) open; explicit commit() calls are no-ops.
# Autocommit: a write that raises mid-method can't leave a dangling transaction (and its
# write lock) open; the explicit commit() calls in _write are then harmless no-ops.
conn = sqlite3.connect(self._key, check_same_thread=False, timeout=10.0, isolation_level=None)
conn.row_factory = sqlite3.Row
entry = MemoryStore._shared[self._key] = {"conn": conn, "lock": threading.RLock(), "refs": 0, "ready": False}
@@ -134,8 +131,7 @@ class MemoryStore:
self._entry["ready"] = True
def _init_db(self) -> None:
"""Create schema, enable WAL via the shared fallback helper (NFS/SMB/FUSE HERMES_HOME
degrades gracefully), and add hrr_vector to pre-HRR databases."""
"""Create schema, enable WAL via the shared fallback helper (NFS/SMB/FUSE degrade gracefully), add hrr_vector to pre-HRR DBs."""
from hermes_state import apply_wal_with_fallback
apply_wal_with_fallback(self._conn, db_label="memory_store.db (holographic)")
self._conn.executescript(_SCHEMA)
@@ -152,8 +148,8 @@ class MemoryStore:
return cur
def add_fact(self, content: str, category: str = "general", tags: str = "") -> int:
"""Insert a fact and return its fact_id; on duplicate content (UNIQUE) return the
existing fact_id untouched. Links extracted entities and rebuilds the category bank."""
"""Insert a fact and return its fact_id; on duplicate content (UNIQUE) return the existing fact_id untouched.
Links extracted entities and rebuilds the category bank."""
with self._lock:
content = content.strip()
if not content: