Files
hermes-agent/plugins/memory/holographic/holographic.py
T

166 lines
6.5 KiB
Python

"""Holographic Reduced Representations (HRR) with phase encoding.
Each concept is a vector of angles in [0, 2π). Operations:
bind — circular convolution (phase addition) — associates two concepts
unbind — circular correlation (phase subtraction) — retrieves a bound value
bundle — superposition (circular mean) — merges multiple concepts
Phase encoding avoids the magnitude collapse of complex-number HRRs and maps
cleanly to cosine similarity. Atoms derive deterministically from SHA-256 so
representations are identical across processes, machines, and Python versions.
References: Plate (1995) HRRs; Gayler (2004) Vector Symbolic Architectures.
"""
import hashlib
import logging
import struct
import math
try:
import numpy as np
_HAS_NUMPY = True
except ImportError:
np = None # type: ignore[assignment]
_HAS_NUMPY = False
logger = logging.getLogger(__name__)
_TWO_PI = 2.0 * math.pi
_FLOAT32_BLOB_PREFIX = b"HRR1"
def _require_numpy() -> None:
if not _HAS_NUMPY:
raise RuntimeError("numpy is required for holographic operations")
def encode_atom(word: str, dim: int = 1024) -> "np.ndarray":
"""Deterministic phase vector: SHA-256 counter blocks of f"{word}:{i}" -> uint16 -> [0, 2π).
hashlib rather than numpy RNG so atoms are reproducible across platforms.
"""
_require_numpy()
uint16_values: list[int] = []
for i in range(math.ceil(dim / 16)): # 32-byte digest = 16 uint16 values
digest = hashlib.sha256(f"{word}:{i}".encode()).digest()
uint16_values.extend(struct.unpack("<16H", digest))
return np.array(uint16_values[:dim], dtype=np.float64) * (_TWO_PI / 65536.0)
def bind(a: "np.ndarray", b: "np.ndarray") -> "np.ndarray":
"""Circular convolution = phase addition; result is quasi-orthogonal to both inputs."""
_require_numpy()
return (a + b) % _TWO_PI
def unbind(memory: "np.ndarray", key: "np.ndarray") -> "np.ndarray":
"""Circular correlation = phase subtraction; unbind(bind(a, b), a) ≈ b."""
_require_numpy()
return (memory - key) % _TWO_PI
def bundle(*vectors: "np.ndarray") -> "np.ndarray":
"""Superposition via circular mean; holds O(sqrt(dim)) items before similarity degrades."""
_require_numpy()
complex_sum = np.sum([np.exp(1j * v) for v in vectors], axis=0)
return np.angle(complex_sum) % _TWO_PI
def similarity(a: "np.ndarray", b: "np.ndarray") -> float:
"""Phase cosine similarity in [-1, 1]; ~0 for unrelated vectors."""
_require_numpy()
return float(np.mean(np.cos(a - b)))
def encode_text(text: str, dim: int = 1024) -> "np.ndarray":
"""Bag-of-words bundle of token atoms; empty text -> encode_atom("__hrr_empty__")."""
_require_numpy()
tokens = [t for t in (tok.strip(".,!?;:\"'()[]{}") for tok in text.lower().split()) if t]
if not tokens:
return encode_atom("__hrr_empty__", dim)
return bundle(*[encode_atom(token, dim) for token in tokens])
def encode_fact(content: str, entities: list[str], dim: int = 1024) -> "np.ndarray":
"""bundle(bind(text, ROLE_CONTENT), bind(entity_i, ROLE_ENTITY)...), so
unbind(fact, bind(entity, ROLE_ENTITY)) ≈ content_vector."""
_require_numpy()
role_content = encode_atom("__hrr_role_content__", dim)
role_entity = encode_atom("__hrr_role_entity__", dim)
components = [bind(encode_text(content, dim), role_content)]
for entity in entities:
components.append(bind(encode_atom(entity.lower(), dim), role_entity))
return bundle(*components)
def phases_to_bytes(phases: "np.ndarray", dim: int | None = None) -> bytes:
"""Serialize as prefixed float32 (half the size of legacy float64 blobs).
At dim=1 the prefixed float32 blob and a raw float64 blob are both 8 bytes,
so we write legacy float64 there to keep ``bytes_to_phases`` unambiguous.
"""
_require_numpy()
if dim is None:
dim = int(phases.shape[0])
if len(_FLOAT32_BLOB_PREFIX) + dim * np.dtype(np.float32).itemsize == dim * np.dtype(np.float64).itemsize:
return np.asarray(phases, dtype=np.float64).tobytes()
return _FLOAT32_BLOB_PREFIX + np.asarray(phases, dtype=np.float32).tobytes()
def bytes_to_phases(data: bytes, dim: int | None = None) -> "np.ndarray":
"""Deserialize prefixed float32 or legacy raw float64 blobs (always returns float64).
With ``dim`` given, a prefixed blob whose size equals the float64 size
(dim=1) is read as legacy float64: ``phases_to_bytes`` never writes a
prefixed blob at that size, so such a blob must be legacy data.
"""
_require_numpy()
f32, f64 = np.dtype(np.float32).itemsize, np.dtype(np.float64).itemsize
plen = len(_FLOAT32_BLOB_PREFIX)
prefixed = data.startswith(_FLOAT32_BLOB_PREFIX)
if dim is None:
if prefixed:
payload = data[plen:]
if len(payload) % f32 != 0:
raise ValueError(f"HRR float32 vector blob has invalid payload byte length: {len(payload)}")
return np.frombuffer(payload, dtype=np.float32).astype(np.float64)
if len(data) % f64 != 0:
raise ValueError(f"HRR legacy vector blob has invalid byte length: {len(data)}")
return np.frombuffer(data, dtype=np.float64).copy()
float32_blob_bytes = plen + dim * f32
float64_bytes = dim * f64
collides = float32_blob_bytes == float64_bytes
if not collides and prefixed and len(data) == float32_blob_bytes:
return np.frombuffer(data[plen:], dtype=np.float32).astype(np.float64)
if len(data) == float64_bytes:
return np.frombuffer(data, dtype=np.float64).copy()
if prefixed:
expected = (f"{float64_bytes} (legacy float64)" if collides
else f"{float32_blob_bytes} (prefixed float32) or {float64_bytes} (legacy float64)")
raise ValueError(
f"HRR vector blob has {len(data)} bytes ({len(data) - plen} payload bytes after "
f"the float32 prefix); expected {expected} for dim={dim}"
)
raise ValueError(
f"HRR legacy vector blob has {len(data)} bytes; expected "
f"{float64_bytes} (float64) for dim={dim}"
)
def snr_estimate(dim: int, n_items: int) -> float:
"""SNR = sqrt(dim / n_items) (inf when empty); warns below 2.0 (n_items > dim/4)."""
_require_numpy()
if n_items <= 0:
return float("inf")
snr = math.sqrt(dim / n_items)
if snr < 2.0:
logger.warning(
"HRR storage near capacity: SNR=%.2f (dim=%d, n_items=%d). "
"Retrieval accuracy may degrade. Consider increasing dim or reducing stored items.",
snr, dim, n_items,
)
return snr