166 lines
6.5 KiB
Python
166 lines
6.5 KiB
Python
"""Holographic Reduced Representations (HRR) with phase encoding.
|
|
|
|
Each concept is a vector of angles in [0, 2π). Operations:
|
|
bind — circular convolution (phase addition) — associates two concepts
|
|
unbind — circular correlation (phase subtraction) — retrieves a bound value
|
|
bundle — superposition (circular mean) — merges multiple concepts
|
|
|
|
Phase encoding avoids the magnitude collapse of complex-number HRRs and maps
|
|
cleanly to cosine similarity. Atoms derive deterministically from SHA-256 so
|
|
representations are identical across processes, machines, and Python versions.
|
|
|
|
References: Plate (1995) HRRs; Gayler (2004) Vector Symbolic Architectures.
|
|
"""
|
|
|
|
import hashlib
|
|
import logging
|
|
import struct
|
|
import math
|
|
|
|
try:
|
|
import numpy as np
|
|
_HAS_NUMPY = True
|
|
except ImportError:
|
|
np = None # type: ignore[assignment]
|
|
_HAS_NUMPY = False
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_TWO_PI = 2.0 * math.pi
|
|
_FLOAT32_BLOB_PREFIX = b"HRR1"
|
|
|
|
|
|
def _require_numpy() -> None:
|
|
if not _HAS_NUMPY:
|
|
raise RuntimeError("numpy is required for holographic operations")
|
|
|
|
|
|
def encode_atom(word: str, dim: int = 1024) -> "np.ndarray":
|
|
"""Deterministic phase vector: SHA-256 counter blocks of f"{word}:{i}" -> uint16 -> [0, 2π).
|
|
|
|
hashlib rather than numpy RNG so atoms are reproducible across platforms.
|
|
"""
|
|
_require_numpy()
|
|
uint16_values: list[int] = []
|
|
for i in range(math.ceil(dim / 16)): # 32-byte digest = 16 uint16 values
|
|
digest = hashlib.sha256(f"{word}:{i}".encode()).digest()
|
|
uint16_values.extend(struct.unpack("<16H", digest))
|
|
return np.array(uint16_values[:dim], dtype=np.float64) * (_TWO_PI / 65536.0)
|
|
|
|
|
|
def bind(a: "np.ndarray", b: "np.ndarray") -> "np.ndarray":
|
|
"""Circular convolution = phase addition; result is quasi-orthogonal to both inputs."""
|
|
_require_numpy()
|
|
return (a + b) % _TWO_PI
|
|
|
|
|
|
def unbind(memory: "np.ndarray", key: "np.ndarray") -> "np.ndarray":
|
|
"""Circular correlation = phase subtraction; unbind(bind(a, b), a) ≈ b."""
|
|
_require_numpy()
|
|
return (memory - key) % _TWO_PI
|
|
|
|
|
|
def bundle(*vectors: "np.ndarray") -> "np.ndarray":
|
|
"""Superposition via circular mean; holds O(sqrt(dim)) items before similarity degrades."""
|
|
_require_numpy()
|
|
complex_sum = np.sum([np.exp(1j * v) for v in vectors], axis=0)
|
|
return np.angle(complex_sum) % _TWO_PI
|
|
|
|
|
|
def similarity(a: "np.ndarray", b: "np.ndarray") -> float:
|
|
"""Phase cosine similarity in [-1, 1]; ~0 for unrelated vectors."""
|
|
_require_numpy()
|
|
return float(np.mean(np.cos(a - b)))
|
|
|
|
|
|
def encode_text(text: str, dim: int = 1024) -> "np.ndarray":
|
|
"""Bag-of-words bundle of token atoms; empty text -> encode_atom("__hrr_empty__")."""
|
|
_require_numpy()
|
|
tokens = [t for t in (tok.strip(".,!?;:\"'()[]{}") for tok in text.lower().split()) if t]
|
|
if not tokens:
|
|
return encode_atom("__hrr_empty__", dim)
|
|
return bundle(*[encode_atom(token, dim) for token in tokens])
|
|
|
|
|
|
def encode_fact(content: str, entities: list[str], dim: int = 1024) -> "np.ndarray":
|
|
"""bundle(bind(text, ROLE_CONTENT), bind(entity_i, ROLE_ENTITY)...), so
|
|
unbind(fact, bind(entity, ROLE_ENTITY)) ≈ content_vector."""
|
|
_require_numpy()
|
|
role_content = encode_atom("__hrr_role_content__", dim)
|
|
role_entity = encode_atom("__hrr_role_entity__", dim)
|
|
components = [bind(encode_text(content, dim), role_content)]
|
|
for entity in entities:
|
|
components.append(bind(encode_atom(entity.lower(), dim), role_entity))
|
|
return bundle(*components)
|
|
|
|
|
|
def phases_to_bytes(phases: "np.ndarray", dim: int | None = None) -> bytes:
|
|
"""Serialize as prefixed float32 (half the size of legacy float64 blobs).
|
|
|
|
At dim=1 the prefixed float32 blob and a raw float64 blob are both 8 bytes,
|
|
so we write legacy float64 there to keep ``bytes_to_phases`` unambiguous.
|
|
"""
|
|
_require_numpy()
|
|
if dim is None:
|
|
dim = int(phases.shape[0])
|
|
if len(_FLOAT32_BLOB_PREFIX) + dim * np.dtype(np.float32).itemsize == dim * np.dtype(np.float64).itemsize:
|
|
return np.asarray(phases, dtype=np.float64).tobytes()
|
|
return _FLOAT32_BLOB_PREFIX + np.asarray(phases, dtype=np.float32).tobytes()
|
|
|
|
|
|
def bytes_to_phases(data: bytes, dim: int | None = None) -> "np.ndarray":
|
|
"""Deserialize prefixed float32 or legacy raw float64 blobs (always returns float64).
|
|
|
|
With ``dim`` given, a prefixed blob whose size equals the float64 size
|
|
(dim=1) is read as legacy float64: ``phases_to_bytes`` never writes a
|
|
prefixed blob at that size, so such a blob must be legacy data.
|
|
"""
|
|
_require_numpy()
|
|
f32, f64 = np.dtype(np.float32).itemsize, np.dtype(np.float64).itemsize
|
|
plen = len(_FLOAT32_BLOB_PREFIX)
|
|
prefixed = data.startswith(_FLOAT32_BLOB_PREFIX)
|
|
|
|
if dim is None:
|
|
if prefixed:
|
|
payload = data[plen:]
|
|
if len(payload) % f32 != 0:
|
|
raise ValueError(f"HRR float32 vector blob has invalid payload byte length: {len(payload)}")
|
|
return np.frombuffer(payload, dtype=np.float32).astype(np.float64)
|
|
if len(data) % f64 != 0:
|
|
raise ValueError(f"HRR legacy vector blob has invalid byte length: {len(data)}")
|
|
return np.frombuffer(data, dtype=np.float64).copy()
|
|
|
|
float32_blob_bytes = plen + dim * f32
|
|
float64_bytes = dim * f64
|
|
collides = float32_blob_bytes == float64_bytes
|
|
if not collides and prefixed and len(data) == float32_blob_bytes:
|
|
return np.frombuffer(data[plen:], dtype=np.float32).astype(np.float64)
|
|
if len(data) == float64_bytes:
|
|
return np.frombuffer(data, dtype=np.float64).copy()
|
|
if prefixed:
|
|
expected = (f"{float64_bytes} (legacy float64)" if collides
|
|
else f"{float32_blob_bytes} (prefixed float32) or {float64_bytes} (legacy float64)")
|
|
raise ValueError(
|
|
f"HRR vector blob has {len(data)} bytes ({len(data) - plen} payload bytes after "
|
|
f"the float32 prefix); expected {expected} for dim={dim}"
|
|
)
|
|
raise ValueError(
|
|
f"HRR legacy vector blob has {len(data)} bytes; expected "
|
|
f"{float64_bytes} (float64) for dim={dim}"
|
|
)
|
|
|
|
|
|
def snr_estimate(dim: int, n_items: int) -> float:
|
|
"""SNR = sqrt(dim / n_items) (inf when empty); warns below 2.0 (n_items > dim/4)."""
|
|
_require_numpy()
|
|
if n_items <= 0:
|
|
return float("inf")
|
|
snr = math.sqrt(dim / n_items)
|
|
if snr < 2.0:
|
|
logger.warning(
|
|
"HRR storage near capacity: SNR=%.2f (dim=%d, n_items=%d). "
|
|
"Retrieval accuracy may degrade. Consider increasing dim or reducing stored items.",
|
|
snr, dim, n_items,
|
|
)
|
|
return snr
|