feat(memory): make the mem0 sync char cap configurable via mem0.json
A flat 450-char cap fits 512-token embedders (bge-small-zh-v1.5, all-minilm) but stores only ~5% of the window on 8192-token models (text-embedding-3-small, jina-embeddings-v3, bge-m3), degrading memory quality for users those models served fine before truncation existed. Read `sync_max_chars` from mem0.json once in initialize() (450 default) and pass it to _truncate_for_sync(). Config over auto-detection: the Ollama /api/show probe + known-model table proposed in #37427 adds a network call and a curated list for a number the operator already knows from their embedder choice; the setup wizard's mem0.json is the plugin's behavioral-settings surface (no new HERMES_* env var). Documented in the plugin README and the memory-providers docs page. Dynamic-cap requirement and measurements (450 OK / 600 -> HTTP 500 on bge-small-zh-v1.5:f16) by @szicely in #106235. Refs #37421 #106235 Co-authored-by: szicely <140148567+szicely@users.noreply.github.com> Co-authored-by: liuhao1024 <sunsky.lau@gmail.com>
This commit is contained in:
@@ -30,6 +30,7 @@ Behavioral settings live in `$HERMES_HOME/mem0.json` (set them via `hermes memor
|
||||
| `user_id` | `hermes-user` | User identifier on Mem0 |
|
||||
| `agent_id` | `hermes` | Agent identifier |
|
||||
| `rerank` | `false` | Rerank search results for relevance (platform mode only) |
|
||||
| `sync_max_chars` | `450` | Per-message character cap applied before each turn is sent for fact extraction (cut at the last sentence boundary). Default fits 512-token embedders; raise it (e.g. `6000`) for 8k-token embedders such as `text-embedding-3-small`, `jina-embeddings-v3`, `bge-m3` |
|
||||
|
||||
The plugin has three connection modes:
|
||||
|
||||
|
||||
@@ -38,6 +38,8 @@ _DEFAULT_USER_ID = "hermes-user"
|
||||
# jina-embeddings-v3: 8192), and oversized turns make backend.add() raise — Ollama
|
||||
# answers HTTP 500, hosted APIs return INPUT_TOKEN_LIMIT_EXCEEDED — which _try only
|
||||
# logs, silently dropping the turn's memory extraction. Cap each message up front.
|
||||
# The default fits a 512-token embedder (measured: 450 OK, 600 -> HTTP 500 on
|
||||
# bge-small-zh-v1.5:f16); ``sync_max_chars`` in mem0.json raises it for larger windows.
|
||||
_SYNC_MSG_MAX_CHARS = 450
|
||||
|
||||
|
||||
@@ -115,6 +117,7 @@ class Mem0MemoryProvider(MemoryProvider):
|
||||
self._config = self._backend = self._sync_thread = self._prefetch_thread = None
|
||||
self._mode, self._api_key, self._host, self._user_id, self._agent_id = "platform", "", "", _DEFAULT_USER_ID, "hermes"
|
||||
self._rerank_default, self._channel = False, "cli" # channel = gateway name (cli/telegram/discord/...)
|
||||
self._sync_max_chars = _SYNC_MSG_MAX_CHARS
|
||||
self._prefetch_query = self._prefetch_result = ""
|
||||
self._prefetch_done = self._atexit_registered = False
|
||||
self._consecutive_failures, self._breaker_open_until = 0, 0.0 # circuit breaker state
|
||||
@@ -220,6 +223,7 @@ class Mem0MemoryProvider(MemoryProvider):
|
||||
_rr = cfg.get("rerank", False)
|
||||
self._rerank_default = _rr.lower() in ("true", "1", "yes") if isinstance(_rr, str) else bool(_rr)
|
||||
self._channel = kwargs.get("platform") or "cli"
|
||||
self._sync_max_chars = int(cfg.get("sync_max_chars") or _SYNC_MSG_MAX_CHARS)
|
||||
self._backend = self._create_backend()
|
||||
if self._backend and not self._atexit_registered:
|
||||
atexit.register(self._shutdown_backend)
|
||||
@@ -291,8 +295,8 @@ class Mem0MemoryProvider(MemoryProvider):
|
||||
def _sync():
|
||||
if self._backend is not None:
|
||||
messages = [
|
||||
{"role": "user", "content": _truncate_for_sync(user_content)},
|
||||
{"role": "assistant", "content": _truncate_for_sync(assistant_content)},
|
||||
{"role": "user", "content": _truncate_for_sync(user_content, self._sync_max_chars)},
|
||||
{"role": "assistant", "content": _truncate_for_sync(assistant_content, self._sync_max_chars)},
|
||||
]
|
||||
self._try(lambda: self._add(messages, infer=True), logger.warning, "Mem0 sync failed: %s")
|
||||
|
||||
|
||||
@@ -180,6 +180,17 @@ class TestSyncTurnTruncation:
|
||||
assert len(sent[1]["content"]) <= mem0_plugin._SYNC_MSG_MAX_CHARS and sent[1]["content"].endswith(".")
|
||||
assert provider._consecutive_failures == 0
|
||||
|
||||
def test_sync_max_chars_config_raises_cap(self, monkeypatch, tmp_path):
|
||||
"""8k-token embedders should not be stuck at the 512-token default (#106235)."""
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
monkeypatch.setenv("MEM0_API_KEY", "test-key")
|
||||
(tmp_path / "mem0.json").write_text('{"sync_max_chars": 3000}')
|
||||
backend = FakeBackend()
|
||||
provider = self._make_provider(monkeypatch, backend)
|
||||
provider.sync_turn("hi", "Long answer. " * 200, session_id="s1") # 2600 chars
|
||||
provider._sync_thread.join(timeout=2)
|
||||
assert backend.captured[0][1][1]["content"] == "Long answer. " * 200
|
||||
|
||||
|
||||
class TestMem0Prefetch:
|
||||
"""prefetch() must recall on the CURRENT question, synchronously.
|
||||
|
||||
@@ -414,6 +414,7 @@ The plugin authenticates with `X-API-Key` and uses the server's `/search` / `/me
|
||||
| `user_id` | `hermes-user` | User identifier |
|
||||
| `agent_id` | `hermes` | Agent identifier |
|
||||
| `rerank` | `false` | Rerank search results for relevance (platform mode only) |
|
||||
| `sync_max_chars` | `450` | Per-message character cap applied before each turn is sent for fact extraction, cut at the last sentence boundary. The default fits 512-token embedders (Ollama `bge-small-zh-v1.5`, `all-minilm`); raise it (e.g. `6000`) for 8k-token embedders such as `text-embedding-3-small`, `jina-embeddings-v3` or `bge-m3` |
|
||||
|
||||
**OSS supported providers:**
|
||||
|
||||
|
||||
Reference in New Issue
Block a user