feat(mem0-plugin): add /mem0:dream consolidation + retention policies

Tier 4 implementation:
- #13+15: /mem0:dream skill — fetches all memories, finds near-dupes,
  proposes merges, flags contradictions, prunes stale entries per
  retention policy, outputs terminal diff for user approval
- #14: --schedule weekly registers cron via Claude Code schedule skill
  (Claude Code only, noted in docs)
- #16: parse_mem0_config.py parses ## Retention section from mem0.md
  (e.g. session_state: 90d, architecture_decisions: forever)
This commit is contained in:
kartik-mem0
2026-05-21 17:43:14 +05:30
parent 6ed5fcf76e
commit b79be63e30
3 changed files with 637 additions and 0 deletions
+137
View File
@@ -0,0 +1,137 @@
#!/usr/bin/env python3
"""Parse mem0.md project configuration file.
Reads the optional ``mem0.md`` file in a project directory and extracts
retention policies from a ``## Retention`` section.
Retention format (inside the section):
<category>: <N>d — keep for N days
<category>: forever — never prune (returned as None)
Usage (CLI):
python3 parse_mem0_config.py [<cwd>]
Prints a JSON object mapping category names to day counts (int) or null
(forever) on stdout. Prints ``{}`` when no mem0.md or no ## Retention
section is found.
"""
from __future__ import annotations
import json
import os
import re
import sys
def find_mem0_config(cwd: str) -> str | None:
"""Look for ``mem0.md`` in *cwd*.
Returns the absolute path to ``mem0.md`` if found, else ``None``.
"""
candidate = os.path.join(cwd, "mem0.md")
return candidate if os.path.isfile(candidate) else None
def parse_retention(content: str) -> dict[str, int | None]:
"""Parse the ``## Retention`` section of *content*.
Scans for a heading that matches ``## Retention`` (case-insensitive),
then reads lines until the next ``##``-level heading or end of string.
Each non-blank, non-comment line inside the section is expected to be::
<category>: <N>d → days=N (int)
<category>: forever → days=None
Malformed lines are silently skipped.
Args:
content: Full text of a mem0.md file.
Returns:
Dict mapping category name (str) to day count (int) or ``None``
(forever). Empty dict when no ``## Retention`` section is found.
"""
# Find the ## Retention section (allow any amount of trailing whitespace /
# extra words, but the heading must start with "## Retention").
section_match = re.search(
r"^##\s+Retention[^\n]*\n(.*?)(?=^##\s|\Z)",
content,
flags=re.MULTILINE | re.DOTALL | re.IGNORECASE,
)
if not section_match:
return {}
section_text = section_match.group(1)
policies: dict[str, int | None] = {}
for line in section_text.splitlines():
# Strip comments and whitespace
line = re.sub(r"#.*$", "", line).strip()
if not line:
continue
# Match "<category>: <value>"
line_match = re.match(r"^([^:]+):\s*(.+)$", line)
if not line_match:
continue
category = line_match.group(1).strip()
value = line_match.group(2).strip().lower()
if value == "forever":
policies[category] = None
else:
days_match = re.match(r"^(\d+)d$", value)
if days_match:
policies[category] = int(days_match.group(1))
# else: malformed value — skip silently
return policies
def load_retention_policies(cwd: str | None = None) -> dict[str, int | None]:
"""Load retention policies from the mem0.md in *cwd*.
Combines :func:`find_mem0_config` and :func:`parse_retention` into a
single convenience function.
Args:
cwd: Directory to search. Defaults to ``os.getcwd()``.
Returns:
Retention dict (category → days or ``None``). Empty dict if no
``mem0.md`` exists in *cwd* or it contains no ``## Retention``
section.
"""
if cwd is None:
cwd = os.getcwd()
config_path = find_mem0_config(cwd)
if config_path is None:
return {}
try:
with open(config_path, encoding="utf-8") as fh:
content = fh.read()
except OSError:
return {}
return parse_retention(content)
def main() -> int:
"""CLI entry point.
Reads cwd from ``sys.argv[1]`` (or ``os.getcwd()``), prints JSON to
stdout.
"""
cwd = sys.argv[1] if len(sys.argv) > 1 else os.getcwd()
policies = load_retention_policies(cwd)
print(json.dumps(policies))
return 0
if __name__ == "__main__":
sys.exit(main())
+250
View File
@@ -0,0 +1,250 @@
---
name: mem0-dream
description: >
Memory consolidation pass. Fetches all project memories, finds near-duplicates,
merges them, flags contradictions, prunes stale entries per retention policy.
Outputs a diff for user approval before applying changes.
TRIGGER: user runs /mem0:dream, or asks "consolidate memories", "clean up memories",
"merge duplicate memories", "run dream".
---
# Mem0 Dream — Memory Consolidation
This skill performs a memory consolidation pass: it fetches all project memories,
identifies near-duplicates, flags contradictions, and prunes stale entries based on
configured retention policies. All proposed changes are shown as a diff for user
approval before anything is modified.
---
## Step 1: Load Retention Policies
Determine the active retention policy by running the parser script. Use the
appropriate `PLUGIN_ROOT` variable for the current platform (`${CLAUDE_PLUGIN_ROOT}`,
`${CODEX_PLUGIN_ROOT}`, or `${CURSOR_PLUGIN_ROOT}`):
```bash
python3 "<PLUGIN_ROOT>/scripts/parse_mem0_config.py" "<cwd>"
```
Parse the JSON output (a dict of `category → days | null`). If the script fails
or returns `{}`, fall back to these built-in defaults:
| `metadata.type` | Default retention |
|---|---|
| `session_state` | 90 days |
| `compact_summary` | 90 days |
| all others | no pruning |
Store the resolved policies for use in Step 3.
---
## Step 2: Fetch ALL Project Memories
Call `get_memories` to retrieve every memory for the active project:
```python
get_memories(
user_id="<active_user_id>",
app_id="<active_project_id>",
page_size=200,
)
```
If the response indicates more pages exist, paginate until all memories are fetched.
Collect the full list before proceeding. If zero memories are found, print:
```
No memories found for project <project_id>. Nothing to consolidate.
```
…and stop.
---
## Step 3: Analyze — Find Issues
Work entirely in-memory; do not modify anything yet.
Group memories by `metadata.type` (use `"unknown"` when the field is absent).
For each group, identify the following:
### 3a. Near-duplicate pairs (merge candidates)
Two memories are near-duplicates when they express the same fact or decision but
phrased differently (e.g., "Use PostgreSQL for auth" and "Auth DB is PostgreSQL").
Heuristics:
- Significant noun/keyword overlap in the memory text.
- Same `metadata.type`.
- Neither memory is pinned (`metadata.pinned != true`).
For each qualifying pair, draft a merged version that is more complete and specific
than either original.
### 3b. Contradictions
Two memories contradict when they assert opposing facts about the same topic
(e.g., "Deploy to ECS" vs. "Deploy to Vercel").
Identify the likely winner: the more recent memory with higher confidence wins.
Store both IDs and their content for user review.
### 3c. Prune candidates
A memory is a prune candidate when **any** of the following is true:
1. Its `metadata.type` has a retention policy and the memory is older than the
configured number of days (compare `created_at` to today).
2. Its confidence score is below 0.3 AND it contains no information unique to
this project (no file paths, identifiers, or domain-specific nouns).
**Always skip memories where `metadata.pinned == true`**, regardless of age or
confidence.
---
## Step 4: Print Diff Report (item 15)
Print a structured diff to the terminal before making any changes. Use exactly
this format:
```
## Dream — Memory Consolidation Report
### Merge proposals (<N> pairs)
MERGE [mem0:<id1>] + [mem0:<id2>] → NEW
- Original 1: "<content of memory 1, truncated to 120 chars>"
- Original 2: "<content of memory 2, truncated to 120 chars>"
- Merged: "<drafted merged content>"
### Contradictions (<N> pairs)
CONFLICT [mem0:<idA>] vs [mem0:<idB>]
- A: "<content>" (<created_at date>, confidence: <score>)
- B: "<content>" (<created_at date>, confidence: <score>)
Which is current? [A/B/skip]
### Prune candidates (<N> memories)
PRUNE [mem0:<id>] — <metadata.type>, <age>d old (policy: <policy_days>d)
---
Proposed: <N> merges, <N> prunes, <N> conflicts
Apply? [Y/n]
```
If there are zero items in any category, omit that section entirely.
If there are zero total proposals (no merges, no prunes, no conflicts), print:
```
Dream complete. No duplicate, contradictory, or stale memories found.
```
…and stop.
---
## Step 5: Wait for User Input and Apply
### 5a. Contradictions
For each `CONFLICT` pair in the report, wait for the user to type `A`, `B`, or
`skip` (case-insensitive). If they enter nothing (empty), treat as `skip`.
Record the winner for each pair before proceeding to the final apply confirmation.
### 5b. Final confirmation
After all conflict resolutions are collected, prompt:
```
Apply? [Y/n]
```
If the user types `n` or `no` (case-insensitive), print `Cancelled. No changes made.`
and stop.
If the user confirms (`Y`, `yes`, or empty / Enter), apply all changes in this order:
#### Merges
For each approved merge pair:
1. `delete_memory(<id1>)`
2. `delete_memory(<id2>)`
3. `add_memory` with:
- `messages=[{"role": "user", "content": "<merged content>"}]`
- `user_id=<active_user_id>`
- `app_id=<active_project_id>` (top-level, not in metadata)
- `metadata={"type": "<original type>", "branch": "<active_branch>", "confidence": <higher of the two original scores>, "source": "mem0-dream"}`
- `infer=False`
#### Contradictions (resolved)
For each resolved conflict where the user chose A or B:
- Identify the loser (the non-chosen memory).
- `update_memory(<loser_id>, data={"metadata": {"confidence": 0.1, "superseded_by": "<winner_id>", "source": "mem0-dream"}})`
Contradictions where the user chose `skip` are left untouched.
#### Prunes
For each prune candidate:
- `delete_memory(<memory_id>)`
---
## Step 6: Print Summary
After all changes are applied, print:
```
Dream complete.
Merged: <N> pairs → <N> new memories
Pruned: <N> memories deleted
Flagged: <N> contradictions resolved, <N> skipped
```
---
## Scheduling (Claude Code only)
### Running dream on a schedule
**Note**: Scheduled dream is only available in Claude Code, which natively supports
the `schedule` skill. Codex and Cursor do not have a scheduling primitive — users
on those platforms should run `/mem0:dream` manually.
When the user runs `/mem0:dream --schedule weekly` (or similar), register a
recurring scheduled task using the Claude Code `schedule` skill.
Use `--auto` flag to enable non-interactive mode, which skips the interactive
contradiction resolution UI and applies only safe operations automatically:
- **Merges**: applied automatically (no contradiction, both are compatible).
- **Prunes**: applied automatically (age/confidence-based, no ambiguity).
- **Contradictions**: skipped in `--auto` mode; they require human judgment.
Example scheduling invocation:
```
/schedule weekly mem0:dream --auto
```
When running with `--auto`:
1. Load policies and fetch memories (Steps 1–3) as normal.
2. Apply merges and prunes silently without printing the diff or prompting.
3. Log a compact summary to stdout (suitable for a cron log):
```
[mem0-dream --auto] project=<id> merged=<N> pruned=<N> conflicts_skipped=<N>
```
4. If contradictions were detected but skipped, store a reminder memory:
```python
add_memory(
messages=[{"role": "user", "content": "mem0-dream detected <N> contradiction(s) requiring manual review. Run /mem0:dream to resolve them interactively."}],
user_id="<active_user_id>",
app_id="<active_project_id>",
metadata={"type": "task_learning", "source": "mem0-dream-auto", "branch": "<active_branch>"},
infer=False,
)
```
+250
View File
@@ -0,0 +1,250 @@
"""Tests for parse_mem0_config.py — mem0.md retention policy parser."""
from __future__ import annotations
import json
import os
import subprocess
import sys
SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts")
# ---------------------------------------------------------------------------
# parse_retention — unit tests
# ---------------------------------------------------------------------------
def test_parse_retention_valid_section():
"""parse_retention extracts day-count policies correctly."""
from parse_mem0_config import parse_retention
content = """\
# Project Config
## Retention
session_state: 90d
compact_summary: 60d
decision: 180d
"""
result = parse_retention(content)
assert result == {
"session_state": 90,
"compact_summary": 60,
"decision": 180,
}
def test_parse_retention_forever_returns_none():
"""parse_retention maps 'forever' to None."""
from parse_mem0_config import parse_retention
content = """\
## Retention
user_preference: forever
anti_pattern: forever
session_state: 30d
"""
result = parse_retention(content)
assert result["user_preference"] is None
assert result["anti_pattern"] is None
assert result["session_state"] == 30
def test_parse_retention_no_section_returns_empty():
"""parse_retention returns {} when there is no ## Retention heading."""
from parse_mem0_config import parse_retention
content = """\
# Project Config
## Some Other Section
key: value
"""
result = parse_retention(content)
assert result == {}
def test_parse_retention_stops_at_next_heading():
"""parse_retention stops reading at the next ## heading."""
from parse_mem0_config import parse_retention
content = """\
## Retention
session_state: 7d
## Other Section
other_key: 999d
"""
result = parse_retention(content)
assert "session_state" in result
assert "other_key" not in result
def test_parse_retention_malformed_lines_skipped():
"""Malformed lines (no colon, bad day format) are silently ignored."""
from parse_mem0_config import parse_retention
content = """\
## Retention
session_state: 90d
bad_line_no_colon
another: badvalue
decision: 30d
"""
result = parse_retention(content)
assert result == {"session_state": 90, "decision": 30}
def test_parse_retention_comments_ignored():
"""Inline # comments are stripped before parsing."""
from parse_mem0_config import parse_retention
content = """\
## Retention
session_state: 90d # rolling 90-day window
user_preference: forever # never prune preferences
"""
result = parse_retention(content)
assert result["session_state"] == 90
assert result["user_preference"] is None
def test_parse_retention_case_insensitive_heading():
"""## retention (lowercase) is matched the same as ## Retention."""
from parse_mem0_config import parse_retention
content = """\
## retention
session_state: 14d
"""
result = parse_retention(content)
assert result == {"session_state": 14}
def test_parse_retention_empty_section_returns_empty():
"""A ## Retention section with no valid lines returns {}."""
from parse_mem0_config import parse_retention
content = """\
## Retention
# only comments here
## Next Section
"""
result = parse_retention(content)
assert result == {}
# ---------------------------------------------------------------------------
# load_retention_policies — integration tests with tmp files
# ---------------------------------------------------------------------------
def test_load_retention_policies_with_tmp_file(tmp_path):
"""load_retention_policies reads a real mem0.md from disk."""
from parse_mem0_config import load_retention_policies
mem0_md = tmp_path / "mem0.md"
mem0_md.write_text(
"""\
# My Project
## Retention
session_state: 90d
compact_summary: 60d
decision: forever
""",
encoding="utf-8",
)
result = load_retention_policies(str(tmp_path))
assert result == {
"session_state": 90,
"compact_summary": 60,
"decision": None,
}
def test_load_retention_policies_no_mem0_md_returns_empty(tmp_path):
"""load_retention_policies returns {} when no mem0.md exists."""
from parse_mem0_config import load_retention_policies
result = load_retention_policies(str(tmp_path))
assert result == {}
def test_load_retention_policies_no_retention_section_returns_empty(tmp_path):
"""load_retention_policies returns {} when mem0.md has no ## Retention."""
from parse_mem0_config import load_retention_policies
mem0_md = tmp_path / "mem0.md"
mem0_md.write_text(
"""\
# My Project
Some general project notes here.
No retention section.
""",
encoding="utf-8",
)
result = load_retention_policies(str(tmp_path))
assert result == {}
def test_load_retention_policies_defaults_to_cwd(tmp_path, monkeypatch):
"""load_retention_policies uses os.getcwd() when cwd is None."""
from parse_mem0_config import load_retention_policies
monkeypatch.chdir(tmp_path)
mem0_md = tmp_path / "mem0.md"
mem0_md.write_text("## Retention\nsession_state: 45d\n", encoding="utf-8")
result = load_retention_policies() # no cwd arg
assert result == {"session_state": 45}
# ---------------------------------------------------------------------------
# CLI / main() — subprocess test
# ---------------------------------------------------------------------------
def test_cli_main_prints_json(tmp_path):
"""CLI: python parse_mem0_config.py <cwd> prints valid JSON."""
mem0_md = tmp_path / "mem0.md"
mem0_md.write_text(
"## Retention\nsession_state: 90d\nuser_preference: forever\n",
encoding="utf-8",
)
result = subprocess.run(
[sys.executable, os.path.join(SCRIPTS_DIR, "parse_mem0_config.py"), str(tmp_path)],
capture_output=True,
text=True,
)
assert result.returncode == 0
data = json.loads(result.stdout)
assert data["session_state"] == 90
assert data["user_preference"] is None
def test_cli_main_no_file_prints_empty_json(tmp_path):
"""CLI: prints '{}' when no mem0.md exists."""
result = subprocess.run(
[sys.executable, os.path.join(SCRIPTS_DIR, "parse_mem0_config.py"), str(tmp_path)],
capture_output=True,
text=True,
)
assert result.returncode == 0
assert json.loads(result.stdout) == {}