feat(mem0-plugin): add /mem0:dream consolidation + retention policies
Tier 4 implementation: - #13+15: /mem0:dream skill — fetches all memories, finds near-dupes, proposes merges, flags contradictions, prunes stale entries per retention policy, outputs terminal diff for user approval - #14: --schedule weekly registers cron via Claude Code schedule skill (Claude Code only, noted in docs) - #16: parse_mem0_config.py parses ## Retention section from mem0.md (e.g. session_state: 90d, architecture_decisions: forever)
This commit is contained in:
@@ -0,0 +1,137 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Parse mem0.md project configuration file.
|
||||
|
||||
Reads the optional ``mem0.md`` file in a project directory and extracts
|
||||
retention policies from a ``## Retention`` section.
|
||||
|
||||
Retention format (inside the section):
|
||||
<category>: <N>d — keep for N days
|
||||
<category>: forever — never prune (returned as None)
|
||||
|
||||
Usage (CLI):
|
||||
python3 parse_mem0_config.py [<cwd>]
|
||||
|
||||
Prints a JSON object mapping category names to day counts (int) or null
|
||||
(forever) on stdout. Prints ``{}`` when no mem0.md or no ## Retention
|
||||
section is found.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
|
||||
def find_mem0_config(cwd: str) -> str | None:
|
||||
"""Look for ``mem0.md`` in *cwd*.
|
||||
|
||||
Returns the absolute path to ``mem0.md`` if found, else ``None``.
|
||||
"""
|
||||
candidate = os.path.join(cwd, "mem0.md")
|
||||
return candidate if os.path.isfile(candidate) else None
|
||||
|
||||
|
||||
def parse_retention(content: str) -> dict[str, int | None]:
|
||||
"""Parse the ``## Retention`` section of *content*.
|
||||
|
||||
Scans for a heading that matches ``## Retention`` (case-insensitive),
|
||||
then reads lines until the next ``##``-level heading or end of string.
|
||||
|
||||
Each non-blank, non-comment line inside the section is expected to be::
|
||||
|
||||
<category>: <N>d → days=N (int)
|
||||
<category>: forever → days=None
|
||||
|
||||
Malformed lines are silently skipped.
|
||||
|
||||
Args:
|
||||
content: Full text of a mem0.md file.
|
||||
|
||||
Returns:
|
||||
Dict mapping category name (str) to day count (int) or ``None``
|
||||
(forever). Empty dict when no ``## Retention`` section is found.
|
||||
"""
|
||||
# Find the ## Retention section (allow any amount of trailing whitespace /
|
||||
# extra words, but the heading must start with "## Retention").
|
||||
section_match = re.search(
|
||||
r"^##\s+Retention[^\n]*\n(.*?)(?=^##\s|\Z)",
|
||||
content,
|
||||
flags=re.MULTILINE | re.DOTALL | re.IGNORECASE,
|
||||
)
|
||||
if not section_match:
|
||||
return {}
|
||||
|
||||
section_text = section_match.group(1)
|
||||
policies: dict[str, int | None] = {}
|
||||
|
||||
for line in section_text.splitlines():
|
||||
# Strip comments and whitespace
|
||||
line = re.sub(r"#.*$", "", line).strip()
|
||||
if not line:
|
||||
continue
|
||||
|
||||
# Match "<category>: <value>"
|
||||
line_match = re.match(r"^([^:]+):\s*(.+)$", line)
|
||||
if not line_match:
|
||||
continue
|
||||
|
||||
category = line_match.group(1).strip()
|
||||
value = line_match.group(2).strip().lower()
|
||||
|
||||
if value == "forever":
|
||||
policies[category] = None
|
||||
else:
|
||||
days_match = re.match(r"^(\d+)d$", value)
|
||||
if days_match:
|
||||
policies[category] = int(days_match.group(1))
|
||||
# else: malformed value — skip silently
|
||||
|
||||
return policies
|
||||
|
||||
|
||||
def load_retention_policies(cwd: str | None = None) -> dict[str, int | None]:
|
||||
"""Load retention policies from the mem0.md in *cwd*.
|
||||
|
||||
Combines :func:`find_mem0_config` and :func:`parse_retention` into a
|
||||
single convenience function.
|
||||
|
||||
Args:
|
||||
cwd: Directory to search. Defaults to ``os.getcwd()``.
|
||||
|
||||
Returns:
|
||||
Retention dict (category → days or ``None``). Empty dict if no
|
||||
``mem0.md`` exists in *cwd* or it contains no ``## Retention``
|
||||
section.
|
||||
"""
|
||||
if cwd is None:
|
||||
cwd = os.getcwd()
|
||||
|
||||
config_path = find_mem0_config(cwd)
|
||||
if config_path is None:
|
||||
return {}
|
||||
|
||||
try:
|
||||
with open(config_path, encoding="utf-8") as fh:
|
||||
content = fh.read()
|
||||
except OSError:
|
||||
return {}
|
||||
|
||||
return parse_retention(content)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
"""CLI entry point.
|
||||
|
||||
Reads cwd from ``sys.argv[1]`` (or ``os.getcwd()``), prints JSON to
|
||||
stdout.
|
||||
"""
|
||||
cwd = sys.argv[1] if len(sys.argv) > 1 else os.getcwd()
|
||||
policies = load_retention_policies(cwd)
|
||||
print(json.dumps(policies))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,250 @@
|
||||
---
|
||||
name: mem0-dream
|
||||
description: >
|
||||
Memory consolidation pass. Fetches all project memories, finds near-duplicates,
|
||||
merges them, flags contradictions, prunes stale entries per retention policy.
|
||||
Outputs a diff for user approval before applying changes.
|
||||
TRIGGER: user runs /mem0:dream, or asks "consolidate memories", "clean up memories",
|
||||
"merge duplicate memories", "run dream".
|
||||
---
|
||||
|
||||
# Mem0 Dream — Memory Consolidation
|
||||
|
||||
This skill performs a memory consolidation pass: it fetches all project memories,
|
||||
identifies near-duplicates, flags contradictions, and prunes stale entries based on
|
||||
configured retention policies. All proposed changes are shown as a diff for user
|
||||
approval before anything is modified.
|
||||
|
||||
---
|
||||
|
||||
## Step 1: Load Retention Policies
|
||||
|
||||
Determine the active retention policy by running the parser script. Use the
|
||||
appropriate `PLUGIN_ROOT` variable for the current platform (`${CLAUDE_PLUGIN_ROOT}`,
|
||||
`${CODEX_PLUGIN_ROOT}`, or `${CURSOR_PLUGIN_ROOT}`):
|
||||
|
||||
```bash
|
||||
python3 "<PLUGIN_ROOT>/scripts/parse_mem0_config.py" "<cwd>"
|
||||
```
|
||||
|
||||
Parse the JSON output (a dict of `category → days | null`). If the script fails
|
||||
or returns `{}`, fall back to these built-in defaults:
|
||||
|
||||
| `metadata.type` | Default retention |
|
||||
|---|---|
|
||||
| `session_state` | 90 days |
|
||||
| `compact_summary` | 90 days |
|
||||
| all others | no pruning |
|
||||
|
||||
Store the resolved policies for use in Step 3.
|
||||
|
||||
---
|
||||
|
||||
## Step 2: Fetch ALL Project Memories
|
||||
|
||||
Call `get_memories` to retrieve every memory for the active project:
|
||||
|
||||
```python
|
||||
get_memories(
|
||||
user_id="<active_user_id>",
|
||||
app_id="<active_project_id>",
|
||||
page_size=200,
|
||||
)
|
||||
```
|
||||
|
||||
If the response indicates more pages exist, paginate until all memories are fetched.
|
||||
Collect the full list before proceeding. If zero memories are found, print:
|
||||
|
||||
```
|
||||
No memories found for project <project_id>. Nothing to consolidate.
|
||||
```
|
||||
|
||||
…and stop.
|
||||
|
||||
---
|
||||
|
||||
## Step 3: Analyze — Find Issues
|
||||
|
||||
Work entirely in-memory; do not modify anything yet.
|
||||
|
||||
Group memories by `metadata.type` (use `"unknown"` when the field is absent).
|
||||
For each group, identify the following:
|
||||
|
||||
### 3a. Near-duplicate pairs (merge candidates)
|
||||
|
||||
Two memories are near-duplicates when they express the same fact or decision but
|
||||
phrased differently (e.g., "Use PostgreSQL for auth" and "Auth DB is PostgreSQL").
|
||||
|
||||
Heuristics:
|
||||
- Significant noun/keyword overlap in the memory text.
|
||||
- Same `metadata.type`.
|
||||
- Neither memory is pinned (`metadata.pinned != true`).
|
||||
|
||||
For each qualifying pair, draft a merged version that is more complete and specific
|
||||
than either original.
|
||||
|
||||
### 3b. Contradictions
|
||||
|
||||
Two memories contradict when they assert opposing facts about the same topic
|
||||
(e.g., "Deploy to ECS" vs. "Deploy to Vercel").
|
||||
|
||||
Identify the likely winner: the more recent memory with higher confidence wins.
|
||||
Store both IDs and their content for user review.
|
||||
|
||||
### 3c. Prune candidates
|
||||
|
||||
A memory is a prune candidate when **any** of the following is true:
|
||||
|
||||
1. Its `metadata.type` has a retention policy and the memory is older than the
|
||||
configured number of days (compare `created_at` to today).
|
||||
2. Its confidence score is below 0.3 AND it contains no information unique to
|
||||
this project (no file paths, identifiers, or domain-specific nouns).
|
||||
|
||||
**Always skip memories where `metadata.pinned == true`**, regardless of age or
|
||||
confidence.
|
||||
|
||||
---
|
||||
|
||||
## Step 4: Print Diff Report (item 15)
|
||||
|
||||
Print a structured diff to the terminal before making any changes. Use exactly
|
||||
this format:
|
||||
|
||||
```
|
||||
## Dream — Memory Consolidation Report
|
||||
|
||||
### Merge proposals (<N> pairs)
|
||||
MERGE [mem0:<id1>] + [mem0:<id2>] → NEW
|
||||
- Original 1: "<content of memory 1, truncated to 120 chars>"
|
||||
- Original 2: "<content of memory 2, truncated to 120 chars>"
|
||||
- Merged: "<drafted merged content>"
|
||||
|
||||
### Contradictions (<N> pairs)
|
||||
CONFLICT [mem0:<idA>] vs [mem0:<idB>]
|
||||
- A: "<content>" (<created_at date>, confidence: <score>)
|
||||
- B: "<content>" (<created_at date>, confidence: <score>)
|
||||
Which is current? [A/B/skip]
|
||||
|
||||
### Prune candidates (<N> memories)
|
||||
PRUNE [mem0:<id>] — <metadata.type>, <age>d old (policy: <policy_days>d)
|
||||
|
||||
---
|
||||
Proposed: <N> merges, <N> prunes, <N> conflicts
|
||||
Apply? [Y/n]
|
||||
```
|
||||
|
||||
If there are zero items in any category, omit that section entirely.
|
||||
|
||||
If there are zero total proposals (no merges, no prunes, no conflicts), print:
|
||||
|
||||
```
|
||||
Dream complete. No duplicate, contradictory, or stale memories found.
|
||||
```
|
||||
|
||||
…and stop.
|
||||
|
||||
---
|
||||
|
||||
## Step 5: Wait for User Input and Apply
|
||||
|
||||
### 5a. Contradictions
|
||||
|
||||
For each `CONFLICT` pair in the report, wait for the user to type `A`, `B`, or
|
||||
`skip` (case-insensitive). If they enter nothing (empty), treat as `skip`.
|
||||
|
||||
Record the winner for each pair before proceeding to the final apply confirmation.
|
||||
|
||||
### 5b. Final confirmation
|
||||
|
||||
After all conflict resolutions are collected, prompt:
|
||||
|
||||
```
|
||||
Apply? [Y/n]
|
||||
```
|
||||
|
||||
If the user types `n` or `no` (case-insensitive), print `Cancelled. No changes made.`
|
||||
and stop.
|
||||
|
||||
If the user confirms (`Y`, `yes`, or empty / Enter), apply all changes in this order:
|
||||
|
||||
#### Merges
|
||||
|
||||
For each approved merge pair:
|
||||
1. `delete_memory(<id1>)`
|
||||
2. `delete_memory(<id2>)`
|
||||
3. `add_memory` with:
|
||||
- `messages=[{"role": "user", "content": "<merged content>"}]`
|
||||
- `user_id=<active_user_id>`
|
||||
- `app_id=<active_project_id>` (top-level, not in metadata)
|
||||
- `metadata={"type": "<original type>", "branch": "<active_branch>", "confidence": <higher of the two original scores>, "source": "mem0-dream"}`
|
||||
- `infer=False`
|
||||
|
||||
#### Contradictions (resolved)
|
||||
|
||||
For each resolved conflict where the user chose A or B:
|
||||
- Identify the loser (the non-chosen memory).
|
||||
- `update_memory(<loser_id>, data={"metadata": {"confidence": 0.1, "superseded_by": "<winner_id>", "source": "mem0-dream"}})`
|
||||
|
||||
Contradictions where the user chose `skip` are left untouched.
|
||||
|
||||
#### Prunes
|
||||
|
||||
For each prune candidate:
|
||||
- `delete_memory(<memory_id>)`
|
||||
|
||||
---
|
||||
|
||||
## Step 6: Print Summary
|
||||
|
||||
After all changes are applied, print:
|
||||
|
||||
```
|
||||
Dream complete.
|
||||
Merged: <N> pairs → <N> new memories
|
||||
Pruned: <N> memories deleted
|
||||
Flagged: <N> contradictions resolved, <N> skipped
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Scheduling (Claude Code only)
|
||||
|
||||
### Running dream on a schedule
|
||||
|
||||
**Note**: Scheduled dream is only available in Claude Code, which natively supports
|
||||
the `schedule` skill. Codex and Cursor do not have a scheduling primitive — users
|
||||
on those platforms should run `/mem0:dream` manually.
|
||||
|
||||
When the user runs `/mem0:dream --schedule weekly` (or similar), register a
|
||||
recurring scheduled task using the Claude Code `schedule` skill.
|
||||
|
||||
Use `--auto` flag to enable non-interactive mode, which skips the interactive
|
||||
contradiction resolution UI and applies only safe operations automatically:
|
||||
|
||||
- **Merges**: applied automatically (no contradiction, both are compatible).
|
||||
- **Prunes**: applied automatically (age/confidence-based, no ambiguity).
|
||||
- **Contradictions**: skipped in `--auto` mode; they require human judgment.
|
||||
|
||||
Example scheduling invocation:
|
||||
|
||||
```
|
||||
/schedule weekly mem0:dream --auto
|
||||
```
|
||||
|
||||
When running with `--auto`:
|
||||
1. Load policies and fetch memories (Steps 1–3) as normal.
|
||||
2. Apply merges and prunes silently without printing the diff or prompting.
|
||||
3. Log a compact summary to stdout (suitable for a cron log):
|
||||
```
|
||||
[mem0-dream --auto] project=<id> merged=<N> pruned=<N> conflicts_skipped=<N>
|
||||
```
|
||||
4. If contradictions were detected but skipped, store a reminder memory:
|
||||
```python
|
||||
add_memory(
|
||||
messages=[{"role": "user", "content": "mem0-dream detected <N> contradiction(s) requiring manual review. Run /mem0:dream to resolve them interactively."}],
|
||||
user_id="<active_user_id>",
|
||||
app_id="<active_project_id>",
|
||||
metadata={"type": "task_learning", "source": "mem0-dream-auto", "branch": "<active_branch>"},
|
||||
infer=False,
|
||||
)
|
||||
```
|
||||
@@ -0,0 +1,250 @@
|
||||
"""Tests for parse_mem0_config.py — mem0.md retention policy parser."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# parse_retention — unit tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_parse_retention_valid_section():
|
||||
"""parse_retention extracts day-count policies correctly."""
|
||||
from parse_mem0_config import parse_retention
|
||||
|
||||
content = """\
|
||||
# Project Config
|
||||
|
||||
## Retention
|
||||
|
||||
session_state: 90d
|
||||
compact_summary: 60d
|
||||
decision: 180d
|
||||
"""
|
||||
result = parse_retention(content)
|
||||
assert result == {
|
||||
"session_state": 90,
|
||||
"compact_summary": 60,
|
||||
"decision": 180,
|
||||
}
|
||||
|
||||
|
||||
def test_parse_retention_forever_returns_none():
|
||||
"""parse_retention maps 'forever' to None."""
|
||||
from parse_mem0_config import parse_retention
|
||||
|
||||
content = """\
|
||||
## Retention
|
||||
|
||||
user_preference: forever
|
||||
anti_pattern: forever
|
||||
session_state: 30d
|
||||
"""
|
||||
result = parse_retention(content)
|
||||
assert result["user_preference"] is None
|
||||
assert result["anti_pattern"] is None
|
||||
assert result["session_state"] == 30
|
||||
|
||||
|
||||
def test_parse_retention_no_section_returns_empty():
|
||||
"""parse_retention returns {} when there is no ## Retention heading."""
|
||||
from parse_mem0_config import parse_retention
|
||||
|
||||
content = """\
|
||||
# Project Config
|
||||
|
||||
## Some Other Section
|
||||
|
||||
key: value
|
||||
"""
|
||||
result = parse_retention(content)
|
||||
assert result == {}
|
||||
|
||||
|
||||
def test_parse_retention_stops_at_next_heading():
|
||||
"""parse_retention stops reading at the next ## heading."""
|
||||
from parse_mem0_config import parse_retention
|
||||
|
||||
content = """\
|
||||
## Retention
|
||||
|
||||
session_state: 7d
|
||||
|
||||
## Other Section
|
||||
|
||||
other_key: 999d
|
||||
"""
|
||||
result = parse_retention(content)
|
||||
assert "session_state" in result
|
||||
assert "other_key" not in result
|
||||
|
||||
|
||||
def test_parse_retention_malformed_lines_skipped():
|
||||
"""Malformed lines (no colon, bad day format) are silently ignored."""
|
||||
from parse_mem0_config import parse_retention
|
||||
|
||||
content = """\
|
||||
## Retention
|
||||
|
||||
session_state: 90d
|
||||
bad_line_no_colon
|
||||
another: badvalue
|
||||
decision: 30d
|
||||
"""
|
||||
result = parse_retention(content)
|
||||
assert result == {"session_state": 90, "decision": 30}
|
||||
|
||||
|
||||
def test_parse_retention_comments_ignored():
|
||||
"""Inline # comments are stripped before parsing."""
|
||||
from parse_mem0_config import parse_retention
|
||||
|
||||
content = """\
|
||||
## Retention
|
||||
|
||||
session_state: 90d # rolling 90-day window
|
||||
user_preference: forever # never prune preferences
|
||||
"""
|
||||
result = parse_retention(content)
|
||||
assert result["session_state"] == 90
|
||||
assert result["user_preference"] is None
|
||||
|
||||
|
||||
def test_parse_retention_case_insensitive_heading():
|
||||
"""## retention (lowercase) is matched the same as ## Retention."""
|
||||
from parse_mem0_config import parse_retention
|
||||
|
||||
content = """\
|
||||
## retention
|
||||
|
||||
session_state: 14d
|
||||
"""
|
||||
result = parse_retention(content)
|
||||
assert result == {"session_state": 14}
|
||||
|
||||
|
||||
def test_parse_retention_empty_section_returns_empty():
|
||||
"""A ## Retention section with no valid lines returns {}."""
|
||||
from parse_mem0_config import parse_retention
|
||||
|
||||
content = """\
|
||||
## Retention
|
||||
|
||||
# only comments here
|
||||
|
||||
## Next Section
|
||||
"""
|
||||
result = parse_retention(content)
|
||||
assert result == {}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# load_retention_policies — integration tests with tmp files
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_load_retention_policies_with_tmp_file(tmp_path):
|
||||
"""load_retention_policies reads a real mem0.md from disk."""
|
||||
from parse_mem0_config import load_retention_policies
|
||||
|
||||
mem0_md = tmp_path / "mem0.md"
|
||||
mem0_md.write_text(
|
||||
"""\
|
||||
# My Project
|
||||
|
||||
## Retention
|
||||
|
||||
session_state: 90d
|
||||
compact_summary: 60d
|
||||
decision: forever
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
result = load_retention_policies(str(tmp_path))
|
||||
assert result == {
|
||||
"session_state": 90,
|
||||
"compact_summary": 60,
|
||||
"decision": None,
|
||||
}
|
||||
|
||||
|
||||
def test_load_retention_policies_no_mem0_md_returns_empty(tmp_path):
|
||||
"""load_retention_policies returns {} when no mem0.md exists."""
|
||||
from parse_mem0_config import load_retention_policies
|
||||
|
||||
result = load_retention_policies(str(tmp_path))
|
||||
assert result == {}
|
||||
|
||||
|
||||
def test_load_retention_policies_no_retention_section_returns_empty(tmp_path):
|
||||
"""load_retention_policies returns {} when mem0.md has no ## Retention."""
|
||||
from parse_mem0_config import load_retention_policies
|
||||
|
||||
mem0_md = tmp_path / "mem0.md"
|
||||
mem0_md.write_text(
|
||||
"""\
|
||||
# My Project
|
||||
|
||||
Some general project notes here.
|
||||
No retention section.
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
result = load_retention_policies(str(tmp_path))
|
||||
assert result == {}
|
||||
|
||||
|
||||
def test_load_retention_policies_defaults_to_cwd(tmp_path, monkeypatch):
|
||||
"""load_retention_policies uses os.getcwd() when cwd is None."""
|
||||
from parse_mem0_config import load_retention_policies
|
||||
|
||||
monkeypatch.chdir(tmp_path)
|
||||
mem0_md = tmp_path / "mem0.md"
|
||||
mem0_md.write_text("## Retention\nsession_state: 45d\n", encoding="utf-8")
|
||||
|
||||
result = load_retention_policies() # no cwd arg
|
||||
assert result == {"session_state": 45}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# CLI / main() — subprocess test
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_cli_main_prints_json(tmp_path):
|
||||
"""CLI: python parse_mem0_config.py <cwd> prints valid JSON."""
|
||||
mem0_md = tmp_path / "mem0.md"
|
||||
mem0_md.write_text(
|
||||
"## Retention\nsession_state: 90d\nuser_preference: forever\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
result = subprocess.run(
|
||||
[sys.executable, os.path.join(SCRIPTS_DIR, "parse_mem0_config.py"), str(tmp_path)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert result.returncode == 0
|
||||
data = json.loads(result.stdout)
|
||||
assert data["session_state"] == 90
|
||||
assert data["user_preference"] is None
|
||||
|
||||
|
||||
def test_cli_main_no_file_prints_empty_json(tmp_path):
|
||||
"""CLI: prints '{}' when no mem0.md exists."""
|
||||
result = subprocess.run(
|
||||
[sys.executable, os.path.join(SCRIPTS_DIR, "parse_mem0_config.py"), str(tmp_path)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert result.returncode == 0
|
||||
assert json.loads(result.stdout) == {}
|
||||
Reference in New Issue
Block a user