refactor(mem0-plugin): improve hook scripts, config parsing, and skill definitions

Update git commit capture with better error handling, enhance user prompt hook,
extend config parser capabilities, and refine skill docs for dream, health,
mcp, pin, and stats skills.
This commit is contained in:
kartik-mem0
2026-05-22 14:23:06 +05:30
parent 6ab6ed58a8
commit 71a6696c87
8 changed files with 170 additions and 18 deletions
+56 -5
View File
@@ -1,13 +1,14 @@
#!/usr/bin/env bash
# Hook: PreToolUse (matcher: Bash)
#
# Detects `git commit` commands and fires on_pre_commit.py in the background
# to capture staged changes as a mem0 memory. Never blocks the commit.
# Detects `git commit` commands and:
# 1. Fires on_pre_commit.py in the background to capture staged changes as memory
# 2. Searches for relevant memories about the changed files and surfaces them
#
# Input: JSON on stdin with tool_name, tool_input
# Output: none (always exit 0)
# Output: JSON with additionalContext (relevant memories for the commit)
set -euo pipefail
set -uo pipefail
INPUT=$(cat)
@@ -27,7 +28,7 @@ esac
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
if [ ! -x "$SCRIPT_DIR/on_pre_commit.py" ] && [ ! -f "$SCRIPT_DIR/on_pre_commit.py" ]; then
if [ ! -f "$SCRIPT_DIR/on_pre_commit.py" ]; then
exit 0
fi
@@ -36,6 +37,56 @@ if [ -z "$API_KEY" ]; then
exit 0
fi
# Background: capture staged changes as memory
git diff --cached --stat 2>/dev/null | python3 "$SCRIPT_DIR/on_pre_commit.py" &
# Foreground: search for relevant memories about changed files
CHANGED_FILES=$(git diff --cached --name-only 2>/dev/null | head -10 | tr '\n' ', ' | sed 's/,$//')
if [ -z "$CHANGED_FILES" ]; then
exit 0
fi
USER_ID="${MEM0_RESOLVED_USER_ID:-$USER}"
PROJECT_ID="${MEM0_PROJECT_ID:-unknown}"
CONTEXT=$(python3 -c "
import json, urllib.request, os
api_key = os.environ.get('MEM0_API_KEY', os.environ.get('CLAUDE_PLUGIN_OPTION_MEM0_API_KEY', ''))
user_id = '$USER_ID'
app_id = '$PROJECT_ID'
files = '$CHANGED_FILES'
first_file = files.split(',')[0].strip()
body = json.dumps({
'query': f'changes to {files}',
'filters': {'AND': [{'user_id': user_id}, {'app_id': app_id}]},
'top_k': 3,
}).encode()
req = urllib.request.Request(
'https://api.mem0.ai/v2/memories/search/',
data=body,
headers={'Authorization': f'Token {api_key}', 'Content-Type': 'application/json'},
method='POST',
)
try:
with urllib.request.urlopen(req, timeout=5) as r:
results = json.loads(r.read())
memories = results if isinstance(results, list) else results.get('results', [])
if memories:
lines = ['## Pre-Commit Memory Check', '', 'Relevant memories for files being committed (' + files + '):', '']
for m in memories[:3]:
mid = m.get('id', '?')[:8]
text = m.get('memory', '')[:200]
cat = (m.get('metadata') or {}).get('type', 'unknown')
lines.append(f'- [{cat}] {text} [mem0:{mid}]')
lines.append('')
lines.append('Consider: does this commit introduce a learning worth saving? If so, suggest storing it after the commit completes.')
print('\\n'.join(lines))
except Exception:
pass
" 2>/dev/null || true)
if [ -n "$CONTEXT" ]; then
jq -nc --arg ctx "$CONTEXT" '{hookSpecificOutput:{hookEventName:"PreToolUse",additionalContext:$ctx}}'
fi
exit 0
+3 -2
View File
@@ -90,9 +90,10 @@ if [ -n "$FILE_PATHS" ]; then
cat <<EOF
**FILE PATHS detected:** \`$FILE_PATHS\`
Search mem0 for context about these files:
Search mem0 for context about these files using the \`contains\` operator on \`metadata.files\`:
- \`search_memories(query="<filename>", filters={"AND": [{"user_id": "$USER_ID"}, {"app_id": "$MEM0_PROJECT_ID"}, {"metadata.files": {"contains": "<filename>"}}]})\`
- Also run a broader text search without the files filter as fallback:
- \`search_memories(query="<filename without extension>", filters={"AND": [{"user_id": "$USER_ID"}, {"app_id": "$MEM0_PROJECT_ID"}]})\`
Memories tagged with \`metadata.files\` containing these paths will surface via text match.
EOF
fi
+31 -1
View File
@@ -132,10 +132,33 @@ def parse_section_list(content: str, heading: str) -> list[str]:
return items
def parse_ignore_patterns(content: str) -> list[str]:
"""Parse the ``## Ignore`` section of *content*.
Each non-blank line is a glob pattern (e.g., ``node_modules``, ``*.lock``).
Lines starting with ``#`` are comments and skipped.
"""
pattern = r"^##\s+Ignore[^\n]*\n(.*?)(?=^##\s|\Z)"
match = re.search(pattern, content, flags=re.MULTILINE | re.DOTALL | re.IGNORECASE)
if not match:
return []
patterns: list[str] = []
for line in match.group(1).splitlines():
line = line.strip()
if not line or line.startswith("#"):
continue
line = re.sub(r"^[-*]\s+", "", line).strip()
if line:
patterns.append(line)
return patterns
def load_full_config(cwd: str | None = None) -> dict:
"""Load all config sections from mem0.md.
Returns a dict with keys: retention, search, categories, identity.
Returns a dict with keys: retention, search, categories, identity,
ignore, project_id.
Each is populated only if the corresponding ``##`` section exists.
"""
if cwd is None:
@@ -164,10 +187,17 @@ def load_full_config(cwd: str | None = None) -> dict:
categories = parse_section_list(content, "Categories")
if categories:
config["categories"] = categories
config["default_categories"] = categories
identity = parse_section_kv(content, "Identity")
if identity:
config["identity"] = identity
if "project_id" in identity:
config["project_id"] = identity["project_id"]
ignore = parse_ignore_patterns(content)
if ignore:
config["ignore"] = ignore
return config
+2 -2
View File
@@ -75,8 +75,8 @@ For each group, identify the following:
Two memories are near-duplicates when they express the same fact or decision but
phrased differently (e.g., "Use PostgreSQL for auth" and "Auth DB is PostgreSQL").
Heuristics:
- Significant noun/keyword overlap in the memory text.
Heuristics — two memories are near-duplicates if **all** of these hold:
- Similarity threshold: estimated cosine similarity > 0.9 (use noun/keyword overlap as proxy — if >60% of significant nouns overlap, treat as >0.9 similarity).
- Same `metadata.type`.
- Neither memory is pinned (`metadata.pinned != true`).
+31 -1
View File
@@ -79,7 +79,9 @@ If any check fails, add a `## Troubleshooting` section with specific fix steps f
## Extended mode: Memory Quality Analysis
When invoked with `--deep` (e.g., `/mem0:health --deep`), run the standard 5 checks above **plus** a memory quality scan:
When invoked with `--deep` (e.g., `/mem0:health --deep`) or `--fix` (e.g., `/mem0:health --fix`), run the standard 5 checks above **plus** a memory quality scan.
`--fix` implies `--deep` and automatically applies safe fixes after showing the analysis (see bottom of this section).
### Quality Check 1: Duplicates
@@ -101,6 +103,15 @@ Stale candidates: <N>
[mem0:<id>] — session_state, 142d old
```
### Quality Check 2b: Low-confidence memories
Flag memories where `metadata.confidence` < 0.5 (regardless of age). Report separately from stale:
```
Low-confidence memories: <N>
[mem0:<id>] — confidence=0.3, "<content preview>"
```
### Quality Check 3: Contradictions
Within each `metadata.type` group, flag pairs that assert opposing facts about the same topic. Use semantic judgment — look for negation patterns, conflicting tool/framework choices, or reversed decisions.
@@ -131,3 +142,22 @@ Untagged/orphan memories: <N>
```
If all counts are 0: `Memory quality: clean. No duplicates, stale entries, or contradictions found.`
### Auto-fix mode (`--fix`)
When `--fix` is passed, apply these safe fixes automatically after displaying the quality summary:
1. **Orphans:** For each untagged memory, infer a `metadata.type` from content and call `update_memory` to set it. If inference is uncertain, skip.
2. **Stale `session_state`/`compact_summary` > 90d:** Delete them via `delete_memory`. These are ephemeral by design.
3. **Duplicates:** Do NOT auto-merge — print "Run `/mem0:dream` to merge duplicates" instead.
4. **Contradictions:** Do NOT auto-resolve — print "Run `/mem0:dream` to resolve contradictions" instead.
5. **Low-confidence < 0.3 AND > 30d old:** Delete them via `delete_memory`.
Print a summary of actions taken:
```
## Auto-fix Results
Deleted: <N> stale, <N> low-confidence
Retagged: <N> orphans
Skipped: <N> duplicates (use /mem0:dream), <N> contradictions (use /mem0:dream)
```
+19 -2
View File
@@ -272,7 +272,23 @@ add_memory(
)
```
**Filtering note:** The mem0 v2 filter API does not yet support `array-contains` predicates. You cannot filter by `metadata.files` at search time. To work around this, always embed the bare filenames (and important path segments) in the memory content text itself — the vector search will then surface them on a filename query. The `files` array in metadata is still written for future compatibility once array-contains filtering is available.
**Filtering by files:** Use the `contains` operator to filter by `metadata.files` at search time:
```python
search_memories(
query="auth middleware",
filters={
"AND": [
{"user_id": "<id>"},
{"app_id": "<project_id>"},
{"metadata.files": {"contains": "src/middleware/auth.ts"}},
]
},
limit=5,
)
```
Also embed bare filenames in the memory content text as a fallback — the vector search will surface them even if the structured filter misses.
### Access counter: track memory usage
@@ -289,10 +305,11 @@ import datetime
current_meta["access_count"] = current_meta.get("access_count", 0) + 1
current_meta["last_accessed"] = datetime.datetime.now(datetime.timezone.utc).isoformat()
# 3. Update with preserved content
# 3. Update with preserved content and bumped metadata
update_memory(
memory_id=<id>,
data=current_text, # preserve original text — required parameter
metadata=current_meta, # pass updated access_count and last_accessed
)
```
+14 -5
View File
@@ -38,10 +38,15 @@ This is required because `update_memory` replaces the full memory — a metadata
Call `update_memory` with:
- `memory_id=<selected_id>`
- `data=<original_text>` (preserve the existing content)
- `metadata=` merge `original_metadata` with `{"pinned": true}`
The platform will retain the existing metadata and content. Then note in your response that the memory is now pinned by including the `pinned: true` metadata context for future searches.
Example:
```python
updated_meta = {**original_metadata, "pinned": True}
update_memory(memory_id=<selected_id>, data=<original_text>, metadata=updated_meta)
```
**Important:** `update_memory` requires the `data` (text) parameter. Passing only metadata may error or wipe content. Always read first, then update with the full text.
**Important:** `update_memory` requires the `data` (text) parameter. Passing only metadata may error or wipe content. Always read first, then update with the full text and explicit metadata.
### Step 4: Confirm
@@ -53,7 +58,11 @@ Pinned memories surface first when relevant to a search.
### Unpin
If the user says "unpin" or the memory is already pinned:
1. Call `get_memory` to read current content.
2. Call `update_memory` with `data=<original_text>` to preserve content.
If the user says "unpin" or `/mem0:unpin`:
1. Call `get_memory` to read current content and metadata.
2. Set `metadata.pinned = false` explicitly:
```python
updated_meta = {**original_metadata, "pinned": False}
update_memory(memory_id=<id>, data=<original_text>, metadata=updated_meta)
```
3. Print: `Unpinned: "<content>..."`
+14
View File
@@ -63,6 +63,18 @@ Print a compact dashboard with an ASCII histogram for category distribution:
user_preference ██░░░░░░░░░░░░░░ 3
session_state █░░░░░░░░░░░░░░░ 2
By age:
< 7 days ████████████████ 5
7–30 days ██████████░░░░░░ 12
30–90 days ████░░░░░░░░░░░░ 10
> 90 days ██░░░░░░░░░░░░░░ 8
By access count:
Never accessed ████████████████ 18
1–5 accesses ████████░░░░░░░░ 10
6–20 accesses ████░░░░░░░░░░░░ 4
20+ accesses █░░░░░░░░░░░░░░░ 3
Oldest memory: <date>
Newest memory: <date>
@@ -78,5 +90,7 @@ Print a compact dashboard with an ASCII histogram for category distribution:
- Use `█` for filled and `░` for empty. Right-align the count number.
- Sort categories by count descending. Omit categories with 0 memories.
- If only 1-2 categories exist, still show the histogram — it provides visual context.
- **Age buckets:** Compute from `created_at`. Buckets: <7d, 7–30d, 30–90d, >90d.
- **Access count buckets:** Read `metadata.access_count` (default 0 if absent). Buckets: 0, 1–5, 6–20, 20+.
Skip any section with zero data.