From 71a6696c875c4e3a83337593e894b9ddc00798da Mon Sep 17 00:00:00 2001 From: kartik-mem0 Date: Fri, 22 May 2026 14:23:06 +0530 Subject: [PATCH] refactor(mem0-plugin): improve hook scripts, config parsing, and skill definitions Update git commit capture with better error handling, enhance user prompt hook, extend config parser capabilities, and refine skill docs for dream, health, mcp, pin, and stats skills. --- mem0-plugin/scripts/on_git_commit_capture.sh | 61 ++++++++++++++++++-- mem0-plugin/scripts/on_user_prompt.sh | 5 +- mem0-plugin/scripts/parse_mem0_config.py | 32 +++++++++- mem0-plugin/skills/mem0-dream/SKILL.md | 4 +- mem0-plugin/skills/mem0-health/SKILL.md | 32 +++++++++- mem0-plugin/skills/mem0-mcp/SKILL.md | 21 ++++++- mem0-plugin/skills/mem0-pin/SKILL.md | 19 ++++-- mem0-plugin/skills/mem0-stats/SKILL.md | 14 +++++ 8 files changed, 170 insertions(+), 18 deletions(-) diff --git a/mem0-plugin/scripts/on_git_commit_capture.sh b/mem0-plugin/scripts/on_git_commit_capture.sh index 1ced327a0..b2aba1b6f 100755 --- a/mem0-plugin/scripts/on_git_commit_capture.sh +++ b/mem0-plugin/scripts/on_git_commit_capture.sh @@ -1,13 +1,14 @@ #!/usr/bin/env bash # Hook: PreToolUse (matcher: Bash) # -# Detects `git commit` commands and fires on_pre_commit.py in the background -# to capture staged changes as a mem0 memory. Never blocks the commit. +# Detects `git commit` commands and: +# 1. Fires on_pre_commit.py in the background to capture staged changes as memory +# 2. Searches for relevant memories about the changed files and surfaces them # # Input: JSON on stdin with tool_name, tool_input -# Output: none (always exit 0) +# Output: JSON with additionalContext (relevant memories for the commit) -set -euo pipefail +set -uo pipefail INPUT=$(cat) @@ -27,7 +28,7 @@ esac SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" -if [ ! -x "$SCRIPT_DIR/on_pre_commit.py" ] && [ ! -f "$SCRIPT_DIR/on_pre_commit.py" ]; then +if [ ! -f "$SCRIPT_DIR/on_pre_commit.py" ]; then exit 0 fi @@ -36,6 +37,56 @@ if [ -z "$API_KEY" ]; then exit 0 fi +# Background: capture staged changes as memory git diff --cached --stat 2>/dev/null | python3 "$SCRIPT_DIR/on_pre_commit.py" & +# Foreground: search for relevant memories about changed files +CHANGED_FILES=$(git diff --cached --name-only 2>/dev/null | head -10 | tr '\n' ', ' | sed 's/,$//') +if [ -z "$CHANGED_FILES" ]; then + exit 0 +fi + +USER_ID="${MEM0_RESOLVED_USER_ID:-$USER}" +PROJECT_ID="${MEM0_PROJECT_ID:-unknown}" + +CONTEXT=$(python3 -c " +import json, urllib.request, os +api_key = os.environ.get('MEM0_API_KEY', os.environ.get('CLAUDE_PLUGIN_OPTION_MEM0_API_KEY', '')) +user_id = '$USER_ID' +app_id = '$PROJECT_ID' +files = '$CHANGED_FILES' +first_file = files.split(',')[0].strip() +body = json.dumps({ + 'query': f'changes to {files}', + 'filters': {'AND': [{'user_id': user_id}, {'app_id': app_id}]}, + 'top_k': 3, +}).encode() +req = urllib.request.Request( + 'https://api.mem0.ai/v2/memories/search/', + data=body, + headers={'Authorization': f'Token {api_key}', 'Content-Type': 'application/json'}, + method='POST', +) +try: + with urllib.request.urlopen(req, timeout=5) as r: + results = json.loads(r.read()) + memories = results if isinstance(results, list) else results.get('results', []) + if memories: + lines = ['## Pre-Commit Memory Check', '', 'Relevant memories for files being committed (' + files + '):', ''] + for m in memories[:3]: + mid = m.get('id', '?')[:8] + text = m.get('memory', '')[:200] + cat = (m.get('metadata') or {}).get('type', 'unknown') + lines.append(f'- [{cat}] {text} [mem0:{mid}]') + lines.append('') + lines.append('Consider: does this commit introduce a learning worth saving? If so, suggest storing it after the commit completes.') + print('\\n'.join(lines)) +except Exception: + pass +" 2>/dev/null || true) + +if [ -n "$CONTEXT" ]; then + jq -nc --arg ctx "$CONTEXT" '{hookSpecificOutput:{hookEventName:"PreToolUse",additionalContext:$ctx}}' +fi + exit 0 diff --git a/mem0-plugin/scripts/on_user_prompt.sh b/mem0-plugin/scripts/on_user_prompt.sh index 96b0570f5..b93b81e2d 100755 --- a/mem0-plugin/scripts/on_user_prompt.sh +++ b/mem0-plugin/scripts/on_user_prompt.sh @@ -90,9 +90,10 @@ if [ -n "$FILE_PATHS" ]; then cat <"}}]})\` +- Also run a broader text search without the files filter as fallback: - \`search_memories(query="", filters={"AND": [{"user_id": "$USER_ID"}, {"app_id": "$MEM0_PROJECT_ID"}]})\` -Memories tagged with \`metadata.files\` containing these paths will surface via text match. EOF fi diff --git a/mem0-plugin/scripts/parse_mem0_config.py b/mem0-plugin/scripts/parse_mem0_config.py index 6b93b0999..0db049c22 100644 --- a/mem0-plugin/scripts/parse_mem0_config.py +++ b/mem0-plugin/scripts/parse_mem0_config.py @@ -132,10 +132,33 @@ def parse_section_list(content: str, heading: str) -> list[str]: return items +def parse_ignore_patterns(content: str) -> list[str]: + """Parse the ``## Ignore`` section of *content*. + + Each non-blank line is a glob pattern (e.g., ``node_modules``, ``*.lock``). + Lines starting with ``#`` are comments and skipped. + """ + pattern = r"^##\s+Ignore[^\n]*\n(.*?)(?=^##\s|\Z)" + match = re.search(pattern, content, flags=re.MULTILINE | re.DOTALL | re.IGNORECASE) + if not match: + return [] + + patterns: list[str] = [] + for line in match.group(1).splitlines(): + line = line.strip() + if not line or line.startswith("#"): + continue + line = re.sub(r"^[-*]\s+", "", line).strip() + if line: + patterns.append(line) + return patterns + + def load_full_config(cwd: str | None = None) -> dict: """Load all config sections from mem0.md. - Returns a dict with keys: retention, search, categories, identity. + Returns a dict with keys: retention, search, categories, identity, + ignore, project_id. Each is populated only if the corresponding ``##`` section exists. """ if cwd is None: @@ -164,10 +187,17 @@ def load_full_config(cwd: str | None = None) -> dict: categories = parse_section_list(content, "Categories") if categories: config["categories"] = categories + config["default_categories"] = categories identity = parse_section_kv(content, "Identity") if identity: config["identity"] = identity + if "project_id" in identity: + config["project_id"] = identity["project_id"] + + ignore = parse_ignore_patterns(content) + if ignore: + config["ignore"] = ignore return config diff --git a/mem0-plugin/skills/mem0-dream/SKILL.md b/mem0-plugin/skills/mem0-dream/SKILL.md index 3466a4568..9964b72ca 100644 --- a/mem0-plugin/skills/mem0-dream/SKILL.md +++ b/mem0-plugin/skills/mem0-dream/SKILL.md @@ -75,8 +75,8 @@ For each group, identify the following: Two memories are near-duplicates when they express the same fact or decision but phrased differently (e.g., "Use PostgreSQL for auth" and "Auth DB is PostgreSQL"). -Heuristics: -- Significant noun/keyword overlap in the memory text. +Heuristics — two memories are near-duplicates if **all** of these hold: +- Similarity threshold: estimated cosine similarity > 0.9 (use noun/keyword overlap as proxy — if >60% of significant nouns overlap, treat as >0.9 similarity). - Same `metadata.type`. - Neither memory is pinned (`metadata.pinned != true`). diff --git a/mem0-plugin/skills/mem0-health/SKILL.md b/mem0-plugin/skills/mem0-health/SKILL.md index 5901d271d..d23fd8c28 100644 --- a/mem0-plugin/skills/mem0-health/SKILL.md +++ b/mem0-plugin/skills/mem0-health/SKILL.md @@ -79,7 +79,9 @@ If any check fails, add a `## Troubleshooting` section with specific fix steps f ## Extended mode: Memory Quality Analysis -When invoked with `--deep` (e.g., `/mem0:health --deep`), run the standard 5 checks above **plus** a memory quality scan: +When invoked with `--deep` (e.g., `/mem0:health --deep`) or `--fix` (e.g., `/mem0:health --fix`), run the standard 5 checks above **plus** a memory quality scan. + +`--fix` implies `--deep` and automatically applies safe fixes after showing the analysis (see bottom of this section). ### Quality Check 1: Duplicates @@ -101,6 +103,15 @@ Stale candidates: [mem0:] — session_state, 142d old ``` +### Quality Check 2b: Low-confidence memories + +Flag memories where `metadata.confidence` < 0.5 (regardless of age). Report separately from stale: + +``` +Low-confidence memories: + [mem0:] — confidence=0.3, "" +``` + ### Quality Check 3: Contradictions Within each `metadata.type` group, flag pairs that assert opposing facts about the same topic. Use semantic judgment — look for negation patterns, conflicting tool/framework choices, or reversed decisions. @@ -131,3 +142,22 @@ Untagged/orphan memories: ``` If all counts are 0: `Memory quality: clean. No duplicates, stale entries, or contradictions found.` + +### Auto-fix mode (`--fix`) + +When `--fix` is passed, apply these safe fixes automatically after displaying the quality summary: + +1. **Orphans:** For each untagged memory, infer a `metadata.type` from content and call `update_memory` to set it. If inference is uncertain, skip. +2. **Stale `session_state`/`compact_summary` > 90d:** Delete them via `delete_memory`. These are ephemeral by design. +3. **Duplicates:** Do NOT auto-merge — print "Run `/mem0:dream` to merge duplicates" instead. +4. **Contradictions:** Do NOT auto-resolve — print "Run `/mem0:dream` to resolve contradictions" instead. +5. **Low-confidence < 0.3 AND > 30d old:** Delete them via `delete_memory`. + +Print a summary of actions taken: + +``` +## Auto-fix Results + Deleted: stale, low-confidence + Retagged: orphans + Skipped: duplicates (use /mem0:dream), contradictions (use /mem0:dream) +``` diff --git a/mem0-plugin/skills/mem0-mcp/SKILL.md b/mem0-plugin/skills/mem0-mcp/SKILL.md index 4f560081e..750e0275f 100644 --- a/mem0-plugin/skills/mem0-mcp/SKILL.md +++ b/mem0-plugin/skills/mem0-mcp/SKILL.md @@ -272,7 +272,23 @@ add_memory( ) ``` -**Filtering note:** The mem0 v2 filter API does not yet support `array-contains` predicates. You cannot filter by `metadata.files` at search time. To work around this, always embed the bare filenames (and important path segments) in the memory content text itself — the vector search will then surface them on a filename query. The `files` array in metadata is still written for future compatibility once array-contains filtering is available. +**Filtering by files:** Use the `contains` operator to filter by `metadata.files` at search time: + +```python +search_memories( + query="auth middleware", + filters={ + "AND": [ + {"user_id": ""}, + {"app_id": ""}, + {"metadata.files": {"contains": "src/middleware/auth.ts"}}, + ] + }, + limit=5, +) +``` + +Also embed bare filenames in the memory content text as a fallback — the vector search will surface them even if the structured filter misses. ### Access counter: track memory usage @@ -289,10 +305,11 @@ import datetime current_meta["access_count"] = current_meta.get("access_count", 0) + 1 current_meta["last_accessed"] = datetime.datetime.now(datetime.timezone.utc).isoformat() -# 3. Update with preserved content +# 3. Update with preserved content and bumped metadata update_memory( memory_id=, data=current_text, # preserve original text — required parameter + metadata=current_meta, # pass updated access_count and last_accessed ) ``` diff --git a/mem0-plugin/skills/mem0-pin/SKILL.md b/mem0-plugin/skills/mem0-pin/SKILL.md index 2cb111cd6..84e97aaea 100644 --- a/mem0-plugin/skills/mem0-pin/SKILL.md +++ b/mem0-plugin/skills/mem0-pin/SKILL.md @@ -38,10 +38,15 @@ This is required because `update_memory` replaces the full memory — a metadata Call `update_memory` with: - `memory_id=` - `data=` (preserve the existing content) +- `metadata=` merge `original_metadata` with `{"pinned": true}` -The platform will retain the existing metadata and content. Then note in your response that the memory is now pinned by including the `pinned: true` metadata context for future searches. +Example: +```python +updated_meta = {**original_metadata, "pinned": True} +update_memory(memory_id=, data=, metadata=updated_meta) +``` -**Important:** `update_memory` requires the `data` (text) parameter. Passing only metadata may error or wipe content. Always read first, then update with the full text. +**Important:** `update_memory` requires the `data` (text) parameter. Passing only metadata may error or wipe content. Always read first, then update with the full text and explicit metadata. ### Step 4: Confirm @@ -53,7 +58,11 @@ Pinned memories surface first when relevant to a search. ### Unpin -If the user says "unpin" or the memory is already pinned: -1. Call `get_memory` to read current content. -2. Call `update_memory` with `data=` to preserve content. +If the user says "unpin" or `/mem0:unpin`: +1. Call `get_memory` to read current content and metadata. +2. Set `metadata.pinned = false` explicitly: + ```python + updated_meta = {**original_metadata, "pinned": False} + update_memory(memory_id=, data=, metadata=updated_meta) + ``` 3. Print: `Unpinned: "..."` diff --git a/mem0-plugin/skills/mem0-stats/SKILL.md b/mem0-plugin/skills/mem0-stats/SKILL.md index a461838fc..d2d67c259 100644 --- a/mem0-plugin/skills/mem0-stats/SKILL.md +++ b/mem0-plugin/skills/mem0-stats/SKILL.md @@ -63,6 +63,18 @@ Print a compact dashboard with an ASCII histogram for category distribution: user_preference ██░░░░░░░░░░░░░░ 3 session_state █░░░░░░░░░░░░░░░ 2 + By age: + < 7 days ████████████████ 5 + 7–30 days ██████████░░░░░░ 12 + 30–90 days ████░░░░░░░░░░░░ 10 + > 90 days ██░░░░░░░░░░░░░░ 8 + + By access count: + Never accessed ████████████████ 18 + 1–5 accesses ████████░░░░░░░░ 10 + 6–20 accesses ████░░░░░░░░░░░░ 4 + 20+ accesses █░░░░░░░░░░░░░░░ 3 + Oldest memory: Newest memory: @@ -78,5 +90,7 @@ Print a compact dashboard with an ASCII histogram for category distribution: - Use `█` for filled and `░` for empty. Right-align the count number. - Sort categories by count descending. Omit categories with 0 memories. - If only 1-2 categories exist, still show the histogram — it provides visual context. +- **Age buckets:** Compute from `created_at`. Buckets: <7d, 7–30d, 30–90d, >90d. +- **Access count buckets:** Read `metadata.access_count` (default 0 if absent). Buckets: 0, 1–5, 6–20, 20+. Skip any section with zero data.