diff --git a/mem0-plugin/scripts/on_git_commit_capture.sh b/mem0-plugin/scripts/on_git_commit_capture.sh index 1ced327a0..b2aba1b6f 100755 --- a/mem0-plugin/scripts/on_git_commit_capture.sh +++ b/mem0-plugin/scripts/on_git_commit_capture.sh @@ -1,13 +1,14 @@ #!/usr/bin/env bash # Hook: PreToolUse (matcher: Bash) # -# Detects `git commit` commands and fires on_pre_commit.py in the background -# to capture staged changes as a mem0 memory. Never blocks the commit. +# Detects `git commit` commands and: +# 1. Fires on_pre_commit.py in the background to capture staged changes as memory +# 2. Searches for relevant memories about the changed files and surfaces them # # Input: JSON on stdin with tool_name, tool_input -# Output: none (always exit 0) +# Output: JSON with additionalContext (relevant memories for the commit) -set -euo pipefail +set -uo pipefail INPUT=$(cat) @@ -27,7 +28,7 @@ esac SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" -if [ ! -x "$SCRIPT_DIR/on_pre_commit.py" ] && [ ! -f "$SCRIPT_DIR/on_pre_commit.py" ]; then +if [ ! -f "$SCRIPT_DIR/on_pre_commit.py" ]; then exit 0 fi @@ -36,6 +37,56 @@ if [ -z "$API_KEY" ]; then exit 0 fi +# Background: capture staged changes as memory git diff --cached --stat 2>/dev/null | python3 "$SCRIPT_DIR/on_pre_commit.py" & +# Foreground: search for relevant memories about changed files +CHANGED_FILES=$(git diff --cached --name-only 2>/dev/null | head -10 | tr '\n' ', ' | sed 's/,$//') +if [ -z "$CHANGED_FILES" ]; then + exit 0 +fi + +USER_ID="${MEM0_RESOLVED_USER_ID:-$USER}" +PROJECT_ID="${MEM0_PROJECT_ID:-unknown}" + +CONTEXT=$(python3 -c " +import json, urllib.request, os +api_key = os.environ.get('MEM0_API_KEY', os.environ.get('CLAUDE_PLUGIN_OPTION_MEM0_API_KEY', '')) +user_id = '$USER_ID' +app_id = '$PROJECT_ID' +files = '$CHANGED_FILES' +first_file = files.split(',')[0].strip() +body = json.dumps({ + 'query': f'changes to {files}', + 'filters': {'AND': [{'user_id': user_id}, {'app_id': app_id}]}, + 'top_k': 3, +}).encode() +req = urllib.request.Request( + 'https://api.mem0.ai/v2/memories/search/', + data=body, + headers={'Authorization': f'Token {api_key}', 'Content-Type': 'application/json'}, + method='POST', +) +try: + with urllib.request.urlopen(req, timeout=5) as r: + results = json.loads(r.read()) + memories = results if isinstance(results, list) else results.get('results', []) + if memories: + lines = ['## Pre-Commit Memory Check', '', 'Relevant memories for files being committed (' + files + '):', ''] + for m in memories[:3]: + mid = m.get('id', '?')[:8] + text = m.get('memory', '')[:200] + cat = (m.get('metadata') or {}).get('type', 'unknown') + lines.append(f'- [{cat}] {text} [mem0:{mid}]') + lines.append('') + lines.append('Consider: does this commit introduce a learning worth saving? If so, suggest storing it after the commit completes.') + print('\\n'.join(lines)) +except Exception: + pass +" 2>/dev/null || true) + +if [ -n "$CONTEXT" ]; then + jq -nc --arg ctx "$CONTEXT" '{hookSpecificOutput:{hookEventName:"PreToolUse",additionalContext:$ctx}}' +fi + exit 0 diff --git a/mem0-plugin/scripts/on_user_prompt.sh b/mem0-plugin/scripts/on_user_prompt.sh index 96b0570f5..b93b81e2d 100755 --- a/mem0-plugin/scripts/on_user_prompt.sh +++ b/mem0-plugin/scripts/on_user_prompt.sh @@ -90,9 +90,10 @@ if [ -n "$FILE_PATHS" ]; then cat <"}}]})\` +- Also run a broader text search without the files filter as fallback: - \`search_memories(query="", filters={"AND": [{"user_id": "$USER_ID"}, {"app_id": "$MEM0_PROJECT_ID"}]})\` -Memories tagged with \`metadata.files\` containing these paths will surface via text match. EOF fi diff --git a/mem0-plugin/scripts/parse_mem0_config.py b/mem0-plugin/scripts/parse_mem0_config.py index 6b93b0999..0db049c22 100644 --- a/mem0-plugin/scripts/parse_mem0_config.py +++ b/mem0-plugin/scripts/parse_mem0_config.py @@ -132,10 +132,33 @@ def parse_section_list(content: str, heading: str) -> list[str]: return items +def parse_ignore_patterns(content: str) -> list[str]: + """Parse the ``## Ignore`` section of *content*. + + Each non-blank line is a glob pattern (e.g., ``node_modules``, ``*.lock``). + Lines starting with ``#`` are comments and skipped. + """ + pattern = r"^##\s+Ignore[^\n]*\n(.*?)(?=^##\s|\Z)" + match = re.search(pattern, content, flags=re.MULTILINE | re.DOTALL | re.IGNORECASE) + if not match: + return [] + + patterns: list[str] = [] + for line in match.group(1).splitlines(): + line = line.strip() + if not line or line.startswith("#"): + continue + line = re.sub(r"^[-*]\s+", "", line).strip() + if line: + patterns.append(line) + return patterns + + def load_full_config(cwd: str | None = None) -> dict: """Load all config sections from mem0.md. - Returns a dict with keys: retention, search, categories, identity. + Returns a dict with keys: retention, search, categories, identity, + ignore, project_id. Each is populated only if the corresponding ``##`` section exists. """ if cwd is None: @@ -164,10 +187,17 @@ def load_full_config(cwd: str | None = None) -> dict: categories = parse_section_list(content, "Categories") if categories: config["categories"] = categories + config["default_categories"] = categories identity = parse_section_kv(content, "Identity") if identity: config["identity"] = identity + if "project_id" in identity: + config["project_id"] = identity["project_id"] + + ignore = parse_ignore_patterns(content) + if ignore: + config["ignore"] = ignore return config diff --git a/mem0-plugin/skills/mem0-dream/SKILL.md b/mem0-plugin/skills/mem0-dream/SKILL.md index 3466a4568..9964b72ca 100644 --- a/mem0-plugin/skills/mem0-dream/SKILL.md +++ b/mem0-plugin/skills/mem0-dream/SKILL.md @@ -75,8 +75,8 @@ For each group, identify the following: Two memories are near-duplicates when they express the same fact or decision but phrased differently (e.g., "Use PostgreSQL for auth" and "Auth DB is PostgreSQL"). -Heuristics: -- Significant noun/keyword overlap in the memory text. +Heuristics — two memories are near-duplicates if **all** of these hold: +- Similarity threshold: estimated cosine similarity > 0.9 (use noun/keyword overlap as proxy — if >60% of significant nouns overlap, treat as >0.9 similarity). - Same `metadata.type`. - Neither memory is pinned (`metadata.pinned != true`). diff --git a/mem0-plugin/skills/mem0-health/SKILL.md b/mem0-plugin/skills/mem0-health/SKILL.md index 5901d271d..d23fd8c28 100644 --- a/mem0-plugin/skills/mem0-health/SKILL.md +++ b/mem0-plugin/skills/mem0-health/SKILL.md @@ -79,7 +79,9 @@ If any check fails, add a `## Troubleshooting` section with specific fix steps f ## Extended mode: Memory Quality Analysis -When invoked with `--deep` (e.g., `/mem0:health --deep`), run the standard 5 checks above **plus** a memory quality scan: +When invoked with `--deep` (e.g., `/mem0:health --deep`) or `--fix` (e.g., `/mem0:health --fix`), run the standard 5 checks above **plus** a memory quality scan. + +`--fix` implies `--deep` and automatically applies safe fixes after showing the analysis (see bottom of this section). ### Quality Check 1: Duplicates @@ -101,6 +103,15 @@ Stale candidates: [mem0:] — session_state, 142d old ``` +### Quality Check 2b: Low-confidence memories + +Flag memories where `metadata.confidence` < 0.5 (regardless of age). Report separately from stale: + +``` +Low-confidence memories: + [mem0:] — confidence=0.3, "" +``` + ### Quality Check 3: Contradictions Within each `metadata.type` group, flag pairs that assert opposing facts about the same topic. Use semantic judgment — look for negation patterns, conflicting tool/framework choices, or reversed decisions. @@ -131,3 +142,22 @@ Untagged/orphan memories: ``` If all counts are 0: `Memory quality: clean. No duplicates, stale entries, or contradictions found.` + +### Auto-fix mode (`--fix`) + +When `--fix` is passed, apply these safe fixes automatically after displaying the quality summary: + +1. **Orphans:** For each untagged memory, infer a `metadata.type` from content and call `update_memory` to set it. If inference is uncertain, skip. +2. **Stale `session_state`/`compact_summary` > 90d:** Delete them via `delete_memory`. These are ephemeral by design. +3. **Duplicates:** Do NOT auto-merge — print "Run `/mem0:dream` to merge duplicates" instead. +4. **Contradictions:** Do NOT auto-resolve — print "Run `/mem0:dream` to resolve contradictions" instead. +5. **Low-confidence < 0.3 AND > 30d old:** Delete them via `delete_memory`. + +Print a summary of actions taken: + +``` +## Auto-fix Results + Deleted: stale, low-confidence + Retagged: orphans + Skipped: duplicates (use /mem0:dream), contradictions (use /mem0:dream) +``` diff --git a/mem0-plugin/skills/mem0-mcp/SKILL.md b/mem0-plugin/skills/mem0-mcp/SKILL.md index 4f560081e..750e0275f 100644 --- a/mem0-plugin/skills/mem0-mcp/SKILL.md +++ b/mem0-plugin/skills/mem0-mcp/SKILL.md @@ -272,7 +272,23 @@ add_memory( ) ``` -**Filtering note:** The mem0 v2 filter API does not yet support `array-contains` predicates. You cannot filter by `metadata.files` at search time. To work around this, always embed the bare filenames (and important path segments) in the memory content text itself — the vector search will then surface them on a filename query. The `files` array in metadata is still written for future compatibility once array-contains filtering is available. +**Filtering by files:** Use the `contains` operator to filter by `metadata.files` at search time: + +```python +search_memories( + query="auth middleware", + filters={ + "AND": [ + {"user_id": ""}, + {"app_id": ""}, + {"metadata.files": {"contains": "src/middleware/auth.ts"}}, + ] + }, + limit=5, +) +``` + +Also embed bare filenames in the memory content text as a fallback — the vector search will surface them even if the structured filter misses. ### Access counter: track memory usage @@ -289,10 +305,11 @@ import datetime current_meta["access_count"] = current_meta.get("access_count", 0) + 1 current_meta["last_accessed"] = datetime.datetime.now(datetime.timezone.utc).isoformat() -# 3. Update with preserved content +# 3. Update with preserved content and bumped metadata update_memory( memory_id=, data=current_text, # preserve original text — required parameter + metadata=current_meta, # pass updated access_count and last_accessed ) ``` diff --git a/mem0-plugin/skills/mem0-pin/SKILL.md b/mem0-plugin/skills/mem0-pin/SKILL.md index 2cb111cd6..84e97aaea 100644 --- a/mem0-plugin/skills/mem0-pin/SKILL.md +++ b/mem0-plugin/skills/mem0-pin/SKILL.md @@ -38,10 +38,15 @@ This is required because `update_memory` replaces the full memory — a metadata Call `update_memory` with: - `memory_id=` - `data=` (preserve the existing content) +- `metadata=` merge `original_metadata` with `{"pinned": true}` -The platform will retain the existing metadata and content. Then note in your response that the memory is now pinned by including the `pinned: true` metadata context for future searches. +Example: +```python +updated_meta = {**original_metadata, "pinned": True} +update_memory(memory_id=, data=, metadata=updated_meta) +``` -**Important:** `update_memory` requires the `data` (text) parameter. Passing only metadata may error or wipe content. Always read first, then update with the full text. +**Important:** `update_memory` requires the `data` (text) parameter. Passing only metadata may error or wipe content. Always read first, then update with the full text and explicit metadata. ### Step 4: Confirm @@ -53,7 +58,11 @@ Pinned memories surface first when relevant to a search. ### Unpin -If the user says "unpin" or the memory is already pinned: -1. Call `get_memory` to read current content. -2. Call `update_memory` with `data=` to preserve content. +If the user says "unpin" or `/mem0:unpin`: +1. Call `get_memory` to read current content and metadata. +2. Set `metadata.pinned = false` explicitly: + ```python + updated_meta = {**original_metadata, "pinned": False} + update_memory(memory_id=, data=, metadata=updated_meta) + ``` 3. Print: `Unpinned: "..."` diff --git a/mem0-plugin/skills/mem0-stats/SKILL.md b/mem0-plugin/skills/mem0-stats/SKILL.md index a461838fc..d2d67c259 100644 --- a/mem0-plugin/skills/mem0-stats/SKILL.md +++ b/mem0-plugin/skills/mem0-stats/SKILL.md @@ -63,6 +63,18 @@ Print a compact dashboard with an ASCII histogram for category distribution: user_preference ██░░░░░░░░░░░░░░ 3 session_state █░░░░░░░░░░░░░░░ 2 + By age: + < 7 days ████████████████ 5 + 7–30 days ██████████░░░░░░ 12 + 30–90 days ████░░░░░░░░░░░░ 10 + > 90 days ██░░░░░░░░░░░░░░ 8 + + By access count: + Never accessed ████████████████ 18 + 1–5 accesses ████████░░░░░░░░ 10 + 6–20 accesses ████░░░░░░░░░░░░ 4 + 20+ accesses █░░░░░░░░░░░░░░░ 3 + Oldest memory: Newest memory: @@ -78,5 +90,7 @@ Print a compact dashboard with an ASCII histogram for category distribution: - Use `█` for filled and `░` for empty. Right-align the count number. - Sort categories by count descending. Omit categories with 0 memories. - If only 1-2 categories exist, still show the histogram — it provides visual context. +- **Age buckets:** Compute from `created_at`. Buckets: <7d, 7–30d, 30–90d, >90d. +- **Access count buckets:** Read `metadata.access_count` (default 0 if absent). Buckets: 0, 1–5, 6–20, 20+. Skip any section with zero data.