fix(mem0-plugin): remove noisy session-state dumps, match openclaw pattern

- Remove all background REST API transcript dumps (on_pre_compact.py,
  capture_compact_summary.py, on_pre_commit.py calls) that were creating
  noisy session_state and commit_context memories automatically
- Rewrite pre-compaction prompt to extract individual durable facts
  (decisions, learnings, anti-patterns) instead of session summary blobs
- Agent now decides what to store, matching openclaw's triage pattern:
  "Most sessions produce zero memory operations. That is correct."
- Fix auto_import.py to search git repo root for CLAUDE.md/AGENTS.md
  (was only searching plugin cwd, missing files in parent directory)
- Fix health check session tracker to check stats file directly instead
  of depending on CLAUDE_PLUGIN_ROOT env var
- Fix session-end report to print in SessionEnd hook (Stop hook output
  doesn't render on /exit)
- Update v2 API endpoint to v3 in git commit capture search
- Keep auto_import.py as only direct REST write path (project profiles,
  idempotent, hash-gated)
This commit is contained in:
kartik-mem0
2026-05-22 16:20:31 +05:30
parent c2290299cf
commit 271c7c23c5
10 changed files with 99 additions and 123 deletions
+28 -3
View File
@@ -46,6 +46,21 @@ TARGET_FILES = ["CLAUDE.md", "AGENTS.md", ".cursorrules", ".windsurfrules", "mem
HASH_STORE = os.path.expanduser("~/.mem0/file_hashes.json")
def _git_root(cwd: str) -> str:
"""Return the git repo root, or empty string if not in a repo."""
import subprocess
try:
result = subprocess.run(
["git", "rev-parse", "--show-toplevel"],
cwd=cwd, capture_output=True, text=True, timeout=5,
)
if result.returncode == 0:
return result.stdout.strip()
except (OSError, subprocess.TimeoutExpired):
pass
return ""
def sha256_file(path: str) -> str:
"""Return the hex SHA-256 digest of a file."""
h = hashlib.sha256()
@@ -133,15 +148,25 @@ def main() -> None:
project_id = resolve_project_id(cwd)
branch = resolve_branch(cwd)
log.debug("Auto-import started: cwd=%s project=%s user=%s branch=%s", cwd, project_id, user_id, branch)
git_root = _git_root(cwd)
search_dirs = [cwd]
if git_root and os.path.realpath(git_root) != os.path.realpath(cwd):
search_dirs.append(git_root)
log.debug("Auto-import started: cwd=%s git_root=%s project=%s user=%s branch=%s", cwd, git_root or "(none)", project_id, user_id, branch)
hashes = load_hashes()
updated = False
for filename in TARGET_FILES:
filepath = os.path.join(cwd, filename)
filepath = ""
for search_dir in search_dirs:
candidate = os.path.join(search_dir, filename)
if os.path.isfile(candidate):
filepath = candidate
break
if not os.path.isfile(filepath):
if not filepath:
log.debug("Not found, skipping: %s", filename)
continue
@@ -109,7 +109,7 @@ def store_summary(api_key: str, summary: str, user_id: str, session_id: str, pro
"user_id": user_id,
"app_id": project_id,
"metadata": metadata,
"infer": False,
"infer": True,
"expiration_date": expires,
}
@@ -28,18 +28,11 @@ esac
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
if [ ! -f "$SCRIPT_DIR/on_pre_commit.py" ]; then
exit 0
fi
API_KEY="${MEM0_API_KEY:-${CLAUDE_PLUGIN_OPTION_MEM0_API_KEY:-}}"
if [ -z "$API_KEY" ]; then
exit 0
fi
# Background: capture staged changes as memory
git diff --cached --stat 2>/dev/null | python3 "$SCRIPT_DIR/on_pre_commit.py" &
# Foreground: search for relevant memories about changed files
CHANGED_FILES=$(git diff --cached --name-only 2>/dev/null | head -10 | tr '\n' ', ' | sed 's/,$//')
if [ -z "$CHANGED_FILES" ]; then
+1 -2
View File
@@ -40,9 +40,8 @@ You lost most conversation history. Recover context NOW:
1. \`search_memories(query="session state current task", filters={"AND": [{"user_id": "$USER_ID"}, {"app_id": "$PROJECT_ID"}, {"metadata": {"type": "session_state"}}]})\`
2. \`search_memories(query="recent decisions and learnings", filters={"AND": [{"user_id": "$USER_ID"}, {"app_id": "$PROJECT_ID"}, {"metadata": {"type": "decision"}}]})\`
3. \`search_memories(query="compact summary", filters={"AND": [{"user_id": "$USER_ID"}, {"app_id": "$PROJECT_ID"}, {"metadata": {"type": "compact_summary"}}]})\`
Run all 3 in parallel. Use results to resume work without asking user to repeat context.
Run both in parallel. Use results to resume work without asking user to repeat context.
EOF
exit 0
+28 -30
View File
@@ -137,33 +137,21 @@ def parse_transcript(lines: list[str]) -> dict:
def build_content(state: dict, source: str) -> str:
"""Build structured markdown from parsed state."""
parts = [f"## Session State ({source})\n"]
"""Build minimal context — only what's needed to resume work.
This is a FALLBACK safety net, not the primary capture path.
The agent handles rich memory storage via on_pre_compact.sh prompts.
This script only fires when the agent didn't store enough on its own.
Keep it short — mem0 infer=True will extract structured facts.
"""
parts = []
if state["user_messages"]:
parts.append("### What the user was working on")
for msg in state["user_messages"]:
truncated = msg[:5000] + "..." if len(msg) > 5000 else msg
parts.append(f"- {truncated}")
parts.append("")
parts.append(f"Working on: {state['user_messages'][-1][:300]}")
if state["files_modified"]:
parts.append("### Files modified this session")
for fp in state["files_modified"]:
parts.append(f"- `{fp}`")
parts.append("")
if state["bash_commands"]:
parts.append("### Recent commands")
for cmd in state["bash_commands"]:
truncated = cmd[:1000] + "..." if len(cmd) > 1000 else cmd
parts.append(f"- `{truncated}`")
parts.append("")
if state["last_assistant_text"]:
parts.append("### Last context")
parts.append(state["last_assistant_text"])
parts.append("")
parts.append(f"Files touched: {', '.join(state['files_modified'][:15])}")
return "\n".join(parts)
@@ -186,7 +174,7 @@ def store_memory(api_key: str, content: str, user_id: str, source: str, session_
"app_id": project_id,
"metadata": metadata,
"expiration_date": expires,
"infer": False,
"infer": True,
}
data = json.dumps(body).encode("utf-8")
@@ -240,6 +228,18 @@ def main():
project_id = resolve_project_id(cwd)
branch = resolve_branch(cwd)
# Skip if agent already stored memories this session — avoid duplicate writes.
# This script is a fallback, not the primary capture path.
stats_file = f"/tmp/mem0_session_stats_{os.environ.get('USER', 'default')}.json"
try:
with open(stats_file) as f:
stats = json.load(f)
if stats.get("adds", 0) >= 2:
log.info("Agent stored %d memories this session — skipping fallback capture", stats["adds"])
return
except (OSError, json.JSONDecodeError):
pass # no stats = agent didn't store anything, proceed with fallback
lines = tail_lines(transcript_path, MAX_TAIL_LINES)
if not lines:
log.debug("Transcript empty or unreadable: %s", transcript_path)
@@ -251,13 +251,11 @@ def main():
return
content = build_content(state, source)
if not content.strip():
log.debug("No content to store")
return
log.info(
"Capturing session state: %d user msgs, %d files, %d commands",
len(state["user_messages"]),
len(state["files_modified"]),
len(state["bash_commands"]),
)
log.info("Fallback capture: %d files modified", len(state["files_modified"]))
store_memory(api_key, content, user_id, source, session_id, project_id, branch)
+23 -45
View File
@@ -18,69 +18,47 @@ fi
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
INPUT=$(cat)
# Fire REST API transcript backup in background (safety net if agent
# can't complete the add_memory call before compaction finishes)
echo "$INPUT" | python3 "$SCRIPT_DIR/on_pre_compact.py" --source=pre-compaction 2>/dev/null &
python3 "$SCRIPT_DIR/telemetry.py" pre_compact 2>/dev/null &
cat <<'EOF'
## CRITICAL: Pre-Compaction Session Summary
## Pre-Compaction: Extract and store durable facts
Context compaction is about to happen. You are about to lose most of your conversation history. You MUST store a comprehensive session summary NOW using the mem0 `add_memory` tool.
Context compaction is about to happen. Review the conversation and store only facts that would help a future agent with ZERO context.
### Step 1: Store session summary
### What to store
Call `add_memory` with `infer=False` and a thorough summary covering ALL of the following.
For each fact, ask: "Would a new agent — with no prior context — benefit from knowing this?" If no, skip it. Most sessions produce 0-3 facts worth storing.
`infer=False` is critical here: you've already done the extraction work yourself using full context. Without it, the platform runs a second LLM pass that loses your structure and pulls fragmented facts. With it, your summary is preserved verbatim.
Store each fact as a SEPARATE `add_memory` call. One fact per call. 15-50 words each. Third person. Include file paths when relevant.
```
## Session Summary (Pre-Compaction)
Categories and when to use them:
- `decision` — architectural choices, trade-offs made ("Chose PostgreSQL over MongoDB for auth because of ACID requirements")
- `task_learning` — patterns that worked ("Running migrations before seed in this repo avoids FK violations")
- `anti_pattern` — approaches that failed ("Don't use batch insert for users table — triggers deadlock with audit log")
- `convention` — coding standards discovered ("This repo uses snake_case for all Python files, camelCase for TS")
- `user_preference` — how the user likes to work ("User prefers short PRs, one feature per branch")
### User's Goal
[What the user originally asked for and their intent]
### What NOT to store
### What Was Accomplished
[Numbered list of tasks completed, features built, bugs fixed]
- Session summaries or "what we did today" blobs
- Raw file lists or command histories
- Anything already stored in a prior `add_memory` this session
- One-time information that won't recur
- Transient state ("currently debugging X")
### Key Decisions Made
[Architectural choices, design decisions, trade-offs discussed]
### How to store
### Files Created or Modified
[List of important file paths with what changed in each]
### Current State
[What is in progress RIGHT NOW — the task you were in the middle of]
[Any pending items, blockers, or next steps]
### Important Context
[User preferences observed, coding patterns, anything that would help
the post-compaction agent continue without asking redundant questions]
```
Tool call shape:
```
add_memory(
messages=[{"role":"user","content":"<the summary above>"}],
user_id="<the active user_id from the SessionStart bootstrap>",
app_id="<the active project_id / app_id from SessionStart>",
metadata={"type":"session_state","source":"pre-compaction","branch":"<active branch>"},
messages=[{"role":"user","content":"<one fact, 15-50 words>"}],
user_id="<active user_id>",
app_id="<active project_id>",
metadata={"type":"<category>","branch":"<active branch>","confidence":0.8},
infer=False,
)
```
### Step 2: Store any unstored learnings
If there are learnings from this session that you haven't stored yet, store them as separate memories with `infer=False` (same reasoning -- you've already extracted the fact, don't re-extract):
- Failed approaches -> metadata `{"type": "anti_pattern"}`
- Successful strategies -> metadata `{"type": "task_learning"}`
- Architecture decisions -> metadata `{"type": "decision"}`
### Step 3: Acknowledge
After storing, briefly tell the user that session state has been saved and you're ready for compaction.
Do this NOW. Do not skip any section. The quality of this summary directly determines whether you can continue the user's task after compaction.
If nothing durable happened this session, store nothing. That is correct.
EOF
exit 0
+15 -17
View File
@@ -20,25 +20,23 @@ INPUT=$(cat)
REASON=$(echo "$INPUT" | jq -r '.reason // "other"' 2>/dev/null || echo "other")
SESSION_ID=$(echo "$INPUT" | jq -r '.session_id // ""' 2>/dev/null || echo "")
# Print session-end report (last chance — Stop hook output may not render on /exit)
REPORT=$(python3 "$SCRIPT_DIR/session_stats.py" report 2>/dev/null || echo "")
if [ -n "$REPORT" ] && [ "$REPORT" != "Session: no memory operations." ]; then
echo ""
echo "---"
echo "mem0 $REPORT"
echo "---"
# Append to persistent session log
mkdir -p "$HOME/.mem0" 2>/dev/null || true
echo "$(date -u +%Y-%m-%dT%H:%M:%SZ) | $REPORT" >> "$HOME/.mem0/session-log.md" 2>/dev/null || true
fi
# Telemetry (fire-and-forget — session dying, best-effort)
python3 "$SCRIPT_DIR/telemetry.py" session_end --reason="$REASON" 2>/dev/null &
# Check if on_stop.sh already captured this session (avoid duplicates)
CAPTURE_MARKER="$HOME/.mem0/.captured_${SESSION_ID}"
if [ -n "$SESSION_ID" ] && [ -f "$CAPTURE_MARKER" ]; then
exit 0
fi
# Last-chance transcript capture (synchronous — session is ending anyway)
if [ -n "${MEM0_API_KEY:-}" ]; then
echo "$INPUT" | python3 "$SCRIPT_DIR/on_pre_compact.py" --source=session-end 2>/dev/null || true
# Mark as captured to prevent duplicate if Stop also ran
if [ -n "$SESSION_ID" ]; then
mkdir -p "$HOME/.mem0" 2>/dev/null || true
touch "$CAPTURE_MARKER" 2>/dev/null || true
# Clean up old markers (> 7 days)
find "$HOME/.mem0" -name ".captured_*" -mtime +7 -delete 2>/dev/null || true
fi
fi
# Clean up old capture markers (> 7 days)
find "$HOME/.mem0" -name ".captured_*" -mtime +7 -delete 2>/dev/null || true
exit 0
+2 -9
View File
@@ -152,20 +152,13 @@ Continue where you left off.
EOF
elif [ "$SOURCE" = "compact" ]; then
# Capture the just-generated compact summary in the background.
# PreCompact fires too early to see this entry; SessionStart-compact
# is the first place isCompactSummary=true is in the transcript.
echo "$INPUT" | python3 "$SCRIPT_DIR/capture_compact_summary.py" 2>/dev/null &
cat <<'EOF'
## Mem0 Post-Compaction Recovery
Context was just compacted. The Claude Code-generated compact summary
is being captured to mem0 in the background as `metadata.type=compact_summary`.
Context was just compacted. Reload your context from mem0.
1. Call `search_memories` to reload context, layering up to three angles:
- `metadata.type=session_state` -- the rich pre-compaction summary you wrote
- `metadata.type=compact_summary` -- the platform-generated condensed summary just now
- `metadata.type=session_state` -- the pre-compaction summary you wrote before compaction
- `metadata.type=decision` / `anti_pattern` -- specific facts you stored during the session
2. Continue working from the recovered context.
EOF
-8
View File
@@ -62,14 +62,6 @@ If nothing notable happened in this interaction, it's fine to skip. Only store g
Always include `app_id` (the active project_id from SessionStart) as a top-level parameter in every `add_memory` call.
EOF
# Capture transcript state in the background via Mem0 REST API
SESSION_ID=$(echo "$INPUT" | jq -r '.session_id // ""' 2>/dev/null || echo "")
echo "$INPUT" | python3 "$SCRIPT_DIR/on_pre_compact.py" --source=session-end 2>/dev/null &
# Mark session as captured so SessionEnd hook can skip duplicate capture
if [ -n "$SESSION_ID" ]; then
mkdir -p "$HOME/.mem0" 2>/dev/null || true
touch "$HOME/.mem0/.captured_${SESSION_ID}" 2>/dev/null || true
fi
exit 0
+1 -1
View File
@@ -135,7 +135,7 @@ def test_capture_compact_summary_store_uses_app_id():
assert "project_id" not in captured.get("metadata", {})
assert captured["metadata"]["type"] == "compact_summary"
assert captured["metadata"]["branch"] == "main"
assert captured["infer"] is False
assert captured["infer"] is True
assert "expiration_date" in captured