fix(plugin): preserve structure on PreCompact + capture compact summary

Two changes to the compaction flow:

PreCompact: agent stores its summary with infer=False
  on_pre_compact.sh now instructs the agent to call add_memory with
  infer=False. The agent has full conversation context and writes a
  structured summary; without infer=False the platform runs a second
  LLM extraction pass over that summary, losing structure and
  producing fragmented facts. With it, the structured text is
  preserved verbatim. Same guidance added to mem0-mcp/SKILL.md so
  the agent applies it to other already-extracted writes (decisions,
  anti-patterns, conventions).

SessionStart-compact: capture the platform-generated summary
  PreCompact fires BEFORE Claude Code generates its compact summary,
  so the summary is unreachable from that hook. SessionStart-compact
  fires AFTER, so a new background helper (capture_compact_summary.py)
  reads the transcript at that point, finds the isCompactSummary=true
  entry, and stores its content as a separate memory tagged
  metadata.type=compact_summary. The post-compaction bootstrap text
  now tells the agent to layer its recall queries across
  session_state (its own pre-compact summary), compact_summary (the
  platform's condensed view), and topical types like decision /
  anti_pattern.

Net effect: three complementary memory types from one session
boundary, none of which used to be filterable or even captured
correctly.
This commit is contained in:
Mgeeeek
2026-05-08 21:13:00 +05:30
parent ed15fc1386
commit 0fd7cb9c5a
4 changed files with 210 additions and 10 deletions
@@ -0,0 +1,167 @@
#!/usr/bin/env python3
"""Capture the post-compaction summary into mem0.
PreCompact hooks fire BEFORE the summary is generated, so they can't
store the actual compact-summary text. This script runs at
SessionStart with source=compact, reads the transcript, finds the
most recent entry flagged isCompactSummary=true, and stores it as a
memory tagged metadata.type=compact_summary.
Input: JSON on stdin with transcript_path, session_id, source
Output: stderr logs only (exit 0 always -- must not block)
Spawned in the background by on_session_start.sh; the user-facing
bootstrap text continues without waiting on the network.
"""
from __future__ import annotations
import json
import logging
import os
import sys
import urllib.error
import urllib.request
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from _identity import resolve_user_id
log = logging.getLogger("mem0-compact-summary")
log.setLevel(logging.DEBUG)
_handler = logging.StreamHandler(sys.stderr)
_handler.setFormatter(logging.Formatter("[mem0-compact-summary] %(message)s"))
log.addHandler(_handler)
if os.environ.get("MEM0_DEBUG"):
_log_dir = os.path.expanduser("~/.mem0")
try:
os.makedirs(_log_dir, exist_ok=True)
_file_handler = logging.FileHandler(os.path.join(_log_dir, "hooks.log"))
_file_handler.setFormatter(logging.Formatter("[mem0-compact-summary] %(asctime)s %(message)s"))
log.addHandler(_file_handler)
except OSError:
pass
API_URL = "https://api.mem0.ai"
MAX_TAIL_LINES = 2000
MAX_SUMMARY_CHARS = 50000
def tail_lines(filepath: str, n: int) -> list[str]:
try:
with open(filepath, "rb") as f:
f.seek(0, 2)
file_size = f.tell()
if file_size == 0:
return []
chunk_size = min(file_size, n * 4096)
f.seek(max(0, file_size - chunk_size))
data = f.read().decode("utf-8", errors="replace")
return data.splitlines()[-n:]
except OSError:
return []
def find_compact_summary(lines: list[str]) -> str:
"""Walk transcript backwards, return text content of the most recent
entry flagged isCompactSummary=true. Empty string if none found."""
for line in reversed(lines):
line = line.strip()
if not line:
continue
try:
entry = json.loads(line)
except json.JSONDecodeError:
continue
if not entry.get("isCompactSummary"):
continue
message = entry.get("message", {})
content = message.get("content", [])
if isinstance(content, str):
return content[:MAX_SUMMARY_CHARS]
if isinstance(content, list):
parts = []
for block in content:
if isinstance(block, str):
parts.append(block)
elif isinstance(block, dict) and block.get("type") == "text":
parts.append(block.get("text", ""))
return "\n".join(parts).strip()[:MAX_SUMMARY_CHARS]
return ""
def store_summary(api_key: str, summary: str, user_id: str, session_id: str) -> bool:
body = {
"messages": [{"role": "user", "content": summary}],
"user_id": user_id,
"metadata": {
"type": "compact_summary",
"source": "session-start-compact",
"session_id": session_id,
},
"infer": False,
}
data = json.dumps(body).encode("utf-8")
req = urllib.request.Request(
f"{API_URL}/v1/memories/",
data=data,
headers={
"Content-Type": "application/json",
"Authorization": f"Token {api_key}",
},
method="POST",
)
try:
with urllib.request.urlopen(req, timeout=15) as resp:
if resp.status in (200, 201):
log.info("Compact summary stored")
return True
log.warning("API returned status %d", resp.status)
return False
except urllib.error.URLError as e:
log.warning("API call failed: %s", e)
return False
def main():
api_key = os.environ.get("MEM0_API_KEY", "")
if not api_key:
log.debug("MEM0_API_KEY not set, skipping capture")
return
try:
hook_input = json.loads(sys.stdin.read())
except (json.JSONDecodeError, OSError):
log.debug("No valid JSON on stdin")
return
transcript_path = hook_input.get("transcript_path", "")
if not transcript_path:
log.debug("No transcript_path provided")
return
session_id = hook_input.get("session_id", "")
user_id = resolve_user_id()
lines = tail_lines(transcript_path, MAX_TAIL_LINES)
if not lines:
log.debug("Transcript empty or unreadable: %s", transcript_path)
return
summary = find_compact_summary(lines)
if not summary:
log.debug("No isCompactSummary entry found")
return
log.info("Capturing compact summary (%d chars)", len(summary))
store_summary(api_key, summary, user_id, session_id)
if __name__ == "__main__":
try:
main()
except Exception as e:
log.error("Unexpected error: %s", e)
sys.exit(0)
+16 -6
View File
@@ -5,9 +5,9 @@
# the full context before it gets compressed.
#
# Output: Text instructions injected into Claude's context.
# Claude still has the full conversation and can write an accurate summary.
# A companion Python script (on_pre_compact.py) also runs to capture
# transcript state directly via the Mem0 REST API as a safety net.
# Claude still has the full conversation and can write an accurate summary,
# which it stores via add_memory(infer=False) so the platform preserves
# the structure verbatim instead of running a second extraction pass.
set -euo pipefail
@@ -22,7 +22,9 @@ Context compaction is about to happen. You are about to lose most of your conver
### Step 1: Store session summary
Call `add_memory` with a thorough summary covering ALL of the following:
Call `add_memory` with `infer=False` and a thorough summary covering ALL of the following.
`infer=False` is critical here: you've already done the extraction work yourself using full context. Without it, the platform runs a second LLM pass that loses your structure and pulls fragmented facts. With it, your summary is preserved verbatim.
```
## Session Summary (Pre-Compaction)
@@ -48,11 +50,19 @@ Call `add_memory` with a thorough summary covering ALL of the following:
the post-compaction agent continue without asking redundant questions]
```
Include metadata: `{"type": "session_state", "source": "pre-compaction"}`
Tool call shape:
```
add_memory(
messages=[{"role":"user","content":"<the summary above>"}],
user_id="<the active user_id from the SessionStart bootstrap>",
metadata={"type":"session_state","source":"pre-compaction"},
infer=False,
)
```
### Step 2: Store any unstored learnings
If there are learnings from this session that you haven't stored yet, store them as separate memories:
If there are learnings from this session that you haven't stored yet, store them as separate memories with `infer=False` (same reasoning -- you've already extracted the fact, don't re-extract):
- Failed approaches -> metadata `{"type": "anti_pattern"}`
- Successful strategies -> metadata `{"type": "task_learning"}`
- Architecture decisions -> metadata `{"type": "decision"}`
+12 -4
View File
@@ -65,14 +65,22 @@ Continue where you left off.
EOF
elif [ "$SOURCE" = "compact" ]; then
# Capture the just-generated compact summary in the background.
# PreCompact fires too early to see this entry; SessionStart-compact
# is the first place isCompactSummary=true is in the transcript.
echo "$INPUT" | python3 "$SCRIPT_DIR/capture_compact_summary.py" 2>/dev/null &
cat <<'EOF'
## Mem0 Post-Compaction Recovery
Context was just compacted. You may have lost important session context.
Context was just compacted. The Claude Code-generated compact summary
is being captured to mem0 in the background as `metadata.type=compact_summary`.
1. Call `search_memories` with queries related to what you were working on to reload relevant knowledge.
2. Check for any session state memories that were saved before compaction.
3. Continue working based on the recovered context.
1. Call `search_memories` to reload context, layering up to three angles:
- `metadata.type=session_state` -- the rich pre-compaction summary you wrote
- `metadata.type=compact_summary` -- the platform-generated condensed summary just now
- `metadata.type=decision` / `anti_pattern` -- specific facts you stored during the session
2. Continue working from the recovered context.
EOF
fi