feat(mem0-plugin): add export/import skills + competing tool importers

Tier 3 implementation:
- #10: Unified project_id resolver already complete (shared scripts/)
- #11: /mem0:export dumps all project memories as YAML-frontmatter
  markdown; /mem0:import restores from same format. Round-trip capable.
  parse_export_file.py handles parsing (stdlib-only).
- #12: import_competing_tools.py imports from .cursorrules, copilot
  instructions, cline memory-bank, continue.dev rules. Splits by
  section headers, POSTs with app_id + infer=False.
This commit is contained in:
kartik-mem0
2026-05-21 17:43:30 +05:30
parent b79be63e30
commit 63641f5573
7 changed files with 1418 additions and 0 deletions
@@ -0,0 +1,327 @@
#!/usr/bin/env python3
"""Import memories from competing AI tool configuration files into mem0.
Sub-commands (via sys.argv[1]):
cursorrules [--path .cursorrules]
copilot [--path .github/copilot-instructions.md]
cline [--path memory-bank/]
continue [--path .continue/rules.md]
Each sub-command reads configuration files from competing tools,
splits them into chunks, and POSTs each chunk to the mem0 API as a
project_profile memory.
Output: progress messages to stdout, errors to stderr
Exit: 0 always
"""
from __future__ import annotations
import json
import os
import sys
import urllib.error
import urllib.request
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from _identity import resolve_api_key, resolve_user_id
from _project import resolve_branch, resolve_project_id
API_URL = "https://api.mem0.ai"
MIN_CHUNK_CHARS = 50
MAX_CHUNK_CHARS = 10_000
# ---------------------------------------------------------------------------
# Content splitting utilities
# ---------------------------------------------------------------------------
def split_by_headers(content: str, header_prefix: str = "## ") -> list[str]:
"""Split content by Markdown header lines (e.g. '## ').
The header line is included at the start of each chunk.
Returns a list of non-empty chunk strings.
"""
chunks: list[str] = []
current_lines: list[str] = []
for line in content.splitlines(keepends=True):
if line.startswith(header_prefix) and current_lines:
chunk = "".join(current_lines).strip()
if chunk:
chunks.append(chunk)
current_lines = [line]
else:
current_lines.append(line)
if current_lines:
chunk = "".join(current_lines).strip()
if chunk:
chunks.append(chunk)
return chunks
def split_by_hr_or_headers(content: str) -> list[str]:
"""Split content by '---' horizontal rules or '## ' headers.
Used for .continue/rules.md which may use either convention.
"""
import re
# Split on lines that are exactly "---" or start with "## "
chunks: list[str] = []
current_lines: list[str] = []
for line in content.splitlines(keepends=True):
is_hr = re.match(r"^---\s*$", line)
is_h2 = line.startswith("## ")
if (is_hr or is_h2) and current_lines:
chunk = "".join(current_lines).strip()
if chunk:
chunks.append(chunk)
current_lines = [] if is_hr else [line]
else:
current_lines.append(line)
if current_lines:
chunk = "".join(current_lines).strip()
if chunk:
chunks.append(chunk)
return chunks
def filter_and_truncate(chunks: list[str]) -> list[str]:
"""Filter out chunks shorter than MIN_CHUNK_CHARS, truncate long chunks."""
result: list[str] = []
for chunk in chunks:
if len(chunk) < MIN_CHUNK_CHARS:
continue
if len(chunk) > MAX_CHUNK_CHARS:
chunk = chunk[:MAX_CHUNK_CHARS]
result.append(chunk)
return result
# ---------------------------------------------------------------------------
# API helpers
# ---------------------------------------------------------------------------
def post_memory(api_key: str, content: str, user_id: str, project_id: str, branch: str, source: str) -> bool:
"""POST a single memory chunk to the mem0 API."""
metadata: dict = {
"type": "project_profile",
"source": source,
}
if branch:
metadata["branch"] = branch
body = {
"messages": [{"role": "user", "content": content}],
"user_id": user_id,
"app_id": project_id,
"metadata": metadata,
"infer": False,
}
data = json.dumps(body).encode("utf-8")
req = urllib.request.Request(
f"{API_URL}/v1/memories/",
data=data,
headers={
"Content-Type": "application/json",
"Authorization": f"Token {api_key}",
},
method="POST",
)
try:
with urllib.request.urlopen(req, timeout=20) as resp:
return resp.status in (200, 201)
except urllib.error.URLError as e:
print(f" [warn] API call failed: {e}", file=sys.stderr)
return False
def import_chunks(chunks: list[str], api_key: str, user_id: str, project_id: str, branch: str, source: str) -> int:
"""Import a list of content chunks; return number of successful imports."""
success = 0
for chunk in chunks:
if post_memory(api_key, chunk, user_id, project_id, branch, source):
success += 1
return success
# ---------------------------------------------------------------------------
# Sub-command implementations
# ---------------------------------------------------------------------------
def _parse_path_arg(args: list[str], flag: str, default: str) -> str:
"""Extract --path <value> from args list, falling back to default."""
for i, arg in enumerate(args):
if arg == flag and i + 1 < len(args):
return args[i + 1]
if arg.startswith(f"{flag}="):
return arg[len(flag) + 1:]
return default
def cmd_cursorrules(args: list[str]) -> None:
path = _parse_path_arg(args, "--path", ".cursorrules")
source = "cursor-import"
api_key = resolve_api_key()
user_id = resolve_user_id()
project_id = resolve_project_id()
branch = resolve_branch()
if not api_key:
print("Error: MEM0_API_KEY not set", file=sys.stderr)
return
if not os.path.isfile(path):
print(f"File not found: {path}", file=sys.stderr)
return
with open(path, encoding="utf-8", errors="replace") as f:
content = f.read()
raw_chunks = split_by_headers(content, "## ")
# Fall back to treating the whole file as one chunk if no headers found
if not raw_chunks:
raw_chunks = [content.strip()] if content.strip() else []
chunks = filter_and_truncate(raw_chunks)
n = import_chunks(chunks, api_key, user_id, project_id, branch, source)
print(f"Imported {n} memories from {source} ({path})")
def cmd_copilot(args: list[str]) -> None:
path = _parse_path_arg(args, "--path", ".github/copilot-instructions.md")
source = "copilot-import"
api_key = resolve_api_key()
user_id = resolve_user_id()
project_id = resolve_project_id()
branch = resolve_branch()
if not api_key:
print("Error: MEM0_API_KEY not set", file=sys.stderr)
return
if not os.path.isfile(path):
print(f"File not found: {path}", file=sys.stderr)
return
with open(path, encoding="utf-8", errors="replace") as f:
content = f.read()
raw_chunks = split_by_headers(content, "## ")
if not raw_chunks:
raw_chunks = [content.strip()] if content.strip() else []
chunks = filter_and_truncate(raw_chunks)
n = import_chunks(chunks, api_key, user_id, project_id, branch, source)
print(f"Imported {n} memories from {source} ({path})")
def cmd_cline(args: list[str]) -> None:
dir_path = _parse_path_arg(args, "--path", "memory-bank/")
source = "cline-import"
api_key = resolve_api_key()
user_id = resolve_user_id()
project_id = resolve_project_id()
branch = resolve_branch()
if not api_key:
print("Error: MEM0_API_KEY not set", file=sys.stderr)
return
if not os.path.isdir(dir_path):
print(f"Directory not found: {dir_path}", file=sys.stderr)
return
md_files = sorted(
f for f in os.listdir(dir_path) if f.endswith(".md")
)
if not md_files:
print(f"No .md files found in {dir_path}", file=sys.stderr)
return
total = 0
for filename in md_files:
filepath = os.path.join(dir_path, filename)
with open(filepath, encoding="utf-8", errors="replace") as f:
content = f.read().strip()
if not content:
continue
chunks = filter_and_truncate([content])
n = import_chunks(chunks, api_key, user_id, project_id, branch, source)
total += n
print(f"Imported {total} memories from {source} ({dir_path})")
def cmd_continue(args: list[str]) -> None:
path = _parse_path_arg(args, "--path", ".continue/rules.md")
source = "continue-import"
api_key = resolve_api_key()
user_id = resolve_user_id()
project_id = resolve_project_id()
branch = resolve_branch()
if not api_key:
print("Error: MEM0_API_KEY not set", file=sys.stderr)
return
if not os.path.isfile(path):
print(f"File not found: {path}", file=sys.stderr)
return
with open(path, encoding="utf-8", errors="replace") as f:
content = f.read()
raw_chunks = split_by_hr_or_headers(content)
if not raw_chunks:
raw_chunks = [content.strip()] if content.strip() else []
chunks = filter_and_truncate(raw_chunks)
n = import_chunks(chunks, api_key, user_id, project_id, branch, source)
print(f"Imported {n} memories from {source} ({path})")
# ---------------------------------------------------------------------------
# Entry point
# ---------------------------------------------------------------------------
COMMANDS = {
"cursorrules": cmd_cursorrules,
"copilot": cmd_copilot,
"cline": cmd_cline,
"continue": cmd_continue,
}
def main() -> None:
if len(sys.argv) < 2 or sys.argv[1] not in COMMANDS:
available = ", ".join(COMMANDS.keys())
print("Usage: import_competing_tools.py <subcommand> [--path <path>]", file=sys.stderr)
print(f"Subcommands: {available}", file=sys.stderr)
sys.exit(0)
subcommand = sys.argv[1]
remaining_args = sys.argv[2:]
COMMANDS[subcommand](remaining_args)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Unexpected error: {e}", file=sys.stderr)
sys.exit(0)
+147
View File
@@ -0,0 +1,147 @@
#!/usr/bin/env python3
"""Parse a mem0 export file and output JSON.
Input: path to a mem0-export-*.md file (sys.argv[1])
Output: JSON array of memory records to stdout
Exit: 0 always
Each block in the file is delimited by lines containing exactly "---".
Blocks have a YAML-like frontmatter section (key: value lines) followed
by a blank line and the memory content text.
Example block format:
---
id: abc123
created_at: 2024-01-01T00:00:00Z
type: task_learnings
confidence: 0.9
branch: main
files: src/foo.py, src/bar.py
categories: coding_conventions, task_learnings
---
The actual memory content text goes here.
"""
from __future__ import annotations
import json
import re
import sys
def parse_blocks(content: str) -> list[dict]:
"""Split content on '---' boundaries and parse each block.
Returns a list of dicts with keys:
id, type, confidence, branch, files (list), categories (list), content (str)
Blocks with empty content are skipped.
Missing optional fields default to "" (scalar) or [] (list fields).
"""
# Normalise line endings
content = content.replace("\r\n", "\n").replace("\r", "\n")
# Split on lines that are exactly "---"
raw_blocks = re.split(r"(?m)^---\s*$", content)
# After splitting on "---", the structure for each memory is:
# raw_blocks[0] = preamble (before first ---, typically empty)
# raw_blocks[1] = frontmatter for block 1
# raw_blocks[2] = content for block 1
# raw_blocks[3] = frontmatter for block 2
# raw_blocks[4] = content for block 2
# ...
# So frontmatter blocks are at odd indices (1, 3, 5, ...) and
# content blocks at even indices (2, 4, 6, ...).
results: list[dict] = []
# Pair up frontmatter + content starting at index 1
i = 1
while i < len(raw_blocks):
frontmatter_raw = raw_blocks[i]
content_raw = raw_blocks[i + 1] if i + 1 < len(raw_blocks) else ""
# Parse the frontmatter key-value pairs
fm = _parse_frontmatter(frontmatter_raw)
# Strip leading/trailing whitespace from content
memory_content = content_raw.strip()
# Skip blocks with empty content
if not memory_content:
i += 2
continue
record = {
"id": fm.get("id", ""),
"type": fm.get("type", ""),
"confidence": fm.get("confidence", ""),
"branch": fm.get("branch", ""),
"files": _parse_list_field(fm.get("files", "")),
"categories": _parse_list_field(fm.get("categories", "")),
"content": memory_content,
}
# Include created_at if present
if "created_at" in fm:
record["created_at"] = fm["created_at"]
results.append(record)
i += 2
return results
def _parse_frontmatter(text: str) -> dict[str, str]:
"""Parse simple 'key: value' lines from frontmatter text.
Only the first colon is used as the delimiter — values may contain colons.
Lines not matching 'key: value' are ignored.
"""
result: dict[str, str] = {}
for line in text.splitlines():
line = line.strip()
if not line:
continue
match = re.match(r"^([A-Za-z_][A-Za-z0-9_]*)\s*:\s*(.*)$", line)
if match:
key = match.group(1).strip()
value = match.group(2).strip()
result[key] = value
return result
def _parse_list_field(value: str) -> list[str]:
"""Split a comma-separated value into a list, stripping whitespace.
Returns [] for empty/whitespace-only input.
"""
if not value or not value.strip():
return []
return [item.strip() for item in value.split(",") if item.strip()]
def main() -> None:
if len(sys.argv) < 2:
print("Usage: parse_export_file.py <path-to-export-file>", file=sys.stderr)
print("[]")
sys.exit(0)
filepath = sys.argv[1]
try:
with open(filepath, encoding="utf-8", errors="replace") as f:
content = f.read()
except OSError as e:
print(f"Error reading file: {e}", file=sys.stderr)
print("[]")
sys.exit(0)
records = parse_blocks(content)
print(json.dumps(records, ensure_ascii=False, indent=2))
sys.exit(0)
if __name__ == "__main__":
main()
+81
View File
@@ -0,0 +1,81 @@
---
name: mem0-export
description: >
Export all memories for the current project to a local Markdown file.
Each memory is written as a YAML-frontmatter block that can be re-imported later.
TRIGGER: user runs /mem0:export, or asks "export memories", "backup memories",
"download my memories", "save memories to file".
---
# Mem0 Export
Export all memories for the current project to a portable Markdown file.
## Execution
### Step 1: Resolve identity
Determine the active identity:
- `user_id` from `MEM0_USER_ID` env var, else `$USER`, else `"default"`
- `project_id` (used as `app_id`) from `MEM0_PROJECT_ID` env var, or via the project resolver
### Step 2: Fetch all memories
Call `get_memories` with:
- `user_id=<active_user_id>`
- `app_id=<active_project_id>`
- `page_size=200`
If the response is paginated (i.e. the result contains a `next` cursor or the count equals `page_size`), continue fetching pages until all memories are retrieved.
### Step 3: Format each memory as a YAML-frontmatter block
For each memory record, produce a block in this exact format:
```
---
id: <memory.id>
created_at: <memory.created_at>
type: <memory.metadata.type or "">
confidence: <memory.metadata.confidence or "">
branch: <memory.metadata.branch or "">
files: <memory.metadata.files joined with ", " or "">
categories: <memory.categories joined with ", " or "">
---
<memory.memory or memory content string>
```
Notes:
- The `---` delimiters must be on their own lines with no extra whitespace.
- `files` and `categories` are written as comma-separated values on a single line.
- Leave a blank line after the content before the next `---` (for readability).
- If a field is missing or null, write an empty string (not "null").
### Step 4: Write the export file
Determine the output filename:
```
mem0-export-<project_id>-<YYYY-MM-DD>.md
```
Where `<YYYY-MM-DD>` is today's date in UTC.
Write all formatted blocks to this file using the Write tool (or equivalent). The file is written to the current working directory.
### Step 5: Print summary
```
Exported <N> memories to <filename>
```
Where `<N>` is the total number of memory blocks written.
## Error Handling
- If `get_memories` returns an error or zero memories, print:
```
No memories found for project <project_id>. Nothing exported.
```
- If the write fails, report the error to the user.
@@ -0,0 +1,106 @@
---
name: mem0-import-tools
description: >
Import memories from competing AI tool configuration files into mem0.
Supports Cursor (.cursorrules), GitHub Copilot (.github/copilot-instructions.md),
Cline (memory-bank/), and Continue (.continue/rules.md).
TRIGGER: user runs /mem0:import-tools, or asks "import from cursor",
"import cursorrules", "import from cline", "import from copilot",
"import from continue", "migrate from cursor", "migrate memories".
---
# Mem0 Import from Competing Tools
Import configuration and memory files from other AI coding tools into mem0.
## Supported Tools
| Tool | Default file/directory |
|------|----------------------|
| Cursor | `.cursorrules` |
| GitHub Copilot | `.github/copilot-instructions.md` |
| Cline | `memory-bank/` (directory of `.md` files) |
| Continue | `.continue/rules.md` |
## Execution
### Step 1: Detect which tool files exist
Check for the presence of each tool's file/directory in the current working directory:
```bash
# Check each location
test -f .cursorrules && echo "cursor: .cursorrules"
test -f .github/copilot-instructions.md && echo "copilot: .github/copilot-instructions.md"
test -d memory-bank/ && echo "cline: memory-bank/"
test -f .continue/rules.md && echo "continue: .continue/rules.md"
```
### Step 2: Report findings and ask user
List all found files to the user. For example:
```
Found the following tool configuration files:
[1] Cursor rules: .cursorrules
[2] Cline memory bank: memory-bank/
Which would you like to import? (enter numbers, comma-separated, or "all"):
```
If no files are found, print:
```
No competing tool configuration files found in the current directory.
Checked: .cursorrules, .github/copilot-instructions.md, memory-bank/, .continue/rules.md
```
and stop.
### Step 3: Run the import script for each selected tool
Determine the plugin root. Use the appropriate variable for the current platform:
- Claude Code: `${CLAUDE_PLUGIN_ROOT}`
- Codex: `${CODEX_PLUGIN_ROOT}`
- Cursor: `${CURSOR_PLUGIN_ROOT}`
For each tool the user selected, run the corresponding sub-command:
**Cursor (.cursorrules):**
```bash
python3 "<PLUGIN_ROOT>/scripts/import_competing_tools.py" cursorrules --path .cursorrules
```
**GitHub Copilot:**
```bash
python3 "<PLUGIN_ROOT>/scripts/import_competing_tools.py" copilot --path .github/copilot-instructions.md
```
**Cline:**
```bash
python3 "<PLUGIN_ROOT>/scripts/import_competing_tools.py" cline --path memory-bank/
```
**Continue:**
```bash
python3 "<PLUGIN_ROOT>/scripts/import_competing_tools.py" continue --path .continue/rules.md
```
### Step 4: Report results
After each script runs, echo its output to the user. Then print a combined summary:
```
Import complete.
Cursor: <N> memories
Copilot: <N> memories
Total: <N> memories imported into project <project_id>
```
Adjust the summary to reflect only the tools that were actually imported.
## Notes
- Memories are imported with `infer=False` — no AI inference is applied, content is stored as-is.
- Each section or file becomes a separate memory tagged with `metadata.source=<tool>-import` and `metadata.type=project_profile`.
- Sections shorter than 50 characters are automatically skipped (too short to be useful).
- Content longer than 10,000 characters is automatically truncated per chunk.
- You can re-run this skill safely — duplicate content will be caught by mem0's deduplication.
+104
View File
@@ -0,0 +1,104 @@
---
name: mem0-import
description: >
Import memories from a mem0 export file back into the current project.
Reads a YAML-frontmatter Markdown file produced by /mem0:export and
adds each memory block to mem0.
TRIGGER: user runs /mem0:import, or asks "import memories", "restore memories",
"load memories from file", "reimport backup".
---
# Mem0 Import
Import memories from a mem0 export file into the current project.
## Execution
### Step 1: Determine the export file to import
If the user provided a filename as an argument to `/mem0:import <filename>`, use that file.
Otherwise, list `.md` files in the current directory whose names contain `mem0-export`:
```bash
ls -1 *.md 2>/dev/null | grep mem0-export || echo "No export files found"
```
If multiple files are found, ask the user which one to import. If none are found, print:
```
No mem0-export files found in the current directory.
Run /mem0:export first, or provide the filename: /mem0:import <path-to-file>
```
### Step 2: Parse the export file
Determine the plugin root. Use the appropriate variable for the current platform:
- Claude Code: `${CLAUDE_PLUGIN_ROOT}`
- Codex: `${CODEX_PLUGIN_ROOT}`
- Cursor: `${CURSOR_PLUGIN_ROOT}`
Run the parser script to extract memory records as JSON:
```bash
python3 "<PLUGIN_ROOT>/scripts/parse_export_file.py" "<path-to-export-file>"
```
This outputs a JSON array where each element has:
- `id` — original memory ID (for reference only; a new ID will be assigned on import)
- `type` — metadata type
- `confidence` — metadata confidence value
- `branch` — metadata branch
- `files` — list of associated files
- `categories` — list of categories
- `content` — the memory text
If the script fails or outputs `[]`, print:
```
Failed to parse <filename> or file contains no valid memory blocks.
```
and stop.
### Step 3: Resolve identity
Determine the active identity:
- `user_id` from `MEM0_USER_ID` env var, else `$USER`, else `"default"`
- `project_id` (used as `app_id`) from `MEM0_PROJECT_ID` env var, or via the project resolver
### Step 4: Import each memory
For each record in the parsed JSON array, call `add_memory` with:
- `messages=[{"role": "user", "content": "<record.content>"}]`
- `user_id=<active_user_id>`
- `app_id=<active_project_id>`
- `metadata={`
- `"type": "<record.type>"` (if non-empty)
- `"confidence": "<record.confidence>"` (if non-empty)
- `"branch": "<record.branch>"` (if non-empty)
- `"files": <record.files>` (the list, if non-empty)
- `"source": "import"`
- `}`
- `infer=False`
Notes:
- Do NOT pass the original `id` — the platform assigns a new ID.
- Skip records where `content` is empty (the parser already filters these, but be defensive).
- Continue importing even if individual records fail; track the count of successes.
### Step 5: Print results
```
Imported <N> memories into project <project_id>
```
Where `<N>` is the number of successfully imported memories.
If any failed:
```
Imported <N>/<total> memories into project <project_id> (<failed> failed)
```
## Error Handling
- If the parser script is not found at `<PLUGIN_ROOT>/scripts/parse_export_file.py`, print an error and stop.
- If `add_memory` calls fail consistently (e.g. auth error), report the issue and stop early.
@@ -0,0 +1,355 @@
"""Tests for import_competing_tools.py — competing tool file importers."""
from __future__ import annotations
import json
import os
import sys
from unittest import mock
SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts")
# ---------------------------------------------------------------------------
# split_sections tests (unit tests on the splitter functions)
# ---------------------------------------------------------------------------
def test_split_by_headers_cursorrules():
"""split_by_headers correctly splits .cursorrules content on ## headers."""
from import_competing_tools import split_by_headers
content = """\
# My Cursor Rules
Some preamble text that belongs to the first section.
## TypeScript Conventions
Always use strict mode. Prefer const over let.
Never use var.
## React Patterns
Use functional components with hooks.
Avoid class components.
## Testing
Write tests for all utility functions.
"""
chunks = split_by_headers(content, "## ")
assert len(chunks) == 4 # preamble + 3 sections
# First chunk is the preamble (before any ## header)
assert "preamble text" in chunks[0]
# Remaining chunks start with their header
assert chunks[1].startswith("## TypeScript Conventions")
assert "strict mode" in chunks[1]
assert chunks[2].startswith("## React Patterns")
assert "functional components" in chunks[2]
assert chunks[3].startswith("## Testing")
assert "utility functions" in chunks[3]
def test_split_by_headers_copilot():
"""split_by_headers correctly splits copilot-instructions.md on ## headers."""
from import_competing_tools import split_by_headers
content = """\
## Code Style
Use 2-space indentation. Always add trailing commas.
## Architecture
Follow clean architecture principles. Keep business logic in domain layer.
"""
chunks = split_by_headers(content, "## ")
assert len(chunks) == 2
assert chunks[0].startswith("## Code Style")
assert "2-space indentation" in chunks[0]
assert chunks[1].startswith("## Architecture")
assert "clean architecture" in chunks[1]
def test_split_by_headers_no_headers():
"""split_by_headers returns entire content as one chunk if no headers found."""
from import_competing_tools import split_by_headers
content = "This file has no headers at all. Just plain text."
chunks = split_by_headers(content, "## ")
assert len(chunks) == 1
assert "Just plain text" in chunks[0]
def test_split_cline_multiple_md_files(tmp_path):
"""cmd_cline processes multiple .md files from memory-bank/ directory."""
from import_competing_tools import filter_and_truncate
# Create a temporary memory-bank directory with .md files
mb_dir = tmp_path / "memory-bank"
mb_dir.mkdir()
(mb_dir / "architecture.md").write_text(
"# Architecture Decisions\n\nUse microservices architecture with event sourcing."
)
(mb_dir / "conventions.md").write_text(
"# Code Conventions\n\nAll functions must have type hints. Use black formatter."
)
(mb_dir / "empty.md").write_text("") # empty file should be skipped
# Read and verify we can split the files
md_files = sorted(f for f in os.listdir(str(mb_dir)) if f.endswith(".md"))
assert "architecture.md" in md_files
assert "conventions.md" in md_files
assert "empty.md" in md_files
non_empty = []
for filename in md_files:
filepath = os.path.join(str(mb_dir), filename)
with open(filepath) as f:
content = f.read().strip()
if content:
chunks = filter_and_truncate([content])
non_empty.extend(chunks)
assert len(non_empty) == 2
assert any("microservices" in c for c in non_empty)
assert any("type hints" in c for c in non_empty)
def test_split_by_hr_or_headers_continue():
"""split_by_hr_or_headers correctly splits .continue/rules.md."""
from import_competing_tools import split_by_hr_or_headers
content = """\
## First Section
Content of first section.
---
## Second Section
Content of second section.
---
Third section without a header (just after HR).
"""
chunks = split_by_hr_or_headers(content)
# Should split into meaningful chunks
assert len(chunks) >= 2
assert any("First Section" in c for c in chunks)
assert any("Second Section" in c for c in chunks)
def test_filter_and_truncate_skips_short():
"""filter_and_truncate skips chunks shorter than MIN_CHUNK_CHARS (50)."""
from import_competing_tools import filter_and_truncate
chunks = [
"Short", # < 50 chars, should be filtered
"A" * 49, # exactly 49 chars, should be filtered
"A" * 50, # exactly 50 chars, should be kept
"A long enough chunk that definitely passes the minimum length filter.",
]
result = filter_and_truncate(chunks)
assert len(result) == 2
assert all(len(c) >= 50 for c in result)
def test_filter_and_truncate_truncates_long():
"""filter_and_truncate truncates chunks over MAX_CHUNK_CHARS (10000)."""
from import_competing_tools import MAX_CHUNK_CHARS, filter_and_truncate
long_chunk = "X" * (MAX_CHUNK_CHARS + 500)
result = filter_and_truncate([long_chunk])
assert len(result) == 1
assert len(result[0]) == MAX_CHUNK_CHARS
# ---------------------------------------------------------------------------
# Mock API tests
# ---------------------------------------------------------------------------
def _make_mock_response(status: int = 201, body: dict | None = None) -> mock.MagicMock:
"""Create a mock HTTP response object."""
if body is None:
body = {"id": "new-mem-id", "memory": "test"}
resp = mock.MagicMock()
resp.status = status
resp.read.return_value = json.dumps(body).encode()
resp.__enter__ = lambda s: s
resp.__exit__ = mock.MagicMock(return_value=False)
return resp
def test_cursorrules_import_api_call(tmp_path):
"""cmd_cursorrules calls the API with correct app_id (top-level), infer=False, and correct source."""
from import_competing_tools import cmd_cursorrules
# Create a .cursorrules file with enough content
cursorrules = tmp_path / ".cursorrules"
cursorrules.write_text(
"## TypeScript Rules\n\nAlways use strict TypeScript. Never use 'any' type. "
"Prefer interfaces over type aliases for object shapes."
)
captured_requests: list[dict] = []
def mock_urlopen(req, timeout=None):
body = json.loads(req.data.decode())
captured_requests.append(body)
return _make_mock_response(201)
with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-testkey"), \
mock.patch("import_competing_tools.resolve_user_id", return_value="testuser"), \
mock.patch("import_competing_tools.resolve_project_id", return_value="my-project"), \
mock.patch("import_competing_tools.resolve_branch", return_value="main"), \
mock.patch("urllib.request.urlopen", side_effect=mock_urlopen):
original_cwd = os.getcwd()
os.chdir(str(tmp_path))
try:
cmd_cursorrules(["--path", str(cursorrules)])
finally:
os.chdir(original_cwd)
assert len(captured_requests) >= 1
req_body = captured_requests[0]
# app_id must be top-level (not inside metadata)
assert req_body["app_id"] == "my-project", f"Expected app_id at top level, got: {req_body}"
# infer must be False (not "false", but the boolean False)
assert req_body["infer"] is False, f"Expected infer=False, got: {req_body['infer']}"
# source must be cursor-import
assert req_body["metadata"]["source"] == "cursor-import", (
f"Expected source=cursor-import, got: {req_body['metadata'].get('source')}"
)
# user_id must be set
assert req_body["user_id"] == "testuser"
# messages must be a list with role/content
assert isinstance(req_body["messages"], list)
assert req_body["messages"][0]["role"] == "user"
assert len(req_body["messages"][0]["content"]) > 0
def test_copilot_import_api_call(tmp_path):
"""cmd_copilot calls the API with source=copilot-import."""
from import_competing_tools import cmd_copilot
copilot_dir = tmp_path / ".github"
copilot_dir.mkdir()
copilot_file = copilot_dir / "copilot-instructions.md"
copilot_file.write_text(
"## Code Style\n\nUse 2-space indentation. Always add trailing commas in multi-line structures. "
"Prefer const over let. Never use var in JavaScript code."
)
captured: list[dict] = []
def mock_urlopen(req, timeout=None):
captured.append(json.loads(req.data.decode()))
return _make_mock_response(201)
with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-key"), \
mock.patch("import_competing_tools.resolve_user_id", return_value="user1"), \
mock.patch("import_competing_tools.resolve_project_id", return_value="proj1"), \
mock.patch("import_competing_tools.resolve_branch", return_value="main"), \
mock.patch("urllib.request.urlopen", side_effect=mock_urlopen):
cmd_copilot(["--path", str(copilot_file)])
assert len(captured) >= 1
assert captured[0]["metadata"]["source"] == "copilot-import"
assert captured[0]["app_id"] == "proj1"
assert captured[0]["infer"] is False
def test_cline_import_multiple_files(tmp_path):
"""cmd_cline imports one memory per non-empty .md file."""
from import_competing_tools import cmd_cline
mb = tmp_path / "memory-bank"
mb.mkdir()
(mb / "arch.md").write_text(
"Architecture: microservices with event-sourcing. Each service owns its database. "
"Communication via message bus only. No direct service-to-service HTTP calls."
)
(mb / "style.md").write_text(
"Code style: PEP 8 for Python. Black formatter. isort for imports. "
"Line length 120. Type hints required on all public functions and methods."
)
(mb / "empty.md").write_text("")
captured: list[dict] = []
def mock_urlopen(req, timeout=None):
captured.append(json.loads(req.data.decode()))
return _make_mock_response(201)
with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-key"), \
mock.patch("import_competing_tools.resolve_user_id", return_value="user1"), \
mock.patch("import_competing_tools.resolve_project_id", return_value="proj1"), \
mock.patch("import_competing_tools.resolve_branch", return_value="main"), \
mock.patch("urllib.request.urlopen", side_effect=mock_urlopen):
cmd_cline(["--path", str(mb)])
# Should have exactly 2 imports (empty.md skipped)
assert len(captured) == 2
sources = {r["metadata"]["source"] for r in captured}
assert sources == {"cline-import"}
for r in captured:
assert r["infer"] is False
assert r["app_id"] == "proj1"
def test_no_api_key_does_not_call_api(tmp_path):
"""When no API key is set, no HTTP call is made."""
from import_competing_tools import cmd_cursorrules
cursorrules = tmp_path / ".cursorrules"
cursorrules.write_text("## Rules\n\n" + "x" * 100)
with mock.patch("import_competing_tools.resolve_api_key", return_value=""), \
mock.patch("urllib.request.urlopen") as mock_url:
cmd_cursorrules(["--path", str(cursorrules)])
mock_url.assert_not_called()
def test_missing_file_does_not_call_api(tmp_path):
"""When the source file doesn't exist, no HTTP call is made."""
from import_competing_tools import cmd_cursorrules
with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-key"), \
mock.patch("urllib.request.urlopen") as mock_url:
cmd_cursorrules(["--path", str(tmp_path / "nonexistent.cursorrules")])
mock_url.assert_not_called()
def test_main_unknown_subcommand_exits_zero():
"""Calling main() with an unknown subcommand exits 0."""
import subprocess
result = subprocess.run(
[sys.executable, os.path.join(SCRIPTS_DIR, "import_competing_tools.py"), "unknown"],
capture_output=True,
text=True,
env={**os.environ, "MEM0_API_KEY": ""},
)
assert result.returncode == 0
+298
View File
@@ -0,0 +1,298 @@
"""Tests for parse_export_file.py — mem0 export file parser."""
from __future__ import annotations
import json
import os
import subprocess
import sys
SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts")
# ---------------------------------------------------------------------------
# Direct function tests
# ---------------------------------------------------------------------------
def test_parse_blocks_single_valid_block():
"""parse_blocks returns one record for a single valid block."""
from parse_export_file import parse_blocks
content = """\
---
id: abc123
created_at: 2024-01-15T10:00:00Z
type: task_learnings
confidence: 0.85
branch: main
files: src/foo.py, src/bar.py
categories: coding_conventions, task_learnings
---
Always use context managers when opening files.
"""
records = parse_blocks(content)
assert len(records) == 1
r = records[0]
assert r["id"] == "abc123"
assert r["type"] == "task_learnings"
assert r["confidence"] == "0.85"
assert r["branch"] == "main"
assert r["files"] == ["src/foo.py", "src/bar.py"]
assert r["categories"] == ["coding_conventions", "task_learnings"]
assert "Always use context managers" in r["content"]
def test_parse_blocks_multiple_blocks():
"""parse_blocks returns the correct number of records for multiple blocks."""
from parse_export_file import parse_blocks
content = """\
---
id: mem001
type: architecture_decisions
confidence: 0.9
branch: main
files:
categories: architecture_decisions
---
Use hexagonal architecture for the core domain.
---
id: mem002
type: anti_patterns
confidence: 0.75
branch: feat/refactor
files: src/legacy.py
categories: anti_patterns
---
Avoid direct database calls from view layer.
---
id: mem003
type: coding_conventions
confidence: 0.8
branch: main
files: src/utils.py, src/helpers.py
categories:
---
Use snake_case for all Python identifiers.
"""
records = parse_blocks(content)
assert len(records) == 3
assert records[0]["id"] == "mem001"
assert records[0]["categories"] == ["architecture_decisions"]
assert "hexagonal architecture" in records[0]["content"]
assert records[1]["id"] == "mem002"
assert records[1]["files"] == ["src/legacy.py"]
assert "direct database calls" in records[1]["content"]
assert records[2]["id"] == "mem003"
assert records[2]["files"] == ["src/utils.py", "src/helpers.py"]
assert records[2]["categories"] == []
assert "snake_case" in records[2]["content"]
def test_parse_blocks_missing_optional_fields():
"""parse_blocks uses defaults when optional fields are absent."""
from parse_export_file import parse_blocks
# confidence, branch, files, categories all absent
content = """\
---
id: xyz789
type: task_learnings
---
Run tests before committing.
"""
records = parse_blocks(content)
assert len(records) == 1
r = records[0]
assert r["id"] == "xyz789"
assert r["confidence"] == "" # default empty string
assert r["branch"] == "" # default empty string
assert r["files"] == [] # default empty list
assert r["categories"] == [] # default empty list
assert "Run tests" in r["content"]
def test_parse_blocks_filters_empty_content():
"""parse_blocks skips blocks whose content is empty or whitespace-only."""
from parse_export_file import parse_blocks
content = """\
---
id: empty1
type: task_learnings
---
---
id: real1
type: task_learnings
---
This block has real content.
---
id: empty2
type: coding_conventions
---
"""
records = parse_blocks(content)
# Only the block with actual content should be returned
assert len(records) == 1
assert records[0]["id"] == "real1"
assert "real content" in records[0]["content"]
def test_parse_blocks_round_trip():
"""Content formatted by export matches what parse_blocks expects."""
from parse_export_file import parse_blocks
# Simulate the exact format produced by the export skill
memory_id = "test-id-001"
created_at = "2024-06-01T12:00:00Z"
mem_type = "architecture_decisions"
confidence = "0.92"
branch = "feat/new-feature"
files = ["src/main.py", "tests/test_main.py"]
categories = ["architecture_decisions", "coding_conventions"]
memory_content = "Use dependency injection for all service classes."
# Format exactly as the export skill would
block = (
"---\n"
f"id: {memory_id}\n"
f"created_at: {created_at}\n"
f"type: {mem_type}\n"
f"confidence: {confidence}\n"
f"branch: {branch}\n"
f"files: {', '.join(files)}\n"
f"categories: {', '.join(categories)}\n"
"---\n"
f"{memory_content}\n"
"\n"
)
records = parse_blocks(block)
assert len(records) == 1
r = records[0]
assert r["id"] == memory_id
assert r["type"] == mem_type
assert r["confidence"] == confidence
assert r["branch"] == branch
assert r["files"] == files
assert r["categories"] == categories
assert r["content"] == memory_content
def test_parse_blocks_multiline_content():
"""parse_blocks correctly captures multi-line memory content."""
from parse_export_file import parse_blocks
content = """\
---
id: multi001
type: task_learnings
---
Line one of the memory.
Line two of the memory.
Line four after blank line.
"""
records = parse_blocks(content)
assert len(records) == 1
assert "Line one" in records[0]["content"]
assert "Line two" in records[0]["content"]
assert "Line four" in records[0]["content"]
def test_parse_blocks_empty_input():
"""parse_blocks returns empty list for empty input."""
from parse_export_file import parse_blocks
assert parse_blocks("") == []
assert parse_blocks(" \n ") == []
def test_parse_blocks_no_blocks():
"""parse_blocks returns empty list for content without any --- delimiters."""
from parse_export_file import parse_blocks
assert parse_blocks("Just some text without any delimiters.") == []
def test_parse_blocks_value_with_colon():
"""parse_blocks handles values that themselves contain colons."""
from parse_export_file import parse_blocks
content = """\
---
id: colon-test
type: task_learnings
created_at: 2024-01-01T10:00:00Z
---
Timestamp values contain colons and should parse correctly.
"""
records = parse_blocks(content)
assert len(records) == 1
assert records[0]["id"] == "colon-test"
# created_at field should be captured (it's in the record if present)
assert "2024-01-01T10:00:00Z" in records[0].get("created_at", "")
# ---------------------------------------------------------------------------
# CLI / subprocess tests
# ---------------------------------------------------------------------------
def test_main_cli_outputs_json(tmp_path):
"""Running parse_export_file.py as a script outputs valid JSON."""
export_file = tmp_path / "mem0-export-test.md"
export_file.write_text("""\
---
id: cli-test-001
type: task_learnings
confidence: 0.8
branch: main
files:
categories: task_learnings
---
Prefer composition over inheritance.
""")
result = subprocess.run(
[sys.executable, os.path.join(SCRIPTS_DIR, "parse_export_file.py"), str(export_file)],
capture_output=True,
text=True,
)
assert result.returncode == 0
records = json.loads(result.stdout)
assert isinstance(records, list)
assert len(records) == 1
assert records[0]["id"] == "cli-test-001"
def test_main_cli_no_args_exits_zero():
"""Running parse_export_file.py with no arguments exits 0 and prints []."""
result = subprocess.run(
[sys.executable, os.path.join(SCRIPTS_DIR, "parse_export_file.py")],
capture_output=True,
text=True,
)
assert result.returncode == 0
assert result.stdout.strip() == "[]"
def test_main_cli_missing_file_exits_zero(tmp_path):
"""Running parse_export_file.py with a missing file exits 0 and prints []."""
result = subprocess.run(
[sys.executable, os.path.join(SCRIPTS_DIR, "parse_export_file.py"),
str(tmp_path / "nonexistent.md")],
capture_output=True,
text=True,
)
assert result.returncode == 0
assert result.stdout.strip() == "[]"