From 63641f5573f13039f062e776127da9bfb03f5dc7 Mon Sep 17 00:00:00 2001 From: kartik-mem0 Date: Thu, 21 May 2026 17:43:30 +0530 Subject: [PATCH] feat(mem0-plugin): add export/import skills + competing tool importers Tier 3 implementation: - #10: Unified project_id resolver already complete (shared scripts/) - #11: /mem0:export dumps all project memories as YAML-frontmatter markdown; /mem0:import restores from same format. Round-trip capable. parse_export_file.py handles parsing (stdlib-only). - #12: import_competing_tools.py imports from .cursorrules, copilot instructions, cline memory-bank, continue.dev rules. Splits by section headers, POSTs with app_id + infer=False. --- mem0-plugin/scripts/import_competing_tools.py | 327 ++++++++++++++++ mem0-plugin/scripts/parse_export_file.py | 147 ++++++++ mem0-plugin/skills/mem0-export/SKILL.md | 81 ++++ mem0-plugin/skills/mem0-import-tools/SKILL.md | 106 ++++++ mem0-plugin/skills/mem0-import/SKILL.md | 104 +++++ .../tests/test_import_competing_tools.py | 355 ++++++++++++++++++ mem0-plugin/tests/test_parse_export_file.py | 298 +++++++++++++++ 7 files changed, 1418 insertions(+) create mode 100644 mem0-plugin/scripts/import_competing_tools.py create mode 100644 mem0-plugin/scripts/parse_export_file.py create mode 100644 mem0-plugin/skills/mem0-export/SKILL.md create mode 100644 mem0-plugin/skills/mem0-import-tools/SKILL.md create mode 100644 mem0-plugin/skills/mem0-import/SKILL.md create mode 100644 mem0-plugin/tests/test_import_competing_tools.py create mode 100644 mem0-plugin/tests/test_parse_export_file.py diff --git a/mem0-plugin/scripts/import_competing_tools.py b/mem0-plugin/scripts/import_competing_tools.py new file mode 100644 index 000000000..89bdce4ae --- /dev/null +++ b/mem0-plugin/scripts/import_competing_tools.py @@ -0,0 +1,327 @@ +#!/usr/bin/env python3 +"""Import memories from competing AI tool configuration files into mem0. + +Sub-commands (via sys.argv[1]): + cursorrules [--path .cursorrules] + copilot [--path .github/copilot-instructions.md] + cline [--path memory-bank/] + continue [--path .continue/rules.md] + +Each sub-command reads configuration files from competing tools, +splits them into chunks, and POSTs each chunk to the mem0 API as a +project_profile memory. + +Output: progress messages to stdout, errors to stderr +Exit: 0 always +""" + +from __future__ import annotations + +import json +import os +import sys +import urllib.error +import urllib.request + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from _identity import resolve_api_key, resolve_user_id +from _project import resolve_branch, resolve_project_id + +API_URL = "https://api.mem0.ai" +MIN_CHUNK_CHARS = 50 +MAX_CHUNK_CHARS = 10_000 + + +# --------------------------------------------------------------------------- +# Content splitting utilities +# --------------------------------------------------------------------------- + + +def split_by_headers(content: str, header_prefix: str = "## ") -> list[str]: + """Split content by Markdown header lines (e.g. '## '). + + The header line is included at the start of each chunk. + Returns a list of non-empty chunk strings. + """ + chunks: list[str] = [] + current_lines: list[str] = [] + + for line in content.splitlines(keepends=True): + if line.startswith(header_prefix) and current_lines: + chunk = "".join(current_lines).strip() + if chunk: + chunks.append(chunk) + current_lines = [line] + else: + current_lines.append(line) + + if current_lines: + chunk = "".join(current_lines).strip() + if chunk: + chunks.append(chunk) + + return chunks + + +def split_by_hr_or_headers(content: str) -> list[str]: + """Split content by '---' horizontal rules or '## ' headers. + + Used for .continue/rules.md which may use either convention. + """ + import re + + # Split on lines that are exactly "---" or start with "## " + chunks: list[str] = [] + current_lines: list[str] = [] + + for line in content.splitlines(keepends=True): + is_hr = re.match(r"^---\s*$", line) + is_h2 = line.startswith("## ") + + if (is_hr or is_h2) and current_lines: + chunk = "".join(current_lines).strip() + if chunk: + chunks.append(chunk) + current_lines = [] if is_hr else [line] + else: + current_lines.append(line) + + if current_lines: + chunk = "".join(current_lines).strip() + if chunk: + chunks.append(chunk) + + return chunks + + +def filter_and_truncate(chunks: list[str]) -> list[str]: + """Filter out chunks shorter than MIN_CHUNK_CHARS, truncate long chunks.""" + result: list[str] = [] + for chunk in chunks: + if len(chunk) < MIN_CHUNK_CHARS: + continue + if len(chunk) > MAX_CHUNK_CHARS: + chunk = chunk[:MAX_CHUNK_CHARS] + result.append(chunk) + return result + + +# --------------------------------------------------------------------------- +# API helpers +# --------------------------------------------------------------------------- + + +def post_memory(api_key: str, content: str, user_id: str, project_id: str, branch: str, source: str) -> bool: + """POST a single memory chunk to the mem0 API.""" + metadata: dict = { + "type": "project_profile", + "source": source, + } + if branch: + metadata["branch"] = branch + + body = { + "messages": [{"role": "user", "content": content}], + "user_id": user_id, + "app_id": project_id, + "metadata": metadata, + "infer": False, + } + data = json.dumps(body).encode("utf-8") + req = urllib.request.Request( + f"{API_URL}/v1/memories/", + data=data, + headers={ + "Content-Type": "application/json", + "Authorization": f"Token {api_key}", + }, + method="POST", + ) + try: + with urllib.request.urlopen(req, timeout=20) as resp: + return resp.status in (200, 201) + except urllib.error.URLError as e: + print(f" [warn] API call failed: {e}", file=sys.stderr) + return False + + +def import_chunks(chunks: list[str], api_key: str, user_id: str, project_id: str, branch: str, source: str) -> int: + """Import a list of content chunks; return number of successful imports.""" + success = 0 + for chunk in chunks: + if post_memory(api_key, chunk, user_id, project_id, branch, source): + success += 1 + return success + + +# --------------------------------------------------------------------------- +# Sub-command implementations +# --------------------------------------------------------------------------- + + +def _parse_path_arg(args: list[str], flag: str, default: str) -> str: + """Extract --path from args list, falling back to default.""" + for i, arg in enumerate(args): + if arg == flag and i + 1 < len(args): + return args[i + 1] + if arg.startswith(f"{flag}="): + return arg[len(flag) + 1:] + return default + + +def cmd_cursorrules(args: list[str]) -> None: + path = _parse_path_arg(args, "--path", ".cursorrules") + source = "cursor-import" + + api_key = resolve_api_key() + user_id = resolve_user_id() + project_id = resolve_project_id() + branch = resolve_branch() + + if not api_key: + print("Error: MEM0_API_KEY not set", file=sys.stderr) + return + + if not os.path.isfile(path): + print(f"File not found: {path}", file=sys.stderr) + return + + with open(path, encoding="utf-8", errors="replace") as f: + content = f.read() + + raw_chunks = split_by_headers(content, "## ") + # Fall back to treating the whole file as one chunk if no headers found + if not raw_chunks: + raw_chunks = [content.strip()] if content.strip() else [] + + chunks = filter_and_truncate(raw_chunks) + n = import_chunks(chunks, api_key, user_id, project_id, branch, source) + print(f"Imported {n} memories from {source} ({path})") + + +def cmd_copilot(args: list[str]) -> None: + path = _parse_path_arg(args, "--path", ".github/copilot-instructions.md") + source = "copilot-import" + + api_key = resolve_api_key() + user_id = resolve_user_id() + project_id = resolve_project_id() + branch = resolve_branch() + + if not api_key: + print("Error: MEM0_API_KEY not set", file=sys.stderr) + return + + if not os.path.isfile(path): + print(f"File not found: {path}", file=sys.stderr) + return + + with open(path, encoding="utf-8", errors="replace") as f: + content = f.read() + + raw_chunks = split_by_headers(content, "## ") + if not raw_chunks: + raw_chunks = [content.strip()] if content.strip() else [] + + chunks = filter_and_truncate(raw_chunks) + n = import_chunks(chunks, api_key, user_id, project_id, branch, source) + print(f"Imported {n} memories from {source} ({path})") + + +def cmd_cline(args: list[str]) -> None: + dir_path = _parse_path_arg(args, "--path", "memory-bank/") + source = "cline-import" + + api_key = resolve_api_key() + user_id = resolve_user_id() + project_id = resolve_project_id() + branch = resolve_branch() + + if not api_key: + print("Error: MEM0_API_KEY not set", file=sys.stderr) + return + + if not os.path.isdir(dir_path): + print(f"Directory not found: {dir_path}", file=sys.stderr) + return + + md_files = sorted( + f for f in os.listdir(dir_path) if f.endswith(".md") + ) + if not md_files: + print(f"No .md files found in {dir_path}", file=sys.stderr) + return + + total = 0 + for filename in md_files: + filepath = os.path.join(dir_path, filename) + with open(filepath, encoding="utf-8", errors="replace") as f: + content = f.read().strip() + if not content: + continue + chunks = filter_and_truncate([content]) + n = import_chunks(chunks, api_key, user_id, project_id, branch, source) + total += n + + print(f"Imported {total} memories from {source} ({dir_path})") + + +def cmd_continue(args: list[str]) -> None: + path = _parse_path_arg(args, "--path", ".continue/rules.md") + source = "continue-import" + + api_key = resolve_api_key() + user_id = resolve_user_id() + project_id = resolve_project_id() + branch = resolve_branch() + + if not api_key: + print("Error: MEM0_API_KEY not set", file=sys.stderr) + return + + if not os.path.isfile(path): + print(f"File not found: {path}", file=sys.stderr) + return + + with open(path, encoding="utf-8", errors="replace") as f: + content = f.read() + + raw_chunks = split_by_hr_or_headers(content) + if not raw_chunks: + raw_chunks = [content.strip()] if content.strip() else [] + + chunks = filter_and_truncate(raw_chunks) + n = import_chunks(chunks, api_key, user_id, project_id, branch, source) + print(f"Imported {n} memories from {source} ({path})") + + +# --------------------------------------------------------------------------- +# Entry point +# --------------------------------------------------------------------------- + +COMMANDS = { + "cursorrules": cmd_cursorrules, + "copilot": cmd_copilot, + "cline": cmd_cline, + "continue": cmd_continue, +} + + +def main() -> None: + if len(sys.argv) < 2 or sys.argv[1] not in COMMANDS: + available = ", ".join(COMMANDS.keys()) + print("Usage: import_competing_tools.py [--path ]", file=sys.stderr) + print(f"Subcommands: {available}", file=sys.stderr) + sys.exit(0) + + subcommand = sys.argv[1] + remaining_args = sys.argv[2:] + COMMANDS[subcommand](remaining_args) + + +if __name__ == "__main__": + try: + main() + except Exception as e: + print(f"Unexpected error: {e}", file=sys.stderr) + sys.exit(0) diff --git a/mem0-plugin/scripts/parse_export_file.py b/mem0-plugin/scripts/parse_export_file.py new file mode 100644 index 000000000..bdf8a67e0 --- /dev/null +++ b/mem0-plugin/scripts/parse_export_file.py @@ -0,0 +1,147 @@ +#!/usr/bin/env python3 +"""Parse a mem0 export file and output JSON. + +Input: path to a mem0-export-*.md file (sys.argv[1]) +Output: JSON array of memory records to stdout +Exit: 0 always + +Each block in the file is delimited by lines containing exactly "---". +Blocks have a YAML-like frontmatter section (key: value lines) followed +by a blank line and the memory content text. + +Example block format: +--- +id: abc123 +created_at: 2024-01-01T00:00:00Z +type: task_learnings +confidence: 0.9 +branch: main +files: src/foo.py, src/bar.py +categories: coding_conventions, task_learnings +--- +The actual memory content text goes here. + +""" + +from __future__ import annotations + +import json +import re +import sys + + +def parse_blocks(content: str) -> list[dict]: + """Split content on '---' boundaries and parse each block. + + Returns a list of dicts with keys: + id, type, confidence, branch, files (list), categories (list), content (str) + + Blocks with empty content are skipped. + Missing optional fields default to "" (scalar) or [] (list fields). + """ + # Normalise line endings + content = content.replace("\r\n", "\n").replace("\r", "\n") + + # Split on lines that are exactly "---" + raw_blocks = re.split(r"(?m)^---\s*$", content) + + # After splitting on "---", the structure for each memory is: + # raw_blocks[0] = preamble (before first ---, typically empty) + # raw_blocks[1] = frontmatter for block 1 + # raw_blocks[2] = content for block 1 + # raw_blocks[3] = frontmatter for block 2 + # raw_blocks[4] = content for block 2 + # ... + # So frontmatter blocks are at odd indices (1, 3, 5, ...) and + # content blocks at even indices (2, 4, 6, ...). + + results: list[dict] = [] + + # Pair up frontmatter + content starting at index 1 + i = 1 + while i < len(raw_blocks): + frontmatter_raw = raw_blocks[i] + content_raw = raw_blocks[i + 1] if i + 1 < len(raw_blocks) else "" + + # Parse the frontmatter key-value pairs + fm = _parse_frontmatter(frontmatter_raw) + + # Strip leading/trailing whitespace from content + memory_content = content_raw.strip() + + # Skip blocks with empty content + if not memory_content: + i += 2 + continue + + record = { + "id": fm.get("id", ""), + "type": fm.get("type", ""), + "confidence": fm.get("confidence", ""), + "branch": fm.get("branch", ""), + "files": _parse_list_field(fm.get("files", "")), + "categories": _parse_list_field(fm.get("categories", "")), + "content": memory_content, + } + + # Include created_at if present + if "created_at" in fm: + record["created_at"] = fm["created_at"] + + results.append(record) + i += 2 + + return results + + +def _parse_frontmatter(text: str) -> dict[str, str]: + """Parse simple 'key: value' lines from frontmatter text. + + Only the first colon is used as the delimiter — values may contain colons. + Lines not matching 'key: value' are ignored. + """ + result: dict[str, str] = {} + for line in text.splitlines(): + line = line.strip() + if not line: + continue + match = re.match(r"^([A-Za-z_][A-Za-z0-9_]*)\s*:\s*(.*)$", line) + if match: + key = match.group(1).strip() + value = match.group(2).strip() + result[key] = value + return result + + +def _parse_list_field(value: str) -> list[str]: + """Split a comma-separated value into a list, stripping whitespace. + + Returns [] for empty/whitespace-only input. + """ + if not value or not value.strip(): + return [] + return [item.strip() for item in value.split(",") if item.strip()] + + +def main() -> None: + if len(sys.argv) < 2: + print("Usage: parse_export_file.py ", file=sys.stderr) + print("[]") + sys.exit(0) + + filepath = sys.argv[1] + try: + with open(filepath, encoding="utf-8", errors="replace") as f: + content = f.read() + except OSError as e: + print(f"Error reading file: {e}", file=sys.stderr) + print("[]") + sys.exit(0) + + records = parse_blocks(content) + print(json.dumps(records, ensure_ascii=False, indent=2)) + sys.exit(0) + + +if __name__ == "__main__": + main() diff --git a/mem0-plugin/skills/mem0-export/SKILL.md b/mem0-plugin/skills/mem0-export/SKILL.md new file mode 100644 index 000000000..f94611fa0 --- /dev/null +++ b/mem0-plugin/skills/mem0-export/SKILL.md @@ -0,0 +1,81 @@ +--- +name: mem0-export +description: > + Export all memories for the current project to a local Markdown file. + Each memory is written as a YAML-frontmatter block that can be re-imported later. + TRIGGER: user runs /mem0:export, or asks "export memories", "backup memories", + "download my memories", "save memories to file". +--- + +# Mem0 Export + +Export all memories for the current project to a portable Markdown file. + +## Execution + +### Step 1: Resolve identity + +Determine the active identity: +- `user_id` from `MEM0_USER_ID` env var, else `$USER`, else `"default"` +- `project_id` (used as `app_id`) from `MEM0_PROJECT_ID` env var, or via the project resolver + +### Step 2: Fetch all memories + +Call `get_memories` with: +- `user_id=` +- `app_id=` +- `page_size=200` + +If the response is paginated (i.e. the result contains a `next` cursor or the count equals `page_size`), continue fetching pages until all memories are retrieved. + +### Step 3: Format each memory as a YAML-frontmatter block + +For each memory record, produce a block in this exact format: + +``` +--- +id: +created_at: +type: +confidence: +branch: +files: +categories: +--- + + +``` + +Notes: +- The `---` delimiters must be on their own lines with no extra whitespace. +- `files` and `categories` are written as comma-separated values on a single line. +- Leave a blank line after the content before the next `---` (for readability). +- If a field is missing or null, write an empty string (not "null"). + +### Step 4: Write the export file + +Determine the output filename: + +``` +mem0-export--.md +``` + +Where `` is today's date in UTC. + +Write all formatted blocks to this file using the Write tool (or equivalent). The file is written to the current working directory. + +### Step 5: Print summary + +``` +Exported memories to +``` + +Where `` is the total number of memory blocks written. + +## Error Handling + +- If `get_memories` returns an error or zero memories, print: + ``` + No memories found for project . Nothing exported. + ``` +- If the write fails, report the error to the user. diff --git a/mem0-plugin/skills/mem0-import-tools/SKILL.md b/mem0-plugin/skills/mem0-import-tools/SKILL.md new file mode 100644 index 000000000..9336fe654 --- /dev/null +++ b/mem0-plugin/skills/mem0-import-tools/SKILL.md @@ -0,0 +1,106 @@ +--- +name: mem0-import-tools +description: > + Import memories from competing AI tool configuration files into mem0. + Supports Cursor (.cursorrules), GitHub Copilot (.github/copilot-instructions.md), + Cline (memory-bank/), and Continue (.continue/rules.md). + TRIGGER: user runs /mem0:import-tools, or asks "import from cursor", + "import cursorrules", "import from cline", "import from copilot", + "import from continue", "migrate from cursor", "migrate memories". +--- + +# Mem0 Import from Competing Tools + +Import configuration and memory files from other AI coding tools into mem0. + +## Supported Tools + +| Tool | Default file/directory | +|------|----------------------| +| Cursor | `.cursorrules` | +| GitHub Copilot | `.github/copilot-instructions.md` | +| Cline | `memory-bank/` (directory of `.md` files) | +| Continue | `.continue/rules.md` | + +## Execution + +### Step 1: Detect which tool files exist + +Check for the presence of each tool's file/directory in the current working directory: + +```bash +# Check each location +test -f .cursorrules && echo "cursor: .cursorrules" +test -f .github/copilot-instructions.md && echo "copilot: .github/copilot-instructions.md" +test -d memory-bank/ && echo "cline: memory-bank/" +test -f .continue/rules.md && echo "continue: .continue/rules.md" +``` + +### Step 2: Report findings and ask user + +List all found files to the user. For example: + +``` +Found the following tool configuration files: + [1] Cursor rules: .cursorrules + [2] Cline memory bank: memory-bank/ + +Which would you like to import? (enter numbers, comma-separated, or "all"): +``` + +If no files are found, print: +``` +No competing tool configuration files found in the current directory. +Checked: .cursorrules, .github/copilot-instructions.md, memory-bank/, .continue/rules.md +``` +and stop. + +### Step 3: Run the import script for each selected tool + +Determine the plugin root. Use the appropriate variable for the current platform: +- Claude Code: `${CLAUDE_PLUGIN_ROOT}` +- Codex: `${CODEX_PLUGIN_ROOT}` +- Cursor: `${CURSOR_PLUGIN_ROOT}` + +For each tool the user selected, run the corresponding sub-command: + +**Cursor (.cursorrules):** +```bash +python3 "/scripts/import_competing_tools.py" cursorrules --path .cursorrules +``` + +**GitHub Copilot:** +```bash +python3 "/scripts/import_competing_tools.py" copilot --path .github/copilot-instructions.md +``` + +**Cline:** +```bash +python3 "/scripts/import_competing_tools.py" cline --path memory-bank/ +``` + +**Continue:** +```bash +python3 "/scripts/import_competing_tools.py" continue --path .continue/rules.md +``` + +### Step 4: Report results + +After each script runs, echo its output to the user. Then print a combined summary: + +``` +Import complete. + Cursor: memories + Copilot: memories + Total: memories imported into project +``` + +Adjust the summary to reflect only the tools that were actually imported. + +## Notes + +- Memories are imported with `infer=False` — no AI inference is applied, content is stored as-is. +- Each section or file becomes a separate memory tagged with `metadata.source=-import` and `metadata.type=project_profile`. +- Sections shorter than 50 characters are automatically skipped (too short to be useful). +- Content longer than 10,000 characters is automatically truncated per chunk. +- You can re-run this skill safely — duplicate content will be caught by mem0's deduplication. diff --git a/mem0-plugin/skills/mem0-import/SKILL.md b/mem0-plugin/skills/mem0-import/SKILL.md new file mode 100644 index 000000000..3f3032364 --- /dev/null +++ b/mem0-plugin/skills/mem0-import/SKILL.md @@ -0,0 +1,104 @@ +--- +name: mem0-import +description: > + Import memories from a mem0 export file back into the current project. + Reads a YAML-frontmatter Markdown file produced by /mem0:export and + adds each memory block to mem0. + TRIGGER: user runs /mem0:import, or asks "import memories", "restore memories", + "load memories from file", "reimport backup". +--- + +# Mem0 Import + +Import memories from a mem0 export file into the current project. + +## Execution + +### Step 1: Determine the export file to import + +If the user provided a filename as an argument to `/mem0:import `, use that file. + +Otherwise, list `.md` files in the current directory whose names contain `mem0-export`: + +```bash +ls -1 *.md 2>/dev/null | grep mem0-export || echo "No export files found" +``` + +If multiple files are found, ask the user which one to import. If none are found, print: +``` +No mem0-export files found in the current directory. +Run /mem0:export first, or provide the filename: /mem0:import +``` + +### Step 2: Parse the export file + +Determine the plugin root. Use the appropriate variable for the current platform: +- Claude Code: `${CLAUDE_PLUGIN_ROOT}` +- Codex: `${CODEX_PLUGIN_ROOT}` +- Cursor: `${CURSOR_PLUGIN_ROOT}` + +Run the parser script to extract memory records as JSON: + +```bash +python3 "/scripts/parse_export_file.py" "" +``` + +This outputs a JSON array where each element has: +- `id` — original memory ID (for reference only; a new ID will be assigned on import) +- `type` — metadata type +- `confidence` — metadata confidence value +- `branch` — metadata branch +- `files` — list of associated files +- `categories` — list of categories +- `content` — the memory text + +If the script fails or outputs `[]`, print: +``` +Failed to parse or file contains no valid memory blocks. +``` +and stop. + +### Step 3: Resolve identity + +Determine the active identity: +- `user_id` from `MEM0_USER_ID` env var, else `$USER`, else `"default"` +- `project_id` (used as `app_id`) from `MEM0_PROJECT_ID` env var, or via the project resolver + +### Step 4: Import each memory + +For each record in the parsed JSON array, call `add_memory` with: + +- `messages=[{"role": "user", "content": ""}]` +- `user_id=` +- `app_id=` +- `metadata={` + - `"type": ""` (if non-empty) + - `"confidence": ""` (if non-empty) + - `"branch": ""` (if non-empty) + - `"files": ` (the list, if non-empty) + - `"source": "import"` + - `}` +- `infer=False` + +Notes: +- Do NOT pass the original `id` — the platform assigns a new ID. +- Skip records where `content` is empty (the parser already filters these, but be defensive). +- Continue importing even if individual records fail; track the count of successes. + +### Step 5: Print results + +``` +Imported memories into project +``` + +Where `` is the number of successfully imported memories. + +If any failed: +``` +Imported / memories into project ( failed) +``` + +## Error Handling + +- If the parser script is not found at `/scripts/parse_export_file.py`, print an error and stop. +- If `add_memory` calls fail consistently (e.g. auth error), report the issue and stop early. diff --git a/mem0-plugin/tests/test_import_competing_tools.py b/mem0-plugin/tests/test_import_competing_tools.py new file mode 100644 index 000000000..5c78a5a0c --- /dev/null +++ b/mem0-plugin/tests/test_import_competing_tools.py @@ -0,0 +1,355 @@ +"""Tests for import_competing_tools.py — competing tool file importers.""" + +from __future__ import annotations + +import json +import os +import sys +from unittest import mock + +SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") + + +# --------------------------------------------------------------------------- +# split_sections tests (unit tests on the splitter functions) +# --------------------------------------------------------------------------- + + +def test_split_by_headers_cursorrules(): + """split_by_headers correctly splits .cursorrules content on ## headers.""" + from import_competing_tools import split_by_headers + + content = """\ +# My Cursor Rules + +Some preamble text that belongs to the first section. + +## TypeScript Conventions + +Always use strict mode. Prefer const over let. +Never use var. + +## React Patterns + +Use functional components with hooks. +Avoid class components. + +## Testing + +Write tests for all utility functions. +""" + chunks = split_by_headers(content, "## ") + assert len(chunks) == 4 # preamble + 3 sections + + # First chunk is the preamble (before any ## header) + assert "preamble text" in chunks[0] + + # Remaining chunks start with their header + assert chunks[1].startswith("## TypeScript Conventions") + assert "strict mode" in chunks[1] + + assert chunks[2].startswith("## React Patterns") + assert "functional components" in chunks[2] + + assert chunks[3].startswith("## Testing") + assert "utility functions" in chunks[3] + + +def test_split_by_headers_copilot(): + """split_by_headers correctly splits copilot-instructions.md on ## headers.""" + from import_competing_tools import split_by_headers + + content = """\ +## Code Style + +Use 2-space indentation. Always add trailing commas. + +## Architecture + +Follow clean architecture principles. Keep business logic in domain layer. +""" + chunks = split_by_headers(content, "## ") + assert len(chunks) == 2 + assert chunks[0].startswith("## Code Style") + assert "2-space indentation" in chunks[0] + assert chunks[1].startswith("## Architecture") + assert "clean architecture" in chunks[1] + + +def test_split_by_headers_no_headers(): + """split_by_headers returns entire content as one chunk if no headers found.""" + from import_competing_tools import split_by_headers + + content = "This file has no headers at all. Just plain text." + chunks = split_by_headers(content, "## ") + assert len(chunks) == 1 + assert "Just plain text" in chunks[0] + + +def test_split_cline_multiple_md_files(tmp_path): + """cmd_cline processes multiple .md files from memory-bank/ directory.""" + from import_competing_tools import filter_and_truncate + + # Create a temporary memory-bank directory with .md files + mb_dir = tmp_path / "memory-bank" + mb_dir.mkdir() + + (mb_dir / "architecture.md").write_text( + "# Architecture Decisions\n\nUse microservices architecture with event sourcing." + ) + (mb_dir / "conventions.md").write_text( + "# Code Conventions\n\nAll functions must have type hints. Use black formatter." + ) + (mb_dir / "empty.md").write_text("") # empty file should be skipped + + # Read and verify we can split the files + md_files = sorted(f for f in os.listdir(str(mb_dir)) if f.endswith(".md")) + assert "architecture.md" in md_files + assert "conventions.md" in md_files + assert "empty.md" in md_files + + non_empty = [] + for filename in md_files: + filepath = os.path.join(str(mb_dir), filename) + with open(filepath) as f: + content = f.read().strip() + if content: + chunks = filter_and_truncate([content]) + non_empty.extend(chunks) + + assert len(non_empty) == 2 + assert any("microservices" in c for c in non_empty) + assert any("type hints" in c for c in non_empty) + + +def test_split_by_hr_or_headers_continue(): + """split_by_hr_or_headers correctly splits .continue/rules.md.""" + from import_competing_tools import split_by_hr_or_headers + + content = """\ +## First Section + +Content of first section. + +--- + +## Second Section + +Content of second section. + +--- + +Third section without a header (just after HR). +""" + chunks = split_by_hr_or_headers(content) + # Should split into meaningful chunks + assert len(chunks) >= 2 + assert any("First Section" in c for c in chunks) + assert any("Second Section" in c for c in chunks) + + +def test_filter_and_truncate_skips_short(): + """filter_and_truncate skips chunks shorter than MIN_CHUNK_CHARS (50).""" + from import_competing_tools import filter_and_truncate + + chunks = [ + "Short", # < 50 chars, should be filtered + "A" * 49, # exactly 49 chars, should be filtered + "A" * 50, # exactly 50 chars, should be kept + "A long enough chunk that definitely passes the minimum length filter.", + ] + result = filter_and_truncate(chunks) + assert len(result) == 2 + assert all(len(c) >= 50 for c in result) + + +def test_filter_and_truncate_truncates_long(): + """filter_and_truncate truncates chunks over MAX_CHUNK_CHARS (10000).""" + from import_competing_tools import MAX_CHUNK_CHARS, filter_and_truncate + + long_chunk = "X" * (MAX_CHUNK_CHARS + 500) + result = filter_and_truncate([long_chunk]) + assert len(result) == 1 + assert len(result[0]) == MAX_CHUNK_CHARS + + +# --------------------------------------------------------------------------- +# Mock API tests +# --------------------------------------------------------------------------- + + +def _make_mock_response(status: int = 201, body: dict | None = None) -> mock.MagicMock: + """Create a mock HTTP response object.""" + if body is None: + body = {"id": "new-mem-id", "memory": "test"} + resp = mock.MagicMock() + resp.status = status + resp.read.return_value = json.dumps(body).encode() + resp.__enter__ = lambda s: s + resp.__exit__ = mock.MagicMock(return_value=False) + return resp + + +def test_cursorrules_import_api_call(tmp_path): + """cmd_cursorrules calls the API with correct app_id (top-level), infer=False, and correct source.""" + from import_competing_tools import cmd_cursorrules + + # Create a .cursorrules file with enough content + cursorrules = tmp_path / ".cursorrules" + cursorrules.write_text( + "## TypeScript Rules\n\nAlways use strict TypeScript. Never use 'any' type. " + "Prefer interfaces over type aliases for object shapes." + ) + + captured_requests: list[dict] = [] + + def mock_urlopen(req, timeout=None): + body = json.loads(req.data.decode()) + captured_requests.append(body) + return _make_mock_response(201) + + with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-testkey"), \ + mock.patch("import_competing_tools.resolve_user_id", return_value="testuser"), \ + mock.patch("import_competing_tools.resolve_project_id", return_value="my-project"), \ + mock.patch("import_competing_tools.resolve_branch", return_value="main"), \ + mock.patch("urllib.request.urlopen", side_effect=mock_urlopen): + + original_cwd = os.getcwd() + os.chdir(str(tmp_path)) + try: + cmd_cursorrules(["--path", str(cursorrules)]) + finally: + os.chdir(original_cwd) + + assert len(captured_requests) >= 1 + + req_body = captured_requests[0] + + # app_id must be top-level (not inside metadata) + assert req_body["app_id"] == "my-project", f"Expected app_id at top level, got: {req_body}" + + # infer must be False (not "false", but the boolean False) + assert req_body["infer"] is False, f"Expected infer=False, got: {req_body['infer']}" + + # source must be cursor-import + assert req_body["metadata"]["source"] == "cursor-import", ( + f"Expected source=cursor-import, got: {req_body['metadata'].get('source')}" + ) + + # user_id must be set + assert req_body["user_id"] == "testuser" + + # messages must be a list with role/content + assert isinstance(req_body["messages"], list) + assert req_body["messages"][0]["role"] == "user" + assert len(req_body["messages"][0]["content"]) > 0 + + +def test_copilot_import_api_call(tmp_path): + """cmd_copilot calls the API with source=copilot-import.""" + from import_competing_tools import cmd_copilot + + copilot_dir = tmp_path / ".github" + copilot_dir.mkdir() + copilot_file = copilot_dir / "copilot-instructions.md" + copilot_file.write_text( + "## Code Style\n\nUse 2-space indentation. Always add trailing commas in multi-line structures. " + "Prefer const over let. Never use var in JavaScript code." + ) + + captured: list[dict] = [] + + def mock_urlopen(req, timeout=None): + captured.append(json.loads(req.data.decode())) + return _make_mock_response(201) + + with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-key"), \ + mock.patch("import_competing_tools.resolve_user_id", return_value="user1"), \ + mock.patch("import_competing_tools.resolve_project_id", return_value="proj1"), \ + mock.patch("import_competing_tools.resolve_branch", return_value="main"), \ + mock.patch("urllib.request.urlopen", side_effect=mock_urlopen): + + cmd_copilot(["--path", str(copilot_file)]) + + assert len(captured) >= 1 + assert captured[0]["metadata"]["source"] == "copilot-import" + assert captured[0]["app_id"] == "proj1" + assert captured[0]["infer"] is False + + +def test_cline_import_multiple_files(tmp_path): + """cmd_cline imports one memory per non-empty .md file.""" + from import_competing_tools import cmd_cline + + mb = tmp_path / "memory-bank" + mb.mkdir() + (mb / "arch.md").write_text( + "Architecture: microservices with event-sourcing. Each service owns its database. " + "Communication via message bus only. No direct service-to-service HTTP calls." + ) + (mb / "style.md").write_text( + "Code style: PEP 8 for Python. Black formatter. isort for imports. " + "Line length 120. Type hints required on all public functions and methods." + ) + (mb / "empty.md").write_text("") + + captured: list[dict] = [] + + def mock_urlopen(req, timeout=None): + captured.append(json.loads(req.data.decode())) + return _make_mock_response(201) + + with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-key"), \ + mock.patch("import_competing_tools.resolve_user_id", return_value="user1"), \ + mock.patch("import_competing_tools.resolve_project_id", return_value="proj1"), \ + mock.patch("import_competing_tools.resolve_branch", return_value="main"), \ + mock.patch("urllib.request.urlopen", side_effect=mock_urlopen): + + cmd_cline(["--path", str(mb)]) + + # Should have exactly 2 imports (empty.md skipped) + assert len(captured) == 2 + sources = {r["metadata"]["source"] for r in captured} + assert sources == {"cline-import"} + for r in captured: + assert r["infer"] is False + assert r["app_id"] == "proj1" + + +def test_no_api_key_does_not_call_api(tmp_path): + """When no API key is set, no HTTP call is made.""" + from import_competing_tools import cmd_cursorrules + + cursorrules = tmp_path / ".cursorrules" + cursorrules.write_text("## Rules\n\n" + "x" * 100) + + with mock.patch("import_competing_tools.resolve_api_key", return_value=""), \ + mock.patch("urllib.request.urlopen") as mock_url: + + cmd_cursorrules(["--path", str(cursorrules)]) + + mock_url.assert_not_called() + + +def test_missing_file_does_not_call_api(tmp_path): + """When the source file doesn't exist, no HTTP call is made.""" + from import_competing_tools import cmd_cursorrules + + with mock.patch("import_competing_tools.resolve_api_key", return_value="m0-key"), \ + mock.patch("urllib.request.urlopen") as mock_url: + + cmd_cursorrules(["--path", str(tmp_path / "nonexistent.cursorrules")]) + + mock_url.assert_not_called() + + +def test_main_unknown_subcommand_exits_zero(): + """Calling main() with an unknown subcommand exits 0.""" + import subprocess + result = subprocess.run( + [sys.executable, os.path.join(SCRIPTS_DIR, "import_competing_tools.py"), "unknown"], + capture_output=True, + text=True, + env={**os.environ, "MEM0_API_KEY": ""}, + ) + assert result.returncode == 0 diff --git a/mem0-plugin/tests/test_parse_export_file.py b/mem0-plugin/tests/test_parse_export_file.py new file mode 100644 index 000000000..66dccbd1d --- /dev/null +++ b/mem0-plugin/tests/test_parse_export_file.py @@ -0,0 +1,298 @@ +"""Tests for parse_export_file.py — mem0 export file parser.""" + +from __future__ import annotations + +import json +import os +import subprocess +import sys + +SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") + + +# --------------------------------------------------------------------------- +# Direct function tests +# --------------------------------------------------------------------------- + + +def test_parse_blocks_single_valid_block(): + """parse_blocks returns one record for a single valid block.""" + from parse_export_file import parse_blocks + + content = """\ +--- +id: abc123 +created_at: 2024-01-15T10:00:00Z +type: task_learnings +confidence: 0.85 +branch: main +files: src/foo.py, src/bar.py +categories: coding_conventions, task_learnings +--- +Always use context managers when opening files. +""" + records = parse_blocks(content) + assert len(records) == 1 + r = records[0] + assert r["id"] == "abc123" + assert r["type"] == "task_learnings" + assert r["confidence"] == "0.85" + assert r["branch"] == "main" + assert r["files"] == ["src/foo.py", "src/bar.py"] + assert r["categories"] == ["coding_conventions", "task_learnings"] + assert "Always use context managers" in r["content"] + + +def test_parse_blocks_multiple_blocks(): + """parse_blocks returns the correct number of records for multiple blocks.""" + from parse_export_file import parse_blocks + + content = """\ +--- +id: mem001 +type: architecture_decisions +confidence: 0.9 +branch: main +files: +categories: architecture_decisions +--- +Use hexagonal architecture for the core domain. + +--- +id: mem002 +type: anti_patterns +confidence: 0.75 +branch: feat/refactor +files: src/legacy.py +categories: anti_patterns +--- +Avoid direct database calls from view layer. + +--- +id: mem003 +type: coding_conventions +confidence: 0.8 +branch: main +files: src/utils.py, src/helpers.py +categories: +--- +Use snake_case for all Python identifiers. +""" + records = parse_blocks(content) + assert len(records) == 3 + + assert records[0]["id"] == "mem001" + assert records[0]["categories"] == ["architecture_decisions"] + assert "hexagonal architecture" in records[0]["content"] + + assert records[1]["id"] == "mem002" + assert records[1]["files"] == ["src/legacy.py"] + assert "direct database calls" in records[1]["content"] + + assert records[2]["id"] == "mem003" + assert records[2]["files"] == ["src/utils.py", "src/helpers.py"] + assert records[2]["categories"] == [] + assert "snake_case" in records[2]["content"] + + +def test_parse_blocks_missing_optional_fields(): + """parse_blocks uses defaults when optional fields are absent.""" + from parse_export_file import parse_blocks + + # confidence, branch, files, categories all absent + content = """\ +--- +id: xyz789 +type: task_learnings +--- +Run tests before committing. +""" + records = parse_blocks(content) + assert len(records) == 1 + r = records[0] + assert r["id"] == "xyz789" + assert r["confidence"] == "" # default empty string + assert r["branch"] == "" # default empty string + assert r["files"] == [] # default empty list + assert r["categories"] == [] # default empty list + assert "Run tests" in r["content"] + + +def test_parse_blocks_filters_empty_content(): + """parse_blocks skips blocks whose content is empty or whitespace-only.""" + from parse_export_file import parse_blocks + + content = """\ +--- +id: empty1 +type: task_learnings +--- + +--- +id: real1 +type: task_learnings +--- +This block has real content. + +--- +id: empty2 +type: coding_conventions +--- + +""" + records = parse_blocks(content) + # Only the block with actual content should be returned + assert len(records) == 1 + assert records[0]["id"] == "real1" + assert "real content" in records[0]["content"] + + +def test_parse_blocks_round_trip(): + """Content formatted by export matches what parse_blocks expects.""" + from parse_export_file import parse_blocks + + # Simulate the exact format produced by the export skill + memory_id = "test-id-001" + created_at = "2024-06-01T12:00:00Z" + mem_type = "architecture_decisions" + confidence = "0.92" + branch = "feat/new-feature" + files = ["src/main.py", "tests/test_main.py"] + categories = ["architecture_decisions", "coding_conventions"] + memory_content = "Use dependency injection for all service classes." + + # Format exactly as the export skill would + block = ( + "---\n" + f"id: {memory_id}\n" + f"created_at: {created_at}\n" + f"type: {mem_type}\n" + f"confidence: {confidence}\n" + f"branch: {branch}\n" + f"files: {', '.join(files)}\n" + f"categories: {', '.join(categories)}\n" + "---\n" + f"{memory_content}\n" + "\n" + ) + + records = parse_blocks(block) + assert len(records) == 1 + r = records[0] + assert r["id"] == memory_id + assert r["type"] == mem_type + assert r["confidence"] == confidence + assert r["branch"] == branch + assert r["files"] == files + assert r["categories"] == categories + assert r["content"] == memory_content + + +def test_parse_blocks_multiline_content(): + """parse_blocks correctly captures multi-line memory content.""" + from parse_export_file import parse_blocks + + content = """\ +--- +id: multi001 +type: task_learnings +--- +Line one of the memory. +Line two of the memory. + +Line four after blank line. +""" + records = parse_blocks(content) + assert len(records) == 1 + assert "Line one" in records[0]["content"] + assert "Line two" in records[0]["content"] + assert "Line four" in records[0]["content"] + + +def test_parse_blocks_empty_input(): + """parse_blocks returns empty list for empty input.""" + from parse_export_file import parse_blocks + + assert parse_blocks("") == [] + assert parse_blocks(" \n ") == [] + + +def test_parse_blocks_no_blocks(): + """parse_blocks returns empty list for content without any --- delimiters.""" + from parse_export_file import parse_blocks + + assert parse_blocks("Just some text without any delimiters.") == [] + + +def test_parse_blocks_value_with_colon(): + """parse_blocks handles values that themselves contain colons.""" + from parse_export_file import parse_blocks + + content = """\ +--- +id: colon-test +type: task_learnings +created_at: 2024-01-01T10:00:00Z +--- +Timestamp values contain colons and should parse correctly. +""" + records = parse_blocks(content) + assert len(records) == 1 + assert records[0]["id"] == "colon-test" + # created_at field should be captured (it's in the record if present) + assert "2024-01-01T10:00:00Z" in records[0].get("created_at", "") + + +# --------------------------------------------------------------------------- +# CLI / subprocess tests +# --------------------------------------------------------------------------- + + +def test_main_cli_outputs_json(tmp_path): + """Running parse_export_file.py as a script outputs valid JSON.""" + export_file = tmp_path / "mem0-export-test.md" + export_file.write_text("""\ +--- +id: cli-test-001 +type: task_learnings +confidence: 0.8 +branch: main +files: +categories: task_learnings +--- +Prefer composition over inheritance. +""") + + result = subprocess.run( + [sys.executable, os.path.join(SCRIPTS_DIR, "parse_export_file.py"), str(export_file)], + capture_output=True, + text=True, + ) + assert result.returncode == 0 + records = json.loads(result.stdout) + assert isinstance(records, list) + assert len(records) == 1 + assert records[0]["id"] == "cli-test-001" + + +def test_main_cli_no_args_exits_zero(): + """Running parse_export_file.py with no arguments exits 0 and prints [].""" + result = subprocess.run( + [sys.executable, os.path.join(SCRIPTS_DIR, "parse_export_file.py")], + capture_output=True, + text=True, + ) + assert result.returncode == 0 + assert result.stdout.strip() == "[]" + + +def test_main_cli_missing_file_exits_zero(tmp_path): + """Running parse_export_file.py with a missing file exits 0 and prints [].""" + result = subprocess.run( + [sys.executable, os.path.join(SCRIPTS_DIR, "parse_export_file.py"), + str(tmp_path / "nonexistent.md")], + capture_output=True, + text=True, + ) + assert result.returncode == 0 + assert result.stdout.strip() == "[]"