diff --git a/openclaw/CHANGELOG.md b/openclaw/CHANGELOG.md index 3173aeb51..d464f7d31 100644 --- a/openclaw/CHANGELOG.md +++ b/openclaw/CHANGELOG.md @@ -2,8 +2,46 @@ All notable changes to the `@mem0/openclaw-mem0` plugin will be documented in this file. +## [0.4.0] - 2026-03-16 + +### Added +- **Non-interactive trigger filtering**: Skips recall and capture for `cron`, `heartbeat`, `automation`, and `schedule` triggers — prevents system-generated noise from polluting memory +- **Subagent hallucination prevention**: `isSubagentSession()` detects ephemeral subagent sessions and routes recall to the parent (main user) namespace instead of empty ephemeral namespaces; skips capture to prevent orphaned memories +- **Subagent-specific preamble**: Subagents receive "You are a subagent — use these memories for context but do not assume you are this user" to prevent identity assumption +- **User identity in recall preamble**: Recalled memories now include `userId` attribution for better context +- **User identity in extraction preamble**: Extraction context includes user identity and current date for accurate attribution and temporal anchoring +- **User-content guard**: Skips extraction when no meaningful user messages remain after filtering +- **Dynamic recall thresholding**: Memories scoring less than 50% of the top result are dropped to filter out the long tail of weak matches +- **SQLite resilience for OSS mode**: Init error recovery with automatic retry (history disabled) when native SQLite bindings fail under jiti +- **`disableHistory` config option**: New `oss.disableHistory` flag to explicitly skip history DB initialization +- **Updated minimum package version of mem0ai package**: Updated minimum package version of mem0ai package to ^2.3.0 to force old users to migrate to better-sqlite3 +- 78 unit tests covering filtering, isolation, trigger filtering, subagent detection, and SQLite resilience + +### Changed +- Auto-recall threshold raised from 0.5 to 0.6 for stricter precision during automatic injection (explicit tool searches remain at 0.5) +- Recall candidate pool increased to `topK * 2` for better filtering headroom +- Provider init promises now reset on failure, allowing retry on subsequent calls +- Relaxed extraction instructions: related facts are kept together to preserve context (removed atomic memory requirement) + +### Fixed +- **Concurrent session race condition**: Lifecycle hooks (`before_agent_start`, `agent_end`) now use `ctx.sessionKey` directly from the event context instead of a shared mutable `currentSessionId` variable, preventing cross-session data leaks when multiple sessions run simultaneously + ## [0.3.1] - 2026-03-12 +### Added +- **Message filtering pipeline**: Multi-stage noise removal before extraction — drops heartbeats, timestamps, single-word acks, system routing metadata, compaction audit logs, and generic assistant acknowledgments +- **Broad recall for new sessions**: Short or new-session prompts trigger a secondary broad search to avoid cold-start blindness +- **Client-side threshold filtering**: Safety net that drops low-relevance results even if the API doesn't honor the threshold parameter +- **Temporal anchoring**: Extraction instructions now include current date so memories are prefixed with "As of YYYY-MM-DD, ..." +- **Summary message inclusion**: Earlier assistant messages containing work summaries are included in extraction context even if outside the recent-message window +- 55 unit tests covering filtering and isolation helpers + +### Changed +- Default `searchThreshold` remains at 0.5, with client-side filtering as a safety net +- Extraction window expanded from last 10 → last 20 messages for richer context +- Rewritten custom extraction instructions: conciseness, outcome-over-intent, deduplication guidance, language preservation +- **Refactored** monolithic `index.ts` (1772 lines) into 6 focused modules: `types.ts`, `providers.ts`, `config.ts`, `filtering.ts`, `isolation.ts`, `index.ts` + ### Fixed - **README image on npmjs.com**: Changed architecture diagram from relative path to absolute GitHub URL so it renders correctly on the npm registry diff --git a/openclaw/README.md b/openclaw/README.md index 517b2ca6b..951781bd6 100644 --- a/openclaw/README.md +++ b/openclaw/README.md @@ -12,10 +12,19 @@ Your agent forgets everything between sessions. This plugin fixes that. It watch **Auto-Recall** — Before the agent responds, the plugin searches Mem0 for memories that match the current message and injects them into context. -**Auto-Capture** — After the agent responds, the plugin sends the exchange to Mem0. Mem0 decides what's worth keeping — new facts get stored, stale ones updated, duplicates merged. +**Auto-Capture** — After the agent responds, the plugin filters the conversation through a noise-removal pipeline, then sends the cleaned exchange to Mem0. Mem0 decides what's worth keeping — new facts get stored, stale ones updated, duplicates merged. Both run silently. No prompting, no configuration, no manual calls. +### Message filtering + +Before extraction, messages pass through a multi-stage filtering pipeline: + +1. **Noise detection** — Drops entire messages that are system noise: heartbeats (`HEARTBEAT_OK`, `NO_REPLY`), timestamps, single-word acknowledgments (`ok`, `sure`, `done`), system routing metadata, and compaction audit logs. +2. **Generic assistant detection** — Drops short assistant messages that are boilerplate acknowledgments with no extractable facts (e.g. "I see you've shared an update. How can I help?"). +3. **Content stripping** — Removes embedded noise fragments (media boilerplate, routing metadata, compaction blocks) from otherwise useful messages. +4. **Truncation** — Caps messages at 2000 characters to avoid sending excessive context. + ### Short-term vs long-term memory Memories are organized into two scopes: @@ -40,6 +49,13 @@ In multi-agent setups, each agent automatically gets its own memory namespace. S - If the key matches `agent::`, memories are stored under `userId:agent:` - Different agents never see each other's memories unless explicitly queried +**Subagent handling:** + +Ephemeral subagents (session keys like `agent:main:subagent:`) are handled specially: +- **Recall** is routed to the parent (main user) namespace — subagents get the user's long-term context instead of searching their empty ephemeral namespace +- **Capture** is skipped entirely — the main agent's `agent_end` hook captures the consolidated result including subagent output, preventing orphaned memories +- A **subagent-specific preamble** is used: "You are a subagent — use these memories for context but do not assume you are this user" + **Explicit cross-agent queries:** All memory tools (`memory_search`, `memory_store`, `memory_list`, `memory_forget`) accept an optional `agentId` parameter to query another agent's namespace: @@ -48,7 +64,17 @@ All memory tools (`memory_search`, `memory_store`, `memory_list`, `memory_forget memory_search({ query: "user's tech stack", agentId: "researcher" }) ``` -Resolution priority: explicit `agentId` > explicit `userId` > session-derived > configured default. +The `agentId` is always namespaced under the configured `userId` (e.g. `agentId: "researcher"` → `utkarsh:agent:researcher`), so it cannot be used to access other users' namespaces. + +### Concurrency safety + +Lifecycle hooks (`before_agent_start`, `agent_end`) use `ctx.sessionKey` directly from the event context rather than shared mutable state. This prevents race conditions when multiple sessions run concurrently (e.g. multiple Telegram users chatting simultaneously). + +Tools still read from a best-effort `currentSessionId` variable (since tools don't receive `ctx`), but hooks — where the critical recall and capture logic runs — are fully concurrency-safe. + +### Non-interactive trigger filtering + +The plugin automatically skips recall and capture for non-interactive triggers: `cron`, `heartbeat`, `automation`, and `schedule`. Detection works via both `ctx.trigger` and session key patterns (`:cron:`, `:heartbeat:`). This prevents system-generated noise from polluting long-term memory. ## Setup @@ -121,10 +147,10 @@ The agent gets five tools it can call during conversations: | Tool | Description | |------|-------------| -| `memory_search` | Search memories by natural language. Optional `agentId` to scope to a specific agent. | -| `memory_list` | List all stored memories for a user. Optional `agentId` to scope to a specific agent. | -| `memory_store` | Explicitly save a fact. Optional `agentId` to store under a specific agent's namespace. | -| `memory_get` | Retrieve a memory by ID | +| `memory_search` | Search memories by natural language. Optional `agentId` to scope to a specific agent, `scope` to filter by session/long-term. | +| `memory_list` | List all stored memories. Optional `agentId` to scope to a specific agent, `scope` to filter. | +| `memory_store` | Explicitly save a fact. Optional `agentId` to store under a specific agent's namespace, `longTerm` to choose scope. | +| `memory_get` | Retrieve a memory by ID. | | `memory_forget` | Delete by ID or by query. Optional `agentId` to scope deletion to a specific agent. | ## CLI @@ -160,7 +186,7 @@ openclaw mem0 stats --agent researcher | `autoRecall` | `boolean` | `true` | Inject memories before each turn | | `autoCapture` | `boolean` | `true` | Store facts after each turn | | `topK` | `number` | `5` | Max memories per recall | -| `searchThreshold` | `number` | `0.3` | Min similarity (0–1) | +| `searchThreshold` | `number` | `0.5` | Min similarity (0–1) | ### Platform mode @@ -170,7 +196,7 @@ openclaw mem0 stats --agent researcher | `orgId` | `string` | — | Organization ID | | `projectId` | `string` | — | Project ID | | `enableGraph` | `boolean` | `false` | Entity graph for relationships | -| `customInstructions` | `string` | *(built-in)* | Extraction rules — what to store, how to format | +| `customInstructions` | `string` | *(built-in)* | Extraction rules — what to store, how to format. Built-in instructions include temporal anchoring, conciseness, outcome-over-intent, deduplication, and language preservation guidelines. | | `customCategories` | `object` | *(12 defaults)* | Category name → description map for tagging | ### Open-source mode @@ -187,9 +213,12 @@ Works with zero extra config. The `oss` block lets you swap out any component: | `oss.llm.provider` | `string` | `"openai"` | LLM provider (`"openai"`, `"anthropic"`, `"ollama"`, `"lmstudio"`, etc.) | | `oss.llm.config` | `object` | — | Provider config: `apiKey`, `model`, `baseURL`, `temperature` | | `oss.historyDbPath` | `string` | — | SQLite path for memory edit history | +| `oss.disableHistory` | `boolean` | `false` | Skip history DB initialization (useful when native SQLite bindings fail) | Everything inside `oss` is optional — defaults use OpenAI embeddings (`text-embedding-3-small`), in-memory vector store, and OpenAI LLM. Override only what you need. +> **SQLite resilience:** If the history DB fails to initialize (e.g. native binding resolution under jiti), the plugin automatically retries with history disabled. Core memory operations (add, search, get, delete) work without the history DB. + ## License Apache 2.0 diff --git a/openclaw/config.ts b/openclaw/config.ts new file mode 100644 index 000000000..f564f237a --- /dev/null +++ b/openclaw/config.ts @@ -0,0 +1,243 @@ +/** + * Configuration parsing, env var resolution, and default instructions/categories. + */ + +import type { Mem0Config, Mem0Mode } from "./types.ts"; + +// ============================================================================ +// Env Var Resolution +// ============================================================================ + +function resolveEnvVars(value: string): string { + return value.replace(/\$\{([^}]+)\}/g, (_, envVar) => { + const envValue = process.env[envVar]; + if (!envValue) { + throw new Error(`Environment variable ${envVar} is not set`); + } + return envValue; + }); +} + +function resolveEnvVarsDeep(obj: Record): Record { + const result: Record = {}; + for (const [key, value] of Object.entries(obj)) { + if (typeof value === "string") { + result[key] = resolveEnvVars(value); + } else if (value && typeof value === "object" && !Array.isArray(value)) { + result[key] = resolveEnvVarsDeep(value as Record); + } else { + result[key] = value; + } + } + return result; +} + +// ============================================================================ +// Default Custom Instructions & Categories +// ============================================================================ + +export const DEFAULT_CUSTOM_INSTRUCTIONS = `Your Task: Extract durable, actionable facts from conversations between a user and an AI assistant. Only store information that would be useful to an agent in a FUTURE session, days or weeks later. + +Before storing any fact, ask: "Would a new agent — with no prior context — benefit from knowing this?" If the answer is no, do not store it. + +Information to Extract (in priority order): + +1. Configuration & System State Changes: + - Tools/services configured, installed, or removed (with versions/dates) + - Model assignments for agents, API keys configured (NEVER the key itself — see Exclude) + - Cron schedules, automation pipelines, deployment configurations + - Architecture decisions (agent hierarchy, system design, deployment strategy) + - Specific identifiers: file paths, sheet IDs, channel IDs, user IDs, folder IDs + +2. Standing Rules & Policies: + - Explicit user directives about behavior ("never create accounts without consent") + - Workflow policies ("each agent must review model selection before completing a task") + - Security constraints, permission boundaries, access patterns + +3. Identity & Demographics: + - Name, location, timezone, language preferences + - Occupation, employer, job role, industry + +4. Preferences & Opinions: + - Communication style preferences + - Tool and technology preferences (with specifics: versions, configs) + - Strong opinions or values explicitly stated + - The WHY behind preferences when stated + +5. Goals, Projects & Milestones: + - Active projects (name, description, current status) + - Completed setup milestones ("ElevenLabs fully configured as of 2026-02-20") + - Deadlines, roadmaps, and progress tracking + - Problems actively being solved + +6. Technical Context: + - Tech stack, tools, development environment + - Agent ecosystem structure (names, roles, relationships) + - Skill levels in different areas + +7. Relationships & People: + - Names and roles of people mentioned (colleagues, family, clients) + - Team structure, key contacts + +8. Decisions & Lessons: + - Important decisions made and their reasoning + - Lessons learned, strategies that worked or failed + +Guidelines: + +TEMPORAL ANCHORING (critical): +- ALWAYS include temporal context for time-sensitive facts using "As of YYYY-MM-DD, ..." +- Extract dates from message timestamps, dates mentioned in the text, or the system-provided current date +- If no date is available, note "date unknown" rather than omitting temporal context +- Examples: "As of 2026-02-20, ElevenLabs setup is complete" NOT "ElevenLabs setup is complete" + +CONCISENESS: +- Use third person ("User prefers..." not "I prefer...") +- Keep related facts together in a single memory to preserve context +- "User's Tailscale machine 'mac' (IP 100.71.135.41) is configured under beau@rizedigital.io (as of 2026-02-20)" +- NOT a paragraph retelling the whole conversation + +OUTCOMES OVER INTENT: +- When an assistant message summarizes completed work, extract the durable OUTCOMES +- "Call scripts sheet (ID: 146Qbb...) was updated with truth-based templates" NOT "User wants to update call scripts" +- Extract what WAS DONE, not what was requested + +DEDUPLICATION: +- Before creating a new memory, check if a substantially similar fact already exists +- If so, UPDATE the existing memory with any new details rather than creating a duplicate + +LANGUAGE: +- ALWAYS preserve the original language of the conversation +- If the user speaks Spanish, store the memory in Spanish; do not translate + +Exclude (NEVER store): +- Passwords, API keys, tokens, secrets, or any credentials — even if shared in conversation. Instead store: "Tavily API key was configured and saved to .env (as of 2026-02-20)" +- One-time commands or instructions ("stop the script", "continue where you left off") +- Acknowledgments or emotional reactions ("ok", "sounds good", "you're right", "sir") +- Transient UI/navigation states ("user is in the admin panel", "relay is attached") +- Ephemeral process status ("download at 50%", "daemon not running", "still syncing") +- Cron heartbeat outputs, NO_REPLY responses, compaction flush directives +- System routing metadata (message IDs, sender IDs, channel routing info) +- Generic small talk with no informational content +- Raw code snippets (capture the intent/decision, not the code itself) +- Information the user explicitly asks not to remember`; + +export const DEFAULT_CUSTOM_CATEGORIES: Record = { + identity: + "Personal identity information: name, age, location, timezone, occupation, employer, education, demographics", + preferences: + "Explicitly stated likes, dislikes, preferences, opinions, and values across any domain", + goals: + "Current and future goals, aspirations, objectives, targets the user is working toward", + projects: + "Specific projects, initiatives, or endeavors the user is working on, including status and details", + technical: + "Technical skills, tools, tech stack, development environment, programming languages, frameworks", + decisions: + "Important decisions made, reasoning behind choices, strategy changes, and their outcomes", + relationships: + "People mentioned by the user: colleagues, family, friends, their roles and relevance", + routines: + "Daily habits, work patterns, schedules, productivity routines, health and wellness habits", + life_events: + "Significant life events, milestones, transitions, upcoming plans and changes", + lessons: + "Lessons learned, insights gained, mistakes acknowledged, changed opinions or beliefs", + work: + "Work-related context: job responsibilities, workplace dynamics, career progression, professional challenges", + health: + "Health-related information voluntarily shared: conditions, medications, fitness, wellness goals", +}; + +// ============================================================================ +// Config Schema +// ============================================================================ + +const ALLOWED_KEYS = [ + "mode", + "apiKey", + "userId", + "orgId", + "projectId", + "autoCapture", + "autoRecall", + "customInstructions", + "customCategories", + "customPrompt", + "enableGraph", + "searchThreshold", + "topK", + "oss", +]; + +function assertAllowedKeys( + value: Record, + allowed: string[], + label: string, +) { + const unknown = Object.keys(value).filter((key) => !allowed.includes(key)); + if (unknown.length === 0) return; + throw new Error(`${label} has unknown keys: ${unknown.join(", ")}`); +} + +export const mem0ConfigSchema = { + parse(value: unknown): Mem0Config { + if (!value || typeof value !== "object" || Array.isArray(value)) { + throw new Error("openclaw-mem0 config required"); + } + const cfg = value as Record; + assertAllowedKeys(cfg, ALLOWED_KEYS, "openclaw-mem0 config"); + + // Accept both "open-source" and legacy "oss" as open-source mode; everything else is platform + const mode: Mem0Mode = + cfg.mode === "oss" || cfg.mode === "open-source" ? "open-source" : "platform"; + + // Platform mode requires apiKey + if (mode === "platform") { + if (typeof cfg.apiKey !== "string" || !cfg.apiKey) { + throw new Error( + "apiKey is required for platform mode (set mode: \"open-source\" for self-hosted)", + ); + } + } + + // Resolve env vars in oss config + let ossConfig: Mem0Config["oss"]; + if (cfg.oss && typeof cfg.oss === "object" && !Array.isArray(cfg.oss)) { + ossConfig = resolveEnvVarsDeep( + cfg.oss as Record, + ) as unknown as Mem0Config["oss"]; + } + + return { + mode, + apiKey: + typeof cfg.apiKey === "string" ? resolveEnvVars(cfg.apiKey) : undefined, + userId: + typeof cfg.userId === "string" && cfg.userId ? cfg.userId : "default", + orgId: typeof cfg.orgId === "string" ? cfg.orgId : undefined, + projectId: typeof cfg.projectId === "string" ? cfg.projectId : undefined, + autoCapture: cfg.autoCapture !== false, + autoRecall: cfg.autoRecall !== false, + customInstructions: + typeof cfg.customInstructions === "string" + ? cfg.customInstructions + : DEFAULT_CUSTOM_INSTRUCTIONS, + customCategories: + cfg.customCategories && + typeof cfg.customCategories === "object" && + !Array.isArray(cfg.customCategories) + ? (cfg.customCategories as Record) + : DEFAULT_CUSTOM_CATEGORIES, + customPrompt: + typeof cfg.customPrompt === "string" + ? cfg.customPrompt + : DEFAULT_CUSTOM_INSTRUCTIONS, + enableGraph: cfg.enableGraph === true, + searchThreshold: + typeof cfg.searchThreshold === "number" ? cfg.searchThreshold : 0.5, + topK: typeof cfg.topK === "number" ? cfg.topK : 5, + oss: ossConfig, + }; + }, +}; diff --git a/openclaw/filtering.ts b/openclaw/filtering.ts new file mode 100644 index 000000000..654d4e93f --- /dev/null +++ b/openclaw/filtering.ts @@ -0,0 +1,115 @@ +/** + * Pre-extraction message filtering: noise detection, content stripping, + * generic assistant detection, truncation, and deduplication. + */ + +import type { MemoryItem } from "./types.ts"; + +// ============================================================================ +// Noise Detection +// ============================================================================ + +/** Patterns that indicate an entire message is noise and should be dropped. */ +const NOISE_MESSAGE_PATTERNS: RegExp[] = [ + /^(HEARTBEAT_OK|NO_REPLY)$/i, + /^Current time:.*\d{4}/, + /^Pre-compaction memory flush/i, + /^(ok|yes|no|sir|sure|thanks|done|good|nice|cool|got it|it's on|continue)$/i, + /^System: \[.*\] (Slack message edited|Gateway restart|Exec (failed|completed))/, + /^System: \[.*\] ⚠️ Post-Compaction Audit:/, +]; + +/** Content fragments that should be stripped from otherwise-valid messages. */ +const NOISE_CONTENT_PATTERNS: Array<{ pattern: RegExp; replacement: string }> = [ + { pattern: /Conversation info \(untrusted metadata\):\s*```json\s*\{[\s\S]*?\}\s*```/g, replacement: "" }, + { pattern: /\[media attached:.*?\]/g, replacement: "" }, + { pattern: /To send an image back, prefer the message tool[\s\S]*?Keep caption in the text body\./g, replacement: "" }, + { pattern: /System: \[\d{4}-\d{2}-\d{2}.*?\] ⚠️ Post-Compaction Audit:[\s\S]*?after memory compaction\./g, replacement: "" }, + { pattern: /Replied message \(untrusted, for context\):\s*```json[\s\S]*?```/g, replacement: "" }, +]; + +const MAX_MESSAGE_LENGTH = 2000; + +/** + * Patterns indicating an assistant message is a generic acknowledgment with + * no extractable facts. These are produced when the agent receives a + * transcript dump or forwarded message and responds with a boilerplate reply. + */ +const GENERIC_ASSISTANT_PATTERNS: RegExp[] = [ + /^(I see you'?ve shared|Thanks for sharing|Got it[.!]?\s*(I see|Let me|How can)|I understand[.!]?\s*(How can|Is there|Would you))/i, + /^(How can I help|Is there anything|Would you like me to|Let me know (if|how|what))/i, + /^(I('?ll| will) (help|assist|look into|review|take a look))/i, + /^(Sure[.!]?\s*(How|What|Is)|Understood[.!]?\s*(How|What|Is))/i, + /^(That('?s| is) (noted|understood|clear))/i, +]; + +// ============================================================================ +// Public Functions +// ============================================================================ + +/** + * Check whether a message's content is entirely noise (cron heartbeats, + * single-word acknowledgments, system routing metadata, etc.). + */ +export function isNoiseMessage(content: string): boolean { + const trimmed = content.trim(); + if (!trimmed) return true; + return NOISE_MESSAGE_PATTERNS.some((p) => p.test(trimmed)); +} + +/** + * Check whether an assistant message is a generic acknowledgment with no + * extractable facts (e.g. "I see you've shared an update. How can I help?"). + * Only applies to short assistant messages — longer responses likely contain + * substantive content even if they start with a generic opener. + */ +export function isGenericAssistantMessage(content: string): boolean { + const trimmed = content.trim(); + // Only flag short messages — longer ones likely have substance after the opener + if (trimmed.length > 300) return false; + return GENERIC_ASSISTANT_PATTERNS.some((p) => p.test(trimmed)); +} + +/** + * Remove embedded noise fragments (routing metadata, media boilerplate, + * compaction audit blocks) from a message while preserving the useful content. + */ +export function stripNoiseFromContent(content: string): string { + let cleaned = content; + for (const { pattern, replacement } of NOISE_CONTENT_PATTERNS) { + cleaned = cleaned.replace(pattern, replacement); + } + // Collapse excessive whitespace left behind after stripping + cleaned = cleaned.replace(/\n{3,}/g, "\n\n").trim(); + return cleaned; +} + +/** + * Truncate a message to `MAX_MESSAGE_LENGTH` characters, preserving the + * opening (which typically contains the summary/conclusion) and appending + * a truncation marker so the extraction model knows content was cut. + */ +function truncateMessage(content: string): string { + if (content.length <= MAX_MESSAGE_LENGTH) return content; + return content.slice(0, MAX_MESSAGE_LENGTH) + "\n[...truncated]"; +} + +/** + * Full pre-extraction pipeline: drop noise messages, strip noise fragments, + * and truncate remaining messages to a reasonable length. + */ +export function filterMessagesForExtraction( + messages: Array<{ role: string; content: string }>, +): Array<{ role: string; content: string }> { + const filtered: Array<{ role: string; content: string }> = []; + for (const msg of messages) { + if (isNoiseMessage(msg.content)) continue; + // Drop generic assistant acknowledgments that contain no facts + if (msg.role === "assistant" && isGenericAssistantMessage(msg.content)) continue; + const cleaned = stripNoiseFromContent(msg.content); + if (!cleaned) continue; + filtered.push({ role: msg.role, content: truncateMessage(cleaned) }); + } + return filtered; +} + diff --git a/openclaw/index.test.ts b/openclaw/index.test.ts index b5151e4ff..6c34c6f7d 100644 --- a/openclaw/index.test.ts +++ b/openclaw/index.test.ts @@ -1,8 +1,6 @@ /** - * Regression tests for per-agent memory isolation helpers. - * - * Addresses review feedback: targeted coverage for auth/session state, - * malformed input, and the resolveUserId priority chain. + * Regression tests for per-agent memory isolation helpers and + * message filtering logic. */ import { describe, it, expect } from "vitest"; import { @@ -10,16 +8,33 @@ import { effectiveUserId, agentUserId, resolveUserId, + isNonInteractiveTrigger, + isSubagentSession, + isNoiseMessage, + isGenericAssistantMessage, + stripNoiseFromContent, + filterMessagesForExtraction, } from "./index.ts"; // --------------------------------------------------------------------------- // extractAgentId // --------------------------------------------------------------------------- describe("extractAgentId", () => { - it("returns agentId from a well-formed session key", () => { + it("returns agentId from a named agent session key", () => { expect(extractAgentId("agent:researcher:550e8400-e29b")).toBe("researcher"); }); + it("returns subagent namespace from subagent session key", () => { + // OpenClaw subagent format: agent:main:subagent: + expect(extractAgentId("agent:main:subagent:3b85177f-69e0-412d-8ecd-fbe542f362ce")).toBe( + "subagent-3b85177f-69e0-412d-8ecd-fbe542f362ce", + ); + }); + + it("returns undefined for the main agent session (agent:main:main)", () => { + expect(extractAgentId("agent:main:main")).toBeUndefined(); + }); + it("returns undefined for the 'main' sentinel", () => { expect(extractAgentId("agent:main:abc-123")).toBeUndefined(); }); @@ -163,3 +178,312 @@ describe("multi-agent isolation", () => { expect(mainId).toBe(base); }); }); + +// --------------------------------------------------------------------------- +// isNonInteractiveTrigger +// --------------------------------------------------------------------------- +describe("isNonInteractiveTrigger", () => { + it("returns true for cron trigger", () => { + expect(isNonInteractiveTrigger("cron", undefined)).toBe(true); + }); + + it("returns true for heartbeat trigger", () => { + expect(isNonInteractiveTrigger("heartbeat", undefined)).toBe(true); + }); + + it("returns true for automation trigger", () => { + expect(isNonInteractiveTrigger("automation", undefined)).toBe(true); + }); + + it("returns true for schedule trigger", () => { + expect(isNonInteractiveTrigger("schedule", undefined)).toBe(true); + }); + + it("is case-insensitive for trigger", () => { + expect(isNonInteractiveTrigger("CRON", undefined)).toBe(true); + expect(isNonInteractiveTrigger("Heartbeat", undefined)).toBe(true); + }); + + it("returns false for user-initiated triggers", () => { + expect(isNonInteractiveTrigger("user", undefined)).toBe(false); + expect(isNonInteractiveTrigger("webchat", undefined)).toBe(false); + expect(isNonInteractiveTrigger("telegram", undefined)).toBe(false); + }); + + it("returns false when trigger is undefined and session key is normal", () => { + expect(isNonInteractiveTrigger(undefined, "agent:main:main")).toBe(false); + }); + + it("detects cron from session key as fallback", () => { + expect(isNonInteractiveTrigger(undefined, "agent:main:cron:c85abdb2-d900-4cd8-8601-9dd960c560c9")).toBe(true); + }); + + it("detects heartbeat from session key as fallback", () => { + expect(isNonInteractiveTrigger(undefined, "agent:main:heartbeat:abc123")).toBe(true); + }); + + it("returns false when both trigger and sessionKey are undefined", () => { + expect(isNonInteractiveTrigger(undefined, undefined)).toBe(false); + }); +}); + +// --------------------------------------------------------------------------- +// isSubagentSession +// --------------------------------------------------------------------------- +describe("isSubagentSession", () => { + it("returns true for subagent session keys", () => { + expect(isSubagentSession("agent:main:subagent:3b85177f-69e0-412d-8ecd-fbe542f362ce")).toBe(true); + }); + + it("returns false for main agent session", () => { + expect(isSubagentSession("agent:main:main")).toBe(false); + }); + + it("returns false for named agent session", () => { + expect(isSubagentSession("agent:researcher:550e8400-e29b")).toBe(false); + }); + + it("returns false for undefined", () => { + expect(isSubagentSession(undefined)).toBe(false); + }); +}); + +// --------------------------------------------------------------------------- +// isNoiseMessage +// --------------------------------------------------------------------------- +describe("isNoiseMessage", () => { + it("detects HEARTBEAT_OK", () => { + expect(isNoiseMessage("HEARTBEAT_OK")).toBe(true); + expect(isNoiseMessage("heartbeat_ok")).toBe(true); + }); + + it("detects NO_REPLY", () => { + expect(isNoiseMessage("NO_REPLY")).toBe(true); + }); + + it("detects current-time stamps", () => { + expect( + isNoiseMessage("Current time: Friday, February 20th, 2026 — 3:58 AM (America/New_York)"), + ).toBe(true); + }); + + it("detects single-word acknowledgments", () => { + for (const word of ["ok", "yes", "sir", "done", "cool", "Got it", "it's on"]) { + expect(isNoiseMessage(word)).toBe(true); + } + }); + + it("detects system routing messages", () => { + expect( + isNoiseMessage("System: [2026-02-19 19:51:31 PST] Slack message edited in #D0AFV2LDGDS."), + ).toBe(true); + expect( + isNoiseMessage("System: [2026-02-19 22:15:42 PST] Exec failed (gentle-b, signal 15)"), + ).toBe(true); + }); + + it("detects compaction audit messages", () => { + expect( + isNoiseMessage( + "System: [2026-02-20 16:12:04 EST] ⚠️ Post-Compaction Audit: The following required startup files were not read", + ), + ).toBe(true); + }); + + it("preserves real content", () => { + expect(isNoiseMessage("Beau runs Rize Digital LLC")).toBe(false); + expect(isNoiseMessage("Can you check the lovable discord?")).toBe(false); + expect(isNoiseMessage("I approve the Tailscale installation")).toBe(false); + }); + + it("treats empty/whitespace as noise", () => { + expect(isNoiseMessage("")).toBe(true); + expect(isNoiseMessage(" ")).toBe(true); + }); +}); + +// --------------------------------------------------------------------------- +// isGenericAssistantMessage +// --------------------------------------------------------------------------- +describe("isGenericAssistantMessage", () => { + it("detects 'I see you've shared' openers", () => { + expect(isGenericAssistantMessage("I see you've shared an update. How can I help?")).toBe(true); + expect(isGenericAssistantMessage("I see you've shared a summary of the Atlas configuration update. Is there anything specific you'd like me to help with?")).toBe(true); + }); + + it("detects 'Thanks for sharing' openers", () => { + expect(isGenericAssistantMessage("Thanks for sharing that update! Would you like me to review the changes?")).toBe(true); + }); + + it("detects 'How can I help' standalone", () => { + expect(isGenericAssistantMessage("How can I help you with this?")).toBe(true); + }); + + it("detects 'Got it' + follow-up", () => { + expect(isGenericAssistantMessage("Got it! How can I assist?")).toBe(true); + expect(isGenericAssistantMessage("Got it. Let me know what you need.")).toBe(true); + }); + + it("detects 'I'll help/review/look into'", () => { + expect(isGenericAssistantMessage("I'll review that for you.")).toBe(true); + expect(isGenericAssistantMessage("I'll look into this right away.")).toBe(true); + }); + + it("preserves substantive assistant content", () => { + expect(isGenericAssistantMessage("## What I Accomplished\n\nDeployed the API to production with Vercel.")).toBe(false); + expect(isGenericAssistantMessage("The ElevenLabs SDK has been installed and configured. Voice skill is ready.")).toBe(false); + expect(isGenericAssistantMessage("Updated the call scripts sheet with truth-based messaging templates.")).toBe(false); + }); + + it("preserves long messages even with generic openers", () => { + const longMsg = "I see you've shared an update. " + "Here are the detailed changes I made to the configuration. ".repeat(10); + expect(isGenericAssistantMessage(longMsg)).toBe(false); + }); +}); + +// --------------------------------------------------------------------------- +// stripNoiseFromContent +// --------------------------------------------------------------------------- +describe("stripNoiseFromContent", () => { + it("removes conversation metadata JSON blocks", () => { + const input = `Conversation info (untrusted metadata): +\`\`\`json +{ + "message_id": "499", + "sender": "6039555582" +} +\`\`\` + +What models are you currently using?`; + const result = stripNoiseFromContent(input); + expect(result).toBe("What models are you currently using?"); + }); + + it("removes media attachment lines", () => { + const input = "[media attached: /path/to/file.jpg (image/jpeg) | /path/to/file.jpg]\nActual question here"; + const result = stripNoiseFromContent(input); + expect(result).toContain("Actual question here"); + expect(result).not.toContain("[media attached:"); + }); + + it("removes image sending boilerplate", () => { + const input = + "To send an image back, prefer the message tool (media/path/filePath). If you must inline, use MEDIA:https://example.com/image.jpg. Keep caption in the text body.\nReal content here"; + const result = stripNoiseFromContent(input); + expect(result).toContain("Real content here"); + expect(result).not.toContain("prefer the message tool"); + }); + + it("preserves content when no noise is present", () => { + const input = "User wants to deploy to production via Vercel."; + expect(stripNoiseFromContent(input)).toBe(input); + }); + + it("collapses excessive blank lines after stripping", () => { + const input = "Line one\n\n\n\n\nLine two"; + expect(stripNoiseFromContent(input)).toBe("Line one\n\nLine two"); + }); +}); + +// --------------------------------------------------------------------------- +// filterMessagesForExtraction +// --------------------------------------------------------------------------- +describe("filterMessagesForExtraction", () => { + it("drops noise messages entirely", () => { + const messages = [ + { role: "user", content: "HEARTBEAT_OK" }, + { role: "assistant", content: "Real response with durable facts." }, + { role: "user", content: "ok" }, + ]; + const result = filterMessagesForExtraction(messages); + expect(result).toHaveLength(1); + expect(result[0].content).toBe("Real response with durable facts."); + }); + + it("strips noise fragments but keeps the rest", () => { + const messages = [ + { + role: "user", + content: `Conversation info (untrusted metadata): +\`\`\`json +{ + "message_id": "123", + "sender": "456" +} +\`\`\` + +What is the deployment plan?`, + }, + ]; + const result = filterMessagesForExtraction(messages); + expect(result).toHaveLength(1); + expect(result[0].content).toBe("What is the deployment plan?"); + }); + + it("truncates long messages", () => { + const longContent = "A".repeat(3000); + const messages = [{ role: "assistant", content: longContent }]; + const result = filterMessagesForExtraction(messages); + expect(result).toHaveLength(1); + expect(result[0].content.length).toBeLessThan(2100); + expect(result[0].content).toContain("[...truncated]"); + }); + + it("returns empty array when all messages are noise", () => { + const messages = [ + { role: "user", content: "NO_REPLY" }, + { role: "user", content: "ok" }, + { role: "user", content: "Current time: Friday, February 20th, 2026" }, + ]; + expect(filterMessagesForExtraction(messages)).toHaveLength(0); + }); + + it("handles a realistic mixed payload", () => { + const messages = [ + { role: "user", content: "Pre-compaction memory flush. Store durable memories now." }, + { + role: "assistant", + content: "## What I Accomplished\n\nDeployed the API to production with Vercel.", + }, + { role: "user", content: "sir" }, + ]; + const result = filterMessagesForExtraction(messages); + expect(result).toHaveLength(1); + expect(result[0].content).toContain("Deployed the API"); + }); + + it("drops generic assistant acknowledgments", () => { + const messages = [ + { role: "user", content: "[ASSISTANT]: Updated the Google Sheet with truth-based scripts." }, + { role: "assistant", content: "I see you've shared an update. How can I help?" }, + ]; + const result = filterMessagesForExtraction(messages); + expect(result).toHaveLength(1); + expect(result[0].role).toBe("user"); + expect(result[0].content).toContain("Google Sheet"); + }); + + it("returns only assistant messages when all user messages are noise", () => { + // This scenario triggers the #2 guard: no user content remains + const messages = [ + { role: "user", content: "ok" }, + { role: "user", content: "HEARTBEAT_OK" }, + { role: "assistant", content: "I deployed the API to production." }, + ]; + const result = filterMessagesForExtraction(messages); + expect(result).toHaveLength(1); + expect(result[0].role).toBe("assistant"); + // The capture hook checks: if no user messages remain, skip add() + expect(result.some((m) => m.role === "user")).toBe(false); + }); + + it("keeps substantive assistant messages even with generic opener", () => { + const messages = [ + { role: "user", content: "What did you do?" }, + { role: "assistant", content: "I deployed the API to production and configured the webhook endpoints for Stripe integration." }, + ]; + const result = filterMessagesForExtraction(messages); + expect(result).toHaveLength(2); + }); +}); + diff --git a/openclaw/index.ts b/openclaw/index.ts index c49d53b58..8386dd77e 100644 --- a/openclaw/index.ts +++ b/openclaw/index.ts @@ -19,614 +19,39 @@ import { Type } from "@sinclair/typebox"; import type { OpenClawPluginApi } from "openclaw/plugin-sdk"; -// ============================================================================ -// Types -// ============================================================================ - -type Mem0Mode = "platform" | "open-source"; - -type Mem0Config = { - mode: Mem0Mode; - // Platform-specific - apiKey?: string; - orgId?: string; - projectId?: string; - customInstructions: string; - customCategories: Record; - enableGraph: boolean; - // OSS-specific - customPrompt?: string; - oss?: { - embedder?: { provider: string; config: Record }; - vectorStore?: { provider: string; config: Record }; - llm?: { provider: string; config: Record }; - historyDbPath?: string; - disableHistory?: boolean; - }; - // Shared - userId: string; - autoCapture: boolean; - autoRecall: boolean; - searchThreshold: number; - topK: number; -}; - -// Unified types for the provider interface -interface AddOptions { - user_id: string; - run_id?: string; - custom_instructions?: string; - custom_categories?: Array>; - enable_graph?: boolean; - output_format?: string; - source?: string; -} - -interface SearchOptions { - user_id: string; - run_id?: string; - top_k?: number; - threshold?: number; - limit?: number; - keyword_search?: boolean; - reranking?: boolean; - source?: string; -} - -interface ListOptions { - user_id: string; - run_id?: string; - page_size?: number; - source?: string; -} - -interface MemoryItem { - id: string; - memory: string; - user_id?: string; - score?: number; - categories?: string[]; - metadata?: Record; - created_at?: string; - updated_at?: string; -} - -interface AddResultItem { - id: string; - memory: string; - event: "ADD" | "UPDATE" | "DELETE" | "NOOP"; -} - -interface AddResult { - results: AddResultItem[]; -} +import type { + Mem0Config, + Mem0Provider, + MemoryItem, + AddOptions, + SearchOptions, +} from "./types.ts"; +import { createProvider } from "./providers.ts"; +import { mem0ConfigSchema } from "./config.ts"; +import { + filterMessagesForExtraction, +} from "./filtering.ts"; +import { + effectiveUserId, + agentUserId, + resolveUserId, + isNonInteractiveTrigger, + isSubagentSession, +} from "./isolation.ts"; // ============================================================================ -// Unified Provider Interface +// Re-exports (for tests and external consumers) // ============================================================================ -interface Mem0Provider { - add( - messages: Array<{ role: string; content: string }>, - options: AddOptions, - ): Promise; - search(query: string, options: SearchOptions): Promise; - get(memoryId: string): Promise; - getAll(options: ListOptions): Promise; - delete(memoryId: string): Promise; -} - -// ============================================================================ -// Platform Provider (Mem0 Cloud) -// ============================================================================ - -class PlatformProvider implements Mem0Provider { - private client: any; // MemoryClient from mem0ai - private initPromise: Promise | null = null; - - constructor( - private readonly apiKey: string, - private readonly orgId?: string, - private readonly projectId?: string, - ) { } - - private async ensureClient(): Promise { - if (this.client) return; - if (this.initPromise) return this.initPromise; - this.initPromise = this._init().catch((err) => { - this.initPromise = null; - throw err; - }); - return this.initPromise; - } - - private async _init(): Promise { - const { default: MemoryClient } = await import("mem0ai"); - const opts: { apiKey: string; org_id?: string; project_id?: string } = { apiKey: this.apiKey }; - if (this.orgId) opts.org_id = this.orgId; - if (this.projectId) opts.project_id = this.projectId; - this.client = new MemoryClient(opts); - } - - async add( - messages: Array<{ role: string; content: string }>, - options: AddOptions, - ): Promise { - await this.ensureClient(); - const opts: Record = { user_id: options.user_id }; - if (options.run_id) opts.run_id = options.run_id; - if (options.custom_instructions) - opts.custom_instructions = options.custom_instructions; - if (options.custom_categories) - opts.custom_categories = options.custom_categories; - if (options.enable_graph) opts.enable_graph = options.enable_graph; - if (options.output_format) opts.output_format = options.output_format; - if (options.source) opts.source = options.source; - - const result = await this.client.add(messages, opts); - return normalizeAddResult(result); - } - - async search(query: string, options: SearchOptions): Promise { - await this.ensureClient(); - const filters: Record = { user_id: options.user_id }; - if (options.run_id) filters.run_id = options.run_id; - - const opts: Record = { - api_version: "v2", - filters, - }; - if (options.top_k != null) opts.top_k = options.top_k; - if (options.threshold != null) opts.threshold = options.threshold; - if (options.keyword_search != null) opts.keyword_search = options.keyword_search; - if (options.reranking != null) opts.rerank = options.reranking; - - const results = await this.client.search(query, opts); - return normalizeSearchResults(results); - } - - async get(memoryId: string): Promise { - await this.ensureClient(); - const result = await this.client.get(memoryId); - return normalizeMemoryItem(result); - } - - async getAll(options: ListOptions): Promise { - await this.ensureClient(); - const opts: Record = { user_id: options.user_id }; - if (options.run_id) opts.run_id = options.run_id; - if (options.page_size != null) opts.page_size = options.page_size; - if (options.source) opts.source = options.source; - - const results = await this.client.getAll(opts); - if (Array.isArray(results)) return results.map(normalizeMemoryItem); - // Some versions return { results: [...] } - if (results?.results && Array.isArray(results.results)) - return results.results.map(normalizeMemoryItem); - return []; - } - - async delete(memoryId: string): Promise { - await this.ensureClient(); - await this.client.delete(memoryId); - } -} - -// ============================================================================ -// Open-Source Provider (Self-hosted) -// ============================================================================ - -class OSSProvider implements Mem0Provider { - private memory: any; // Memory from mem0ai/oss - private initPromise: Promise | null = null; - - constructor( - private readonly ossConfig?: Mem0Config["oss"], - private readonly customPrompt?: string, - private readonly resolvePath?: (p: string) => string, - ) { } - - private async ensureMemory(): Promise { - if (this.memory) return; - if (this.initPromise) return this.initPromise; - this.initPromise = this._init().catch((err) => { - this.initPromise = null; - throw err; - }); - return this.initPromise; - } - - private async _init(): Promise { - const { Memory } = await import("mem0ai/oss"); - - const config: Record = { version: "v1.1" }; - - if (this.ossConfig?.embedder) config.embedder = this.ossConfig.embedder; - if (this.ossConfig?.vectorStore) - config.vectorStore = this.ossConfig.vectorStore; - if (this.ossConfig?.llm) config.llm = this.ossConfig.llm; - - if (this.ossConfig?.historyDbPath) { - const dbPath = this.resolvePath - ? this.resolvePath(this.ossConfig.historyDbPath) - : this.ossConfig.historyDbPath; - config.historyDbPath = dbPath; - } - - if (this.ossConfig?.disableHistory) { - config.disableHistory = true; - } - - if (this.customPrompt) config.customPrompt = this.customPrompt; - - try { - this.memory = new Memory(config); - } catch (err) { - // If initialization fails (e.g. native SQLite binding resolution under - // jiti), retry with history disabled — the history DB is the most common - // source of native-binding failures and is not required for core - // memory operations. - if (!config.disableHistory) { - console.warn( - "[mem0] Memory initialization failed, retrying with history disabled:", - err instanceof Error ? err.message : err, - ); - config.disableHistory = true; - this.memory = new Memory(config); - } else { - throw err; - } - } - } - - async add( - messages: Array<{ role: string; content: string }>, - options: AddOptions, - ): Promise { - await this.ensureMemory(); - // OSS SDK uses camelCase: userId/runId, not user_id/run_id - const addOpts: Record = { userId: options.user_id }; - if (options.run_id) addOpts.runId = options.run_id; - if (options.source) addOpts.source = options.source; - const result = await this.memory.add(messages, addOpts); - return normalizeAddResult(result); - } - - async search(query: string, options: SearchOptions): Promise { - await this.ensureMemory(); - // OSS SDK uses camelCase: userId/runId, not user_id/run_id - const opts: Record = { userId: options.user_id }; - if (options.run_id) opts.runId = options.run_id; - if (options.limit != null) opts.limit = options.limit; - else if (options.top_k != null) opts.limit = options.top_k; - if (options.keyword_search != null) opts.keyword_search = options.keyword_search; - if (options.reranking != null) opts.reranking = options.reranking; - if (options.source) opts.source = options.source; - if (options.threshold != null) opts.threshold = options.threshold; - - const results = await this.memory.search(query, opts); - const normalized = normalizeSearchResults(results); - - // Filter results by threshold if specified (client-side filtering as fallback) - if (options.threshold != null) { - return normalized.filter(item => (item.score ?? 0) >= options.threshold!); - } - - return normalized; - } - - async get(memoryId: string): Promise { - await this.ensureMemory(); - const result = await this.memory.get(memoryId); - return normalizeMemoryItem(result); - } - - async getAll(options: ListOptions): Promise { - await this.ensureMemory(); - // OSS SDK uses camelCase: userId/runId, not user_id/run_id - const getAllOpts: Record = { userId: options.user_id }; - if (options.run_id) getAllOpts.runId = options.run_id; - if (options.source) getAllOpts.source = options.source; - const results = await this.memory.getAll(getAllOpts); - if (Array.isArray(results)) return results.map(normalizeMemoryItem); - if (results?.results && Array.isArray(results.results)) - return results.results.map(normalizeMemoryItem); - return []; - } - - async delete(memoryId: string): Promise { - await this.ensureMemory(); - await this.memory.delete(memoryId); - } -} - -// ============================================================================ -// Result Normalizers -// ============================================================================ - -function normalizeMemoryItem(raw: any): MemoryItem { - return { - id: raw.id ?? raw.memory_id ?? "", - memory: raw.memory ?? raw.text ?? raw.content ?? "", - // Handle both platform (user_id, created_at) and OSS (userId, createdAt) field names - user_id: raw.user_id ?? raw.userId, - score: raw.score, - categories: raw.categories, - metadata: raw.metadata, - created_at: raw.created_at ?? raw.createdAt, - updated_at: raw.updated_at ?? raw.updatedAt, - }; -} - -function normalizeSearchResults(raw: any): MemoryItem[] { - // Platform API returns flat array, OSS returns { results: [...] } - if (Array.isArray(raw)) return raw.map(normalizeMemoryItem); - if (raw?.results && Array.isArray(raw.results)) - return raw.results.map(normalizeMemoryItem); - return []; -} - -function normalizeAddResult(raw: any): AddResult { - // Handle { results: [...] } shape (both platform and OSS) - if (raw?.results && Array.isArray(raw.results)) { - return { - results: raw.results.map((r: any) => ({ - id: r.id ?? r.memory_id ?? "", - memory: r.memory ?? r.text ?? "", - // Platform API may return PENDING status (async processing) - // OSS stores event in metadata.event - event: r.event ?? r.metadata?.event ?? (r.status === "PENDING" ? "ADD" : "ADD"), - })), - }; - } - // Platform API without output_format returns flat array - if (Array.isArray(raw)) { - return { - results: raw.map((r: any) => ({ - id: r.id ?? r.memory_id ?? "", - memory: r.memory ?? r.text ?? "", - event: r.event ?? r.metadata?.event ?? (r.status === "PENDING" ? "ADD" : "ADD"), - })), - }; - } - return { results: [] }; -} - -// ============================================================================ -// Config Parser -// ============================================================================ - -function resolveEnvVars(value: string): string { - return value.replace(/\$\{([^}]+)\}/g, (_, envVar) => { - const envValue = process.env[envVar]; - if (!envValue) { - throw new Error(`Environment variable ${envVar} is not set`); - } - return envValue; - }); -} - -function resolveEnvVarsDeep(obj: Record): Record { - const result: Record = {}; - for (const [key, value] of Object.entries(obj)) { - if (typeof value === "string") { - result[key] = resolveEnvVars(value); - } else if (value && typeof value === "object" && !Array.isArray(value)) { - result[key] = resolveEnvVarsDeep(value as Record); - } else { - result[key] = value; - } - } - return result; -} - -// ============================================================================ -// Default Custom Instructions & Categories -// ============================================================================ - -const DEFAULT_CUSTOM_INSTRUCTIONS = `Your Task: Extract and maintain a structured, evolving profile of the user from their conversations with an AI assistant. Capture information that would help the assistant provide personalized, context-aware responses in future interactions. - -Information to Extract: - -1. Identity & Demographics: - - Name, age, location, timezone, language preferences - - Occupation, employer, job role, industry - - Education background - -2. Preferences & Opinions: - - Communication style preferences (formal/casual, verbose/concise) - - Tool and technology preferences (languages, frameworks, editors, OS) - - Content preferences (topics of interest, learning style) - - Strong opinions or values they've expressed - - Likes and dislikes they've explicitly stated - -3. Goals & Projects: - - Current projects they're working on (name, description, status) - - Short-term and long-term goals - - Deadlines and milestones mentioned - - Problems they're actively trying to solve - -4. Technical Context: - - Tech stack and tools they use - - Skill level in different areas (beginner/intermediate/expert) - - Development environment and setup details - - Recurring technical challenges - -5. Relationships & People: - - Names and roles of people they mention (colleagues, family, friends) - - Team structure and dynamics - - Key contacts and their relevance - -6. Decisions & Lessons: - - Important decisions made and their reasoning - - Lessons learned from past experiences - - Strategies that worked or failed - - Changed opinions or updated beliefs - -7. Routines & Habits: - - Daily routines and schedules mentioned - - Work patterns (when they're productive, how they organize work) - - Health and wellness habits if voluntarily shared - -8. Life Events: - - Significant events (new job, moving, milestones) - - Upcoming events or plans - - Changes in circumstances - -Guidelines: -- Store memories as clear, self-contained statements (each memory should make sense on its own) -- Use third person: "User prefers..." not "I prefer..." -- Include temporal context when relevant: "As of [date], user is working on..." -- When information updates, UPDATE the existing memory rather than creating duplicates -- Merge related facts into single coherent memories when possible -- Preserve specificity: "User uses Next.js 14 with App Router" is better than "User uses React" -- Capture the WHY behind preferences when stated: "User prefers Vim because of keyboard-driven workflow" - -Exclude: -- Passwords, API keys, tokens, or any authentication credentials -- Exact financial amounts (account balances, salaries) unless the user explicitly asks to remember them -- Temporary or ephemeral information (one-time questions, debugging sessions with no lasting insight) -- Generic small talk with no informational content -- The assistant's own responses unless they contain a commitment or promise to the user -- Raw code snippets (capture the intent/decision, not the code itself) -- Information the user explicitly asks not to remember`; - -const DEFAULT_CUSTOM_CATEGORIES: Record = { - identity: - "Personal identity information: name, age, location, timezone, occupation, employer, education, demographics", - preferences: - "Explicitly stated likes, dislikes, preferences, opinions, and values across any domain", - goals: - "Current and future goals, aspirations, objectives, targets the user is working toward", - projects: - "Specific projects, initiatives, or endeavors the user is working on, including status and details", - technical: - "Technical skills, tools, tech stack, development environment, programming languages, frameworks", - decisions: - "Important decisions made, reasoning behind choices, strategy changes, and their outcomes", - relationships: - "People mentioned by the user: colleagues, family, friends, their roles and relevance", - routines: - "Daily habits, work patterns, schedules, productivity routines, health and wellness habits", - life_events: - "Significant life events, milestones, transitions, upcoming plans and changes", - lessons: - "Lessons learned, insights gained, mistakes acknowledged, changed opinions or beliefs", - work: - "Work-related context: job responsibilities, workplace dynamics, career progression, professional challenges", - health: - "Health-related information voluntarily shared: conditions, medications, fitness, wellness goals", -}; - -// ============================================================================ -// Config Schema -// ============================================================================ - -const ALLOWED_KEYS = [ - "mode", - "apiKey", - "userId", - "orgId", - "projectId", - "autoCapture", - "autoRecall", - "customInstructions", - "customCategories", - "customPrompt", - "enableGraph", - "searchThreshold", - "topK", - "oss", -]; - -function assertAllowedKeys( - value: Record, - allowed: string[], - label: string, -) { - const unknown = Object.keys(value).filter((key) => !allowed.includes(key)); - if (unknown.length === 0) return; - throw new Error(`${label} has unknown keys: ${unknown.join(", ")}`); -} - -export const mem0ConfigSchema = { - parse(value: unknown): Mem0Config { - if (!value || typeof value !== "object" || Array.isArray(value)) { - throw new Error("openclaw-mem0 config required"); - } - const cfg = value as Record; - assertAllowedKeys(cfg, ALLOWED_KEYS, "openclaw-mem0 config"); - - // Accept both "open-source" and legacy "oss" as open-source mode; everything else is platform - const mode: Mem0Mode = - cfg.mode === "oss" || cfg.mode === "open-source" ? "open-source" : "platform"; - - // Platform mode requires apiKey - if (mode === "platform") { - if (typeof cfg.apiKey !== "string" || !cfg.apiKey) { - throw new Error( - "apiKey is required for platform mode (set mode: \"open-source\" for self-hosted)", - ); - } - } - - // Resolve env vars in oss config - let ossConfig: Mem0Config["oss"]; - if (cfg.oss && typeof cfg.oss === "object" && !Array.isArray(cfg.oss)) { - ossConfig = resolveEnvVarsDeep( - cfg.oss as Record, - ) as unknown as Mem0Config["oss"]; - } - - return { - mode, - apiKey: - typeof cfg.apiKey === "string" ? resolveEnvVars(cfg.apiKey) : undefined, - userId: - typeof cfg.userId === "string" && cfg.userId ? cfg.userId : "default", - orgId: typeof cfg.orgId === "string" ? cfg.orgId : undefined, - projectId: typeof cfg.projectId === "string" ? cfg.projectId : undefined, - autoCapture: cfg.autoCapture !== false, - autoRecall: cfg.autoRecall !== false, - customInstructions: - typeof cfg.customInstructions === "string" - ? cfg.customInstructions - : DEFAULT_CUSTOM_INSTRUCTIONS, - customCategories: - cfg.customCategories && - typeof cfg.customCategories === "object" && - !Array.isArray(cfg.customCategories) - ? (cfg.customCategories as Record) - : DEFAULT_CUSTOM_CATEGORIES, - customPrompt: - typeof cfg.customPrompt === "string" - ? cfg.customPrompt - : DEFAULT_CUSTOM_INSTRUCTIONS, - enableGraph: cfg.enableGraph === true, - searchThreshold: - typeof cfg.searchThreshold === "number" ? cfg.searchThreshold : 0.5, - topK: typeof cfg.topK === "number" ? cfg.topK : 5, - oss: ossConfig, - }; - }, -}; - -// ============================================================================ -// Provider Factory -// ============================================================================ - -export function createProvider( - cfg: Mem0Config, - api: OpenClawPluginApi, -): Mem0Provider { - if (cfg.mode === "open-source") { - return new OSSProvider(cfg.oss, cfg.customPrompt, (p) => - api.resolvePath(p), - ); - } - - return new PlatformProvider(cfg.apiKey!, cfg.orgId, cfg.projectId); -} +export { extractAgentId, effectiveUserId, agentUserId, resolveUserId, isNonInteractiveTrigger, isSubagentSession } from "./isolation.ts"; +export { + isNoiseMessage, + isGenericAssistantMessage, + stripNoiseFromContent, + filterMessagesForExtraction, +} from "./filtering.ts"; +export { mem0ConfigSchema } from "./config.ts"; +export { createProvider } from "./providers.ts"; // ============================================================================ // Helpers @@ -639,50 +64,6 @@ function categoriesToArray( return Object.entries(cats).map(([key, value]) => ({ [key]: value })); } -// ============================================================================ -// Per-agent isolation helpers (exported for testability) -// ============================================================================ - -/** - * Parse an agent ID from a session key following the pattern `agent::`. - * Returns undefined for non-agent sessions, the "main" sentinel, or malformed keys. - */ -export function extractAgentId(sessionKey: string | undefined): string | undefined { - if (!sessionKey) return undefined; - const match = sessionKey.match(/^agent:([^:]+):/); - const agentId = match?.[1]; - // "main" is the primary session — fall back to configured userId - if (!agentId || agentId === "main") return undefined; - return agentId; -} - -/** - * Derive the effective user_id from a session key, namespacing per-agent. - * Falls back to baseUserId when the session is not agent-scoped. - */ -export function effectiveUserId(baseUserId: string, sessionKey?: string): string { - const agentId = extractAgentId(sessionKey); - return agentId ? `${baseUserId}:agent:${agentId}` : baseUserId; -} - -/** Build a user_id for an explicit agentId (e.g. from tool params). */ -export function agentUserId(baseUserId: string, agentId: string): string { - return `${baseUserId}:agent:${agentId}`; -} - -/** - * Resolve user_id with priority: explicit agentId > explicit userId > session-derived > configured. - */ -export function resolveUserId( - baseUserId: string, - opts: { agentId?: string; userId?: string }, - currentSessionId?: string, -): string { - if (opts.agentId) return agentUserId(baseUserId, opts.agentId); - if (opts.userId) return opts.userId; - return effectiveUserId(baseUserId, currentSessionId); -} - // ============================================================================ // Plugin Definition // ============================================================================ @@ -699,7 +80,10 @@ const memoryPlugin = { const cfg = mem0ConfigSchema.parse(api.pluginConfig); const provider = createProvider(cfg, api); - // Track current session ID for tool-level session scoping + // Track current session ID for tool-level session scoping. + // NOTE: This is shared mutable state — tools don't receive ctx, so they + // read this as a best-effort fallback. Hooks should use ctx.sessionKey + // directly and avoid relying on this variable. let currentSessionId: string | undefined; // ======================================================================== @@ -755,755 +139,21 @@ const memoryPlugin = { // Tools // ======================================================================== - api.registerTool( - { - name: "memory_search", - label: "Memory Search", - description: - "Search through long-term memories stored in Mem0. Use when you need context about user preferences, past decisions, or previously discussed topics.", - parameters: Type.Object({ - query: Type.String({ description: "Search query" }), - limit: Type.Optional( - Type.Number({ - description: `Max results (default: ${cfg.topK})`, - }), - ), - userId: Type.Optional( - Type.String({ - description: - "User ID to scope search (default: configured userId)", - }), - ), - agentId: Type.Optional( - Type.String({ - description: - "Agent ID to search memories for a specific agent (e.g. \"researcher\"). Overrides userId.", - }), - ), - scope: Type.Optional( - Type.Union([ - Type.Literal("session"), - Type.Literal("long-term"), - Type.Literal("all"), - ], { - description: - 'Memory scope: "session" (current session only), "long-term" (user-scoped only), or "all" (both). Default: "all"', - }), - ), - }), - async execute(_toolCallId, params) { - const { query, limit, userId, agentId, scope = "all" } = params as { - query: string; - limit?: number; - userId?: string; - agentId?: string; - scope?: "session" | "long-term" | "all"; - }; - - try { - let results: MemoryItem[] = []; - const uid = _resolveUserId({ agentId, userId }); - - if (scope === "session") { - if (currentSessionId) { - results = await provider.search( - query, - buildSearchOptions(uid, limit, currentSessionId), - ); - } - } else if (scope === "long-term") { - results = await provider.search( - query, - buildSearchOptions(uid, limit), - ); - } else { - // "all" — search both scopes and combine - const longTermResults = await provider.search( - query, - buildSearchOptions(uid, limit), - ); - let sessionResults: MemoryItem[] = []; - if (currentSessionId) { - sessionResults = await provider.search( - query, - buildSearchOptions(uid, limit, currentSessionId), - ); - } - // Deduplicate by ID, preferring long-term - const seen = new Set(longTermResults.map((r) => r.id)); - results = [ - ...longTermResults, - ...sessionResults.filter((r) => !seen.has(r.id)), - ]; - } - - if (!results || results.length === 0) { - return { - content: [ - { type: "text", text: "No relevant memories found." }, - ], - details: { count: 0 }, - }; - } - - const text = results - .map( - (r, i) => - `${i + 1}. ${r.memory} (score: ${((r.score ?? 0) * 100).toFixed(0)}%, id: ${r.id})`, - ) - .join("\n"); - - const sanitized = results.map((r) => ({ - id: r.id, - memory: r.memory, - score: r.score, - categories: r.categories, - created_at: r.created_at, - })); - - return { - content: [ - { - type: "text", - text: `Found ${results.length} memories:\n\n${text}`, - }, - ], - details: { count: results.length, memories: sanitized }, - }; - } catch (err) { - return { - content: [ - { - type: "text", - text: `Memory search failed: ${String(err)}`, - }, - ], - details: { error: String(err) }, - }; - } - }, - }, - { name: "memory_search" }, - ); - - api.registerTool( - { - name: "memory_store", - label: "Memory Store", - description: - "Save important information in long-term memory via Mem0. Use for preferences, facts, decisions, and anything worth remembering.", - parameters: Type.Object({ - text: Type.String({ description: "Information to remember" }), - userId: Type.Optional( - Type.String({ - description: "User ID to scope this memory", - }), - ), - agentId: Type.Optional( - Type.String({ - description: - "Agent ID to store memory under a specific agent's namespace (e.g. \"researcher\"). Overrides userId.", - }), - ), - metadata: Type.Optional( - Type.Record(Type.String(), Type.Unknown(), { - description: "Optional metadata to attach to this memory", - }), - ), - longTerm: Type.Optional( - Type.Boolean({ - description: - "Store as long-term (user-scoped) memory. Default: true. Set to false for session-scoped memory.", - }), - ), - }), - async execute(_toolCallId, params) { - const { text, userId, agentId, longTerm = true } = params as { - text: string; - userId?: string; - agentId?: string; - metadata?: Record; - longTerm?: boolean; - }; - - try { - const uid = _resolveUserId({ agentId, userId }); - const runId = !longTerm && currentSessionId ? currentSessionId : undefined; - const result = await provider.add( - [{ role: "user", content: text }], - buildAddOptions(uid, runId, currentSessionId), - ); - - const added = - result.results?.filter((r) => r.event === "ADD") ?? []; - const updated = - result.results?.filter((r) => r.event === "UPDATE") ?? []; - - const summary = []; - if (added.length > 0) - summary.push( - `${added.length} new memor${added.length === 1 ? "y" : "ies"} added`, - ); - if (updated.length > 0) - summary.push( - `${updated.length} memor${updated.length === 1 ? "y" : "ies"} updated`, - ); - if (summary.length === 0) - summary.push("No new memories extracted"); - - return { - content: [ - { - type: "text", - text: `Stored: ${summary.join(", ")}. ${result.results?.map((r) => `[${r.event}] ${r.memory}`).join("; ") ?? ""}`, - }, - ], - details: { - action: "stored", - results: result.results, - }, - }; - } catch (err) { - return { - content: [ - { - type: "text", - text: `Memory store failed: ${String(err)}`, - }, - ], - details: { error: String(err) }, - }; - } - }, - }, - { name: "memory_store" }, - ); - - api.registerTool( - { - name: "memory_get", - label: "Memory Get", - description: "Retrieve a specific memory by its ID from Mem0.", - parameters: Type.Object({ - memoryId: Type.String({ description: "The memory ID to retrieve" }), - }), - async execute(_toolCallId, params) { - const { memoryId } = params as { memoryId: string }; - - try { - const memory = await provider.get(memoryId); - - return { - content: [ - { - type: "text", - text: `Memory ${memory.id}:\n${memory.memory}\n\nCreated: ${memory.created_at ?? "unknown"}\nUpdated: ${memory.updated_at ?? "unknown"}`, - }, - ], - details: { memory }, - }; - } catch (err) { - return { - content: [ - { - type: "text", - text: `Memory get failed: ${String(err)}`, - }, - ], - details: { error: String(err) }, - }; - } - }, - }, - { name: "memory_get" }, - ); - - api.registerTool( - { - name: "memory_list", - label: "Memory List", - description: - "List all stored memories for a user or agent. Use this when you want to see everything that's been remembered, rather than searching for something specific.", - parameters: Type.Object({ - userId: Type.Optional( - Type.String({ - description: - "User ID to list memories for (default: configured userId)", - }), - ), - agentId: Type.Optional( - Type.String({ - description: - "Agent ID to list memories for a specific agent (e.g. \"researcher\"). Overrides userId.", - }), - ), - scope: Type.Optional( - Type.Union([ - Type.Literal("session"), - Type.Literal("long-term"), - Type.Literal("all"), - ], { - description: - 'Memory scope: "session" (current session only), "long-term" (user-scoped only), or "all" (both). Default: "all"', - }), - ), - }), - async execute(_toolCallId, params) { - const { userId, agentId, scope = "all" } = params as { userId?: string; agentId?: string; scope?: "session" | "long-term" | "all" }; - - try { - let memories: MemoryItem[] = []; - const uid = _resolveUserId({ agentId, userId }); - - if (scope === "session") { - if (currentSessionId) { - memories = await provider.getAll({ - user_id: uid, - run_id: currentSessionId, - source: "OPENCLAW", - }); - } - } else if (scope === "long-term") { - memories = await provider.getAll({ user_id: uid, source: "OPENCLAW" }); - } else { - // "all" — combine both scopes - const longTerm = await provider.getAll({ user_id: uid, source: "OPENCLAW" }); - let session: MemoryItem[] = []; - if (currentSessionId) { - session = await provider.getAll({ - user_id: uid, - run_id: currentSessionId, - source: "OPENCLAW", - }); - } - const seen = new Set(longTerm.map((r) => r.id)); - memories = [ - ...longTerm, - ...session.filter((r) => !seen.has(r.id)), - ]; - } - - if (!memories || memories.length === 0) { - return { - content: [ - { type: "text", text: "No memories stored yet." }, - ], - details: { count: 0 }, - }; - } - - const text = memories - .map( - (r, i) => - `${i + 1}. ${r.memory} (id: ${r.id})`, - ) - .join("\n"); - - const sanitized = memories.map((r) => ({ - id: r.id, - memory: r.memory, - categories: r.categories, - created_at: r.created_at, - })); - - return { - content: [ - { - type: "text", - text: `${memories.length} memories:\n\n${text}`, - }, - ], - details: { count: memories.length, memories: sanitized }, - }; - } catch (err) { - return { - content: [ - { - type: "text", - text: `Memory list failed: ${String(err)}`, - }, - ], - details: { error: String(err) }, - }; - } - }, - }, - { name: "memory_list" }, - ); - - api.registerTool( - { - name: "memory_forget", - label: "Memory Forget", - description: - "Delete memories from Mem0. Provide a specific memoryId to delete directly, or a query to search and delete matching memories. Supports agent-scoped deletion. GDPR-compliant.", - parameters: Type.Object({ - query: Type.Optional( - Type.String({ - description: "Search query to find memory to delete", - }), - ), - memoryId: Type.Optional( - Type.String({ description: "Specific memory ID to delete" }), - ), - agentId: Type.Optional( - Type.String({ - description: - "Agent ID to scope deletion to a specific agent's memories (e.g. \"researcher\").", - }), - ), - }), - async execute(_toolCallId, params) { - const { query, memoryId, agentId } = params as { - query?: string; - memoryId?: string; - agentId?: string; - }; - - try { - if (memoryId) { - await provider.delete(memoryId); - return { - content: [ - { type: "text", text: `Memory ${memoryId} forgotten.` }, - ], - details: { action: "deleted", id: memoryId }, - }; - } - - if (query) { - const uid = _resolveUserId({ agentId }); - const results = await provider.search( - query, - buildSearchOptions(uid, 5), - ); - - if (!results || results.length === 0) { - return { - content: [ - { type: "text", text: "No matching memories found." }, - ], - details: { found: 0 }, - }; - } - - // If single high-confidence match, delete directly - if ( - results.length === 1 || - (results[0].score ?? 0) > 0.9 - ) { - await provider.delete(results[0].id); - return { - content: [ - { - type: "text", - text: `Forgotten: "${results[0].memory}"`, - }, - ], - details: { action: "deleted", id: results[0].id }, - }; - } - - const list = results - .map( - (r) => - `- [${r.id}] ${r.memory.slice(0, 80)}${r.memory.length > 80 ? "..." : ""} (score: ${((r.score ?? 0) * 100).toFixed(0)}%)`, - ) - .join("\n"); - - const candidates = results.map((r) => ({ - id: r.id, - memory: r.memory, - score: r.score, - })); - - return { - content: [ - { - type: "text", - text: `Found ${results.length} candidates. Specify memoryId to delete:\n${list}`, - }, - ], - details: { action: "candidates", candidates }, - }; - } - - return { - content: [ - { type: "text", text: "Provide a query or memoryId." }, - ], - details: { error: "missing_param" }, - }; - } catch (err) { - return { - content: [ - { - type: "text", - text: `Memory forget failed: ${String(err)}`, - }, - ], - details: { error: String(err) }, - }; - } - }, - }, - { name: "memory_forget" }, - ); + registerTools(api, provider, cfg, _resolveUserId, _effectiveUserId, _agentUserId, buildAddOptions, buildSearchOptions, () => currentSessionId); // ======================================================================== // CLI Commands // ======================================================================== - api.registerCli( - ({ program }) => { - const mem0 = program - .command("mem0") - .description("Mem0 memory plugin commands"); - - mem0 - .command("search") - .description("Search memories in Mem0") - .argument("", "Search query") - .option("--limit ", "Max results", String(cfg.topK)) - .option("--scope ", 'Memory scope: "session", "long-term", or "all"', "all") - .option("--agent ", "Search a specific agent's memory namespace") - .action(async (query: string, opts: { limit: string; scope: string; agent?: string }) => { - try { - const limit = parseInt(opts.limit, 10); - const scope = opts.scope as "session" | "long-term" | "all"; - const uid = opts.agent ? _agentUserId(opts.agent) : _effectiveUserId(currentSessionId); - - let allResults: MemoryItem[] = []; - - if (scope === "session" || scope === "all") { - if (currentSessionId) { - const sessionResults = await provider.search( - query, - buildSearchOptions(uid, limit, currentSessionId), - ); - if (sessionResults?.length) { - allResults.push(...sessionResults.map((r) => ({ ...r, _scope: "session" as const }))); - } - } else if (scope === "session") { - console.log("No active session ID available for session-scoped search."); - return; - } - } - - if (scope === "long-term" || scope === "all") { - const longTermResults = await provider.search( - query, - buildSearchOptions(uid, limit), - ); - if (longTermResults?.length) { - allResults.push(...longTermResults.map((r) => ({ ...r, _scope: "long-term" as const }))); - } - } - - // Deduplicate by ID when searching "all" - if (scope === "all") { - const seen = new Set(); - allResults = allResults.filter((r) => { - if (seen.has(r.id)) return false; - seen.add(r.id); - return true; - }); - } - - if (!allResults.length) { - console.log("No memories found."); - return; - } - - const output = allResults.map((r) => ({ - id: r.id, - memory: r.memory, - score: r.score, - scope: (r as any)._scope, - categories: r.categories, - created_at: r.created_at, - })); - console.log(JSON.stringify(output, null, 2)); - } catch (err) { - console.error(`Search failed: ${String(err)}`); - } - }); - - mem0 - .command("stats") - .description("Show memory statistics from Mem0") - .option("--agent ", "Show stats for a specific agent") - .action(async (opts: { agent?: string }) => { - try { - const uid = opts.agent ? _agentUserId(opts.agent) : cfg.userId; - const memories = await provider.getAll({ - user_id: uid, - source: "OPENCLAW", - }); - console.log(`Mode: ${cfg.mode}`); - console.log(`User: ${uid}${opts.agent ? ` (agent: ${opts.agent})` : ""}`); - console.log( - `Total memories: ${Array.isArray(memories) ? memories.length : "unknown"}`, - ); - console.log(`Graph enabled: ${cfg.enableGraph}`); - console.log( - `Auto-recall: ${cfg.autoRecall}, Auto-capture: ${cfg.autoCapture}`, - ); - } catch (err) { - console.error(`Stats failed: ${String(err)}`); - } - }); - }, - { commands: ["mem0"] }, - ); + registerCli(api, provider, cfg, _effectiveUserId, _agentUserId, buildSearchOptions, () => currentSessionId); // ======================================================================== // Lifecycle Hooks // ======================================================================== - // Auto-recall: inject relevant memories before agent starts - if (cfg.autoRecall) { - api.on("before_agent_start", async (event, ctx) => { - if (!event.prompt || event.prompt.length < 5) return; - - // Track session ID - const sessionId = (ctx as any)?.sessionKey ?? undefined; - if (sessionId) currentSessionId = sessionId; - - try { - // Search long-term memories (user-scoped, isolated per agent) - const longTermResults = await provider.search( - event.prompt, - buildSearchOptions(undefined, undefined, undefined, sessionId), - ); - - // Search session memories (session-scoped) if we have a session ID - let sessionResults: MemoryItem[] = []; - if (currentSessionId) { - sessionResults = await provider.search( - event.prompt, - buildSearchOptions(undefined, undefined, currentSessionId, sessionId), - ); - } - - // Deduplicate session results against long-term - const longTermIds = new Set(longTermResults.map((r) => r.id)); - const uniqueSessionResults = sessionResults.filter( - (r) => !longTermIds.has(r.id), - ); - - if (longTermResults.length === 0 && uniqueSessionResults.length === 0) return; - - // Build context with clear labels - let memoryContext = ""; - if (longTermResults.length > 0) { - memoryContext += longTermResults - .map( - (r) => - `- ${r.memory}${r.categories?.length ? ` [${r.categories.join(", ")}]` : ""}`, - ) - .join("\n"); - } - if (uniqueSessionResults.length > 0) { - if (memoryContext) memoryContext += "\n"; - memoryContext += "\nSession memories:\n"; - memoryContext += uniqueSessionResults - .map((r) => `- ${r.memory}`) - .join("\n"); - } - - const totalCount = longTermResults.length + uniqueSessionResults.length; - api.logger.info( - `openclaw-mem0: injecting ${totalCount} memories into context (${longTermResults.length} long-term, ${uniqueSessionResults.length} session)`, - ); - - return { - prependContext: `\nThe following memories may be relevant to this conversation:\n${memoryContext}\n`, - }; - } catch (err) { - api.logger.warn(`openclaw-mem0: recall failed: ${String(err)}`); - } - }); - } - - // Auto-capture: store conversation context after agent ends - if (cfg.autoCapture) { - api.on("agent_end", async (event, ctx) => { - if (!event.success || !event.messages || event.messages.length === 0) { - return; - } - - // Track session ID - const sessionId = (ctx as any)?.sessionKey ?? undefined; - if (sessionId) currentSessionId = sessionId; - - try { - // Extract messages, limiting to last 10 - const recentMessages = event.messages.slice(-10); - const formattedMessages: Array<{ - role: string; - content: string; - }> = []; - - for (const msg of recentMessages) { - if (!msg || typeof msg !== "object") continue; - const msgObj = msg as Record; - - const role = msgObj.role; - if (role !== "user" && role !== "assistant") continue; - - let textContent = ""; - const content = msgObj.content; - - if (typeof content === "string") { - textContent = content; - } else if (Array.isArray(content)) { - for (const block of content) { - if ( - block && - typeof block === "object" && - "text" in block && - typeof (block as Record).text === "string" - ) { - textContent += - (textContent ? "\n" : "") + - ((block as Record).text as string); - } - } - } - - if (!textContent) continue; - // Strip injected memory context, keep the actual user text - if (textContent.includes("")) { - textContent = textContent.replace(/[\s\S]*?<\/relevant-memories>\s*/g, "").trim(); - if (!textContent) continue; - } - - formattedMessages.push({ - role: role as string, - content: textContent, - }); - } - - if (formattedMessages.length === 0) return; - - const addOpts = buildAddOptions(undefined, currentSessionId, sessionId); - const result = await provider.add( - formattedMessages, - addOpts, - ); - - const capturedCount = result.results?.length ?? 0; - if (capturedCount > 0) { - api.logger.info( - `openclaw-mem0: auto-captured ${capturedCount} memories`, - ); - } - } catch (err) { - api.logger.warn(`openclaw-mem0: capture failed: ${String(err)}`); - } - }); - } + registerHooks(api, provider, cfg, _effectiveUserId, buildAddOptions, buildSearchOptions, { + setCurrentSessionId: (id: string) => { currentSessionId = id; }, + }); // ======================================================================== // Service @@ -1523,4 +173,958 @@ const memoryPlugin = { }, }; +// ============================================================================ +// Tool Registration +// ============================================================================ + +function registerTools( + api: OpenClawPluginApi, + provider: Mem0Provider, + cfg: Mem0Config, + _resolveUserId: (opts: { agentId?: string; userId?: string }) => string, + _effectiveUserId: (sessionKey?: string) => string, + _agentUserId: (id: string) => string, + buildAddOptions: (userIdOverride?: string, runId?: string, sessionKey?: string) => AddOptions, + buildSearchOptions: (userIdOverride?: string, limit?: number, runId?: string, sessionKey?: string) => SearchOptions, + getCurrentSessionId: () => string | undefined, +) { + api.registerTool( + { + name: "memory_search", + label: "Memory Search", + description: + "Search through long-term memories stored in Mem0. Use when you need context about user preferences, past decisions, or previously discussed topics.", + parameters: Type.Object({ + query: Type.String({ description: "Search query" }), + limit: Type.Optional( + Type.Number({ + description: `Max results (default: ${cfg.topK})`, + }), + ), + userId: Type.Optional( + Type.String({ + description: + "User ID to scope search (default: configured userId)", + }), + ), + agentId: Type.Optional( + Type.String({ + description: + "Agent ID to search memories for a specific agent (e.g. \"researcher\"). Overrides userId.", + }), + ), + scope: Type.Optional( + Type.Union([ + Type.Literal("session"), + Type.Literal("long-term"), + Type.Literal("all"), + ], { + description: + 'Memory scope: "session" (current session only), "long-term" (user-scoped only), or "all" (both). Default: "all"', + }), + ), + }), + async execute(_toolCallId, params) { + const { query, limit, userId, agentId, scope = "all" } = params as { + query: string; + limit?: number; + userId?: string; + agentId?: string; + scope?: "session" | "long-term" | "all"; + }; + + try { + let results: MemoryItem[] = []; + const uid = _resolveUserId({ agentId, userId }); + const currentSessionId = getCurrentSessionId(); + + if (scope === "session") { + if (currentSessionId) { + results = await provider.search( + query, + buildSearchOptions(uid, limit, currentSessionId), + ); + } + } else if (scope === "long-term") { + results = await provider.search( + query, + buildSearchOptions(uid, limit), + ); + } else { + // "all" — search both scopes and combine + const longTermResults = await provider.search( + query, + buildSearchOptions(uid, limit), + ); + let sessionResults: MemoryItem[] = []; + if (currentSessionId) { + sessionResults = await provider.search( + query, + buildSearchOptions(uid, limit, currentSessionId), + ); + } + // Deduplicate by ID, preferring long-term + const seen = new Set(longTermResults.map((r) => r.id)); + results = [ + ...longTermResults, + ...sessionResults.filter((r) => !seen.has(r.id)), + ]; + } + + if (!results || results.length === 0) { + return { + content: [ + { type: "text", text: "No relevant memories found." }, + ], + details: { count: 0 }, + }; + } + + const text = results + .map( + (r, i) => + `${i + 1}. ${r.memory} (score: ${((r.score ?? 0) * 100).toFixed(0)}%, id: ${r.id})`, + ) + .join("\n"); + + const sanitized = results.map((r) => ({ + id: r.id, + memory: r.memory, + score: r.score, + categories: r.categories, + created_at: r.created_at, + })); + + return { + content: [ + { + type: "text", + text: `Found ${results.length} memories:\n\n${text}`, + }, + ], + details: { count: results.length, memories: sanitized }, + }; + } catch (err) { + return { + content: [ + { + type: "text", + text: `Memory search failed: ${String(err)}`, + }, + ], + details: { error: String(err) }, + }; + } + }, + }, + { name: "memory_search" }, + ); + + api.registerTool( + { + name: "memory_store", + label: "Memory Store", + description: + "Save important information in long-term memory via Mem0. Use for preferences, facts, decisions, and anything worth remembering.", + parameters: Type.Object({ + text: Type.String({ description: "Information to remember" }), + userId: Type.Optional( + Type.String({ + description: "User ID to scope this memory", + }), + ), + agentId: Type.Optional( + Type.String({ + description: + "Agent ID to store memory under a specific agent's namespace (e.g. \"researcher\"). Overrides userId.", + }), + ), + metadata: Type.Optional( + Type.Record(Type.String(), Type.Unknown(), { + description: "Optional metadata to attach to this memory", + }), + ), + longTerm: Type.Optional( + Type.Boolean({ + description: + "Store as long-term (user-scoped) memory. Default: true. Set to false for session-scoped memory.", + }), + ), + }), + async execute(_toolCallId, params) { + const { text, userId, agentId, longTerm = true } = params as { + text: string; + userId?: string; + agentId?: string; + metadata?: Record; + longTerm?: boolean; + }; + + try { + const uid = _resolveUserId({ agentId, userId }); + const currentSessionId = getCurrentSessionId(); + const runId = !longTerm && currentSessionId ? currentSessionId : undefined; + + // Pre-check for near-duplicates so the extraction model has + // context about existing memories and can UPDATE rather than ADD + const preview = text.slice(0, 200); + const dedupOpts = buildSearchOptions(uid, 3); + dedupOpts.threshold = 0.85; + const existing = await provider.search(preview, dedupOpts); + if (existing.length > 0) { + api.logger.info( + `openclaw-mem0: found ${existing.length} similar existing memories — mem0 may update instead of add`, + ); + } + + const result = await provider.add( + [{ role: "user", content: text }], + buildAddOptions(uid, runId, currentSessionId), + ); + + const added = + result.results?.filter((r) => r.event === "ADD") ?? []; + const updated = + result.results?.filter((r) => r.event === "UPDATE") ?? []; + + const summary = []; + if (added.length > 0) + summary.push( + `${added.length} new memor${added.length === 1 ? "y" : "ies"} added`, + ); + if (updated.length > 0) + summary.push( + `${updated.length} memor${updated.length === 1 ? "y" : "ies"} updated`, + ); + if (summary.length === 0) + summary.push("No new memories extracted"); + + return { + content: [ + { + type: "text", + text: `Stored: ${summary.join(", ")}. ${result.results?.map((r) => `[${r.event}] ${r.memory}`).join("; ") ?? ""}`, + }, + ], + details: { + action: "stored", + results: result.results, + }, + }; + } catch (err) { + return { + content: [ + { + type: "text", + text: `Memory store failed: ${String(err)}`, + }, + ], + details: { error: String(err) }, + }; + } + }, + }, + { name: "memory_store" }, + ); + + api.registerTool( + { + name: "memory_get", + label: "Memory Get", + description: "Retrieve a specific memory by its ID from Mem0.", + parameters: Type.Object({ + memoryId: Type.String({ description: "The memory ID to retrieve" }), + }), + async execute(_toolCallId, params) { + const { memoryId } = params as { memoryId: string }; + + try { + const memory = await provider.get(memoryId); + + return { + content: [ + { + type: "text", + text: `Memory ${memory.id}:\n${memory.memory}\n\nCreated: ${memory.created_at ?? "unknown"}\nUpdated: ${memory.updated_at ?? "unknown"}`, + }, + ], + details: { memory }, + }; + } catch (err) { + return { + content: [ + { + type: "text", + text: `Memory get failed: ${String(err)}`, + }, + ], + details: { error: String(err) }, + }; + } + }, + }, + { name: "memory_get" }, + ); + + api.registerTool( + { + name: "memory_list", + label: "Memory List", + description: + "List all stored memories for a user or agent. Use this when you want to see everything that's been remembered, rather than searching for something specific.", + parameters: Type.Object({ + userId: Type.Optional( + Type.String({ + description: + "User ID to list memories for (default: configured userId)", + }), + ), + agentId: Type.Optional( + Type.String({ + description: + "Agent ID to list memories for a specific agent (e.g. \"researcher\"). Overrides userId.", + }), + ), + scope: Type.Optional( + Type.Union([ + Type.Literal("session"), + Type.Literal("long-term"), + Type.Literal("all"), + ], { + description: + 'Memory scope: "session" (current session only), "long-term" (user-scoped only), or "all" (both). Default: "all"', + }), + ), + }), + async execute(_toolCallId, params) { + const { userId, agentId, scope = "all" } = params as { userId?: string; agentId?: string; scope?: "session" | "long-term" | "all" }; + + try { + let memories: MemoryItem[] = []; + const uid = _resolveUserId({ agentId, userId }); + const currentSessionId = getCurrentSessionId(); + + if (scope === "session") { + if (currentSessionId) { + memories = await provider.getAll({ + user_id: uid, + run_id: currentSessionId, + source: "OPENCLAW", + }); + } + } else if (scope === "long-term") { + memories = await provider.getAll({ user_id: uid, source: "OPENCLAW" }); + } else { + // "all" — combine both scopes + const longTerm = await provider.getAll({ user_id: uid, source: "OPENCLAW" }); + let session: MemoryItem[] = []; + if (currentSessionId) { + session = await provider.getAll({ + user_id: uid, + run_id: currentSessionId, + source: "OPENCLAW", + }); + } + const seen = new Set(longTerm.map((r) => r.id)); + memories = [ + ...longTerm, + ...session.filter((r) => !seen.has(r.id)), + ]; + } + + if (!memories || memories.length === 0) { + return { + content: [ + { type: "text", text: "No memories stored yet." }, + ], + details: { count: 0 }, + }; + } + + const text = memories + .map( + (r, i) => + `${i + 1}. ${r.memory} (id: ${r.id})`, + ) + .join("\n"); + + const sanitized = memories.map((r) => ({ + id: r.id, + memory: r.memory, + categories: r.categories, + created_at: r.created_at, + })); + + return { + content: [ + { + type: "text", + text: `${memories.length} memories:\n\n${text}`, + }, + ], + details: { count: memories.length, memories: sanitized }, + }; + } catch (err) { + return { + content: [ + { + type: "text", + text: `Memory list failed: ${String(err)}`, + }, + ], + details: { error: String(err) }, + }; + } + }, + }, + { name: "memory_list" }, + ); + + api.registerTool( + { + name: "memory_forget", + label: "Memory Forget", + description: + "Delete memories from Mem0. Provide a specific memoryId to delete directly, or a query to search and delete matching memories. Supports agent-scoped deletion. GDPR-compliant.", + parameters: Type.Object({ + query: Type.Optional( + Type.String({ + description: "Search query to find memory to delete", + }), + ), + memoryId: Type.Optional( + Type.String({ description: "Specific memory ID to delete" }), + ), + agentId: Type.Optional( + Type.String({ + description: + "Agent ID to scope deletion to a specific agent's memories (e.g. \"researcher\").", + }), + ), + }), + async execute(_toolCallId, params) { + const { query, memoryId, agentId } = params as { + query?: string; + memoryId?: string; + agentId?: string; + }; + + try { + if (memoryId) { + await provider.delete(memoryId); + return { + content: [ + { type: "text", text: `Memory ${memoryId} forgotten.` }, + ], + details: { action: "deleted", id: memoryId }, + }; + } + + if (query) { + const uid = _resolveUserId({ agentId }); + const results = await provider.search( + query, + buildSearchOptions(uid, 5), + ); + + if (!results || results.length === 0) { + return { + content: [ + { type: "text", text: "No matching memories found." }, + ], + details: { found: 0 }, + }; + } + + // If single high-confidence match, delete directly + if ( + results.length === 1 || + (results[0].score ?? 0) > 0.9 + ) { + await provider.delete(results[0].id); + return { + content: [ + { + type: "text", + text: `Forgotten: "${results[0].memory}"`, + }, + ], + details: { action: "deleted", id: results[0].id }, + }; + } + + const list = results + .map( + (r) => + `- [${r.id}] ${r.memory.slice(0, 80)}${r.memory.length > 80 ? "..." : ""} (score: ${((r.score ?? 0) * 100).toFixed(0)}%)`, + ) + .join("\n"); + + const candidates = results.map((r) => ({ + id: r.id, + memory: r.memory, + score: r.score, + })); + + return { + content: [ + { + type: "text", + text: `Found ${results.length} candidates. Specify memoryId to delete:\n${list}`, + }, + ], + details: { action: "candidates", candidates }, + }; + } + + return { + content: [ + { type: "text", text: "Provide a query or memoryId." }, + ], + details: { error: "missing_param" }, + }; + } catch (err) { + return { + content: [ + { + type: "text", + text: `Memory forget failed: ${String(err)}`, + }, + ], + details: { error: String(err) }, + }; + } + }, + }, + { name: "memory_forget" }, + ); +} + +// ============================================================================ +// CLI Registration +// ============================================================================ + +function registerCli( + api: OpenClawPluginApi, + provider: Mem0Provider, + cfg: Mem0Config, + _effectiveUserId: (sessionKey?: string) => string, + _agentUserId: (id: string) => string, + buildSearchOptions: (userIdOverride?: string, limit?: number, runId?: string, sessionKey?: string) => SearchOptions, + getCurrentSessionId: () => string | undefined, +) { + api.registerCli( + ({ program }) => { + const mem0 = program + .command("mem0") + .description("Mem0 memory plugin commands"); + + mem0 + .command("search") + .description("Search memories in Mem0") + .argument("", "Search query") + .option("--limit ", "Max results", String(cfg.topK)) + .option("--scope ", 'Memory scope: "session", "long-term", or "all"', "all") + .option("--agent ", "Search a specific agent's memory namespace") + .action(async (query: string, opts: { limit: string; scope: string; agent?: string }) => { + try { + const limit = parseInt(opts.limit, 10); + const scope = opts.scope as "session" | "long-term" | "all"; + const currentSessionId = getCurrentSessionId(); + const uid = opts.agent ? _agentUserId(opts.agent) : _effectiveUserId(currentSessionId); + + let allResults: MemoryItem[] = []; + + if (scope === "session" || scope === "all") { + if (currentSessionId) { + const sessionResults = await provider.search( + query, + buildSearchOptions(uid, limit, currentSessionId), + ); + if (sessionResults?.length) { + allResults.push(...sessionResults.map((r) => ({ ...r, _scope: "session" as const }))); + } + } else if (scope === "session") { + console.log("No active session ID available for session-scoped search."); + return; + } + } + + if (scope === "long-term" || scope === "all") { + const longTermResults = await provider.search( + query, + buildSearchOptions(uid, limit), + ); + if (longTermResults?.length) { + allResults.push(...longTermResults.map((r) => ({ ...r, _scope: "long-term" as const }))); + } + } + + // Deduplicate by ID when searching "all" + if (scope === "all") { + const seen = new Set(); + allResults = allResults.filter((r) => { + if (seen.has(r.id)) return false; + seen.add(r.id); + return true; + }); + } + + if (!allResults.length) { + console.log("No memories found."); + return; + } + + const output = allResults.map((r) => ({ + id: r.id, + memory: r.memory, + score: r.score, + scope: (r as any)._scope, + categories: r.categories, + created_at: r.created_at, + })); + console.log(JSON.stringify(output, null, 2)); + } catch (err) { + console.error(`Search failed: ${String(err)}`); + } + }); + + mem0 + .command("stats") + .description("Show memory statistics from Mem0") + .option("--agent ", "Show stats for a specific agent") + .action(async (opts: { agent?: string }) => { + try { + const uid = opts.agent ? _agentUserId(opts.agent) : cfg.userId; + const memories = await provider.getAll({ + user_id: uid, + source: "OPENCLAW", + }); + console.log(`Mode: ${cfg.mode}`); + console.log(`User: ${uid}${opts.agent ? ` (agent: ${opts.agent})` : ""}`); + console.log( + `Total memories: ${Array.isArray(memories) ? memories.length : "unknown"}`, + ); + console.log(`Graph enabled: ${cfg.enableGraph}`); + console.log( + `Auto-recall: ${cfg.autoRecall}, Auto-capture: ${cfg.autoCapture}`, + ); + } catch (err) { + console.error(`Stats failed: ${String(err)}`); + } + }); + }, + { commands: ["mem0"] }, + ); +} + +// ============================================================================ +// Lifecycle Hook Registration +// ============================================================================ + +function registerHooks( + api: OpenClawPluginApi, + provider: Mem0Provider, + cfg: Mem0Config, + _effectiveUserId: (sessionKey?: string) => string, + buildAddOptions: (userIdOverride?: string, runId?: string, sessionKey?: string) => AddOptions, + buildSearchOptions: (userIdOverride?: string, limit?: number, runId?: string, sessionKey?: string) => SearchOptions, + session: { + setCurrentSessionId: (id: string) => void; + }, +) { + // Auto-recall: inject relevant memories before agent starts + if (cfg.autoRecall) { + api.on("before_agent_start", async (event, ctx) => { + if (!event.prompt || event.prompt.length < 5) return; + + // Skip non-interactive triggers (cron, heartbeat, automation) + const trigger = (ctx as any)?.trigger ?? undefined; + const sessionId = (ctx as any)?.sessionKey ?? undefined; + if (isNonInteractiveTrigger(trigger, sessionId)) { + api.logger.info("openclaw-mem0: skipping recall for non-interactive trigger"); + return; + } + + // Update shared state for tools (best-effort — tools don't have ctx) + if (sessionId) session.setCurrentSessionId(sessionId); + + // Detect new session for cold-start broadening + const isNewSession = true; // treat every hook invocation as potentially new + + // Subagents have ephemeral UUIDs — their namespace is always empty. + // Search the parent (main) user namespace instead so subagents get + // the user's long-term context. + const isSubagent = isSubagentSession(sessionId); + const recallSessionKey = isSubagent ? undefined : sessionId; + + try { + // Use a larger candidate pool for recall, then filter down + const recallTopK = Math.max((cfg.topK ?? 5) * 2, 10); + + // Search long-term memories (user-scoped; subagents read from parent namespace) + let longTermResults = await provider.search( + event.prompt, + buildSearchOptions(undefined, recallTopK, undefined, recallSessionKey), + ); + + // Client-side threshold filter for auto-recall — use a stricter + // threshold (0.6) than explicit tool searches (0.5) to avoid + // injecting irrelevant memories into agent context + const recallThreshold = Math.max(cfg.searchThreshold, 0.6); + longTermResults = longTermResults.filter( + (r) => (r.score ?? 0) >= recallThreshold, + ); + + // Dynamic thresholding: drop memories scoring less than 50% of + // the top result's score to filter out the long tail of weak matches + if (longTermResults.length > 1) { + const topScore = longTermResults[0]?.score ?? 0; + if (topScore > 0) { + longTermResults = longTermResults.filter( + (r) => (r.score ?? 0) >= topScore * 0.5, + ); + } + } + + // For short/generic prompts or new sessions, broaden recall + // with a general query to avoid cold-start blindness. + // Use a lower threshold (0.5) since the generic query is + // intentionally broad and strict thresholds defeat the purpose. + if (event.prompt.length < 100 || isNewSession) { + const broadOpts = buildSearchOptions(undefined, 5, undefined, recallSessionKey); + broadOpts.threshold = 0.5; + const broadResults = await provider.search( + "recent decisions, preferences, active projects, and configuration", + broadOpts, + ); + const existingIds = new Set(longTermResults.map((r) => r.id)); + for (const r of broadResults) { + if (!existingIds.has(r.id)) { + longTermResults.push(r); + } + } + } + + // Cap at configured topK after filtering + longTermResults = longTermResults.slice(0, cfg.topK); + + // Search session memories (session-scoped) if we have a session ID + let sessionResults: MemoryItem[] = []; + if (sessionId) { + sessionResults = await provider.search( + event.prompt, + buildSearchOptions(undefined, undefined, sessionId, recallSessionKey), + ); + sessionResults = sessionResults.filter( + (r) => (r.score ?? 0) >= cfg.searchThreshold, + ); + } + + // Deduplicate session results against long-term + const longTermIds = new Set(longTermResults.map((r) => r.id)); + const uniqueSessionResults = sessionResults.filter( + (r) => !longTermIds.has(r.id), + ); + + if (longTermResults.length === 0 && uniqueSessionResults.length === 0) return; + + // Build context with clear labels + let memoryContext = ""; + if (longTermResults.length > 0) { + memoryContext += longTermResults + .map( + (r) => + `- ${r.memory}${r.categories?.length ? ` [${r.categories.join(", ")}]` : ""}`, + ) + .join("\n"); + } + if (uniqueSessionResults.length > 0) { + if (memoryContext) memoryContext += "\n"; + memoryContext += "\nSession memories:\n"; + memoryContext += uniqueSessionResults + .map((r) => `- ${r.memory}`) + .join("\n"); + } + + const totalCount = longTermResults.length + uniqueSessionResults.length; + api.logger.info( + `openclaw-mem0: injecting ${totalCount} memories into context (${longTermResults.length} long-term, ${uniqueSessionResults.length} session)`, + ); + + const preamble = isSubagent + ? `The following are stored memories for user "${cfg.userId}". You are a subagent — use these memories for context but do not assume you are this user.` + : `The following are stored memories for user "${cfg.userId}". Use them to personalize your response:`; + + return { + prependContext: `\n${preamble}\n${memoryContext}\n`, + }; + } catch (err) { + api.logger.warn(`openclaw-mem0: recall failed: ${String(err)}`); + } + }); + } + + // Auto-capture: store conversation context after agent ends + if (cfg.autoCapture) { + api.on("agent_end", async (event, ctx) => { + if (!event.success || !event.messages || event.messages.length === 0) { + return; + } + + // Skip non-interactive triggers (cron, heartbeat, automation) + const trigger = (ctx as any)?.trigger ?? undefined; + const sessionId = (ctx as any)?.sessionKey ?? undefined; + if (isNonInteractiveTrigger(trigger, sessionId)) { + api.logger.info("openclaw-mem0: skipping capture for non-interactive trigger"); + return; + } + + // Skip capture for subagents — their ephemeral UUIDs create orphaned + // namespaces that are never read again. The main agent's agent_end + // hook captures the consolidated result including subagent output. + if (isSubagentSession(sessionId)) { + api.logger.info("openclaw-mem0: skipping capture for subagent (main agent captures consolidated result)"); + return; + } + + // Update shared state for tools (best-effort — tools don't have ctx) + if (sessionId) session.setCurrentSessionId(sessionId); + + try { + // Patterns indicating an assistant message contains a summary of + // completed work — these are high-value for extraction and should + // be included even if they fall outside the recent-message window. + const SUMMARY_PATTERNS = [ + /## What I (Accomplished|Built|Updated)/i, + /✅\s*(Done|Complete|All done)/i, + /Here's (what I updated|the recap|a summary)/i, + /### Changes Made/i, + /Implementation Status/i, + /All locked in\. Quick summary/i, + ]; + + // First pass: extract all messages into a typed array + const allParsed: Array<{ + role: string; + content: string; + index: number; + isSummary: boolean; + }> = []; + + for (let i = 0; i < event.messages.length; i++) { + const msg = event.messages[i]; + if (!msg || typeof msg !== "object") continue; + const msgObj = msg as Record; + + const role = msgObj.role; + if (role !== "user" && role !== "assistant") continue; + + let textContent = ""; + const content = msgObj.content; + + if (typeof content === "string") { + textContent = content; + } else if (Array.isArray(content)) { + for (const block of content) { + if ( + block && + typeof block === "object" && + "text" in block && + typeof (block as Record).text === "string" + ) { + textContent += + (textContent ? "\n" : "") + + ((block as Record).text as string); + } + } + } + + if (!textContent) continue; + // Strip injected memory context, keep the actual user text + if (textContent.includes("")) { + textContent = textContent.replace(/[\s\S]*?<\/relevant-memories>\s*/g, "").trim(); + if (!textContent) continue; + } + + const isSummary = + role === "assistant" && + SUMMARY_PATTERNS.some((p) => p.test(textContent)); + + allParsed.push({ + role: role as string, + content: textContent, + index: i, + isSummary, + }); + } + + if (allParsed.length === 0) return; + + // Select messages: last 20 + any earlier summary messages, + // sorted by original index to preserve chronological order. + const recentWindow = 20; + const recentCutoff = allParsed.length - recentWindow; + + const candidates: typeof allParsed = []; + + // Include summary messages from anywhere in the conversation + for (const msg of allParsed) { + if (msg.isSummary && msg.index < recentCutoff) { + candidates.push(msg); + } + } + + // Include recent messages + const seenIndices = new Set(candidates.map((m) => m.index)); + for (const msg of allParsed) { + if (msg.index >= recentCutoff && !seenIndices.has(msg.index)) { + candidates.push(msg); + } + } + + // Sort by original position so the extraction model sees + // messages in the order they actually occurred + candidates.sort((a, b) => a.index - b.index); + + const selected = candidates.map((m) => ({ + role: m.role, + content: m.content, + })); + + // Apply noise filtering pipeline: drop noise, strip fragments, truncate + const formattedMessages = filterMessagesForExtraction(selected); + + if (formattedMessages.length === 0) return; + + // Skip if no meaningful user content remains after filtering + if (!formattedMessages.some((m) => m.role === "user")) return; + + // Inject a timestamp preamble so the extraction model can anchor + // time-sensitive facts to a concrete date and attribute to the correct user + const timestamp = new Date().toISOString().split("T")[0]; + formattedMessages.unshift({ + role: "system", + content: `Current date: ${timestamp}. The user is identified as "${cfg.userId}". Extract durable facts from this conversation. Include this date when storing time-sensitive information.`, + }); + + const addOpts = buildAddOptions(undefined, sessionId, sessionId); + const result = await provider.add( + formattedMessages, + addOpts, + ); + + const capturedCount = result.results?.length ?? 0; + if (capturedCount > 0) { + api.logger.info( + `openclaw-mem0: auto-captured ${capturedCount} memories`, + ); + } + } catch (err) { + api.logger.warn(`openclaw-mem0: capture failed: ${String(err)}`); + } + }); + } +} + export default memoryPlugin; diff --git a/openclaw/isolation.ts b/openclaw/isolation.ts new file mode 100644 index 000000000..6a42d7098 --- /dev/null +++ b/openclaw/isolation.ts @@ -0,0 +1,101 @@ +/** + * Per-agent memory isolation helpers. + * + * Multi-agent setups write/read from separate userId namespaces + * automatically via sessionKey routing. + */ + +// ============================================================================ +// Trigger filtering — skip non-interactive sessions +// ============================================================================ + +/** + * Triggers that should NOT run autocapture/autorecall. + * These are system-initiated sessions (cron jobs, heartbeats, automation + * pipelines) whose prompts would pollute the user's memory store. + */ +const SKIP_TRIGGERS = new Set(["cron", "heartbeat", "automation", "schedule"]); + +/** + * Returns true if the session trigger is non-interactive and memory + * hooks should be skipped entirely. + * + * Also detects cron-style session keys (e.g. "agent:main:cron:") + * as a fallback when the trigger field is not set. + */ +export function isNonInteractiveTrigger( + trigger: string | undefined, + sessionKey: string | undefined, +): boolean { + if (trigger && SKIP_TRIGGERS.has(trigger.toLowerCase())) return true; + + // Fallback: detect cron/heartbeat from the session key pattern + if (sessionKey) { + if (/:cron:/i.test(sessionKey) || /:heartbeat:/i.test(sessionKey)) return true; + } + + return false; +} + +/** + * Returns true if the session key indicates a subagent (ephemeral) session. + * Subagent UUIDs are random per-spawn, so their namespaces are always empty + * on recall and orphaned after capture. + */ +export function isSubagentSession(sessionKey: string | undefined): boolean { + if (!sessionKey) return false; + return /:subagent:/i.test(sessionKey); +} + +/** + * Parse an agent ID from a session key. + * + * OpenClaw session key formats: + * - Main agent: "agent:main:main" + * - Subagent: "agent:main:subagent:" + * - Named agent: "agent::" + * + * Returns the subagent UUID for subagent sessions, the agentId for + * non-"main" named agents, or undefined for the main agent session. + */ +export function extractAgentId(sessionKey: string | undefined): string | undefined { + if (!sessionKey) return undefined; + + // Check for subagent pattern: "agent::subagent:" + const subagentMatch = sessionKey.match(/:subagent:([^:]+)$/); + if (subagentMatch?.[1]) return `subagent-${subagentMatch[1]}`; + + // Check for named agent pattern: "agent::" + const match = sessionKey.match(/^agent:([^:]+):/); + const agentId = match?.[1]; + // "main" is the primary session — fall back to configured userId + if (!agentId || agentId === "main") return undefined; + return agentId; +} + +/** + * Derive the effective user_id from a session key, namespacing per-agent. + * Falls back to baseUserId when the session is not agent-scoped. + */ +export function effectiveUserId(baseUserId: string, sessionKey?: string): string { + const agentId = extractAgentId(sessionKey); + return agentId ? `${baseUserId}:agent:${agentId}` : baseUserId; +} + +/** Build a user_id for an explicit agentId (e.g. from tool params). */ +export function agentUserId(baseUserId: string, agentId: string): string { + return `${baseUserId}:agent:${agentId}`; +} + +/** + * Resolve user_id with priority: explicit agentId > explicit userId > session-derived > configured. + */ +export function resolveUserId( + baseUserId: string, + opts: { agentId?: string; userId?: string }, + currentSessionId?: string, +): string { + if (opts.agentId) return agentUserId(baseUserId, opts.agentId); + if (opts.userId) return opts.userId; + return effectiveUserId(baseUserId, currentSessionId); +} diff --git a/openclaw/package.json b/openclaw/package.json index 7eee981e7..0b28792fc 100644 --- a/openclaw/package.json +++ b/openclaw/package.json @@ -1,6 +1,6 @@ { "name": "@mem0/openclaw-mem0", - "version": "0.3.3", + "version": "0.4.0", "type": "module", "description": "Mem0 memory backend for OpenClaw — platform or self-hosted open-source", "license": "Apache-2.0", @@ -29,7 +29,7 @@ }, "dependencies": { "@sinclair/typebox": "0.34.47", - "mem0ai": "^2.2.1" + "mem0ai": "^2.3.0" }, "openclaw": { "extensions": [ diff --git a/openclaw/pnpm-lock.yaml b/openclaw/pnpm-lock.yaml index 80b2f3ade..fa6c23b34 100644 --- a/openclaw/pnpm-lock.yaml +++ b/openclaw/pnpm-lock.yaml @@ -12,7 +12,7 @@ importers: specifier: 0.34.47 version: 0.34.47 mem0ai: - specifier: ^2.2.1 + specifier: ^2.3.0 version: 2.4.0(@anthropic-ai/sdk@0.40.1)(@azure/identity@4.13.0)(@azure/search-documents@12.2.0)(@cloudflare/workers-types@4.20260313.1)(@google/genai@1.45.0)(@langchain/core@0.3.80(openai@4.104.0(ws@8.19.0)(zod@3.25.76)))(@mistralai/mistralai@1.15.1)(@qdrant/js-client-rest@1.13.0(typescript@5.9.3))(@supabase/supabase-js@2.99.1)(@types/jest@29.5.14)(@types/pg@8.11.0)(better-sqlite3@12.8.0)(cloudflare@4.5.0)(groq-sdk@0.3.0)(neo4j-driver@5.28.3)(ollama@0.5.18)(pg@8.11.3)(redis@4.7.1)(ws@8.19.0) devDependencies: '@types/node': diff --git a/openclaw/providers.ts b/openclaw/providers.ts new file mode 100644 index 000000000..45ef8f069 --- /dev/null +++ b/openclaw/providers.ts @@ -0,0 +1,307 @@ +/** + * Mem0 provider implementations: Platform (cloud) and OSS (self-hosted). + */ + +import type { OpenClawPluginApi } from "openclaw/plugin-sdk"; +import type { + Mem0Config, + Mem0Provider, + AddOptions, + SearchOptions, + ListOptions, + MemoryItem, + AddResult, +} from "./types.ts"; + +// ============================================================================ +// Result Normalizers +// ============================================================================ + +function normalizeMemoryItem(raw: any): MemoryItem { + return { + id: raw.id ?? raw.memory_id ?? "", + memory: raw.memory ?? raw.text ?? raw.content ?? "", + // Handle both platform (user_id, created_at) and OSS (userId, createdAt) field names + user_id: raw.user_id ?? raw.userId, + score: raw.score, + categories: raw.categories, + metadata: raw.metadata, + created_at: raw.created_at ?? raw.createdAt, + updated_at: raw.updated_at ?? raw.updatedAt, + }; +} + +function normalizeSearchResults(raw: any): MemoryItem[] { + // Platform API returns flat array, OSS returns { results: [...] } + if (Array.isArray(raw)) return raw.map(normalizeMemoryItem); + if (raw?.results && Array.isArray(raw.results)) + return raw.results.map(normalizeMemoryItem); + return []; +} + +function normalizeAddResult(raw: any): AddResult { + // Handle { results: [...] } shape (both platform and OSS) + if (raw?.results && Array.isArray(raw.results)) { + return { + results: raw.results.map((r: any) => ({ + id: r.id ?? r.memory_id ?? "", + memory: r.memory ?? r.text ?? "", + // Platform API may return PENDING status (async processing) + // OSS stores event in metadata.event + event: r.event ?? r.metadata?.event ?? (r.status === "PENDING" ? "ADD" : "ADD"), + })), + }; + } + // Platform API without output_format returns flat array + if (Array.isArray(raw)) { + return { + results: raw.map((r: any) => ({ + id: r.id ?? r.memory_id ?? "", + memory: r.memory ?? r.text ?? "", + event: r.event ?? r.metadata?.event ?? (r.status === "PENDING" ? "ADD" : "ADD"), + })), + }; + } + return { results: [] }; +} + +// ============================================================================ +// Platform Provider (Mem0 Cloud) +// ============================================================================ + +class PlatformProvider implements Mem0Provider { + private client: any; // MemoryClient from mem0ai + private initPromise: Promise | null = null; + + constructor( + private readonly apiKey: string, + private readonly orgId?: string, + private readonly projectId?: string, + ) { } + + private async ensureClient(): Promise { + if (this.client) return; + if (this.initPromise) return this.initPromise; + this.initPromise = this._init().catch((err) => { + this.initPromise = null; + throw err; + }); + return this.initPromise; + } + + private async _init(): Promise { + const { default: MemoryClient } = await import("mem0ai"); + const opts: { apiKey: string; org_id?: string; project_id?: string } = { apiKey: this.apiKey }; + if (this.orgId) opts.org_id = this.orgId; + if (this.projectId) opts.project_id = this.projectId; + this.client = new MemoryClient(opts); + } + + async add( + messages: Array<{ role: string; content: string }>, + options: AddOptions, + ): Promise { + await this.ensureClient(); + const opts: Record = { user_id: options.user_id }; + if (options.run_id) opts.run_id = options.run_id; + if (options.custom_instructions) + opts.custom_instructions = options.custom_instructions; + if (options.custom_categories) + opts.custom_categories = options.custom_categories; + if (options.enable_graph) opts.enable_graph = options.enable_graph; + if (options.output_format) opts.output_format = options.output_format; + if (options.source) opts.source = options.source; + + const result = await this.client.add(messages, opts); + return normalizeAddResult(result); + } + + async search(query: string, options: SearchOptions): Promise { + await this.ensureClient(); + const filters: Record = { user_id: options.user_id }; + if (options.run_id) filters.run_id = options.run_id; + + const opts: Record = { + api_version: "v2", + filters, + }; + if (options.top_k != null) opts.top_k = options.top_k; + if (options.threshold != null) opts.threshold = options.threshold; + if (options.keyword_search != null) opts.keyword_search = options.keyword_search; + if (options.reranking != null) opts.rerank = options.reranking; + + const results = await this.client.search(query, opts); + return normalizeSearchResults(results); + } + + async get(memoryId: string): Promise { + await this.ensureClient(); + const result = await this.client.get(memoryId); + return normalizeMemoryItem(result); + } + + async getAll(options: ListOptions): Promise { + await this.ensureClient(); + const opts: Record = { user_id: options.user_id }; + if (options.run_id) opts.run_id = options.run_id; + if (options.page_size != null) opts.page_size = options.page_size; + if (options.source) opts.source = options.source; + + const results = await this.client.getAll(opts); + if (Array.isArray(results)) return results.map(normalizeMemoryItem); + // Some versions return { results: [...] } + if (results?.results && Array.isArray(results.results)) + return results.results.map(normalizeMemoryItem); + return []; + } + + async delete(memoryId: string): Promise { + await this.ensureClient(); + await this.client.delete(memoryId); + } +} + +// ============================================================================ +// Open-Source Provider (Self-hosted) +// ============================================================================ + +class OSSProvider implements Mem0Provider { + private memory: any; // Memory from mem0ai/oss + private initPromise: Promise | null = null; + + constructor( + private readonly ossConfig?: Mem0Config["oss"], + private readonly customPrompt?: string, + private readonly resolvePath?: (p: string) => string, + ) { } + + private async ensureMemory(): Promise { + if (this.memory) return; + if (this.initPromise) return this.initPromise; + this.initPromise = this._init().catch((err) => { + this.initPromise = null; + throw err; + }); + return this.initPromise; + } + + private async _init(): Promise { + const { Memory } = await import("mem0ai/oss"); + + const config: Record = { version: "v1.1" }; + + if (this.ossConfig?.embedder) config.embedder = this.ossConfig.embedder; + if (this.ossConfig?.vectorStore) + config.vectorStore = this.ossConfig.vectorStore; + if (this.ossConfig?.llm) config.llm = this.ossConfig.llm; + + if (this.ossConfig?.historyDbPath) { + const dbPath = this.resolvePath + ? this.resolvePath(this.ossConfig.historyDbPath) + : this.ossConfig.historyDbPath; + config.historyDbPath = dbPath; + } + + if (this.ossConfig?.disableHistory) { + config.disableHistory = true; + } + + if (this.customPrompt) config.customPrompt = this.customPrompt; + + try { + this.memory = new Memory(config); + } catch (err) { + // If initialization fails (e.g. native SQLite binding resolution under + // jiti), retry with history disabled — the history DB is the most common + // source of native-binding failures and is not required for core + // memory operations. + if (!config.disableHistory) { + console.warn( + "[mem0] Memory initialization failed, retrying with history disabled:", + err instanceof Error ? err.message : err, + ); + config.disableHistory = true; + this.memory = new Memory(config); + } else { + throw err; + } + } + } + + async add( + messages: Array<{ role: string; content: string }>, + options: AddOptions, + ): Promise { + await this.ensureMemory(); + // OSS SDK uses camelCase: userId/runId, not user_id/run_id + const addOpts: Record = { userId: options.user_id }; + if (options.run_id) addOpts.runId = options.run_id; + if (options.source) addOpts.source = options.source; + const result = await this.memory.add(messages, addOpts); + return normalizeAddResult(result); + } + + async search(query: string, options: SearchOptions): Promise { + await this.ensureMemory(); + // OSS SDK uses camelCase: userId/runId, not user_id/run_id + const opts: Record = { userId: options.user_id }; + if (options.run_id) opts.runId = options.run_id; + if (options.limit != null) opts.limit = options.limit; + else if (options.top_k != null) opts.limit = options.top_k; + if (options.keyword_search != null) opts.keyword_search = options.keyword_search; + if (options.reranking != null) opts.reranking = options.reranking; + if (options.source) opts.source = options.source; + if (options.threshold != null) opts.threshold = options.threshold; + + const results = await this.memory.search(query, opts); + const normalized = normalizeSearchResults(results); + + // Filter results by threshold if specified (client-side filtering as fallback) + if (options.threshold != null) { + return normalized.filter(item => (item.score ?? 0) >= options.threshold!); + } + + return normalized; + } + + async get(memoryId: string): Promise { + await this.ensureMemory(); + const result = await this.memory.get(memoryId); + return normalizeMemoryItem(result); + } + + async getAll(options: ListOptions): Promise { + await this.ensureMemory(); + // OSS SDK uses camelCase: userId/runId, not user_id/run_id + const getAllOpts: Record = { userId: options.user_id }; + if (options.run_id) getAllOpts.runId = options.run_id; + if (options.source) getAllOpts.source = options.source; + const results = await this.memory.getAll(getAllOpts); + if (Array.isArray(results)) return results.map(normalizeMemoryItem); + if (results?.results && Array.isArray(results.results)) + return results.results.map(normalizeMemoryItem); + return []; + } + + async delete(memoryId: string): Promise { + await this.ensureMemory(); + await this.memory.delete(memoryId); + } +} + +// ============================================================================ +// Provider Factory +// ============================================================================ + +export function createProvider( + cfg: Mem0Config, + api: OpenClawPluginApi, +): Mem0Provider { + if (cfg.mode === "open-source") { + return new OSSProvider(cfg.oss, cfg.customPrompt, (p) => + api.resolvePath(p), + ); + } + + return new PlatformProvider(cfg.apiKey!, cfg.orgId, cfg.projectId); +} diff --git a/openclaw/tsconfig.json b/openclaw/tsconfig.json index 97496a947..2b2ef9d76 100644 --- a/openclaw/tsconfig.json +++ b/openclaw/tsconfig.json @@ -15,8 +15,10 @@ "skipLibCheck": true, "forceConsistentCasingInFileNames": true, "isolatedModules": true, - "verbatimModuleSyntax": true + "verbatimModuleSyntax": true, + "allowImportingTsExtensions": true, + "noEmit": true }, - "include": ["index.ts", "openclaw-plugin-sdk.d.ts"], + "include": ["index.ts", "types.ts", "providers.ts", "config.ts", "filtering.ts", "isolation.ts", "openclaw-plugin-sdk.d.ts"], "exclude": ["node_modules", "dist", "**/*.test.ts"] } diff --git a/openclaw/types.ts b/openclaw/types.ts new file mode 100644 index 000000000..c93d42296 --- /dev/null +++ b/openclaw/types.ts @@ -0,0 +1,91 @@ +/** + * Shared type definitions for the OpenClaw Mem0 plugin. + */ + +export type Mem0Mode = "platform" | "open-source"; + +export type Mem0Config = { + mode: Mem0Mode; + // Platform-specific + apiKey?: string; + orgId?: string; + projectId?: string; + customInstructions: string; + customCategories: Record; + enableGraph: boolean; + // OSS-specific + customPrompt?: string; + oss?: { + embedder?: { provider: string; config: Record }; + vectorStore?: { provider: string; config: Record }; + llm?: { provider: string; config: Record }; + historyDbPath?: string; + disableHistory?: boolean; + }; + // Shared + userId: string; + autoCapture: boolean; + autoRecall: boolean; + searchThreshold: number; + topK: number; +}; + +export interface AddOptions { + user_id: string; + run_id?: string; + custom_instructions?: string; + custom_categories?: Array>; + enable_graph?: boolean; + output_format?: string; + source?: string; +} + +export interface SearchOptions { + user_id: string; + run_id?: string; + top_k?: number; + threshold?: number; + limit?: number; + keyword_search?: boolean; + reranking?: boolean; + source?: string; +} + +export interface ListOptions { + user_id: string; + run_id?: string; + page_size?: number; + source?: string; +} + +export interface MemoryItem { + id: string; + memory: string; + user_id?: string; + score?: number; + categories?: string[]; + metadata?: Record; + created_at?: string; + updated_at?: string; +} + +export interface AddResultItem { + id: string; + memory: string; + event: "ADD" | "UPDATE" | "DELETE" | "NOOP"; +} + +export interface AddResult { + results: AddResultItem[]; +} + +export interface Mem0Provider { + add( + messages: Array<{ role: string; content: string }>, + options: AddOptions, + ): Promise; + search(query: string, options: SearchOptions): Promise; + get(memoryId: string): Promise; + getAll(options: ListOptions): Promise; + delete(memoryId: string): Promise; +}