diff --git a/docs/cookbooks/essentials/building-ai-companion.mdx b/docs/cookbooks/essentials/building-ai-companion.mdx index 95e6df5c9..e6c7e3a57 100644 --- a/docs/cookbooks/essentials/building-ai-companion.mdx +++ b/docs/cookbooks/essentials/building-ai-companion.mdx @@ -10,6 +10,32 @@ Problem: LLMs are stateless. GPT doesn't remember conversations. You could stuff The solution: Mem0. It extracts and stores what matters from conversations, then retrieves it when needed. Your companion remembers user preferences, past events, and history. + + + + + Here we use **Mem0 open source** (`Memory`): all local, no API keys needed for memory. Vectors in **Qdrant**, LLM and embeddings via **Ollama**. The **OpenAI** Python SDK calls Ollama's **OpenAI-compatible** `/v1` endpoint for Ray's chat replies. + + ## Installation + + Install the required dependencies: + + ```bash + pip install mem0ai qdrant-client openai ollama + ``` + + Then start Qdrant and pull the Ollama models: + + ```bash + docker run -d -p 6333:6333 qdrant/qdrant + ollama pull llama3.1:latest + ollama pull nomic-embed-text:latest + ``` + + You can swap `nomic-embed-text` for any Ollama-supported embedding model (e.g., `snowflake-arctic-embed`, `mxbai-embed-large`). Just update the `model` in the `embedder` config and set `embedding_model_dims` in the Qdrant config to match the model's output dimensions (768 for `nomic-embed-text`). + + + In this cookbook we'll build a **fitness companion** that: - Remembers user goals across sessions @@ -25,6 +51,8 @@ By the end, you'll have a working fitness companion and know how to handle commo Max wants to train for a marathon. He starts chatting with Ray, an AI running coach. + + ```python from openai import OpenAI from mem0 import MemoryClient @@ -53,8 +81,73 @@ def chat(user_input, user_id): ], user_id=user_id) return response - ``` + + +```python +from openai import OpenAI +from mem0 import Memory + +OLLAMA_URL = "http://localhost:11434" +CHAT_MODEL = "llama3.1:latest" + +memory = Memory.from_config({ + "vector_store": { + "provider": "qdrant", + "config": { + "collection_name": "fitness_companion", + "host": "localhost", + "port": 6333, + "embedding_model_dims": 768, + }, + }, + "llm": { + "provider": "ollama", + "config": { + "model": CHAT_MODEL, + "temperature": 0, + "max_tokens": 2000, + "ollama_base_url": OLLAMA_URL, + }, + }, + "embedder": { + "provider": "ollama", + "config": { + "model": "nomic-embed-text:latest", + "ollama_base_url": OLLAMA_URL, + }, + }, +}) + +ollama_chat = OpenAI(base_url=f"{OLLAMA_URL}/v1", api_key="ollama") + +def chat(user_input, user_id): + # Retrieve relevant memories + memories = memory.search(user_input, user_id=user_id, limit=5) + context = "\n".join(m["memory"] for m in memories["results"]) + + # Call LLM with memory context (Ollama via OpenAI-compatible API) + response = ollama_chat.chat.completions.create( + model=CHAT_MODEL, + messages=[ + {"role": "system", "content": f"You're Ray, a running coach. Memories:\n{context}"}, + {"role": "user", "content": user_input}, + ], + ).choices[0].message.content + + # Store the exchange + memory.add( + [ + {"role": "user", "content": user_input}, + {"role": "assistant", "content": response}, + ], + user_id=user_id, + ) + + return response +``` + + **Session 1:** @@ -62,7 +155,6 @@ def chat(user_input, user_id): chat("I want to run a marathon in under 4 hours", user_id="max") # Output: "That's a solid goal. What's your current weekly mileage?" # Stored in Mem0: "Max wants to run sub-4 marathon" - ``` **Session 2 (next day, app restarted):** @@ -70,7 +162,6 @@ chat("I want to run a marathon in under 4 hours", user_id="max") ```python chat("What should I focus on today?", user_id="max") # Output: "Based on your sub-4 marathon goal, let's work on building your aerobic base..." - ``` @@ -87,6 +178,8 @@ Ray remembers. Restart the app, and the goal persists. From here on, we'll focus Max mentions his knee hurts. That's different from his marathon goal - one is temporary, the other is long-term. + + **Categories vs Metadata:** - **Categories**: AI-assigned by Mem0 based on content (you can't force them) @@ -100,7 +193,6 @@ mem0_client.project.update(custom_categories=[ {"constraints": "Injuries, limitations, recovery needs"}, {"preferences": "Training style, surfaces, schedules"} ]) - ``` @@ -121,7 +213,6 @@ mem0_client.add( [{"role": "user", "content": "My right knee flares up on downhills"}], user_id="max" ) - ``` Mem0 reads the content and intelligently picks which categories apply. You define the palette, it handles the tagging. @@ -135,13 +226,50 @@ mem0_client.add( user_id="max", metadata={"workout_type": "speed", "forced_tag": "custom_label"} ) - ``` + + +**Categories via Metadata:** + +In open source, model categories with a stable field in `metadata`—here we use `memory_bucket`: + +```python +# Add goal +memory.add( + [{"role": "user", "content": "Sub-4 marathon is my A-race"}], + user_id="max", + metadata={"memory_bucket": "goals"}, +) + +# Add constraint +memory.add( + [{"role": "user", "content": "My right knee flares up on downhills"}], + user_id="max", + metadata={"memory_bucket": "constraints"}, +) +``` + + +**Categories vs Metadata:** In open source, categories are modeled as `metadata` fields you set on each `add`. Filters only see what you put on `add`. + + +```python +# Force tag using metadata +memory.add( + [{"role": "user", "content": "Some workout note"}], + user_id="max", + metadata={"memory_bucket": "goals", "workout_type": "speed", "forced_tag": "custom_label"}, +) +``` + + ### Filtering by Category Retrieve just constraints for workout planning: + + ```python constraints = mem0_client.search( query="injury concerns", @@ -155,8 +283,21 @@ constraints = mem0_client.search( ) print([m["memory"] for m in constraints["results"]]) # Output: ["Max's right knee flares up on downhills"] - ``` + + +```python +constraints = memory.search( + query="injury concerns", + user_id="max", + filters={"memory_bucket": {"in": ["constraints"]}}, + threshold=0.0 # optional: widen recall for short phrases +) +print([m["memory"] for m in constraints["results"]]) +# Output: ["Max's right knee flares up on downhills"] +``` + + Ray can plan workouts that avoid aggravating Max's knee, without pulling in race goals or other unrelated memories. @@ -168,12 +309,22 @@ Ray can plan workouts that avoid aggravating Max's knee, without pulling in race Run the basic loop for a week and check what's stored: + + ```python memories = mem0_client.get_all(filters={"AND": [{"user_id": "max"}]}) print([m["memory"] for m in memories["results"]]) # Output: ["Max wants to run marathon under 4 hours", "hey", "lol ok", "cool thanks", "gtg bye"] - ``` + + +```python +memories = memory.get_all(user_id="max") +print([m["memory"] for m in memories["results"]]) +# Output: ["Max wants to run marathon under 4 hours", "hey", "lol ok", "cool thanks", "gtg bye"] +``` + + Without filters, Mem0 stores everything—greetings, filler, and casual chat. This pollutes retrieval: instead of pulling "marathon goal," you get "lol ok." Set custom instructions to keep memory clean. @@ -183,6 +334,8 @@ Noise. Greetings and filler clutter the memory. ### Custom Instructions + + Tell Mem0 what matters: ```python @@ -198,11 +351,38 @@ Exclude: - Casual chatter - Hypotheticals unless planning related """) - ``` + + +Tell Mem0 what matters by including `custom_fact_extraction_prompt` in the config dict: + +```python +MEMORY_CONFIG["custom_fact_extraction_prompt"] = """ +Extract from running coach conversations: +- Training goals and race targets +- Physical constraints or injuries +- Training preferences (time of day, surfaces, weather) +- Progress milestones + +Exclude: +- Greetings and filler +- Casual chatter +- Hypotheticals unless planning related + +Return JSON with key "facts" as a list of strings (use [] if nothing to store). +""" + +memory = Memory.from_config(MEMORY_CONFIG) +``` + +`custom_fact_extraction_prompt` is a top-level key in the config dictionary passed to `Memory.from_config()`. Make sure it's set before creating the Memory instance — not after. + + Now chat again: + + ```python chat("hey how's it going", user_id="max") chat("I prefer trail running over roads", user_id="max") @@ -210,8 +390,19 @@ chat("I prefer trail running over roads", user_id="max") memories = mem0_client.get_all(filters={"AND": [{"user_id": "max"}]}) print([m["memory"] for m in memories["results"]]) # Output: ["Max wants to run marathon under 4 hours", "Max prefers trail running over roads"] - ``` + + +```python +chat("hey how's it going", user_id="max") +chat("I prefer trail running over roads", user_id="max") + +memories = memory.get_all(user_id="max") +print([m["memory"] for m in memories["results"]]) +# Output: ["Max wants to run marathon under 4 hours", "Max prefers trail running over roads"] +``` + + **Expected output:** Only 2 memories stored—the marathon goal and trail preference. The greeting "hey how's it going" was filtered out automatically. Custom instructions are working. @@ -221,8 +412,6 @@ Only meaningful facts. Filler gets dropped automatically. --- ---- - ## Agent Memory for Personality ### Why Agents Need Memory Too @@ -231,16 +420,30 @@ Max prefers direct feedback, not motivational fluff. Ray needs to remember how t Store agent personality: + + ```python mem0_client.add( [{"role": "system", "content": "Max wants direct, data-driven feedback. Skip motivational language."}], agent_id="ray_coach" ) - ``` + + +```python +memory.add( + [{"role": "user", "content": "Max wants direct, data-driven feedback. Skip motivational language."}], + agent_id="ray_coach", + infer=False, +) +``` + + Retrieve agent style alongside user memories: + + ```python # Get coach personality agent_memories = mem0_client.search("coaching style", agent_id="ray_coach") @@ -251,8 +454,26 @@ mem0_client.add([ {"role": "user", "content": "How'd my run look today?"}, {"role": "assistant", "content": "Pace was 8:15/mile. Heart rate 152, zone 2."} ], user_id="max", agent_id="ray_coach") - ``` + + +```python +# Get coach personality +agent_memories = memory.search("coaching style", agent_id="ray_coach") +# Output: ["Max wants direct, data-driven feedback. Skip motivational language."] + +# Store conversations with agent_id +memory.add( + [ + {"role": "user", "content": "How'd my run look today?"}, + {"role": "assistant", "content": "Pace was 8:15/mile. Heart rate 152, zone 2."}, + ], + user_id="max", + agent_id="ray_coach", +) +``` + + **Expected behavior:** Ray's responses are now data-driven and direct. The agent memory stored the coaching style preference, so future responses adapt automatically without Max having to repeat his preference. @@ -268,6 +489,8 @@ No "Great job!" or "Keep it up!" - just data. Ray adapts to Max's preference. Don't send every single message to Mem0. Keep recent context in memory, let Mem0 handle the important long-term facts. + + ```python # Store only meaningful exchanges in Mem0 mem0_client.add([ @@ -280,8 +503,27 @@ mem0_client.add([ # "cool thanks" → don't store # Or rely on custom_instructions to filter automatically - ``` + + +```python +# Store only meaningful exchanges in Mem0 +memory.add( + [ + {"role": "user", "content": "I want to run a marathon"}, + {"role": "assistant", "content": "Let's build a training plan"}, + ], + user_id="max", +) + +# Skip storing filler +# "hey" → don't store +# "cool thanks" → don't store + +# Or rely on custom_fact_extraction_prompt to filter automatically +``` + + Last 10 messages in your app's buffer. Important facts in Mem0. Faster, cheaper, still works. @@ -293,6 +535,8 @@ Last 10 messages in your app's buffer. Important facts in Mem0. Faster, cheaper, Max tweaks his ankle. It'll heal in two weeks - the memory should expire too. + + ```python from datetime import datetime, timedelta @@ -303,10 +547,26 @@ mem0_client.add( user_id="max", expiration_date=expiration ) - ``` In 14 days, this memory disappears automatically. Ray stops asking about the ankle. + + +```python +from datetime import datetime, timedelta + +expiration = (datetime.now() + timedelta(days=14)).strftime("%Y-%m-%d") + +memory.add( + [{"role": "user", "content": "Rolled my left ankle, needs rest"}], + user_id="max", + metadata={"memory_bucket": "constraints", "expires_on": expiration}, +) +``` + +Store `expires_on` in metadata and prune expired memories in your app. Ray stops asking about the ankle once it's removed. + + --- @@ -314,6 +574,8 @@ In 14 days, this memory disappears automatically. Ray stops asking about the ank Here's the Mem0 setup combining everything: + + ```python from mem0 import MemoryClient from datetime import datetime, timedelta @@ -332,11 +594,55 @@ mem0_client.project.update( {"name": "preferences", "description": "Training style"} ] ) - ``` + + +```python +from mem0 import Memory +from datetime import datetime, timedelta + +MEMORY_CONFIG = { + "vector_store": { + "provider": "qdrant", + "config": { + "collection_name": "fitness_companion", + "host": "localhost", + "port": 6333, + "embedding_model_dims": 768, + }, + }, + "llm": { + "provider": "ollama", + "config": { + "model": "llama3.1:latest", + "temperature": 0, + "max_tokens": 2000, + "ollama_base_url": "http://localhost:11434", + }, + }, + "embedder": { + "provider": "ollama", + "config": { + "model": "nomic-embed-text:latest", + "ollama_base_url": "http://localhost:11434", + }, + }, + "custom_fact_extraction_prompt": """ + Extract: goals, constraints, preferences, progress + Exclude: greetings, filler, casual chat + Return JSON with key "facts" as a list of strings. + """, +} + +memory = Memory.from_config(MEMORY_CONFIG) +``` + + **Week 1 - Store goals and preferences:** + + ```python mem0_client.add([ {"role": "user", "content": "I want to run a sub-4 marathon"}, @@ -346,11 +652,33 @@ mem0_client.add([ mem0_client.add([ {"role": "user", "content": "I prefer trail running over roads"} ], user_id="max", categories=["preferences"]) - ``` + + +```python +memory.add( + [ + {"role": "user", "content": "I want to run a sub-4 marathon"}, + {"role": "assistant", "content": "Got it. Let's build a training plan."}, + ], + user_id="max", + agent_id="ray", + metadata={"memory_bucket": "goals"}, +) + +memory.add( + [{"role": "user", "content": "I prefer trail running over roads"}], + user_id="max", + metadata={"memory_bucket": "preferences"}, +) +``` + + **Week 3 - Temporary injury with expiration:** + + ```python expiration = (datetime.now() + timedelta(days=14)).strftime("%Y-%m-%d") mem0_client.add( @@ -359,16 +687,36 @@ mem0_client.add( categories=["constraints"], expiration_date=expiration ) - ``` + + +```python +expiration = (datetime.now() + timedelta(days=14)).strftime("%Y-%m-%d") +memory.add( + [{"role": "user", "content": "Rolled ankle, need light workouts"}], + user_id="max", + metadata={"memory_bucket": "constraints", "expires_on": expiration}, +) +``` + + **Retrieve for context:** + + ```python memories = mem0_client.search("training plan", user_id="max", limit=5) # Gets: marathon goal, trail preference, ankle injury (if still valid) - ``` + + +```python +memories = memory.search("training plan", user_id="max", limit=5) +# Gets: marathon goal, trail preference, ankle injury (if still valid / not pruned) +``` + + Ray remembers goals, preferences, and personality. Handles temporary injuries. Works across sessions. @@ -380,6 +728,8 @@ Ray remembers goals, preferences, and personality. Handles temporary injuries. W Training for Boston is different from training for New York. Separate the memory threads: + + ```python mem0_client.add(messages, user_id="max", run_id="boston-2025") mem0_client.add(messages, user_id="max", run_id="nyc-2025") @@ -390,8 +740,22 @@ boston_memories = mem0_client.search( user_id="max", run_id="boston-2025" ) - ``` + + +```python +memory.add(messages, user_id="max", run_id="boston-2025") +memory.add(messages, user_id="max", run_id="nyc-2025") + +# Retrieve only Boston memories +boston_memories = memory.search( + "training plan", + user_id="max", + run_id="boston-2025", +) +``` + + Each race gets its own episodic boundary. No cross-contamination. @@ -399,6 +763,8 @@ Each race gets its own episodic boundary. No cross-contamination. Max has 6 months of training logs to backfill: + + ```python old_logs = [ [{"role": "user", "content": "Completed 20-mile long run"}], @@ -407,13 +773,27 @@ old_logs = [ for log in old_logs: mem0_client.add(log, user_id="max") - ``` + + +```python +old_logs = [ + [{"role": "user", "content": "Completed 20-mile long run"}], + [{"role": "user", "content": "Hit 8:00 pace on tempo run"}], +] + +for log in old_logs: + memory.add(log, user_id="max") +``` + + ### Handling Contradictions Max changes his goal from sub-4 to sub-3:45: + + ```python # Find the old memory memories = mem0_client.get_all(filters={"AND": [{"user_id": "max"}]}) @@ -421,8 +801,19 @@ goal_memory = [m for m in memories["results"] if "sub-4" in m["memory"]][0] # Update it mem0_client.update(goal_memory["id"], "Max wants to run sub-3:45 marathon") - ``` + + +```python +# Find the old memory +memories = memory.get_all(user_id="max") +goal_memory = [m for m in memories["results"] if "sub-4" in m["memory"]][0] + +# Update it +memory.update(goal_memory["id"], "Max wants to run sub-3:45 marathon") +``` + + Update instead of creating duplicates. @@ -430,11 +821,20 @@ Update instead of creating duplicates. Max works with Ray for running and Jordan for strength training: + + ```python chat("easy run today", user_id="max", agent_id="ray") chat("leg day workout", user_id="max", agent_id="jordan") - ``` + + +```python +chat("easy run today", user_id="max", agent_id="ray") +chat("leg day workout", user_id="max", agent_id="jordan") +``` + + Each coach maintains separate personality memory while sharing user context. @@ -442,19 +842,44 @@ Each coach maintains separate personality memory while sharing user context. Prioritize recent training over old data: + + ```python recent = mem0_client.search( "training progress", user_id="max", filters={"created_at": {"gte": "2025-10-01"}} ) - ``` + + +```python +# Qdrant range filters require numbers — store an epoch timestamp in metadata +from datetime import datetime + +epoch = int(datetime(2025, 10, 15).timestamp()) +memory.add( + [{"role": "user", "content": "Completed 18-mile long run"}], + user_id="max", + metadata={"logged_epoch": epoch}, +) + +cutoff = int(datetime(2025, 10, 1).timestamp()) +recent = memory.search( + "training progress", + user_id="max", + filters={"logged_epoch": {"gte": cutoff}}, +) +``` + + ### Metadata Tagging Tag workouts by type: + + ```python mem0_client.add( [{"role": "user", "content": "10x400m intervals"}], @@ -468,20 +893,48 @@ speed_sessions = mem0_client.search( user_id="max", filters={"metadata": {"workout_type": "speed"}} ) - ``` + + +```python +memory.add( + [{"role": "user", "content": "10x400m intervals"}], + user_id="max", + metadata={"workout_type": "speed", "intensity": "high"}, +) + +# Later, find all speed workouts +speed_sessions = memory.search( + "speed work", + user_id="max", + filters={"workout_type": "speed"}, +) +``` + + ### Pruning Old Memories Delete irrelevant memories: + + ```python mem0_client.delete(memory_id="mem_xyz") # Or clear an entire run_id mem0_client.delete_all(user_id="max", run_id="old-training-cycle") - ``` + + +```python +memory.delete(memory_id="mem_xyz") + +# Or clear an entire run_id +memory.delete_all(user_id="max", run_id="old-training-cycle") +``` + + --- @@ -515,7 +968,7 @@ Before launching: - Define 2-3 categories (goals, constraints, preferences) - Add expiration strategy for time-bound facts - Implement error handling for API calls -- Monitor memory quality in Mem0 dashboard +- Monitor memory quality (Mem0 dashboard or `get_all` / Qdrant when local) - Clear test data from production project ---