docs: remove duplicate multimodal page and normalize em/en-dashes (#6112)
This commit is contained in:
@@ -14,7 +14,7 @@ Multimodal support lets Mem0 extract facts from images alongside regular text. A
|
||||
</Info>
|
||||
|
||||
<Warning>
|
||||
Images larger than 20 MB are rejected. Compress or resize files before sending them to avoid errors.
|
||||
Mem0 passes images straight to your configured vision model, so per-image size and resolution limits come from that provider (for example, OpenAI caps images at 20 MB). Compress or resize large files to stay within your provider's limits and keep processing fast.
|
||||
</Warning>
|
||||
|
||||
---
|
||||
@@ -49,6 +49,10 @@ Multimodal support lets Mem0 extract facts from images alongside regular text. A
|
||||
```
|
||||
</Warning>
|
||||
|
||||
<Note>
|
||||
`vision_details` maps to the vision model's image `detail` setting and accepts `"auto"` (the default), `"low"`, or `"high"`. Use `"high"` for dense images like receipts or documents; `"low"` is faster and cheaper for simple photos.
|
||||
</Note>
|
||||
|
||||
### Add image messages from URLs
|
||||
|
||||
<CodeGroup>
|
||||
@@ -161,7 +165,7 @@ await client.add(messages, { userId: "alice" });
|
||||
</CodeGroup>
|
||||
|
||||
<Tip>
|
||||
Keep base64 payloads under 5 MB to speed up uploads and avoid hitting the 20 MB limit.
|
||||
Smaller images upload and process faster. Compress or resize before encoding to base64, and check your vision provider's per-image size limit for large files.
|
||||
</Tip>
|
||||
|
||||
---
|
||||
@@ -234,7 +238,6 @@ client.add(messages, user_id="user123")
|
||||
<CodeGroup>
|
||||
```python Python
|
||||
from mem0 import Memory
|
||||
from mem0.exceptions import ValidationError
|
||||
|
||||
client = Memory()
|
||||
|
||||
@@ -250,10 +253,12 @@ try:
|
||||
client.add(messages, user_id="user123")
|
||||
print("Image processed successfully")
|
||||
|
||||
except ValidationError as exc:
|
||||
print(f"Image validation error: {exc}")
|
||||
except ValueError as exc:
|
||||
# Raised when an image_url part is missing its url
|
||||
print(f"Invalid image message: {exc}")
|
||||
except Exception as exc:
|
||||
print(f"Unexpected error: {exc}")
|
||||
# Raised when the image can't be downloaded or the vision model call fails
|
||||
print(f"Could not process image: {exc}")
|
||||
```
|
||||
|
||||
```ts TypeScript
|
||||
@@ -273,13 +278,8 @@ try {
|
||||
await client.add(messages, { userId: "user123" });
|
||||
console.log("Image processed successfully");
|
||||
} catch (error: any) {
|
||||
if (error.type === "invalid_image") {
|
||||
console.log("Invalid image format or corrupted file");
|
||||
} else if (error.type === "file_size_exceeded") {
|
||||
console.log("Image file too large");
|
||||
} else {
|
||||
console.log(`Unexpected error: ${error.message}`);
|
||||
}
|
||||
// A missing image_url, a failed download, or a vision model error surfaces here
|
||||
console.log(`Could not process image: ${error.message}`);
|
||||
}
|
||||
```
|
||||
</CodeGroup>
|
||||
@@ -313,10 +313,10 @@ try {
|
||||
|
||||
| Issue | Cause | Fix |
|
||||
| --- | --- | --- |
|
||||
| Upload rejected | File larger than 20 MB | Compress or resize before sending. |
|
||||
| Upload rejected | Image exceeds your vision provider's size limit | Compress or resize before sending. |
|
||||
| Memory missing image data | Low-quality or blurry image | Retake the photo with better lighting. |
|
||||
| Invalid format error | Unsupported file type | Convert to JPEG or PNG first. |
|
||||
| Slow processing | High-resolution images | Downscale or compress to under 5 MB. |
|
||||
| Slow processing | High-resolution images | Downscale or compress to under 5 MB. |
|
||||
| Base64 errors | Incorrect prefix or encoding | Ensure `data:image/<type>;base64,` is present and the string is valid. |
|
||||
|
||||
---
|
||||
|
||||
@@ -32,11 +32,11 @@ Reranker-enhanced search adds a second scoring pass after vector retrieval so Me
|
||||
|
||||
<AccordionGroup>
|
||||
<Accordion title="Supported providers">
|
||||
- **[Cohere](/components/rerankers/models/cohere)** – Multilingual hosted reranker with API-based scoring.
|
||||
- **[Sentence Transformer](/components/rerankers/models/sentence_transformer)** – Local Hugging Face cross-encoders for GPU or CPU.
|
||||
- **[Hugging Face](/components/rerankers/models/huggingface)** – Bring any hosted or on-prem reranker model ID.
|
||||
- **[LLM Reranker](/components/rerankers/models/llm_reranker)** – Use your preferred LLM (OpenAI, etc.) for prompt-driven scoring.
|
||||
- **[Zero Entropy](/components/rerankers/models/zero_entropy)** – High-quality neural reranking tuned for retrieval tasks.
|
||||
- **[Cohere](/components/rerankers/models/cohere)**: Multilingual hosted reranker with API-based scoring.
|
||||
- **[Sentence Transformer](/components/rerankers/models/sentence_transformer)**: Local Hugging Face cross-encoders for GPU or CPU.
|
||||
- **[Hugging Face](/components/rerankers/models/huggingface)**: Bring any hosted or on-prem reranker model ID.
|
||||
- **[LLM Reranker](/components/rerankers/models/llm_reranker)**: Use your preferred LLM (OpenAI, etc.) for prompt-driven scoring.
|
||||
- **[Zero Entropy](/components/rerankers/models/zero_entropy)**: High-quality neural reranking tuned for retrieval tasks.
|
||||
</Accordion>
|
||||
<Accordion title="Provider comparison">
|
||||
| Provider | Latency | Quality | Cost | Local deploy |
|
||||
|
||||
@@ -1,269 +0,0 @@
|
||||
---
|
||||
title: Multimodal Support
|
||||
description: "Enable multimodal support in Mem0 to process images alongside text and extract visual information into memory."
|
||||
icon: "image"
|
||||
iconType: "solid"
|
||||
---
|
||||
|
||||
Mem0 extends its capabilities beyond text by supporting multimodal data, including images. You can seamlessly integrate images into your interactions, allowing Mem0 to extract pertinent information from visual content and enrich the memory system.
|
||||
|
||||
## How It Works
|
||||
|
||||
When you provide an image, Mem0 processes it to extract textual information and relevant details, which are then added to your memory. This feature enhances the system's ability to understand and remember details based on visual inputs.
|
||||
|
||||
<Note>
|
||||
To enable multimodal support, you must set `enable_vision = True` in your configuration. The `vision_details` parameter can be set to "auto" (default), "low", or "high" to control the level of detail in image processing.
|
||||
</Note>
|
||||
|
||||
<CodeGroup>
|
||||
```python Code
|
||||
from mem0 import Memory
|
||||
|
||||
config = {
|
||||
"llm": {
|
||||
"provider": "openai",
|
||||
"config": {
|
||||
"enable_vision": True,
|
||||
"vision_details": "high"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
client = Memory.from_config(config=config)
|
||||
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Hi, my name is Alice."
|
||||
},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "Nice to meet you, Alice! What do you like to eat?"
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": {
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": "https://www.superhealthykids.com/wp-content/uploads/2021/10/best-veggie-pizza-featured-image-square-2.jpg"
|
||||
}
|
||||
}
|
||||
},
|
||||
]
|
||||
|
||||
# Calling the add method to ingest messages into the memory system
|
||||
client.add(messages, user_id="alice")
|
||||
```
|
||||
|
||||
```typescript TypeScript
|
||||
import { Memory, Message } from "mem0ai/oss";
|
||||
|
||||
const client = new Memory();
|
||||
|
||||
const messages: Message[] = [
|
||||
{
|
||||
role: "user",
|
||||
content: "Hi, my name is Alice."
|
||||
},
|
||||
{
|
||||
role: "assistant",
|
||||
content: "Nice to meet you, Alice! What do you like to eat?"
|
||||
},
|
||||
{
|
||||
role: "user",
|
||||
content: {
|
||||
type: "image_url",
|
||||
image_url: {
|
||||
url: "https://www.superhealthykids.com/wp-content/uploads/2021/10/best-veggie-pizza-featured-image-square-2.jpg"
|
||||
}
|
||||
}
|
||||
},
|
||||
]
|
||||
|
||||
await client.add(messages, { userId: "alice" })
|
||||
```
|
||||
|
||||
```json Output
|
||||
{
|
||||
"results": [
|
||||
{
|
||||
"memory": "Name is Alice",
|
||||
"event": "ADD",
|
||||
"id": "7ae113a3-3cb5-46e9-b6f7-486c36391847"
|
||||
},
|
||||
{
|
||||
"memory": "Likes large pizza with toppings including cherry tomatoes, black olives, green spinach, yellow bell peppers, diced ham, and sliced mushrooms",
|
||||
"event": "ADD",
|
||||
"id": "56545065-7dee-4acf-8bf2-a5b2535aabb3"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
## Image Integration Methods
|
||||
|
||||
Mem0 allows you to add images to user interactions through two primary methods: by providing an image URL or by using a Base64-encoded image. Below are examples demonstrating each approach.
|
||||
|
||||
### Using an Image URL (Recommended)
|
||||
|
||||
You can include an image by passing its direct URL. This method is simple and efficient for online images.
|
||||
|
||||
<CodeGroup>
|
||||
```python
|
||||
# Define the image URL
|
||||
image_url = "https://www.superhealthykids.com/wp-content/uploads/2021/10/best-veggie-pizza-featured-image-square-2.jpg"
|
||||
|
||||
# Create the message dictionary with the image URL
|
||||
image_message = {
|
||||
"role": "user",
|
||||
"content": {
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": image_url
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
```typescript TypeScript
|
||||
import { Memory, Message } from "mem0ai/oss";
|
||||
|
||||
const client = new Memory();
|
||||
|
||||
const imageUrl = "https://www.superhealthykids.com/wp-content/uploads/2021/10/best-veggie-pizza-featured-image-square-2.jpg";
|
||||
|
||||
const imageMessage: Message = {
|
||||
role: "user",
|
||||
content: {
|
||||
type: "image_url",
|
||||
image_url: {
|
||||
url: imageUrl
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
await client.add([imageMessage], { userId: "alice" })
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
### Using Base64 Image Encoding for Local Files
|
||||
|
||||
For local images or scenarios where embedding the image directly is preferable, you can use a Base64-encoded string.
|
||||
|
||||
<CodeGroup>
|
||||
```python Python
|
||||
import base64
|
||||
|
||||
# Path to the image file
|
||||
image_path = "path/to/your/image.jpg"
|
||||
|
||||
# Encode the image in Base64
|
||||
with open(image_path, "rb") as image_file:
|
||||
base64_image = base64.b64encode(image_file.read()).decode("utf-8")
|
||||
|
||||
# Create the message dictionary with the Base64-encoded image
|
||||
image_message = {
|
||||
"role": "user",
|
||||
"content": {
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": f"data:image/jpeg;base64,{base64_image}"
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
```typescript TypeScript
|
||||
import { Memory, Message } from "mem0ai/oss";
|
||||
|
||||
const client = new Memory();
|
||||
|
||||
const imagePath = "path/to/your/image.jpg";
|
||||
|
||||
const base64Image = fs.readFileSync(imagePath, { encoding: 'base64' });
|
||||
|
||||
const imageMessage: Message = {
|
||||
role: "user",
|
||||
content: {
|
||||
type: "image_url",
|
||||
image_url: {
|
||||
url: `data:image/jpeg;base64,${base64Image}`
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
await client.add([imageMessage], { userId: "alice" })
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
### OpenAI-Compatible Message Format
|
||||
|
||||
You can also use the OpenAI-compatible format to combine text and images in a single message:
|
||||
|
||||
<CodeGroup>
|
||||
```python Python
|
||||
import base64
|
||||
|
||||
# Path to the image file
|
||||
image_path = "path/to/your/image.jpg"
|
||||
|
||||
# Encode the image in Base64
|
||||
with open(image_path, "rb") as image_file:
|
||||
base64_image = base64.b64encode(image_file.read()).decode("utf-8")
|
||||
|
||||
# Create the message using OpenAI-compatible format
|
||||
message = {
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "What is in this image?",
|
||||
},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {"url": f"data:image/jpeg;base64,{base64_image}"},
|
||||
},
|
||||
],
|
||||
}
|
||||
|
||||
# Add the message to memory
|
||||
client.add([message], user_id="alice")
|
||||
```
|
||||
|
||||
```typescript TypeScript
|
||||
import { Memory, Message } from "mem0ai/oss";
|
||||
|
||||
const client = new Memory();
|
||||
|
||||
const imagePath = "path/to/your/image.jpg";
|
||||
|
||||
const base64Image = fs.readFileSync(imagePath, { encoding: 'base64' });
|
||||
|
||||
const message: Message = {
|
||||
role: "user",
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "What is in this image?",
|
||||
},
|
||||
{
|
||||
type: "image_url",
|
||||
image_url: {
|
||||
url: `data:image/jpeg;base64,${base64Image}`
|
||||
}
|
||||
},
|
||||
],
|
||||
}
|
||||
|
||||
await client.add([message], { userId: "alice" })
|
||||
```
|
||||
</CodeGroup>
|
||||
|
||||
This format allows you to combine text and images in a single message, making it easier to provide context along with visual content.
|
||||
|
||||
By utilizing these methods, you can effectively incorporate images into user interactions, enhancing the multimodal capabilities of your Mem0 instance.
|
||||
|
||||
If you have any questions, please feel free to reach out to us using one of the following methods:
|
||||
|
||||
<Snippet file="get-help.mdx" />
|
||||
Reference in New Issue
Block a user