feat(ai-sdk): added file support for multimodal capabilities with memory context (#3500)

This commit is contained in:
Saket Aryan
2025-09-25 10:32:06 +05:30
committed by GitHub
parent a199ee4ff8
commit 6e1d02c137
5 changed files with 262 additions and 7 deletions
+5
View File
@@ -1087,6 +1087,11 @@ mode: "wide"
<Tab title="Vercel AI SDK">
<Update label="2025-09-25" description="v2.0.3">
**New Features:**
- **Vercel AI SDK:** Added file support for multimodal capabilities with memory context
</Update>
<Update label="2025-09-03" description="v2.0.2">
**Bug Fix:**
- **Vercel AI SDK:** Fixed streaming response in the AI SDK.
+66
View File
@@ -203,6 +203,72 @@ console.log(sources);
The same can be done for `streamText` as well.
### 6. File Support with Memory Context
Mem0 AI SDK supports file processing with memory context. Here's an example of analyzing a PDF file:
```typescript
import { streamText } from "ai";
import { createMem0 } from "@mem0/vercel-ai-provider";
import { readFileSync } from 'fs';
import { join } from 'path';
const mem0 = createMem0({
provider: "google",
mem0ApiKey: "m0-xxx",
config: {
apiKey: "google-api-key"
},
mem0Config: {
user_id: "alice",
},
});
async function main() {
// Read the PDF file
const filePath = join(process.cwd(), 'my_pdf.pdf');
const fileBuffer = readFileSync(filePath);
// Convert the file's arrayBuffer to a Base64 data URL
const arrayBuffer = fileBuffer.buffer.slice(fileBuffer.byteOffset, fileBuffer.byteOffset + fileBuffer.byteLength);
const uint8Array = new Uint8Array(arrayBuffer);
// Convert Uint8Array to an array of characters
const charArray = Array.from(uint8Array, byte => String.fromCharCode(byte));
const binaryString = charArray.join('');
const base64Data = Buffer.from(binaryString, 'binary').toString('base64');
const fileDataUrl = `data:application/pdf;base64,${base64Data}`;
const { textStream } = streamText({
model: mem0("gemini-2.5-flash"),
messages: [
{
role: 'user',
content: [
{
type: 'text',
text: 'Analyze the following PDF and generate a summary.',
},
{
type: 'file',
data: fileDataUrl,
mediaType: 'application/pdf',
},
],
},
],
});
for await (const textPart of textStream) {
process.stdout.write(textPart);
}
}
main();
```
> **Note**: File support is available with providers that support multimodal capabilities like Google's Gemini models. The example shows how to process PDF files, but you can also work with images, text files, and other supported formats.
## Graph Memory
Mem0 AI SDK now supports Graph Memory. You can enable it by setting `enable_graph` to `true` in the `mem0Config` object.
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "@mem0/vercel-ai-provider",
"version": "2.0.2",
"version": "2.0.3",
"description": "Vercel AI Provider for providing memory to LLMs",
"main": "./dist/index.js",
"module": "./dist/index.mjs",
+185 -6
View File
@@ -1,19 +1,62 @@
import { LanguageModelV2Prompt } from '@ai-sdk/provider';
import { Mem0ConfigSettings } from './mem0-types';
import { loadApiKey } from '@ai-sdk/provider-utils';
interface MultimodalContent {
type: 'text' | 'image_url' | 'mdx_url' | 'pdf_url';
text?: string;
image_url?: {
url: string;
};
mdx_url?: {
url: string;
};
pdf_url?: {
url: string;
};
}
interface FileContent {
type: 'file';
data: string; // fileDataUrl
mediaType: string; // e.g., 'application/pdf', 'text/markdown', 'image/jpeg'
}
interface Message {
role: string;
content: string | Array<{type: string, text: string}>;
content: string | MultimodalContent | Array<MultimodalContent>;
}
const flattenPrompt = (prompt: LanguageModelV2Prompt) => {
try {
return prompt.map((part) => {
if (part.role === "user") {
return part.content
.filter((obj) => obj.type === 'text')
.map((obj) => obj.text)
.join(" ");
if (typeof part.content === 'string') {
return part.content;
} else if (Array.isArray(part.content)) {
return part.content
.filter((obj) => obj.type === 'text')
.map((obj) => obj.text)
.join(" ");
} else if (part.content && typeof part.content === 'object' && 'type' in part.content) {
const content = part.content as any;
if (content.type === 'text' && content.text) {
return content.text;
} else if (content.type === 'file') {
// For file content, we'll include a descriptive placeholder
if (content.mediaType === 'application/pdf') {
return '[PDF document]';
} else if (content.mediaType === 'text/markdown' || content.mediaType === 'application/mdx') {
return '[Markdown document]';
} else if (content.mediaType && content.mediaType.startsWith('image/')) {
return '[Image]';
} else {
return '[File attachment]';
}
}
}
// For non-text content (images, pdfs, mdx), we'll include a placeholder
// This helps maintain context for memory search while not breaking the text flow
return "[multimodal content]";
}
return "";
}).join(" ");
@@ -33,7 +76,7 @@ const convertToMem0Format = (messages: LanguageModelV2Prompt) => {
content: message.content,
};
}
else {
else if (Array.isArray(message.content)) {
return message.content.map((obj: any) => {
try {
if (obj.type === "text") {
@@ -41,6 +84,69 @@ const convertToMem0Format = (messages: LanguageModelV2Prompt) => {
role: message.role,
content: obj.text,
};
} else if (obj.type === "file") {
// Handle LanguageModelV2Prompt file format
if (obj.mediaType === "application/pdf") {
return {
role: message.role,
content: {
type: "pdf_url",
pdf_url: {
url: obj.data
}
}
};
} else if (obj.mediaType === "text/markdown" || obj.mediaType === "application/mdx") {
return {
role: message.role,
content: {
type: "mdx_url",
mdx_url: {
url: obj.data
}
}
};
} else if (obj.mediaType && obj.mediaType.startsWith("image/")) {
return {
role: message.role,
content: {
type: "image_url",
image_url: {
url: obj.data
}
}
};
}
} else if (obj.type === "image_url" || obj.type === "image") {
return {
role: message.role,
content: {
type: "image_url",
image_url: {
url: obj.image_url?.url || obj.image?.url || obj.url
}
}
};
} else if (obj.type === "mdx_url" || obj.type === "mdx") {
return {
role: message.role,
content: {
type: "mdx_url",
mdx_url: {
url: obj.mdx_url?.url || obj.mdx?.url || obj.url
}
}
};
} else if (obj.type === "pdf_url" || obj.type === "pdf") {
return {
role: message.role,
content: {
type: "pdf_url",
pdf_url: {
url: obj.pdf_url?.url || obj.pdf?.url || obj.url
}
}
};
}
return null;
} catch (error) {
@@ -48,6 +154,79 @@ const convertToMem0Format = (messages: LanguageModelV2Prompt) => {
return null;
}
}).filter((item: null) => item !== null);
} else {
// Handle single multimodal content object
const obj = message.content;
if (obj.type === "text") {
return {
role: message.role,
content: obj.text,
};
} else if (obj.type === "file") {
// Handle LanguageModelV2Prompt file format
if (obj.mediaType === "application/pdf") {
return {
role: message.role,
content: {
type: "pdf_url",
pdf_url: {
url: obj.data
}
}
};
} else if (obj.mediaType === "text/markdown" || obj.mediaType === "application/mdx") {
return {
role: message.role,
content: {
type: "mdx_url",
mdx_url: {
url: obj.data
}
}
};
} else if (obj.mediaType && obj.mediaType.startsWith("image/")) {
return {
role: message.role,
content: {
type: "image_url",
image_url: {
url: obj.data
}
}
};
}
} else if (obj.type === "image_url" || obj.type === "image") {
return {
role: message.role,
content: {
type: "image_url",
image_url: {
url: obj.image_url?.url || obj.image?.url || obj.url
}
}
};
} else if (obj.type === "mdx_url" || obj.type === "mdx") {
return {
role: message.role,
content: {
type: "mdx_url",
mdx_url: {
url: obj.mdx_url?.url || obj.mdx?.url || obj.url
}
}
};
} else if (obj.type === "pdf_url" || obj.type === "pdf") {
return {
role: message.role,
content: {
type: "pdf_url",
pdf_url: {
url: obj.pdf_url?.url || obj.pdf?.url || obj.url
}
}
};
}
return null;
}
} catch (error) {
console.error("Error processing message:", error);
@@ -59,6 +59,11 @@ class Mem0AITextGenerator implements LanguageModelV2 {
})(modelId);
break;
case "google":
this.languageModel = createGoogleGenerativeAI({
apiKey: config?.apiKey,
...provider_config as GoogleGenerativeAIProviderSettings,
})(modelId);
break;
case "gemini":
this.languageModel = createGoogleGenerativeAI({
apiKey: config?.apiKey,