docs: update memory tool list, CLI usage, and config file reading logic (#4861)
Co-authored-by: Livia Ellen <liviaellen@msn.com>
This commit is contained in:
@@ -883,38 +883,160 @@ export function removeCodeBlocks(text: string): string {
|
||||
/**
|
||||
* Extracts a JSON object from text that may be wrapped in explanation text.
|
||||
*
|
||||
* Some LLMs (especially local models like Ollama/LM Studio) return JSON
|
||||
* wrapped in conversational text without code fences, e.g.:
|
||||
* Some LLMs (especially local models like Ollama/LM Studio, or OpenRouter)
|
||||
* return JSON wrapped in conversational text without code fences, e.g.:
|
||||
*
|
||||
* "Here are the facts I extracted:\n{\"facts\": [\"fact1\"]}\nI hope this helps!"
|
||||
*
|
||||
* This function first tries `removeCodeBlocks` for code-fence-wrapped JSON,
|
||||
* then falls back to locating the first `{` and last `}` to extract the
|
||||
* outermost JSON object.
|
||||
* This function:
|
||||
* 1. Strips known noise tokens from OpenRouter and other providers
|
||||
* 2. Removes code fences and <think> blocks
|
||||
* 3. Tries to find a valid JSON object by testing each `{` as a starting point
|
||||
* 4. Falls back to first/last brace matching if validation isn't possible
|
||||
*
|
||||
* @param text - The raw LLM response text
|
||||
* @returns The extracted JSON string, or the original text if no JSON object
|
||||
* boundaries are found
|
||||
*/
|
||||
export function extractJson(text: string): string {
|
||||
// Step 1: Strip code fences if present
|
||||
const cleaned = removeCodeBlocks(text);
|
||||
// Step 1: Strip known noise tokens from OpenRouter/local models
|
||||
let cleaned = text
|
||||
.replace(/<\|end_of_text\|>/g, "")
|
||||
.replace(/<\|eot_id\|>/g, "")
|
||||
.replace(/<\|im_end\|>/g, "")
|
||||
.replace(/<\|im_start\|>/g, "")
|
||||
.replace(/<\|endoftext\|>/g, "");
|
||||
|
||||
// Step 2: Strip code fences and <think> blocks
|
||||
cleaned = removeCodeBlocks(cleaned);
|
||||
const trimmed = cleaned.trim();
|
||||
|
||||
// Step 2: Try to locate a JSON object by first `{` and last `}` boundaries
|
||||
if (!trimmed) return "";
|
||||
|
||||
// Step 3: Try to find valid JSON object by testing each `{` as potential start
|
||||
// This handles cases like "Here's the {formatted} output: {...actual json...}"
|
||||
const braceIndices: number[] = [];
|
||||
for (let i = 0; i < trimmed.length; i++) {
|
||||
if (trimmed[i] === "{") braceIndices.push(i);
|
||||
}
|
||||
|
||||
for (const start of braceIndices) {
|
||||
// Find the matching closing brace by tracking depth
|
||||
let depth = 0;
|
||||
let inString = false;
|
||||
let escapeNext = false;
|
||||
|
||||
for (let i = start; i < trimmed.length; i++) {
|
||||
const char = trimmed[i];
|
||||
|
||||
if (escapeNext) {
|
||||
escapeNext = false;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (char === "\\") {
|
||||
escapeNext = true;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (char === '"' && !escapeNext) {
|
||||
inString = !inString;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (inString) continue;
|
||||
|
||||
if (char === "{") depth++;
|
||||
else if (char === "}") {
|
||||
depth--;
|
||||
if (depth === 0) {
|
||||
const candidate = trimmed.substring(start, i + 1);
|
||||
try {
|
||||
JSON.parse(candidate);
|
||||
return candidate; // Valid JSON found
|
||||
} catch {
|
||||
// Not valid JSON, try next starting brace
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Step 4: Fallback - try first/last brace (original behavior for edge cases)
|
||||
// Only use this if it produces valid JSON
|
||||
const firstBrace = trimmed.indexOf("{");
|
||||
const lastBrace = trimmed.lastIndexOf("}");
|
||||
if (firstBrace !== -1 && lastBrace > firstBrace) {
|
||||
return trimmed.substring(firstBrace, lastBrace + 1);
|
||||
const candidate = trimmed.substring(firstBrace, lastBrace + 1);
|
||||
try {
|
||||
JSON.parse(candidate);
|
||||
return candidate;
|
||||
} catch {
|
||||
// Not valid JSON, continue to array extraction
|
||||
}
|
||||
}
|
||||
|
||||
// Step 3: Try to locate a JSON array by first `[` and last `]` boundaries
|
||||
// Step 5: Try to locate a JSON array by testing each `[` as potential start
|
||||
const bracketIndices: number[] = [];
|
||||
for (let i = 0; i < trimmed.length; i++) {
|
||||
if (trimmed[i] === "[") bracketIndices.push(i);
|
||||
}
|
||||
|
||||
for (const start of bracketIndices) {
|
||||
let depth = 0;
|
||||
let inString = false;
|
||||
let escapeNext = false;
|
||||
|
||||
for (let i = start; i < trimmed.length; i++) {
|
||||
const char = trimmed[i];
|
||||
|
||||
if (escapeNext) {
|
||||
escapeNext = false;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (char === "\\") {
|
||||
escapeNext = true;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (char === '"' && !escapeNext) {
|
||||
inString = !inString;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (inString) continue;
|
||||
|
||||
if (char === "[") depth++;
|
||||
else if (char === "]") {
|
||||
depth--;
|
||||
if (depth === 0) {
|
||||
const candidate = trimmed.substring(start, i + 1);
|
||||
try {
|
||||
JSON.parse(candidate);
|
||||
return candidate;
|
||||
} catch {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback for arrays - validate before returning
|
||||
const firstBracket = trimmed.indexOf("[");
|
||||
const lastBracket = trimmed.lastIndexOf("]");
|
||||
if (firstBracket !== -1 && lastBracket > firstBracket) {
|
||||
return trimmed.substring(firstBracket, lastBracket + 1);
|
||||
const candidate = trimmed.substring(firstBracket, lastBracket + 1);
|
||||
try {
|
||||
JSON.parse(candidate);
|
||||
return candidate;
|
||||
} catch {
|
||||
// Not valid JSON
|
||||
}
|
||||
}
|
||||
|
||||
// No JSON boundaries found — return as-is and let the caller handle the error
|
||||
// No valid JSON found — return as-is and let the caller handle the error
|
||||
return trimmed;
|
||||
}
|
||||
|
||||
@@ -148,4 +148,85 @@ That's all I found.`;
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({ facts: ["test"] });
|
||||
});
|
||||
|
||||
it("strips <|end_of_text|> tokens from OpenRouter responses", () => {
|
||||
const input = '{"facts": ["test"]}<|end_of_text|>';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({ facts: ["test"] });
|
||||
});
|
||||
|
||||
it("strips <|eot_id|> tokens from OpenRouter responses", () => {
|
||||
const input = '{"facts": ["hello"]}<|eot_id|>';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({ facts: ["hello"] });
|
||||
});
|
||||
|
||||
it("strips <|im_end|> tokens from ChatML responses", () => {
|
||||
const input = '{"memory": [{"text": "test"}]}<|im_end|>';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({ memory: [{ text: "test" }] });
|
||||
});
|
||||
|
||||
it("strips multiple noise tokens", () => {
|
||||
const input =
|
||||
'<|im_start|>assistant\n{"facts": ["data"]}<|im_end|><|end_of_text|>';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({ facts: ["data"] });
|
||||
});
|
||||
|
||||
// Issue #4737: Leading text with braces that aren't JSON
|
||||
it("handles leading text containing braces before actual JSON", () => {
|
||||
const input = 'Here\'s the {formatted} output: {"facts": ["real data"]}';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({ facts: ["real data"] });
|
||||
});
|
||||
|
||||
it("handles multiple fake braces in leading text", () => {
|
||||
const input =
|
||||
"I'll format this {nicely} with {proper} structure:\n" +
|
||||
'{"memory": [{"id": "1", "text": "actual memory"}]}';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({
|
||||
memory: [{ id: "1", text: "actual memory" }],
|
||||
});
|
||||
});
|
||||
|
||||
it("handles incomplete JSON-like structures in leading text", () => {
|
||||
const input =
|
||||
"Based on {user preferences} I found:\n" +
|
||||
'{"facts": ["User likes TypeScript"]}';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({ facts: ["User likes TypeScript"] });
|
||||
});
|
||||
|
||||
it("validates JSON and skips malformed candidates", () => {
|
||||
const input = 'The result is {broken and {"facts": ["valid"]} is here';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({ facts: ["valid"] });
|
||||
});
|
||||
|
||||
it("handles deeply nested valid JSON after invalid starts", () => {
|
||||
const input =
|
||||
"Here {is some {context}} for you:\n" +
|
||||
'{"memory": [{"nested": {"deep": "value"}}]}';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({
|
||||
memory: [{ nested: { deep: "value" } }],
|
||||
});
|
||||
});
|
||||
|
||||
it("handles JSON with escaped quotes correctly", () => {
|
||||
const input =
|
||||
'Output: {"facts": ["User said \\"hello\\"", "Has a \\"test\\" project"]}';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual({
|
||||
facts: ['User said "hello"', 'Has a "test" project'],
|
||||
});
|
||||
});
|
||||
|
||||
it("handles arrays with leading brace-like text", () => {
|
||||
const input = 'Here\'s {some context}. The array is: ["fact1", "fact2"]';
|
||||
const result = extractJson(input);
|
||||
expect(JSON.parse(result)).toEqual(["fact1", "fact2"]);
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user