docs: update memory tool list, CLI usage, and config file reading logic (#4861)

Co-authored-by: Livia Ellen <liviaellen@msn.com>
This commit is contained in:
Kartik
2026-04-20 20:09:45 +05:30
committed by GitHub
parent 5520226b5b
commit 4e611e8dba
31 changed files with 1531 additions and 365 deletions
+134 -12
View File
@@ -883,38 +883,160 @@ export function removeCodeBlocks(text: string): string {
/**
* Extracts a JSON object from text that may be wrapped in explanation text.
*
* Some LLMs (especially local models like Ollama/LM Studio) return JSON
* wrapped in conversational text without code fences, e.g.:
* Some LLMs (especially local models like Ollama/LM Studio, or OpenRouter)
* return JSON wrapped in conversational text without code fences, e.g.:
*
* "Here are the facts I extracted:\n{\"facts\": [\"fact1\"]}\nI hope this helps!"
*
* This function first tries `removeCodeBlocks` for code-fence-wrapped JSON,
* then falls back to locating the first `{` and last `}` to extract the
* outermost JSON object.
* This function:
* 1. Strips known noise tokens from OpenRouter and other providers
* 2. Removes code fences and <think> blocks
* 3. Tries to find a valid JSON object by testing each `{` as a starting point
* 4. Falls back to first/last brace matching if validation isn't possible
*
* @param text - The raw LLM response text
* @returns The extracted JSON string, or the original text if no JSON object
* boundaries are found
*/
export function extractJson(text: string): string {
// Step 1: Strip code fences if present
const cleaned = removeCodeBlocks(text);
// Step 1: Strip known noise tokens from OpenRouter/local models
let cleaned = text
.replace(/<\|end_of_text\|>/g, "")
.replace(/<\|eot_id\|>/g, "")
.replace(/<\|im_end\|>/g, "")
.replace(/<\|im_start\|>/g, "")
.replace(/<\|endoftext\|>/g, "");
// Step 2: Strip code fences and <think> blocks
cleaned = removeCodeBlocks(cleaned);
const trimmed = cleaned.trim();
// Step 2: Try to locate a JSON object by first `{` and last `}` boundaries
if (!trimmed) return "";
// Step 3: Try to find valid JSON object by testing each `{` as potential start
// This handles cases like "Here's the {formatted} output: {...actual json...}"
const braceIndices: number[] = [];
for (let i = 0; i < trimmed.length; i++) {
if (trimmed[i] === "{") braceIndices.push(i);
}
for (const start of braceIndices) {
// Find the matching closing brace by tracking depth
let depth = 0;
let inString = false;
let escapeNext = false;
for (let i = start; i < trimmed.length; i++) {
const char = trimmed[i];
if (escapeNext) {
escapeNext = false;
continue;
}
if (char === "\\") {
escapeNext = true;
continue;
}
if (char === '"' && !escapeNext) {
inString = !inString;
continue;
}
if (inString) continue;
if (char === "{") depth++;
else if (char === "}") {
depth--;
if (depth === 0) {
const candidate = trimmed.substring(start, i + 1);
try {
JSON.parse(candidate);
return candidate; // Valid JSON found
} catch {
// Not valid JSON, try next starting brace
break;
}
}
}
}
}
// Step 4: Fallback - try first/last brace (original behavior for edge cases)
// Only use this if it produces valid JSON
const firstBrace = trimmed.indexOf("{");
const lastBrace = trimmed.lastIndexOf("}");
if (firstBrace !== -1 && lastBrace > firstBrace) {
return trimmed.substring(firstBrace, lastBrace + 1);
const candidate = trimmed.substring(firstBrace, lastBrace + 1);
try {
JSON.parse(candidate);
return candidate;
} catch {
// Not valid JSON, continue to array extraction
}
}
// Step 3: Try to locate a JSON array by first `[` and last `]` boundaries
// Step 5: Try to locate a JSON array by testing each `[` as potential start
const bracketIndices: number[] = [];
for (let i = 0; i < trimmed.length; i++) {
if (trimmed[i] === "[") bracketIndices.push(i);
}
for (const start of bracketIndices) {
let depth = 0;
let inString = false;
let escapeNext = false;
for (let i = start; i < trimmed.length; i++) {
const char = trimmed[i];
if (escapeNext) {
escapeNext = false;
continue;
}
if (char === "\\") {
escapeNext = true;
continue;
}
if (char === '"' && !escapeNext) {
inString = !inString;
continue;
}
if (inString) continue;
if (char === "[") depth++;
else if (char === "]") {
depth--;
if (depth === 0) {
const candidate = trimmed.substring(start, i + 1);
try {
JSON.parse(candidate);
return candidate;
} catch {
break;
}
}
}
}
}
// Fallback for arrays - validate before returning
const firstBracket = trimmed.indexOf("[");
const lastBracket = trimmed.lastIndexOf("]");
if (firstBracket !== -1 && lastBracket > firstBracket) {
return trimmed.substring(firstBracket, lastBracket + 1);
const candidate = trimmed.substring(firstBracket, lastBracket + 1);
try {
JSON.parse(candidate);
return candidate;
} catch {
// Not valid JSON
}
}
// No JSON boundaries found — return as-is and let the caller handle the error
// No valid JSON found — return as-is and let the caller handle the error
return trimmed;
}
@@ -148,4 +148,85 @@ That's all I found.`;
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({ facts: ["test"] });
});
it("strips <|end_of_text|> tokens from OpenRouter responses", () => {
const input = '{"facts": ["test"]}<|end_of_text|>';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({ facts: ["test"] });
});
it("strips <|eot_id|> tokens from OpenRouter responses", () => {
const input = '{"facts": ["hello"]}<|eot_id|>';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({ facts: ["hello"] });
});
it("strips <|im_end|> tokens from ChatML responses", () => {
const input = '{"memory": [{"text": "test"}]}<|im_end|>';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({ memory: [{ text: "test" }] });
});
it("strips multiple noise tokens", () => {
const input =
'<|im_start|>assistant\n{"facts": ["data"]}<|im_end|><|end_of_text|>';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({ facts: ["data"] });
});
// Issue #4737: Leading text with braces that aren't JSON
it("handles leading text containing braces before actual JSON", () => {
const input = 'Here\'s the {formatted} output: {"facts": ["real data"]}';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({ facts: ["real data"] });
});
it("handles multiple fake braces in leading text", () => {
const input =
"I'll format this {nicely} with {proper} structure:\n" +
'{"memory": [{"id": "1", "text": "actual memory"}]}';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({
memory: [{ id: "1", text: "actual memory" }],
});
});
it("handles incomplete JSON-like structures in leading text", () => {
const input =
"Based on {user preferences} I found:\n" +
'{"facts": ["User likes TypeScript"]}';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({ facts: ["User likes TypeScript"] });
});
it("validates JSON and skips malformed candidates", () => {
const input = 'The result is {broken and {"facts": ["valid"]} is here';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({ facts: ["valid"] });
});
it("handles deeply nested valid JSON after invalid starts", () => {
const input =
"Here {is some {context}} for you:\n" +
'{"memory": [{"nested": {"deep": "value"}}]}';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({
memory: [{ nested: { deep: "value" } }],
});
});
it("handles JSON with escaped quotes correctly", () => {
const input =
'Output: {"facts": ["User said \\"hello\\"", "Has a \\"test\\" project"]}';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual({
facts: ['User said "hello"', 'Has a "test" project'],
});
});
it("handles arrays with leading brace-like text", () => {
const input = 'Here\'s {some context}. The array is: ["fact1", "fact2"]';
const result = extractJson(input);
expect(JSON.parse(result)).toEqual(["fact1", "fact2"]);
});
});