fix: local-model tool-call robustness (partial-JSON repair, fallback parser, empty-response retry)
Three upgrades targeting the most common local-model failure modes where a tool call is intended but never executes: #2 Partial/truncated-JSON recovery (native streaming): - src/toolcalling/partialJson.ts: cheap structural repair for tool-call arguments that fail JSON.parse — close unterminated strings, balance open braces/brackets (max_tokens truncation), strip trailing commas and stray trailing tokens. Never invents keys/values; a repaired call still goes through schema validation. - agent/loop.ts: on a parse failure in accumulated native tool calls, try repair before falling back to a full non-streaming regeneration (which is expensive on a local backend and fails identically when the cause was max_tokens). Repaired args replace the broken ones; only unrepairable calls trigger the retry. - partialJson.test.ts: 10 cases (truncation, trailing comma nesting, braces inside strings, escaped quotes, stray trailing content). #3 Fallback tool-call parser robustness: - src/toolcalling/fallbackParser.ts: now accepts ```tool_call blocks (multiline-anchored so an inner ```json fence isn't read as the terminator), ```json blocks, AND bare unfenced tool-call objects in prose. Inner ```json fences are stripped. Every candidate runs through partial-JSON repair, so a truncated fence (no closing ```) still recovers. Only accepts bare braces that contain "name"+"arguments" keys. - fallbackParser.test.ts: 11 cases. #6 Empty-response retry: - agent/loop.ts: a bare empty `stop` (no text, no tool calls — common from small/quantized local models) now retries once with a nudge instead of returning "" or hard-erroring. After the retry budget is exhausted, a genuine empty response throws "Empty response from model." - loop.test.ts: retry-then-succeed and retry-exhausted-throws cases. Verified: typecheck clean, build 251.48 KB, 242 tests pass (+22).
This commit is contained in:
@@ -59,6 +59,73 @@ describe("runTurn / max_tokens", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("runTurn / empty response retry", () => {
|
||||
it("retries once on a bare empty stop, then succeeds on the next call", async () => {
|
||||
// Small local models sometimes emit a stream that ends with finish_reason "stop" but no
|
||||
// content and no tool calls. Without a retry this aborted the whole turn as a hard error.
|
||||
// Here the first create() returns an empty stop; the second returns real text, so the turn
|
||||
// should recover and return that text (with a nudge message inserted in between).
|
||||
let call = 0;
|
||||
const fakeClient = {
|
||||
chat: {
|
||||
completions: {
|
||||
create: vi.fn(async (_params: any) => {
|
||||
call++;
|
||||
const chunk = call === 1
|
||||
? { choices: [{ delta: {}, finish_reason: "stop" }] } // empty stop, no content
|
||||
: { choices: [{ delta: { content: "all done" }, finish_reason: "stop" }] };
|
||||
let yielded = false;
|
||||
return {
|
||||
[Symbol.asyncIterator]: () => ({
|
||||
next: async () => {
|
||||
if (yielded) return { done: true, value: undefined };
|
||||
yielded = true;
|
||||
return { done: false, value: chunk };
|
||||
},
|
||||
}),
|
||||
};
|
||||
}),
|
||||
},
|
||||
},
|
||||
} as any;
|
||||
|
||||
const session = createSession(fakeClient, "test-model", process.cwd(), async () => "once", "native", []);
|
||||
const result = await runTurn(session, "hi", () => {});
|
||||
expect(result).toBe("all done");
|
||||
expect(call).toBe(2); // one empty + one successful
|
||||
// A nudge user message must have been inserted after the empty response.
|
||||
expect(session.messages.some((m) => m.role === "user" && typeof m.content === "string" && m.content.includes("empty response"))).toBe(true);
|
||||
});
|
||||
|
||||
it("throws after the retry budget is exhausted on repeated empty responses", async () => {
|
||||
let call = 0;
|
||||
const fakeClient = {
|
||||
chat: {
|
||||
completions: {
|
||||
create: vi.fn(async (_params: any) => {
|
||||
call++;
|
||||
let yielded = false;
|
||||
return {
|
||||
[Symbol.asyncIterator]: () => ({
|
||||
next: async () => {
|
||||
if (yielded) return { done: true, value: undefined };
|
||||
yielded = true;
|
||||
return { done: false, value: { choices: [{ delta: {}, finish_reason: "stop" }] } };
|
||||
},
|
||||
}),
|
||||
};
|
||||
}),
|
||||
},
|
||||
},
|
||||
} as any;
|
||||
|
||||
const session = createSession(fakeClient, "test-model", process.cwd(), async () => "once", "native", []);
|
||||
await expect(runTurn(session, "hi", () => {})).rejects.toThrow("Empty response from model.");
|
||||
// One initial empty + one retry = 2 calls total.
|
||||
expect(call).toBe(2);
|
||||
});
|
||||
});
|
||||
|
||||
describe("runTurn / max iterations", () => {
|
||||
it("throws MaxIterationsError (not a generic error) when a model keeps calling tools forever", async () => {
|
||||
// A no-op tool the fake model calls on every single turn, forever — simulates a model that
|
||||
|
||||
+39
-3
@@ -13,6 +13,7 @@ import { FALLBACK_RETRY_NUDGE } from "../toolcalling/fallbackPrompt.js";
|
||||
import { parseFallbackToolCalls } from "../toolcalling/fallbackParser.js";
|
||||
import { resolveToolCall } from "../toolcalling/nativeAdapter.js";
|
||||
import { resolveToolInvocation, runTool, type ResolvedToolCall } from "../toolcalling/resolve.js";
|
||||
import { repairPartialJson } from "../toolcalling/partialJson.js";
|
||||
import { notifyFileChanged } from "../codeintel/lspManager.js";
|
||||
import { formatCallLabel, summarizeToolResult } from "../ui/toolSummary.js";
|
||||
import { estimateTokens } from "../utils/tokens.js";
|
||||
@@ -122,6 +123,10 @@ export function createIdleAbort(
|
||||
}
|
||||
|
||||
const MAX_MALFORMED_RETRIES = 2;
|
||||
/** Local models (especially small quantized ones) occasionally emit a bare empty `stop`
|
||||
* chunk — no text, no tool calls. Rather than abort the whole turn as a hard error, retry
|
||||
* once with a short nudge so the model continues, mirroring the malformed-tool-call path. */
|
||||
const MAX_EMPTY_RESPONSE_RETRIES = 1;
|
||||
|
||||
// Sub-agent safety limits. The toolset already excludes `agent` for sub-agents (so a model can't
|
||||
// spawn nested sub-agents through normal tool use), but these are independent, explicit backstops
|
||||
@@ -884,6 +889,7 @@ export async function runTurn(
|
||||
if (session.subAgentDepth === 0) session.stats.turns++;
|
||||
session.mutationCommitLength = null;
|
||||
let malformedRetries = 0;
|
||||
let emptyResponseRetries = 0;
|
||||
|
||||
for (let i = 0; i < session.maxIterations; i++) {
|
||||
pruneOldImages(session);
|
||||
@@ -1026,13 +1032,23 @@ export async function runTurn(
|
||||
|
||||
// --- Handle native tool calls from streaming ---
|
||||
if (session.mode === "native" && accumulatedToolCalls.length > 0) {
|
||||
// Validate accumulated arguments — if any fail to parse (Ollama streaming bug),
|
||||
// retry the entire turn non-streaming
|
||||
// Validate accumulated arguments. Local-model streaming (Ollama/LM Studio deltas)
|
||||
// commonly truncates the JSON (max_tokens clipped mid-value) or leaves a trailing
|
||||
// comma / unbalanced brace. Before paying for a full non-streaming regeneration of
|
||||
// the turn (expensive on a local backend, and it fails identically when the cause was
|
||||
// max_tokens), try a cheap structural repair — close unterminated strings, balance
|
||||
// braces/brackets, strip stray trailing tokens. A repaired call still goes through
|
||||
// schema validation at resolve time, so a bad repair can't execute a wrong call.
|
||||
let allValid = true;
|
||||
for (const tc of accumulatedToolCalls) {
|
||||
try {
|
||||
JSON.parse(tc.arguments);
|
||||
} catch {
|
||||
const repaired = repairPartialJson(tc.arguments);
|
||||
if (repaired !== null) {
|
||||
tc.arguments = JSON.stringify(repaired);
|
||||
continue; // repaired successfully — keep checking the rest
|
||||
}
|
||||
allValid = false;
|
||||
break;
|
||||
}
|
||||
@@ -1160,6 +1176,19 @@ export async function runTurn(
|
||||
}
|
||||
}
|
||||
|
||||
// A bare empty `stop` (no text, no tool calls) — common from small/quantized local models.
|
||||
// Treat it as retryable rather than a final empty answer: nudge once and continue, so a
|
||||
// transient empty response doesn't end the task with no output (mirroring malformed path).
|
||||
if (!fullText.trim()) {
|
||||
if (emptyResponseRetries < MAX_EMPTY_RESPONSE_RETRIES) {
|
||||
emptyResponseRetries++;
|
||||
session.messages.push({ role: "user", content: "You returned an empty response. Please continue with your answer or call a tool." } as ChatCompletionMessageParam);
|
||||
continue;
|
||||
}
|
||||
// Retry budget exhausted — surface a hard error instead of silently returning "".
|
||||
throw new AgentError("Empty response from model.");
|
||||
}
|
||||
|
||||
// Final text answer
|
||||
emit({ type: "text_done", fullText });
|
||||
session.messages.push({ role: "assistant", content: fullText });
|
||||
@@ -1167,7 +1196,14 @@ export async function runTurn(
|
||||
return fullText;
|
||||
}
|
||||
|
||||
// Empty response with no tool calls and no text — this shouldn't happen normally
|
||||
// Empty response with no tool calls and no text. Small/quantized local models sometimes
|
||||
// emit a bare empty `stop` chunk; retry once with a short nudge before giving up, so a
|
||||
// transient empty response doesn't abort the whole turn (mirroring the malformed path).
|
||||
if (emptyResponseRetries < MAX_EMPTY_RESPONSE_RETRIES) {
|
||||
emptyResponseRetries++;
|
||||
session.messages.push({ role: "user", content: "You returned an empty response. Please continue with your answer or call a tool." } as ChatCompletionMessageParam);
|
||||
continue;
|
||||
}
|
||||
throw new AgentError("Empty response from model.");
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { parseFallbackToolCalls } from "./fallbackParser.js";
|
||||
|
||||
describe("parseFallbackToolCalls", () => {
|
||||
it("parses the instructed ```tool_call fenced format", () => {
|
||||
const out = parseFallbackToolCalls(
|
||||
'```tool_call\n{"name":"read_file","arguments":{"path":"src/a.ts"}}\n```',
|
||||
);
|
||||
expect(out.calls).toEqual([{ name: "read_file", arguments: { path: "src/a.ts" } }]);
|
||||
expect(out.malformed).toBe(false);
|
||||
});
|
||||
|
||||
it("returns no calls and not-malformed for plain prose with no tool attempt", () => {
|
||||
const out = parseFallbackToolCalls("Here is the answer: use foo().");
|
||||
expect(out.calls).toEqual([]);
|
||||
expect(out.malformed).toBe(false);
|
||||
});
|
||||
|
||||
it("flags a tool_call block whose JSON is unparseable as malformed", () => {
|
||||
const out = parseFallbackToolCalls("```tool_call\n{not valid json\n```");
|
||||
expect(out.calls).toEqual([]);
|
||||
expect(out.malformed).toBe(true);
|
||||
});
|
||||
|
||||
it("accepts a ```json fenced block when no tool_call block is present", () => {
|
||||
const out = parseFallbackToolCalls(
|
||||
'```json\n{"name":"grep","arguments":{"pattern":"foo"}}\n```',
|
||||
);
|
||||
expect(out.calls).toEqual([{ name: "grep", arguments: { pattern: "foo" } }]);
|
||||
expect(out.malformed).toBe(false);
|
||||
});
|
||||
|
||||
it("prefers a ```tool_call block over a ```json block when both appear", () => {
|
||||
const out = parseFallbackToolCalls(
|
||||
'```tool_call\n{"name":"read_file","arguments":{"path":"a"}}\n```\n' +
|
||||
'```json\n{"name":"grep","arguments":{"pattern":"x"}}\n```',
|
||||
);
|
||||
expect(out.calls).toHaveLength(1);
|
||||
expect(out.calls[0]!.name).toBe("read_file");
|
||||
});
|
||||
|
||||
it("does not treat a ```json fence inside a tool_call block as the terminator", () => {
|
||||
const out = parseFallbackToolCalls(
|
||||
"```tool_call\n```json\n{\"name\":\"read_file\",\"arguments\":{\"path\":\"a\"}}\n```\n```",
|
||||
);
|
||||
expect(out.calls).toEqual([{ name: "read_file", arguments: { path: "a" } }]);
|
||||
});
|
||||
|
||||
it("extracts a bare (unfenced) tool-call object from surrounding prose", () => {
|
||||
const out = parseFallbackToolCalls(
|
||||
'Let me read that file.\n{"name":"read_file","arguments":{"path":"src/loop.ts"}}\nThat should help.',
|
||||
);
|
||||
expect(out.calls).toEqual([{ name: "read_file", arguments: { path: "src/loop.ts" } }]);
|
||||
});
|
||||
|
||||
it("ignores bare braces in prose that don't look like a tool call", () => {
|
||||
const out = parseFallbackToolCalls("The config is { key: value } and that's it.");
|
||||
expect(out.calls).toEqual([]);
|
||||
expect(out.malformed).toBe(false);
|
||||
});
|
||||
|
||||
it("repairs a truncated JSON tool_call block via partial-JSON repair", () => {
|
||||
// max_tokens clipped the closing brace and quote
|
||||
const out = parseFallbackToolCalls('```tool_call\n{"name":"read_file","arguments":{"path":"src/lo');
|
||||
expect(out.calls).toEqual([{ name: "read_file", arguments: { path: "src/lo" } }]);
|
||||
});
|
||||
|
||||
it("accepts a tool call with omitted arguments as empty arguments", () => {
|
||||
const out = parseFallbackToolCalls('```tool_call\n{"name":"git_status"}\n```');
|
||||
expect(out.calls).toEqual([{ name: "git_status", arguments: {} }]);
|
||||
});
|
||||
|
||||
it("flags a tool call whose arguments is not an object", () => {
|
||||
const out = parseFallbackToolCalls('```tool_call\n{"name":"x","arguments":"foo"}\n```');
|
||||
expect(out.calls).toEqual([]);
|
||||
expect(out.malformed).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -1,3 +1,5 @@
|
||||
import { repairPartialJson } from "./partialJson.js";
|
||||
|
||||
export interface FallbackToolCall {
|
||||
name: string;
|
||||
arguments: Record<string, unknown>;
|
||||
@@ -8,25 +10,155 @@ export interface FallbackParseResult {
|
||||
malformed: boolean;
|
||||
}
|
||||
|
||||
const BLOCK_RE = /```tool_call\s*([\s\S]*?)```/g;
|
||||
// Local models in fallback mode (no native function calling) are asked to emit tool calls as a
|
||||
// fenced ```tool_call block. In practice they frequently deviate, so the parser is lenient about
|
||||
// FORMAT but strict about CONTENT: anything we extract must still parse to { name, arguments }.
|
||||
// Accepted shapes, in priority order:
|
||||
// 1. A ```tool_call fenced block (the instructed format). The opening fence is ```tool_call on
|
||||
// its own line; the closing fence is ``` on its own line — anchored so a ```json block INSIDE
|
||||
// isn't mistaken for the terminator. Some models wrap the JSON in an inner ```json fence; we
|
||||
// strip that inner fence before parsing.
|
||||
// 2. A ```json fenced block whose content is a tool-call object (name + arguments). Models that
|
||||
// ignore the custom "tool_call" fence name but reach for the familiar "json" one.
|
||||
// 3. A bare tool-call object appearing in the response with no fence at all. We scan for the
|
||||
// first balanced {...} that contains a string "name" and an object "arguments". To avoid
|
||||
// matching arbitrary prose-embedded JSON, we require the recognizable keys.
|
||||
//
|
||||
// Every extracted candidate goes through `coerce`, which parses (with partial-JSON repair as a
|
||||
// last resort for truncated streaming output) and validates the name/arguments shape. A candidate
|
||||
// that doesn't yield a valid call sets `malformed` — the loop nudges the model to retry rather than
|
||||
// silently ending the task with no tool executed.
|
||||
|
||||
// Opening fence ```tool_call on its own line; closing ``` on its own line. Multiline-anchored so
|
||||
// an inner ```json fence can't be read as the terminator.
|
||||
const TOOL_CALL_FENCE_RE = /^```tool_call\s*\n([\s\S]*?)\n```(?:\n|$)/gm;
|
||||
// A ```json block — only used if no ```tool_call block matched, since a json fence may carry prose.
|
||||
const JSON_FENCE_RE = /^```json\s*\n([\s\S]*?)\n```(?:\n|$)/gm;
|
||||
|
||||
/** Pull the textual content out of a fenced block, stripping any inner ```json fence a model may
|
||||
* have nested inside it. Returns the cleaned, trimmed body. */
|
||||
function cleanFencedBody(raw: string): string {
|
||||
return raw.replace(/^```(?:json)?\s*|\s*```$/g, "").trim();
|
||||
}
|
||||
|
||||
/** Parse a candidate string into a tool call, or null if it isn't one. Tries a direct parse, then a
|
||||
* partial-JSON repair (truncation / trailing comma / unbalanced braces) for clipped streaming. */
|
||||
function coerce(candidate: string): FallbackToolCall | null {
|
||||
const trimmed = candidate.trim();
|
||||
if (!trimmed) return null;
|
||||
// Direct parse first.
|
||||
let parsed: unknown = null;
|
||||
try {
|
||||
parsed = JSON.parse(trimmed);
|
||||
} catch {
|
||||
parsed = null;
|
||||
}
|
||||
// Repair pass for truncated / sloppy JSON from local streaming.
|
||||
if (parsed === null) parsed = repairPartialJson(trimmed);
|
||||
if (!parsed || typeof parsed !== "object") return null;
|
||||
const obj = parsed as Record<string, unknown>;
|
||||
if (typeof obj.name !== "string") return null;
|
||||
if (obj.arguments === undefined || obj.arguments === null) {
|
||||
// Some models omit arguments entirely when the tool takes none — treat as empty.
|
||||
return { name: obj.name, arguments: {} };
|
||||
}
|
||||
if (typeof obj.arguments !== "object" || Array.isArray(obj.arguments)) return null;
|
||||
return { name: obj.name, arguments: obj.arguments as Record<string, unknown> };
|
||||
}
|
||||
|
||||
/** Scan `content` for the first balanced {...} object containing a `"name"` string and an
|
||||
* `"arguments"` object, with no fence at all. Tracks string literals so braces inside strings
|
||||
* don't affect nesting, and cuts to the first complete top-level object. */
|
||||
function findBareToolCall(content: string): string | null {
|
||||
const start = content.indexOf("{");
|
||||
if (start < 0) return null;
|
||||
let depth = 0;
|
||||
let inStr = false;
|
||||
for (let i = start; i < content.length; i++) {
|
||||
const c = content[i]!;
|
||||
if (inStr) {
|
||||
if (c === "\\") {
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
if (c === '"') inStr = false;
|
||||
continue;
|
||||
}
|
||||
if (c === '"') {
|
||||
inStr = true;
|
||||
continue;
|
||||
}
|
||||
if (c === "{") depth++;
|
||||
else if (c === "}") {
|
||||
depth--;
|
||||
if (depth === 0) {
|
||||
const candidate = content.slice(start, i + 1);
|
||||
// Only accept it if it actually looks like a tool call — otherwise keep scanning.
|
||||
if (/"name"\s*:/.test(candidate) && /"arguments"\s*:/.test(candidate)) {
|
||||
return candidate;
|
||||
}
|
||||
// Reset to the next brace past this point to keep looking.
|
||||
const next = content.indexOf("{", i + 1);
|
||||
if (next < 0) return null;
|
||||
i = next - 1;
|
||||
depth = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Ran off the end with an unclosed object — a truncated tool call (max_tokens clipped the
|
||||
// closing brace, and possibly the closing fence too). Hand the unbalanced substring back;
|
||||
// `coerce` will run it through partial-JSON repair to close what's open.
|
||||
if (depth > 0) {
|
||||
const candidate = content.slice(start);
|
||||
if (/"name"\s*:/.test(candidate) && /"arguments"\s*:/.test(candidate)) {
|
||||
return candidate;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
export function parseFallbackToolCalls(content: string): FallbackParseResult {
|
||||
const calls: FallbackToolCall[] = [];
|
||||
let malformed = false;
|
||||
let sawAnyCandidate = false;
|
||||
|
||||
for (const match of content.matchAll(BLOCK_RE)) {
|
||||
const raw = match[1]?.trim() ?? "";
|
||||
try {
|
||||
const parsed = JSON.parse(raw);
|
||||
if (parsed && typeof parsed.name === "string" && typeof parsed.arguments === "object") {
|
||||
calls.push({ name: parsed.name, arguments: parsed.arguments ?? {} });
|
||||
} else {
|
||||
malformed = true;
|
||||
}
|
||||
} catch {
|
||||
malformed = true;
|
||||
// (1) ```tool_call fenced blocks (the instructed format) — there may be several.
|
||||
for (const match of content.matchAll(TOOL_CALL_FENCE_RE)) {
|
||||
sawAnyCandidate = true;
|
||||
const body = cleanFencedBody(match[1] ?? "");
|
||||
const call = coerce(body);
|
||||
if (call) calls.push(call);
|
||||
else malformed = true;
|
||||
}
|
||||
if (calls.length > 0) return { calls, malformed };
|
||||
|
||||
// (2) ```json fenced blocks — only if no tool_call block matched. Take the first that coerces.
|
||||
for (const match of content.matchAll(JSON_FENCE_RE)) {
|
||||
const call = coerce(cleanFencedBody(match[1] ?? ""));
|
||||
if (call) {
|
||||
calls.push(call);
|
||||
return { calls, malformed: false };
|
||||
}
|
||||
sawAnyCandidate = true;
|
||||
malformed = true;
|
||||
}
|
||||
if (calls.length > 0) return { calls, malformed };
|
||||
|
||||
// (3) Bare (unfenced) tool-call object — last resort. At most one: the prompt says one call per
|
||||
// response, and extracting multiple bare objects from prose is too error-prone.
|
||||
const bare = findBareToolCall(content);
|
||||
if (bare !== null) {
|
||||
sawAnyCandidate = true;
|
||||
const call = coerce(bare);
|
||||
if (call) {
|
||||
calls.push(call);
|
||||
return { calls, malformed: false };
|
||||
}
|
||||
malformed = true;
|
||||
}
|
||||
|
||||
// If we never saw anything that even looked like a tool-call attempt, that's not malformed —
|
||||
// the model simply answered in prose (no tool needed). `malformed` stays false.
|
||||
void sawAnyCandidate;
|
||||
return { calls, malformed };
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
import { describe, it, expect } from "vitest";
|
||||
import { repairPartialJson } from "./partialJson.js";
|
||||
|
||||
describe("repairPartialJson", () => {
|
||||
it("parses already-valid JSON unchanged", () => {
|
||||
expect(repairPartialJson('{"path":"src/a.ts"}')).toEqual({ path: "src/a.ts" });
|
||||
expect(repairPartialJson("[]")).toEqual([]);
|
||||
expect(repairPartialJson(" 42 ")).toBe(42);
|
||||
expect(repairPartialJson('{"a":1}\n')).toEqual({ a: 1 });
|
||||
});
|
||||
|
||||
it("returns null for empty / whitespace input", () => {
|
||||
expect(repairPartialJson("")).toBeNull();
|
||||
expect(repairPartialJson(" ")).toBeNull();
|
||||
});
|
||||
|
||||
it("strips a trailing comma before a closing bracket", () => {
|
||||
expect(repairPartialJson('{"a":1,}')).toEqual({ a: 1 });
|
||||
expect(repairPartialJson('[1,2,]')).toEqual([1, 2]);
|
||||
expect(repairPartialJson('{"a":{"b":2,},}')).toEqual({ a: { b: 2 } });
|
||||
});
|
||||
|
||||
it("strips stray trailing content after a complete value", () => {
|
||||
expect(repairPartialJson('{"a":1}\n```')).toEqual({ a: 1 });
|
||||
expect(repairPartialJson('{"a":1}garbage')).toEqual({ a: 1 });
|
||||
expect(repairPartialJson('[1,2] }')).toEqual([1, 2]);
|
||||
});
|
||||
|
||||
it("closes a truncated string value", () => {
|
||||
// max_tokens clipped mid-value: {"path":"src/lo → needs closing quote + brace
|
||||
expect(repairPartialJson('{"path":"src/lo')).toEqual({ path: "src/lo" });
|
||||
expect(repairPartialJson('{"a":"hello wor')).toEqual({ a: "hello wor" });
|
||||
});
|
||||
|
||||
it("balances unclosed braces and brackets from truncation", () => {
|
||||
expect(repairPartialJson('{"a":1')).toEqual({ a: 1 });
|
||||
expect(repairPartialJson('{"a":{"b":2')).toEqual({ a: { b: 2 } });
|
||||
expect(repairPartialJson("[1,2")).toEqual([1, 2]);
|
||||
expect(repairPartialJson('{"items":[1,2')).toEqual({ items: [1, 2] });
|
||||
});
|
||||
|
||||
it("does not count braces inside string literals", () => {
|
||||
// The braces/brackets inside the string are content, not nesting.
|
||||
expect(repairPartialJson('{"code":"func() { return [1] "')).toEqual({
|
||||
code: "func() { return [1] ",
|
||||
});
|
||||
expect(repairPartialJson('{"s":"\\\"escaped\\\""}')).toEqual({ s: '"escaped"' });
|
||||
});
|
||||
|
||||
it("handles escaped quotes inside strings during truncation repair", () => {
|
||||
// Unterminated string with an escaped quote inside: {"s":"a\"b
|
||||
expect(repairPartialJson('{"s":"a\\"b')).toEqual({ s: 'a"b' });
|
||||
});
|
||||
|
||||
it("combined: trailing comma exposed after balancing", () => {
|
||||
// {"a":1,"b":2, (truncated with trailing comma) → close brace, then strip comma
|
||||
expect(repairPartialJson('{"a":1,"b":2,')).toEqual({ a: 1, b: 2 });
|
||||
});
|
||||
|
||||
it("returns null when input is not salvageable as object/array/scalar", () => {
|
||||
expect(repairPartialJson("just prose with no json")).toBeNull();
|
||||
expect(repairPartialJson("{:}")).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,197 @@
|
||||
// Partial/truncated JSON repair for native streaming tool-call arguments.
|
||||
//
|
||||
// Local-model backends (Ollama, LM Studio) streaming tool calls accumulate the `arguments` string
|
||||
// across deltas. Two common pathologies produce a string that JSON.parse rejects but that contains
|
||||
// all the semantic content the model intended:
|
||||
//
|
||||
// 1. TRUNCATION — max_tokens clipped the JSON mid-value. The string ends inside a string value,
|
||||
// an array, or an object: `{"path": "src/lo`, `{"items": [1, 2`, `{"a": {"b": 1`.
|
||||
// 2. LOCAL-MODEL SLOPPINESS — a trailing comma, an unbalanced brace/bracket, or a trailing
|
||||
// garbage token after the closing brace: `{"path": "x.ts",}`, `{"a": 1 `.
|
||||
//
|
||||
// This module attempts a cheap, conservative repair BEFORE the caller falls back to a full
|
||||
// non-streaming regeneration (which is expensive on a local backend and often fails identically
|
||||
// when the cause was max_tokens). It only closes what's open and trims what's stray — it never
|
||||
// invents keys or values, so a genuinely malformed call still fails downstream at schema validation.
|
||||
//
|
||||
// The repair is best-effort: if it can't produce parseable JSON, it returns null and the caller
|
||||
// keeps its existing retry path. It is deliberately string-based (no AST) so it's trivially fast and
|
||||
// has no dependencies, and so it handles truncated input that a strict parser can't even build an
|
||||
// AST from.
|
||||
|
||||
/** Parse `s` as JSON; on success return the value, on failure return null (never throws). */
|
||||
function tryParse(s: string): unknown {
|
||||
try {
|
||||
return JSON.parse(s);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Skip past the next JSON string literal starting at `i` (the opening quote). Returns the index
|
||||
* just past the closing quote. Strings are the only place braces/brackets can appear without
|
||||
* affecting nesting, so we must not count them while inside one. Handles `\"` and other escapes. */
|
||||
function skipString(s: string, i: number): number {
|
||||
let j = i + 1; // past opening quote
|
||||
for (; j < s.length; j++) {
|
||||
const c = s[j]!;
|
||||
if (c === "\\") {
|
||||
j++; // skip the escaped char (covers \", \\, etc.)
|
||||
continue;
|
||||
}
|
||||
if (c === '"') return j + 1; // past closing quote
|
||||
}
|
||||
return j; // ran off the end — unterminated string
|
||||
}
|
||||
|
||||
/** Attempts to repair `raw` into parseable JSON. Returns the parsed value on success, or null if no
|
||||
* repair produced valid JSON. Steps, applied in order of how cheap and safe they are:
|
||||
*
|
||||
* 1. Maybe it already parses (trailing whitespace/newlines are fine for JSON.parse) — return as-is.
|
||||
* 2. Strip a trailing comma before an expected-but-absent `}` or `]` (common local-model slip).
|
||||
* 3. Strip stray non-JSON tokens after the first complete top-level value (`{"a":1}\n` → `{"a":1}`,
|
||||
* and `{"a":1}garbage` → `{"a":1}` — JSON.parse rejects trailing content, so trim to the first
|
||||
* complete value).
|
||||
* 4. Close unterminated strings, then balance still-open braces/brackets (truncation repair).
|
||||
*
|
||||
* Each step re-attempts a parse, so the cheapest fix that works wins. */
|
||||
export function repairPartialJson(raw: string): unknown | null {
|
||||
if (!raw) return null;
|
||||
const trimmed = raw.trim();
|
||||
if (!trimmed) return null;
|
||||
|
||||
// (1) Already valid?
|
||||
const direct = tryParse(trimmed);
|
||||
if (direct !== null) return direct;
|
||||
|
||||
// (2) Trailing comma before end-of-object/array: `{"a":1,}` or `[1,2,]`. Repeat until none
|
||||
// left so a nested shape like `{"a":{"b":2,},}` clears both commas (innermost-first).
|
||||
let noTrailing = trimmed;
|
||||
let prev: string;
|
||||
do {
|
||||
prev = noTrailing;
|
||||
noTrailing = noTrailing.replace(/,\s*([\]}]+\s*$)/, "$1");
|
||||
} while (noTrailing !== prev);
|
||||
if (noTrailing !== trimmed) {
|
||||
const v = tryParse(noTrailing);
|
||||
if (v !== null) return v;
|
||||
}
|
||||
|
||||
// (3) Stray trailing content after the first complete value. JSON.parse refuses trailing tokens,
|
||||
// but a model often emits a closing brace then a stray newline, a repeated token, or prose.
|
||||
// Find the end of the first balanced top-level value and cut there.
|
||||
const cut = cutToFirstCompleteValue(trimmed);
|
||||
if (cut !== null && cut !== trimmed) {
|
||||
const v = tryParse(cut);
|
||||
if (v !== null) return v;
|
||||
}
|
||||
|
||||
// (4) Truncation repair: close an unterminated string, then balance open braces/brackets.
|
||||
const balanced = balanceAndClose(trimmed);
|
||||
if (balanced !== null && balanced !== trimmed) {
|
||||
// Re-run the earlier cheap fixes on the balanced result (a trailing comma may now be exposed).
|
||||
const v = tryParse(balanced);
|
||||
if (v !== null) return v;
|
||||
const v2 = tryParse(balanced.replace(/,\s*([\]}]\s*$)/, "$1"));
|
||||
if (v2 !== null) return v2;
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
/** If `s` starts with a complete top-level JSON value followed by stray content, return just that
|
||||
* value (as a substring). Returns null if we can't find a clean boundary (e.g. the value is itself
|
||||
* truncated). Walks the string tracking string literals and nesting depth. */
|
||||
function cutToFirstCompleteValue(s: string): string | null {
|
||||
let i = 0;
|
||||
// Skip leading whitespace.
|
||||
while (i < s.length && /\s/.test(s[i]!)) i++;
|
||||
if (i >= s.length) return null;
|
||||
|
||||
const start = i;
|
||||
const stack: string[] = [];
|
||||
let inStr = false;
|
||||
|
||||
while (i < s.length) {
|
||||
const c = s[i]!;
|
||||
if (inStr) {
|
||||
if (c === "\\") {
|
||||
i += 2;
|
||||
continue;
|
||||
}
|
||||
if (c === '"') inStr = false;
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
if (c === '"') {
|
||||
inStr = true;
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
if (c === "{" || c === "[") {
|
||||
stack.push(c);
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
if (c === "}" || c === "]") {
|
||||
stack.pop();
|
||||
i++;
|
||||
// If the stack is empty, this was the end of the top-level value — cut here.
|
||||
if (stack.length === 0) return s.slice(start, i);
|
||||
continue;
|
||||
}
|
||||
// A bare scalar (number/true/false/null) ends at the next delimiter/comma/whitespace.
|
||||
if (stack.length === 0 && (c === "," || c === "}" || c === "]" || /\s/.test(c))) {
|
||||
return s.slice(start, i);
|
||||
}
|
||||
i++;
|
||||
}
|
||||
|
||||
// Ran off the end without closing the top-level value → it's truncated, not "complete + stray".
|
||||
if (stack.length > 0) return null;
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Closes an unterminated trailing string and balances any open braces/brackets. Returns the
|
||||
* repaired string, or null if nothing needed closing (caller can compare to skip a no-op parse). */
|
||||
function balanceAndClose(s: string): string | null {
|
||||
let out = s;
|
||||
const stack: string[] = [];
|
||||
let inStr = false;
|
||||
let i = 0;
|
||||
|
||||
for (; i < out.length; i++) {
|
||||
const c = out[i]!;
|
||||
if (inStr) {
|
||||
if (c === "\\") {
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
if (c === '"') inStr = false;
|
||||
continue;
|
||||
}
|
||||
if (c === '"') {
|
||||
inStr = true;
|
||||
continue;
|
||||
}
|
||||
if (c === "{") stack.push("}");
|
||||
else if (c === "[") stack.push("]");
|
||||
else if (c === "}" || c === "]") stack.pop();
|
||||
}
|
||||
|
||||
// If we ended inside a string, close it. A truncated value like `{"path":"src/lo` needs a closing
|
||||
// quote before we can balance the outer braces.
|
||||
if (inStr) {
|
||||
out += '"';
|
||||
}
|
||||
|
||||
// Close anything still open, innermost-first. Truncation mid-array/object → append the closers.
|
||||
if (stack.length === 0 && !inStr) {
|
||||
// Nothing to close — but a trailing comma may have been the only issue; let the caller handle it.
|
||||
return inStr ? out : null;
|
||||
}
|
||||
while (stack.length) {
|
||||
out += stack.pop();
|
||||
}
|
||||
return out;
|
||||
}
|
||||
Reference in New Issue
Block a user