diff --git a/src/agent/loop.test.ts b/src/agent/loop.test.ts index 6d98ff9..80a9ccf 100644 --- a/src/agent/loop.test.ts +++ b/src/agent/loop.test.ts @@ -59,6 +59,73 @@ describe("runTurn / max_tokens", () => { }); }); +describe("runTurn / empty response retry", () => { + it("retries once on a bare empty stop, then succeeds on the next call", async () => { + // Small local models sometimes emit a stream that ends with finish_reason "stop" but no + // content and no tool calls. Without a retry this aborted the whole turn as a hard error. + // Here the first create() returns an empty stop; the second returns real text, so the turn + // should recover and return that text (with a nudge message inserted in between). + let call = 0; + const fakeClient = { + chat: { + completions: { + create: vi.fn(async (_params: any) => { + call++; + const chunk = call === 1 + ? { choices: [{ delta: {}, finish_reason: "stop" }] } // empty stop, no content + : { choices: [{ delta: { content: "all done" }, finish_reason: "stop" }] }; + let yielded = false; + return { + [Symbol.asyncIterator]: () => ({ + next: async () => { + if (yielded) return { done: true, value: undefined }; + yielded = true; + return { done: false, value: chunk }; + }, + }), + }; + }), + }, + }, + } as any; + + const session = createSession(fakeClient, "test-model", process.cwd(), async () => "once", "native", []); + const result = await runTurn(session, "hi", () => {}); + expect(result).toBe("all done"); + expect(call).toBe(2); // one empty + one successful + // A nudge user message must have been inserted after the empty response. + expect(session.messages.some((m) => m.role === "user" && typeof m.content === "string" && m.content.includes("empty response"))).toBe(true); + }); + + it("throws after the retry budget is exhausted on repeated empty responses", async () => { + let call = 0; + const fakeClient = { + chat: { + completions: { + create: vi.fn(async (_params: any) => { + call++; + let yielded = false; + return { + [Symbol.asyncIterator]: () => ({ + next: async () => { + if (yielded) return { done: true, value: undefined }; + yielded = true; + return { done: false, value: { choices: [{ delta: {}, finish_reason: "stop" }] } }; + }, + }), + }; + }), + }, + }, + } as any; + + const session = createSession(fakeClient, "test-model", process.cwd(), async () => "once", "native", []); + await expect(runTurn(session, "hi", () => {})).rejects.toThrow("Empty response from model."); + // One initial empty + one retry = 2 calls total. + expect(call).toBe(2); + }); +}); + describe("runTurn / max iterations", () => { it("throws MaxIterationsError (not a generic error) when a model keeps calling tools forever", async () => { // A no-op tool the fake model calls on every single turn, forever — simulates a model that diff --git a/src/agent/loop.ts b/src/agent/loop.ts index f0f1ce6..275e243 100644 --- a/src/agent/loop.ts +++ b/src/agent/loop.ts @@ -13,6 +13,7 @@ import { FALLBACK_RETRY_NUDGE } from "../toolcalling/fallbackPrompt.js"; import { parseFallbackToolCalls } from "../toolcalling/fallbackParser.js"; import { resolveToolCall } from "../toolcalling/nativeAdapter.js"; import { resolveToolInvocation, runTool, type ResolvedToolCall } from "../toolcalling/resolve.js"; +import { repairPartialJson } from "../toolcalling/partialJson.js"; import { notifyFileChanged } from "../codeintel/lspManager.js"; import { formatCallLabel, summarizeToolResult } from "../ui/toolSummary.js"; import { estimateTokens } from "../utils/tokens.js"; @@ -122,6 +123,10 @@ export function createIdleAbort( } const MAX_MALFORMED_RETRIES = 2; +/** Local models (especially small quantized ones) occasionally emit a bare empty `stop` + * chunk — no text, no tool calls. Rather than abort the whole turn as a hard error, retry + * once with a short nudge so the model continues, mirroring the malformed-tool-call path. */ +const MAX_EMPTY_RESPONSE_RETRIES = 1; // Sub-agent safety limits. The toolset already excludes `agent` for sub-agents (so a model can't // spawn nested sub-agents through normal tool use), but these are independent, explicit backstops @@ -884,6 +889,7 @@ export async function runTurn( if (session.subAgentDepth === 0) session.stats.turns++; session.mutationCommitLength = null; let malformedRetries = 0; + let emptyResponseRetries = 0; for (let i = 0; i < session.maxIterations; i++) { pruneOldImages(session); @@ -1026,13 +1032,23 @@ export async function runTurn( // --- Handle native tool calls from streaming --- if (session.mode === "native" && accumulatedToolCalls.length > 0) { - // Validate accumulated arguments — if any fail to parse (Ollama streaming bug), - // retry the entire turn non-streaming + // Validate accumulated arguments. Local-model streaming (Ollama/LM Studio deltas) + // commonly truncates the JSON (max_tokens clipped mid-value) or leaves a trailing + // comma / unbalanced brace. Before paying for a full non-streaming regeneration of + // the turn (expensive on a local backend, and it fails identically when the cause was + // max_tokens), try a cheap structural repair — close unterminated strings, balance + // braces/brackets, strip stray trailing tokens. A repaired call still goes through + // schema validation at resolve time, so a bad repair can't execute a wrong call. let allValid = true; for (const tc of accumulatedToolCalls) { try { JSON.parse(tc.arguments); } catch { + const repaired = repairPartialJson(tc.arguments); + if (repaired !== null) { + tc.arguments = JSON.stringify(repaired); + continue; // repaired successfully — keep checking the rest + } allValid = false; break; } @@ -1160,6 +1176,19 @@ export async function runTurn( } } + // A bare empty `stop` (no text, no tool calls) — common from small/quantized local models. + // Treat it as retryable rather than a final empty answer: nudge once and continue, so a + // transient empty response doesn't end the task with no output (mirroring malformed path). + if (!fullText.trim()) { + if (emptyResponseRetries < MAX_EMPTY_RESPONSE_RETRIES) { + emptyResponseRetries++; + session.messages.push({ role: "user", content: "You returned an empty response. Please continue with your answer or call a tool." } as ChatCompletionMessageParam); + continue; + } + // Retry budget exhausted — surface a hard error instead of silently returning "". + throw new AgentError("Empty response from model."); + } + // Final text answer emit({ type: "text_done", fullText }); session.messages.push({ role: "assistant", content: fullText }); @@ -1167,7 +1196,14 @@ export async function runTurn( return fullText; } - // Empty response with no tool calls and no text — this shouldn't happen normally + // Empty response with no tool calls and no text. Small/quantized local models sometimes + // emit a bare empty `stop` chunk; retry once with a short nudge before giving up, so a + // transient empty response doesn't abort the whole turn (mirroring the malformed path). + if (emptyResponseRetries < MAX_EMPTY_RESPONSE_RETRIES) { + emptyResponseRetries++; + session.messages.push({ role: "user", content: "You returned an empty response. Please continue with your answer or call a tool." } as ChatCompletionMessageParam); + continue; + } throw new AgentError("Empty response from model."); } diff --git a/src/toolcalling/fallbackParser.test.ts b/src/toolcalling/fallbackParser.test.ts new file mode 100644 index 0000000..408819c --- /dev/null +++ b/src/toolcalling/fallbackParser.test.ts @@ -0,0 +1,78 @@ +import { describe, it, expect } from "vitest"; +import { parseFallbackToolCalls } from "./fallbackParser.js"; + +describe("parseFallbackToolCalls", () => { + it("parses the instructed ```tool_call fenced format", () => { + const out = parseFallbackToolCalls( + '```tool_call\n{"name":"read_file","arguments":{"path":"src/a.ts"}}\n```', + ); + expect(out.calls).toEqual([{ name: "read_file", arguments: { path: "src/a.ts" } }]); + expect(out.malformed).toBe(false); + }); + + it("returns no calls and not-malformed for plain prose with no tool attempt", () => { + const out = parseFallbackToolCalls("Here is the answer: use foo()."); + expect(out.calls).toEqual([]); + expect(out.malformed).toBe(false); + }); + + it("flags a tool_call block whose JSON is unparseable as malformed", () => { + const out = parseFallbackToolCalls("```tool_call\n{not valid json\n```"); + expect(out.calls).toEqual([]); + expect(out.malformed).toBe(true); + }); + + it("accepts a ```json fenced block when no tool_call block is present", () => { + const out = parseFallbackToolCalls( + '```json\n{"name":"grep","arguments":{"pattern":"foo"}}\n```', + ); + expect(out.calls).toEqual([{ name: "grep", arguments: { pattern: "foo" } }]); + expect(out.malformed).toBe(false); + }); + + it("prefers a ```tool_call block over a ```json block when both appear", () => { + const out = parseFallbackToolCalls( + '```tool_call\n{"name":"read_file","arguments":{"path":"a"}}\n```\n' + + '```json\n{"name":"grep","arguments":{"pattern":"x"}}\n```', + ); + expect(out.calls).toHaveLength(1); + expect(out.calls[0]!.name).toBe("read_file"); + }); + + it("does not treat a ```json fence inside a tool_call block as the terminator", () => { + const out = parseFallbackToolCalls( + "```tool_call\n```json\n{\"name\":\"read_file\",\"arguments\":{\"path\":\"a\"}}\n```\n```", + ); + expect(out.calls).toEqual([{ name: "read_file", arguments: { path: "a" } }]); + }); + + it("extracts a bare (unfenced) tool-call object from surrounding prose", () => { + const out = parseFallbackToolCalls( + 'Let me read that file.\n{"name":"read_file","arguments":{"path":"src/loop.ts"}}\nThat should help.', + ); + expect(out.calls).toEqual([{ name: "read_file", arguments: { path: "src/loop.ts" } }]); + }); + + it("ignores bare braces in prose that don't look like a tool call", () => { + const out = parseFallbackToolCalls("The config is { key: value } and that's it."); + expect(out.calls).toEqual([]); + expect(out.malformed).toBe(false); + }); + + it("repairs a truncated JSON tool_call block via partial-JSON repair", () => { + // max_tokens clipped the closing brace and quote + const out = parseFallbackToolCalls('```tool_call\n{"name":"read_file","arguments":{"path":"src/lo'); + expect(out.calls).toEqual([{ name: "read_file", arguments: { path: "src/lo" } }]); + }); + + it("accepts a tool call with omitted arguments as empty arguments", () => { + const out = parseFallbackToolCalls('```tool_call\n{"name":"git_status"}\n```'); + expect(out.calls).toEqual([{ name: "git_status", arguments: {} }]); + }); + + it("flags a tool call whose arguments is not an object", () => { + const out = parseFallbackToolCalls('```tool_call\n{"name":"x","arguments":"foo"}\n```'); + expect(out.calls).toEqual([]); + expect(out.malformed).toBe(true); + }); +}); \ No newline at end of file diff --git a/src/toolcalling/fallbackParser.ts b/src/toolcalling/fallbackParser.ts index 1158f6e..ed02e03 100644 --- a/src/toolcalling/fallbackParser.ts +++ b/src/toolcalling/fallbackParser.ts @@ -1,3 +1,5 @@ +import { repairPartialJson } from "./partialJson.js"; + export interface FallbackToolCall { name: string; arguments: Record; @@ -8,25 +10,155 @@ export interface FallbackParseResult { malformed: boolean; } -const BLOCK_RE = /```tool_call\s*([\s\S]*?)```/g; +// Local models in fallback mode (no native function calling) are asked to emit tool calls as a +// fenced ```tool_call block. In practice they frequently deviate, so the parser is lenient about +// FORMAT but strict about CONTENT: anything we extract must still parse to { name, arguments }. +// Accepted shapes, in priority order: +// 1. A ```tool_call fenced block (the instructed format). The opening fence is ```tool_call on +// its own line; the closing fence is ``` on its own line — anchored so a ```json block INSIDE +// isn't mistaken for the terminator. Some models wrap the JSON in an inner ```json fence; we +// strip that inner fence before parsing. +// 2. A ```json fenced block whose content is a tool-call object (name + arguments). Models that +// ignore the custom "tool_call" fence name but reach for the familiar "json" one. +// 3. A bare tool-call object appearing in the response with no fence at all. We scan for the +// first balanced {...} that contains a string "name" and an object "arguments". To avoid +// matching arbitrary prose-embedded JSON, we require the recognizable keys. +// +// Every extracted candidate goes through `coerce`, which parses (with partial-JSON repair as a +// last resort for truncated streaming output) and validates the name/arguments shape. A candidate +// that doesn't yield a valid call sets `malformed` — the loop nudges the model to retry rather than +// silently ending the task with no tool executed. + +// Opening fence ```tool_call on its own line; closing ``` on its own line. Multiline-anchored so +// an inner ```json fence can't be read as the terminator. +const TOOL_CALL_FENCE_RE = /^```tool_call\s*\n([\s\S]*?)\n```(?:\n|$)/gm; +// A ```json block — only used if no ```tool_call block matched, since a json fence may carry prose. +const JSON_FENCE_RE = /^```json\s*\n([\s\S]*?)\n```(?:\n|$)/gm; + +/** Pull the textual content out of a fenced block, stripping any inner ```json fence a model may + * have nested inside it. Returns the cleaned, trimmed body. */ +function cleanFencedBody(raw: string): string { + return raw.replace(/^```(?:json)?\s*|\s*```$/g, "").trim(); +} + +/** Parse a candidate string into a tool call, or null if it isn't one. Tries a direct parse, then a + * partial-JSON repair (truncation / trailing comma / unbalanced braces) for clipped streaming. */ +function coerce(candidate: string): FallbackToolCall | null { + const trimmed = candidate.trim(); + if (!trimmed) return null; + // Direct parse first. + let parsed: unknown = null; + try { + parsed = JSON.parse(trimmed); + } catch { + parsed = null; + } + // Repair pass for truncated / sloppy JSON from local streaming. + if (parsed === null) parsed = repairPartialJson(trimmed); + if (!parsed || typeof parsed !== "object") return null; + const obj = parsed as Record; + if (typeof obj.name !== "string") return null; + if (obj.arguments === undefined || obj.arguments === null) { + // Some models omit arguments entirely when the tool takes none — treat as empty. + return { name: obj.name, arguments: {} }; + } + if (typeof obj.arguments !== "object" || Array.isArray(obj.arguments)) return null; + return { name: obj.name, arguments: obj.arguments as Record }; +} + +/** Scan `content` for the first balanced {...} object containing a `"name"` string and an + * `"arguments"` object, with no fence at all. Tracks string literals so braces inside strings + * don't affect nesting, and cuts to the first complete top-level object. */ +function findBareToolCall(content: string): string | null { + const start = content.indexOf("{"); + if (start < 0) return null; + let depth = 0; + let inStr = false; + for (let i = start; i < content.length; i++) { + const c = content[i]!; + if (inStr) { + if (c === "\\") { + i++; + continue; + } + if (c === '"') inStr = false; + continue; + } + if (c === '"') { + inStr = true; + continue; + } + if (c === "{") depth++; + else if (c === "}") { + depth--; + if (depth === 0) { + const candidate = content.slice(start, i + 1); + // Only accept it if it actually looks like a tool call — otherwise keep scanning. + if (/"name"\s*:/.test(candidate) && /"arguments"\s*:/.test(candidate)) { + return candidate; + } + // Reset to the next brace past this point to keep looking. + const next = content.indexOf("{", i + 1); + if (next < 0) return null; + i = next - 1; + depth = 0; + } + } + } + // Ran off the end with an unclosed object — a truncated tool call (max_tokens clipped the + // closing brace, and possibly the closing fence too). Hand the unbalanced substring back; + // `coerce` will run it through partial-JSON repair to close what's open. + if (depth > 0) { + const candidate = content.slice(start); + if (/"name"\s*:/.test(candidate) && /"arguments"\s*:/.test(candidate)) { + return candidate; + } + } + return null; +} export function parseFallbackToolCalls(content: string): FallbackParseResult { const calls: FallbackToolCall[] = []; let malformed = false; + let sawAnyCandidate = false; - for (const match of content.matchAll(BLOCK_RE)) { - const raw = match[1]?.trim() ?? ""; - try { - const parsed = JSON.parse(raw); - if (parsed && typeof parsed.name === "string" && typeof parsed.arguments === "object") { - calls.push({ name: parsed.name, arguments: parsed.arguments ?? {} }); - } else { - malformed = true; - } - } catch { - malformed = true; + // (1) ```tool_call fenced blocks (the instructed format) — there may be several. + for (const match of content.matchAll(TOOL_CALL_FENCE_RE)) { + sawAnyCandidate = true; + const body = cleanFencedBody(match[1] ?? ""); + const call = coerce(body); + if (call) calls.push(call); + else malformed = true; + } + if (calls.length > 0) return { calls, malformed }; + + // (2) ```json fenced blocks — only if no tool_call block matched. Take the first that coerces. + for (const match of content.matchAll(JSON_FENCE_RE)) { + const call = coerce(cleanFencedBody(match[1] ?? "")); + if (call) { + calls.push(call); + return { calls, malformed: false }; } + sawAnyCandidate = true; + malformed = true; + } + if (calls.length > 0) return { calls, malformed }; + + // (3) Bare (unfenced) tool-call object — last resort. At most one: the prompt says one call per + // response, and extracting multiple bare objects from prose is too error-prone. + const bare = findBareToolCall(content); + if (bare !== null) { + sawAnyCandidate = true; + const call = coerce(bare); + if (call) { + calls.push(call); + return { calls, malformed: false }; + } + malformed = true; } + // If we never saw anything that even looked like a tool-call attempt, that's not malformed — + // the model simply answered in prose (no tool needed). `malformed` stays false. + void sawAnyCandidate; return { calls, malformed }; -} +} \ No newline at end of file diff --git a/src/toolcalling/partialJson.test.ts b/src/toolcalling/partialJson.test.ts new file mode 100644 index 0000000..f6556bb --- /dev/null +++ b/src/toolcalling/partialJson.test.ts @@ -0,0 +1,64 @@ +import { describe, it, expect } from "vitest"; +import { repairPartialJson } from "./partialJson.js"; + +describe("repairPartialJson", () => { + it("parses already-valid JSON unchanged", () => { + expect(repairPartialJson('{"path":"src/a.ts"}')).toEqual({ path: "src/a.ts" }); + expect(repairPartialJson("[]")).toEqual([]); + expect(repairPartialJson(" 42 ")).toBe(42); + expect(repairPartialJson('{"a":1}\n')).toEqual({ a: 1 }); + }); + + it("returns null for empty / whitespace input", () => { + expect(repairPartialJson("")).toBeNull(); + expect(repairPartialJson(" ")).toBeNull(); + }); + + it("strips a trailing comma before a closing bracket", () => { + expect(repairPartialJson('{"a":1,}')).toEqual({ a: 1 }); + expect(repairPartialJson('[1,2,]')).toEqual([1, 2]); + expect(repairPartialJson('{"a":{"b":2,},}')).toEqual({ a: { b: 2 } }); + }); + + it("strips stray trailing content after a complete value", () => { + expect(repairPartialJson('{"a":1}\n```')).toEqual({ a: 1 }); + expect(repairPartialJson('{"a":1}garbage')).toEqual({ a: 1 }); + expect(repairPartialJson('[1,2] }')).toEqual([1, 2]); + }); + + it("closes a truncated string value", () => { + // max_tokens clipped mid-value: {"path":"src/lo → needs closing quote + brace + expect(repairPartialJson('{"path":"src/lo')).toEqual({ path: "src/lo" }); + expect(repairPartialJson('{"a":"hello wor')).toEqual({ a: "hello wor" }); + }); + + it("balances unclosed braces and brackets from truncation", () => { + expect(repairPartialJson('{"a":1')).toEqual({ a: 1 }); + expect(repairPartialJson('{"a":{"b":2')).toEqual({ a: { b: 2 } }); + expect(repairPartialJson("[1,2")).toEqual([1, 2]); + expect(repairPartialJson('{"items":[1,2')).toEqual({ items: [1, 2] }); + }); + + it("does not count braces inside string literals", () => { + // The braces/brackets inside the string are content, not nesting. + expect(repairPartialJson('{"code":"func() { return [1] "')).toEqual({ + code: "func() { return [1] ", + }); + expect(repairPartialJson('{"s":"\\\"escaped\\\""}')).toEqual({ s: '"escaped"' }); + }); + + it("handles escaped quotes inside strings during truncation repair", () => { + // Unterminated string with an escaped quote inside: {"s":"a\"b + expect(repairPartialJson('{"s":"a\\"b')).toEqual({ s: 'a"b' }); + }); + + it("combined: trailing comma exposed after balancing", () => { + // {"a":1,"b":2, (truncated with trailing comma) → close brace, then strip comma + expect(repairPartialJson('{"a":1,"b":2,')).toEqual({ a: 1, b: 2 }); + }); + + it("returns null when input is not salvageable as object/array/scalar", () => { + expect(repairPartialJson("just prose with no json")).toBeNull(); + expect(repairPartialJson("{:}")).toBeNull(); + }); +}); \ No newline at end of file diff --git a/src/toolcalling/partialJson.ts b/src/toolcalling/partialJson.ts new file mode 100644 index 0000000..e3cb960 --- /dev/null +++ b/src/toolcalling/partialJson.ts @@ -0,0 +1,197 @@ +// Partial/truncated JSON repair for native streaming tool-call arguments. +// +// Local-model backends (Ollama, LM Studio) streaming tool calls accumulate the `arguments` string +// across deltas. Two common pathologies produce a string that JSON.parse rejects but that contains +// all the semantic content the model intended: +// +// 1. TRUNCATION — max_tokens clipped the JSON mid-value. The string ends inside a string value, +// an array, or an object: `{"path": "src/lo`, `{"items": [1, 2`, `{"a": {"b": 1`. +// 2. LOCAL-MODEL SLOPPINESS — a trailing comma, an unbalanced brace/bracket, or a trailing +// garbage token after the closing brace: `{"path": "x.ts",}`, `{"a": 1 `. +// +// This module attempts a cheap, conservative repair BEFORE the caller falls back to a full +// non-streaming regeneration (which is expensive on a local backend and often fails identically +// when the cause was max_tokens). It only closes what's open and trims what's stray — it never +// invents keys or values, so a genuinely malformed call still fails downstream at schema validation. +// +// The repair is best-effort: if it can't produce parseable JSON, it returns null and the caller +// keeps its existing retry path. It is deliberately string-based (no AST) so it's trivially fast and +// has no dependencies, and so it handles truncated input that a strict parser can't even build an +// AST from. + +/** Parse `s` as JSON; on success return the value, on failure return null (never throws). */ +function tryParse(s: string): unknown { + try { + return JSON.parse(s); + } catch { + return null; + } +} + +/** Skip past the next JSON string literal starting at `i` (the opening quote). Returns the index + * just past the closing quote. Strings are the only place braces/brackets can appear without + * affecting nesting, so we must not count them while inside one. Handles `\"` and other escapes. */ +function skipString(s: string, i: number): number { + let j = i + 1; // past opening quote + for (; j < s.length; j++) { + const c = s[j]!; + if (c === "\\") { + j++; // skip the escaped char (covers \", \\, etc.) + continue; + } + if (c === '"') return j + 1; // past closing quote + } + return j; // ran off the end — unterminated string +} + +/** Attempts to repair `raw` into parseable JSON. Returns the parsed value on success, or null if no + * repair produced valid JSON. Steps, applied in order of how cheap and safe they are: + * + * 1. Maybe it already parses (trailing whitespace/newlines are fine for JSON.parse) — return as-is. + * 2. Strip a trailing comma before an expected-but-absent `}` or `]` (common local-model slip). + * 3. Strip stray non-JSON tokens after the first complete top-level value (`{"a":1}\n` → `{"a":1}`, + * and `{"a":1}garbage` → `{"a":1}` — JSON.parse rejects trailing content, so trim to the first + * complete value). + * 4. Close unterminated strings, then balance still-open braces/brackets (truncation repair). + * + * Each step re-attempts a parse, so the cheapest fix that works wins. */ +export function repairPartialJson(raw: string): unknown | null { + if (!raw) return null; + const trimmed = raw.trim(); + if (!trimmed) return null; + + // (1) Already valid? + const direct = tryParse(trimmed); + if (direct !== null) return direct; + + // (2) Trailing comma before end-of-object/array: `{"a":1,}` or `[1,2,]`. Repeat until none + // left so a nested shape like `{"a":{"b":2,},}` clears both commas (innermost-first). + let noTrailing = trimmed; + let prev: string; + do { + prev = noTrailing; + noTrailing = noTrailing.replace(/,\s*([\]}]+\s*$)/, "$1"); + } while (noTrailing !== prev); + if (noTrailing !== trimmed) { + const v = tryParse(noTrailing); + if (v !== null) return v; + } + + // (3) Stray trailing content after the first complete value. JSON.parse refuses trailing tokens, + // but a model often emits a closing brace then a stray newline, a repeated token, or prose. + // Find the end of the first balanced top-level value and cut there. + const cut = cutToFirstCompleteValue(trimmed); + if (cut !== null && cut !== trimmed) { + const v = tryParse(cut); + if (v !== null) return v; + } + + // (4) Truncation repair: close an unterminated string, then balance open braces/brackets. + const balanced = balanceAndClose(trimmed); + if (balanced !== null && balanced !== trimmed) { + // Re-run the earlier cheap fixes on the balanced result (a trailing comma may now be exposed). + const v = tryParse(balanced); + if (v !== null) return v; + const v2 = tryParse(balanced.replace(/,\s*([\]}]\s*$)/, "$1")); + if (v2 !== null) return v2; + } + + return null; +} + +/** If `s` starts with a complete top-level JSON value followed by stray content, return just that + * value (as a substring). Returns null if we can't find a clean boundary (e.g. the value is itself + * truncated). Walks the string tracking string literals and nesting depth. */ +function cutToFirstCompleteValue(s: string): string | null { + let i = 0; + // Skip leading whitespace. + while (i < s.length && /\s/.test(s[i]!)) i++; + if (i >= s.length) return null; + + const start = i; + const stack: string[] = []; + let inStr = false; + + while (i < s.length) { + const c = s[i]!; + if (inStr) { + if (c === "\\") { + i += 2; + continue; + } + if (c === '"') inStr = false; + i++; + continue; + } + if (c === '"') { + inStr = true; + i++; + continue; + } + if (c === "{" || c === "[") { + stack.push(c); + i++; + continue; + } + if (c === "}" || c === "]") { + stack.pop(); + i++; + // If the stack is empty, this was the end of the top-level value — cut here. + if (stack.length === 0) return s.slice(start, i); + continue; + } + // A bare scalar (number/true/false/null) ends at the next delimiter/comma/whitespace. + if (stack.length === 0 && (c === "," || c === "}" || c === "]" || /\s/.test(c))) { + return s.slice(start, i); + } + i++; + } + + // Ran off the end without closing the top-level value → it's truncated, not "complete + stray". + if (stack.length > 0) return null; + return null; +} + +/** Closes an unterminated trailing string and balances any open braces/brackets. Returns the + * repaired string, or null if nothing needed closing (caller can compare to skip a no-op parse). */ +function balanceAndClose(s: string): string | null { + let out = s; + const stack: string[] = []; + let inStr = false; + let i = 0; + + for (; i < out.length; i++) { + const c = out[i]!; + if (inStr) { + if (c === "\\") { + i++; + continue; + } + if (c === '"') inStr = false; + continue; + } + if (c === '"') { + inStr = true; + continue; + } + if (c === "{") stack.push("}"); + else if (c === "[") stack.push("]"); + else if (c === "}" || c === "]") stack.pop(); + } + + // If we ended inside a string, close it. A truncated value like `{"path":"src/lo` needs a closing + // quote before we can balance the outer braces. + if (inStr) { + out += '"'; + } + + // Close anything still open, innermost-first. Truncation mid-array/object → append the closers. + if (stack.length === 0 && !inStr) { + // Nothing to close — but a trailing comma may have been the only issue; let the caller handle it. + return inStr ? out : null; + } + while (stack.length) { + out += stack.pop(); + } + return out; +} \ No newline at end of file