Major upgrades (15 items): 1. System prompt overhaul: tool usage guide, local model guidance, error recovery, safety guidelines, read/mutating tool categorization (12 tests) 2. Multiline input: Shift+Enter for newlines, bracketed paste support, Enter submits single-line, Ctrl+Enter forces submit on multiline 3. /undo command: removes last user turn + assistant/tool messages, preserves filesystem changes, shows count (7 tests) 4. Thinking/reasoning token support: capture `reasoning_content` from DeepSeek/QwQ-style models, render as dimmed collapsible block 5. @ mention improvements: fuzzy path matching (character-order matching), 5-minute cache TTL for file list refresh 6. readFile binary guard: 10MB size limit, extension-based binary detection, null-byte heuristic, CRLF line-ending fix 7. writeFile atomic write: temp file + rename pattern, explicit ENOENT check 8. bash string accumulation: O(n²) → O(n) with array chunks 9. estimateTokens regex speedup: per-character loop → regex bulk counting 10. handleCompletedMessage: `any` → `Record<string, unknown>` 11. Session restore: persist `allowedTools` in SessionRecord, restore on resume 12. Context window parallel detection: Promise.allSettled for Ollama + LM Studio 13. Streaming text accumulation: O(n²) → O(n) with text chunks array 14. /dashboard & context bar already implemented (no changes needed) 15. Cost estimation already implemented (no changes needed) Tests: 322 passing across 42 files, typecheck clean, build 289.56 KB
115 lines
5.1 KiB
TypeScript
115 lines
5.1 KiB
TypeScript
import { readFile as fsReadFile, stat as fsStat } from "node:fs/promises";
|
|
import path from "node:path";
|
|
import { z } from "zod";
|
|
import { imageMimeType, MAX_IMAGE_BYTES } from "../utils/image.js";
|
|
import type { ToolDef } from "./types.js";
|
|
|
|
function withinWorkspace(resolved: string, workspace: string): boolean {
|
|
const rel = path.relative(workspace, resolved);
|
|
return !rel.startsWith("..") && !path.isAbsolute(rel);
|
|
}
|
|
|
|
// Reject files larger than this so we never accidentally OOM on a huge binary or log file.
|
|
const MAX_FILE_SIZE = 10 * 1024 * 1024; // 10 MB
|
|
|
|
// Common binary extensions — if the file extension matches, reject without reading.
|
|
const BINARY_EXTENSIONS = new Set([
|
|
".exe", ".dll", ".so", ".dylib", ".bin", ".dat", ".o", ".obj", ".pyc", ".pyo",
|
|
".class", ".jar", ".war", ".zip", ".tar", ".gz", ".bz2", ".7z", ".rar",
|
|
".iso", ".dmg", ".pdb", ".lib", ".a", ".woff", ".woff2", ".eot", ".ttf", ".otf",
|
|
".pdf", ".doc", ".docx", ".xls", ".xlsx", ".ppt", ".pptx",
|
|
".sqlite", ".db", ".ico", ".cur",
|
|
]);
|
|
|
|
// Cap on how much text a single read_file call returns, so a huge file can't blow up the context
|
|
// in one call. Cut on a line boundary (never mid-line) and report the exact next offset, so the
|
|
// model can page through the rest with `offset` instead of re-reading the same truncated prefix in
|
|
// a loop — which is what happened before, when the generic "... [truncated N more characters]"
|
|
// notice never mentioned offset and the model just re-issued the same call.
|
|
const MAX_READ_CHARS = 20_000;
|
|
|
|
const schema = z.object({
|
|
path: z.string().describe("File path, relative to working directory or absolute."),
|
|
offset: z.number().int().min(1).optional().describe("1-indexed start line (text only)."),
|
|
limit: z.number().int().min(1).max(2000).optional().describe("Max lines to read (text only)."),
|
|
});
|
|
|
|
export const readFileTool: ToolDef<z.infer<typeof schema>> = {
|
|
name: "read_file",
|
|
description:
|
|
"Read a local file. Text files return 1-indexed lines; large files are paginated (use nextOffset for next page). " +
|
|
"Image files (png, jpg, jpeg, gif, webp, bmp) are returned as image content (requires vision-capable model). " +
|
|
"Use to inspect file contents before editing, or to understand existing code. Prefer this over bash cat for files.",
|
|
schema,
|
|
mutating: false,
|
|
handler: async ({ path: filePath, offset, limit }, ctx) => {
|
|
const resolved = path.resolve(ctx.cwd, filePath);
|
|
if (!withinWorkspace(resolved, ctx.cwd)) {
|
|
throw new Error(`File ${filePath} resolves outside the workspace.`);
|
|
}
|
|
|
|
// --- Size guard: reject files over MAX_FILE_SIZE before reading ---
|
|
const stats = await fsStat(resolved);
|
|
if (stats.size > MAX_FILE_SIZE) {
|
|
throw new Error(
|
|
`${filePath} is ${(stats.size / 1_048_576).toFixed(1)}MB, over the ${MAX_FILE_SIZE / 1_048_576}MB read limit.`,
|
|
);
|
|
}
|
|
|
|
const mimeType = imageMimeType(resolved);
|
|
if (mimeType) {
|
|
if (stats.size > MAX_IMAGE_BYTES) {
|
|
throw new Error(
|
|
`${filePath} is ${(stats.size / 1_048_576).toFixed(1)}MB, over the ${MAX_IMAGE_BYTES / 1_048_576}MB limit for image reads.`,
|
|
);
|
|
}
|
|
const buffer = await fsReadFile(resolved);
|
|
return { path: resolved, image: true, mimeType, bytes: buffer.byteLength, base64: buffer.toString("base64") };
|
|
}
|
|
|
|
// --- Binary guard: reject by extension ---
|
|
const ext = path.extname(resolved).toLowerCase();
|
|
if (BINARY_EXTENSIONS.has(ext)) {
|
|
throw new Error(
|
|
`${filePath} looks like a binary file (${ext}). Use bash for binary inspection.`,
|
|
);
|
|
}
|
|
|
|
const content = await fsReadFile(resolved, "utf-8");
|
|
|
|
// --- Binary guard: null-byte heuristic (catches extensionless binaries) ---
|
|
const nullIndex = content.indexOf("\0");
|
|
if (nullIndex !== -1) {
|
|
throw new Error(
|
|
`${filePath} appears to be a binary file (null byte at position ${nullIndex}). Use bash for binary inspection.`,
|
|
);
|
|
}
|
|
|
|
const lines = content.split(/\r?\n/);
|
|
const start = offset ? offset - 1 : 0;
|
|
const requestedEnd = limit ? Math.min(start + limit, lines.length) : lines.length;
|
|
|
|
// Accumulate whole lines until the next line would push past the char cap. The first line is
|
|
// always included even if it alone exceeds the cap (a single minified 200k-char line, say) —
|
|
// paging within a line isn't possible with offset/limit, so there's nothing better to do there.
|
|
let end = start;
|
|
let chars = 0;
|
|
while (end < requestedEnd) {
|
|
const lineLen = `${end + 1}\t${lines[end]}\n`.length;
|
|
if (end > start && chars + lineLen > MAX_READ_CHARS) break;
|
|
chars += lineLen;
|
|
end++;
|
|
}
|
|
const slice = lines.slice(start, end);
|
|
const numbered = slice.map((line, i) => `${start + i + 1}\t${line}`).join("\n");
|
|
const hasMore = end < requestedEnd;
|
|
return {
|
|
path: resolved,
|
|
totalLines: lines.length,
|
|
content: hasMore
|
|
? `${numbered}\n\n... [truncated — ${requestedEnd - end} more line(s) in range; call read_file again with offset=${end + 1} to read the next page]`
|
|
: numbered,
|
|
...(hasMore ? { truncated: true, nextOffset: end + 1 } : {}),
|
|
};
|
|
},
|
|
}; |