Files
locode/src/tools/readFile.ts
T
kim 506599020c feat: 15 upgrades — system prompt, multiline input, /undo, thinking tokens, and more
Major upgrades (15 items):

1. System prompt overhaul: tool usage guide, local model guidance, error recovery,
   safety guidelines, read/mutating tool categorization (12 tests)
2. Multiline input: Shift+Enter for newlines, bracketed paste support,
   Enter submits single-line, Ctrl+Enter forces submit on multiline
3. /undo command: removes last user turn + assistant/tool messages,
   preserves filesystem changes, shows count (7 tests)
4. Thinking/reasoning token support: capture `reasoning_content` from
   DeepSeek/QwQ-style models, render as dimmed collapsible block
5. @ mention improvements: fuzzy path matching (character-order matching),
   5-minute cache TTL for file list refresh
6. readFile binary guard: 10MB size limit, extension-based binary detection,
   null-byte heuristic, CRLF line-ending fix
7. writeFile atomic write: temp file + rename pattern, explicit ENOENT check
8. bash string accumulation: O(n²) → O(n) with array chunks
9. estimateTokens regex speedup: per-character loop → regex bulk counting
10. handleCompletedMessage: `any` → `Record<string, unknown>`
11. Session restore: persist `allowedTools` in SessionRecord, restore on resume
12. Context window parallel detection: Promise.allSettled for Ollama + LM Studio
13. Streaming text accumulation: O(n²) → O(n) with text chunks array
14. /dashboard & context bar already implemented (no changes needed)
15. Cost estimation already implemented (no changes needed)

Tests: 322 passing across 42 files, typecheck clean, build 289.56 KB
2026-08-21 17:07:49 +09:00

115 lines
5.1 KiB
TypeScript

import { readFile as fsReadFile, stat as fsStat } from "node:fs/promises";
import path from "node:path";
import { z } from "zod";
import { imageMimeType, MAX_IMAGE_BYTES } from "../utils/image.js";
import type { ToolDef } from "./types.js";
function withinWorkspace(resolved: string, workspace: string): boolean {
const rel = path.relative(workspace, resolved);
return !rel.startsWith("..") && !path.isAbsolute(rel);
}
// Reject files larger than this so we never accidentally OOM on a huge binary or log file.
const MAX_FILE_SIZE = 10 * 1024 * 1024; // 10 MB
// Common binary extensions — if the file extension matches, reject without reading.
const BINARY_EXTENSIONS = new Set([
".exe", ".dll", ".so", ".dylib", ".bin", ".dat", ".o", ".obj", ".pyc", ".pyo",
".class", ".jar", ".war", ".zip", ".tar", ".gz", ".bz2", ".7z", ".rar",
".iso", ".dmg", ".pdb", ".lib", ".a", ".woff", ".woff2", ".eot", ".ttf", ".otf",
".pdf", ".doc", ".docx", ".xls", ".xlsx", ".ppt", ".pptx",
".sqlite", ".db", ".ico", ".cur",
]);
// Cap on how much text a single read_file call returns, so a huge file can't blow up the context
// in one call. Cut on a line boundary (never mid-line) and report the exact next offset, so the
// model can page through the rest with `offset` instead of re-reading the same truncated prefix in
// a loop — which is what happened before, when the generic "... [truncated N more characters]"
// notice never mentioned offset and the model just re-issued the same call.
const MAX_READ_CHARS = 20_000;
const schema = z.object({
path: z.string().describe("File path, relative to working directory or absolute."),
offset: z.number().int().min(1).optional().describe("1-indexed start line (text only)."),
limit: z.number().int().min(1).max(2000).optional().describe("Max lines to read (text only)."),
});
export const readFileTool: ToolDef<z.infer<typeof schema>> = {
name: "read_file",
description:
"Read a local file. Text files return 1-indexed lines; large files are paginated (use nextOffset for next page). " +
"Image files (png, jpg, jpeg, gif, webp, bmp) are returned as image content (requires vision-capable model). " +
"Use to inspect file contents before editing, or to understand existing code. Prefer this over bash cat for files.",
schema,
mutating: false,
handler: async ({ path: filePath, offset, limit }, ctx) => {
const resolved = path.resolve(ctx.cwd, filePath);
if (!withinWorkspace(resolved, ctx.cwd)) {
throw new Error(`File ${filePath} resolves outside the workspace.`);
}
// --- Size guard: reject files over MAX_FILE_SIZE before reading ---
const stats = await fsStat(resolved);
if (stats.size > MAX_FILE_SIZE) {
throw new Error(
`${filePath} is ${(stats.size / 1_048_576).toFixed(1)}MB, over the ${MAX_FILE_SIZE / 1_048_576}MB read limit.`,
);
}
const mimeType = imageMimeType(resolved);
if (mimeType) {
if (stats.size > MAX_IMAGE_BYTES) {
throw new Error(
`${filePath} is ${(stats.size / 1_048_576).toFixed(1)}MB, over the ${MAX_IMAGE_BYTES / 1_048_576}MB limit for image reads.`,
);
}
const buffer = await fsReadFile(resolved);
return { path: resolved, image: true, mimeType, bytes: buffer.byteLength, base64: buffer.toString("base64") };
}
// --- Binary guard: reject by extension ---
const ext = path.extname(resolved).toLowerCase();
if (BINARY_EXTENSIONS.has(ext)) {
throw new Error(
`${filePath} looks like a binary file (${ext}). Use bash for binary inspection.`,
);
}
const content = await fsReadFile(resolved, "utf-8");
// --- Binary guard: null-byte heuristic (catches extensionless binaries) ---
const nullIndex = content.indexOf("\0");
if (nullIndex !== -1) {
throw new Error(
`${filePath} appears to be a binary file (null byte at position ${nullIndex}). Use bash for binary inspection.`,
);
}
const lines = content.split(/\r?\n/);
const start = offset ? offset - 1 : 0;
const requestedEnd = limit ? Math.min(start + limit, lines.length) : lines.length;
// Accumulate whole lines until the next line would push past the char cap. The first line is
// always included even if it alone exceeds the cap (a single minified 200k-char line, say) —
// paging within a line isn't possible with offset/limit, so there's nothing better to do there.
let end = start;
let chars = 0;
while (end < requestedEnd) {
const lineLen = `${end + 1}\t${lines[end]}\n`.length;
if (end > start && chars + lineLen > MAX_READ_CHARS) break;
chars += lineLen;
end++;
}
const slice = lines.slice(start, end);
const numbered = slice.map((line, i) => `${start + i + 1}\t${line}`).join("\n");
const hasMore = end < requestedEnd;
return {
path: resolved,
totalLines: lines.length,
content: hasMore
? `${numbered}\n\n... [truncated — ${requestedEnd - end} more line(s) in range; call read_file again with offset=${end + 1} to read the next page]`
: numbered,
...(hasMore ? { truncated: true, nextOffset: end + 1 } : {}),
};
},
};