diff --git a/README.md b/README.md index b1b96b9..a60f974 100644 --- a/README.md +++ b/README.md @@ -87,6 +87,7 @@ locode is a full-screen terminal app built with [Ink](https://github.com/vadimde - **Ctrl+O**: print the full text of the last `/compact` (or auto-compact) summary. The collapsed notice you see right after compacting only shows a one-line hint — press Ctrl+O any time afterward to print the whole thing. - **Ctrl+B**: while a `bash` command is running, detaches it into the background and returns control to you immediately — the turn continues with a `bash_output`-checkable job id instead of waiting for the command to finish. A notice appears in the transcript once the backgrounded command actually completes. Only `bash` supports this today. The model can kill a still-running backgrounded job with `bash_kill`; any jobs still running when locode itself exits are killed too, so they don't outlive the process as orphans. +- **Ctrl+F**: open and focus a file panel docked to the right of the chat (hidden by default); press again to close it. It has two tabs — **Files**, the project's collapsible file tree (directories in cyan, same `node_modules`/`.git`/`dist` exclusions as `@` mentions), and **Activity**, the files `read_file`/`write_file`/`edit_file` have touched so far this session, most recent first, with a status glyph (`·` read, `+` written, `~` edited) and a repeat count. **Ctrl+G** switches between the two tabs. While the panel is focused, `↑`/`↓` move the selection (auto-scrolling to keep it in view), `↵`/`←`/`→` expand or collapse the selected folder, and **Esc** hands keyboard focus back to the chat input without closing the panel — typing is disabled while the panel has focus, so the same arrow key doesn't simultaneously recall chat history. - **Backends**: `--backend ollama` (default) or `--backend lmstudio`, or `--base-url ` for anything else that speaks the same API. - **Tools**: `read_file`, `list_files`, `grep`, `web_search`, `web_fetch`, `git_status`, `bash_output`, `todo_write` run automatically. `write_file`, `edit_file`, `bash`, `bash_kill`, and `git_commit` show a diff/preview in a bordered box and ask you to pick Yes / Yes-always-this-session / No with the arrow keys before running. @@ -100,7 +101,7 @@ locode is a full-screen terminal app built with [Ink](https://github.com/vadimde - **MCP servers**: locode connects to any [MCP](https://modelcontextprotocol.io) servers configured via `locode mcp add` or a project's `.mcp.json` (stdio and remote/streamable-HTTP transports), and adds their tools to every session, namespaced as `mcp____`. Every MCP tool is treated as mutating (confirmation required on every call) regardless of what it reports — the MCP `readOnlyHint` annotation is advisory and could be wrong (or set by a malicious server specifically to skip confirmation), so locode never trusts it. One misconfigured server doesn't block the others — check `/mcp` for per-server connection status. - **Claude Code plugins**: `locode plugin add ` installs a Claude Code-compatible plugin — locode reads its `.claude-plugin/plugin.json`, then loads it all directly: MCP servers (its `.mcp.json` or manifest `mcpServers`, merged in like any other MCP server), slash commands (`commands/*.md` — frontmatter `description`/`argument-hint`, body is a template expanded with `$ARGUMENTS`/`$1..$9` and submitted as your message), agents (`agents/*.md` — the body becomes a sub-agent's system prompt, exposed as a callable tool named `agent____`; a `tools:` frontmatter list restricts what it can use, with Claude Code's built-in tool names — Read, Grep, Edit, etc. — automatically mapped to locode's equivalents), hooks (`hooks/hooks.json`, see below), and skills (`skills/*/SKILL.md`, see below). Check `/plugins` for what's loaded. - **Skills**: named instructions the model loads on demand rather than a hook or a sub-agent — every installed skill (`skills//SKILL.md`) is exposed through one shared `skill` tool, whose own description lists every skill's name and "use this when..." blurb so the model knows when to call it. You can also invoke one directly with `/ [request]`, which skips the model's own judgment and submits the skill's instructions (plus your request, if any) as the turn. Check `/skills` for what's installed. -- **Hooks**: shell commands that fire on session lifecycle events — `SessionStart`, `UserPromptSubmit`, `PreToolUse`, `PostToolUse`, `PermissionRequest`, `SubagentStart`, `SubagentStop`, `CwdChanged`, `FileChanged`, `ConfigChange`, `Stop`, `SessionEnd`. Configured the same way MCP servers are — plugin-bundled (`hooks/hooks.json`), user-level (`locode hooks path`, hand-edited), and project-level (`.locode/hooks.json`) all merge together, every hook from every source runs. A hook receives a JSON payload on stdin (`session_id`, `cwd`, `hook_event_name`, plus event-specific fields like `prompt` or `tool_name`/`tool_input`); exit 0 allows (stdout becomes injected context for `SessionStart`/`UserPromptSubmit`), exit 2 blocks (stderr is the reason shown), anything else is a non-blocking warning. `PreToolUse`, `UserPromptSubmit`, and `PermissionRequest` can block; the rest are fire-and-forget. `PreToolUse` fires before permission modes apply, so a hook's block can't be bypassed by auto-accept. `command` and `http` hook types are supported (`prompt` is declared in the config format but not yet executed); command hooks can opt into structured JSON output via `outputSchema: "json"`. `SessionEnd` fires on every exit path (including Ctrl+C and external `SIGTERM`/`SIGHUP`). Check `/hooks` for what's configured. +- **Hooks**: shell commands that fire on session lifecycle events — `SessionStart`, `UserPromptSubmit`, `PreToolUse`, `PostToolUse`, `PermissionRequest`, `SubagentStart`, `SubagentStop`, `CwdChanged`, `FileChanged`, `ConfigChange`, `Stop`, `SessionEnd`. Configured the same way MCP servers are — plugin-bundled (`hooks/hooks.json`), user-level (`locode hooks path`, hand-edited), and project-level (`.locode/hooks.json`) all merge together, every hook from every source runs. A hook receives a JSON payload on stdin (`session_id`, `cwd`, `hook_event_name`, plus event-specific fields like `prompt` or `tool_name`/`tool_input`); exit 0 allows (stdout becomes injected context for `SessionStart`/`UserPromptSubmit`), exit 2 blocks (stderr is the reason shown), anything else is a non-blocking warning. `PreToolUse`, `UserPromptSubmit`, and `PermissionRequest` can block; the rest are fire-and-forget. `PreToolUse` fires before permission modes apply, so a hook's block can't be bypassed by auto-accept. `command` and `http` hook types run an external process/request; a `prompt` hook just injects its static `message` as additional context instead. Command/http hooks can opt into structured JSON output via `outputSchema: "json"`. `SessionEnd` fires on every exit path (including Ctrl+C and external `SIGTERM`/`SIGHUP`). Check `/hooks` for what's configured. - **Tool-calling mode**: on connect, locode probes whether the model reliably uses native OpenAI-style function calling. If not, it switches to a prompt-based fallback mode where the model is instructed to emit tool calls as fenced ` ```tool_call ``` ` JSON blocks, which locode parses itself. The result is cached per backend+model so future sessions skip the probe. Override with `--tool-mode native|fallback|auto` or the in-session `/mode` command. - **Context tracking & compaction**: the status bar shows `ctx NN%` — context window usage, from real `usage.prompt_tokens` when the backend reports it (requested via `stream_options.include_usage`), or a `~`-prefixed char-based estimate otherwise. The window size itself is auto-detected (Ollama's `/api/show`, then LM Studio's `/api/v0/models`) and cached per backend+model; falls back to a configurable default (`locode config set contextWindow `, or `$LOCODE_CONTEXT_WINDOW`) if neither responds. At 85% usage, locode automatically asks the model to summarize the conversation and replaces the history with that summary (a notice tells you when this happens) — or trigger it yourself anytime with `/compact`. @@ -151,13 +152,12 @@ locode config path - Requires a real interactive terminal (TTY) — you can't pipe input into it or run it from a non-interactive script. - Native tool-calling reliability varies by model and is non-deterministic even for capable models (see above). -- No sandboxing beyond the confirmation prompts — mutating tools operate on the real filesystem/shell with the permissions of the user running `locode`. Only approve commands you understand. -- Session resume replays prior user/assistant text so you can see it, but it doesn't re-display prior tool-call/tool-result lines from before the resume (the model still has that history — it's just not re-rendered). -- No in-app scrollback — once a message scrolls off the top of the window it's gone until you resize the terminal taller (the conversation itself is still intact and sent to the model; this only affects what you can visually re-read). +- No OS-level sandboxing (no container/VM isolation) — mutating tools operate on the real filesystem/shell with the permissions of the user running `locode`. Only approve commands you understand. Two lightweight guardrails run unconditionally regardless of permission mode (including `auto-accept`), as a safety floor rather than a full sandbox: `write_file`/`edit_file`/`bash`'s `cwd` override can't target a path outside the working directory (`../` traversal, an absolute path elsewhere, or — on Windows — a different drive all refuse), and `bash` refuses a short list of unambiguously catastrophic commands (wiping the filesystem root or home directory, a fork bomb, formatting/wiping a whole drive, writing raw data to a block device) before they'd ever run. Neither guard stops a model from doing damage confined to *within* the project directory, or running something merely inadvisable — see `src/tools/pathGuard.ts` and `src/tools/bashGuard.ts`. +- In-app scrollback is manual: PageUp/PageDown scroll the conversation view a page at a time (the terminal's native scrollback isn't available in the alternate screen buffer). Scrolling back up unpins the view from the latest message; PageDown back to the bottom (or sending a new message) re-pins it so new messages auto-scroll into view again. - Windows shell quoting for the `bash` tool has only had light testing; behavior may differ from Unix shells for complex quoting. - `git_commit` covers add/commit/create_branch/checkout/push/reset/stash/merge/rebase/delete_branch. Use `bash` for anything beyond that. - MCP tool results support text, image, audio, and resource content blocks. Images are returned in the same shape as `read_file` so vision-capable models can see them; audio and binary resources are summarized. Remote (HTTP) MCP servers support static headers (e.g. a bearer token) but not OAuth flows. - Compaction (`/compact` or automatic) replaces history with a model-generated prose summary — it costs one extra model call and loses tool-call/tool-result detail (the model's own account of what happened survives; the raw record doesn't). The auto-compact threshold defaults to 85% and is configurable via `locode config set autoCompactThreshold` or `LOCODE_AUTO_COMPACT_THRESHOLD`. -- Plugin support (`locode plugin add`) now covers every part of a plugin: MCP servers, slash commands, agents, hooks, and skills. A plugin's `allowed-tools` restriction on a command isn't enforced (the expanded prompt just runs as a normal turn with the full toolset). Duplicate MCP server names across sources are now detected and surfaced in `/mcp` — project-level wins over user-level wins over plugin-level. Duplicate slash-command and skill names are also surfaced in `/plugins` and `/skills`. +- Plugin support (`locode plugin add`) now covers every part of a plugin: MCP servers, slash commands, agents, hooks, and skills. A slash command's `allowed-tools` frontmatter restricts that one invocation's toolset (same tool-name translation as an agent's `tools:` — see `/plugins`); a skill invoked directly via `/` isn't restricted this way, since skills have no `allowed-tools` field of their own. Duplicate MCP server names across sources are now detected and surfaced in `/mcp` — project-level wins over user-level wins over plugin-level. Duplicate slash-command and skill names are also surfaced in `/plugins` and `/skills`. - Skills are exposed as one shared `skill` tool rather than one tool per skill — if two plugins install a skill with the same name, the first plugin in load order wins and the collision is shown in `/skills`. Skills can bundle sibling `references/*.md` files that are included when the skill is invoked. -- Hooks cover the most useful subset of Claude Code's lifecycle events: `SessionStart`, `UserPromptSubmit`, `PreToolUse`, `PostToolUse`, `PermissionRequest`, `SubagentStart`, `SubagentStop`, `CwdChanged`, `FileChanged`, `ConfigChange`, `Stop`, `SessionEnd`. `CwdChanged` is declared in the config format but not yet fired anywhere — a cwd never changes mid-session in locode today, so configuring it is a no-op for now. `command` hooks and `http` hooks are supported; `prompt` hooks are declared in the config format but not yet executed. Command hooks can opt into structured JSON output via `outputSchema: "json"`. `PreToolUse`, `UserPromptSubmit`, and `PermissionRequest` can block; the rest are fire-and-forget. `SessionEnd` fires on every exit path (including Ctrl+C and external `SIGTERM`/`SIGHUP`), so it doesn't always have a real session id to report. All hooks for an event run in parallel with no defined ordering, and every configured hook always runs — there's no way to disable one without editing the file it came from. +- Hooks cover the most useful subset of Claude Code's lifecycle events: `SessionStart`, `UserPromptSubmit`, `PreToolUse`, `PostToolUse`, `PermissionRequest`, `SubagentStart`, `SubagentStop`, `CwdChanged`, `FileChanged`, `ConfigChange`, `Stop`, `SessionEnd`. `CwdChanged` is declared in the config format but not yet fired anywhere — a cwd never changes mid-session in locode today, so configuring it is a no-op for now. All three hook types are supported: `command` and `http` run an external process/request, and `prompt` just injects its static `message` as additional context (the same way a command/http hook's stdout does) — it has no process to fail, so it can't block an event the way a command hook's exit code 2 can. Command/http hooks can opt into structured JSON output via `outputSchema: "json"`. `PreToolUse`, `UserPromptSubmit`, and `PermissionRequest` can block; the rest are fire-and-forget. `SessionEnd` fires on every exit path (including Ctrl+C and external `SIGTERM`/`SIGHUP`), so it doesn't always have a real session id to report. All hooks for an event run in parallel with no defined ordering, and every configured hook always runs — there's no way to disable one without editing the file it came from. diff --git a/locode-upgrade-memo.md b/locode-upgrade-memo.md new file mode 100644 index 0000000..442da84 --- /dev/null +++ b/locode-upgrade-memo.md @@ -0,0 +1,247 @@ +# locode 업그레이드 작업 메모 + +## 프로젝트 개요 +**locode** — Claude Code의 설계 철학을 가져와 로컬 모델(Ollama/LM Studio)용으로 재구현한 에이전트 코딩 CLI. TypeScript + Ink(React-for-CLI) 기반. 백엔드는 OpenAI 호환 `/v1/chat/completions` 엔드포인트 사용. + +- 핵심 철학: "신뢰할 수 없고 느리고 비전/툴콜 지원이 불확실한 로컬 모델"이라는 현실에 맞춰 모든 가정을 비관적으로 재단 +- Claude Code 플러그인 포맷을 직접 소비하는 하위호환 브리지 (`.claude-plugin/plugin.json`, commands/agents/skills/hooks/MCP) + +## 빌드/테스트 상태 +- `npm run build` (tsup) — 깨끗 +- `npx tsc --noEmit` — 깨끗 +- `npm test` (vitest) — 29 파일 215개 전부 통과 +- 파일 인코딩: **CRLF** (edit_file 도구가 LF로 정규화해서 매칭 실패 → Python 스크립트로 바이너리 편집해야 함) + +--- + +## 발견한 업그레이드 후보 (12개) + +| # | 항목 | 난이도 | 효과 | 로컬 특화 | 상태 | +|---|---|:---:|:---:|:---:|:---:| +| 1 | 병렬 툴 실행 (read-only) | 중 | 대 | ★ | **완료** ✅ | +| 2 | 재시도 정책 설정화 (`maxRetries`) | 하 | 중 | ★ | **완료** ✅ | +| 3 | 정확한 토큰 추정 (`/api/tokenize` 또는 BPE) | 중 | 대 | ★★ | **완료** ✅ | +| 4 | 스마트 출력 캡 (head+tail, 라인 길이) | 하 | 중 | ★ | **완료** ✅ | +| 5 | 부분 히스토리 보존 컴팩션 | **상** | 대 | ★ | **완료** ✅ | +| 6 | 동적 `max_tokens` | 하 | 대 | ★ | **완료** ✅ | +| 7 | 툴 설명 풍부화 + 동적 툴 선택 | 중 | 중 | ★ | **완료** ✅ | +| 8 | MCP 연결 재시도·재연결 | 중 | 중 | | **완료** ✅ | +| 9 | 컨텍스트 윈도우 캐시 TTL | 하 | 중 | ★ | **완료** ✅ | +| 10 | `edit_file` 유사 매치 제안 | 중 | 대 | ★★ | **완료** ✅ | +| 11 | git_status 출력 head+tail | 하 | 중 | | **완료** ✅ | +| 12 | `auto-accept` 모드 세분화 | 중 | 중 | | **완료** ✅ | + +--- + +## ✅ 완료: #6 동적 `max_tokens` + +### 문제 +`src/agent/loop.ts`의 두 생성 요청이 `max_tokens: 4096` 하드코딩: +- 729행: 스트리밍 요청 (`session.client.chat.completions.create`, `stream: true`) +- 840행(→이제 856행): 비스트리밍 재시도 (native 툴콜 인자가 깨졌을 때) + +로컬 모델이 파일을 통째로 다시 쓸 때(fallback 모드에서 정밀 edit이 어려워 흔함) 4096 토큰으로 부족 → 응답 중간 잘림 → 툴콜 JSON 불완전 → malformed 에러 반복. 이게 "자꾸 에러가 나"던 원인. + +### 해결 +`shouldAutoCompact` 뒤(157행 근처)에 `resolveMaxTokens(session)` 헬퍼 추가: + +```ts +function resolveMaxTokens(session: Session): number { + const MARGIN = 512; + const MIN = 2048; + const available = session.contextWindow - session.lastContextTokens - MARGIN; + return Math.max(MIN, Math.min(available, session.contextWindow)); +} +``` + +두 사이트 모두 `max_tokens: 4096` → `max_tokens: resolveMaxTokens(session)` 교체. + +### 동작 +- `contextWindow − lastContextTokens − 512`를 출력 예산으로 할당 +- 하한 2048: 컨텍스트 거의 찼어도 최소 출력 보장 +- 상한 `contextWindow`: 윈도우 커도 그 이상 요구 안 함 +- 32k 윈도우 / 8k 사용 중 → 약 23k 출력 (기존 4096의 5.7배) +- 8k 윈도우 / 6k 사용 중 → 2048 (기존과 동일) + +### 남겨둔 것 +- `compactSession`의 `max_tokens: 1024` (loop.ts:186) — 짧은 산문 요약용이라 동적 계산 불필요, 그대로 유지 +- `capabilityProbe.ts`의 `max_tokens: 200` — 핑 툴용, 그대로 유지 + +### 검증 +- `tsc --noEmit` ✓ +- `npm run build` ✓ (217.48 KB) +- `npm test` ✓ 27파일 179개 전부 통과 + +### 편집 메모 +- 파일이 CRLF라 edit_file 도구가 매칭 실패함 → `src/agent/_patch.py` 임시 스크립트로 바이너리 교체 후 삭제 +- 향후 이 프로젝트 edit_file 시도 전 `file `로 인코딩 확인; CRLF면 Python 바이너리 편집 또는 `sed` 사용 + +--- + +## ✅ 완료: #3 정확한 토큰 추정 (스크립트 인식 휴리스틱) + +### 문제 +`src/utils/tokens.ts`의 `estimateTokens`가 `JSON.stringify(messages).length / 4` — 두 가지 실패 모드: +1. JSON 직렬화 오버헤드(따옴표, 중괄호, 이스케이프)를 콘텐츠 토큰으로 계산 → 추정치 15–25% 부풀림 (모델에 안 보내는 것들) +2. 동일한 chars/token 비율을 모든 스크립트에 적용 → 영어 산문 ~4, 코드/기호 ~3.5, CJK(한국어/중국어/일본어) ~1.5인데 무시 → CJK 컨텍스트 과소평가, 컴팩션 타이밍 부정확 + +실제 usage가 오면 이미 정확하지만(`lastContextTokensIsEstimate = false`), 추정은 첫 턴 전/컴팩션 직후/서브에이전트 생성 시 사용 → 이 시점의 부정확이 컴팩션 타이밍을 빗나가게 함. + +### 해결 +`tokens.ts`를 메시지 구조 순회 + 스크립트 인식 휴리스틱으로 재작성: + +- **PER_MESSAGE_OVERHEAD = 4**: 채팅 템플릿이 각 메시지에 추가하는 역할/구분자 토큰(~3–5) 반영 +- **메시지별 콘텐츠 순회**: 시스템/사용자/어시스턴트 텍스트, tool_calls 구조, tool 결과를 JSON이 아닌 모델이 실제로 보는 텍스트로 추출 +- **스크립트 인식 가중치** (`weightedChars`): + - CJK(히라가나/가타카나/한자/한글) ×2.4 → ~1.5 chars/token (각 코드 포인트가 보통 자체 BPE 토큰) + - 조밀 기호(구두점/연산자/괄호, 코드에 흔함) ×1.15 → ~3.5 chars/token + - 라틴 기본 ×1 → ~4 chars/token +- **멀티파트 콘텐츠**: 텍스트 파트는 텍스트, 이미지/오디오 파트는 flat 8 토큰(base64가 아닌 placeholder 토큰) +- 동기 순수 추정 유지(백엔드 호출 없음) → 첫 턴/컴팩션/서브에이전트에 안전 + +Ollama `/api/tokenize`는 백엔드 분기 + 매 턴 지연이 필요해 제외(의존성·복잡도 대비 효과 부족). 휴리스틱 개선으로 즉시 효과. + +### 동작 +- 한국어 메시지 40자: 기존 10 토큰 → ~27 토큰 (실제에 가까움) +- 코드/기호: 기존보다 약간 높게 → 컴팩션 조기 트리거 (OOM 방지) +- 영어 산문: 기존과 유사하되 JSON 오버헤드 제거 → 약간 낮아짐 +- 구조(tool_calls, 멀티파트) 비용 반영 + +### 검증 +- `tsc --noEmit` ✓ +- `npm run build` ✓ (224.26 KB) +- `npm test` ✓ 29파일 203개 전부 통과 (신규 9개: tokens.test.ts) +- `loop.test.ts`는 `lastContextTokens` 직접 설정 → 추정값 변화에 영향 없음 확인 + +### 편집 메모 +- `tokens.ts` CRLF → write_file + Python 변환 +- `tokens.test.ts` LF → edit_file 사용 +- `ChatCompletionMessageParam` 멀티파트 타입 캐스트 `as unknown as` 필요 (OpenAI 타입 narrow) + +### 남겨둔 것 +- Ollama `/api/tokenize` 캐싱: 백엔드별 분기 + 비동기 필요 → 별도 작업. 현재 휴리스틱으로 충분히 개선됨 +- 실제 usage 도착 후에는 항상 정확한 값 사용(`updateContextTracking`의 `lastContextTokensIsEstimate = false`) + +--- + +## ✅ 완료: #4 스마트 출력 캡 (head+tail 보존) + +### 문제 +`src/utils/truncate.ts`의 `truncate()`가 head만 보존. 긴 명령 출력에서 tail의 에러/상태 줄이 잘림 → 모델이 실패 원인을 못 봄. 특히 로컬 모델에서 bash/git 출력이 길면 마지막 에러 메시지가 사라져 디버깅 불가. + +### 해결 +`truncate.ts`를 head+tail 보존(중간 생략)으로 재작성: + +- 라인 단위로 잘라 가독성 유지 (반 줄 잘림 방지) +- 예산의 60% head, 40% tail 할당 (tail이 에러/상태 줄을 담는 경우가 많아 비중 높임) +- head/tail 오버랩 가드 (예산 초과가 적을 때 중복 라인 방지) +- 모든 라인이 예산보다 길면 문자 단위 폴백 +- 생략된 문자 수 + 보존된 head/tail 라인 수 표시 + +### 적용 범위 +`truncate()` 시그니처 유지 → 모든 기존 호출자 자동 개선: +- `bash.ts` (stdout/stderr) — 가장 큰 효과 +- `git.ts` (status/diff/log/show/branches 출력) — **#11도 함께 해결** +- `bashOutput.ts`, `backgroundJobs.ts`, `webFetch.ts`, `importFile.ts` + +`readFile.ts`는 자체 페이지네이션(nextOffset)을 쓰므로 그대로 유지. `grep.ts`/`listFiles.ts`는 자체 limit 잘라내기 사용. + +### 동작 +- 짧으면 그대로 반환 +- 길면 head 일부 + `... [truncated N more characters — middle omitted, X head + Y tail lines kept] ...` + tail 일부 +- tail에 에러 줄이 있으면 모델이 볼 수 있음 + +### 검증 +- `tsc --noEmit` ✓ +- `npm run build` ✓ (222.30 KB) +- `npm test` ✓ 28파일 194개 전부 통과 (신규 7개: truncate.test.ts) +- 기존 호출자 테스트(bash/git/grep 등) 전부 통과 → 호환성 확인 + +### 편집 메모 +- `truncate.ts`는 CRLF → write_file 후 Python으로 CRLF 변환 +- `truncate.test.ts`는 LF → edit_file 도구 사용 가능 + +--- + +## ✅ 완료: #10 `edit_file` 유사 매치 제안 + +### 문제 +fallback 모델(그리고 정밀 edit이 어려운 로컬 모델)이 `old_string`을 거의 정확히 but not exactly 제공 → `occurrences === 0` → 단순 "not found" 에러 → 모델이 맥락 없이 재시도, 실패 반복. 정확한 텍스트를 어디서 가져와야 할지 힌트가 없음. + +### 해결 +`src/tools/editFile.ts`에 유사 매치 제안 추가: + +- `normaliseForCompare(s)`: 공백 연속을 단일 스페이스로 정규화 → 들여쓰기/줄바꿈 차이에 강건 +- `boundedLevenshtein(a, b, maxDist)`: 조기 종료 Levenshtein. `maxDist` 초과 시 즉시 반환 → 큰 파일에서도 저렴 +- `findSimilarMatch(content, needle)`: 파일 전체를 슬라이딩 윈도우(needle 길이 ±50%, step = needle/8)로 순회하며 정규화된 텍스트로 유사도 측정. 최고 점수 ≥ 0.5일 때만 반환 +- `similarHint(original, oldString)`: 매치 실패 시 에러/preview 메시지에 "The closest match in the file (line N, ~X% similar):" + snippet 추가 + +handler와 preview 양쪽의 `occurrences === 0` 경로에 적용. 기존 "not found" 메시지 뒤에 힌트가 붙음. + +### 동작 +- 정확히 일치하는 부분이 있으면 기존 동작 유지 (힌트 없음) +- 유사한 부분이 있으면 위치·유사도·snippet 제안 → 모델이 정확한 `old_string`으로 재시도 가능 +- 전혀 다르면 힌트 없이 "not found"만 (노이즈 방지) + +### 검증 +- `tsc --noEmit` ✓ +- `npm run build` ✓ (220.91 KB) +- `npm test` ✓ 27파일 187개 전부 통과 (신규 3개: closest match 제안/preview/유사도 임계값) + +### 편집 메모 +- `editFile.ts`는 CRLF → Python 바이너리 편집으로 교체 + CRLF 유지 +- `editFile.test.ts`는 LF → edit_file 도구 사용 가능 +- `noUncheckedIndexedAccess` 활성화 → 배열 인덱스 접근 시 `?? 기본값` 처리 필요 + +--- + +## 남은 우선순위 — 모두 완료 ✅ + +12개 업그레이드 후보 전부 완료. 아래는 구현 요약. + +### 즉시 효과 (구현 가벼움) +- ✅ **#4 스마트 출력 캡** — `truncate.ts` head+tail 보존, 모든 호출자 자동 개선 +- ✅ **#10 `edit_file` 유사 매치 제안** — 매치 실패 시 Levenshtein 유사 위치 제안 +- ✅ **#6 동적 max_tokens** — `resolveMaxTokens(session)` +- ✅ **#11 git 출력 head+tail** — #4로 함께 해결 + +### 정확도에 큰 영향 +- ✅ **#3 정확한 토큰 추정** — 스크립트 인식 휴리스틱 (CJK/기호/구조 비용) +- ✅ **#5 부분 히스토리 보존 컴팩션** — 최근 N턴 원본 보존 + 이전 요약 + +### 성능 +- ✅ **#1 병렬 툴 실행** — `runToolBatch`: read-only 툴 `Promise.all` 병렬, mutating 순차. 4개 루프에 적용 + +### 회복력 +- ✅ **#2 재시도 정책** — `maxRetries` 설정화 (기본 0, SDK 지수 백오프) +- ✅ **#8 MCP 재연결** — `connectMcpServer` 재시도 + `/mcp reconnect` 명령 + 세션 toolset 갱신 +- ✅ **#9 캐시 TTL** — `cachedAt` 타임스탬프 + N일(기본 7) 경과 재감지 + +### 기타 +- ✅ **#7 툴 설명 풍부화** — 8개 핵심 툴 description에 "use when…"/예시 추가 +- ✅ **#12 auto-accept 세분화** — `auto-accept` = 모든 mutating 툴 자동 승인, `auto-edit` = 편집 툴만 (설명-동작 일치) + +--- + +## 핵심 파일 맵 +- `src/agent/loop.ts` (1138행) — 메인 에이전트 루프, 턴/스트리밍/툴콜/컴팩션/서브에이전트/병렬 툴 배치 +- `src/agent/session.ts` — Session 객체, 통계, 상태 +- `src/agent/systemPrompt.ts` — 시스템 프롬프트 빌더 (매우 간결, 로컬 준수율 우선) +- `src/tools/` — 14개 내장 툴 (read_file, list_files, grep, web_search, web_fetch, git_status, write_file, edit_file, bash, bash_output, bash_kill, git_commit, todo_write, agent) +- `src/toolcalling/` — native 어댑터, fallback 파서/프롬프트, resolve (Ollama 빈키 복구 포함) +- `src/mcp/` — MCP 클라이언트/매니저/어댑터/config (모든 MCP 툴 mutating 강제) +- `src/hooks/` — 훅 러너 (병렬 실행, SSRF 가드, exit 0/2 시맨틱스) +- `src/plugins/` — Claude Code 플러그인 로더 (commands/agents/skills/hooks/MCP, 툴명 매핑) +- `src/backend/` — client, capabilityProbe, contextWindow(자동 탐지), capabilityCache +- `src/config/` — config (CLI > env > 저장 > 기본값 우선순위), defaults, store, types +- `src/permissions/` — permissionManager (default/plan/auto-edit/auto-accept), types +- `src/persistence/` — sessionStore (원자 쓰기, 큐잉), exportSession, replayHistory +- `src/utils/` — tokens, truncate, shell, processTree, image, html, mentions, projectInstructions +- `src/ui/ink/` — Ink(React) 풀스크린 TUI 컴포넌트 + +## 트러블슈팅 힌트 +- "자꾸 에러"의 주요 원인: `max_tokens: 4096` 잘림 → malformed 툴콜 (✅ 해결됨) +- 컨텍스트 윈도우가 8192 기본값이면 `locode config set contextWindow <실제값>` 필요 — `resolveMaxTokens`가 제값을 내려면 +- `/status`로 현재 model/backend/mode/cwd 확인 가능 +- fallback 모드 툴콜 실패 시 `/mode fallback` 강제 또는 모델 교체 +- **"Paused after N steps"가 자주 뜨면**: 1 스텝 = 1 모델 요청. 로컬 모델은 한 번에 1 툴만 호출하는 경향이 있어 다수 파일 작업이 50스텝을 쉽게 초과. 기본값 50→**100** 상향(일상 작업용). 큰 배치 작업 시 `locode config set maxIterations ` (예: 200). 정지 시 작업 내용은 보존되므로 "continue"로 이어서 진행 가능. \ No newline at end of file diff --git a/package-lock.json b/package-lock.json index 7da4434..fbacdaa 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "locode", - "version": "0.5.1", + "version": "0.5.2", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "locode", - "version": "0.5.1", + "version": "0.5.2", "dependencies": { "@modelcontextprotocol/sdk": "^1.29.0", "@vscode/ripgrep": "^1.18.0", @@ -94,9 +94,9 @@ } }, "node_modules/@esbuild/aix-ppc64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.27.7.tgz", - "integrity": "sha512-EKX3Qwmhz1eMdEJokhALr0YiD0lhQNwDqkPYyPhiSwKrh7/4KRjQc04sZ8db+5DVVnZ1LmbNDI1uAMPEUBnQPg==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.27.2.tgz", + "integrity": "sha512-GZMB+a0mOMZs4MpDbj8RJp4cw+w1WV5NYD6xzgvzUJ5Ek2jerwfO2eADyI6ExDSUED+1X8aMbegahsJi+8mgpw==", "cpu": [ "ppc64" ], @@ -111,9 +111,9 @@ } }, "node_modules/@esbuild/android-arm": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.27.7.tgz", - "integrity": "sha512-jbPXvB4Yj2yBV7HUfE2KHe4GJX51QplCN1pGbYjvsyCZbQmies29EoJbkEc+vYuU5o45AfQn37vZlyXy4YJ8RQ==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.27.2.tgz", + "integrity": "sha512-DVNI8jlPa7Ujbr1yjU2PfUSRtAUZPG9I1RwW4F4xFB1Imiu2on0ADiI/c3td+KmDtVKNbi+nffGDQMfcIMkwIA==", "cpu": [ "arm" ], @@ -128,9 +128,9 @@ } }, "node_modules/@esbuild/android-arm64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.27.7.tgz", - "integrity": "sha512-62dPZHpIXzvChfvfLJow3q5dDtiNMkwiRzPylSCfriLvZeq0a1bWChrGx/BbUbPwOrsWKMn8idSllklzBy+dgQ==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.27.2.tgz", + "integrity": "sha512-pvz8ZZ7ot/RBphf8fv60ljmaoydPU12VuXHImtAs0XhLLw+EXBi2BLe3OYSBslR4rryHvweW5gmkKFwTiFy6KA==", "cpu": [ "arm64" ], @@ -145,9 +145,9 @@ } }, "node_modules/@esbuild/android-x64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.27.7.tgz", - "integrity": "sha512-x5VpMODneVDb70PYV2VQOmIUUiBtY3D3mPBG8NxVk5CogneYhkR7MmM3yR/uMdITLrC1ml/NV1rj4bMJuy9MCg==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.27.2.tgz", + "integrity": "sha512-z8Ank4Byh4TJJOh4wpz8g2vDy75zFL0TlZlkUkEwYXuPSgX8yzep596n6mT7905kA9uHZsf/o2OJZubl2l3M7A==", "cpu": [ "x64" ], @@ -162,9 +162,9 @@ } }, "node_modules/@esbuild/darwin-arm64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.27.7.tgz", - "integrity": "sha512-5lckdqeuBPlKUwvoCXIgI2D9/ABmPq3Rdp7IfL70393YgaASt7tbju3Ac+ePVi3KDH6N2RqePfHnXkaDtY9fkw==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.27.2.tgz", + "integrity": "sha512-davCD2Zc80nzDVRwXTcQP/28fiJbcOwvdolL0sOiOsbwBa72kegmVU0Wrh1MYrbuCL98Omp5dVhQFWRKR2ZAlg==", "cpu": [ "arm64" ], @@ -179,9 +179,9 @@ } }, "node_modules/@esbuild/darwin-x64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.27.7.tgz", - "integrity": "sha512-rYnXrKcXuT7Z+WL5K980jVFdvVKhCHhUwid+dDYQpH+qu+TefcomiMAJpIiC2EM3Rjtq0sO3StMV/+3w3MyyqQ==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.27.2.tgz", + "integrity": "sha512-ZxtijOmlQCBWGwbVmwOF/UCzuGIbUkqB1faQRf5akQmxRJ1ujusWsb3CVfk/9iZKr2L5SMU5wPBi1UWbvL+VQA==", "cpu": [ "x64" ], @@ -196,9 +196,9 @@ } }, "node_modules/@esbuild/freebsd-arm64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.27.7.tgz", - "integrity": "sha512-B48PqeCsEgOtzME2GbNM2roU29AMTuOIN91dsMO30t+Ydis3z/3Ngoj5hhnsOSSwNzS+6JppqWsuhTp6E82l2w==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.27.2.tgz", + "integrity": "sha512-lS/9CN+rgqQ9czogxlMcBMGd+l8Q3Nj1MFQwBZJyoEKI50XGxwuzznYdwcav6lpOGv5BqaZXqvBSiB/kJ5op+g==", "cpu": [ "arm64" ], @@ -213,9 +213,9 @@ } }, "node_modules/@esbuild/freebsd-x64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.27.7.tgz", - "integrity": "sha512-jOBDK5XEjA4m5IJK3bpAQF9/Lelu/Z9ZcdhTRLf4cajlB+8VEhFFRjWgfy3M1O4rO2GQ/b2dLwCUGpiF/eATNQ==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.27.2.tgz", + "integrity": "sha512-tAfqtNYb4YgPnJlEFu4c212HYjQWSO/w/h/lQaBK7RbwGIkBOuNKQI9tqWzx7Wtp7bTPaGC6MJvWI608P3wXYA==", "cpu": [ "x64" ], @@ -230,9 +230,9 @@ } }, "node_modules/@esbuild/linux-arm": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.27.7.tgz", - "integrity": "sha512-RkT/YXYBTSULo3+af8Ib0ykH8u2MBh57o7q/DAs3lTJlyVQkgQvlrPTnjIzzRPQyavxtPtfg0EopvDyIt0j1rA==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.27.2.tgz", + "integrity": "sha512-vWfq4GaIMP9AIe4yj1ZUW18RDhx6EPQKjwe7n8BbIecFtCQG4CfHGaHuh7fdfq+y3LIA2vGS/o9ZBGVxIDi9hw==", "cpu": [ "arm" ], @@ -247,9 +247,9 @@ } }, "node_modules/@esbuild/linux-arm64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.27.7.tgz", - "integrity": "sha512-RZPHBoxXuNnPQO9rvjh5jdkRmVizktkT7TCDkDmQ0W2SwHInKCAV95GRuvdSvA7w4VMwfCjUiPwDi0ZO6Nfe9A==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.27.2.tgz", + "integrity": "sha512-hYxN8pr66NsCCiRFkHUAsxylNOcAQaxSSkHMMjcpx0si13t1LHFphxJZUiGwojB1a/Hd5OiPIqDdXONia6bhTw==", "cpu": [ "arm64" ], @@ -264,9 +264,9 @@ } }, "node_modules/@esbuild/linux-ia32": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.27.7.tgz", - "integrity": "sha512-GA48aKNkyQDbd3KtkplYWT102C5sn/EZTY4XROkxONgruHPU72l+gW+FfF8tf2cFjeHaRbWpOYa/uRBz/Xq1Pg==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.27.2.tgz", + "integrity": "sha512-MJt5BRRSScPDwG2hLelYhAAKh9imjHK5+NE/tvnRLbIqUWa+0E9N4WNMjmp/kXXPHZGqPLxggwVhz7QP8CTR8w==", "cpu": [ "ia32" ], @@ -281,9 +281,9 @@ } }, "node_modules/@esbuild/linux-loong64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.27.7.tgz", - "integrity": "sha512-a4POruNM2oWsD4WKvBSEKGIiWQF8fZOAsycHOt6JBpZ+JN2n2JH9WAv56SOyu9X5IqAjqSIPTaJkqN8F7XOQ5Q==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.27.2.tgz", + "integrity": "sha512-lugyF1atnAT463aO6KPshVCJK5NgRnU4yb3FUumyVz+cGvZbontBgzeGFO1nF+dPueHD367a2ZXe1NtUkAjOtg==", "cpu": [ "loong64" ], @@ -298,9 +298,9 @@ } }, "node_modules/@esbuild/linux-mips64el": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.27.7.tgz", - "integrity": "sha512-KabT5I6StirGfIz0FMgl1I+R1H73Gp0ofL9A3nG3i/cYFJzKHhouBV5VWK1CSgKvVaG4q1RNpCTR2LuTVB3fIw==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.27.2.tgz", + "integrity": "sha512-nlP2I6ArEBewvJ2gjrrkESEZkB5mIoaTswuqNFRv/WYd+ATtUpe9Y09RnJvgvdag7he0OWgEZWhviS1OTOKixw==", "cpu": [ "mips64el" ], @@ -315,9 +315,9 @@ } }, "node_modules/@esbuild/linux-ppc64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.27.7.tgz", - "integrity": "sha512-gRsL4x6wsGHGRqhtI+ifpN/vpOFTQtnbsupUF5R5YTAg+y/lKelYR1hXbnBdzDjGbMYjVJLJTd2OFmMewAgwlQ==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.27.2.tgz", + "integrity": "sha512-C92gnpey7tUQONqg1n6dKVbx3vphKtTHJaNG2Ok9lGwbZil6DrfyecMsp9CrmXGQJmZ7iiVXvvZH6Ml5hL6XdQ==", "cpu": [ "ppc64" ], @@ -332,9 +332,9 @@ } }, "node_modules/@esbuild/linux-riscv64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.27.7.tgz", - "integrity": "sha512-hL25LbxO1QOngGzu2U5xeXtxXcW+/GvMN3ejANqXkxZ/opySAZMrc+9LY/WyjAan41unrR3YrmtTsUpwT66InQ==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.27.2.tgz", + "integrity": "sha512-B5BOmojNtUyN8AXlK0QJyvjEZkWwy/FKvakkTDCziX95AowLZKR6aCDhG7LeF7uMCXEJqwa8Bejz5LTPYm8AvA==", "cpu": [ "riscv64" ], @@ -349,9 +349,9 @@ } }, "node_modules/@esbuild/linux-s390x": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.27.7.tgz", - "integrity": "sha512-2k8go8Ycu1Kb46vEelhu1vqEP+UeRVj2zY1pSuPdgvbd5ykAw82Lrro28vXUrRmzEsUV0NzCf54yARIK8r0fdw==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.27.2.tgz", + "integrity": "sha512-p4bm9+wsPwup5Z8f4EpfN63qNagQ47Ua2znaqGH6bqLlmJ4bx97Y9JdqxgGZ6Y8xVTixUnEkoKSHcpRlDnNr5w==", "cpu": [ "s390x" ], @@ -366,9 +366,9 @@ } }, "node_modules/@esbuild/linux-x64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.27.7.tgz", - "integrity": "sha512-hzznmADPt+OmsYzw1EE33ccA+HPdIqiCRq7cQeL1Jlq2gb1+OyWBkMCrYGBJ+sxVzve2ZJEVeePbLM2iEIZSxA==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.27.2.tgz", + "integrity": "sha512-uwp2Tip5aPmH+NRUwTcfLb+W32WXjpFejTIOWZFw/v7/KnpCDKG66u4DLcurQpiYTiYwQ9B7KOeMJvLCu/OvbA==", "cpu": [ "x64" ], @@ -383,9 +383,9 @@ } }, "node_modules/@esbuild/netbsd-arm64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.27.7.tgz", - "integrity": "sha512-b6pqtrQdigZBwZxAn1UpazEisvwaIDvdbMbmrly7cDTMFnw/+3lVxxCTGOrkPVnsYIosJJXAsILG9XcQS+Yu6w==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.27.2.tgz", + "integrity": "sha512-Kj6DiBlwXrPsCRDeRvGAUb/LNrBASrfqAIok+xB0LxK8CHqxZ037viF13ugfsIpePH93mX7xfJp97cyDuTZ3cw==", "cpu": [ "arm64" ], @@ -400,9 +400,9 @@ } }, "node_modules/@esbuild/netbsd-x64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.27.7.tgz", - "integrity": "sha512-OfatkLojr6U+WN5EDYuoQhtM+1xco+/6FSzJJnuWiUw5eVcicbyK3dq5EeV/QHT1uy6GoDhGbFpprUiHUYggrw==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.27.2.tgz", + "integrity": "sha512-HwGDZ0VLVBY3Y+Nw0JexZy9o/nUAWq9MlV7cahpaXKW6TOzfVno3y3/M8Ga8u8Yr7GldLOov27xiCnqRZf0tCA==", "cpu": [ "x64" ], @@ -417,9 +417,9 @@ } }, "node_modules/@esbuild/openbsd-arm64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.27.7.tgz", - "integrity": "sha512-AFuojMQTxAz75Fo8idVcqoQWEHIXFRbOc1TrVcFSgCZtQfSdc1RXgB3tjOn/krRHENUB4j00bfGjyl2mJrU37A==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.27.2.tgz", + "integrity": "sha512-DNIHH2BPQ5551A7oSHD0CKbwIA/Ox7+78/AWkbS5QoRzaqlev2uFayfSxq68EkonB+IKjiuxBFoV8ESJy8bOHA==", "cpu": [ "arm64" ], @@ -434,9 +434,9 @@ } }, "node_modules/@esbuild/openbsd-x64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.27.7.tgz", - "integrity": "sha512-+A1NJmfM8WNDv5CLVQYJ5PshuRm/4cI6WMZRg1by1GwPIQPCTs1GLEUHwiiQGT5zDdyLiRM/l1G0Pv54gvtKIg==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.27.2.tgz", + "integrity": "sha512-/it7w9Nb7+0KFIzjalNJVR5bOzA9Vay+yIPLVHfIQYG/j+j9VTH84aNB8ExGKPU4AzfaEvN9/V4HV+F+vo8OEg==", "cpu": [ "x64" ], @@ -451,9 +451,9 @@ } }, "node_modules/@esbuild/openharmony-arm64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.27.7.tgz", - "integrity": "sha512-+KrvYb/C8zA9CU/g0sR6w2RBw7IGc5J2BPnc3dYc5VJxHCSF1yNMxTV5LQ7GuKteQXZtspjFbiuW5/dOj7H4Yw==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.27.2.tgz", + "integrity": "sha512-LRBbCmiU51IXfeXk59csuX/aSaToeG7w48nMwA6049Y4J4+VbWALAuXcs+qcD04rHDuSCSRKdmY63sruDS5qag==", "cpu": [ "arm64" ], @@ -468,9 +468,9 @@ } }, "node_modules/@esbuild/sunos-x64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.27.7.tgz", - "integrity": "sha512-ikktIhFBzQNt/QDyOL580ti9+5mL/YZeUPKU2ivGtGjdTYoqz6jObj6nOMfhASpS4GU4Q/Clh1QtxWAvcYKamA==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.27.2.tgz", + "integrity": "sha512-kMtx1yqJHTmqaqHPAzKCAkDaKsffmXkPHThSfRwZGyuqyIeBvf08KSsYXl+abf5HDAPMJIPnbBfXvP2ZC2TfHg==", "cpu": [ "x64" ], @@ -485,9 +485,9 @@ } }, "node_modules/@esbuild/win32-arm64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.27.7.tgz", - "integrity": "sha512-7yRhbHvPqSpRUV7Q20VuDwbjW5kIMwTHpptuUzV+AA46kiPze5Z7qgt6CLCK3pWFrHeNfDd1VKgyP4O+ng17CA==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.27.2.tgz", + "integrity": "sha512-Yaf78O/B3Kkh+nKABUF++bvJv5Ijoy9AN1ww904rOXZFLWVc5OLOfL56W+C8F9xn5JQZa3UX6m+IktJnIb1Jjg==", "cpu": [ "arm64" ], @@ -502,9 +502,9 @@ } }, "node_modules/@esbuild/win32-ia32": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.27.7.tgz", - "integrity": "sha512-SmwKXe6VHIyZYbBLJrhOoCJRB/Z1tckzmgTLfFYOfpMAx63BJEaL9ExI8x7v0oAO3Zh6D/Oi1gVxEYr5oUCFhw==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.27.2.tgz", + "integrity": "sha512-Iuws0kxo4yusk7sw70Xa2E2imZU5HoixzxfGCdxwBdhiDgt9vX9VUCBhqcwY7/uh//78A1hMkkROMJq9l27oLQ==", "cpu": [ "ia32" ], @@ -519,9 +519,9 @@ } }, "node_modules/@esbuild/win32-x64": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.27.7.tgz", - "integrity": "sha512-56hiAJPhwQ1R4i+21FVF7V8kSD5zZTdHcVuRFMW0hn753vVfQN8xlx4uOPT4xoGH0Z/oVATuR82AiqSTDIpaHg==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.27.2.tgz", + "integrity": "sha512-sRdU18mcKf7F+YgheI/zGf5alZatMUTKj/jNS6l744f9u3WFu4v7twcUI9vu4mknF4Y9aDlblIie0IM+5xxaqQ==", "cpu": [ "x64" ], @@ -536,9 +536,9 @@ } }, "node_modules/@hono/node-server": { - "version": "1.19.14", - "resolved": "https://registry.npmjs.org/@hono/node-server/-/node-server-1.19.14.tgz", - "integrity": "sha512-GwtvgtXxnWsucXvbQXkRgqksiH2Qed37H9xHZocE5sA3N8O8O8/8FA3uclQXxXVzc9XBZuEOMK7+r02FmSpHtw==", + "version": "1.19.17", + "resolved": "https://registry.npmjs.org/@hono/node-server/-/node-server-1.19.17.tgz", + "integrity": "sha512-dSneS5qhiauZWGDCeK4o695Xd9nUNjviSZCMQrj10eetr8Uln1ucn6bbphOM6UynAMMtNIzZNSpL9vnASJwrPQ==", "license": "MIT", "engines": { "node": ">=18.14.1" @@ -2169,9 +2169,9 @@ ] }, "node_modules/esbuild": { - "version": "0.27.7", - "resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.27.7.tgz", - "integrity": "sha512-IxpibTjyVnmrIQo5aqNpCgoACA/dTKLTlhMHihVHhdkxKyPO1uBBthumT0rdHmcsk9uMonIWS0m4FljWzILh3w==", + "version": "0.27.2", + "resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.27.2.tgz", + "integrity": "sha512-HyNQImnsOC7X9PMNaCIeAm4ISCQXs5a5YasTXVliKv4uuBo1dKrG0A+uQS8M5eXjVMnLg3WgXaKvprHlFJQffw==", "dev": true, "hasInstallScript": true, "license": "MIT", @@ -2182,32 +2182,32 @@ "node": ">=18" }, "optionalDependencies": { - "@esbuild/aix-ppc64": "0.27.7", - "@esbuild/android-arm": "0.27.7", - "@esbuild/android-arm64": "0.27.7", - "@esbuild/android-x64": "0.27.7", - "@esbuild/darwin-arm64": "0.27.7", - "@esbuild/darwin-x64": "0.27.7", - "@esbuild/freebsd-arm64": "0.27.7", - "@esbuild/freebsd-x64": "0.27.7", - "@esbuild/linux-arm": "0.27.7", - "@esbuild/linux-arm64": "0.27.7", - "@esbuild/linux-ia32": "0.27.7", - "@esbuild/linux-loong64": "0.27.7", - "@esbuild/linux-mips64el": "0.27.7", - "@esbuild/linux-ppc64": "0.27.7", - "@esbuild/linux-riscv64": "0.27.7", - "@esbuild/linux-s390x": "0.27.7", - "@esbuild/linux-x64": "0.27.7", - "@esbuild/netbsd-arm64": "0.27.7", - "@esbuild/netbsd-x64": "0.27.7", - "@esbuild/openbsd-arm64": "0.27.7", - "@esbuild/openbsd-x64": "0.27.7", - "@esbuild/openharmony-arm64": "0.27.7", - "@esbuild/sunos-x64": "0.27.7", - "@esbuild/win32-arm64": "0.27.7", - "@esbuild/win32-ia32": "0.27.7", - "@esbuild/win32-x64": "0.27.7" + "@esbuild/aix-ppc64": "0.27.2", + "@esbuild/android-arm": "0.27.2", + "@esbuild/android-arm64": "0.27.2", + "@esbuild/android-x64": "0.27.2", + "@esbuild/darwin-arm64": "0.27.2", + "@esbuild/darwin-x64": "0.27.2", + "@esbuild/freebsd-arm64": "0.27.2", + "@esbuild/freebsd-x64": "0.27.2", + "@esbuild/linux-arm": "0.27.2", + "@esbuild/linux-arm64": "0.27.2", + "@esbuild/linux-ia32": "0.27.2", + "@esbuild/linux-loong64": "0.27.2", + "@esbuild/linux-mips64el": "0.27.2", + "@esbuild/linux-ppc64": "0.27.2", + "@esbuild/linux-riscv64": "0.27.2", + "@esbuild/linux-s390x": "0.27.2", + "@esbuild/linux-x64": "0.27.2", + "@esbuild/netbsd-arm64": "0.27.2", + "@esbuild/netbsd-x64": "0.27.2", + "@esbuild/openbsd-arm64": "0.27.2", + "@esbuild/openbsd-x64": "0.27.2", + "@esbuild/openharmony-arm64": "0.27.2", + "@esbuild/sunos-x64": "0.27.2", + "@esbuild/win32-arm64": "0.27.2", + "@esbuild/win32-ia32": "0.27.2", + "@esbuild/win32-x64": "0.27.2" } }, "node_modules/escalade": { @@ -2394,9 +2394,9 @@ } }, "node_modules/fast-uri": { - "version": "3.1.3", - "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.3.tgz", - "integrity": "sha512-i70LwGWUduXqzicKXWshooq+sWL1K3WUU5rKZNG/0i3a1OSoX3HqhH5WbWwTmqWfor4urUakGPiRQcleRZTwOg==", + "version": "3.1.5", + "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.5.tgz", + "integrity": "sha512-gHwA1O9LDIcKunMKhObS/HimwtehO1nPUECKAu5TpKgaO19fcWEl4bliWe1jWxVFvIXztJjjQ4L8XQ1EU9f7Jw==", "funding": [ { "type": "github", @@ -2679,9 +2679,9 @@ } }, "node_modules/hono": { - "version": "4.12.27", - "resolved": "https://registry.npmjs.org/hono/-/hono-4.12.27.tgz", - "integrity": "sha512-1yrb/+w6HWQJrUCLkJ2IF5jNIPvvFkblV5RNOYl6bV+OA6p9GLcMpHFFGTosSvHvcAUibuUukRqhlYI4z32C7Q==", + "version": "4.13.2", + "resolved": "https://registry.npmjs.org/hono/-/hono-4.13.2.tgz", + "integrity": "sha512-JydRilDRkYBQMt9qR9U92mXxmbGqsqSn/IKOrh4e7/gEbn+0zSr8igTu0obwJoNGN4sez28DIql7FBHWydoJpA==", "license": "MIT", "engines": { "node": ">=16.9.0" @@ -2985,9 +2985,9 @@ } }, "node_modules/ip-address": { - "version": "10.2.0", - "resolved": "https://registry.npmjs.org/ip-address/-/ip-address-10.2.0.tgz", - "integrity": "sha512-/+S6j4E9AHvW9SWMSEY9Xfy66O5PWvVEJ08O0y5JGyEKQpojb0K0GKpz/v5HJ/G0vi3D2sjGK78119oXZeE0qA==", + "version": "10.5.0", + "resolved": "https://registry.npmjs.org/ip-address/-/ip-address-10.5.0.tgz", + "integrity": "sha512-R5SnVLJmgYYvf2F2ZgwSBnelz5G4q5AxIC277GDfUaNbrZKNANcBC7RHqYYePlszf4kBolVkJauG0ZjHHFh55g==", "license": "MIT", "engines": { "node": ">= 12" @@ -3363,9 +3363,9 @@ } }, "node_modules/nanoid": { - "version": "3.3.15", - "resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.15.tgz", - "integrity": "sha512-y7Wygv/7mEOvxTuEQDB8StXdMRBWf1kR/tlhAzBRUFkB2jfcLOAxO/SHmOO2zgz1pVgK29/kyupn059/bCHdjA==", + "version": "3.3.18", + "resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.18.tgz", + "integrity": "sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w==", "dev": true, "funding": [ { @@ -3644,9 +3644,9 @@ } }, "node_modules/postcss": { - "version": "8.5.16", - "resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.16.tgz", - "integrity": "sha512-vuwillviilfKZsg0VGj5R/YwwcHx4SLsIOI/7K6mQkWx+l5cUHTjj5g0AasTBcyXsbfTgrwsUNmVUb5xVwyPwg==", + "version": "8.5.26", + "resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.26.tgz", + "integrity": "sha512-u82N74LFzG8ca+dD8puPnplTXoGH4fTPpVGuIbt36G3qvNlkvfD0lEAZSxaly3KX8TS/L1A1gsCEmvKmBcVbkQ==", "dev": true, "funding": [ { @@ -3664,7 +3664,7 @@ ], "license": "MIT", "dependencies": { - "nanoid": "^3.3.12", + "nanoid": "^3.3.17", "picocolors": "^1.1.1", "source-map-js": "^1.2.1" }, diff --git a/package.json b/package.json index 12cd781..008071d 100644 --- a/package.json +++ b/package.json @@ -47,5 +47,8 @@ "tsx": "^4.19.0", "typescript": "^5.7.0", "vitest": "^3.0.0" + }, + "allowScripts": { + "esbuild@0.27.2": true } } diff --git a/src/agent/events.ts b/src/agent/events.ts index 2d194b0..2975d4e 100644 --- a/src/agent/events.ts +++ b/src/agent/events.ts @@ -7,8 +7,13 @@ export type AgentEvent = * emitted when a partially-streamed native tool-call turn turns out to have malformed args and is * retried non-streaming, so the UI doesn't carry the stale partial into the retry's output. */ | { type: "stream_discard" } - | { type: "tool_call"; label: string } - | { type: "tool_result"; summary: string; isError: boolean } + /** `name`/`args` are only populated for an actually-resolved tool call (not the "unknown tool" + * error path) — the file panel's Activity tab (App.tsx) uses them to track which files a + * read_file/write_file/edit_file call touched, without having to re-parse the display `label`. */ + | { type: "tool_call"; label: string; name?: string; args?: unknown } + /** `name`/`result` mirror `tool_call`'s — only populated when a tool actually ran (not a + * hook-blocked/denied/unknown-tool result), for the same file-panel tracking purpose. */ + | { type: "tool_result"; summary: string; isError: boolean; name?: string; result?: unknown } /** A sub-agent's tool call or result, forwarded to the parent so its work is visible while it * runs headless. Routed to the dedicated sub-agent panel below the input (not the main * scrollback) — see App.tsx. */ diff --git a/src/agent/loop.test.ts b/src/agent/loop.test.ts index 0d008da..6d98ff9 100644 --- a/src/agent/loop.test.ts +++ b/src/agent/loop.test.ts @@ -1,6 +1,6 @@ import { describe, expect, it, vi } from "vitest"; import { z } from "zod"; -import { MaxIterationsError, runTurn, shouldAutoCompact } from "./loop.js"; +import { MaxIterationsError, compactSession, runTurn, shouldAutoCompact } from "./loop.js"; import { agentTool } from "../tools/agentTool.js"; import { createSession } from "./session.js"; import { buildToolSet } from "../tools/toolset.js"; @@ -22,6 +22,43 @@ describe("shouldAutoCompact", () => { }); }); +describe("runTurn / max_tokens", () => { + it("caps max_tokens independent of a large contextWindow", async () => { + // Regression: some backends advertise a huge context window but cap a single response's + // max_tokens far below it (e.g. Ollama's glm-5.2:cloud: 1,000,000-token context, 8192-token max + // output). resolveMaxTokens used to request up to the whole remaining window, which such + // backends reject outright — worse the larger (or user-raised) contextWindow got. + let capturedMaxTokens: number | undefined; + const fakeClient = { + chat: { + completions: { + create: vi.fn(async (params: any) => { + capturedMaxTokens = params.max_tokens; + let yielded = false; + return { + [Symbol.asyncIterator]: () => ({ + next: async () => { + if (yielded) return { done: true, value: undefined }; + yielded = true; + return { done: false, value: { choices: [{ delta: { content: "done" }, finish_reason: "stop" }] } }; + }, + }), + }; + }), + }, + }, + } as any; + + const session = createSession(fakeClient, "test-model", process.cwd(), async () => "once", "native", []); + session.contextWindow = 1_000_000; + session.lastContextTokens = 500; + + await runTurn(session, "hi", () => {}); + + expect(capturedMaxTokens).toBeLessThanOrEqual(8192); + }); +}); + describe("runTurn / max iterations", () => { it("throws MaxIterationsError (not a generic error) when a model keeps calling tools forever", async () => { // A no-op tool the fake model calls on every single turn, forever — simulates a model that @@ -284,6 +321,66 @@ describe("runTurn / max iterations", () => { expect(capturedSignal).toBe(ac.signal); }); + it("stops mid-stream when the caller's signal aborts, without misreporting it as an idle timeout", async () => { + // Regression for Escape-to-interrupt (App.tsx now threads a per-turn AbortController's signal + // into the top-level runTurn call the same way sub-agents already did). This covers the other + // half of that feature from the previous test: interrupting while the model is still streaming + // plain prose, with no tool call or confirm() involved at all. + let sawSecondNext = false; + const fakeClient = { + chat: { + completions: { + create: vi.fn(async (_params: any, options: any) => { + const signal: AbortSignal | undefined = options?.signal; + let calls = 0; + return { + [Symbol.asyncIterator]: () => ({ + next: async () => { + calls++; + if (calls === 1) { + return { done: false, value: { choices: [{ delta: { content: "Hello" }, index: 0 }] } }; + } + sawSecondNext = true; + // A real aborted fetch stream rejects rather than yielding forever — mimic that, + // gated on the actual signal so this test depends on abort reaching the request. + return new Promise((_resolve, reject) => { + signal?.addEventListener( + "abort", + () => { + const err = new Error("The operation was aborted."); + err.name = "AbortError"; + reject(err); + }, + { once: true }, + ); + }); + }, + }), + }; + }), + }, + }, + } as any; + + const session = createSession(fakeClient, "test-model", process.cwd(), async () => "once", "native", []); + const ac = new AbortController(); + + const turnPromise = runTurn(session, "say something long", () => {}, undefined, ac.signal); + // Let the first content chunk land (setting the streaming text in motion) before interrupting. + await new Promise((resolve) => setTimeout(resolve, 10)); + ac.abort(); + + await expect(turnPromise).rejects.toThrow(); + expect(sawSecondNext).toBe(true); + // The whole point of threading a real external signal (vs. relying on the idle guard alone) is + // that App.tsx can tell "the user stopped this" apart from "the backend actually hung" — which + // it does by checking its own AbortController's `.aborted` flag, not by string-matching the + // error. But the idle-timeout message specifically must not leak out here, since it would be an + // actively misleading claim (the backend didn't go silent; the user interrupted it) if App.tsx's + // fallback error-message path (or any future caller) ever surfaced it directly. + await expect(turnPromise).rejects.not.toThrow(/Backend stopped responding mid-stream/); + }); + it("prunes older images from history, keeping only the most recent few at full resolution", async () => { // Unlike text tool results, an image's base64 payload has no per-call cap and (without pruning) // gets resent in full on every subsequent request for the rest of the session — a handful of @@ -497,3 +594,92 @@ describe("runTurn / max iterations", () => { expect(String((toolResultMessage as any).content)).toContain("plan mode is active"); }); }); + +describe("compactSession — partial-history preservation", () => { + function fakeClientReturning(summary: string) { + return { + chat: { + completions: { + create: vi.fn(async () => ({ + choices: [{ message: { role: "assistant", content: summary } }], + usage: { prompt_tokens: 100, completion_tokens: 20 }, + })), + }, + }, + } as any; + } + + it("preserves recent tail messages verbatim and summarizes the older prefix", async () => { + const client = fakeClientReturning("SUMMARY OF EARLIER WORK"); + const session = createSession(client, "test-model", process.cwd(), async () => "once", "native", []); + session.contextWindow = 4096; + + // Build a history longer than the preserved tail: system + several user/assistant turns. + // Index 0 is the system prompt that createSession already added. + for (let i = 0; i < 12; i++) { + session.messages.push({ role: "user", content: `user message ${i}` } as any); + session.messages.push({ role: "assistant", content: `assistant reply ${i}` } as any); + } + + const beforeLen = session.messages.length; + const summary = await compactSession(session); + + expect(summary).toBe("SUMMARY OF EARLIER WORK"); + // The recap must be present. + const recap = session.messages.find((m) => m.role === "assistant" && typeof m.content === "string" && (m.content as string).includes("[Earlier conversation compacted")); + expect(recap).toBeDefined(); + // The most recent assistant reply must survive compaction (it is in the tail). + expect(session.messages.some((m) => m.role === "assistant" && (m.content as string) === "assistant reply 11")).toBe(true); + // The oldest user message (well before the tail) must NOT survive verbatim — it was summarized. + expect(session.messages.some((m) => m.role === "user" && (m.content as string) === "user message 0")).toBe(false); + // History shrunk but is not empty. + expect(session.messages.length).toBeLessThan(beforeLen); + expect(session.messages.length).toBeGreaterThan(1); + }); + + it("keeps an assistant tool_calls message grouped with its tool result messages in the tail", async () => { + const client = fakeClientReturning("SUMMARY"); + const session = createSession(client, "test-model", process.cwd(), async () => "once", "native", []); + session.contextWindow = 4096; + + // Add enough history that the boundary lands inside the tool-call group. + for (let i = 0; i < 10; i++) { + session.messages.push({ role: "user", content: `u${i}` } as any); + session.messages.push({ role: "assistant", content: `a${i}` } as any); + } + // Now the most recent turns: an assistant tool_calls message + its tool results. + session.messages.push({ + role: "assistant", + content: null, + tool_calls: [{ id: "call_0", type: "function", function: { name: "read_file", arguments: "{}" } }], + } as any); + session.messages.push({ role: "tool", tool_call_id: "call_0", content: "file contents" } as any); + + await compactSession(session); + + // The tool_calls assistant message and its tool result must both be present in the tail. + const hasCall = session.messages.some( + (m) => m.role === "assistant" && Array.isArray((m as any).tool_calls) && (m as any).tool_calls.some((c: any) => c.id === "call_0"), + ); + const hasResult = session.messages.some((m) => m.role === "tool" && (m as any).tool_call_id === "call_0"); + // Both or neither — never one without the other (that would be a malformed history). + expect(hasCall).toBe(hasResult); + expect(hasCall).toBe(true); + }); + + it("does not call the model when there is no older prefix to summarize", async () => { + const client = fakeClientReturning("SHOULD NOT BE USED"); + const session = createSession(client, "test-model", process.cwd(), async () => "once", "native", []); + session.contextWindow = 4096; + // Only system + a couple of messages: the tail covers everything, so no summary request. + session.messages.push({ role: "user", content: "hi" } as any); + session.messages.push({ role: "assistant", content: "hello" } as any); + + const summary = await compactSession(session); + + expect(summary).toBe(""); + expect((client.chat.completions.create as any).mock.calls.length).toBe(0); + // The recent messages are still there verbatim. + expect(session.messages.some((m) => m.role === "assistant" && (m.content as string) === "hello")).toBe(true); + }); +}); diff --git a/src/agent/loop.ts b/src/agent/loop.ts index 9dfcb42..fd3c185 100644 --- a/src/agent/loop.ts +++ b/src/agent/loop.ts @@ -16,7 +16,7 @@ import { resolveToolInvocation, runTool, type ResolvedToolCall } from "../toolca import { formatCallLabel, summarizeToolResult } from "../ui/toolSummary.js"; import { estimateTokens } from "../utils/tokens.js"; import { runHooksForEvent } from "../hooks/runner.js"; -import { resolveRequestTimeoutMs, resolveSubagentTimeoutMs } from "../config/config.js"; +import { resolveMaxOutputTokens, resolveRequestTimeoutMs, resolveSubagentTimeoutMs } from "../config/config.js"; import { buildSystemPrompt } from "./systemPrompt.js"; import type { Session } from "./session.js"; @@ -155,63 +155,187 @@ export function shouldAutoCompact(session: Session): boolean { return contextUsageRatio(session) >= session.autoCompactThreshold; } +/** Computes a dynamic `max_tokens` for a generation request, instead of the old hardcoded 4096 + * that was far too small for local models — a full-file rewrite (common in fallback mode, where + * precise edits are hard) can easily exceed 4096 tokens and get truncated mid-tool-call, turning + * a valid call into malformed JSON the parser then rejects. + * + * Reserves `contextWindow − lastContextTokens` for the response, minus a small safety margin so + * the prompt+response never overshoots the window (a local backend will typically OOM or error + * on an oversized request rather than gracefully truncating). Clamped to [2048, contextWindow] + * so a nearly-full context still gets a usable (if small) output budget, and a huge window does + * not ask for more than the model could ever produce — but also capped at resolveMaxOutputTokens() + * independent of contextWindow, since many backends cap a single response far below their total + * context window (e.g. Ollama's glm-5.2:cloud: 1,000,000-token context, 8192-token max output). + * Without that second cap, a large (or user-raised, see `contextWindow` config) window made + * resolveMaxTokens request far more than such a backend allows, which it rejects outright as a + * context/length error even on the very first turn — raising contextWindow to fix truncation made + * this worse, not better. */ +function resolveMaxTokens(session: Session): number { + const MARGIN = 512; + const MIN = 2048; + const available = session.contextWindow - session.lastContextTokens - MARGIN; + const cap = Math.min(session.contextWindow, resolveMaxOutputTokens()); + return Math.min(cap, Math.max(MIN, Math.min(available, cap))); +} + /** - * Replaces the conversation history with a model-generated summary of everything so far, to free - * up context. Runs as a plain (non-tool-calling) request so the model just produces prose, not - * more tool calls. The new history is just [system, a synthetic assistant "recap"] — framing the - * summary as an assistant turn keeps proper role alternation with whatever real user message - * follows next. + * Maximum number of trailing messages to keep verbatim across a compaction. Sized so the + * preserved tail is large enough to carry a couple of recent tool calls + results (the context + * the model most needs to continue), but small enough that summarizing the older prefix actually + * frees meaningful context. The tail is also capped by a fraction of the context window so a + * handful of very large messages (e.g. big file reads) can not crowd out the summary. + */ +const MAX_PRESERVED_TAIL_MESSAGES = 8; +const MAX_PRESERVED_TAIL_FRACTION = 0.3; + +/** + * Finds the index into `session.messages` at which the preserved tail should begin, so that the + * tail keeps the most recent turns verbatim while the older prefix gets summarized. The boundary + * is always placed on a safe edge — never between an assistant tool_calls message and its `tool` + * result messages, since splitting that group would leave dangling tool calls (no results) or + * orphaned results (no triggering call), which OpenAI-compatible backends reject as malformed. + * + * Returns at least 1 (never splits before the system prompt at index 0). + */ +function computeKeepBoundary(session: Session): number { + const msgs = session.messages; + if (msgs.length <= 1) return 1; + + // Start from the end and walk backward, collecting messages into the tail until we hit the + // message budget. Tool-result messages must stay grouped with the assistant tool_calls message + // that precedes them, so when we reach one we keep scanning left until that call is included. + let boundary = msgs.length; + let kept = 0; + const tailTokenBudget = Math.floor(session.contextWindow * MAX_PRESERVED_TAIL_FRACTION); + let tailTokens = 0; + + for (let i = msgs.length - 1; i >= 1; i--) { + const msg = msgs[i]!; + const isToolResult = msg.role === "tool" || isFallbackToolResult(msg); + const isAssistantToolCall = msg.role === "assistant" && Array.isArray((msg as any).tool_calls) && (msg as any).tool_calls.length > 0; + + // If we already started a tail and the next message left is a tool result whose triggering + // assistant call is even further left, keep extending left to include that call — otherwise + // the tail would start with orphaned tool results (no triggering call), which backends reject. + if (isToolResult && boundary < msgs.length && boundary > i + 1) { + tailTokens += estimateTokens([msg]); + boundary = i; + continue; + } + + if (kept >= MAX_PRESERVED_TAIL_MESSAGES || tailTokens >= tailTokenBudget) { + break; + } + + tailTokens += estimateTokens([msg]); + boundary = i; + kept++; + } + + // Never return 0 — the system prompt at index 0 always belongs to the summarized prefix + // (the summary request needs it as framing, and the post-compaction history rebuilds it). + return Math.max(1, boundary); +} + +/** Detects a fallback-mode tool result: a user-role message whose content is a tool_result block. */ +function isFallbackToolResult(msg: ChatCompletionMessageParam): boolean { + if (msg.role !== "user") return false; + const content = (msg as { content?: unknown }).content; + if (typeof content !== "string") return false; + return content.startsWith("```tool_result\n"); +} + +/** + * Partial-history compaction: summarizes the older prefix of the conversation and keeps the most + * recent turns verbatim, to free context without discarding the exact tool calls + results the + * model most needs to continue its task. Runs a plain (non-tool-calling) request for the summary. + * The new history is [system, a synthetic assistant "recap", ...preserved tail] — framing the + * summary as an assistant turn keeps proper role alternation with the preserved messages that + * follow (which already begin with whatever role naturally came next in the original history). */ export async function compactSession(session: Session): Promise { - const requestMessages: ChatCompletionMessageParam[] = [ - ...session.messages, - { - role: "user", - content: - "Summarize this entire conversation so far, concisely but completely: the user's goals, key decisions " + - "made, files created/changed and why, current task state, and anything still outstanding. Write it as " + - "background context for continuing the conversation — plain prose, no meta-commentary about summarizing.", - }, - ]; + // Partial-history preservation: keep the most recent turns verbatim and only summarize the + // older prefix. On a local backend a full compaction is expensive (one extra model request) and + // the freshly-discarded turns are exactly the ones the model most needs to continue its task — + // losing the last tool call + its result, for example, leaves it unable to reference what it + // just did. So split the history at a safe boundary (never mid-tool-call) and keep the tail. + const keepFromIndex = computeKeepBoundary(session); - const requestStart = Date.now(); - const compactGuard = createIdleAbort(resolveRequestTimeoutMs()); - let res; - try { - res = await session.client.chat.completions.create( + // The prefix to summarize is everything before the preserved tail (always including the + // system prompt at index 0, which the model needs as framing for the summary request). + const prefix = session.messages.slice(0, keepFromIndex); + const tail = session.messages.slice(keepFromIndex); + + // If there's almost nothing old to summarize (e.g. a very short history or one where the tail + // already covers most of it), don't waste a model request producing a near-empty summary — + // just keep the tail verbatim. This also avoids a degenerate "[system, recap, single-message]" + // result when compaction triggers early on a small history. + const prefixHasContent = prefix.length > 1; // more than just the system prompt + let summary: string | null = null; + + if (prefixHasContent) { + const requestMessages: ChatCompletionMessageParam[] = [ + ...prefix, { - model: session.model, - messages: requestMessages, - stream: false, - max_tokens: 1024, + role: "user", + content: + "Summarize the conversation up to this point, concisely but completely: the user's goals, key decisions " + + "made, files created/changed and why, current task state, and anything still outstanding. Write it as " + + "background context for continuing the conversation — plain prose, no meta-commentary about summarizing. " + + "(The most recent turns are kept verbatim after this summary, so focus on the earlier history.)", }, - { signal: compactGuard.signal }, - ); - } catch (err) { - if (compactGuard.didTimeOut()) { - throw new AgentError( - `Compaction failed: backend stopped responding (no data for ${Math.round(resolveRequestTimeoutMs() / 1000)}s).`, + ]; + + const requestStart = Date.now(); + const compactGuard = createIdleAbort(resolveRequestTimeoutMs()); + let res; + try { + res = await session.client.chat.completions.create( + { + model: session.model, + messages: requestMessages, + stream: false, + max_tokens: 1024, + }, + { signal: compactGuard.signal }, ); + } catch (err) { + if (compactGuard.didTimeOut()) { + throw new AgentError( + `Compaction failed: backend stopped responding (no data for ${Math.round(resolveRequestTimeoutMs() / 1000)}s).`, + ); + } + throw err; + } finally { + compactGuard.dispose(); + } + session.stats.modelTimeMs += Date.now() - requestStart; + recordUsage(session, res.usage); + summary = res.choices[0]?.message?.content ?? null; + if (!summary) { + throw new AgentError("Compaction failed: the model returned no summary."); } - throw err; - } finally { - compactGuard.dispose(); - } - session.stats.modelTimeMs += Date.now() - requestStart; - recordUsage(session, res.usage); - const summary = res.choices[0]?.message?.content; - if (!summary) { - throw new AgentError("Compaction failed: the model returned no summary."); } - session.messages = [ - { role: "system", content: buildSystemPrompt(session.toolset.tools, session.mode, session.projectInstructions) }, - { role: "assistant", content: `[Earlier conversation compacted to save context]\n\n${summary}` }, - ]; + if (summary) { + session.messages = [ + { role: "system", content: buildSystemPrompt(session.toolset.tools, session.mode, session.projectInstructions) }, + { role: "assistant", content: `[Earlier conversation compacted to save context]\n\n${summary}` }, + ...tail, + ]; + } else { + // Nothing to summarize: keep the tail as-is, but still rebuild the system prompt in case + // the tail's first message was a stale system prompt we want to replace. + session.messages = [ + { role: "system", content: buildSystemPrompt(session.toolset.tools, session.mode, session.projectInstructions) }, + ...tail, + ]; + } // res.usage describes the old (now-discarded) prompt, not the new shorter history — estimate fresh. session.lastContextTokens = estimateTokens(session.messages); session.lastContextTokensIsEstimate = true; - return summary; + return summary ?? ""; } /** Accumulates streaming tool-call deltas into complete tool calls. */ @@ -363,7 +487,7 @@ async function gateAndRun( return { error: resolved.error }; } const { tool, args } = resolved; - emit({ type: "tool_call", label: formatCallLabel(tool.name, args) }); + emit({ type: "tool_call", label: formatCallLabel(tool.name, args), name: tool.name, args }); // Only `bash` is backgroundable today — the control object is how a mid-flight Ctrl+B (flipped // on session.activeBackground by the UI) reaches into this specific call's polling loop. Reset @@ -440,7 +564,7 @@ async function gateAndRun( .then((r) => emitHookWarnings(r.warnings, emit)) .catch(() => {}); } - emit({ type: "tool_result", summary: summarizeToolResult(tool.name, result), isError }); + emit({ type: "tool_result", summary: summarizeToolResult(tool.name, result), isError, name: tool.name, result }); // PostToolUse can't *veto* a tool call that already ran (unlike PreToolUse), but it IS awaited: // an auto-format hook that rewrites the file the tool just wrote needs to finish before the next @@ -455,6 +579,39 @@ async function gateAndRun( } } +/** + * Runs a batch of tool calls, parallelizing read-only tools for throughput (a local backend can + * serve several independent file reads / greps / web fetches concurrently, where sequential + * execution just adds latency), while keeping mutating tools strictly sequential — they share + * order-sensitive session state (the permission prompt, `activeBackground`, mutationCommitLength, + * FileChanged hooks) that concurrent execution would race on. If every call in the batch is + * read-only the whole batch runs concurrently; if any are mutating, the batch runs sequentially to + * preserve the existing ordering guarantees (and keep permission prompts in a deterministic order). + * + * Returns results in the *original* call order regardless of execution order, so callers can push + * the corresponding tool-result messages in the order the model emitted the calls. + */ +async function runToolBatch( + calls: { resolved: ResolvedToolCall; label: string }[], + session: Session, + emit: AgentEventHandler, + signal?: AbortSignal, +): Promise { + if (calls.length === 0) return []; + const allReadOnly = calls.every((c) => !("error" in c.resolved) && !c.resolved.tool.mutating); + if (!allReadOnly || calls.length === 1) { + const results: unknown[] = []; + for (const c of calls) { + results.push(await gateAndRun(c.resolved, c.label, session, emit, signal)); + } + return results; + } + // All read-only: run concurrently. gateAndRun's shared-state side effects are safe for read-only + // tools (no permission prompt, activeBackground stays null, no mutating hooks), and the + // tool_call/tool_result emit events interleaving is purely cosmetic. + return Promise.all(calls.map((c) => gateAndRun(c.resolved, c.label, session, emit, signal))); +} + /** * Runs a sub-agent as a fresh, isolated turn loop that shares the parent's client/model/cwd/ * permissions, but starts with no conversation history beyond the delegated task. Runs headless — @@ -614,10 +771,20 @@ async function handleCompletedMessage( } as ChatCompletionMessageParam); const pendingImages: ImageAttachment[] = []; - for (const call of message.tool_calls) { + const nativeCalls: { resolved: ResolvedToolCall; label: string; call: any }[] = message.tool_calls.map((call: any) => { const resolved = resolveToolCall(call as any, toolset.registry); const label = call.type === "function" ? `${call.function.name}(${call.function.arguments})` : call.type; - const result = await gateAndRun(resolved, label, session, emit, signal); + return { resolved, label, call }; + }); + const results = await runToolBatch( + nativeCalls.map((c) => ({ resolved: c.resolved, label: c.label })), + session, + emit, + signal, + ); + for (let i = 0; i < nativeCalls.length; i++) { + const { resolved, call } = nativeCalls[i]!; + const result = results[i]!; const image = pushToolResultMessage(session, "native", call.id, call.type === "function" ? call.function.name : call.type, result); noteMutationCommit(session, resolved, result); if (image) pendingImages.push(image); @@ -635,12 +802,21 @@ async function handleCompletedMessage( if (parsed.calls.length) { session.messages.push({ role: "assistant", content: text }); - for (const call of parsed.calls) { + const fbCalls: { resolved: ResolvedToolCall; label: string; call: { name: string; arguments: any } }[] = parsed.calls.map((call) => { const resolved = resolveToolInvocation(call.name, call.arguments, toolset.registry); const label = `${call.name}(${JSON.stringify(call.arguments)})`; - const result = await gateAndRun(resolved, label, session, emit, signal); - pushToolResultMessage(session, "fallback", "", call.name, result); - noteMutationCommit(session, resolved, result); + return { resolved, label, call }; + }); + const fbResults = await runToolBatch( + fbCalls.map((c) => ({ resolved: c.resolved, label: c.label })), + session, + emit, + signal, + ); + for (let i = 0; i < fbCalls.length; i++) { + const { resolved, call } = fbCalls[i]!; + pushToolResultMessage(session, "fallback", "", call.name, fbResults[i]!); + noteMutationCommit(session, resolved, fbResults[i]!); } return { text: "", hadToolCalls: true }; } @@ -681,13 +857,13 @@ export async function runTurn( try { await compactSession(session); emit({ type: "notice", text: "Context was getting full — auto-compacted mid-turn.", isError: false }); - // compactSession leaves the history as [system, assistant-recap] with no user turn. Sending - // that to the model gives it nothing to respond to — local models routinely answer with an - // empty stop (which runTurn then returns as ""), which is exactly why sub-agents that - // compacted mid-task came back as "Sub-agent finished (0 chars)", and why a tool-heavy main - // turn appeared to hang/stop after compacting. Re-add a user turn so the model resumes the - // task instead of going empty. (Between-turn compaction in App.tsx doesn't need this — the - // user's next message supplies the turn.) + // compactSession now leaves [system, recap, ...preserved tail]. The tail usually ends with + // a tool result (role: tool in native mode, a user-role tool_result block in fallback) or an + // assistant turn — either way the model needs an explicit cue to resume rather than stop. A + // bare [system, recap] used to make local models answer with an empty stop (runTurn returns ""), + // which is why sub-agents that compacted mid-task came back "Sub-agent finished (0 chars)" and + // tool-heavy main turns appeared to hang. The synthetic user turn fixes that. (Between-turn + // compaction in App.tsx doesn't need this — the user's next message supplies the turn.) session.messages.push({ role: "user", content: @@ -726,7 +902,7 @@ export async function runTurn( tools: session.mode === "native" ? toolset.openaiTools : undefined, stream: true, stream_options: { include_usage: true }, - max_tokens: 4096, + max_tokens: resolveMaxTokens(session), }, { signal: idleGuard.signal }, ); @@ -837,7 +1013,7 @@ export async function runTurn( messages: session.messages, tools: toolset.openaiTools, stream: false, - max_tokens: 4096, + max_tokens: resolveMaxTokens(session), }, { signal: retryGuard.signal }, ); @@ -881,13 +1057,23 @@ export async function runTurn( } as ChatCompletionMessageParam); const pendingImages: ImageAttachment[] = []; - for (const tc of accumulatedToolCalls) { + const streamNativeCalls: { resolved: ResolvedToolCall; label: string; tc: AccumulatedToolCall }[] = accumulatedToolCalls.map((tc) => { const resolved = resolveToolCall( { id: tc.id, type: "function", function: { name: tc.name, arguments: tc.arguments } } as any, toolset.registry, ); const label = `${tc.name}(${tc.arguments})`; - const result = await gateAndRun(resolved, label, session, emit, signal); + return { resolved, label, tc }; + }); + const streamResults = await runToolBatch( + streamNativeCalls.map((c) => ({ resolved: c.resolved, label: c.label })), + session, + emit, + signal, + ); + for (let i = 0; i < streamNativeCalls.length; i++) { + const { resolved, tc } = streamNativeCalls[i]!; + const result = streamResults[i]!; const image = pushToolResultMessage(session, "native", tc.id, tc.name, result); noteMutationCommit(session, resolved, result); if (image) pendingImages.push(image); @@ -905,12 +1091,21 @@ export async function runTurn( if (parsed.calls.length) { emit({ type: "text_done", fullText }); session.messages.push({ role: "assistant", content: fullText }); - for (const call of parsed.calls) { + const streamFbCalls: { resolved: ResolvedToolCall; label: string; call: { name: string; arguments: any } }[] = parsed.calls.map((call) => { const resolved = resolveToolInvocation(call.name, call.arguments, toolset.registry); const label = `${call.name}(${JSON.stringify(call.arguments)})`; - const result = await gateAndRun(resolved, label, session, emit, signal); - pushToolResultMessage(session, "fallback", "", call.name, result); - noteMutationCommit(session, resolved, result); + return { resolved, label, call }; + }); + const streamFbResults = await runToolBatch( + streamFbCalls.map((c) => ({ resolved: c.resolved, label: c.label })), + session, + emit, + signal, + ); + for (let i = 0; i < streamFbCalls.length; i++) { + const { resolved, call } = streamFbCalls[i]!; + pushToolResultMessage(session, "fallback", "", call.name, streamFbResults[i]!); + noteMutationCommit(session, resolved, streamFbResults[i]!); } continue; } @@ -936,6 +1131,9 @@ export async function runTurn( } throw new MaxIterationsError( - `Paused after ${session.maxIterations} steps in this turn. Everything done so far (including any file edits) is saved — send another message to continue.`, + `Paused after ${session.maxIterations} steps in this turn (each step is one model request). ` + + `Everything done so far — including any file edits or tool results — is saved in the conversation. ` + + `Send another message to continue from where it stopped (e.g. "continue" or "keep going"). ` + + `If this happens often, raise the limit with: locode config set maxIterations .`, ); } \ No newline at end of file diff --git a/src/backend/client.ts b/src/backend/client.ts index 127ba79..19eae90 100644 --- a/src/backend/client.ts +++ b/src/backend/client.ts @@ -1,5 +1,5 @@ import OpenAI from "openai"; -import { resolveRequestTimeoutMs } from "../config/config.js"; +import { resolveMaxRetries, resolveRequestTimeoutMs } from "../config/config.js"; import type { AppConfig } from "../config/types.js"; export function makeClient(cfg: AppConfig): OpenAI { @@ -15,6 +15,10 @@ export function makeClient(cfg: AppConfig): OpenAI { // requests behind a concurrency limit (e.g. Ollama's OLLAMA_NUM_PARALLEL) can legitimately take // longer than the 180s default to even start serving a request under contention. timeout: resolveRequestTimeoutMs(), - maxRetries: 0, + // Configurable retries on transient failures (connection errors, 429, 5xx) with exponential + // backoff. Defaults to 0 (fail immediately) to preserve the old behavior, since a local + // backend's slow response usually means the model is stuck rather than a transient blip — but + // raise via `maxRetries` / LOCODE_MAX_RETRIES for setups with occasional connection drops. + maxRetries: resolveMaxRetries(), }); } diff --git a/src/backend/contextWindowCache.test.ts b/src/backend/contextWindowCache.test.ts index d562a33..5d52dd0 100644 --- a/src/backend/contextWindowCache.test.ts +++ b/src/backend/contextWindowCache.test.ts @@ -4,6 +4,8 @@ import envPaths from "env-paths"; import { afterEach, beforeEach, describe, expect, it } from "vitest"; import { getCachedContextWindow } from "./contextWindowCache.js"; +const KEY = "http://localhost:11434/v1::ttl-model"; + const cacheFile = path.join(envPaths("locode", { suffix: "" }).config, "context-windows.json"); describe("contextWindowCache", () => { @@ -19,16 +21,18 @@ describe("contextWindowCache", () => { } else if (existsSync(cacheFile)) { rmSync(cacheFile); } + delete process.env.LOCODE_CONTEXT_WINDOW_CACHE_TTL_DAYS; }); it("reads a well-formed cached entry", () => { // Written directly (rather than via setCachedContextWindow, whose write is fire-and-forget // async and would race this file-backed cache's always-read-from-disk load()) so the test is - // deterministic and can't leak a pending write past its own afterEach cleanup. + // deterministic and can't leak a pending write past its own afterEach cleanup. Includes a + // fresh cachedAt so the TTL check treats it as current. mkdirSync(path.dirname(cacheFile), { recursive: true }); - writeFileSync(cacheFile, JSON.stringify({ "http://localhost:11434/v1::test-model": { value: 32768, isEstimate: false } }), "utf-8"); + writeFileSync(cacheFile, JSON.stringify({ "http://localhost:11434/v1::test-model": { value: 32768, isEstimate: false, cachedAt: Date.now() } }), "utf-8"); - expect(getCachedContextWindow("http://localhost:11434/v1", "test-model")).toEqual({ value: 32768, isEstimate: false }); + expect(getCachedContextWindow("http://localhost:11434/v1", "test-model")).toEqual({ value: 32768, isEstimate: false, cachedAt: expect.any(Number) }); }); it("treats a legacy bare-number cache entry as a miss instead of returning {value: undefined}", () => { @@ -42,4 +46,41 @@ describe("contextWindowCache", () => { const result = getCachedContextWindow("http://localhost:11434/v1", "legacy-model"); expect(result).toBeUndefined(); }); + + it("treats an entry without cachedAt as expired (re-detect)", () => { + mkdirSync(path.dirname(cacheFile), { recursive: true }); + writeFileSync(cacheFile, JSON.stringify({ [KEY]: { value: 32768, isEstimate: false } }), "utf-8"); + // No cachedAt field — legacy entry from before TTL was added; should be a miss. + expect(getCachedContextWindow("http://localhost:11434/v1", "ttl-model")).toBeUndefined(); + }); + + it("treats a fresh entry (recent cachedAt) as a hit", () => { + mkdirSync(path.dirname(cacheFile), { recursive: true }); + writeFileSync(cacheFile, JSON.stringify({ [KEY]: { value: 32768, isEstimate: false, cachedAt: Date.now() } }), "utf-8"); + expect(getCachedContextWindow("http://localhost:11434/v1", "ttl-model")).toEqual({ value: 32768, isEstimate: false, cachedAt: expect.any(Number) }); + }); + + it("treats an old entry (cachedAt beyond TTL) as a miss", () => { + mkdirSync(path.dirname(cacheFile), { recursive: true }); + // 30 days ago, default TTL is 7 days — stale. + const old = Date.now() - 30 * 24 * 60 * 60 * 1000; + writeFileSync(cacheFile, JSON.stringify({ [KEY]: { value: 32768, isEstimate: false, cachedAt: old } }), "utf-8"); + expect(getCachedContextWindow("http://localhost:11434/v1", "ttl-model")).toBeUndefined(); + }); + + it("respects a configured TTL of 0 (always re-detect)", () => { + process.env.LOCODE_CONTEXT_WINDOW_CACHE_TTL_DAYS = "0"; + mkdirSync(path.dirname(cacheFile), { recursive: true }); + writeFileSync(cacheFile, JSON.stringify({ [KEY]: { value: 32768, isEstimate: false, cachedAt: Date.now() } }), "utf-8"); + expect(getCachedContextWindow("http://localhost:11434/v1", "ttl-model")).toBeUndefined(); + }); + + it("respects a longer configured TTL", () => { + process.env.LOCODE_CONTEXT_WINDOW_CACHE_TTL_DAYS = "365"; + mkdirSync(path.dirname(cacheFile), { recursive: true }); + // 30 days ago, but TTL is now 365 days — fresh. + const old = Date.now() - 30 * 24 * 60 * 60 * 1000; + writeFileSync(cacheFile, JSON.stringify({ [KEY]: { value: 32768, isEstimate: false, cachedAt: old } }), "utf-8"); + expect(getCachedContextWindow("http://localhost:11434/v1", "ttl-model")).toBeDefined(); + }); }); diff --git a/src/backend/contextWindowCache.ts b/src/backend/contextWindowCache.ts index 0939cb8..ac86a22 100644 --- a/src/backend/contextWindowCache.ts +++ b/src/backend/contextWindowCache.ts @@ -6,9 +6,23 @@ import { writeFileAtomic } from "../utils/writeFileAtomic.js"; const paths = envPaths("locode", { suffix: "" }); const cacheFile = path.join(paths.config, "context-windows.json"); +/** How long a cached context-window detection stays fresh before locode re-detects it. A model's + * context window rarely changes, but a backend can be reconfigured (quantization swapped, a + * different model loaded under the same id, Ollama's `num_ctx` raised) — a TTL avoids pinning a + * stale value forever. Set to 0 to disable caching (re-detect every session). */ +const DEFAULT_CACHE_TTL_DAYS = 7; + +export function resolveCacheTtlDays(): number { + const envValue = Number(process.env.LOCODE_CONTEXT_WINDOW_CACHE_TTL_DAYS); + if (Number.isFinite(envValue) && envValue >= 0 && envValue <= 365) return envValue; + return DEFAULT_CACHE_TTL_DAYS; +} + export interface CachedContextWindow { value: number; isEstimate: boolean; + /** Unix epoch ms when this entry was cached. Absent on legacy entries (treated as expired). */ + cachedAt?: number; } type Cache = Record; @@ -39,7 +53,17 @@ function save(cache: Cache): void { void writeFileAtomic(cacheFile, JSON.stringify(cache, null, 2)); } +/** Returns true when the entry is stale given the configured TTL. A TTL of 0 means "always + * re-detect", so every entry is stale; a missing `cachedAt` (legacy entry) is also stale. */ +function isStale(entry: CachedContextWindow, ttlDays: number): boolean { + if (ttlDays <= 0) return true; + if (typeof entry.cachedAt !== "number") return true; + const ageMs = Date.now() - entry.cachedAt; + return ageMs > ttlDays * 24 * 60 * 60 * 1000; +} + export function getCachedContextWindow(baseURL: string, model: string): CachedContextWindow | undefined { + const ttlDays = resolveCacheTtlDays(); const entry = load()[keyFor(baseURL, model)]; if (entry === undefined) return undefined; // An older locode version cached a bare number instead of { value, isEstimate }. Treat that @@ -50,11 +74,13 @@ export function getCachedContextWindow(baseURL: string, model: string): CachedCo if (typeof entry !== "object" || entry === null || typeof (entry as CachedContextWindow).value !== "number") { return undefined; } + // Expired entries are treated as a miss so the backend is re-queried and the entry refreshed. + if (isStale(entry, ttlDays)) return undefined; return entry; } export function setCachedContextWindow(baseURL: string, model: string, contextWindow: CachedContextWindow): void { const cache = load(); - cache[keyFor(baseURL, model)] = contextWindow; + cache[keyFor(baseURL, model)] = { ...contextWindow, cachedAt: Date.now() }; save(cache); -} +} \ No newline at end of file diff --git a/src/cli.ts b/src/cli.ts index 80ec3cb..de5d7d2 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -114,20 +114,24 @@ configCmd configCmd .command("set ") - .description("Persist a config value (backend, model, baseUrl, contextWindow, maxIterations, autoCompactThreshold, requestTimeoutMs, subagentTimeoutMs)") + .description( + "Persist a config value (backend, model, baseUrl, contextWindow, maxOutputTokens, maxIterations, autoCompactThreshold, requestTimeoutMs, subagentTimeoutMs, maxRetries)", + ) .action((key: string, value: string) => { if ( key !== "backend" && key !== "model" && key !== "baseUrl" && key !== "contextWindow" && + key !== "maxOutputTokens" && key !== "maxIterations" && key !== "autoCompactThreshold" && key !== "requestTimeoutMs" && - key !== "subagentTimeoutMs" + key !== "subagentTimeoutMs" && + key !== "maxRetries" ) { console.error( - `Unknown config key "${key}". Valid keys: backend, model, baseUrl, contextWindow, maxIterations, autoCompactThreshold, requestTimeoutMs, subagentTimeoutMs`, + `Unknown config key "${key}". Valid keys: backend, model, baseUrl, contextWindow, maxOutputTokens, maxIterations, autoCompactThreshold, requestTimeoutMs, subagentTimeoutMs, maxRetries`, ); process.exit(1); } @@ -139,6 +143,13 @@ configCmd process.exit(1); } stored[key] = n; + } else if (key === "maxOutputTokens") { + const n = Number(value); + if (!Number.isFinite(n) || n < 256 || n > 1_000_000) { + console.error(`maxOutputTokens must be between 256 and 1000000, got "${value}".`); + process.exit(1); + } + stored[key] = n; } else if (key === "autoCompactThreshold") { const n = Number(value); if (!Number.isFinite(n) || n < 0.1 || n > 0.95) { @@ -160,6 +171,13 @@ configCmd process.exit(1); } stored[key] = n; + } else if (key === "maxRetries") { + const n = Number(value); + if (!Number.isFinite(n) || n < 0 || n > 10) { + console.error(`maxRetries must be between 0 and 10, got "${value}".`); + process.exit(1); + } + stored[key] = n; } else { stored[key] = value; } diff --git a/src/config/config.test.ts b/src/config/config.test.ts index 8168df4..d283955 100644 --- a/src/config/config.test.ts +++ b/src/config/config.test.ts @@ -3,8 +3,8 @@ import { existsSync, mkdtempSync, rmSync } from "node:fs"; import path from "node:path"; import os from "node:os"; import { _setConfigFilePathForTest, loadStoredConfig, saveStoredConfig } from "./store.js"; -import { resolveAutoCompactThreshold, resolveContextWindowDefault, resolveMaxIterations, resolveRequestTimeoutMs, resolveSubagentTimeoutMs } from "./config.js"; -import { DEFAULT_AUTO_COMPACT_THRESHOLD, DEFAULT_CONTEXT_WINDOW, DEFAULT_MAX_ITERATIONS, DEFAULT_REQUEST_TIMEOUT_MS, DEFAULT_SUBAGENT_TIMEOUT_MS } from "./defaults.js"; +import { resolveAutoCompactThreshold, resolveContextWindowDefault, resolveMaxIterations, resolveMaxOutputTokens, resolveMaxRetries, resolveRequestTimeoutMs, resolveSubagentTimeoutMs } from "./config.js"; +import { DEFAULT_AUTO_COMPACT_THRESHOLD, DEFAULT_CONTEXT_WINDOW, DEFAULT_MAX_ITERATIONS, DEFAULT_MAX_OUTPUT_TOKENS, DEFAULT_MAX_RETRIES, DEFAULT_REQUEST_TIMEOUT_MS, DEFAULT_SUBAGENT_TIMEOUT_MS } from "./defaults.js"; // Isolate the persisted config to a temp directory so the suite never reads or overwrites the // user's real ~/.config/locode/config.json (the previous afterEach { saveStoredConfig({}) } wiped @@ -25,6 +25,9 @@ describe("config resolution", () => { delete process.env.LOCODE_AUTO_COMPACT_THRESHOLD; delete process.env.LOCODE_CONTEXT_WINDOW; delete process.env.LOCODE_MAX_ITERATIONS; + delete process.env.LOCODE_MAX_OUTPUT_TOKENS; + delete process.env.LOCODE_MAX_OUTPUT_TOKENS; + delete process.env.LOCODE_MAX_RETRIES; delete process.env.LOCODE_REQUEST_TIMEOUT_MS; delete process.env.LOCODE_SUBAGENT_TIMEOUT_MS; }); @@ -58,6 +61,27 @@ describe("config resolution", () => { expect(resolveMaxIterations()).toBe(DEFAULT_MAX_ITERATIONS); }); + it("resolves max output tokens default", () => { + expect(resolveMaxOutputTokens()).toBe(DEFAULT_MAX_OUTPUT_TOKENS); + }); + + it("reads max output tokens from env", () => { + process.env.LOCODE_MAX_OUTPUT_TOKENS = "16384"; + expect(resolveMaxOutputTokens()).toBe(16_384); + }); + + it("reads max output tokens from stored config", () => { + saveStoredConfig({ maxOutputTokens: 4096 }); + expect(resolveMaxOutputTokens()).toBe(4096); + }); + + it("rejects out-of-range max output tokens", () => { + saveStoredConfig({ maxOutputTokens: 100 }); + expect(resolveMaxOutputTokens()).toBe(DEFAULT_MAX_OUTPUT_TOKENS); + process.env.LOCODE_MAX_OUTPUT_TOKENS = "5000000"; + expect(resolveMaxOutputTokens()).toBe(DEFAULT_MAX_OUTPUT_TOKENS); + }); + it("resolves request timeout default", () => { expect(resolveRequestTimeoutMs()).toBe(DEFAULT_REQUEST_TIMEOUT_MS); }); @@ -101,4 +125,25 @@ describe("config resolution", () => { saveStoredConfig({ subagentTimeoutMs: 500 }); // stored below floor expect(resolveSubagentTimeoutMs()).toBe(DEFAULT_SUBAGENT_TIMEOUT_MS); }); + + it("resolves max retries default", () => { + expect(resolveMaxRetries()).toBe(DEFAULT_MAX_RETRIES); + }); + + it("reads max retries from env", () => { + process.env.LOCODE_MAX_RETRIES = "3"; + expect(resolveMaxRetries()).toBe(3); + }); + + it("reads max retries from stored config", () => { + saveStoredConfig({ maxRetries: 5 }); + expect(resolveMaxRetries()).toBe(5); + }); + + it("rejects out-of-range max retries", () => { + saveStoredConfig({ maxRetries: 11 }); + expect(resolveMaxRetries()).toBe(DEFAULT_MAX_RETRIES); + process.env.LOCODE_MAX_RETRIES = "-1"; + expect(resolveMaxRetries()).toBe(DEFAULT_MAX_RETRIES); + }); }); \ No newline at end of file diff --git a/src/config/config.ts b/src/config/config.ts index e0cb27d..b731aab 100644 --- a/src/config/config.ts +++ b/src/config/config.ts @@ -2,6 +2,8 @@ import { DEFAULT_AUTO_COMPACT_THRESHOLD, DEFAULT_CONTEXT_WINDOW, DEFAULT_MAX_ITERATIONS, + DEFAULT_MAX_OUTPUT_TOKENS, + DEFAULT_MAX_RETRIES, DEFAULT_REQUEST_TIMEOUT_MS, DEFAULT_SUBAGENT_TIMEOUT_MS, KNOWN_BACKENDS, @@ -56,6 +58,18 @@ export function resolveContextWindowDefault(): number { return DEFAULT_CONTEXT_WINDOW; } +/** Ceiling on a single response's max_tokens (see DEFAULT_MAX_OUTPUT_TOKENS), independent of the + * context window. Bounded to 256–1,000,000 to reject pathological values. */ +export function resolveMaxOutputTokens(): number { + const stored = loadStoredConfig(); + const envValue = Number(process.env.LOCODE_MAX_OUTPUT_TOKENS); + if (Number.isFinite(envValue) && envValue >= 256 && envValue <= 1_000_000) return envValue; + if (typeof stored.maxOutputTokens === "number" && stored.maxOutputTokens >= 256 && stored.maxOutputTokens <= 1_000_000) { + return stored.maxOutputTokens; + } + return DEFAULT_MAX_OUTPUT_TOKENS; +} + /** Max tool calls allowed per turn before locode gives up. */ export function resolveMaxIterations(): number { const stored = loadStoredConfig(); @@ -88,8 +102,22 @@ export function resolveAutoCompactThreshold(): number { return DEFAULT_AUTO_COMPACT_THRESHOLD; } +/** Max retry attempts the OpenAI SDK makes on transient failures (connection errors, 429, 5xx) + * with exponential backoff. Bounded to 0–10 to reject pathological values. 0 = fail immediately, + * matching locode's old behavior of never retrying (a slow local backend usually means the model + * is genuinely stuck, not a transient blip — but some setups have occasional connection drops). */ +export function resolveMaxRetries(): number { + const stored = loadStoredConfig(); + const envValue = Number(process.env.LOCODE_MAX_RETRIES); + if (Number.isFinite(envValue) && envValue >= 0 && envValue <= 10) return envValue; + if (typeof stored.maxRetries === "number" && stored.maxRetries >= 0 && stored.maxRetries <= 10) { + return stored.maxRetries; + } + return DEFAULT_MAX_RETRIES; +} + /** Milliseconds to wait on a single chat completion request before giving up (see backend/client.ts - * for why locode doesn't retry on top of this). Bounded to 10s–30min to reject pathological values. */ + * for why locode defaults to no retries). Bounded to 10s–30min to reject pathological values. */ export function resolveRequestTimeoutMs(): number { const stored = loadStoredConfig(); const envValue = Number(process.env.LOCODE_REQUEST_TIMEOUT_MS); diff --git a/src/config/defaults.ts b/src/config/defaults.ts index fc52d89..9a244f6 100644 --- a/src/config/defaults.ts +++ b/src/config/defaults.ts @@ -12,19 +12,38 @@ export type BackendName = keyof typeof KNOWN_BACKENDS; * and the user hasn't configured one — a conservative size common among smaller local models. */ export const DEFAULT_CONTEXT_WINDOW = 8192; -/** Max tool calls per turn before locode gives up rather than looping forever. 50 gives real - * multi-file tasks room to breathe (local models often issue one tool call per turn, so a - * multi-file edit + verify sequence can easily run past 25); still bounded so a genuinely stuck - * model fails fast, and hitting the cap is a soft pause, not a failure (see MaxIterationsError). */ -export const DEFAULT_MAX_ITERATIONS = 50; +/** Ceiling on a single response's `max_tokens`, independent of the model's context window. Most + * backends cap how much a single completion can generate well below the total context window they + * advertise (e.g. Ollama's glm-5.2:cloud reports a 1,000,000-token context window but only ever + * generates up to 8192 tokens per response) — resolveMaxTokens (agent/loop.ts) used to request up + * to the whole remaining window, which such backends rejected outright as a context/length error + * even on the very first turn. 8192 is a safe default most backends support; raise it via + * `locode config set maxOutputTokens` for backends known to allow more. */ +export const DEFAULT_MAX_OUTPUT_TOKENS = 8192; + +/** Max model requests per turn before locode pauses rather than looping forever. Each iteration + * is one model generation request (one tool-call round-trip), and local models commonly issue a + * single tool call per request — so a real multi-file task (read several files, edit each, grep + * to verify, re-read) easily needs 40–60 requests. 50 was too tight and caused frequent + * "Paused after 50 steps" soft-stops on legitimate work; 100 gives real tasks room to finish + * while still bounding a genuinely stuck model. Hitting the cap is a soft pause, not a failure + * (the work so far is intact — send another message to resume). Configurable via `maxIterations`, + * e.g. `locode config set maxIterations 200` for large batch jobs. */ +export const DEFAULT_MAX_ITERATIONS = 100; /** Fraction of the context window at which locode automatically summarizes the conversation. * User-configurable via `locode config set autoCompactThreshold`. */ export const DEFAULT_AUTO_COMPACT_THRESHOLD = 0.85; -/** How long to wait on a single chat completion request before giving up (no retries — see - * backend/client.ts). Raise this via `requestTimeoutMs` if your backend queues requests behind a - * concurrency limit (e.g. Ollama's `OLLAMA_NUM_PARALLEL`) rather than serving them immediately. */ +/** How long to wait on a single chat completion request before giving up. The OpenAI SDK retries + * transient failures (connection errors, 429, 5xx) up to `maxRetries` times with exponential + * backoff before surfacing the error; set to 0 to fail immediately like older locode versions. + * Raise this via `maxRetries` if your backend has occasional transient blips. See backend/client.ts. */ +export const DEFAULT_MAX_RETRIES = 0; + +/** How long to wait on a single chat completion request before giving up. Raise this via + * `requestTimeoutMs` if your backend queues requests behind a concurrency limit (e.g. Ollama's + * `OLLAMA_NUM_PARALLEL`) rather than serving them immediately. */ export const DEFAULT_REQUEST_TIMEOUT_MS = 180_000; /** Wall-clock budget for a single sub-agent turn. Sub-agents make their own sequence of model diff --git a/src/config/store.ts b/src/config/store.ts index ba01c45..854bfb2 100644 --- a/src/config/store.ts +++ b/src/config/store.ts @@ -16,6 +16,12 @@ export interface StoredConfig { requestTimeoutMs?: number; /** Milliseconds of wall-clock budget for a single sub-agent turn. */ subagentTimeoutMs?: number; + /** Ceiling on a single response's max_tokens, independent of contextWindow. */ + maxOutputTokens?: number; + /** Max retry attempts the OpenAI SDK makes on transient failures (connection errors, 429, 5xx) + * before surfacing the error. 0 = fail immediately (old behavior); the SDK uses exponential + * backoff between attempts. */ + maxRetries?: number; } const paths = envPaths("locode", { suffix: "" }); diff --git a/src/hooks/runner.test.ts b/src/hooks/runner.test.ts index 1300184..93d1d56 100644 --- a/src/hooks/runner.test.ts +++ b/src/hooks/runner.test.ts @@ -91,11 +91,22 @@ describe("runHooksForEvent", () => { expect(result.additionalContext).toBeUndefined(); }); - it("warns for unsupported prompt hooks", async () => { + it("injects a prompt hook's message as additional context", async () => { loadMergedHooks.loadMergedHooks.mockReturnValueOnce({ - SessionStart: [{ hooks: [{ type: "prompt", message: "ok?" } as any] }], + SessionStart: [{ hooks: [{ type: "prompt", message: "Remember to check the changelog." }] }], }); const result = await runHooksForEvent("SessionStart", ctx, {}); - expect(result.warnings).toContain("1 prompt hook(s) skipped (not yet implemented)."); + expect(result.additionalContext).toBe("Remember to check the changelog."); + expect(result.warnings).toEqual([]); + }); + + it("combines a prompt hook's message with a command hook's stdout", async () => { + loadMergedHooks.loadMergedHooks.mockReturnValueOnce({ + SessionStart: [{ hooks: [{ type: "prompt", message: "prompt message" }, { type: "command", command: "echo cmd" }] }], + }); + execa.execa.mockResolvedValueOnce({ exitCode: 0, stdout: "cmd output", stderr: "", timedOut: false }); + const result = await runHooksForEvent("SessionStart", ctx, {}); + expect(result.additionalContext).toContain("prompt message"); + expect(result.additionalContext).toContain("cmd output"); }); }); diff --git a/src/hooks/runner.ts b/src/hooks/runner.ts index 026845c..53dd1f6 100644 --- a/src/hooks/runner.ts +++ b/src/hooks/runner.ts @@ -2,7 +2,7 @@ import { execa } from "execa"; import path from "node:path"; import { loadMergedHooks } from "./config.js"; import { matcherMatches } from "./matcher.js"; -import type { Hook, HookCommand, HookEventName, HookHttp } from "./types.js"; +import type { Hook, HookCommand, HookEventName, HookHttp, HookPrompt } from "./types.js"; const DEFAULT_TIMEOUT_SECONDS = 30; @@ -37,6 +37,20 @@ function isHttpHook(hook: Hook): hook is HookHttp { return hook.type === "http"; } +function isPromptHook(hook: Hook): hook is HookPrompt { + return hook.type === "prompt"; +} + +/** A prompt hook has no process/response to run — it's just a static message that always + * "succeeds" and folds into additionalContext the same way a command/http hook's plain-text + * stdout does (see the outcome-handling loop in runHooksForEvent). Wrapped in a resolved promise + * so it can share the same Promise.all as the other hook kinds. */ +async function runPromptHook( + hook: HookPrompt, +): Promise<{ exitCode: number; stdout: string; stderr: string; timedOut: boolean; json?: unknown; warning?: string }> { + return { exitCode: 0, stdout: hook.message, stderr: "", timedOut: false }; +} + function parseOutput(output: string, schema?: "json"): { text?: string; json?: unknown; warning?: string } { if (!schema) return { text: output }; if (schema !== "json") return { text: output }; @@ -195,26 +209,18 @@ export async function runHooksForEvent( .flatMap((entry) => entry.hooks); if (hooks.length === 0) return EMPTY_RESULT; - // Prompt hooks are declared in the type system but not implemented in this pass — they need a - // blocking UI flow the current runner doesn't have. Skip them with a warning so a config that - // includes them doesn't silently do nothing. - const skippedPrompts = hooks.filter((h) => h.type === "prompt").length; - const runnableHooks = hooks.filter((hook) => hook.type !== "prompt"); - const stdinPayload = JSON.stringify({ hook_event_name: event, session_id: ctx.sessionId, cwd: ctx.cwd, ...payload }); const outcomes = await Promise.all( - runnableHooks.map(async (hook) => { + hooks.map(async (hook) => { if (isCommandHook(hook)) return runCommandHook(hook, stdinPayload, ctx); if (isHttpHook(hook)) return runHttpHook(hook, stdinPayload, ctx); + if (isPromptHook(hook)) return runPromptHook(hook); return { exitCode: 1, stdout: "", stderr: `Unsupported hook type: ${(hook as Hook).type}`, timedOut: false }; }), ); const result: HookResult = { blocked: false, warnings: [] }; - if (skippedPrompts > 0) { - result.warnings.push(`${skippedPrompts} prompt hook(s) skipped (not yet implemented).`); - } for (const outcome of outcomes) { if (outcome.warning) { result.warnings.push(outcome.warning); diff --git a/src/hooks/types.ts b/src/hooks/types.ts index 6076352..8ddf321 100644 --- a/src/hooks/types.ts +++ b/src/hooks/types.ts @@ -40,8 +40,10 @@ export interface HookHttp extends HookBase { export interface HookPrompt extends HookBase { type: "prompt"; + /** Injected verbatim as additional context, the same way a command/http hook's stdout is — + * see runner.ts. Always "succeeds" (there's no process/response to fail); a prompt hook can't + * block an event the way a command hook's exit code 2 can. */ message: string; - /** Not yet implemented — prompt hooks require a UI blocking flow the current runner doesn't support. */ } export type Hook = HookCommand | HookHttp | HookPrompt; diff --git a/src/mcp/client.ts b/src/mcp/client.ts index 88ebcf4..ed36e92 100644 --- a/src/mcp/client.ts +++ b/src/mcp/client.ts @@ -54,30 +54,53 @@ export interface ConnectedMcpServer { tools: McpToolInfo[]; } -export async function connectMcpServer(name: string, config: McpServerConfig): Promise { - const transport: Transport = isHttpServerConfig(config) - ? new StreamableHTTPClientTransport(new URL(config.url), { - requestInit: config.headers ? { headers: config.headers } : undefined, - }) - : new StdioClientTransport({ - command: config.command, - args: config.args, - env: config.env, - // Default is "inherit", which would leak the child's stderr straight into the terminal - // and corrupt Ink's alternate-screen UI. Pipe it instead so it's just discarded. - stderr: "pipe", - }); +/** Max connection attempts for an MCP server that fails transiently (process slow to start, + * HTTP 503, etc.). A stdio server whose command genuinely doesn't exist fails immediately every + * time, so retries only help the transient case — kept small to avoid stalling startup. */ +const MCP_CONNECT_MAX_ATTEMPTS = 3; +const MCP_CONNECT_BASE_DELAY_MS = 500; - const client = new Client({ name: "locode", version: "0.3.1" }); - await client.connect(transport); - try { - const { tools } = await client.listTools(); - return { name, client, transport, tools: tools as McpToolInfo[] }; - } catch (err) { - // listTools failed (server connected but never responded to the listing) — close the - // transport so the stdio subprocess / HTTP connection isn't orphaned. manager.ts catches - // the rejection as an error status, but without this the child process keeps running. - await transport.close().catch(() => {}); - throw err; +async function sleep(ms: number, signal?: AbortSignal): Promise { + return new Promise((resolve, reject) => { + const t = setTimeout(resolve, ms); + signal?.addEventListener("abort", () => { clearTimeout(t); reject(new Error("aborted")); }, { once: true }); + }); +} + +export async function connectMcpServer(name: string, config: McpServerConfig): Promise { + let lastErr: unknown; + for (let attempt = 1; attempt <= MCP_CONNECT_MAX_ATTEMPTS; attempt++) { + const transport: Transport = isHttpServerConfig(config) + ? new StreamableHTTPClientTransport(new URL(config.url), { + requestInit: config.headers ? { headers: config.headers } : undefined, + }) + : new StdioClientTransport({ + command: config.command, + args: config.args, + env: config.env, + // Default is "inherit", which would leak the child's stderr straight into the terminal + // and corrupt Ink's alternate-screen UI. Pipe it instead so it's just discarded. + stderr: "pipe", + }); + + const client = new Client({ name: "locode", version: "0.3.1" }); + try { + await client.connect(transport); + const { tools } = await client.listTools(); + return { name, client, transport, tools: tools as McpToolInfo[] }; + } catch (err) { + // listTools failed or connect failed — close the transport so the stdio subprocess / HTTP + // connection isn't orphaned. manager.ts catches the rejection as an error status, but + // without this the child process keeps running. + await transport.close().catch(() => {}); + lastErr = err; + // Retry with exponential backoff for transient failures; the last attempt's error is what + // the caller sees. A genuinely broken config (missing binary, bad URL) fails fast every time, + // so the retries just add a small delay — acceptable for the rare transient-startup case. + if (attempt < MCP_CONNECT_MAX_ATTEMPTS) { + await sleep(MCP_CONNECT_BASE_DELAY_MS * Math.pow(2, attempt - 1)); + } + } } + throw lastErr; } diff --git a/src/mcp/manager.ts b/src/mcp/manager.ts index 73a7f81..8a2a716 100644 --- a/src/mcp/manager.ts +++ b/src/mcp/manager.ts @@ -76,3 +76,12 @@ export async function disconnectAllMcpServers(): Promise { await Promise.allSettled(connections.map((c) => c.transport.close())); connections = []; } + +/** Re-connects to every configured MCP server (e.g. after the user edits `.mcp.json` or restarts a + * server process), swapping in the new connections and returning the fresh tool list so the caller + * can rebuild the session's toolset. Equivalent to a fresh `connectConfiguredMcpServers` call, + * exposed separately so callers can name the intent ("reconnect") without implying the first-run + * setup path. */ +export async function reconnectMcpServers(cwd: string): Promise { + return connectConfiguredMcpServers(cwd); +} diff --git a/src/permissions/permissionManager.ts b/src/permissions/permissionManager.ts index a6384bb..04e4d50 100644 --- a/src/permissions/permissionManager.ts +++ b/src/permissions/permissionManager.ts @@ -15,10 +15,12 @@ export class PermissionManager { /** Check whether a mutating tool should be auto-approved (no confirmation needed). */ isAutoApproved(toolName: string): boolean { - // auto-accept only covers the same file-edit tools as auto-edit, not arbitrary mutating tools - // such as bash or git_commit. This prevents a user who intended "approve edits" from silently - // approving every dangerous operation. - if (this.mode === "auto-accept" && AUTO_EDIT_TOOLS.has(toolName)) return true; + // auto-accept approves EVERY mutating tool (bash, git_commit, write_file, edit_file, ...) — + // the "I trust everything, don't ask" mode. auto-edit is the narrower "approve file edits only" + // mode, auto-approving just the file-edit tools so a user can batch-edit without approving each + // one but still gate dangerous shell/git operations. default falls back to the session-allowed + // list (tools the user approved "for this session" in a prior prompt). + if (this.mode === "auto-accept") return true; if (this.mode === "auto-edit" && AUTO_EDIT_TOOLS.has(toolName)) return true; // "default" — check session-allowed list return this.allowedForSession.has(toolName); diff --git a/src/persistence/exportSession.ts b/src/persistence/exportSession.ts index 295bf7f..223cc8f 100644 --- a/src/persistence/exportSession.ts +++ b/src/persistence/exportSession.ts @@ -3,9 +3,10 @@ import path from "node:path"; import type { ChatCompletionMessageParam } from "openai/resources/chat/completions"; import { writeFileAtomic } from "../utils/writeFileAtomic.js"; -/** Mirrors the filtering used when replaying a resumed session (see App.tsx initSessionFromRecord): - * only plain user/assistant text turns are human-readable — raw tool-call/tool-result payloads - * and fallback-mode `tool_result` blocks are internal bookkeeping, not conversation content. */ +/** A markdown export is meant to be read as prose, so only plain user/assistant text turns are + * included — raw tool-call/tool-result payloads and fallback-mode `tool_result` blocks are + * internal bookkeeping, not conversation content (unlike session resume, which does replay them + * as their own history items — see replayHistory.ts — since that's an interactive transcript). */ function messageSection(m: ChatCompletionMessageParam): string | null { if (m.role === "user" && typeof m.content === "string" && !m.content.startsWith("```tool_result")) { return `### You\n\n${m.content}`; diff --git a/src/persistence/replayHistory.test.ts b/src/persistence/replayHistory.test.ts new file mode 100644 index 0000000..26dbc38 --- /dev/null +++ b/src/persistence/replayHistory.test.ts @@ -0,0 +1,81 @@ +import { describe, expect, it } from "vitest"; +import type { ChatCompletionMessageParam } from "openai/resources/chat/completions"; +import { buildReplayHistory } from "./replayHistory.js"; + +describe("buildReplayHistory", () => { + it("replays plain user/assistant text turns", () => { + const messages: ChatCompletionMessageParam[] = [ + { role: "user", content: "hi" }, + { role: "assistant", content: "hello there" }, + ]; + expect(buildReplayHistory(messages)).toMatchObject([ + { kind: "user", text: "hi" }, + { kind: "assistant", text: "hello there" }, + ]); + }); + + it("replays native-mode tool calls and results, correlating the result's name via tool_call_id", () => { + const messages: ChatCompletionMessageParam[] = [ + { role: "user", content: "read foo.txt" }, + { + role: "assistant", + content: null, + tool_calls: [ + { id: "call_1", type: "function", function: { name: "read_file", arguments: JSON.stringify({ path: "foo.txt" }) } }, + ], + }, + { role: "tool", tool_call_id: "call_1", content: JSON.stringify({ totalLines: 3 }) }, + { role: "assistant", content: "It has 3 lines." }, + ]; + expect(buildReplayHistory(messages)).toMatchObject([ + { kind: "user", text: "read foo.txt" }, + { kind: "tool_call", label: "Read(foo.txt)" }, + { kind: "tool_result", summary: "Read 3 lines", isError: false }, + { kind: "assistant", text: "It has 3 lines." }, + ]); + }); + + it("marks a native-mode tool error result", () => { + const messages: ChatCompletionMessageParam[] = [ + { + role: "assistant", + content: null, + tool_calls: [{ id: "call_1", type: "function", function: { name: "bash", arguments: "{}" } }], + }, + { role: "tool", tool_call_id: "call_1", content: JSON.stringify({ error: "command not found" }) }, + ]; + expect(buildReplayHistory(messages)).toMatchObject([ + { kind: "tool_call", label: "Bash()" }, + { kind: "tool_result", summary: "command not found", isError: true }, + ]); + }); + + it("replays fallback-mode tool_call blocks embedded in assistant text", () => { + const messages: ChatCompletionMessageParam[] = [ + { + role: "assistant", + content: 'Let me check.\n```tool_call\n{"name":"read_file","arguments":{"path":"foo.txt"}}\n```', + }, + { + role: "user", + content: '```tool_result\n{"name":"read_file","result":{"totalLines":3}}\n```', + }, + ]; + expect(buildReplayHistory(messages)).toMatchObject([ + { kind: "assistant", text: "Let me check." }, + { kind: "tool_call", label: "Read(foo.txt)" }, + { kind: "tool_result", summary: "Read 3 lines", isError: false }, + ]); + }); + + it("drops the empty assistant text bubble when a native tool call has no accompanying prose", () => { + const messages: ChatCompletionMessageParam[] = [ + { + role: "assistant", + content: null, + tool_calls: [{ id: "call_1", type: "function", function: { name: "grep", arguments: "{}" } }], + }, + ]; + expect(buildReplayHistory(messages)).toMatchObject([{ kind: "tool_call", label: "Grep()" }]); + }); +}); diff --git a/src/persistence/replayHistory.ts b/src/persistence/replayHistory.ts new file mode 100644 index 0000000..8c8f09c --- /dev/null +++ b/src/persistence/replayHistory.ts @@ -0,0 +1,108 @@ +import type { ChatCompletionMessageParam } from "openai/resources/chat/completions"; +import { parseFallbackToolCalls } from "../toolcalling/fallbackParser.js"; +import { nextId, type HistoryItem, type NewHistoryItem } from "../ui/ink/types.js"; +import { formatCallLabel, summarizeToolResult } from "../ui/toolSummary.js"; + +const TOOL_CALL_BLOCK_RE = /```tool_call\s*[\s\S]*?```/g; +const TOOL_RESULT_BLOCK_RE = /^```tool_result\n([\s\S]*?)\n```$/; + +/** A message's `content` can be a plain string or an array of content parts (text/image_url) — + * see pushToolResultMessage in agent/loop.ts. Only the text part matters for replay display. */ +function textContent(content: ChatCompletionMessageParam["content"]): string | null { + if (typeof content === "string") return content; + if (Array.isArray(content)) { + const part = content.find((p): p is { type: "text"; text: string } => (p as { type?: string }).type === "text"); + return part?.text ?? null; + } + return null; +} + +function isErrorResult(result: unknown): boolean { + return !!(result && typeof result === "object" && "error" in (result as object)); +} + +/** Rebuilds the tool_call/tool_result HistoryItems a resumed session's transcript is otherwise + * missing (see App.tsx initSessionFromRecord) — the persisted record (SessionRecord.messages) is + * the raw OpenAI-shape history, which carries everything needed (tool name, arguments, result) + * even though it was never saved as a pre-rendered display label. Handles both native mode + * (assistant `tool_calls` + matching `tool`-role messages, correlated by `tool_call_id`) and + * fallback mode (` ```tool_call``` ` blocks embedded in assistant text, ` ```tool_result``` ` + * blocks embedded in user text — see fallbackParser.ts and pushToolResultMessage). */ +export function buildReplayHistory(messages: ChatCompletionMessageParam[]): HistoryItem[] { + const items: HistoryItem[] = []; + // Native mode: `tool` messages only carry a tool_call_id, not the tool's name — remember the + // name from the assistant message that made the call so its later result can be labeled. + const pendingCallNames = new Map(); + + const push = (item: NewHistoryItem) => items.push({ id: nextId(), ...item } as HistoryItem); + + for (const m of messages) { + if (m.role === "user") { + const text = textContent(m.content); + if (text === null) continue; + const fallbackResult = TOOL_RESULT_BLOCK_RE.exec(text.trim()); + if (fallbackResult) { + try { + const { name, result } = JSON.parse(fallbackResult[1]!) as { name: string; result: unknown }; + push({ kind: "tool_result", summary: summarizeToolResult(name, result), isError: isErrorResult(result) }); + } catch { + // Malformed persisted block (shouldn't happen since we wrote it) — drop rather than + // show the raw JSON to the user. + } + continue; + } + if (text) push({ kind: "user", text }); + continue; + } + + if (m.role === "assistant") { + const toolCalls = m.tool_calls; + if (toolCalls?.length) { + const text = textContent(m.content); + if (text) push({ kind: "assistant", text }); + for (const call of toolCalls) { + if (call.type !== "function") continue; + let args: unknown = {}; + try { + args = JSON.parse(call.function.arguments); + } catch { + // Leave args as {} — formatCallLabel degrades gracefully for a missing field. + } + pendingCallNames.set(call.id, call.function.name); + push({ kind: "tool_call", label: formatCallLabel(call.function.name, args) }); + } + continue; + } + + const text = textContent(m.content); + if (!text) continue; + const parsed = parseFallbackToolCalls(text); + if (parsed.calls.length) { + const stripped = text.replace(TOOL_CALL_BLOCK_RE, "").trim(); + if (stripped) push({ kind: "assistant", text: stripped }); + for (const call of parsed.calls) { + push({ kind: "tool_call", label: formatCallLabel(call.name, call.arguments) }); + } + } else { + push({ kind: "assistant", text }); + } + continue; + } + + if (m.role === "tool") { + const name = pendingCallNames.get(m.tool_call_id) ?? "unknown"; + const text = textContent(m.content); + let result: unknown = text; + if (text) { + try { + result = JSON.parse(text); + } catch { + result = text; + } + } + push({ kind: "tool_result", summary: summarizeToolResult(name, result), isError: isErrorResult(result) }); + } + } + + return items; +} diff --git a/src/plugins/loader.test.ts b/src/plugins/loader.test.ts index 33a67dc..a0907db 100644 --- a/src/plugins/loader.test.ts +++ b/src/plugins/loader.test.ts @@ -46,4 +46,21 @@ describe("loadPlugin", () => { expect(plugin.agents).toHaveLength(1); expect(plugin.agents[0]).toMatchObject({ name: "review", description: "Code reviewer" }); }); + + it("parses a command's allowed-tools frontmatter, translating Claude Code tool names", () => { + mkdirSync(path.join(tempDir, ".claude-plugin"), { recursive: true }); + writeFileSync(path.join(tempDir, ".claude-plugin", "plugin.json"), JSON.stringify({ name: "test-plugin" })); + mkdirSync(path.join(tempDir, "commands"), { recursive: true }); + writeFileSync( + path.join(tempDir, "commands", "readonly.md"), + "---\ndescription: Look but don't touch\nallowed-tools: Read, Grep\n---\nInvestigate $ARGUMENTS", + ); + writeFileSync(path.join(tempDir, "commands", "unrestricted.md"), "---\ndescription: No restriction\n---\nDo $ARGUMENTS"); + + const plugin = loadPlugin(tempDir); + const readonly = plugin.commands.find((c) => c.name === "readonly"); + const unrestricted = plugin.commands.find((c) => c.name === "unrestricted"); + expect(readonly?.allowedTools).toEqual(["read_file", "grep"]); + expect(unrestricted?.allowedTools).toBeUndefined(); + }); }); diff --git a/src/plugins/loader.ts b/src/plugins/loader.ts index 2d97fda..382fc88 100644 --- a/src/plugins/loader.ts +++ b/src/plugins/loader.ts @@ -20,6 +20,17 @@ function listMarkdownFiles(dir: string): string[] { return readdirSync(dir).filter((f) => f.endsWith(".md")); } +/** Parses a comma-separated tool-name list from frontmatter (agents' `tools`, commands' + * `allowed-tools`) and translates each from Claude Code's built-in names to locode's. Returns + * undefined for an absent/empty field so callers can treat that as "no restriction". */ +function parseToolList(field: string | undefined): string[] | undefined { + const tools = field + ?.split(",") + .map((t) => resolveToolName(t.trim())) + .filter(Boolean); + return tools?.length ? tools : undefined; +} + function loadCommands(pluginRoot: string, pluginName: string): PluginCommand[] { const dir = path.join(pluginRoot, "commands"); return listMarkdownFiles(dir).map((entry) => { @@ -30,6 +41,7 @@ function loadCommands(pluginRoot: string, pluginName: string): PluginCommand[] { description: frontmatter.description, argumentHint: frontmatter["argument-hint"], template: body, + allowedTools: parseToolList(frontmatter["allowed-tools"]), }; }); } @@ -39,15 +51,11 @@ function loadAgents(pluginRoot: string, pluginName: string): PluginAgentDef[] { return listMarkdownFiles(dir).map((entry) => { const { frontmatter, body } = parseFrontmatter(readFileSync(path.join(dir, entry), "utf-8")); const name = frontmatter.name || entry.replace(/\.md$/, ""); - const tools = frontmatter.tools - ?.split(",") - .map((t) => resolveToolName(t.trim())) - .filter(Boolean); return { pluginName, name, description: frontmatter.description || name, - tools: tools?.length ? tools : undefined, + tools: parseToolList(frontmatter.tools), systemPrompt: body, }; }); diff --git a/src/plugins/types.ts b/src/plugins/types.ts index de735c5..b870ce9 100644 --- a/src/plugins/types.ts +++ b/src/plugins/types.ts @@ -16,6 +16,10 @@ export interface PluginCommand { /** The markdown body — expanded via $ARGUMENTS/$1../$9 (see expandTemplate.ts) and submitted * as the turn's input when the command is invoked. */ template: string; + /** Tool names this command's turn is restricted to (already translated from Claude Code's + * built-in tool names — see toolNameMap.ts), from frontmatter `allowed-tools`; undefined means + * no restriction (the command runs with the session's full toolset). */ + allowedTools?: string[]; } export interface PluginAgentDef { diff --git a/src/tools/bash.test.ts b/src/tools/bash.test.ts index 645640f..6ccd54b 100644 --- a/src/tools/bash.test.ts +++ b/src/tools/bash.test.ts @@ -1,4 +1,5 @@ import { beforeEach, describe, expect, it, vi } from "vitest"; +import { execa } from "execa"; import type { ResultPromise } from "execa"; import { bashTool } from "./bash.js"; import { killProcessTree } from "../utils/processTree.js"; @@ -80,4 +81,36 @@ describe("bash tool — sub-agent abort", () => { expect(killProcessTree).not.toHaveBeenCalled(); }); +}); + +describe("bash tool — safety guards", () => { + beforeEach(() => { + vi.mocked(execa).mockClear(); + }); + + it("refuses a command that wipes the filesystem root before ever spawning it", async () => { + const ctx: ToolContext = { cwd: process.cwd() }; + await expect(bashTool.handler({ command: "rm -rf /" }, ctx)).rejects.toThrow(/Refusing to run/); + expect(execa).not.toHaveBeenCalled(); + }); + + it("does not flag an ordinary rm -rf on a project subdirectory", async () => { + // preview (not handler) so this doesn't spawn the fake child, which only ever resolves when + // killed/backgrounded — nothing here would do either, so awaiting handler() would hang. + const ctx: ToolContext = { cwd: process.cwd() }; + const preview = await bashTool.preview!({ command: "rm -rf node_modules" }, ctx); + expect(preview).not.toMatch(/^Blocked:/); + }); + + it("refuses a cwd override that escapes the working directory", async () => { + const ctx: ToolContext = { cwd: process.cwd() }; + await expect(bashTool.handler({ command: "ls", cwd: "../../" }, ctx)).rejects.toThrow(/Refusing to write outside/); + expect(execa).not.toHaveBeenCalled(); + }); + + it("preview surfaces the block reason instead of running the command", async () => { + const ctx: ToolContext = { cwd: process.cwd() }; + const preview = await bashTool.preview!({ command: "mkfs.ext4 /dev/sda1" }, ctx); + expect(preview).toMatch(/^Blocked:/); + }); }); \ No newline at end of file diff --git a/src/tools/bash.ts b/src/tools/bash.ts index 1807e6d..8f8af0b 100644 --- a/src/tools/bash.ts +++ b/src/tools/bash.ts @@ -1,7 +1,8 @@ -import path from "node:path"; import { execa } from "execa"; import { z } from "zod"; import { registerBackgroundJob } from "./backgroundJobs.js"; +import { riskyBashCommandReason } from "./bashGuard.js"; +import { resolveWithinCwd } from "./pathGuard.js"; import { killProcessTree } from "../utils/processTree.js"; import { truncate } from "../utils/truncate.js"; import { resolveShell } from "../utils/shell.js"; @@ -21,12 +22,27 @@ function delay(ms: number): Promise<"pending"> { export const bashTool: ToolDef> = { name: "bash", - description: "Run a shell command and return its stdout, stderr, and exit code.", + description: "Run a shell command and return its stdout, stderr, and exit code. Use for building, running tests, git operations, or inspecting the environment. Output is capped (head+tail preserved); long-running commands can be backgrounded with Ctrl+B and checked with bash_output.", schema, mutating: true, - preview: async ({ command, cwd }) => `Run shell command: ${command}${cwd ? ` (cwd: ${cwd})` : ""}`, + preview: async ({ command, cwd }, ctx) => { + const riskyReason = riskyBashCommandReason(command); + if (riskyReason) return `Blocked: this command ${riskyReason}.`; + if (cwd) { + try { + resolveWithinCwd(ctx.cwd, cwd); + } catch (err) { + return (err as Error).message; + } + } + return `Run shell command: ${command}${cwd ? ` (cwd: ${cwd})` : ""}`; + }, handler: async ({ command, cwd, timeout_ms }, ctx) => { - const workDir = cwd ? path.resolve(ctx.cwd, cwd) : ctx.cwd; + const riskyReason = riskyBashCommandReason(command); + if (riskyReason) { + throw new Error(`Refusing to run: this command ${riskyReason}.`); + } + const workDir = cwd ? resolveWithinCwd(ctx.cwd, cwd) : ctx.cwd; // Timeout is enforced by our own timer rather than execa's built-in `timeout` option, so that // backgrounding via Ctrl+B can cancel it below — execa's own timeout kills the process on a // fixed schedule regardless of what happens to it afterward, which would silently kill a diff --git a/src/tools/bashGuard.test.ts b/src/tools/bashGuard.test.ts new file mode 100644 index 0000000..7c123ea --- /dev/null +++ b/src/tools/bashGuard.test.ts @@ -0,0 +1,45 @@ +import { describe, expect, it } from "vitest"; +import { riskyBashCommandReason } from "./bashGuard.js"; + +describe("riskyBashCommandReason", () => { + describe("blocks", () => { + const cases: [name: string, command: string][] = [ + ["rm -rf /", "rm -rf /"], + ["rm -fr / (flag order swapped)", "rm -fr /"], + ["rm -rf / with a trailing slash-star", "rm -rf /*"], + ["rm -Rf ~ (home dir)", "rm -Rf ~"], + ["rm --recursive --force /", "rm --recursive --force /"], + ["sudo rm -rf /", "sudo rm -rf /"], + ["classic fork bomb", ":(){ :|:& };:"], + ["fork bomb with extra whitespace", ": ( ) { : | : & } ; :"], + ["mkfs.ext4", "mkfs.ext4 /dev/sda1"], + ["dd to a raw device", "dd if=/dev/zero of=/dev/sda bs=1M"], + ["redirect onto a raw device", "echo oops > /dev/sda"], + ["Windows format", "format C:"], + ["Windows rd /s /q on a drive root", "rd /s /q C:\\"], + ["PowerShell Remove-Item -Recurse -Force on a drive", "Remove-Item -Recurse -Force C:\\"], + ]; + for (const [name, command] of cases) { + it(name, () => { + expect(riskyBashCommandReason(command)).not.toBeNull(); + }); + } + }); + + describe("does not block", () => { + const cases: [name: string, command: string][] = [ + ["rm -rf on a project subdirectory", "rm -rf node_modules"], + ["rm -rf on a relative build dir", "rm -rf ./dist"], + ["rm without force/recursive on root-looking arg", "rm /tmp/foo.txt"], + ["dd between two regular files", "dd if=file.img of=out.img"], + ["a command that merely contains the word format", "echo 'please format your PR title'"], + ["a normal git command", "git status"], + ["listing a directory named format", "ls format"], + ]; + for (const [name, command] of cases) { + it(name, () => { + expect(riskyBashCommandReason(command)).toBeNull(); + }); + } + }); +}); diff --git a/src/tools/bashGuard.ts b/src/tools/bashGuard.ts new file mode 100644 index 0000000..345ffae --- /dev/null +++ b/src/tools/bashGuard.ts @@ -0,0 +1,87 @@ +/** Blocks a small set of unambiguously catastrophic shell commands — wiping the whole filesystem + * or a whole drive, formatting a device, a fork bomb — before they ever reach the confirmation + * prompt (or, under auto-accept, before they'd run with no prompt at all). This is not a general + * command sandbox: it doesn't stop a model from `rm -rf`-ing some *other* directory it shouldn't, + * running a slow fork loop that isn't the canonical bomb syntax, or anything merely inadvisable — + * only the handful of patterns whose only realistic purpose is destroying the whole machine, where + * a false negative is far more likely than a false positive. Deliberately narrow so it doesn't + * reject legitimate commands like `rm -rf node_modules` or `dd if=file.img of=out.img`. */ + +interface RiskyPattern { + test: (command: string) => boolean; + reason: string; +} + +/** Splits on whitespace for a crude token scan — good enough for a blocklist (not a security + * boundary; execa still runs the raw string through a real shell either way) and avoids a brittle + * do-everything regex that has to encode flag ordering itself. */ +function tokenize(command: string): string[] { + return command.trim().split(/\s+/); +} + +const ROOT_TARGETS = new Set(["/", "/*", "~", "~/", "~/*", "$home", "${home}"]); + +/** `rm -rf /`, `rm -fr ~`, `sudo rm -Rf --no-preserve-root /`, etc. — recursive+forced deletion + * whose target is the filesystem root or the whole home directory, in any flag order/spelling. */ +function isRmWipingRootOrHome(command: string): boolean { + const tokens = tokenize(command).map((t) => t.toLowerCase()); + const rmIdx = tokens.findIndex((t) => t === "rm" || t.endsWith("/rm")); + if (rmIdx === -1) return false; + const rest = tokens.slice(rmIdx + 1); + const isFlag = (t: string) => t.startsWith("-"); + const hasForce = rest.some((t) => (isFlag(t) && !t.startsWith("--") && t.includes("f")) || t === "--force"); + const hasRecursive = rest.some((t) => (isFlag(t) && !t.startsWith("--") && (t.includes("r") || t.includes("R"))) || t === "--recursive"); + const targets = rest.filter((t) => !isFlag(t)); + return hasForce && hasRecursive && targets.some((t) => ROOT_TARGETS.has(t)); +} + +const RISKY_PATTERNS: RiskyPattern[] = [ + { + test: isRmWipingRootOrHome, + reason: "recursively force-deletes the filesystem root or home directory", + }, + { + // Classic bash fork bomb: ":(){ :|:& };:" (whitespace-tolerant). + test: (cmd) => /:\s*\(\s*\)\s*\{\s*:\s*\|\s*:\s*&?\s*;?\s*\}\s*;\s*:/.test(cmd), + reason: "is a fork bomb (unbounded process spawning)", + }, + { + test: (cmd) => /\bmkfs(\.\w+)?\b/i.test(cmd), + reason: "formats a filesystem (mkfs)", + }, + { + test: (cmd) => /\bdd\b[^\n]*\bof=\/dev\/(sd|hd|nvme|disk|xvd|rdisk)\w*/i.test(cmd), + reason: "writes raw data directly to a block device (dd of=/dev/...)", + }, + { + test: (cmd) => />\s*\/dev\/(sd|hd|nvme|disk|xvd|rdisk)\w*\b/i.test(cmd), + reason: "redirects output directly onto a block device", + }, + { + // `format C:`, `format /Y D:` — Windows drive format. + test: (cmd) => /\bformat\b[^\n]*\b[a-zA-Z]:/i.test(cmd), + reason: "formats a Windows drive (format)", + }, + { + // `rd /s /q C:\`, `rmdir /s /q D:\` — recursive quiet delete of a bare drive root. + test: (cmd) => /\b(rd|rmdir)\b[^\n]*\/s\b[^\n]*\b[a-zA-Z]:\\?\s*(\/q\b[^\n]*)?$/im.test(cmd), + reason: "recursively deletes an entire Windows drive", + }, + { + // PowerShell `Remove-Item -Recurse -Force C:\` (or -Path C:\, or $env:SystemDrive), flag order-tolerant. + test: (cmd) => + /remove-item\b/i.test(cmd) && + /-recurse\b/i.test(cmd) && + /-force\b/i.test(cmd) && + (/\b[a-zA-Z]:\\?\s*($|['")\s;])/.test(cmd) || /\$env:systemdrive\b/i.test(cmd)), + reason: "recursively force-deletes an entire Windows drive (Remove-Item)", + }, +]; + +/** Returns a human-readable reason if `command` matches a known catastrophic pattern, else null. */ +export function riskyBashCommandReason(command: string): string | null { + for (const pattern of RISKY_PATTERNS) { + if (pattern.test(command)) return pattern.reason; + } + return null; +} diff --git a/src/tools/editFile.test.ts b/src/tools/editFile.test.ts new file mode 100644 index 0000000..fed9723 --- /dev/null +++ b/src/tools/editFile.test.ts @@ -0,0 +1,79 @@ +import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { editFileTool } from "./editFile.js"; +import type { ToolContext } from "./types.js"; + +describe("editFile tool — path containment", () => { + let cwd: string; + let ctx: ToolContext; + + beforeEach(() => { + cwd = mkdtempSync(path.join(os.tmpdir(), "locode-editfile-")); + ctx = { cwd }; + }); + + afterEach(() => { + rmSync(cwd, { recursive: true, force: true }); + }); + + it("edits a file inside the working directory", async () => { + writeFileSync(path.join(cwd, "note.txt"), "hello world"); + const result = (await editFileTool.handler({ path: "note.txt", old_string: "world", new_string: "there" }, ctx)) as { replacements: number }; + expect(result.replacements).toBe(1); + }); + + it("refuses to edit a file outside the working directory via ../ traversal", async () => { + await expect( + editFileTool.handler({ path: "../escape.txt", old_string: "a", new_string: "b" }, ctx), + ).rejects.toThrow(/outside the working directory/); + }); + + it("preview reports the block instead of reading the target file", async () => { + const preview = await editFileTool.preview!({ path: "../escape.txt", old_string: "a", new_string: "b" }, ctx); + expect(preview).toMatch(/outside the working directory/); + }); + + it("handler suggests the closest match when old_string is not found", async () => { + writeFileSync( + path.join(cwd, "code.ts"), + "function greet(name: string): string {\n return `Hello, ${name}!`;\n}\n", + ); + // Close but wrong: single quotes instead of backticks, "Hi" instead of "Hello". + await expect( + editFileTool.handler( + { path: "code.ts", old_string: "return 'Hi, ${name}!';", new_string: "return `Hi, ${name}!`;" }, + ctx, + ), + ).rejects.toThrow(/closest match/); + }); + + it("preview warns and shows the closest match when old_string is not found", async () => { + writeFileSync(path.join(cwd, "note.txt"), "the quick brown fox jumps over the lazy dog"); + const preview = await editFileTool.preview!( + { path: "note.txt", old_string: "the quick red fox jumps over the lazy cat", new_string: "x" }, + ctx, + ); + expect(preview).toMatch(/not found/); + expect(preview).toMatch(/closest match/); + expect(preview).toContain("quick brown fox"); + }); + + it("does not suggest a match when nothing is remotely similar", async () => { + writeFileSync(path.join(cwd, "note.txt"), "aaaaaaaaaaaaaaaaaaaaaaaa"); + await expect( + editFileTool.handler( + { path: "note.txt", old_string: "completely different text xyz", new_string: "b" }, + ctx, + ), + ).rejects.toThrow(/not found/); + // No "closest match" suffix when similarity is below the threshold. + await expect( + editFileTool.handler( + { path: "note.txt", old_string: "completely different text xyz", new_string: "b" }, + ctx, + ), + ).rejects.not.toThrow(/closest match/); + }); +}); diff --git a/src/tools/editFile.ts b/src/tools/editFile.ts index 3522366..9443f55 100644 --- a/src/tools/editFile.ts +++ b/src/tools/editFile.ts @@ -1,8 +1,8 @@ import { createPatch } from "diff"; import { randomBytes } from "node:crypto"; import { readFile as fsReadFile, rename as fsRename, unlink as fsUnlink, writeFile as fsWriteFile } from "node:fs/promises"; -import path from "node:path"; import { z } from "zod"; +import { resolveWithinCwd } from "./pathGuard.js"; import type { ToolDef } from "./types.js"; const schema = z.object({ @@ -24,14 +24,130 @@ function applyEdit(original: string, oldString: string, newString: string, repla return replaceAll ? original.split(oldString).join(newString) : original.replace(oldString, () => newString); } +/** + * Normalise a candidate snippet for fuzzy comparison: collapse runs of whitespace to single spaces + * and trim. This makes the similarity score tolerant to indentation/line-ending differences, which + * are the most common reasons a local model's old_string almost-matches but not quite. + */ +function normaliseForCompare(s: string): string { + return s.replace(/\s+/g, " ").trim(); +} + +/** + * Compute a Levenshtein distance limited to `maxDist` — early-exits once the distance exceeds it, + * making it O(n*m) worst case but far cheaper in practice when we only care about "close enough". + */ +function boundedLevenshtein(a: string, b: string, maxDist: number): number { + const al = a.length; + const bl = b.length; + if (Math.abs(al - bl) > maxDist) return maxDist + 1; + if (al === 0) return bl; + if (bl === 0) return al; + let prev: number[] = new Array(bl + 1); + let curr: number[] = new Array(bl + 1); + for (let j = 0; j <= bl; j++) prev[j] = j; + for (let i = 1; i <= al; i++) { + curr[0] = i; + let rowMin = i; + const ai = a.charCodeAt(i - 1); + for (let j = 1; j <= bl; j++) { + const cost = ai === b.charCodeAt(j - 1) ? 0 : 1; + const del = (prev[j] ?? 0) + 1; + const ins = (curr[j - 1] ?? 0) + 1; + const sub = (prev[j - 1] ?? 0) + cost; + curr[j] = Math.min(del, ins, sub); + const cell = curr[j] ?? 0; + if (cell < rowMin) rowMin = cell; + } + // If every cell in this row already exceeds maxDist, the final answer can only be worse. + if (rowMin > maxDist) return maxDist + 1; + [prev, curr] = [curr, prev]; + } + return prev[bl] ?? maxDist + 1; +} + +interface SimilarMatch { + /** 0..1 similarity ratio (1 = identical, 0 = unrelated). */ + score: number; + /** The exact text from the file at the best matching window. */ + snippet: string; + /** 1-indexed line number where the snippet starts. */ + line: number; +} + +/** + * Find the region of `content` most similar to `needle`. Slides a window of the needle's length + * (±50%) across the file in word steps, scoring normalised text with bounded Levenshtein. Returns + * the best candidate when its similarity is at least 0.5 — clearly worth suggesting to the model. + * Returns null when nothing is close enough, in which case the caller falls back to the plain + * "not found" message. + */ +function findSimilarMatch(content: string, needle: string): SimilarMatch | null { + const needleNorm = normaliseForCompare(needle); + if (needleNorm.length < 3) return null; + + const words = needleNorm.split(" "); + const minLen = Math.floor(needleNorm.length * 0.5); + const maxLen = Math.ceil(needleNorm.length * 1.5); + + let best: SimilarMatch | null = null; + let bestDist = Infinity; + + // Walk the file by character, treating every position as a potential window start is O(n*len) + // and too slow for big files. Instead, step at every Nth character (≈ word boundaries) to keep + // it cheap while still landing near real matches. + const step = Math.max(1, Math.floor(needleNorm.length / 8)); + const contentLen = content.length; + + for (let start = 0; start < contentLen; start += step) { + for (let len = minLen; len <= maxLen; len += step) { + const end = Math.min(start + len, contentLen); + const candidate = content.slice(start, end); + const candNorm = normaliseForCompare(candidate); + if (candNorm.length < minLen) continue; + + // Only spend Levenshtein effort if the lengths are plausibly close. + const maxDist = Math.floor(needleNorm.length * 0.5); + const dist = boundedLevenshtein(needleNorm, candNorm, maxDist); + if (dist >= bestDist) continue; + + bestDist = dist; + const score = 1 - dist / Math.max(needleNorm.length, candNorm.length); + // 1-indexed line: count newlines before `start`. + let line = 1; + for (let k = 0; k < start; k++) if (content.charCodeAt(k) === 10) line++; + best = { score, snippet: candidate.trim(), line }; + } + } + + if (best && best.score >= 0.5) return best; + return null; +} + +/** Build a "did you mean" suffix for error/preview messages. Returns "" if nothing useful. */ +function similarHint(original: string, oldString: string): string { + const m = findSimilarMatch(original, oldString); + if (!m) return ""; + // Truncate long snippets so the message stays readable. + const snippet = + m.snippet.length > 300 ? `${m.snippet.slice(0, 300)}…` : m.snippet; + return `\n\nThe closest match in the file (line ${m.line}, ~${Math.round(m.score * 100)}% similar):\n"""\n${snippet}\n"""\nUse this exact text (or a unique subset of it) as old_string.`; +} + export const editFileTool: ToolDef> = { name: "edit_file", description: - "Replace exact text in a file. old_string must match exactly. Unless replace_all is set, it must be unique — include enough context.", + "Replace exact text in a file. old_string must match exactly. Unless replace_all is set, it must be unique — include enough context. " + + "Use for small, targeted changes to an existing file. On a mismatch, the closest similar text is suggested to help retry.", schema, mutating: true, preview: async ({ path: filePath, old_string, new_string, replace_all }, ctx) => { - const resolved = path.resolve(ctx.cwd, filePath); + let resolved: string; + try { + resolved = resolveWithinCwd(ctx.cwd, filePath); + } catch (err) { + return (err as Error).message; + } let original: string; try { original = await fsReadFile(resolved, "utf-8"); @@ -40,7 +156,7 @@ export const editFileTool: ToolDef> = { } const occurrences = countOccurrences(original, old_string); if (occurrences === 0) { - return `Warning: old_string not found in ${resolved} — this edit will fail.`; + return `Warning: old_string not found in ${resolved} — this edit will fail.${similarHint(original, old_string)}`; } if (occurrences > 1 && !replace_all) { return `Warning: old_string appears ${occurrences} times in ${resolved} — this edit will fail unless replace_all is set.`; @@ -49,12 +165,12 @@ export const editFileTool: ToolDef> = { return createPatch(resolved, original, updated, "", ""); }, handler: async ({ path: filePath, old_string, new_string, replace_all }, ctx) => { - const resolved = path.resolve(ctx.cwd, filePath); + const resolved = resolveWithinCwd(ctx.cwd, filePath); const original = await fsReadFile(resolved, "utf-8"); const occurrences = countOccurrences(original, old_string); if (occurrences === 0) { throw new Error( - `old_string not found in ${filePath}. Make sure it matches the file exactly, including whitespace.`, + `old_string not found in ${filePath}. Make sure it matches the file exactly, including whitespace.${similarHint(original, old_string)}`, ); } if (occurrences > 1 && !replace_all) { @@ -77,4 +193,4 @@ export const editFileTool: ToolDef> = { } return { path: resolved, replacements: replace_all ? occurrences : 1 }; }, -}; +}; \ No newline at end of file diff --git a/src/tools/grep.ts b/src/tools/grep.ts index d260e2c..c45fa7e 100644 --- a/src/tools/grep.ts +++ b/src/tools/grep.ts @@ -14,7 +14,7 @@ const schema = z.object({ export const grepTool: ToolDef> = { name: "grep", - description: "Search file contents for a regular expression pattern using ripgrep.", + description: "Search file contents for a regular expression pattern using ripgrep. Use to find where a symbol/function/word is used across the codebase, or to locate files containing specific text. Faster than reading files one by one.", schema, mutating: false, handler: async ({ pattern, path: searchPath, glob, case_insensitive, max_results }, ctx) => { diff --git a/src/tools/listFiles.ts b/src/tools/listFiles.ts index 8895fc4..4a23434 100644 --- a/src/tools/listFiles.ts +++ b/src/tools/listFiles.ts @@ -17,7 +17,7 @@ const MAX_MATCHES = 500; export const listFilesTool: ToolDef> = { name: "list_files", - description: "List files matching a glob pattern.", + description: "List files matching a glob pattern (e.g. `src/**/*.ts`). Use to explore the project structure or find files by name/extension before reading them.", schema, mutating: false, handler: async ({ pattern, cwd }, ctx) => { diff --git a/src/tools/pathGuard.test.ts b/src/tools/pathGuard.test.ts new file mode 100644 index 0000000..ce8a1f6 --- /dev/null +++ b/src/tools/pathGuard.test.ts @@ -0,0 +1,36 @@ +import path from "node:path"; +import { describe, expect, it } from "vitest"; +import { PathOutsideCwdError, resolveWithinCwd } from "./pathGuard.js"; + +describe("resolveWithinCwd", () => { + const cwd = path.resolve("/project"); + + it("resolves a plain relative path inside cwd", () => { + expect(resolveWithinCwd(cwd, "src/index.ts")).toBe(path.join(cwd, "src", "index.ts")); + }); + + it("resolves an absolute path that happens to already be inside cwd", () => { + const inside = path.join(cwd, "foo.txt"); + expect(resolveWithinCwd(cwd, inside)).toBe(inside); + }); + + it("resolves cwd itself", () => { + expect(resolveWithinCwd(cwd, ".")).toBe(cwd); + }); + + it("rejects a ../ escape", () => { + expect(() => resolveWithinCwd(cwd, "../outside.txt")).toThrow(PathOutsideCwdError); + }); + + it("rejects a deeper ../../ escape", () => { + expect(() => resolveWithinCwd(cwd, "sub/../../outside.txt")).toThrow(PathOutsideCwdError); + }); + + it("rejects an absolute path outside cwd", () => { + expect(() => resolveWithinCwd(cwd, path.resolve("/etc/passwd"))).toThrow(PathOutsideCwdError); + }); + + it("rejects the filesystem root", () => { + expect(() => resolveWithinCwd(cwd, path.parse(cwd).root)).toThrow(PathOutsideCwdError); + }); +}); diff --git a/src/tools/pathGuard.ts b/src/tools/pathGuard.ts new file mode 100644 index 0000000..9358a2f --- /dev/null +++ b/src/tools/pathGuard.ts @@ -0,0 +1,27 @@ +import path from "node:path"; + +/** Thrown by resolveWithinCwd — kept as its own class only so callers can recognize it (via + * instanceof) if they ever need to react differently than a plain thrown Error. */ +export class PathOutsideCwdError extends Error {} + +/** Resolves `targetPath` against `cwd` and hard-blocks the result if it would land outside the + * project root (the working directory locode was launched in) — an absolute path elsewhere on + * disk, a `../` escape, or (on Windows) a path on a different drive all reject. This applies + * unconditionally, regardless of permission mode: even `auto-accept` skips a tool's `preview` + * entirely (see gateAndRun in agent/loop.ts), so this check has to live in each tool's `handler` + * — which always runs — to actually hold as a floor rather than just a confirmation-dialog hint. + * It's deliberately not configurable; a model tricked (or simply mistaken) into targeting a path + * outside the project shouldn't be one auto-approved call away from touching it. */ +export function resolveWithinCwd(cwd: string, targetPath: string): string { + const resolved = path.resolve(cwd, targetPath); + const rel = path.relative(cwd, resolved); + // rel === "" is targetPath resolving to cwd itself — fine. Anything starting with ".." walked + // upward out of cwd; an absolute rel (Windows: a different drive, e.g. "D:\foo") never went + // through cwd's tree in the first place. Either way, it's outside. + if (rel !== "" && (rel.startsWith(`..${path.sep}`) || rel === ".." || path.isAbsolute(rel))) { + throw new PathOutsideCwdError( + `Refusing to write outside the working directory: "${targetPath}" resolves to ${resolved}, which is not inside ${cwd}.`, + ); + } + return resolved; +} diff --git a/src/tools/readFile.ts b/src/tools/readFile.ts index 15ee7f6..a53f4b3 100644 --- a/src/tools/readFile.ts +++ b/src/tools/readFile.ts @@ -26,7 +26,8 @@ export const readFileTool: ToolDef> = { name: "read_file", description: "Read a local file. Text files return 1-indexed lines; large files are paginated (use nextOffset for next page). " + - "Image files (png, jpg, jpeg, gif, webp, bmp) are returned as image content (requires vision-capable model).", + "Image files (png, jpg, jpeg, gif, webp, bmp) are returned as image content (requires vision-capable model). " + + "Use to inspect file contents before editing, or to understand existing code. Prefer this over bash cat for files.", schema, mutating: false, handler: async ({ path: filePath, offset, limit }, ctx) => { diff --git a/src/tools/webFetch.ts b/src/tools/webFetch.ts index 4957964..4d5e831 100644 --- a/src/tools/webFetch.ts +++ b/src/tools/webFetch.ts @@ -19,7 +19,8 @@ function isBinaryContentType(contentType: string): boolean { export const webFetchTool: ToolDef> = { name: "web_fetch", description: - "Fetch a URL and return readable text (HTML tags/scripts/styles stripped). Use for specific pages found via web_search.", + "Fetch a URL and return readable text (HTML tags/scripts/styles stripped). Use for specific pages found via web_search — " + + "e.g. to read a doc page or blog post in full when the search snippet was not enough.", schema, mutating: false, handler: async ({ url }) => { diff --git a/src/tools/webSearch.ts b/src/tools/webSearch.ts index ad6a90a..3fc6dc1 100644 --- a/src/tools/webSearch.ts +++ b/src/tools/webSearch.ts @@ -45,7 +45,8 @@ function parseResults(html: string, limit: number): SearchResult[] { export const webSearchTool: ToolDef> = { name: "web_search", description: - "Search the web via DuckDuckGo. Returns title, url, snippet. Use for info not in the local codebase.", + "Search the web via DuckDuckGo. Returns title, url, snippet. Use for info not in the local codebase — " + + "e.g. an unfamiliar API, library docs, or an error message. Follow up with web_fetch on a specific result for full page text.", schema, mutating: false, handler: async ({ query, max_results }) => { diff --git a/src/tools/writeFile.test.ts b/src/tools/writeFile.test.ts new file mode 100644 index 0000000..5672c5f --- /dev/null +++ b/src/tools/writeFile.test.ts @@ -0,0 +1,45 @@ +import { mkdirSync, mkdtempSync, readFileSync, rmSync } from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import { writeFileTool } from "./writeFile.js"; +import type { ToolContext } from "./types.js"; + +describe("writeFile tool — path containment", () => { + let cwd: string; + let ctx: ToolContext; + + beforeEach(() => { + cwd = mkdtempSync(path.join(os.tmpdir(), "locode-writefile-")); + ctx = { cwd }; + }); + + afterEach(() => { + rmSync(cwd, { recursive: true, force: true }); + }); + + it("writes a file inside the working directory", async () => { + const result = (await writeFileTool.handler({ path: "note.txt", content: "hi" }, ctx)) as { path: string }; + expect(readFileSync(result.path, "utf-8")).toBe("hi"); + }); + + it("refuses to write outside the working directory via ../ traversal", async () => { + await expect(writeFileTool.handler({ path: "../escape.txt", content: "oops" }, ctx)).rejects.toThrow(/outside the working directory/); + }); + + it("refuses to write to an absolute path outside the working directory", async () => { + const outside = path.join(os.tmpdir(), "locode-outside-target.txt"); + await expect(writeFileTool.handler({ path: outside, content: "oops" }, ctx)).rejects.toThrow(/outside the working directory/); + }); + + it("preview reports the block instead of showing a diff", async () => { + const preview = await writeFileTool.preview!({ path: "../escape.txt", content: "oops" }, ctx); + expect(preview).toMatch(/outside the working directory/); + }); + + it("still applies even when the escaping subdirectory already exists", async () => { + // Sanity check that the guard runs before mkdir/write, not after. + mkdirSync(path.join(cwd, "sub"), { recursive: true }); + await expect(writeFileTool.handler({ path: "sub/../../escape.txt", content: "oops" }, ctx)).rejects.toThrow(/outside the working directory/); + }); +}); diff --git a/src/tools/writeFile.ts b/src/tools/writeFile.ts index ca02bb4..1fb40b7 100644 --- a/src/tools/writeFile.ts +++ b/src/tools/writeFile.ts @@ -2,6 +2,7 @@ import { createPatch } from "diff"; import { mkdir, readFile as fsReadFile, writeFile as fsWriteFile } from "node:fs/promises"; import path from "node:path"; import { z } from "zod"; +import { resolveWithinCwd } from "./pathGuard.js"; import type { ToolDef } from "./types.js"; const schema = z.object({ @@ -19,11 +20,16 @@ async function readExisting(resolved: string): Promise { export const writeFileTool: ToolDef> = { name: "write_file", - description: "Create or overwrite a file with the given content.", + description: "Create or overwrite a file with the given content. Use for new files or full rewrites. For small changes to an existing file, prefer edit_file instead of rewriting the whole file.", schema, mutating: true, preview: async ({ path: filePath, content }, ctx) => { - const resolved = path.resolve(ctx.cwd, filePath); + let resolved: string; + try { + resolved = resolveWithinCwd(ctx.cwd, filePath); + } catch (err) { + return (err as Error).message; + } const existing = await readExisting(resolved); if (existing === null) { return `Create new file ${resolved} (${content.length} chars)`; @@ -31,7 +37,7 @@ export const writeFileTool: ToolDef> = { return createPatch(resolved, existing, content, "", ""); }, handler: async ({ path: filePath, content }, ctx) => { - const resolved = path.resolve(ctx.cwd, filePath); + const resolved = resolveWithinCwd(ctx.cwd, filePath); await mkdir(path.dirname(resolved), { recursive: true }); await fsWriteFile(resolved, content, "utf-8"); return { path: resolved, bytesWritten: Buffer.byteLength(content, "utf-8") }; diff --git a/src/ui/ink/App.tsx b/src/ui/ink/App.tsx index c8f0de0..0526072 100644 --- a/src/ui/ink/App.tsx +++ b/src/ui/ink/App.tsx @@ -1,5 +1,6 @@ -import { Box, Text, useApp, useBoxMetrics, useInput, useWindowSize, type DOMElement } from "ink"; -import { useCallback, useEffect, useRef, useState } from "react"; +import { Box, Text, useApp, useBoxMetrics, useInput, useStdin, useStdout, useWindowSize, type DOMElement } from "ink"; +import { useCallback, useEffect, useMemo, useRef, useState } from "react"; +import path from "node:path"; import { AgentError, compactSession, @@ -27,7 +28,7 @@ import { resolveContextWindow } from "../../backend/contextWindow.js"; import { resolveToolCallMode } from "../../backend/resolveMode.js"; import { resolveAutoCompactThreshold, resolveMaxIterations } from "../../config/config.js"; import { KNOWN_BACKENDS, type BackendName } from "../../config/defaults.js"; -import { getMcpStatuses } from "../../mcp/manager.js"; +import { getMcpStatuses, reconnectMcpServers } from "../../mcp/manager.js"; import type { PermissionDecision, PermissionMode } from "../../permissions/types.js"; import { defaultExportFilename, exportSession } from "../../persistence/exportSession.js"; import { loadMergedHooks } from "../../hooks/config.js"; @@ -46,13 +47,18 @@ import { type SessionRecord, type SessionSummary, } from "../../persistence/sessionStore.js"; +import { buildReplayHistory } from "../../persistence/replayHistory.js"; import { onBackgroundJobDone } from "../../tools/backgroundJobs.js"; import { TOOLS } from "../../tools/index.js"; import type { ToolDef } from "../../tools/types.js"; +import { buildToolSet, type ToolSet } from "../../tools/toolset.js"; import { ChatInput } from "./ChatInput.js"; +import { makeConfirmFn } from "./confirmFn.js"; import { ExportPrompt } from "./ExportPrompt.js"; +import { FilePanel, type FilePanelTab, type TouchedFile } from "./FilePanel.js"; import { HistoryItemView } from "./HistoryItemView.js"; import { ModelSelect } from "./ModelSelect.js"; +import { matchMouseSequence } from "./mouseInput.js"; import { PermissionPrompt } from "./PermissionPrompt.js"; import { SessionSelect } from "./SessionSelect.js"; import { StatusBar } from "./StatusBar.js"; @@ -63,6 +69,20 @@ import { nextId, type HistoryItem, type NewHistoryItem } from "./types.js"; // Cap for the input-history ring buffer used for ↑/↓ recall in the chat input. const MAX_HISTORY = 100; +// Fixed width (in columns) of the file panel (see FilePanel.tsx) when shown — a compromise between +// filenames actually fitting and leaving enough room for the chat column on an 80-col terminal. +const FILE_PANEL_WIDTH = 30; + +// Maps a tool name to how the file panel's Activity tab should label a successful call that +// touched a file — see the tool_result handling in submitTurn. Every one of these tools returns +// `{ path: , ... }` from its handler (see readFile.ts/writeFile.ts/ +// editFile.ts), which is what makes a single lookup here enough to build a TouchedFile entry. +const FILE_TOUCH_STATUS: Partial> = { + read_file: "read", + write_file: "written", + edit_file: "edited", +}; + // Shared between /perm's explicit-cycle notice and Shift+Tab's cyclePermMode so the two paths to // the same action can't drift out of sync with each other. const PERM_MODE_LABELS: Record = { @@ -107,6 +127,8 @@ export function App({ onSessionIdChange, }: AppProps) { const { exit } = useApp(); + const { stdin, isRawModeSupported, setRawMode } = useStdin(); + const { stdout } = useStdout(); const [staticItems, setStaticItems] = useState([]); const [phase, setPhase] = useState( resumeSessionId ? "connecting" : interactiveResume ? "starting" : initialModel ? "connecting" : "loading-models", @@ -120,13 +142,30 @@ export function App({ const [modelList, setModelList] = useState([]); const [sessionList, setSessionList] = useState([]); const [gitInfo, setGitInfo] = useState(null); + const [filePanelVisible, setFilePanelVisible] = useState(false); + const [filePanelTab, setFilePanelTab] = useState("files"); + // Whether the file panel currently owns keyboard input — see the Ctrl+F handler below and + // ChatInput's isActive prop, which this disables while true so an arrow/Enter/Escape keystroke + // doesn't simultaneously navigate the tree and edit/submit the chat input. + const [filePanelFocused, setFilePanelFocused] = useState(false); + // Keyed by relPath so repeated touches update the same entry (bumping count) instead of + // duplicating it — see the tool_result handling in submitTurn below. + const [touchedFiles, setTouchedFiles] = useState>(new Map()); + // Most-recently-touched first — the Activity tab's whole point is "what's happened lately". + const touchedFilesList = useMemo(() => Array.from(touchedFiles.values()).sort((a, b) => b.lastTouchedAt - a.lastTouchedAt), [touchedFiles]); const [history, setHistory] = useState([]); const baseURLRef = useRef(initialBaseURL); const sessionRef = useRef(null); + // The AbortController for whichever top-level turn is currently in flight — null between turns. + // Escape (see the global useInput handler below) aborts it, which cancels the in-flight backend + // request, kills a running bash child (see gateAndRun/bashTool's ctx.signal), and dismisses/rejects + // a pending permission prompt (via makeConfirmFn) — the same abort plumbing sub-agent timeouts + // already used, just wired up to a top-level turn for the first time. + const turnAbortRef = useRef(null); // Wraps whichever branch the bottom ternary renders (permission/export prompt, a picker, or the // normal StatusBar+ChatInput column) — measured (height only) so the history viewport above it - // knows exactly how much vertical space is left (see historyHeight). + // knows exactly how much vertical space is left (the history viewport above it flexes). const bottomSectionRef = useRef(null); // Measures the history content's own natural (unclipped) height — Yoga still computes a child's // intrinsic size even when its parent has a fixed height + overflowY:hidden, so this reports the @@ -149,23 +188,56 @@ export function App({ // re-registered on every phase change. const phaseRef = useRef(phase); phaseRef.current = phase; - const { height: bottomSectionHeight, hasMeasured: bottomSectionMeasured } = useBoxMetrics(bottomSectionRef); - const { rows: terminalRows } = useWindowSize(); - // Estimated fallback for the one frame before bottomSectionRef's first real measurement lands - // (StatusBar ~2 rows + a single-line bordered ChatInput ~3 rows) — avoids a brief overflow/flash - // of the history viewport claiming the whole terminal height on first mount. - const FALLBACK_BOTTOM_HEIGHT = 6; - const historyHeight = Math.max( - 1, - terminalRows - (bottomSectionMeasured ? bottomSectionHeight : FALLBACK_BOTTOM_HEIGHT), - ); + const { rows: terminalRows, columns: terminalColumns } = useWindowSize(); + // Columns actually left for the chat column once the file panel (see FilePanel.tsx) claims its + // fixed width on the right — ChatInput can't derive this from its own measured width (see the + // comment on its availableColumns prop), so it's computed once here and threaded down. + const chatColumns = terminalColumns - (filePanelVisible ? FILE_PANEL_WIDTH : 0); + // The split between the history viewport and the bottom section is resolved by Yoga in a single + // layout pass (history grows/shrinks, bottom is fixed) rather than computed by hand from a + // measured bottom height — that measurement always lagged one frame behind the bottom section's + // actual height (useBoxMetrics updates in an effect *after* render), so on any frame it grew + // (a permission/export modal mounting, or the @-mention suggestion box opening in ChatInput) + // history was sized too tall and the bottom section overpainted it, producing the overlap. + // We still read the history viewport's own measured height, but only for the top-vs-bottom + // alignment decision — a one-frame lag there only affects alignment, never the split, so it's + // harmless (unlike the split, which is what caused the overlap). + const historyViewportRef = useRef(null); + const { height: measuredHistoryHeight } = useBoxMetrics(historyViewportRef); const { height: historyContentHeight } = useBoxMetrics(historyContentRef); - // Short conversations (or right after connecting, with just the welcome banner) should stay - // top-aligned — flex-end would otherwise glue even a single item to the bottom of the viewport, - // leaving a large empty gap above it. Only once real content actually exceeds the available - // height does it make sense to bottom-align and clip the oldest (off-the-top) content, which is - // what makes the view auto-scroll to the latest message once a conversation grows past one screen. - const historyOverflows = historyContentHeight > historyHeight; + const viewportHeight = Math.max(1, measuredHistoryHeight); + // How far (in rows) the content has scrolled past the viewport — the max meaningful scrollTop. + // Short conversations (content fits entirely) have maxScroll 0, which also naturally keeps them + // top-aligned instead of glued to the bottom with a gap above. + const maxScroll = Math.max(0, historyContentHeight - viewportHeight); + // Whether the view should keep tracking the latest content as it arrives (the normal chat + // behavior) or hold still at a manually scrolled position. PageUp breaks the pin; PageDown + // re-establishes it once scrolled back down to the bottom; sending a new message always re-pins. + const [pinnedToBottom, setPinnedToBottom] = useState(true); + // Rows scrolled down from the content's top edge — only meaningful while not pinned; while + // pinned, the effective scrollTop is just maxScroll (always show the latest content). + const [scrollTop, setScrollTop] = useState(0); + const effectiveScrollTop = pinnedToBottom ? maxScroll : Math.min(scrollTop, maxScroll); + + // Refs mirroring the scroll-relevant values so the (once-registered) mouse-wheel listener can + // read the latest without re-binding on every state change (see mouseMode effect below). + const maxScrollRef = useRef(maxScroll); + maxScrollRef.current = maxScroll; + const pinnedToBottomRef = useRef(pinnedToBottom); + pinnedToBottomRef.current = pinnedToBottom; + const scrollTopRef = useRef(scrollTop); + scrollTopRef.current = scrollTop; + + // Scroll the history viewport by `delta` rows (negative = up, positive = down), using the same + // pin/unpin rules as the PageUp/PageDown handler above. Shared by keyboard paging and the + // mouse-wheel listener so the two paths can't drift. + const scrollBy = useCallback((delta: number) => { + const ms = maxScrollRef.current; + const current = pinnedToBottomRef.current ? ms : Math.min(scrollTopRef.current, ms); + const next = Math.max(0, Math.min(ms, current + delta)); + setScrollTop(next); + setPinnedToBottom(next >= ms); + }, []); const flushStreamingText = useCallback(() => { const accumulated = streamingAccumulatorRef.current; @@ -193,9 +265,23 @@ export function App({ // Fetch once up front; re-fetched after each turn (see submitTurn) since a tool call (git_commit, // bash) can switch branches or change the dirty state mid-session. + // Enable xterm mouse tracking (X11 mode 1000 + SGR-1006 pixel format) so wheel events arrive on + // stdin as escape sequences. ink's input parser passes each mouse sequence through to useInput + // as a single event (with the leading ESC stripped from `input`), where we detect it below. + // Only enabled during the chat phase and only when raw mode is supported; toggling it off on exit + // (and on phase change) restores the terminal so the shell's own mouse mode isn't disturbed. + // Also suspended while the export filename prompt is open: that prompt's text field is + // ink-text-input (third-party) with no guard against raw mouse sequences, so scrolling while it's + // open would otherwise type them straight into the filename — simplest fix is to stop the + // terminal from sending them at all rather than filtering inside a dependency we don't control. useEffect(() => { - refreshGitInfo(); - }, [refreshGitInfo]); + if (!isRawModeSupported || phase !== "input" || exportPrompt) return; + stdout.write("[?1000h[?1006h"); + setRawMode(true); + return () => { + stdout.write("[?1006l[?1000l"); + }; + }, [phase, isRawModeSupported, stdout, setRawMode, exportPrompt]); // A backgrounded bash job (see Ctrl+B below) can finish long after the turn that started it has // ended — this is how its completion still reaches the transcript. @@ -209,9 +295,60 @@ export function App({ // Mounted for the whole App lifetime (unlike ChatInput's own useInput, which only exists while // ChatInput is rendered) so both shortcuts work even mid-turn, when ChatInput is unmounted. useInput((input, key) => { + // Escape interrupts an in-flight turn (like Claude Code) — checked first and unconditionally on + // phase/modal state so it always wins, including while a permission prompt is open (aborting + // dismisses it too, via makeConfirmFn) or the file panel is focused. Gated on isThinking/ + // streamingText rather than the phase/permission/exportPrompt guard below, since those describe + // UI modal state, not whether a turn is actually running. + if (key.escape && (isThinking || streamingText !== null)) { + turnAbortRef.current?.abort(); + return; + } + // Shift+Tab: cycle permission mode. Handled globally (not just inside ChatInput) so it still + // works while a turn is in flight or a permission prompt is open — both unmount ChatInput (see + // the bottom-section ternary below), which is exactly the gap Escape hit above. Requires an + // active session (phase "input"); excluded only for exportPrompt, where it isn't a meaningful + // action while naming a file. + if (key.shift && key.tab && phaseRef.current === "input" && !exportPrompt) { + cyclePermMode(); + return; + } // Only react to global shortcuts during the actual chat phase; ignore them while a modal // (permission/export) or a non-input phase (model/session select, connecting) is open. if (phaseRef.current !== "input" || permission || exportPrompt) return; + // Mouse wheel events (xterm SGR-1006 format). button 64 = wheel up, 65 = wheel down. We only + // react to wheel events, not regular button clicks — but any recognized mouse sequence still + // returns early so it can't fall through to a shortcut check below. + const mouseEvent = matchMouseSequence(input); + if (mouseEvent) { + if (mouseEvent.button === 64) scrollBy(-3); + else if (mouseEvent.button === 65) scrollBy(3); + return; + } + if (key.pageUp || key.pageDown) { + const pageStep = Math.max(1, viewportHeight - 1); + scrollBy(key.pageUp ? -pageStep : pageStep); + return; + } + if (key.ctrl && input === "f") { + // Three-state cycle: hidden -> open+focused -> open+unfocused (via Escape, not here) -> hidden. + // Pressing Ctrl+F while open-but-unfocused (the Escape state) re-focuses it instead of hiding + // it outright, so "peek without hiding" (Escape) and "close" (Ctrl+F again) stay distinct. + if (!filePanelVisible) { + setFilePanelVisible(true); + setFilePanelFocused(true); + } else if (filePanelFocused) { + setFilePanelVisible(false); + setFilePanelFocused(false); + } else { + setFilePanelFocused(true); + } + return; + } + if (key.ctrl && input === "g") { + setFilePanelTab((t) => (t === "files" ? "activity" : "files")); + return; + } if (key.ctrl && input === "o") { const summary = lastCompactSummaryRef.current; push({ @@ -279,10 +416,7 @@ export function App({ // Default to "native" mode immediately — no blocking probe on startup. // The probe runs lazily on the first turn if needed. const mode: ToolCallMode = toolModeOverride ?? "native"; - const confirmFn = (opts: { toolName: string; args: unknown; preview?: string }) => - new Promise((resolve) => { - setPermission({ ...opts, resolve }); - }); + const confirmFn = makeConfirmFn(setPermission); const [extraTools, contextWindow, projectInstructions] = await Promise.all([ extraToolsPromise, resolveContextWindow(baseURLRef.current, model), @@ -331,10 +465,7 @@ export function App({ try { baseURLRef.current = record.baseURL; const client = makeClient({ baseURL: record.baseURL, model: record.model }); - const confirmFn = (opts: { toolName: string; args: unknown; preview?: string }) => - new Promise((resolve) => { - setPermission({ ...opts, resolve }); - }); + const confirmFn = makeConfirmFn(setPermission); const [extraTools, contextWindow, projectInstructions] = await Promise.all([ extraToolsPromise, resolveContextWindow(record.baseURL, record.model), @@ -365,17 +496,9 @@ export function App({ push({ kind: "notice", text: `Hook warning: ${warning}` }); } - // Replay the saved user/assistant turns so the transcript isn't blank — tool - // call/result lines aren't replayed since we don't persist their display labels. - const replayed: HistoryItem[] = []; - for (const m of record.messages) { - if (m.role === "user" && typeof m.content === "string" && !m.content.startsWith("```tool_result")) { - replayed.push({ id: nextId(), kind: "user", text: m.content }); - } else if (m.role === "assistant" && typeof m.content === "string" && m.content) { - replayed.push({ id: nextId(), kind: "assistant", text: m.content }); - } - } - setStaticItems((prev) => [...prev, ...replayed]); + // Replay the saved turns (including tool_call/tool_result lines) so the transcript + // looks the same as it did live, not just the user/assistant text. + setStaticItems((prev) => [...prev, ...buildReplayHistory(record.messages)]); setPhase("input"); } catch (err) { @@ -510,8 +633,10 @@ export function App({ } } - async function submitTurn(session: Session, input: string | ChatCompletionUserContent) { + async function submitTurn(session: Session, input: string | ChatCompletionUserContent, toolset?: ToolSet) { const rollbackLength = session.messages.length; + const ac = new AbortController(); + turnAbortRef.current = ac; setIsThinking(true); setStreamingText(null); streamingAccumulatorRef.current = ""; @@ -565,12 +690,22 @@ export function App({ } else if (event.type === "tool_result") { setRunningToolIsBash(false); setStaticItems((prev) => [...prev, { id: nextId(), kind: "tool_result", summary: event.summary, isError: event.isError } as HistoryItem]); + const touchStatus = event.name && !event.isError ? FILE_TOUCH_STATUS[event.name] : undefined; + const resultPath = touchStatus ? (event.result as { path?: unknown } | undefined)?.path : undefined; + if (touchStatus && typeof resultPath === "string") { + const relPath = path.relative(cwd, resultPath) || resultPath; + setTouchedFiles((prev) => { + const next = new Map(prev); + next.set(relPath, { relPath, status: touchStatus, count: (next.get(relPath)?.count ?? 0) + 1, lastTouchedAt: Date.now() }); + return next; + }); + } } else if (event.type === "hook_notice" || event.type === "notice") { setStaticItems((prev) => [...prev, { id: nextId(), kind: "notice", text: event.text, isError: event.isError } as HistoryItem]); } else if (event.type === "todos_update") { setStaticItems((prev) => [...prev, { id: nextId(), kind: "todos", todos: event.todos } as HistoryItem]); } - }); + }, toolset, ac.signal); // text_done already added the assistant message to staticItems // No fallback needed — the streaming loop always emits text_done void text; @@ -608,7 +743,14 @@ export function App({ // retry the edit on the next turn (which then fails with "old_string not found", etc.). const commitLength = session.mutationCommitLength; session.messages.length = commitLength ?? rollbackLength; - if (err instanceof MaxIterationsError) { + if (ac.signal.aborted) { + // The user pressed Escape (see the global useInput handler) — not a failure, so no + // "Request failed" framing. Whatever error actually surfaced (a raw fetch AbortError, or + // makeConfirmFn's "prompt cancelled" rejection if Escape landed while a permission prompt + // was open) is irrelevant here; `ac` is ours, so its own aborted flag is the one source of + // truth for "this was the user stopping it" regardless of which code path threw. + push({ kind: "notice", text: "Interrupted.", isError: false }); + } else if (err instanceof MaxIterationsError) { // Not a failure — the model just ran out of per-turn budget. No "Request failed" framing, // no isError styling, since nothing actually broke and the work done so far is intact. push({ kind: "notice", text: err.message, isError: false }); @@ -617,6 +759,7 @@ export function App({ push({ kind: "notice", text: `Request failed: ${reason}`, isError: true }); } } finally { + if (turnAbortRef.current === ac) turnAbortRef.current = null; setIsThinking(false); setStreamingText(null); streamingAccumulatorRef.current = ""; @@ -631,6 +774,7 @@ export function App({ setInputValue(""); const trimmed = raw.trim(); if (!trimmed) return; + setPinnedToBottom(true); const session = sessionRef.current; if (!session) return; @@ -743,7 +887,20 @@ export function App({ push({ kind: "sessions", sessions: listSessions() }); return; } - if (trimmed === "/mcp") { + if (trimmed === "/mcp" || trimmed === "/mcp reconnect") { + if (trimmed === "/mcp reconnect" && sessionRef.current) { + try { + const newMcpTools = await reconnectMcpServers(cwd); + // Rebuild the session's toolset: keep non-MCP tools (built-ins + plugin tools already in + // the toolset), drop the old MCP tools (namespaced `mcp__...`), and add the fresh ones. + const kept = sessionRef.current.toolset.tools.filter((t) => !t.name.startsWith("mcp__")); + sessionRef.current.toolset = buildToolSet([...kept, ...newMcpTools]); + push({ kind: "mcp", statuses: getMcpStatuses() }); + } catch (err) { + push({ kind: "mcp", statuses: getMcpStatuses() }); + } + return; + } push({ kind: "mcp", statuses: getMcpStatuses() }); return; } @@ -860,7 +1017,13 @@ export function App({ const pluginCommand = plugins.flatMap((p) => p.commands).find((c) => c.name === cmdName); if (pluginCommand) { const expanded = expandCommandTemplate(pluginCommand.template, argsText); - await submitTurn(session, extraContext ? `${expanded}\n\n${extraContext}` : expanded); + // A command's `allowed-tools` frontmatter restricts only this one turn — build a + // narrowed ToolSet from the session's full one rather than mutating session.toolset, + // the same pattern runSubAgentTurn (agent/loop.ts) uses for a plugin agent's `tools:`. + const restrictedToolset = pluginCommand.allowedTools + ? buildToolSet(session.toolset.tools.filter((t) => pluginCommand.allowedTools!.includes(t.name))) + : undefined; + await submitTurn(session, extraContext ? `${expanded}\n\n${extraContext}` : expanded, restrictedToolset); return; } @@ -936,94 +1099,110 @@ export function App({ } return ( - - {/* Fixed-height, bottom-pinned viewport: overflowY="hidden" clips whatever scrolls past the - * top, and justifyContent="flex-end" keeps the *latest* content flush against the bottom - * edge — together they auto-scroll to the newest message without any manual scroll-offset - * math, the same way a normal chat view does. This replaced (permanent one-shot - * scrollback printing) because Static's already-flushed rows never participate in Yoga - * layout again, which is fundamentally incompatible with letting old items visually scroll - * out of a *bounded* viewport as new ones arrive. Trade-off: every item re-renders on every - * frame now (Static rendered each item exactly once, ever) — fine at the sizes a single - * session reaches before auto-compaction, but worth knowing if a session gets huge. */} - - - {staticItems.map((item) => ( - - ))} - {streamingText !== null && ( - - )} - {isThinking && streamingText === null && !permission && !exportPrompt && ( - + + + {/* Scrollable, bottom-pinned viewport: overflowY="hidden" clips the content box, which is + * shifted up by a negative marginTop equal to effectiveScrollTop rows — at maxScroll (the + * pinned-to-bottom default) that puts the *latest* content flush against the bottom edge, + * auto-scrolling to it without any input; PageUp/PageDown (see the useInput handler above) + * unpin and walk scrollTop up/down a page at a time. This replaced (permanent + * one-shot scrollback printing) because Static's already-flushed rows never participate in + * Yoga layout again, which is fundamentally incompatible with letting old items visually + * scroll within a *bounded* viewport. Trade-off: every item re-renders on every frame now + * (Static rendered each item exactly once, ever) — fine at the sizes a single session + * reaches before auto-compaction, but worth knowing if a session gets huge. */} + + {/* flexShrink={0} is load-bearing: Yoga's default flexShrink is nonzero, so without this the + * content box (and every item inside it) gets squeezed down to the viewport's height instead + * of clipped at it — Yoga distributes the deficit proportionally across every child, which + * rounds most rows down to zero height and leaves only a handful of survivors, rendering as + * scrambled/decimated lines instead of a clean top slice or bottom slice of real content. */} + + {staticItems.map((item) => ( + + ))} + {streamingText !== null && ( + + )} + {isThinking && streamingText === null && !permission && !exportPrompt && ( + + )} + + + + + {permission ? ( + + ) : exportPrompt ? ( + + ) : phase === "starting" ? ( + + ) : phase === "connecting" ? ( + + ) : phase === "loading-models" ? ( + + ) : phase === "session-select" ? ( + + ) : phase === "model-select" ? ( + + ) : ( + <> + {!pinnedToBottom && sessionRef.current && phaseRef.current === "input" && ( + ── scrolled up · PageDown to jump to latest ── + )} + {sessionRef.current && phaseRef.current === "input" && ( + + )} + {isThinking || streamingText !== null ? ( + // Swapped in for ChatInput while a turn is in flight. Closes a real (if rare) race: + // handleSubmit had no guard against firing while a turn was already running, so + // typing and hitting Enter mid-stream could submit a second overlapping turn onto + // the same session. + + Waiting for response… (esc to interrupt) + + ) : ( + + )} + )} - - - {permission ? ( - - ) : exportPrompt ? ( - - ) : phase === "starting" ? ( - - ) : phase === "connecting" ? ( - - ) : phase === "loading-models" ? ( - - ) : phase === "session-select" ? ( - - ) : phase === "model-select" ? ( - - ) : ( - <> - {sessionRef.current && phaseRef.current === "input" && ( - - )} - {isThinking || streamingText !== null ? ( - // Swapped in for ChatInput while a turn is in flight. Closes a real (if rare) race: - // handleSubmit had no guard against firing while a turn was already running, so - // typing and hitting Enter mid-stream could submit a second overlapping turn onto - // the same session. - - Waiting for response… - - ) : ( - - )} - - )} - + setFilePanelFocused(false)} + /> ); } diff --git a/src/ui/ink/ChatInput.tsx b/src/ui/ink/ChatInput.tsx index 4fd80a4..e6b0f38 100644 --- a/src/ui/ink/ChatInput.tsx +++ b/src/ui/ink/ChatInput.tsx @@ -5,14 +5,22 @@ import stringWidth from "string-width"; import { getAbsolutePosition } from "./absolutePosition.js"; import { ACCENT_HEX } from "../theme.js"; import { getActiveMention } from "../../utils/mentions.js"; +import { matchMouseSequence } from "./mouseInput.js"; interface Props { value: string; onChange: (value: string) => void; onSubmit: (value: string) => void; - onCyclePermMode?: () => void; cwd: string; history?: string[]; + /** Actual terminal columns available to this box's content — normally the full terminal width, + * but narrower when the file panel (App.tsx) is showing alongside it. Defaults to the terminal's + * own column count so callers that don't have a side panel can omit it. */ + availableColumns?: number; + /** False while the file panel has keyboard focus (App.tsx) — disables this component's own + * useInput so the same keystroke (arrows, Enter, Escape) doesn't also edit/submit the chat + * input while it's being used to navigate the file tree. Defaults to true. */ + isActive?: boolean; } const MAX_MATCHES = 50; @@ -22,7 +30,7 @@ const PROMPT_WIDTH = 2; // Border (1 col each side) + paddingX={1} (1 col each side) around the bordered box's content. const BOX_CHROME_WIDTH = 4; -export function ChatInput({ value, onChange, onSubmit, onCyclePermMode, cwd, history = [] }: Props) { +export function ChatInput({ value, onChange, onSubmit, cwd, history = [], availableColumns, isActive = true }: Props) { const [allFiles, setAllFiles] = useState(null); const [selectedIndex, setSelectedIndex] = useState(0); const [historyIndex, setHistoryIndex] = useState(-1); @@ -43,14 +51,17 @@ export function ChatInput({ value, onChange, onSubmit, onCyclePermMode, cwd, his // time (through however many wrapper Boxes sit above this one) proved fragile in practice. const { hasMeasured } = useBoxMetrics(boxRef); const { setCursorPosition } = useCursor(); - // Terminal columns, not this box's own measured width: the input box is always width="100%" - // with no horizontal siblings, so its content width is deterministically `columns - - // BOX_CHROME_WIDTH`. Using the measured width instead briefly produced a near-zero value before - // the box's first real layout pass landed (hasMeasured only means the ref is attached, not that - // Yoga has computed a real width yet) — with content width floored at 1, every single typed - // character was computed as needing its own wrapped row, so the reported cursor row grew by one - // per keystroke, visibly "falling" down the screen as you typed. + // Terminal columns, not this box's own measured width: the input box is always width="100%" of + // its column (deterministic), so its content width is `availableColumns - BOX_CHROME_WIDTH`. + // Using the measured width instead briefly produced a near-zero value before the box's first real + // layout pass landed (hasMeasured only means the ref is attached, not that Yoga has computed a + // real width yet) — with content width floored at 1, every single typed character was computed as + // needing its own wrapped row, so the reported cursor row grew by one per keystroke, visibly + // "falling" down the screen as you typed. `availableColumns` defaults to the raw terminal width + // for callers with no horizontal siblings; App.tsx passes the narrower figure when the file panel + // is showing alongside this box, since it isn't full terminal width in that case. const { columns: terminalColumns } = useWindowSize(); + const contentColumns = availableColumns ?? terminalColumns; const mention = getActiveMention(value); @@ -103,7 +114,7 @@ export function ChatInput({ value, onChange, onSubmit, onCyclePermMode, cwd, his // is up to date before this same render's insertion effect runs. if (hasMeasured) { const origin = getAbsolutePosition(boxRef.current); - const contentWidth = Math.max(1, terminalColumns - BOX_CHROME_WIDTH); + const contentWidth = Math.max(1, contentColumns - BOX_CHROME_WIDTH); const totalWidth = PROMPT_WIDTH + stringWidth(value.slice(0, cursorOffset)); const row = Math.floor(totalWidth / contentWidth); const col = totalWidth % contentWidth; @@ -162,11 +173,15 @@ export function ChatInput({ value, onChange, onSubmit, onCyclePermMode, cwd, his } useInput((input, key) => { - // Shift+Tab: cycle permission mode (takes priority even while suggestions are open). - if (key.shift && key.tab) { - onCyclePermMode?.(); - return; - } + // App.tsx's own useInput (mounted for the whole app) already handles xterm SGR mouse sequences + // (wheel scroll, clicks) for the history viewport — but ink broadcasts every raw stdin event to + // every active useInput hook, so this component sees the same sequence too. Without this guard + // it falls through to the catch-all `if (input)` below and types the raw escape text into the + // chat box on every scroll notch. See mouseInput.ts. + if (matchMouseSequence(input)) return; + // Shift+Tab (cycle permission mode) is handled globally in App.tsx now, not here — see its + // useInput handler for why. + if (key.shift && key.tab) return; if (key.escape) { cancelMention(); return; @@ -231,7 +246,7 @@ export function ChatInput({ value, onChange, onSubmit, onCyclePermMode, cwd, his if (input) { replaceValue(value.slice(0, cursorOffset) + input + value.slice(cursorOffset), cursorOffset + input.length); } - }); + }, { isActive }); return ( diff --git a/src/ui/ink/FilePanel.test.ts b/src/ui/ink/FilePanel.test.ts new file mode 100644 index 0000000..ec01980 --- /dev/null +++ b/src/ui/ink/FilePanel.test.ts @@ -0,0 +1,57 @@ +import { describe, expect, it } from "vitest"; +import { buildTree, flattenTree } from "./FilePanel.js"; + +function names(lines: ReturnType): string[] { + return lines.map((l) => `${" ".repeat(l.depth)}${l.isDir ? `${l.name}/` : l.name}`); +} + +describe("buildTree / flattenTree", () => { + it("nests files under their directories, directories sorted before files", () => { + const tree = buildTree(["src/index.ts", "src/utils/foo.ts", "README.md"]); + const lines = flattenTree(tree); + expect(names(lines)).toEqual(["src/", " utils/", " foo.ts", " index.ts", "README.md"]); + }); + + it("sorts directories before files, alphabetically within each group", () => { + const tree = buildTree(["z.ts", "a.ts", "zdir/x.ts", "adir/x.ts"]); + const lines = flattenTree(tree); + expect(names(lines)).toEqual(["adir/", " x.ts", "zdir/", " x.ts", "a.ts", "z.ts"]); + }); + + it("produces stable, unique keys per full relative path", () => { + const tree = buildTree(["a/x.ts", "b/x.ts"]); + const keys = flattenTree(tree).map((l) => l.key); + expect(new Set(keys).size).toBe(keys.length); + expect(keys).toContain("a/x.ts"); + expect(keys).toContain("b/x.ts"); + }); + + it("handles an empty file list", () => { + expect(flattenTree(buildTree([]))).toEqual([]); + }); + + it("marks a directory as isDir with collapsed=false by default", () => { + const tree = buildTree(["src/index.ts"]); + const [srcLine] = flattenTree(tree); + expect(srcLine).toMatchObject({ key: "src", isDir: true, collapsed: false }); + }); + + it("omits a collapsed directory's children but keeps its own line", () => { + const tree = buildTree(["src/index.ts", "src/utils/foo.ts", "README.md"]); + const lines = flattenTree(tree, new Set(["src"])); + expect(names(lines)).toEqual(["src/", "README.md"]); + expect(lines.find((l) => l.key === "src")).toMatchObject({ collapsed: true }); + }); + + it("collapsing a directory doesn't affect a sibling directory's expansion", () => { + const tree = buildTree(["a/x.ts", "b/y.ts"]); + const lines = flattenTree(tree, new Set(["a"])); + expect(names(lines)).toEqual(["a/", "b/", " y.ts"]); + }); + + it("collapsing a nested directory only hides its own subtree", () => { + const tree = buildTree(["src/utils/a.ts", "src/utils/b.ts", "src/index.ts"]); + const lines = flattenTree(tree, new Set(["src/utils"])); + expect(names(lines)).toEqual(["src/", " utils/", " index.ts"]); + }); +}); diff --git a/src/ui/ink/FilePanel.tsx b/src/ui/ink/FilePanel.tsx new file mode 100644 index 0000000..b1e2c73 --- /dev/null +++ b/src/ui/ink/FilePanel.tsx @@ -0,0 +1,256 @@ +import { Box, Text, useBoxMetrics, useInput, type DOMElement } from "ink"; +import fg from "fast-glob"; +import { useEffect, useMemo, useRef, useState } from "react"; +import { ACCENT_HEX } from "../theme.js"; + +export type FilePanelTab = "files" | "activity"; + +export interface TouchedFile { + /** Relative to cwd — what's actually shown, so it stays legible regardless of where the project + * lives on disk. */ + relPath: string; + status: "read" | "written" | "edited"; + /** How many times this file has been touched this session — shown as "(N)" past the first. */ + count: number; + lastTouchedAt: number; +} + +interface Props { + visible: boolean; + activeTab: FilePanelTab; + cwd: string; + touchedFiles: TouchedFile[]; + width: number; + /** Whether the panel currently owns keyboard input (App.tsx disables ChatInput's own useInput + * while this is true, so the same arrow/Enter/Escape keystroke doesn't do both at once). */ + focused: boolean; + /** Escape while focused calls this to hand keyboard focus back to the chat input — the panel + * itself stays visible, only `focused` flips. */ + onExitFocus: () => void; +} + +const STATUS_COLOR: Record = { read: "gray", written: "green", edited: "yellow" }; +const STATUS_GLYPH: Record = { read: "·", written: "+", edited: "~" }; + +interface TreeNode { + children: Map; + isDir: boolean; +} + +export function buildTree(paths: string[]): TreeNode { + const root: TreeNode = { children: new Map(), isDir: true }; + for (const filePath of paths) { + const parts = filePath.split("/"); + let node = root; + parts.forEach((part, i) => { + const isLast = i === parts.length - 1; + let child = node.children.get(part); + if (!child) { + child = { children: new Map(), isDir: !isLast }; + node.children.set(part, child); + } + node = child; + }); + } + return root; +} + +export interface TreeLine { + /** Full path relative to cwd — unique per line, and what collapsedPaths/selection key on. */ + key: string; + depth: number; + name: string; + isDir: boolean; + /** Only meaningful when isDir — whether this directory's children are hidden. */ + collapsed: boolean; +} + +/** Directories first, then alphabetical within each group — matches how most file explorers sort. + * A collapsed directory's own line is still emitted (so it stays selectable/expandable) but its + * children are skipped entirely, the same way a real file explorer hides them. */ +export function flattenTree(node: TreeNode, collapsedPaths: ReadonlySet = new Set(), prefix = "", depth = 0): TreeLine[] { + const entries = [...node.children.entries()].sort(([nameA, a], [nameB, b]) => { + if (a.isDir !== b.isDir) return a.isDir ? -1 : 1; + return nameA.localeCompare(nameB); + }); + const lines: TreeLine[] = []; + for (const [name, child] of entries) { + const fullPath = prefix ? `${prefix}/${name}` : name; + const collapsed = child.isDir && collapsedPaths.has(fullPath); + lines.push({ key: fullPath, depth, name, isDir: child.isDir, collapsed }); + if (child.isDir && !collapsed) { + lines.push(...flattenTree(child, collapsedPaths, fullPath, depth + 1)); + } + } + return lines; +} + +// Same ignore list ChatInput's `@`-mention picker uses (see getActiveMention/fg call there) — +// keeps the two file listings consistent rather than drifting apart independently. +const IGNORE = ["node_modules/**", ".git/**", "dist/**"]; +// A very large repo could produce tens of thousands of paths; capping keeps the tree-build/sort/ +// flatten cheap. Collapsing directories (this component's whole point) is the actual answer to a +// tree that's too big to show at once — this cap is just a hard ceiling under that. +const MAX_FILES = 2000; + +export function FilePanel({ visible, activeTab, cwd, touchedFiles, width, focused, onExitFocus }: Props) { + const [allFiles, setAllFiles] = useState(null); + const [collapsedPaths, setCollapsedPaths] = useState>(new Set()); + const [selectedIndex, setSelectedIndex] = useState(0); + const [scrollTop, setScrollTop] = useState(0); + const scrollRef = useRef(null); + const { height: measuredScrollHeight } = useBoxMetrics(scrollRef); + const visibleRows = Math.max(1, measuredScrollHeight); + + // Fetched lazily on first show (not on mount) and cached for the rest of the session — the panel + // itself stays mounted at all times (hidden via display:none below) specifically so this cache + // survives toggling the panel off and back on, rather than re-globbing the project every time. + useEffect(() => { + if (!visible || allFiles !== null) return; + let cancelled = false; + fg("**/*", { cwd, dot: false, onlyFiles: true, absolute: false, ignore: IGNORE }) + .then((files) => { + if (!cancelled) setAllFiles(files.sort().slice(0, MAX_FILES)); + }) + .catch(() => { + if (!cancelled) setAllFiles([]); + }); + return () => { + cancelled = true; + }; + }, [visible, allFiles, cwd]); + + const treeLines = useMemo(() => (allFiles ? flattenTree(buildTree(allFiles), collapsedPaths) : []), [allFiles, collapsedPaths]); + const lineCount = activeTab === "files" ? treeLines.length : touchedFiles.length; + + // Fresh list each time you switch tabs — a leftover selection/scroll position from the other + // tab's (usually different-length) list would either point at the wrong row or be out of range. + useEffect(() => { + setSelectedIndex(0); + setScrollTop(0); + }, [activeTab]); + + // Clamp on every shrink (collapsing a directory, or the file list finishing its first fetch) + // rather than just when growing, so a collapse that removes the selected row doesn't leave + // selectedIndex pointing past the new end of the list. + useEffect(() => { + setSelectedIndex((i) => Math.max(0, Math.min(i, lineCount - 1))); + }, [lineCount]); + + // Keeps the selected row inside the currently-scrolled window — same marginTop-shift technique + // App.tsx's history viewport uses (see effectiveScrollTop there), just driven by selection moving + // instead of new content arriving. + useEffect(() => { + setScrollTop((top) => { + if (selectedIndex < top) return selectedIndex; + if (selectedIndex >= top + visibleRows) return selectedIndex - visibleRows + 1; + return top; + }); + }, [selectedIndex, visibleRows]); + + function toggleCollapse(dirPath: string) { + setCollapsedPaths((prev) => { + const next = new Set(prev); + if (next.has(dirPath)) next.delete(dirPath); + else next.add(dirPath); + return next; + }); + } + + useInput( + (_input, key) => { + if (key.escape) { + onExitFocus(); + return; + } + if (key.upArrow) { + setSelectedIndex((i) => Math.max(0, i - 1)); + return; + } + if (key.downArrow) { + setSelectedIndex((i) => Math.min(lineCount - 1, i + 1)); + return; + } + if (activeTab !== "files") return; + const line = treeLines[selectedIndex]; + if (!line?.isDir) return; + if (key.return) { + toggleCollapse(line.key); + } else if (key.leftArrow && !line.collapsed) { + toggleCollapse(line.key); + } else if (key.rightArrow && line.collapsed) { + toggleCollapse(line.key); + } + }, + { isActive: visible && focused }, + ); + + return ( + // display "none" (not conditional mounting) so the glob fetch above only ever runs once per + // session regardless of how many times the panel is toggled — see the effect's comment. + + + + Files + + · + + Activity + + + {"─".repeat(Math.max(1, width - 4))} + + {/* flexShrink={0} is load-bearing the same way it is in App.tsx's history content box: + * without it, Yoga shrinks this box (and every line inside it) to fit the panel's height + * instead of letting overflowY:hidden clip it, which renders as scrambled/decimated lines + * rather than a clean top slice. */} + + {activeTab === "files" ? ( + allFiles === null ? ( + Loading… + ) : treeLines.length === 0 ? ( + No files found. + ) : ( + treeLines.map((line, i) => { + const isSelected = focused && i === selectedIndex; + const chevron = line.isDir ? (line.collapsed ? "▸ " : "▾ ") : " "; + const label = line.isDir ? `${line.name}/` : line.name; + return ( + + {isSelected ? "❯" : " "} + {" ".repeat(line.depth)} + {chevron} + {label} + + ); + }) + ) + ) : touchedFiles.length === 0 ? ( + No files touched yet. + ) : ( + touchedFiles.map((f, i) => { + const isSelected = focused && i === selectedIndex; + return ( + + {isSelected ? "❯ " : " "} + {STATUS_GLYPH[f.status]} {f.relPath} + {f.count > 1 ? ({f.count}) : null} + + ); + }) + )} + + + {"─".repeat(Math.max(1, width - 4))} + {focused ? "↑↓ move · ↵/←/→ expand · Esc unfocus" : "Ctrl+G tab · Ctrl+F focus/hide"} + + ); +} diff --git a/src/ui/ink/HistoryItemView.tsx b/src/ui/ink/HistoryItemView.tsx index 771d5ef..89b340e 100644 --- a/src/ui/ink/HistoryItemView.tsx +++ b/src/ui/ink/HistoryItemView.tsx @@ -1,4 +1,5 @@ import { Box, Text } from "ink"; +import { memo } from "react"; import { renderMarkdown } from "../render.js"; import { ACCENT_HEX } from "../theme.js"; import type { HistoryItem } from "./types.js"; @@ -15,6 +16,7 @@ const HELP_LINES = [ " /permissions list mutating tools allowed for the rest of this session", " /sessions list saved conversations you can resume with --resume", " /mcp show connected MCP servers and their tool counts", + " /mcp reconnect re-connect to configured MCP servers (after editing .mcp.json or restarting one)", " /plugins show installed Claude Code-compatible plugins (commands, agents, MCP servers)", " /hooks show configured hooks per lifecycle event", " /skills show installed skills; / [request] invokes one directly", @@ -26,9 +28,14 @@ const HELP_LINES = [ " /exit, /quit exit", "", "Keyboard shortcuts:", + " Esc interrupt the current response", " Shift+Tab cycle permission mode", " Ctrl+O print the full text of the last /compact summary", " Ctrl+B background the currently-running bash command", + " Ctrl+F open/focus the file panel; press again to close it (Esc to unfocus without closing)", + " Ctrl+G switch the file panel's tab (Files / Activity)", + " ↑↓ ↵ ← → (while the file panel is focused) navigate / expand / collapse folders", + " PageUp/PageDown scroll the conversation", ]; function formatDuration(ms: number): string { @@ -41,7 +48,12 @@ function formatDuration(ms: number): string { return `${s}s`; } -export function HistoryItemView({ item }: { item: HistoryItem }) { +// Memoized: staticItems only ever grows by appending, so every previously-committed item keeps +// the same object reference across re-renders. Without this, every historical item (including +// "assistant" ones, whose renderMarkdown call is real CPU work) gets its render function +// re-invoked on every parent re-render — which happens on every ThinkingIndicator spinner tick +// and every throttled streaming-text frame — even though nothing about that item changed. +export const HistoryItemView = memo(function HistoryItemView({ item }: { item: HistoryItem }) { switch (item.kind) { case "banner": return ( @@ -395,4 +407,4 @@ export function HistoryItemView({ item }: { item: HistoryItem }) { ); } -} +}); diff --git a/src/ui/ink/index.tsx b/src/ui/ink/index.tsx index 5b06076..f647411 100644 --- a/src/ui/ink/index.tsx +++ b/src/ui/ink/index.tsx @@ -61,9 +61,9 @@ export async function runInkApp(opts: RunInkAppOptions): Promise { // // The model-select/connecting phases before a session exists stay on the main screen (so any // startup errors remain in normal scrollback); once a session actually starts, we switch to the - // alternate screen buffer for a clean full-screen chat view. This trades away scrollback (no - // mouse-wheel/Shift+PgUp scrolling once in the alt screen) for that full-screen feel — a - // deliberate choice, revisit if the lack of scrollback turns out to matter more in practice. + // alternate screen buffer for a clean full-screen chat view. This trades away the terminal's own + // scrollback (no mouse-wheel/Shift+PgUp once in the alt screen) for that full-screen feel — App.tsx + // provides its own in-app scrollback instead (PageUp/PageDown over the history viewport). const instance = render( [0]); +// marked.parse() + marked-terminal's ANSI formatting is real CPU work (word-wrap, table layout, +// syntax highlighting). Historical assistant messages are immutable once committed, so re-parsing +// the same text on every re-render (every spinner tick, every streamed-token frame) is pure waste +// that scales with total conversation length — cache by input text so each message is parsed once. +const renderCache = new Map(); + export function renderMarkdown(text: string): string { + const cached = renderCache.get(text); + if (cached !== undefined) return cached; const rendered = marked.parse(text); - return typeof rendered === "string" ? rendered.trimEnd() : text; + const result = typeof rendered === "string" ? rendered.trimEnd() : text; + renderCache.set(text, result); + return result; } diff --git a/src/utils/tokens.test.ts b/src/utils/tokens.test.ts new file mode 100644 index 0000000..92b6d33 --- /dev/null +++ b/src/utils/tokens.test.ts @@ -0,0 +1,106 @@ +import { describe, expect, it } from "vitest"; +import { estimateTokens } from "./tokens.js"; +import type { ChatCompletionMessageParam } from "openai/resources/chat/completions"; + +describe("estimateTokens — script-aware heuristic", () => { + it("returns a positive number for a single system message", () => { + const msgs: ChatCompletionMessageParam[] = [{ role: "system", content: "You are helpful." }]; + expect(estimateTokens(msgs)).toBeGreaterThan(0); + }); + + it("scales roughly linearly with English prose length", () => { + const short: ChatCompletionMessageParam[] = [{ role: "user", content: "hello" }]; + const long: ChatCompletionMessageParam[] = [ + { role: "user", content: "hello ".repeat(100) + "world" }, + ]; + // ~605 chars vs 5 chars: the long message should be many times larger (overhead aside). + expect(estimateTokens(long)).toBeGreaterThan(estimateTokens(short) * 15); + }); + + it("estimates CJK text as more tokens than the same length of Latin text", () => { + // Same character count, but Korean characters each map closer to 1:1 token. + const latin: ChatCompletionMessageParam[] = [ + { role: "user", content: "a".repeat(40) }, + ]; + const korean: ChatCompletionMessageParam[] = [ + { role: "user", content: "안".repeat(40) }, + ]; + expect(estimateTokens(korean)).toBeGreaterThan(estimateTokens(latin)); + }); + + it("estimates symbol-heavy (code) text as more tokens than prose of the same length", () => { + const prose: ChatCompletionMessageParam[] = [ + { role: "user", content: "word word word word word word word word" }, + ]; + const code: ChatCompletionMessageParam[] = [ + { role: "user", content: "{}{}{}{}{}{}{}{}()()()()()()()()[][][][]" }, + ]; + // Both 39 chars; the code version has more dense-symbol weight. + expect(estimateTokens(code)).toBeGreaterThan(estimateTokens(prose)); + }); + + it("accounts for tool_calls structure", () => { + const withCalls: ChatCompletionMessageParam[] = [ + { + role: "assistant", + content: null, + tool_calls: [ + { + id: "call_1", + type: "function", + function: { name: "read_file", arguments: '{"path":"src/index.ts"}' }, + }, + ], + } as ChatCompletionMessageParam, + ]; + expect(estimateTokens(withCalls)).toBeGreaterThan(10); + }); + + it("accounts for tool result messages", () => { + const result: ChatCompletionMessageParam[] = [ + { role: "tool", tool_call_id: "call_1", content: "the file contents are here" }, + ] as ChatCompletionMessageParam[]; + expect(estimateTokens(result)).toBeGreaterThan(10); + }); + + it("handles multipart content arrays (text parts)", () => { + const msgs: ChatCompletionMessageParam[] = [ + { + role: "user", + content: [ + { type: "text", text: "describe this" }, + { type: "text", text: "and this" }, + ], + } as ChatCompletionMessageParam, + ]; + expect(estimateTokens(msgs)).toBeGreaterThan( + estimateTokens([{ role: "user", content: "describe this" }]), + ); + }); + + it("charges a flat cost for non-text content parts (images)", () => { + const textOnly: ChatCompletionMessageParam[] = [ + { role: "user", content: [{ type: "text", text: "look" }] } as unknown as ChatCompletionMessageParam, + ]; + const withImage: ChatCompletionMessageParam[] = [ + { + role: "user", + content: [ + { type: "text", text: "look" }, + { type: "image_url", image_url: { url: "data:image/png;base64,abc" } }, + ], + } as unknown as ChatCompletionMessageParam, + ]; + expect(estimateTokens(withImage)).toBeGreaterThan(estimateTokens(textOnly)); + }); + + it("grows with more messages, not just longer content", () => { + const one: ChatCompletionMessageParam[] = [{ role: "user", content: "ab" }]; + const two: ChatCompletionMessageParam[] = [ + { role: "user", content: "ab" }, + { role: "assistant", content: "cd" }, + ]; + // Each message carries a per-message overhead, so two short messages cost more than one. + expect(estimateTokens(two)).toBeGreaterThan(estimateTokens(one)); + }); +}); \ No newline at end of file diff --git a/src/utils/tokens.ts b/src/utils/tokens.ts index 63fa1b8..4f6d538 100644 --- a/src/utils/tokens.ts +++ b/src/utils/tokens.ts @@ -1,8 +1,121 @@ import type { ChatCompletionMessageParam } from "openai/resources/chat/completions"; -/** Rough token estimate (~4 chars/token) used when the backend doesn't report real usage stats - * (via `stream_options: { include_usage: true }`) or before any turn has run yet. */ +/** + * Per-message structural overhead the chat template adds (role tags, delimiters, etc.). + * Most chat templates wrap each message with ~3–5 special tokens (`<|im_start|>role\n`, + * `<|im_end|>\n`, etc.), so charge a flat per-message cost regardless of content length. + */ +const PER_MESSAGE_OVERHEAD = 4; + +/** + * Estimate the token count of a chat message array without backend support. + * + * The old `JSON.stringify(messages).length / 4` heuristic had two failure modes: + * 1. It counted the JSON serialization overhead (quotes, braces, escaped delimiters) as + * content tokens — inflating the estimate by ~15–25% since those aren't sent to the model. + * 2. It used one flat chars-per-token ratio for everything, but that ratio varies a lot by + * script: English prose is ~4 chars/token, code/symbols ~3.5, and CJK (Korean/Chinese/ + * Japanese) is ~1.5 chars/token because each code point is usually its own BPE token. + * + * This walker reconstructs only the text the model actually sees (system/user content strings, + * assistant content, tool calls, tool results) and applies a script-aware ratio. It stays a + * pure synchronous estimate — no backend calls — so it's safe to use before the first turn, + * after compaction, and when creating sub-agents. + */ export function estimateTokens(messages: ChatCompletionMessageParam[]): number { - const chars = JSON.stringify(messages).length; - return Math.ceil(chars / 4); + let tokens = 0; + for (const msg of messages) { + tokens += PER_MESSAGE_OVERHEAD; + tokens += estimateContentTokens(msg); + } + // A trailing assistant generation sentinel / chat-template end tokens. + tokens += 3; + return Math.max(1, tokens); } + +function estimateContentTokens(msg: ChatCompletionMessageParam): number { + let chars = 0; + + const content = (msg as { content?: unknown }).content; + if (typeof content === "string") { + chars += weightedChars(content); + } else if (Array.isArray(content)) { + for (const part of content) { + if (part == null) continue; + if (typeof part === "string") { + chars += weightedChars(part); + } else if (typeof part === "object") { + const p = part as { type?: string; text?: string }; + // Text parts contribute their text; image/audio parts are a flat structural cost + // (the model sees a placeholder image token, not the base64 bytes). + if (p.type === "text" && typeof p.text === "string") chars += weightedChars(p.text); + else chars += 8; + } + } + } + + // Tool calls: the model emits a JSON-ish structure; estimate its serialized size. + const toolCalls = (msg as { tool_calls?: unknown }).tool_calls; + if (Array.isArray(toolCalls)) { + for (const call of toolCalls) { + const c = call as { function?: { name?: string; arguments?: string }; id?: string }; + if (c.function?.name) chars += weightedChars(c.function.name) + 4; + if (c.function?.arguments) chars += weightedChars(c.function.arguments) + 2; + if (c.id) chars += weightedChars(c.id) + 2; + chars += 6; // call/function wrapper tokens + } + } + + // Tool result role: the content is the tool output text. + // `name` and `tool_call_id` are small metadata fields. + const name = (msg as { name?: string }).name; + if (name) chars += weightedChars(name) + 2; + const toolCallId = (msg as { tool_call_id?: string }).tool_call_id; + if (toolCallId) chars += weightedChars(toolCallId) + 2; + + return Math.ceil(chars); +} + +/** + * Map a content string to "effective chars" using a script-aware weight, then divided by a + * base chars-per-token ratio to get tokens. The weight encodes that a single CJK code point + * is worth roughly one token (ratio ~1.5) while Latin code points cluster ~4 per token, and + * dense punctuation/symbols (common in code) are closer to ~3.5. + * + * We accumulate weighted chars and the caller divides by the base ratio once. + */ +function weightedChars(text: string): number { + let weight = 0; + for (let i = 0; i < text.length; i++) { + const code = text.charCodeAt(i); + if (isCjk(code)) { + weight += 2.4; // ~1.5 chars/token instead of 4 → ×2.4 + } else if (isDenseSymbol(code)) { + weight += 1.15; // ~3.5 chars/token → ×1.15 + } else { + weight += 1; // base ~4 chars/token + } + } + return weight / 4; // base ratio +} + +function isCjk(code: number): boolean { + // CJK Unified Ideographs + extensions, Hiragana, Katakana, Hangul syllables/jamo. + return ( + (code >= 0x3040 && code <= 0x30ff) || // Hiragana + Katakana + (code >= 0x3400 && code <= 0x9fff) || // CJK ideographs (incl. ext A) + (code >= 0xac00 && code <= 0xd7af) || // Hangul syllables + (code >= 0x1100 && code <= 0x11ff) // Hangul jamo + ); +} + +function isDenseSymbol(code: number): boolean { + // Punctuation, math, operators, brackets — the kind of characters that dominate code and + // tend to each consume ~1 BPE token rather than sharing a token with neighbours. + return ( + (code >= 0x21 && code <= 0x2f) || // ! " # $ % & ' ( ) * + , - . / + (code >= 0x3a && code <= 0x40) || // : ; < = > ? @ + (code >= 0x5b && code <= 0x60) || // [ \ ] ^ _ ` + (code >= 0x7b && code <= 0x7e) // { | } ~ + ); +} \ No newline at end of file diff --git a/src/utils/truncate.test.ts b/src/utils/truncate.test.ts new file mode 100644 index 0000000..00f7b3b --- /dev/null +++ b/src/utils/truncate.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, it } from "vitest"; +import { truncate } from "./truncate.js"; + +describe("truncate — head+tail preservation", () => { + it("returns short text unchanged", () => { + expect(truncate("hello", 100)).toBe("hello"); + }); + + it("returns text at exactly the limit unchanged", () => { + const text = "x".repeat(100); + expect(truncate(text, 100)).toBe(text); + }); + + it("keeps the head and tail and elides the middle", () => { + // 50 lines of 10 chars each = 599 chars (49 newlines). Cap at 300. + const lines = Array.from({ length: 50 }, (_, i) => `line${String(i).padStart(5, "0")}`); + const text = lines.join("\n"); + const out = truncate(text, 300); + + // Head preserved: first line still present. + expect(out).toContain("line00000"); + // Tail preserved: last line still present. + expect(out).toContain("line00049"); + // Middle dropped: a middle line is gone. + expect(out).not.toContain("line00025"); + // Truncation marker present with a character count. + expect(out).toMatch(/\[truncated \d+ more characters/); + }); + + it("preserves the tail error line a model most needs to see", () => { + const lines = Array.from({ length: 200 }, (_, i) => `row ${i}: data`); + // Append a final error line — the whole point of keeping the tail. + lines.push("Error: compilation failed at line 42"); + const text = lines.join("\n"); + const out = truncate(text, 300); + expect(out).toContain("Error: compilation failed at line 42"); + }); + + it("cuts on line boundaries, never half a line", () => { + const lines = Array.from({ length: 100 }, (_, i) => `line-${i}-${"x".repeat(40)}`); + const text = lines.join("\n"); + const out = truncate(text, 800); + // The head section's last kept line should be a complete line, not a fragment. + const firstPart = out.split("\n\n... [truncated")[0]!; + for (const line of firstPart.split("\n")) { + // Every kept head line should start with the known prefix or be empty. + expect(line === "" || /^line-\d+-x+$/.test(line)).toBe(true); + } + }); + + it("handles a file only slightly over budget without overlapping head and tail", () => { + const lines = Array.from({ length: 20 }, (_, i) => `line ${i}`); + const text = lines.join("\n"); + // Budget just under total length so head and tail would overlap without the guard. + const out = truncate(text, text.length - 1); + // Should still return something readable with a truncation marker. + expect(out).toMatch(/\[truncated/); + // No duplicate lines: each kept line appears at most once as a *whole line* + // (match against line boundaries, not as a substring, since "line 1" is a substring of "line 10"). + const outLines = out.split("\n"); + for (const line of lines) { + const occurrences = outLines.filter((l) => l === line).length; + expect(occurrences).toBeLessThanOrEqual(1); + } + }); + + it("falls back to a character cut when every line is huge", () => { + // One massive line longer than the head budget. + const text = "x".repeat(5000); + const out = truncate(text, 1000); + expect(out).toMatch(/\[truncated/); + expect(out.length).toBeLessThan(text.length); + }); +}); \ No newline at end of file diff --git a/src/utils/truncate.ts b/src/utils/truncate.ts index 379d2af..a2b1c7e 100644 --- a/src/utils/truncate.ts +++ b/src/utils/truncate.ts @@ -1,4 +1,66 @@ +/** + * Cap a string to roughly `maxChars` by keeping the head and the tail and eliding the + * middle — far more useful for command output than a bare head-only cut, because the tail + * usually carries the error/status line a model most needs to see. + * + * Cuts on line boundaries when possible so the result stays readable, and reports exactly + * how many characters were dropped so the model knows there is more it can't see. + */ export function truncate(text: string, maxChars = 20_000): string { if (text.length <= maxChars) return text; - return `${text.slice(0, maxChars)}\n... [truncated ${text.length - maxChars} more characters]`; -} + + // Reserve a few lines for the truncation notice itself so the final string doesn't + // overshoot maxChars after we splice the marker back in. + const NOTICE_SLACK = 120; + const budget = Math.max(maxChars - NOTICE_SLACK, Math.floor(maxChars * 0.9)); + + // Split into lines so we can cut on boundaries. We keep whole lines for both head and tail, + // never half a line — half-lines confuse both the model and the tests. + const lines = text.split("\n"); + + // First pass: keep ~60% of the budget for the head, ~40% for the tail. The head carries + // context and the tail carries the outcome (exit status, error, final summary). + const headBudget = Math.floor(budget * 0.6); + const tailBudget = budget - headBudget; + + const headLines: string[] = []; + let headChars = 0; + for (const line of lines) { + // +1 accounts for the "\n" we'll rejoin with. + if (headChars + line.length + 1 > headBudget) break; + headLines.push(line); + headChars += line.length + 1; + } + + const tailLines: string[] = []; + let tailChars = 0; + for (let i = lines.length - 1; i >= 0; i--) { + const line = lines[i]!; + if (tailChars + line.length + 1 > tailBudget) break; + // Avoid overlapping with the head when the file is only slightly over budget. + if (i < headLines.length) break; + tailLines.unshift(line); + tailChars += line.length + 1; + } + + const keptHead = headLines.join("\n"); + const keptTail = tailLines.join("\n"); + const dropped = text.length - (keptHead.length + keptTail.length); + + // Edge case: the file is over budget but every single line is huge (longer than headBudget), + // so the loop above kept zero head lines. Fall back to a character cut so we still return + // something useful rather than an empty head + the whole tail. + if (headLines.length === 0) { + const headSlice = text.slice(0, headBudget); + const tailSlice = text.slice(text.length - tailBudget); + const droppedChars = text.length - headSlice.length - tailSlice.length; + return `${headSlice}\n\n... [truncated ${droppedChars} more characters — head+tail preserved, middle omitted] ...\n\n${tailSlice}`; + } + + if (tailLines.length === 0) { + // Tail couldn't keep anything without overlapping head: just head + notice. + return `${keptHead}\n\n... [truncated ${dropped} more characters — tail omitted] ...`; + } + + return `${keptHead}\n\n... [truncated ${dropped} more characters — middle omitted, ${headLines.length} head + ${tailLines.length} tail lines kept] ...\n\n${keptTail}`; +} \ No newline at end of file