- Add todo_write tool with live checklist rendering in the UI - Add permission modes (default/plan/auto-edit/auto-accept) cycled via Shift+Tab or /perm - Load CLAUDE.md/AGENTS.md as project instructions into the system prompt - Forward sub-agent tool calls/results to parent UI panel - Auto-compact context mid-turn and keep model moving with a synthetic user turn - Prune older image_url parts to cap vision token cost - Idle-abort guard for stalled streaming backends - Improve background bash jobs, kill whole process trees on timeout/exit - Expand hooks: command/http hooks, blocking, JSON outputSchema - MCP: JSON schema → Zod conversion, parallel connection, duplicate server detection - Add tests for loop, confirmFn, projectInstructions, readFile, bash, todoWrite, contextWindowCache
21 lines
1.1 KiB
TypeScript
21 lines
1.1 KiB
TypeScript
import OpenAI from "openai";
|
|
import { resolveRequestTimeoutMs } from "../config/config.js";
|
|
import type { AppConfig } from "../config/types.js";
|
|
|
|
export function makeClient(cfg: AppConfig): OpenAI {
|
|
return new OpenAI({
|
|
baseURL: cfg.baseURL,
|
|
// Honor explicit API keys for OpenAI-compatible proxies/services; default to a dummy value for
|
|
// local backends that don't check it.
|
|
apiKey: process.env.OPENAI_API_KEY ?? process.env.LOCODE_API_KEY ?? "local",
|
|
// The SDK defaults to a 10-minute timeout with 2 retries (up to 30 min before a request ever
|
|
// fails). For local backends a slow response almost always means the model is genuinely stuck,
|
|
// not a transient network blip, so retrying just compounds the wait — fail faster instead.
|
|
// Configurable (`requestTimeoutMs` / LOCODE_REQUEST_TIMEOUT_MS) because a backend that queues
|
|
// requests behind a concurrency limit (e.g. Ollama's OLLAMA_NUM_PARALLEL) can legitimately take
|
|
// longer than the 180s default to even start serving a request under contention.
|
|
timeout: resolveRequestTimeoutMs(),
|
|
maxRetries: 0,
|
|
});
|
|
}
|