Files
locode/src/backend/client.ts
T
kim 6fe98887d5 v0.6.0: complete all 12 local-model upgrades + raise maxIterations to 100
Upgrade candidates (all 12 done):
- #1 parallel read-only tool execution (runToolBatch, 4 loop sites)
- #2 configurable retry policy (maxRetries + exponential backoff via SDK)
- #3 script-aware token estimation (CJK/symbol/structure-aware heuristic)
- #4 head+tail output capping (truncate.ts), auto-applies to bash/git/etc
- #5 partial-history compaction (preserve recent tail, summarize older prefix)
- #6 dynamic max_tokens (resolveMaxTokens)
- #7 richer tool descriptions with "use when" guidance
- #8 MCP reconnect retry + /mcp reconnect command + session toolset refresh
- #9 context-window cache TTL (cachedAt timestamp, default 7 days)
- #10 edit_file similar-match suggestion on old_string miss (bounded Levenshtein)
- #11 git_status output head+tail (resolved via #4)
- #12 auto-accept now approves all mutating tools; auto-edit stays edit-only

Other:
- DEFAULT_MAX_ITERATIONS 50 -> 100 (local models issue one tool call per step)
- MaxIterationsError message guides resume + config override
- plus prior known-issues work (grep -e/--, session id sanitization, MCP content
  types, 12 hook events, plugin collisions, skill references, git ops expansion,
  configurable autoCompactThreshold, bashGuard, pathGuard, FilePanel, replay)
2026-08-20 16:52:03 +09:00

25 lines
1.5 KiB
TypeScript

import OpenAI from "openai";
import { resolveMaxRetries, resolveRequestTimeoutMs } from "../config/config.js";
import type { AppConfig } from "../config/types.js";
export function makeClient(cfg: AppConfig): OpenAI {
return new OpenAI({
baseURL: cfg.baseURL,
// Honor explicit API keys for OpenAI-compatible proxies/services; default to a dummy value for
// local backends that don't check it.
apiKey: process.env.OPENAI_API_KEY ?? process.env.LOCODE_API_KEY ?? "local",
// The SDK defaults to a 10-minute timeout with 2 retries (up to 30 min before a request ever
// fails). For local backends a slow response almost always means the model is genuinely stuck,
// not a transient network blip, so retrying just compounds the wait — fail faster instead.
// Configurable (`requestTimeoutMs` / LOCODE_REQUEST_TIMEOUT_MS) because a backend that queues
// requests behind a concurrency limit (e.g. Ollama's OLLAMA_NUM_PARALLEL) can legitimately take
// longer than the 180s default to even start serving a request under contention.
timeout: resolveRequestTimeoutMs(),
// Configurable retries on transient failures (connection errors, 429, 5xx) with exponential
// backoff. Defaults to 0 (fail immediately) to preserve the old behavior, since a local
// backend's slow response usually means the model is stuck rather than a transient blip — but
// raise via `maxRetries` / LOCODE_MAX_RETRIES for setups with occasional connection drops.
maxRetries: resolveMaxRetries(),
});
}