Upgrade candidates (all 12 done): - #1 parallel read-only tool execution (runToolBatch, 4 loop sites) - #2 configurable retry policy (maxRetries + exponential backoff via SDK) - #3 script-aware token estimation (CJK/symbol/structure-aware heuristic) - #4 head+tail output capping (truncate.ts), auto-applies to bash/git/etc - #5 partial-history compaction (preserve recent tail, summarize older prefix) - #6 dynamic max_tokens (resolveMaxTokens) - #7 richer tool descriptions with "use when" guidance - #8 MCP reconnect retry + /mcp reconnect command + session toolset refresh - #9 context-window cache TTL (cachedAt timestamp, default 7 days) - #10 edit_file similar-match suggestion on old_string miss (bounded Levenshtein) - #11 git_status output head+tail (resolved via #4) - #12 auto-accept now approves all mutating tools; auto-edit stays edit-only Other: - DEFAULT_MAX_ITERATIONS 50 -> 100 (local models issue one tool call per step) - MaxIterationsError message guides resume + config override - plus prior known-issues work (grep -e/--, session id sanitization, MCP content types, 12 hook events, plugin collisions, skill references, git ops expansion, configurable autoCompactThreshold, bashGuard, pathGuard, FilePanel, replay)
25 lines
1.5 KiB
TypeScript
25 lines
1.5 KiB
TypeScript
import OpenAI from "openai";
|
|
import { resolveMaxRetries, resolveRequestTimeoutMs } from "../config/config.js";
|
|
import type { AppConfig } from "../config/types.js";
|
|
|
|
export function makeClient(cfg: AppConfig): OpenAI {
|
|
return new OpenAI({
|
|
baseURL: cfg.baseURL,
|
|
// Honor explicit API keys for OpenAI-compatible proxies/services; default to a dummy value for
|
|
// local backends that don't check it.
|
|
apiKey: process.env.OPENAI_API_KEY ?? process.env.LOCODE_API_KEY ?? "local",
|
|
// The SDK defaults to a 10-minute timeout with 2 retries (up to 30 min before a request ever
|
|
// fails). For local backends a slow response almost always means the model is genuinely stuck,
|
|
// not a transient network blip, so retrying just compounds the wait — fail faster instead.
|
|
// Configurable (`requestTimeoutMs` / LOCODE_REQUEST_TIMEOUT_MS) because a backend that queues
|
|
// requests behind a concurrency limit (e.g. Ollama's OLLAMA_NUM_PARALLEL) can legitimately take
|
|
// longer than the 180s default to even start serving a request under contention.
|
|
timeout: resolveRequestTimeoutMs(),
|
|
// Configurable retries on transient failures (connection errors, 429, 5xx) with exponential
|
|
// backoff. Defaults to 0 (fail immediately) to preserve the old behavior, since a local
|
|
// backend's slow response usually means the model is stuck rather than a transient blip — but
|
|
// raise via `maxRetries` / LOCODE_MAX_RETRIES for setups with occasional connection drops.
|
|
maxRetries: resolveMaxRetries(),
|
|
});
|
|
}
|