Files
locode/src/config/config.ts
T
kimandClaude Sonnet 5 5bf60c7921 fix: local/cloud model detection, capability cache TTL, chat cursor offset
- isSmallLocalModel(baseURL, model) replaces isLocalBackendURL() for system-prompt
  branching and context-window fallback defaults: Ollama's cloud-routed models
  (glm-5.2:cloud, qwen3.5:397b-cloud, etc.) share a localhost endpoint with
  genuinely local models, so the base URL alone can't tell them apart. Recomputed
  on /model and /backend switches too, not just at session creation.
- Fixed a latent isLocalBackendURL bug found while testing it: URL.hostname keeps
  the brackets on a literal IPv6 host ("[::1]"), so the old "::1" comparison never
  matched.
- capabilityCache entries now carry a cachedAt timestamp with a 30-day TTL
  (LOCODE_CAPABILITY_CACHE_TTL_DAYS), so a stale "fallback" verdict from a
  transient probe failure doesn't permanently disable native tool calls.
- ChatInput's cursor row offset (+2 -> +1): the extra row was empirical padding
  for a bottomSectionRef wrapper Box and virtual-scroll viewport that no longer
  exist since the Static-based rendering change.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LxiAaGhSD4DRVYYQZ5GJjm
2026-08-24 13:51:25 +09:00

143 lines
6.6 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import {
DEFAULT_AUTO_COMPACT_THRESHOLD,
DEFAULT_CONTEXT_WINDOW_LOCAL,
DEFAULT_CONTEXT_WINDOW_CLOUD,
DEFAULT_MAX_ITERATIONS,
DEFAULT_MAX_OUTPUT_TOKENS,
DEFAULT_MAX_RETRIES,
DEFAULT_REQUEST_TIMEOUT_MS,
DEFAULT_SUBAGENT_TIMEOUT_MS,
KNOWN_BACKENDS,
type BackendName,
isSmallLocalModel,
} from "./defaults.js";
import { loadStoredConfig } from "./store.js";
export class ConfigError extends Error {}
export interface CliBackendOpts {
backend?: string;
baseUrl?: string;
}
/** Precedence: CLI flags > env vars > persisted config file > defaults. */
export function resolveBackendConfig(cliOpts: CliBackendOpts): {
backendName: string;
baseURL: string;
} {
const stored = loadStoredConfig();
const backend = cliOpts.backend ?? process.env.LOCODE_BACKEND ?? stored.backend ?? "ollama";
const explicitBaseUrl = cliOpts.baseUrl ?? process.env.LOCODE_BASE_URL ?? stored.baseUrl;
// Allow any backend name when an explicit base URL is provided; only the built-in names have a
// default URL, so a custom name without --base-url is still an error.
if (backend !== "ollama" && backend !== "lmstudio" && !explicitBaseUrl) {
throw new ConfigError(
`Unknown backend "${backend}". Use --backend ollama|lmstudio, or pass --base-url for a custom endpoint.`,
);
}
const knownBaseURL = (KNOWN_BACKENDS as Record<string, string>)[backend];
const baseURL = explicitBaseUrl ?? knownBaseURL;
if (!baseURL) {
throw new ConfigError(`No base URL configured for backend "${backend}".`);
}
return { backendName: backend, baseURL };
}
/** Returns undefined (rather than throwing) when no model is configured, so callers can prompt interactively. */
export function resolveModel(cliModel?: string): string | undefined {
const stored = loadStoredConfig();
return cliModel ?? process.env.LOCODE_MODEL ?? stored.model;
}
/** The fallback context window size to use when it can't be auto-detected from the backend. Picks a
* small-local or cloud-scale default based on isSmallLocalModel() — the backend host alone isn't
* enough, since Ollama's cloud-routed models (e.g. "glm-5.2:cloud") share a local host with
* genuinely local ones. */
export function resolveContextWindowDefault(baseURL: string, model: string): number {
const stored = loadStoredConfig();
const envValue = Number(process.env.LOCODE_CONTEXT_WINDOW);
if (Number.isFinite(envValue) && envValue > 0) return envValue;
if (typeof stored.contextWindow === "number" && stored.contextWindow > 0) return stored.contextWindow;
return isSmallLocalModel(baseURL, model) ? DEFAULT_CONTEXT_WINDOW_LOCAL : DEFAULT_CONTEXT_WINDOW_CLOUD;
}
/** Ceiling on a single response's max_tokens (see DEFAULT_MAX_OUTPUT_TOKENS), independent of the
* context window. Bounded to 256–1,000,000 to reject pathological values. */
export function resolveMaxOutputTokens(): number {
const stored = loadStoredConfig();
const envValue = Number(process.env.LOCODE_MAX_OUTPUT_TOKENS);
if (Number.isFinite(envValue) && envValue >= 256 && envValue <= 1_000_000) return envValue;
if (typeof stored.maxOutputTokens === "number" && stored.maxOutputTokens >= 256 && stored.maxOutputTokens <= 1_000_000) {
return stored.maxOutputTokens;
}
return DEFAULT_MAX_OUTPUT_TOKENS;
}
/** Max tool calls allowed per turn before locode gives up. */
export function resolveMaxIterations(): number {
const stored = loadStoredConfig();
const envValue = Number(process.env.LOCODE_MAX_ITERATIONS);
if (Number.isFinite(envValue) && envValue > 0) return envValue;
if (typeof stored.maxIterations === "number" && stored.maxIterations > 0) return stored.maxIterations;
return DEFAULT_MAX_ITERATIONS;
}
/** Wall-clock budget for a single sub-agent turn (see DEFAULT_SUBAGENT_TIMEOUT_MS). Bounded to
* 1s–1h to reject pathological values. */
export function resolveSubagentTimeoutMs(): number {
const stored = loadStoredConfig();
const envValue = Number(process.env.LOCODE_SUBAGENT_TIMEOUT_MS);
if (Number.isFinite(envValue) && envValue >= 1_000 && envValue <= 3_600_000) return envValue;
if (typeof stored.subagentTimeoutMs === "number" && stored.subagentTimeoutMs >= 1_000 && stored.subagentTimeoutMs <= 3_600_000) {
return stored.subagentTimeoutMs;
}
return DEFAULT_SUBAGENT_TIMEOUT_MS;
}
/** Fraction of the context window at which locode auto-compacts the conversation. */
export function resolveAutoCompactThreshold(): number {
const stored = loadStoredConfig();
const envValue = Number(process.env.LOCODE_AUTO_COMPACT_THRESHOLD);
if (Number.isFinite(envValue) && envValue >= 0.1 && envValue <= 0.95) return envValue;
if (typeof stored.autoCompactThreshold === "number" && stored.autoCompactThreshold >= 0.1 && stored.autoCompactThreshold <= 0.95) {
return stored.autoCompactThreshold;
}
return DEFAULT_AUTO_COMPACT_THRESHOLD;
}
/** Max retry attempts the OpenAI SDK makes on transient failures (connection errors, 429, 5xx)
* with exponential backoff. Bounded to 0–10 to reject pathological values. 0 = fail immediately,
* matching locode's old behavior of never retrying (a slow local backend usually means the model
* is genuinely stuck, not a transient blip — but some setups have occasional connection drops). */
export function resolveMaxRetries(): number {
const stored = loadStoredConfig();
const envValue = Number(process.env.LOCODE_MAX_RETRIES);
if (Number.isFinite(envValue) && envValue >= 0 && envValue <= 10) return envValue;
if (typeof stored.maxRetries === "number" && stored.maxRetries >= 0 && stored.maxRetries <= 10) {
return stored.maxRetries;
}
return DEFAULT_MAX_RETRIES;
}
/** Milliseconds to wait on a single chat completion request before giving up (see backend/client.ts
* for why locode defaults to no retries). Bounded to 10s–30min to reject pathological values. */
export function resolveRequestTimeoutMs(): number {
const stored = loadStoredConfig();
const envValue = Number(process.env.LOCODE_REQUEST_TIMEOUT_MS);
if (Number.isFinite(envValue) && envValue >= 10_000 && envValue <= 1_800_000) return envValue;
if (typeof stored.requestTimeoutMs === "number" && stored.requestTimeoutMs >= 10_000 && stored.requestTimeoutMs <= 1_800_000) {
return stored.requestTimeoutMs;
}
return DEFAULT_REQUEST_TIMEOUT_MS;
}
/** User-configured LSP server overrides/additions (see StoredConfig.lspServers). An empty object
* means "use the built-in language→server mappings only". Validated loosely: entries without a
* command are dropped by configureLanguageSpecs, so we just pass them through. */
export function resolveLspServers(): Record<string, { command: string; args?: string[]; extensions?: string[] }> {
const stored = loadStoredConfig();
return stored.lspServers ?? {};
}