- Separate system prompt for local vs cloud models via isLocal flag (isLocalBackendURL, buildSystemPrompt(..., isLocal), propagate through createSession/createSessionFromRecord/compactSession/spawnSubAgent) - Increase MAX_EMPTY_RESPONSE_RETRIES from 1 to 3 for cloud model resilience - Upgrade mouse input: full SGR-1006 parsing with col/row/pressed/modifiers, logicalButton() helper, and copyToClipboard() via OSC 52 - Add temporary mouse debug logging in App.tsx - Increase DEFAULT_MAX_ITERATIONS from 100 to 300 - Update README mouse/scrollback docs, tweak diff-remove color
70 lines
4.2 KiB
TypeScript
70 lines
4.2 KiB
TypeScript
export const DEFAULT_OLLAMA_BASE_URL = "http://localhost:11434/v1";
|
||
export const DEFAULT_LMSTUDIO_BASE_URL = "http://localhost:1234/v1";
|
||
|
||
export const KNOWN_BACKENDS = {
|
||
ollama: DEFAULT_OLLAMA_BASE_URL,
|
||
lmstudio: DEFAULT_LMSTUDIO_BASE_URL,
|
||
} as const;
|
||
|
||
export type BackendName = keyof typeof KNOWN_BACKENDS;
|
||
|
||
/** Returns true if the given base URL looks like a local backend (Ollama or LM Studio on localhost).
|
||
* Used to tailor the system prompt — local models get extra guidance about their limitations,
|
||
* while cloud models (which are more capable and reliable) get a leaner prompt without that framing. */
|
||
export function isLocalBackendURL(baseURL: string): boolean {
|
||
try {
|
||
const url = new URL(baseURL);
|
||
return url.hostname === "localhost" || url.hostname === "127.0.0.1" || url.hostname === "::1";
|
||
} catch {
|
||
return false;
|
||
}
|
||
}
|
||
|
||
/** Used when the context window can't be auto-detected from the backend (see backend/contextWindow.ts)
|
||
* and the user hasn't configured one — a conservative size common among smaller local models. */
|
||
export const DEFAULT_CONTEXT_WINDOW = 8192;
|
||
|
||
/** Ceiling on a single response's `max_tokens`, independent of the model's context window. Most
|
||
* backends cap how much a single completion can generate well below the total context window they
|
||
* advertise (e.g. Ollama's glm-5.2:cloud reports a 1,000,000-token context window but only ever
|
||
* generates up to 8192 tokens per response) — resolveMaxTokens (agent/loop.ts) used to request up
|
||
* to the whole remaining window, which such backends rejected outright as a context/length error
|
||
* even on the very first turn. 8192 is a safe default most backends support; raise it via
|
||
* `locode config set maxOutputTokens` for backends known to allow more. */
|
||
export const DEFAULT_MAX_OUTPUT_TOKENS = 8192;
|
||
|
||
/** Max model requests per turn before locode pauses rather than looping forever. Each iteration
|
||
* is one model generation request (one tool-call round-trip), and local models commonly issue a
|
||
* single tool call per request — so a real multi-file task (read several files, edit each, grep
|
||
* to verify, re-read) easily needs 40–60 requests. 50 was too tight and caused frequent
|
||
* "Paused after 50 steps" soft-stops on legitimate work; 100 still caused frequent pauses on
|
||
* larger tasks. 300 gives real tasks ample room to finish while still bounding a genuinely stuck
|
||
* model. Hitting the cap is a soft pause, not a failure (the work so far is intact — send another
|
||
* message to resume). Configurable via `maxIterations`, e.g. `locode config set maxIterations 500`
|
||
* for very large batch jobs. */
|
||
export const DEFAULT_MAX_ITERATIONS = 300;
|
||
|
||
/** Fraction of the context window at which locode automatically summarizes the conversation.
|
||
* User-configurable via `locode config set autoCompactThreshold`. */
|
||
export const DEFAULT_AUTO_COMPACT_THRESHOLD = 0.85;
|
||
|
||
/** How long to wait on a single chat completion request before giving up. The OpenAI SDK retries
|
||
* transient failures (connection errors, 429, 5xx) up to `maxRetries` times with exponential
|
||
* backoff before surfacing the error; set to 0 to fail immediately like older locode versions.
|
||
* Raise this via `maxRetries` if your backend has occasional transient blips. See backend/client.ts. */
|
||
export const DEFAULT_MAX_RETRIES = 0;
|
||
|
||
/** How long to wait on a single chat completion request before giving up. Raise this via
|
||
* `requestTimeoutMs` if your backend queues requests behind a concurrency limit (e.g. Ollama's
|
||
* `OLLAMA_NUM_PARALLEL`) rather than serving them immediately. */
|
||
export const DEFAULT_REQUEST_TIMEOUT_MS = 180_000;
|
||
|
||
/** Wall-clock budget for a single sub-agent turn. Sub-agents make their own sequence of model
|
||
* requests (one per file read, grep, etc.), and on a cloud backend each request has real network
|
||
* latency on top of generation time — so a sub-agent reading 20-30 files can legitimately take
|
||
* several minutes. The old 120s hardcoded cap timed those out mid-task (the parent would see
|
||
* "Sub-agent timed out" and give up on delegating). The per-request idle guard
|
||
* (DEFAULT_REQUEST_TIMEOUT_MS) still catches a single hung request; this bounds only the whole
|
||
* sub-agent turn. Configurable via `LOCODE_SUBAGENT_TIMEOUT_MS`. */
|
||
export const DEFAULT_SUBAGENT_TIMEOUT_MS = 600_000;
|