diff --git a/src/providers/ollama-adapter.ts b/src/providers/ollama-adapter.ts index 189f99c..10b61a9 100644 --- a/src/providers/ollama-adapter.ts +++ b/src/providers/ollama-adapter.ts @@ -444,7 +444,33 @@ export class OllamaAdapter implements LLMProvider { return null; } + // A wol-gate-fronted Ollama endpoint (e.g. 지서버, [[project_wol_gate]]) answers a sleeping + // target with an HTML "waking up, refresh in 5s" page instead of proxying — the ollama + // client library then tries to JSON.parse that HTML and throws "Unexpected token '<'" + // (observed live 2026-08-08, repeatedly, on the first request after any idle period). + // Poll a lightweight endpoint first and wait out the wake-up window so the real call only + // fires once the target is actually answering as itself. + private async _waitIfWakingUp(): Promise { + const maxWaitMs = 90_000; + const pollIntervalMs = 5_000; + const deadline = Date.now() + maxWaitMs; + while (Date.now() < deadline) { + try { + const res = await fetch(`${this.endpoint}/api/tags`, { signal: AbortSignal.timeout(4_000) }); + const text = await res.text(); + if (!text.trimStart().startsWith('<')) return; // real (JSON) response — target is up + } catch { + // Connection refused/timeout could mean "still waking" or "genuinely unreachable" — + // either way, keep polling within the budget rather than guessing which. + } + await new Promise((r) => setTimeout(r, pollIntervalMs)); + } + // Waited the full budget without a real response — let the actual call proceed anyway + // and surface whatever error it hits; don't block forever on a target that never wakes. + } + async chat(messages: ChatMessage[], model: string, options?: ChatOptions): Promise { + await this._waitIfWakingUp(); const workspacePath = getConfig().getConfig().workspace?.path; // Ollama requires the last message to be user or tool — strip trailing assistant messages while (messages.length > 1 && messages[messages.length - 1].role === 'assistant') { @@ -569,6 +595,7 @@ export class OllamaAdapter implements LLMProvider { } async generate(prompt: string, model: string, options?: GenerateOptions): Promise { + await this._waitIfWakingUp(); const thinkCandidates = this.buildThinkCandidates(options?.think); const modelCtx = options?.num_ctx ?? await this.getModelCtx(model); let lastError: any = null; @@ -617,6 +644,7 @@ export class OllamaAdapter implements LLMProvider { } async chatStream(messages: ChatMessage[], model: string, options: ChatOptions & { onToken: (text: string) => void }): Promise { + await this._waitIfWakingUp(); const workspacePath = getConfig().getConfig().workspace?.path; // Ollama requires the last message to be user or tool — strip trailing assistant messages while (messages.length > 1 && messages[messages.length - 1].role === 'assistant') {