From 6a4b3f8d68056499d5a85793d527efa2256bfc01 Mon Sep 17 00:00:00 2001 From: kim Date: Sat, 8 Aug 2026 23:29:11 +0900 Subject: [PATCH] =?UTF-8?q?fix:=20wol-gate=EB=A1=9C=20=EA=B9=A8=EC=9A=B0?= =?UTF-8?q?=EB=8A=94=20=EC=A4=91=EC=9D=B8=20Ollama=20=EB=8C=80=EC=83=81?= =?UTF-8?q?=EC=97=90=20=EC=9E=AC=EC=8B=9C=EB=8F=84=20=EB=8C=80=EA=B8=B0=20?= =?UTF-8?q?=EB=A1=9C=EC=A7=81=20=EC=B6=94=EA=B0=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 지서버처럼 wol-gate 프록시 뒤에 있는 Ollama 엔드포인트가 잠들어 있으면, 프록시가 실제 API 응답 대신 "깨우는 중..." HTML 페이지를 돌려주는데 ollama 클라이언트 라이브러리가 이걸 JSON으로 파싱하려다 매번 "Unexpected token '<'" 에러로 죽는 문제 발견(실제 채팅에서 반복 확인, 2026-08-08). chat/generate/chatStream 진입 시점에 /api/tags를 먼저 가볍게 찔러보고, HTML(깨우는 중) 응답이면 최대 90초까지 5초 간격으로 재시도 대기한 뒤 실제 호출을 진행하도록 수정 — 이미 깨어있는 경우엔 응답이 JSON이라 즉시 통과, 지연 거의 없음(실측 확인). Co-Authored-By: Claude Sonnet 5 --- src/providers/ollama-adapter.ts | 28 ++++++++++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/src/providers/ollama-adapter.ts b/src/providers/ollama-adapter.ts index 189f99c..10b61a9 100644 --- a/src/providers/ollama-adapter.ts +++ b/src/providers/ollama-adapter.ts @@ -444,7 +444,33 @@ export class OllamaAdapter implements LLMProvider { return null; } + // A wol-gate-fronted Ollama endpoint (e.g. 지서버, [[project_wol_gate]]) answers a sleeping + // target with an HTML "waking up, refresh in 5s" page instead of proxying — the ollama + // client library then tries to JSON.parse that HTML and throws "Unexpected token '<'" + // (observed live 2026-08-08, repeatedly, on the first request after any idle period). + // Poll a lightweight endpoint first and wait out the wake-up window so the real call only + // fires once the target is actually answering as itself. + private async _waitIfWakingUp(): Promise { + const maxWaitMs = 90_000; + const pollIntervalMs = 5_000; + const deadline = Date.now() + maxWaitMs; + while (Date.now() < deadline) { + try { + const res = await fetch(`${this.endpoint}/api/tags`, { signal: AbortSignal.timeout(4_000) }); + const text = await res.text(); + if (!text.trimStart().startsWith('<')) return; // real (JSON) response — target is up + } catch { + // Connection refused/timeout could mean "still waking" or "genuinely unreachable" — + // either way, keep polling within the budget rather than guessing which. + } + await new Promise((r) => setTimeout(r, pollIntervalMs)); + } + // Waited the full budget without a real response — let the actual call proceed anyway + // and surface whatever error it hits; don't block forever on a target that never wakes. + } + async chat(messages: ChatMessage[], model: string, options?: ChatOptions): Promise { + await this._waitIfWakingUp(); const workspacePath = getConfig().getConfig().workspace?.path; // Ollama requires the last message to be user or tool — strip trailing assistant messages while (messages.length > 1 && messages[messages.length - 1].role === 'assistant') { @@ -569,6 +595,7 @@ export class OllamaAdapter implements LLMProvider { } async generate(prompt: string, model: string, options?: GenerateOptions): Promise { + await this._waitIfWakingUp(); const thinkCandidates = this.buildThinkCandidates(options?.think); const modelCtx = options?.num_ctx ?? await this.getModelCtx(model); let lastError: any = null; @@ -617,6 +644,7 @@ export class OllamaAdapter implements LLMProvider { } async chatStream(messages: ChatMessage[], model: string, options: ChatOptions & { onToken: (text: string) => void }): Promise { + await this._waitIfWakingUp(); const workspacePath = getConfig().getConfig().workspace?.path; // Ollama requires the last message to be user or tool — strip trailing assistant messages while (messages.length > 1 && messages[messages.length - 1].role === 'assistant') {