fix: wol-gate로 깨우는 중인 Ollama 대상에 재시도 대기 로직 추가

지서버처럼 wol-gate 프록시 뒤에 있는 Ollama 엔드포인트가 잠들어 있으면,
프록시가 실제 API 응답 대신 "깨우는 중..." HTML 페이지를 돌려주는데
ollama 클라이언트 라이브러리가 이걸 JSON으로 파싱하려다 매번
"Unexpected token '<'" 에러로 죽는 문제 발견(실제 채팅에서 반복 확인,
2026-08-08). chat/generate/chatStream 진입 시점에 /api/tags를 먼저
가볍게 찔러보고, HTML(깨우는 중) 응답이면 최대 90초까지 5초 간격으로
재시도 대기한 뒤 실제 호출을 진행하도록 수정 — 이미 깨어있는 경우엔
응답이 JSON이라 즉시 통과, 지연 거의 없음(실측 확인).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-08-08 23:29:11 +09:00
co-authored by Claude Sonnet 5
parent b8ee2e1be9
commit 6a4b3f8d68
+28
View File
@@ -444,7 +444,33 @@ export class OllamaAdapter implements LLMProvider {
return null;
}
// A wol-gate-fronted Ollama endpoint (e.g. 지서버, [[project_wol_gate]]) answers a sleeping
// target with an HTML "waking up, refresh in 5s" page instead of proxying — the ollama
// client library then tries to JSON.parse that HTML and throws "Unexpected token '<'"
// (observed live 2026-08-08, repeatedly, on the first request after any idle period).
// Poll a lightweight endpoint first and wait out the wake-up window so the real call only
// fires once the target is actually answering as itself.
private async _waitIfWakingUp(): Promise<void> {
const maxWaitMs = 90_000;
const pollIntervalMs = 5_000;
const deadline = Date.now() + maxWaitMs;
while (Date.now() < deadline) {
try {
const res = await fetch(`${this.endpoint}/api/tags`, { signal: AbortSignal.timeout(4_000) });
const text = await res.text();
if (!text.trimStart().startsWith('<')) return; // real (JSON) response — target is up
} catch {
// Connection refused/timeout could mean "still waking" or "genuinely unreachable" —
// either way, keep polling within the budget rather than guessing which.
}
await new Promise((r) => setTimeout(r, pollIntervalMs));
}
// Waited the full budget without a real response — let the actual call proceed anyway
// and surface whatever error it hits; don't block forever on a target that never wakes.
}
async chat(messages: ChatMessage[], model: string, options?: ChatOptions): Promise<ChatResult> {
await this._waitIfWakingUp();
const workspacePath = getConfig().getConfig().workspace?.path;
// Ollama requires the last message to be user or tool — strip trailing assistant messages
while (messages.length > 1 && messages[messages.length - 1].role === 'assistant') {
@@ -569,6 +595,7 @@ export class OllamaAdapter implements LLMProvider {
}
async generate(prompt: string, model: string, options?: GenerateOptions): Promise<GenerateResult> {
await this._waitIfWakingUp();
const thinkCandidates = this.buildThinkCandidates(options?.think);
const modelCtx = options?.num_ctx ?? await this.getModelCtx(model);
let lastError: any = null;
@@ -617,6 +644,7 @@ export class OllamaAdapter implements LLMProvider {
}
async chatStream(messages: ChatMessage[], model: string, options: ChatOptions & { onToken: (text: string) => void }): Promise<ChatResult> {
await this._waitIfWakingUp();
const workspacePath = getConfig().getConfig().workspace?.path;
// Ollama requires the last message to be user or tool — strip trailing assistant messages
while (messages.length > 1 && messages[messages.length - 1].role === 'assistant') {