fix: wol-gate로 깨우는 중인 Ollama 대상에 재시도 대기 로직 추가
지서버처럼 wol-gate 프록시 뒤에 있는 Ollama 엔드포인트가 잠들어 있으면, 프록시가 실제 API 응답 대신 "깨우는 중..." HTML 페이지를 돌려주는데 ollama 클라이언트 라이브러리가 이걸 JSON으로 파싱하려다 매번 "Unexpected token '<'" 에러로 죽는 문제 발견(실제 채팅에서 반복 확인, 2026-08-08). chat/generate/chatStream 진입 시점에 /api/tags를 먼저 가볍게 찔러보고, HTML(깨우는 중) 응답이면 최대 90초까지 5초 간격으로 재시도 대기한 뒤 실제 호출을 진행하도록 수정 — 이미 깨어있는 경우엔 응답이 JSON이라 즉시 통과, 지연 거의 없음(실측 확인). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -444,7 +444,33 @@ export class OllamaAdapter implements LLMProvider {
|
||||
return null;
|
||||
}
|
||||
|
||||
// A wol-gate-fronted Ollama endpoint (e.g. 지서버, [[project_wol_gate]]) answers a sleeping
|
||||
// target with an HTML "waking up, refresh in 5s" page instead of proxying — the ollama
|
||||
// client library then tries to JSON.parse that HTML and throws "Unexpected token '<'"
|
||||
// (observed live 2026-08-08, repeatedly, on the first request after any idle period).
|
||||
// Poll a lightweight endpoint first and wait out the wake-up window so the real call only
|
||||
// fires once the target is actually answering as itself.
|
||||
private async _waitIfWakingUp(): Promise<void> {
|
||||
const maxWaitMs = 90_000;
|
||||
const pollIntervalMs = 5_000;
|
||||
const deadline = Date.now() + maxWaitMs;
|
||||
while (Date.now() < deadline) {
|
||||
try {
|
||||
const res = await fetch(`${this.endpoint}/api/tags`, { signal: AbortSignal.timeout(4_000) });
|
||||
const text = await res.text();
|
||||
if (!text.trimStart().startsWith('<')) return; // real (JSON) response — target is up
|
||||
} catch {
|
||||
// Connection refused/timeout could mean "still waking" or "genuinely unreachable" —
|
||||
// either way, keep polling within the budget rather than guessing which.
|
||||
}
|
||||
await new Promise((r) => setTimeout(r, pollIntervalMs));
|
||||
}
|
||||
// Waited the full budget without a real response — let the actual call proceed anyway
|
||||
// and surface whatever error it hits; don't block forever on a target that never wakes.
|
||||
}
|
||||
|
||||
async chat(messages: ChatMessage[], model: string, options?: ChatOptions): Promise<ChatResult> {
|
||||
await this._waitIfWakingUp();
|
||||
const workspacePath = getConfig().getConfig().workspace?.path;
|
||||
// Ollama requires the last message to be user or tool — strip trailing assistant messages
|
||||
while (messages.length > 1 && messages[messages.length - 1].role === 'assistant') {
|
||||
@@ -569,6 +595,7 @@ export class OllamaAdapter implements LLMProvider {
|
||||
}
|
||||
|
||||
async generate(prompt: string, model: string, options?: GenerateOptions): Promise<GenerateResult> {
|
||||
await this._waitIfWakingUp();
|
||||
const thinkCandidates = this.buildThinkCandidates(options?.think);
|
||||
const modelCtx = options?.num_ctx ?? await this.getModelCtx(model);
|
||||
let lastError: any = null;
|
||||
@@ -617,6 +644,7 @@ export class OllamaAdapter implements LLMProvider {
|
||||
}
|
||||
|
||||
async chatStream(messages: ChatMessage[], model: string, options: ChatOptions & { onToken: (text: string) => void }): Promise<ChatResult> {
|
||||
await this._waitIfWakingUp();
|
||||
const workspacePath = getConfig().getConfig().workspace?.path;
|
||||
// Ollama requires the last message to be user or tool — strip trailing assistant messages
|
||||
while (messages.length > 1 && messages[messages.length - 1].role === 'assistant') {
|
||||
|
||||
Reference in New Issue
Block a user