feat: 지서버 직결 + wol-gate 폴백 (하이브리드 연결)

지서버 트래픽이 전부 wol-gate(TrueNAS) 프록시를 거치고 있었다. 게이트는 자는
타깃에 HTTP 200 + "깨우는 중" HTML을 반환하는데, ollama 클라이언트가 그 HTML을
JSON.parse해서 "Unexpected token '<'"로 죽었다.

endpoint를 지서버 직결(192.168.0.8:11434)로 두고, wake_url을 새로 받아
게이트는 직결이 안 될 때만 쓴다.

- _isAwake(host): 200이어도 본문이 HTML이면 잠든 것으로 판정
- 깨어있으면 즉시 통과(fast path). 기존 코드는 매 호출마다 /api/tags를 찍고
  무조건 5초를 기다린 뒤 재확인했다.
- 게이트를 초인종이 아니라 폴백 경로로 사용. 직결과 게이트를 같은 주기로 확인해
  둘 중 되는 쪽으로 이번 호출을 넘긴다. 08-09 00:52 지서버는 살아있고 게이트는
  정상 프록시 중인데 직결만 타임아웃인 상황이 실제로 관찰됐고, 직결만 폴링하면
  작동하는 경로를 두고 90초를 버린 뒤 실패한다.
- getModelCtx()도 _activeHost를 따라가게 수정. 폴백 중에 num_ctx 계산용
  /api/show만 따로 실패하는 것을 막는다.

wake_url이 없는 provider(클로서버 등)는 기존 동작 그대로다.

검증: 직결을 죽은 주소로 바꿔 강제 재현 → 게이트 경유로 정상 응답(33초),
로그에 routing via wol-gate 기록. 원복 후 직결 fast path 9초.
실제로 잠든 지서버에 채팅 → 23초 만에 기상 후 응답.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-08-09 01:04:28 +09:00
co-authored by Claude Opus 5
parent 1146b59199
commit cb8d93e03f
2 changed files with 114 additions and 29 deletions
+14 -8
View File
@@ -17,15 +17,16 @@
}
},
"llm": {
"provider": "ollama",
"provider": "ollama_local",
"providers": {
"ollama": {
"endpoint": "http://localhost:11434",
"model": "gemma4:31b-cloud"
"model": "gemma4:26b"
},
"lm_studio": {
"endpoint": "http://host.docker.internal:1234",
"model": ""
"endpoint": "http://192.168.0.7:1234",
"model": "qwen/qwen3.6-35b-a3b",
"disable_thinking": true
},
"llama_cpp": {
"endpoint": "http://host.docker.internal:8080",
@@ -43,6 +44,11 @@
"api_key": "vault:llm.google.api_key",
"model": "gemini-3.5-flash-lite"
},
"ollama_local": {
"endpoint": "http://192.168.0.8:11434",
"model": "gemma4:26b",
"wake_url": "http://192.168.0.240:8100"
},
"anthropic": {
"api_key": "",
"model": "claude-sonnet-4-6"
@@ -50,12 +56,12 @@
}
},
"models": {
"primary": "gemma4:31b-cloud",
"primary": "gemma4:26b",
"fallback": "kimi-k2.6:cloud",
"roles": {
"manager": "gemma4:31b-cloud",
"executor": "gemma4:31b-cloud",
"verifier": "gemma4:31b-cloud",
"manager": "gemma4:26b",
"executor": "gemma4:26b",
"verifier": "gemma4:26b",
"background_task": ""
},
"profiles": {
+100 -21
View File
@@ -346,6 +346,8 @@ export class OllamaAdapter implements LLMProvider {
readonly id: 'ollama' | 'ollama_local';
private client: Ollama;
private endpoint: string;
/** Host the client currently points at — `endpoint`, or the wol-gate when direct is down. */
private _activeHost: string;
private _ctxCache: Map<string, number> = new Map();
// id defaults to 'ollama' (클로서버) for backward compat — pass 'ollama_local' for
@@ -354,12 +356,14 @@ export class OllamaAdapter implements LLMProvider {
constructor(endpoint: string, id: 'ollama' | 'ollama_local' = 'ollama') {
this.id = id;
this.endpoint = endpoint;
this._activeHost = endpoint;
this.client = new Ollama({ host: endpoint });
}
updateEndpoint(endpoint: string) {
if (endpoint !== this.endpoint) {
this.endpoint = endpoint;
this._activeHost = endpoint;
this.client = new Ollama({ host: endpoint });
this._ctxCache.clear();
}
@@ -372,7 +376,9 @@ export class OllamaAdapter implements LLMProvider {
private async getModelCtx(model: string): Promise<number> {
if (this._ctxCache.has(model)) return this._ctxCache.get(model)!;
try {
const res = await fetch(`${this.endpoint}/api/show`, {
// _activeHost, not endpoint: when direct is down the call itself is going through the
// gate, so the context lookup that sizes num_ctx has to follow the same route.
const res = await fetch(`${this._activeHost}/api/show`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ name: model }),
@@ -444,29 +450,102 @@ export class OllamaAdapter implements LLMProvider {
return null;
}
// A wol-gate-fronted Ollama endpoint (e.g. 지서버, [[project_wol_gate]]) answers a sleeping
// target with an HTML "waking up, refresh in 5s" page instead of proxying — the ollama
// client library then tries to JSON.parse that HTML and throws "Unexpected token '<'"
// (observed live 2026-08-08, repeatedly, on the first request after any idle period).
// Poll a lightweight endpoint first and wait out the wake-up window so the real call only
// fires once the target is actually answering as itself.
private async _waitIfWakingUp(): Promise<void> {
const maxWaitMs = 90_000;
const pollIntervalMs = 5_000;
const deadline = Date.now() + maxWaitMs;
/**
* Is `host` answering as the real Ollama (JSON), rather than absent or fronted by a
* wol-gate "waking up" page? The gate answers for a sleeping target with HTTP 200 and
* an HTML "refresh in 5s" body, so a 200 alone proves nothing — the ollama client then
* JSON.parses that HTML and throws "Unexpected token '<'" ([[project_wol_gate]],
* observed live 2026-08-08 on the first request after any idle period).
*/
private async _isAwake(host: string): Promise<boolean> {
try {
const res = await fetch(`${host}/api/tags`, { signal: AbortSignal.timeout(4_000) });
const text = await res.text();
return !text.trimStart().startsWith('<');
} catch {
return false;
}
}
/** Poke the wol-gate so it emits a magic packet. Response body is irrelevant. */
private async _sendWake(wakeUrl: string): Promise<void> {
try {
await fetch(`${wakeUrl}/api/tags`, { signal: AbortSignal.timeout(4_000) });
} catch {
// The gate answers a sleeping target with HTML (or nothing) — either way the
// request itself is what triggers the WOL packet, so failures here are expected.
}
}
/**
* Point this call at whichever host can actually serve it.
*
* With `wake_url` configured, `endpoint` goes straight to the box (지서버) and the gate is
* only consulted when direct fails — so the proxy stays out of the hot path. Direct is the
* optimization; the gate is the path that is known to work, and it is used as a fallback
* rather than merely as a doorbell: 2026-08-09 00:52 the box was demonstrably up and the
* gate was proxying real JSON while direct timed out from this host. Polling direct alone
* would have stalled the full wake budget and then failed, with a working route sitting
* unused the whole time.
*/
private async _selectHost(): Promise<string> {
const direct = this.endpoint;
if (await this._isAwake(direct)) return direct; // fast path — no extra hop when up
const wakeUrl = this._resolveWakeUrl();
if (!wakeUrl) {
// No separate gate: `endpoint` may itself be gate-fronted, in which case requesting
// it is what wakes the target. Wait it out before letting the real call fire.
await this._pollUntilAwake(direct, 90_000);
return direct;
}
await this._sendWake(wakeUrl);
// Give direct a short window to come back (a cold box needs ~30s to boot), but check
// the gate on the same cadence so a reachable-only-via-gate target is picked up early.
const deadline = Date.now() + 90_000;
while (Date.now() < deadline) {
try {
const res = await fetch(`${this.endpoint}/api/tags`, { signal: AbortSignal.timeout(4_000) });
const text = await res.text();
if (!text.trimStart().startsWith('<')) return; // real (JSON) response — target is up
} catch {
// Connection refused/timeout could mean "still waking" or "genuinely unreachable" —
// either way, keep polling within the budget rather than guessing which.
await new Promise((r) => setTimeout(r, 5_000));
if (await this._isAwake(direct)) return direct;
if (await this._isAwake(wakeUrl)) {
console.warn(`[OllamaAdapter:${this.id}] direct ${direct} unreachable; routing via wol-gate ${wakeUrl}`);
return wakeUrl;
}
await new Promise((r) => setTimeout(r, pollIntervalMs));
}
// Neither route answered within the budget — let the real call proceed and surface
// whatever error it hits rather than blocking forever on a target that never wakes.
return direct;
}
private async _pollUntilAwake(host: string, budgetMs: number): Promise<void> {
const deadline = Date.now() + budgetMs;
while (Date.now() < deadline) {
await new Promise((r) => setTimeout(r, 5_000));
if (await this._isAwake(host)) return;
}
}
/**
* Resolve the host to use and point the client at it for this call. Returns without
* touching the client on the common path where direct is already serving.
*/
private async _waitIfWakingUp(): Promise<void> {
const host = await this._selectHost();
if (host !== this._activeHost) {
this._activeHost = host;
this.client = new Ollama({ host });
}
}
private _resolveWakeUrl(): string | undefined {
try {
const cfg: any = getConfig().getConfig();
const raw = cfg?.llm?.providers?.[this.id]?.wake_url;
return raw ? String(raw).replace(/\/+$/, '') : undefined;
} catch {
return undefined;
}
// Waited the full budget without a real response — let the actual call proceed anyway
// and surface whatever error it hits; don't block forever on a target that never wakes.
}
async chat(messages: ChatMessage[], model: string, options?: ChatOptions): Promise<ChatResult> {