feat: 지서버 직결 + wol-gate 폴백 (하이브리드 연결)
지서버 트래픽이 전부 wol-gate(TrueNAS) 프록시를 거치고 있었다. 게이트는 자는 타깃에 HTTP 200 + "깨우는 중" HTML을 반환하는데, ollama 클라이언트가 그 HTML을 JSON.parse해서 "Unexpected token '<'"로 죽었다. endpoint를 지서버 직결(192.168.0.8:11434)로 두고, wake_url을 새로 받아 게이트는 직결이 안 될 때만 쓴다. - _isAwake(host): 200이어도 본문이 HTML이면 잠든 것으로 판정 - 깨어있으면 즉시 통과(fast path). 기존 코드는 매 호출마다 /api/tags를 찍고 무조건 5초를 기다린 뒤 재확인했다. - 게이트를 초인종이 아니라 폴백 경로로 사용. 직결과 게이트를 같은 주기로 확인해 둘 중 되는 쪽으로 이번 호출을 넘긴다. 08-09 00:52 지서버는 살아있고 게이트는 정상 프록시 중인데 직결만 타임아웃인 상황이 실제로 관찰됐고, 직결만 폴링하면 작동하는 경로를 두고 90초를 버린 뒤 실패한다. - getModelCtx()도 _activeHost를 따라가게 수정. 폴백 중에 num_ctx 계산용 /api/show만 따로 실패하는 것을 막는다. wake_url이 없는 provider(클로서버 등)는 기존 동작 그대로다. 검증: 직결을 죽은 주소로 바꿔 강제 재현 → 게이트 경유로 정상 응답(33초), 로그에 routing via wol-gate 기록. 원복 후 직결 fast path 9초. 실제로 잠든 지서버에 채팅 → 23초 만에 기상 후 응답. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
+14
-8
@@ -17,15 +17,16 @@
|
||||
}
|
||||
},
|
||||
"llm": {
|
||||
"provider": "ollama",
|
||||
"provider": "ollama_local",
|
||||
"providers": {
|
||||
"ollama": {
|
||||
"endpoint": "http://localhost:11434",
|
||||
"model": "gemma4:31b-cloud"
|
||||
"model": "gemma4:26b"
|
||||
},
|
||||
"lm_studio": {
|
||||
"endpoint": "http://host.docker.internal:1234",
|
||||
"model": ""
|
||||
"endpoint": "http://192.168.0.7:1234",
|
||||
"model": "qwen/qwen3.6-35b-a3b",
|
||||
"disable_thinking": true
|
||||
},
|
||||
"llama_cpp": {
|
||||
"endpoint": "http://host.docker.internal:8080",
|
||||
@@ -43,6 +44,11 @@
|
||||
"api_key": "vault:llm.google.api_key",
|
||||
"model": "gemini-3.5-flash-lite"
|
||||
},
|
||||
"ollama_local": {
|
||||
"endpoint": "http://192.168.0.8:11434",
|
||||
"model": "gemma4:26b",
|
||||
"wake_url": "http://192.168.0.240:8100"
|
||||
},
|
||||
"anthropic": {
|
||||
"api_key": "",
|
||||
"model": "claude-sonnet-4-6"
|
||||
@@ -50,12 +56,12 @@
|
||||
}
|
||||
},
|
||||
"models": {
|
||||
"primary": "gemma4:31b-cloud",
|
||||
"primary": "gemma4:26b",
|
||||
"fallback": "kimi-k2.6:cloud",
|
||||
"roles": {
|
||||
"manager": "gemma4:31b-cloud",
|
||||
"executor": "gemma4:31b-cloud",
|
||||
"verifier": "gemma4:31b-cloud",
|
||||
"manager": "gemma4:26b",
|
||||
"executor": "gemma4:26b",
|
||||
"verifier": "gemma4:26b",
|
||||
"background_task": ""
|
||||
},
|
||||
"profiles": {
|
||||
|
||||
+100
-21
@@ -346,6 +346,8 @@ export class OllamaAdapter implements LLMProvider {
|
||||
readonly id: 'ollama' | 'ollama_local';
|
||||
private client: Ollama;
|
||||
private endpoint: string;
|
||||
/** Host the client currently points at — `endpoint`, or the wol-gate when direct is down. */
|
||||
private _activeHost: string;
|
||||
private _ctxCache: Map<string, number> = new Map();
|
||||
|
||||
// id defaults to 'ollama' (클로서버) for backward compat — pass 'ollama_local' for
|
||||
@@ -354,12 +356,14 @@ export class OllamaAdapter implements LLMProvider {
|
||||
constructor(endpoint: string, id: 'ollama' | 'ollama_local' = 'ollama') {
|
||||
this.id = id;
|
||||
this.endpoint = endpoint;
|
||||
this._activeHost = endpoint;
|
||||
this.client = new Ollama({ host: endpoint });
|
||||
}
|
||||
|
||||
updateEndpoint(endpoint: string) {
|
||||
if (endpoint !== this.endpoint) {
|
||||
this.endpoint = endpoint;
|
||||
this._activeHost = endpoint;
|
||||
this.client = new Ollama({ host: endpoint });
|
||||
this._ctxCache.clear();
|
||||
}
|
||||
@@ -372,7 +376,9 @@ export class OllamaAdapter implements LLMProvider {
|
||||
private async getModelCtx(model: string): Promise<number> {
|
||||
if (this._ctxCache.has(model)) return this._ctxCache.get(model)!;
|
||||
try {
|
||||
const res = await fetch(`${this.endpoint}/api/show`, {
|
||||
// _activeHost, not endpoint: when direct is down the call itself is going through the
|
||||
// gate, so the context lookup that sizes num_ctx has to follow the same route.
|
||||
const res = await fetch(`${this._activeHost}/api/show`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ name: model }),
|
||||
@@ -444,29 +450,102 @@ export class OllamaAdapter implements LLMProvider {
|
||||
return null;
|
||||
}
|
||||
|
||||
// A wol-gate-fronted Ollama endpoint (e.g. 지서버, [[project_wol_gate]]) answers a sleeping
|
||||
// target with an HTML "waking up, refresh in 5s" page instead of proxying — the ollama
|
||||
// client library then tries to JSON.parse that HTML and throws "Unexpected token '<'"
|
||||
// (observed live 2026-08-08, repeatedly, on the first request after any idle period).
|
||||
// Poll a lightweight endpoint first and wait out the wake-up window so the real call only
|
||||
// fires once the target is actually answering as itself.
|
||||
private async _waitIfWakingUp(): Promise<void> {
|
||||
const maxWaitMs = 90_000;
|
||||
const pollIntervalMs = 5_000;
|
||||
const deadline = Date.now() + maxWaitMs;
|
||||
/**
|
||||
* Is `host` answering as the real Ollama (JSON), rather than absent or fronted by a
|
||||
* wol-gate "waking up" page? The gate answers for a sleeping target with HTTP 200 and
|
||||
* an HTML "refresh in 5s" body, so a 200 alone proves nothing — the ollama client then
|
||||
* JSON.parses that HTML and throws "Unexpected token '<'" ([[project_wol_gate]],
|
||||
* observed live 2026-08-08 on the first request after any idle period).
|
||||
*/
|
||||
private async _isAwake(host: string): Promise<boolean> {
|
||||
try {
|
||||
const res = await fetch(`${host}/api/tags`, { signal: AbortSignal.timeout(4_000) });
|
||||
const text = await res.text();
|
||||
return !text.trimStart().startsWith('<');
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Poke the wol-gate so it emits a magic packet. Response body is irrelevant. */
|
||||
private async _sendWake(wakeUrl: string): Promise<void> {
|
||||
try {
|
||||
await fetch(`${wakeUrl}/api/tags`, { signal: AbortSignal.timeout(4_000) });
|
||||
} catch {
|
||||
// The gate answers a sleeping target with HTML (or nothing) — either way the
|
||||
// request itself is what triggers the WOL packet, so failures here are expected.
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Point this call at whichever host can actually serve it.
|
||||
*
|
||||
* With `wake_url` configured, `endpoint` goes straight to the box (지서버) and the gate is
|
||||
* only consulted when direct fails — so the proxy stays out of the hot path. Direct is the
|
||||
* optimization; the gate is the path that is known to work, and it is used as a fallback
|
||||
* rather than merely as a doorbell: 2026-08-09 00:52 the box was demonstrably up and the
|
||||
* gate was proxying real JSON while direct timed out from this host. Polling direct alone
|
||||
* would have stalled the full wake budget and then failed, with a working route sitting
|
||||
* unused the whole time.
|
||||
*/
|
||||
private async _selectHost(): Promise<string> {
|
||||
const direct = this.endpoint;
|
||||
if (await this._isAwake(direct)) return direct; // fast path — no extra hop when up
|
||||
|
||||
const wakeUrl = this._resolveWakeUrl();
|
||||
if (!wakeUrl) {
|
||||
// No separate gate: `endpoint` may itself be gate-fronted, in which case requesting
|
||||
// it is what wakes the target. Wait it out before letting the real call fire.
|
||||
await this._pollUntilAwake(direct, 90_000);
|
||||
return direct;
|
||||
}
|
||||
|
||||
await this._sendWake(wakeUrl);
|
||||
|
||||
// Give direct a short window to come back (a cold box needs ~30s to boot), but check
|
||||
// the gate on the same cadence so a reachable-only-via-gate target is picked up early.
|
||||
const deadline = Date.now() + 90_000;
|
||||
while (Date.now() < deadline) {
|
||||
try {
|
||||
const res = await fetch(`${this.endpoint}/api/tags`, { signal: AbortSignal.timeout(4_000) });
|
||||
const text = await res.text();
|
||||
if (!text.trimStart().startsWith('<')) return; // real (JSON) response — target is up
|
||||
} catch {
|
||||
// Connection refused/timeout could mean "still waking" or "genuinely unreachable" —
|
||||
// either way, keep polling within the budget rather than guessing which.
|
||||
await new Promise((r) => setTimeout(r, 5_000));
|
||||
if (await this._isAwake(direct)) return direct;
|
||||
if (await this._isAwake(wakeUrl)) {
|
||||
console.warn(`[OllamaAdapter:${this.id}] direct ${direct} unreachable; routing via wol-gate ${wakeUrl}`);
|
||||
return wakeUrl;
|
||||
}
|
||||
await new Promise((r) => setTimeout(r, pollIntervalMs));
|
||||
}
|
||||
// Neither route answered within the budget — let the real call proceed and surface
|
||||
// whatever error it hits rather than blocking forever on a target that never wakes.
|
||||
return direct;
|
||||
}
|
||||
|
||||
private async _pollUntilAwake(host: string, budgetMs: number): Promise<void> {
|
||||
const deadline = Date.now() + budgetMs;
|
||||
while (Date.now() < deadline) {
|
||||
await new Promise((r) => setTimeout(r, 5_000));
|
||||
if (await this._isAwake(host)) return;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the host to use and point the client at it for this call. Returns without
|
||||
* touching the client on the common path where direct is already serving.
|
||||
*/
|
||||
private async _waitIfWakingUp(): Promise<void> {
|
||||
const host = await this._selectHost();
|
||||
if (host !== this._activeHost) {
|
||||
this._activeHost = host;
|
||||
this.client = new Ollama({ host });
|
||||
}
|
||||
}
|
||||
|
||||
private _resolveWakeUrl(): string | undefined {
|
||||
try {
|
||||
const cfg: any = getConfig().getConfig();
|
||||
const raw = cfg?.llm?.providers?.[this.id]?.wake_url;
|
||||
return raw ? String(raw).replace(/\/+$/, '') : undefined;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
// Waited the full budget without a real response — let the actual call proceed anyway
|
||||
// and surface whatever error it hits; don't block forever on a target that never wakes.
|
||||
}
|
||||
|
||||
async chat(messages: ChatMessage[], model: string, options?: ChatOptions): Promise<ChatResult> {
|
||||
|
||||
Reference in New Issue
Block a user