fix: 컨텍스트 조회 실패를 사실처럼 캐시해 fixedNumCtx가 8K로 깎이던 문제
사용자 제보: "지서버 모델 처음 로딩하면 컨텍스트가 항상 8킬로인데, 웹을 리로드하면 256킬로가 된다." getModelCtx()가 /api/show 실패 시 8192를 반환하면서 그 값을 캐시에 넣었다. 그리고 _resolveCtx가 요청값을 native로 깎는다: if (requested && requested > 0) return Math.min(requested, native); → Math.min(262144, 8192) = 8192 지서버가 자고 있을 때 첫 조회가 실패하면 8192가 캐시에 박히고, 캐시는 updateEndpoint()에서 엔드포인트 문자열이 바뀔 때만 비워진다. 그래서 한 번 오염되면 계속 8K로 돌았고, wol-gate 직결/게이트 전환이 일어나 캐시가 비워질 때 비로소 정상으로 돌아왔다 — 리로드하면 고쳐지는 것처럼 보인 이유가 이것. 추측한 값을 사실처럼 캐시하고, 그걸로 사용자 설정을 덮어쓴 게 버그다. 두 가지로 나눠 고쳤다: - getModelCtxInfo()가 native와 함께 known(진짜 조회 결과인지)을 반환한다. **조회에 성공한 값만 캐시한다** — 지금 모델에 못 닿는다는 사실은 그 모델의 컨텍스트 창에 대해 아무것도 말해주지 않으므로, 다음 호출에서 다시 조회한다. Modelfile num_ctx는 선언된 실제 값이라 캐시 대상으로 유지 - 명시적으로 요청된 창(models.profiles의 fixedNumCtx)은 **ceiling을 실제로 알 때만** 깎는다. 폴백 추측값으로 깎는 게 262144를 8192로 만든 경로다 검증: 응답 없는 엔드포인트로 재현 → 수정 전 8192, 수정 후 262144 유지. 실제 지서버 조회는 262144 정상, 동적 사이징의 8192 폴백은 그대로 동작. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -374,7 +374,25 @@ export class OllamaAdapter implements LLMProvider {
|
||||
* over a Modelfile's `PARAMETER num_ctx` (which is just a conservative runtime default
|
||||
* the caller is allowed to exceed up to the native max). */
|
||||
private async getModelCtx(model: string): Promise<number> {
|
||||
if (this._ctxCache.has(model)) return this._ctxCache.get(model)!;
|
||||
return (await this.getModelCtxInfo(model)).native;
|
||||
}
|
||||
|
||||
/**
|
||||
* Same lookup, but says whether the number is REAL or a guess.
|
||||
*
|
||||
* 2026-08-11 bug this fixes: a failed probe was cached as if it were a fact. When 지서버 was
|
||||
* asleep the /api/show call failed, 8192 went into the cache, and every later call clamped the
|
||||
* configured fixedNumCtx (262144) down to it — `Math.min(262144, 8192)`. The cache is only
|
||||
* cleared when the endpoint string changes, so the model stayed at 8K until a wol-gate
|
||||
* direct/gate flip happened to clear it (which is why reloading the page appeared to "fix" it).
|
||||
*
|
||||
* Two rules follow, and both matter:
|
||||
* 1. Only cache a value that came back from a successful lookup. A guess must be re-probed.
|
||||
* 2. A guess must never clamp an explicitly requested window — see _resolveCtx.
|
||||
*/
|
||||
private async getModelCtxInfo(model: string): Promise<{ native: number; known: boolean }> {
|
||||
const cached = this._ctxCache.get(model);
|
||||
if (cached !== undefined) return { native: cached, known: true };
|
||||
try {
|
||||
// _activeHost, not endpoint: when direct is down the call itself is going through the
|
||||
// gate, so the context lookup that sizes num_ctx has to follow the same route.
|
||||
@@ -389,20 +407,22 @@ export class OllamaAdapter implements LLMProvider {
|
||||
for (const key of Object.keys(mi)) {
|
||||
if (key.endsWith('.context_length') && typeof mi[key] === 'number') {
|
||||
this._ctxCache.set(model, mi[key]);
|
||||
return mi[key];
|
||||
return { native: mi[key], known: true };
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {}
|
||||
// No native context_length surfaced. Cloud-routed models (`*:cloud`) don't expose
|
||||
// it via /api/show but typically have large windows — assume a generous default so
|
||||
// dynamic sizing doesn't needlessly clamp them.
|
||||
if (/:cloud$/i.test(model)) { this._ctxCache.set(model, 131072); return 131072; }
|
||||
// Otherwise fall back to the Modelfile num_ctx, then 8192.
|
||||
// Everything below is a guess, so none of it is cached — the next call probes again. The
|
||||
// model being unreachable right now says nothing about its context window.
|
||||
//
|
||||
// Cloud-routed models (`*:cloud`) don't expose context_length via /api/show but typically
|
||||
// have large windows, so assume a generous default rather than clamping them.
|
||||
if (/:cloud$/i.test(model)) return { native: 131072, known: false };
|
||||
// A Modelfile num_ctx is a real declared value, just a conservative runtime default rather
|
||||
// than the native ceiling — good enough to cache and to size dynamically against.
|
||||
const manifestCtx = this._readManifestNumCtx(model);
|
||||
if (manifestCtx) { this._ctxCache.set(model, manifestCtx); return manifestCtx; }
|
||||
this._ctxCache.set(model, 8192);
|
||||
return 8192;
|
||||
if (manifestCtx) { this._ctxCache.set(model, manifestCtx); return { native: manifestCtx, known: true }; }
|
||||
return { native: 8192, known: false };
|
||||
}
|
||||
|
||||
/** Rough token estimate from message chars (~4 chars/token). The optional `extraChars`
|
||||
@@ -423,8 +443,11 @@ export class OllamaAdapter implements LLMProvider {
|
||||
* native context_length — so content is never silently truncated. A 1.3x margin
|
||||
* absorbs chat-template overhead the char estimate can't see. */
|
||||
private async _resolveCtx(messages: any[], model: string, requested: number | undefined, reserve: number, extraChars = 0): Promise<number> {
|
||||
const native = await this.getModelCtx(model);
|
||||
if (requested && requested > 0) return Math.min(requested, native);
|
||||
const { native, known } = await this.getModelCtxInfo(model);
|
||||
// An explicitly requested window is the caller's assertion (models.profiles fixedNumCtx).
|
||||
// Clamp it to the native ceiling only when that ceiling is actually known — clamping to a
|
||||
// fallback guess is how 262144 silently became 8192.
|
||||
if (requested && requested > 0) return known ? Math.min(requested, native) : requested;
|
||||
const want = Math.ceil(this._estPromptTokens(messages, extraChars) * 1.3) + reserve;
|
||||
const stepped = Math.pow(2, Math.ceil(Math.log2(Math.max(want, 1))));
|
||||
return Math.max(8192, Math.min(native, stepped));
|
||||
|
||||
Reference in New Issue
Block a user