Read num_ctx from Ollama manifest params blob for accurate context window
/api/show가 cloud 모델의 PARAMETER num_ctx를 반환 안 하는 문제 해결. 매니페스트 → params 레이어 블롭 직접 파싱해서 num_ctx 우선 사용. kimi-k2.6:cloud 1M으로 Modelfile 수정 후 정확히 표시됨. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -12938,6 +12938,29 @@ app.get('/api/ollama/models', async (_req, res) => {
|
||||
});
|
||||
|
||||
// GET /api/model-context?model=<name> — return context window size for a model
|
||||
/** Read num_ctx from the Ollama manifest's params blob (for cloud/custom models
|
||||
* where /api/show doesn't surface PARAMETER num_ctx in the parameters field). */
|
||||
function _readOllamaManifestNumCtx(modelName: string): number | null {
|
||||
try {
|
||||
const [nameTag, tag = 'latest'] = modelName.split(':');
|
||||
const manifestDir = '/usr/share/ollama/.ollama/models/manifests/registry.ollama.ai/library';
|
||||
const manifestPath = `${manifestDir}/${nameTag}/${tag}`;
|
||||
if (!fs.existsSync(manifestPath)) return null;
|
||||
const manifest = JSON.parse(fs.readFileSync(manifestPath, 'utf8'));
|
||||
const blobsDir = '/usr/share/ollama/.ollama/models/blobs';
|
||||
for (const layer of (manifest.layers || [])) {
|
||||
if (layer.mediaType === 'application/vnd.ollama.image.params') {
|
||||
const blobPath = `${blobsDir}/${layer.digest.replace(':', '-')}`;
|
||||
if (fs.existsSync(blobPath)) {
|
||||
const params = JSON.parse(fs.readFileSync(blobPath, 'utf8'));
|
||||
if (typeof params.num_ctx === 'number') return params.num_ctx;
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {}
|
||||
return null;
|
||||
}
|
||||
|
||||
app.get('/api/model-context', async (req, res) => {
|
||||
const model = String(req.query.model || '').trim().toLowerCase();
|
||||
// Claude models: 200K (no local metadata)
|
||||
@@ -12945,7 +12968,11 @@ app.get('/api/model-context', async (req, res) => {
|
||||
res.json({ contextWindow: 200000 });
|
||||
return;
|
||||
}
|
||||
// For Ollama models, query /api/show directly
|
||||
// Check manifest params blob first (catches PARAMETER num_ctx set via Modelfile)
|
||||
const manifestCtx = _readOllamaManifestNumCtx(model);
|
||||
if (manifestCtx) { res.json({ contextWindow: manifestCtx }); return; }
|
||||
|
||||
// Fall back to /api/show model_info
|
||||
try {
|
||||
const ollamaEndpoint = (getConfig().getConfig() as any).ollama?.endpoint || 'http://localhost:11434';
|
||||
const response = await fetch(`${ollamaEndpoint}/api/show`, {
|
||||
|
||||
@@ -226,6 +226,9 @@ export class OllamaAdapter implements LLMProvider {
|
||||
/** Return the model's native context_length from /api/show, cached per model. */
|
||||
private async getModelCtx(model: string): Promise<number> {
|
||||
if (this._ctxCache.has(model)) return this._ctxCache.get(model)!;
|
||||
// Check manifest params blob first (PARAMETER num_ctx via Modelfile)
|
||||
const manifestCtx = this._readManifestNumCtx(model);
|
||||
if (manifestCtx) { this._ctxCache.set(model, manifestCtx); return manifestCtx; }
|
||||
try {
|
||||
const res = await fetch(`${this.endpoint}/api/show`, {
|
||||
method: 'POST',
|
||||
@@ -241,15 +244,34 @@ export class OllamaAdapter implements LLMProvider {
|
||||
return mi[key];
|
||||
}
|
||||
}
|
||||
// Fallback: num_ctx in parameters string
|
||||
const pm = String(data?.parameters || '').match(/\bnum_ctx\s+(\d+)/);
|
||||
if (pm) { const v = parseInt(pm[1], 10); this._ctxCache.set(model, v); return v; }
|
||||
}
|
||||
} catch {}
|
||||
this._ctxCache.set(model, 8192); // unknown → keep safe default
|
||||
this._ctxCache.set(model, 8192);
|
||||
return 8192;
|
||||
}
|
||||
|
||||
private _readManifestNumCtx(model: string): number | null {
|
||||
try {
|
||||
const [name, tag = 'latest'] = model.split(':');
|
||||
const manifestPath = `/usr/share/ollama/.ollama/models/manifests/registry.ollama.ai/library/${name}/${tag}`;
|
||||
if (!fs.existsSync(manifestPath)) return null;
|
||||
const manifest = JSON.parse(fs.readFileSync(manifestPath, 'utf8'));
|
||||
const blobsDir = '/usr/share/ollama/.ollama/models/blobs';
|
||||
for (const layer of (manifest.layers || [])) {
|
||||
if (layer.mediaType === 'application/vnd.ollama.image.params') {
|
||||
const blobPath = `${blobsDir}/${layer.digest.replace(':', '-')}`;
|
||||
if (fs.existsSync(blobPath)) {
|
||||
const params = JSON.parse(fs.readFileSync(blobPath, 'utf8'));
|
||||
if (typeof params.num_ctx === 'number') return params.num_ctx;
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {}
|
||||
return null;
|
||||
}
|
||||
|
||||
async chat(messages: ChatMessage[], model: string, options?: ChatOptions): Promise<ChatResult> {
|
||||
const workspacePath = getConfig().getConfig().workspace?.path;
|
||||
// Ollama requires the last message to be user or tool — strip trailing assistant messages
|
||||
|
||||
Reference in New Issue
Block a user