From 554b66d299dde508541febb0d9fe1c76fa8cdf96 Mon Sep 17 00:00:00 2001 From: kim Date: Sat, 15 Aug 2026 08:25:36 +0900 Subject: [PATCH] =?UTF-8?q?fix:=20=EB=AA=85=EC=8B=9C=EC=A0=81=20think:fals?= =?UTF-8?q?e=EB=8A=94=20=EC=9E=AC=EC=8B=9C=EB=8F=84=20=EC=82=AC=EB=8B=A4?= =?UTF-8?q?=EB=A6=AC=EB=A5=BC=20=EC=98=A4=EB=A5=B4=EC=A7=80=20=EC=95=8A?= =?UTF-8?q?=EB=8A=94=EB=8B=A4?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 지서버를 muse-glimmer:latest nothink 모드로 붙이면서 다시 넣는다. (86063d3에 함께 있었으나 그 커밋이 통째로 되돌려졌고, 되돌린 사유였던 에러는 서버 로그상 8시간 앞선 별건으로 확인됐다. UI 드롭다운 변경은 제외하고 think 사다리 부분만 복원한다.) 사다리는 glm-5.1:cloud처럼 think를 생략하면 생각만 하고 content를 안 내놓는 모델을 건지려고 만든 것이다. 하지만 false는 성격이 다르다. "이 모델은 생각에 예산을 낭비한다"는 운영자의 결정이라, 빈 응답 한 번에 thinking을 도로 켜면 그 결정이 조용히 뒤집힌다. 08-12 muse-glimmer 실측: 생각비중 85%, 체감 3.1 tok/s. think:false로 20.9 tok/s (7배)에 답변은 오히려 길어지고 2000토큰 캡 잘림도 사라졌으며 20문항 빈 답변 0건이었다. 사다리가 이걸 되돌리면 의도한 7배가 말없이 느린 경로로 돌아간다. 이제 false는 [false, undefined]까지만 — 모델 기본값 폴백은 남겨서 빈 응답이 막다른 길이 되지는 않게 한다. 테스트 목적으로 모듈 레벨 export로 추출(346개 통과). Co-Authored-By: Claude Opus 5 --- src/providers/ollama-adapter.ts | 44 ++++++++++++++++++++------- tests/think-candidates.test.ts | 53 +++++++++++++++++++++++++++++++++ 2 files changed, 87 insertions(+), 10 deletions(-) create mode 100644 tests/think-candidates.test.ts diff --git a/src/providers/ollama-adapter.ts b/src/providers/ollama-adapter.ts index 3e875a3..2e7b2ce 100644 --- a/src/providers/ollama-adapter.ts +++ b/src/providers/ollama-adapter.ts @@ -343,6 +343,39 @@ function repairJson(input: string): string { return s; } +/** + * The ordered `think` values to retry with when a call comes back empty. + * + * The retry ladder exists because some models (e.g. glm-5.1:cloud) default to thinking-only output + * when `think` is omitted, emitting reasoning and no content — nudging them to an explicit mode + * gets a real answer out. + * + * `think: false` is the one request that must not climb that ladder. It is an operator decision + * ("this model wastes its budget thinking"), so silently re-enabling thinking on an empty response + * would undo exactly what was asked for — and for a model whose thinking ratio is 85%, that turns a + * deliberate 7x speedup back into the slow path without a word. Explicit `false` therefore falls + * back only to the model's own default, and only as a last resort so an empty reply is not a dead + * end. Measured 2026-08-12 on muse-glimmer: think:false produced 0 empty answers in 20 questions, + * so this fallback should stay cold in practice. + */ +export function buildThinkCandidates(requested?: boolean | 'high' | 'medium' | 'low') { + const candidates: Array = []; + const push = (v: boolean | 'high' | 'medium' | 'low' | undefined) => { + if (!candidates.some(x => x === v)) candidates.push(v); + }; + if (requested === false) { + push(false); + push(undefined); + return candidates; + } + push(requested); + if (requested !== 'low') push('low'); + push(undefined); + if (requested !== true) push(true); + push('medium'); + return candidates; +} + export class OllamaAdapter implements LLMProvider { readonly id: 'ollama' | 'ollama_local'; private client: Ollama; @@ -864,15 +897,6 @@ export class OllamaAdapter implements LLMProvider { } private buildThinkCandidates(requested?: boolean | 'high' | 'medium' | 'low') { - const candidates: Array = []; - const push = (v: boolean | 'high' | 'medium' | 'low' | undefined) => { - if (!candidates.some(x => x === v)) candidates.push(v); - }; - push(requested); - if (requested !== 'low') push('low'); - push(undefined); - if (requested !== true) push(true); - push('medium'); - return candidates; + return buildThinkCandidates(requested); } } diff --git a/tests/think-candidates.test.ts b/tests/think-candidates.test.ts new file mode 100644 index 0000000..034ccdc --- /dev/null +++ b/tests/think-candidates.test.ts @@ -0,0 +1,53 @@ +/** + * buildThinkCandidates — think 재시도 사다리 + * + * 빈 응답이 오면 다른 think 값으로 재시도한다. glm-5.1:cloud처럼 think를 생략하면 생각만 하고 + * content를 안 내놓는 모델을 건지려고 만든 장치다. + * + * 문제는 `think:false`였다. 2026-08-12 muse-glimmer(생각비중 85%, 체감 3.1 tok/s) 측정 후 + * 생각을 끄기로 했는데, 사다리가 빈 응답 한 번에 thinking을 도로 켜버리면 그 결정이 조용히 + * 뒤집힌다. 명시적 false는 모델 기본값까지만 물러선다. + */ + +import { test, describe } from 'node:test'; +import assert from 'node:assert/strict'; +import { buildThinkCandidates } from '../src/providers/ollama-adapter'; + +describe('think:false는 사다리를 오르지 않는다', () => { + test('false → [false, undefined]까지만', () => { + assert.deepEqual(buildThinkCandidates(false), [false, undefined]); + }); + + test('생각을 켜는 값이 하나도 없다', () => { + const c = buildThinkCandidates(false); + for (const v of [true, 'low', 'medium', 'high']) { + assert.ok(!c.includes(v as any), `${v}가 포함되면 안 됨`); + } + }); + + test('빈 응답이 막다른 길이 되지 않게 기본값 폴백은 남긴다', () => { + assert.ok(buildThinkCandidates(false).includes(undefined)); + }); +}); + +describe('나머지 경로는 그대로', () => { + test('요청값이 맨 앞에 온다', () => { + assert.equal(buildThinkCandidates(true)[0], true); + assert.equal(buildThinkCandidates('low')[0], 'low'); + assert.equal(buildThinkCandidates(undefined)[0], undefined); + }); + + test('true 요청은 여전히 여러 모드로 재시도한다', () => { + const c = buildThinkCandidates(true); + assert.ok(c.includes('low')); + assert.ok(c.includes(undefined)); + assert.ok(c.includes('medium')); + }); + + test('중복이 없다', () => { + for (const req of [undefined, true, false, 'low', 'medium', 'high'] as const) { + const c = buildThinkCandidates(req); + assert.equal(new Set(c).size, c.length, `중복 발생: ${JSON.stringify(c)}`); + } + }); +});