feat: models.profiles[<model>].think — 모델별 think 모드 오버라이드

glm-5.3-flash:cloud 실측(2026-09-22): 오케스트레이터 턴(multiAgentActive)에서
think:true로 돌며 라운드당 reasoning 3~6K 토큰 소모(뉴스 요약 턴 최종 호출
5,868토큰, 72초) — gemma4는 같은 턴을 ~100토큰에 처리. completion 중앙값 331 vs
93(3.5배). 플래시인데도 체감이 느린 진범은 속도가 아니라 출력량.

- server.ts에 getModelProfileThink 신설(forceToolChoice/fixedNumCtx와 같은
  프로파일 패턴) — boolean 또는 high/medium/low
- handle-chat.ts의 think 결정에 프로파일 오버라이드 적용(라운드 루프 안에서
  effectiveModel 기준으로 판정)
- config: glm-5.3-flash:cloud에 think:false 세팅

Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
kim
2026-09-23 01:05:36 +09:00
co-authored by Claude Code
parent f8d1c9f9df
commit b00db7c064
2 changed files with 33 additions and 1 deletions
+8 -1
View File
@@ -127,6 +127,7 @@ export interface HandleChatDeps {
goalLikelyNeedsTextInput: (goal: string) => boolean;
getModelProfileForceToolChoice: (modelName: string) => boolean;
getModelProfileFixedNumCtx: (modelName: string) => number | undefined;
getModelProfileThink: (modelName: string) => boolean | 'high' | 'medium' | 'low' | undefined;
stripExplicitThinkTags: (text: string) => { cleaned: string; thinking: string };
separateThinkingFromContent: (text: string) => { reply: string; thinking: string };
isOrchestrationSkillEnabled: () => boolean;
@@ -181,6 +182,7 @@ export function createHandleChat(deps: HandleChatDeps) {
goalLikelyNeedsTextInput,
getModelProfileForceToolChoice,
getModelProfileFixedNumCtx,
getModelProfileThink,
stripExplicitThinkTags,
separateThinkingFromContent,
isOrchestrationSkillEnabled,
@@ -1434,7 +1436,12 @@ async function handleChat(
// mid-session reloads.
num_ctx: getModelProfileFixedNumCtx(resolvedModelThisRound || getModelForRole('executor')),
num_predict: (isCodeAiSession || isProjSession) ? 32768 : (needsLongOutput ? 32768 : 6000),
think: primaryThinkMode,
// models.profiles[<model>].think로 모델별 오버라이드 가능 (예: glm-5.3-flash는
// 오케스트레이터 턴에서도 reasoning을 3~6K 토큰씩 써서 응답이 1분+ — 2026-09-22 실측)
think: (() => {
const pt = getModelProfileThink(effectiveModel || getModelForRole('executor'));
return pt !== undefined ? pt : primaryThinkMode;
})(),
model: effectiveModel,
});
+25
View File
@@ -1824,6 +1824,30 @@ function getModelProfileFixedNumCtx(modelName: string): number | undefined {
}
// Per-model thinking override for the chat pipeline's primaryThinkMode. Default behavior:
// orchestrator turns (multiAgentActive) think, everything else doesn't. Some models think
// too verbosely to be usable there — glm-5.3-flash:cloud measured 2026-09-22 writing 3–6K
// reasoning tokens per orchestrator round (12K chars of content-inline reasoning on a news
// digest turn, ~72s for the final call) where gemma4 spends ~100 on the same turns. Set
// models.profiles[<model>].think = false (or "high"/"medium"/"low") to pin the think mode
// for that model; omit the field to keep the default behavior.
function getModelProfileThink(modelName: string): boolean | 'high' | 'medium' | 'low' | undefined {
const m = String(modelName || '').trim();
if (!m) return undefined;
try {
const raw = getConfig().getConfig() as any;
const v = raw?.models?.profiles?.[m]?.think;
if (typeof v === 'boolean') return v;
if (typeof v === 'string' && ['high', 'medium', 'low'].includes(v.toLowerCase())) {
return v.toLowerCase() as 'high' | 'medium' | 'low';
}
return undefined;
} catch {
return undefined;
}
}
function resolveWorkspaceFilePath(workspacePath: string, filename: string): string {
if (!filename) return '';
if (path.isAbsolute(filename)) return filename;
@@ -2084,6 +2108,7 @@ const { handleChat, handleCodeChat } = createHandleChat({
goalLikelyNeedsTextInput,
getModelProfileForceToolChoice,
getModelProfileFixedNumCtx,
getModelProfileThink,
stripExplicitThinkTags,
separateThinkingFromContent,
isOrchestrationSkillEnabled,