feat: models.profiles[<model>].think — 모델별 think 모드 오버라이드
glm-5.3-flash:cloud 실측(2026-09-22): 오케스트레이터 턴(multiAgentActive)에서 think:true로 돌며 라운드당 reasoning 3~6K 토큰 소모(뉴스 요약 턴 최종 호출 5,868토큰, 72초) — gemma4는 같은 턴을 ~100토큰에 처리. completion 중앙값 331 vs 93(3.5배). 플래시인데도 체감이 느린 진범은 속도가 아니라 출력량. - server.ts에 getModelProfileThink 신설(forceToolChoice/fixedNumCtx와 같은 프로파일 패턴) — boolean 또는 high/medium/low - handle-chat.ts의 think 결정에 프로파일 오버라이드 적용(라운드 루프 안에서 effectiveModel 기준으로 판정) - config: glm-5.3-flash:cloud에 think:false 세팅 Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
@@ -127,6 +127,7 @@ export interface HandleChatDeps {
|
||||
goalLikelyNeedsTextInput: (goal: string) => boolean;
|
||||
getModelProfileForceToolChoice: (modelName: string) => boolean;
|
||||
getModelProfileFixedNumCtx: (modelName: string) => number | undefined;
|
||||
getModelProfileThink: (modelName: string) => boolean | 'high' | 'medium' | 'low' | undefined;
|
||||
stripExplicitThinkTags: (text: string) => { cleaned: string; thinking: string };
|
||||
separateThinkingFromContent: (text: string) => { reply: string; thinking: string };
|
||||
isOrchestrationSkillEnabled: () => boolean;
|
||||
@@ -181,6 +182,7 @@ export function createHandleChat(deps: HandleChatDeps) {
|
||||
goalLikelyNeedsTextInput,
|
||||
getModelProfileForceToolChoice,
|
||||
getModelProfileFixedNumCtx,
|
||||
getModelProfileThink,
|
||||
stripExplicitThinkTags,
|
||||
separateThinkingFromContent,
|
||||
isOrchestrationSkillEnabled,
|
||||
@@ -1434,7 +1436,12 @@ async function handleChat(
|
||||
// mid-session reloads.
|
||||
num_ctx: getModelProfileFixedNumCtx(resolvedModelThisRound || getModelForRole('executor')),
|
||||
num_predict: (isCodeAiSession || isProjSession) ? 32768 : (needsLongOutput ? 32768 : 6000),
|
||||
think: primaryThinkMode,
|
||||
// models.profiles[<model>].think로 모델별 오버라이드 가능 (예: glm-5.3-flash는
|
||||
// 오케스트레이터 턴에서도 reasoning을 3~6K 토큰씩 써서 응답이 1분+ — 2026-09-22 실측)
|
||||
think: (() => {
|
||||
const pt = getModelProfileThink(effectiveModel || getModelForRole('executor'));
|
||||
return pt !== undefined ? pt : primaryThinkMode;
|
||||
})(),
|
||||
model: effectiveModel,
|
||||
});
|
||||
|
||||
|
||||
@@ -1824,6 +1824,30 @@ function getModelProfileFixedNumCtx(modelName: string): number | undefined {
|
||||
}
|
||||
|
||||
|
||||
// Per-model thinking override for the chat pipeline's primaryThinkMode. Default behavior:
|
||||
// orchestrator turns (multiAgentActive) think, everything else doesn't. Some models think
|
||||
// too verbosely to be usable there — glm-5.3-flash:cloud measured 2026-09-22 writing 3–6K
|
||||
// reasoning tokens per orchestrator round (12K chars of content-inline reasoning on a news
|
||||
// digest turn, ~72s for the final call) where gemma4 spends ~100 on the same turns. Set
|
||||
// models.profiles[<model>].think = false (or "high"/"medium"/"low") to pin the think mode
|
||||
// for that model; omit the field to keep the default behavior.
|
||||
function getModelProfileThink(modelName: string): boolean | 'high' | 'medium' | 'low' | undefined {
|
||||
const m = String(modelName || '').trim();
|
||||
if (!m) return undefined;
|
||||
try {
|
||||
const raw = getConfig().getConfig() as any;
|
||||
const v = raw?.models?.profiles?.[m]?.think;
|
||||
if (typeof v === 'boolean') return v;
|
||||
if (typeof v === 'string' && ['high', 'medium', 'low'].includes(v.toLowerCase())) {
|
||||
return v.toLowerCase() as 'high' | 'medium' | 'low';
|
||||
}
|
||||
return undefined;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
function resolveWorkspaceFilePath(workspacePath: string, filename: string): string {
|
||||
if (!filename) return '';
|
||||
if (path.isAbsolute(filename)) return filename;
|
||||
@@ -2084,6 +2108,7 @@ const { handleChat, handleCodeChat } = createHandleChat({
|
||||
goalLikelyNeedsTextInput,
|
||||
getModelProfileForceToolChoice,
|
||||
getModelProfileFixedNumCtx,
|
||||
getModelProfileThink,
|
||||
stripExplicitThinkTags,
|
||||
separateThinkingFromContent,
|
||||
isOrchestrationSkillEnabled,
|
||||
|
||||
Reference in New Issue
Block a user