From 5eb47ca08b4b730f171205fb4c901b53bf422549 Mon Sep 17 00:00:00 2001 From: kim Date: Wed, 29 Jul 2026 16:16:59 +0900 Subject: [PATCH] =?UTF-8?q?v4.3.24:=20=EA=B2=80=EC=83=89=20=EC=98=88?= =?UTF-8?q?=EC=82=B0=20=ED=8C=90=EB=8B=A8=EC=9D=84=20search-budget.ts?= =?UTF-8?q?=EB=A1=9C=20=EB=B6=84=EB=A6=AC=20+=20=ED=85=8C=EC=8A=A4?= =?UTF-8?q?=ED=8A=B8=2013=EA=B0=9C?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 라운드 루프(1,938줄) 안의 순수 판단 로직 추출. 루프 자체는 옮기지 않음 — 판단 근거는 아래에 정리. - src/gateway/chat/search-budget.ts (신규) — evaluateSearchBudget()가 (이전 쿼리들, 현재 쿼리, 차단횟수) → (허용여부, 사유, 도구차단, 하드캡도달) 을 반환하는 순수 함수. handle-chat.ts 2,936 → 2,908줄 - tests/search-budget.test.ts — 13개. 캡이 "반복에는 물리고 팬아웃에는 안 물린다"는 양방향을 모두 고정 테스트가 찾은 약점 — 한 단어 우회: "어떤 미등장 토큰이라도 있으면 새 주제로 본다"는 규칙이 팬아웃(도시 8곳 날씨 확인 등)을 보호하려고 일부러 느슨한데, 그 탓에 "우크라이나 전황" → "우크라이나 전황 다시"가 완전히 새로운 주제로 통과해 재검색 루프를 그대로 허용했음. 이 모듈이 막으려던 바로 그 패턴. 규칙 자체를 조이면 정당한 쿼리까지 막히므로, 변별력 없는 단어(다시/추가/자세히/더/또/again/more/detail)를 불용어에 추가하는 쪽으로 해결. 라운드 루프를 통째로 옮기지 않은 이유: 루프는 1,938줄이면서 외부 가변 변수 22개(allThinking, toolSkipForcedRetries, toolsDisabledForRestOfTurn, fileOpOwner 등)를 변경하고, 본질적으로 I/O (모델 스트리밍·도구 실행·SSE 전송)라 순수 함수가 될 수 없음. 옮기려면 22개를 상태 객체로 묶어 넘겨야 하는데 회귀 위험만 크고 테스트 가능성은 늘지 않음 — 줄 수만 다른 파일로 이동할 뿐임. 오늘 tool-scope/system-prompt 추출이 값어치 있었던 건 순수 함수가 되어 테스트가 붙었기 때문이지 줄 수가 줄어서가 아니므로, 같은 기준으로 루프 안의 '판단'만 뽑는 방향을 택함. 전체 79개 테스트 통과. 배포 후 뉴스 요청에서 news_search 8회 팬아웃 정상 확인. Co-Authored-By: Claude Opus 5 --- src/gateway/chat/handle-chat.ts | 46 +++--------- src/gateway/chat/search-budget.ts | 98 ++++++++++++++++++++++++ tests/search-budget.test.ts | 120 ++++++++++++++++++++++++++++++ 3 files changed, 227 insertions(+), 37 deletions(-) create mode 100644 src/gateway/chat/search-budget.ts create mode 100644 tests/search-budget.test.ts diff --git a/src/gateway/chat/handle-chat.ts b/src/gateway/chat/handle-chat.ts index 6bbea16..474f12c 100644 --- a/src/gateway/chat/handle-chat.ts +++ b/src/gateway/chat/handle-chat.ts @@ -72,6 +72,7 @@ import { createBrowserDesktopAdvisors, type BrowserDesktopAdvisorCtx } from './b import type { SkillsManager } from '../skills-manager'; import { selectToolsForTurn, bootAllowedTools, codeAiBlockedTools } from './tool-scope'; import { buildChatSystemPrompt } from './system-prompt'; +import { evaluateSearchBudget } from './search-budget'; const MAX_TOOL_ROUNDS = 50; type ExecutionMode = 'interactive' | 'background_task' | 'heartbeat' | 'cron'; @@ -414,17 +415,6 @@ async function handleChat( // web_search/web_fetch query this turn. This lets legitimate fan-out (e.g. checking // weather for 8 different cities) extend past the soft cap, while still blocking // redundant re-searches of the same topic (the original runaway-loop failure mode). - const SEARCH_BUDGET_STOPWORDS = new Set([ - '오늘', '현재', '지금', '최고', '최저', '기온', '날씨', '뉴스', '헤드라인', - '국내', '국제', '알려줘', '검색', '정보', '오늘의', 'today', 'news', 'weather', - ]); - const extractSearchQueryTokens = (q: string): Set => new Set( - String(q || '') - .toLowerCase() - .replace(/[^\p{L}\p{N}\s]/gu, ' ') - .split(/\s+/) - .filter((w) => w.length >= 2 && !SEARCH_BUDGET_STOPWORDS.has(w)), - ); const orchestrationState = new OrchestrationTriggerState(); const orchestrationLog: string[] = []; const orchestrationStats = getOrchestrationSessionStats(sessionId); @@ -2656,34 +2646,16 @@ async function handleChat( // the same "today's news" or "GPT-5.6 pricing" query 7-50+ times) is blocked as before. // A hard cap still applies regardless of distinctness, so extension can't run away either. if (toolName === 'web_search' || toolName === 'web_fetch') { - const SEARCH_SOFT_CAP = 5; - const SEARCH_HARD_CAP = 12; const priorSearchCalls = allToolResults.filter((r) => r.name === 'web_search' || r.name === 'web_fetch'); - const searchFetchCallsSoFar = priorSearchCalls.length; - let allowAsNewTopic = false; - if (searchFetchCallsSoFar >= SEARCH_SOFT_CAP && searchFetchCallsSoFar < SEARCH_HARD_CAP) { - const currentTokens = extractSearchQueryTokens(String(toolArgs?.query || toolArgs?.url || '')); - const seenTokens = new Set(); - for (const r of priorSearchCalls) { - for (const t of extractSearchQueryTokens(String((r.args as any)?.query || (r.args as any)?.url || ''))) seenTokens.add(t); - } - allowAsNewTopic = [...currentTokens].some((t) => !seenTokens.has(t)); - } - if (searchFetchCallsSoFar >= SEARCH_SOFT_CAP && !allowAsNewTopic) { + const _budget = evaluateSearchBudget({ + priorQueries: priorSearchCalls.map((r) => String((r.args as any)?.query || (r.args as any)?.url || '')), + currentQuery: String(toolArgs?.query || toolArgs?.url || ''), + priorBlockedCount: searchBudgetBlockedCount, + }); + if (!_budget.allowed) { searchBudgetBlockedCount++; - const secondPlusStrike = searchBudgetBlockedCount >= 2; - if (secondPlusStrike) toolsDisabledForRestOfTurn = true; - const hitHardCap = searchFetchCallsSoFar >= SEARCH_HARD_CAP; - const _blockedSearch: ToolResult = { - name: toolName, - args: toolArgs, - result: hitHardCap - ? `[BLOCKED] Search hard cap reached (${searchFetchCallsSoFar} calls, max ${SEARCH_HARD_CAP}). Write your final text answer now using whatever results you already have; if that's not enough, say so explicitly.` - : secondPlusStrike - ? `[BLOCKED] Search budget exhausted (${searchFetchCallsSoFar} calls) and you already ignored one stop instruction. Tools are now disabled for the rest of this turn — no more calls of any kind will run. Write your final text answer now using whatever results you already have; if that's not enough, say so explicitly.` - : `[BLOCKED] Search budget exhausted (${searchFetchCallsSoFar} web_search/web_fetch calls already made this turn, soft cap ${SEARCH_SOFT_CAP}) and this query repeats a topic you already searched. Stop re-searching the same thing and answer now with what you already have. (New, genuinely distinct topics — e.g. a different city or subject — are still allowed up to ${SEARCH_HARD_CAP} calls total.)`, - error: true, - }; + if (_budget.disableToolsForRestOfTurn) toolsDisabledForRestOfTurn = true; + const _blockedSearch: ToolResult = { name: toolName, args: toolArgs, result: _budget.blockedReason, error: true }; allToolResults.push(_blockedSearch); sendSSE('tool_result', { action: toolName, result: _blockedSearch.result, error: true, stepNum: allToolResults.length }); messages.push({ role: 'tool', name: toolName, tool_name: toolName, tool_call_id: toolCallId || undefined, content: _blockedSearch.result }); diff --git a/src/gateway/chat/search-budget.ts b/src/gateway/chat/search-budget.ts new file mode 100644 index 0000000..9850b9d --- /dev/null +++ b/src/gateway/chat/search-budget.ts @@ -0,0 +1,98 @@ +/** + * search-budget.ts + * + * Decides whether one more web_search / web_fetch call is allowed this turn. + * + * Exists because of a real runaway: kimi-k2.6 was observed issuing 7+ (once 50+) searches for a + * single broad request like "오늘 국제 뉴스 정리", walking topic by topic (Ukraine, Gaza, + * tariffs, …) instead of doing a couple of broad searches. Each call costs latency and tokens + * and the answer does not improve. + * + * The tension this encodes: a naive "max N calls" cap also kills legitimate fan-out — checking + * weather for 8 different cities is 8 genuinely distinct searches, not a loop. So past the soft + * cap a call is still allowed if its query introduces a topic token nobody has searched yet, + * with a hard cap as the backstop so extension cannot run away either. + * + * Extracted from handleChat() on 2026-07-29. The loop it lived in is I/O-bound and cannot be + * made pure, but this decision can — and a decision is where the bugs actually are. + */ + +/** + * Words too generic to distinguish one search from another. Without these, "오늘 서울 날씨" and + * "오늘 부산 날씨" would look distinct only by city (correct), but "오늘 뉴스" and "현재 뉴스" + * would also look distinct (wrong) and the cap would never bite. + */ +export const SEARCH_BUDGET_STOPWORDS = new Set([ + '오늘', '현재', '지금', '최고', '최저', '기온', '날씨', '뉴스', '헤드라인', + '국내', '국제', '알려줘', '검색', '정보', '오늘의', 'today', 'news', 'weather', + // 2026-07-29: added when the extraction's tests showed a one-word escape hatch. The + // "any unseen token means a new topic" rule is deliberately permissive to protect fan-out, + // but that also meant "우크라이나 전황" → "우크라이나 전황 다시" read as a brand-new topic and + // sailed past the cap — the exact re-search loop this module exists to stop. These words + // never distinguish one search from another, so they belong here rather than tightening the + // rule itself (which would start blocking legitimate distinct queries). + '다시', '추가', '자세히', '더', '또', 'again', 'more', 'detail', 'details', +]); + +export const SEARCH_SOFT_CAP = 5; +export const SEARCH_HARD_CAP = 12; + +/** Content words of a query, lowercased, punctuation stripped, stopwords removed. */ +export function extractSearchQueryTokens(q: string): Set { + return new Set( + String(q || '') + .toLowerCase() + .replace(/[^\p{L}\p{N}\s]/gu, ' ') + .split(/\s+/) + .filter((w) => w.length >= 2 && !SEARCH_BUDGET_STOPWORDS.has(w)), + ); +} + +export interface SearchBudgetDecision { + /** false → the call must be replaced with a [BLOCKED] tool result. */ + allowed: boolean; + /** Message handed back to the model in place of results. Empty when allowed. */ + blockedReason: string; + /** True once the model has ignored a stop instruction — caller disables tools for the turn. */ + disableToolsForRestOfTurn: boolean; + /** Whether the hard cap (not just the soft cap) was the thing that stopped it. */ + hitHardCap: boolean; +} + +export interface SearchBudgetInput { + /** Queries/urls of the web_search + web_fetch calls already made THIS turn, in order. */ + priorQueries: string[]; + /** query (or url) of the call being attempted now. */ + currentQuery: string; + /** How many times this turn a search has already been blocked. 2+ means tools get cut off. */ + priorBlockedCount: number; +} + +export function evaluateSearchBudget(input: SearchBudgetInput): SearchBudgetDecision { + const callsSoFar = input.priorQueries.length; + const ok: SearchBudgetDecision = { allowed: true, blockedReason: '', disableToolsForRestOfTurn: false, hitHardCap: false }; + + if (callsSoFar < SEARCH_SOFT_CAP) return ok; + + // Between soft and hard cap a genuinely new topic still gets through. + if (callsSoFar < SEARCH_HARD_CAP) { + const seen = new Set(); + for (const q of input.priorQueries) { + for (const t of extractSearchQueryTokens(q)) seen.add(t); + } + const introducesNewTopic = [...extractSearchQueryTokens(input.currentQuery)].some(t => !seen.has(t)); + if (introducesNewTopic) return ok; + } + + const blockedCount = input.priorBlockedCount + 1; + const secondPlusStrike = blockedCount >= 2; + const hitHardCap = callsSoFar >= SEARCH_HARD_CAP; + + const blockedReason = hitHardCap + ? `[BLOCKED] Search hard cap reached (${callsSoFar} calls, max ${SEARCH_HARD_CAP}). Write your final text answer now using whatever results you already have; if that's not enough, say so explicitly.` + : secondPlusStrike + ? `[BLOCKED] Search budget exhausted (${callsSoFar} calls) and you already ignored one stop instruction. Tools are now disabled for the rest of this turn — no more calls of any kind will run. Write your final text answer now using whatever results you already have; if that's not enough, say so explicitly.` + : `[BLOCKED] Search budget exhausted (${callsSoFar} web_search/web_fetch calls already made this turn, soft cap ${SEARCH_SOFT_CAP}) and this query repeats a topic you already searched. Stop re-searching the same thing and answer now with what you already have. (New, genuinely distinct topics — e.g. a different city or subject — are still allowed up to ${SEARCH_HARD_CAP} calls total.)`; + + return { allowed: false, blockedReason, disableToolsForRestOfTurn: secondPlusStrike, hitHardCap }; +} diff --git a/tests/search-budget.test.ts b/tests/search-budget.test.ts new file mode 100644 index 0000000..a591ec9 --- /dev/null +++ b/tests/search-budget.test.ts @@ -0,0 +1,120 @@ +/** + * search-budget.test.ts + * + * The runaway this guards against was real: kimi-k2.6 issuing 7+ (once 50+) searches for one + * broad request, walking topic by topic instead of doing a couple of broad searches. + * + * The hard part is that the obvious fix — a flat call cap — breaks legitimate fan-out. Checking + * weather for 8 cities is 8 distinct searches, not a loop. So both directions need pinning: + * the cap must bite on repetition AND must not bite on genuine breadth. A regression in either + * direction is silent (either searches quietly stop working, or the runaway comes back). + */ + +import { test, describe } from 'node:test'; +import assert from 'node:assert/strict'; +import { + evaluateSearchBudget, + extractSearchQueryTokens, + SEARCH_SOFT_CAP, + SEARCH_HARD_CAP, +} from '../src/gateway/chat/search-budget'; + +const evalWith = (priorQueries: string[], currentQuery: string, priorBlockedCount = 0) => + evaluateSearchBudget({ priorQueries, currentQuery, priorBlockedCount }); + +describe('extractSearchQueryTokens', () => { + test('불용어를 걸러 실제 주제어만 남긴다', () => { + const t = extractSearchQueryTokens('오늘 서울 날씨 알려줘'); + assert.ok(t.has('서울')); + for (const stop of ['오늘', '날씨', '알려줘']) assert.ok(!t.has(stop), `${stop} 가 남음`); + }); + + test('대소문자/문장부호를 정규화한다', () => { + const t = extractSearchQueryTokens('RTX 5090, PCIe 5.0!'); + assert.ok(t.has('rtx')); + assert.ok(t.has('5090')); + }); + + test('한 글자 토큰은 버린다 — 변별력이 없다', () => { + assert.ok(!extractSearchQueryTokens('a b 서 울').has('a')); + }); +}); + +describe('소프트캡 이하 — 자유롭게 허용', () => { + test('0회~4회까지는 무조건 통과', () => { + for (let n = 0; n < SEARCH_SOFT_CAP; n++) { + const prior = Array.from({ length: n }, (_, i) => `쿼리${i}`); + assert.equal(evalWith(prior, '쿼리0').allowed, true, `${n}회째에서 막힘`); + } + }); +}); + +describe('소프트캡 초과 — 반복은 막고 새 주제는 통과', () => { + const five = ['우크라이나 전황', '가자 지구', '관세 협상', '금리 인상', '유가 동향']; + + test('이미 검색한 주제를 또 검색하면 막힌다 (원래의 폭주 패턴)', () => { + const d = evalWith(five, '우크라이나 전황 다시'); + assert.equal(d.allowed, false); + assert.match(d.blockedReason, /\[BLOCKED\]/); + assert.match(d.blockedReason, /repeats a topic/); + }); + + test('새로운 주제는 소프트캡을 넘겨도 통과한다 — 정당한 팬아웃 보호', () => { + // 도시 8곳 날씨 확인 같은 경우가 여기 해당한다. + assert.equal(evalWith(five, '삿포로 적설량').allowed, true); + }); + + test('불용어만 다른 쿼리는 새 주제가 아니다', () => { + // "오늘 뉴스" vs "현재 뉴스" — 이게 통과하면 캡이 사실상 무력화된다. + const prior = ['오늘 뉴스', '현재 뉴스', '지금 뉴스', '오늘의 뉴스', '뉴스 알려줘']; + assert.equal(evalWith(prior, '지금 뉴스 알려줘').allowed, false); + }); +}); + +describe('하드캡 — 새 주제라도 결국 멈춘다', () => { + const twelve = Array.from({ length: SEARCH_HARD_CAP }, (_, i) => `고유주제${i}`); + + test('12회에 도달하면 완전히 새로운 주제도 막힌다', () => { + const d = evalWith(twelve, '완전히새로운주제'); + assert.equal(d.allowed, false); + assert.equal(d.hitHardCap, true); + assert.match(d.blockedReason, /hard cap/i); + }); + + test('하드캡 직전(11회)에는 새 주제가 아직 통과한다', () => { + assert.equal(evalWith(twelve.slice(0, SEARCH_HARD_CAP - 1), '완전히새로운주제').allowed, true); + }); +}); + +describe('반복 위반 — 두 번째부터 도구 자체를 끈다', () => { + const five = ['가', '나', '다', '라', '마'].map(s => `주제${s}`); + + test('첫 차단에서는 검색만 막고 도구는 살려둔다', () => { + const d = evalWith(five, '주제가', 0); + assert.equal(d.allowed, false); + assert.equal(d.disableToolsForRestOfTurn, false); + }); + + test('이미 한 번 무시했다면 남은 턴 도구를 전부 끈다', () => { + const d = evalWith(five, '주제가', 1); + assert.equal(d.disableToolsForRestOfTurn, true); + assert.match(d.blockedReason, /already ignored one stop instruction/i); + }); +}); + +describe('차단 메시지 — 모델이 다음에 뭘 해야 하는지 알려준다', () => { + test('세 경우 모두 "지금 답을 쓰라"고 지시한다', () => { + const five = ['주제1', '주제2', '주제3', '주제4', '주제5']; + const twelve = Array.from({ length: SEARCH_HARD_CAP }, (_, i) => `t${i}`); + for (const d of [evalWith(five, '주제1', 0), evalWith(five, '주제1', 1), evalWith(twelve, 'x', 0)]) { + assert.match(d.blockedReason, /answer now|final text answer/i); + } + }); + + test('허용될 때는 사유가 비어 있다', () => { + const d = evalWith([], '아무거나'); + assert.equal(d.blockedReason, ''); + assert.equal(d.disableToolsForRestOfTurn, false); + assert.equal(d.hitHardCap, false); + }); +});