v4.3.24: 검색 예산 판단을 search-budget.ts로 분리 + 테스트 13개

라운드 루프(1,938줄) 안의 순수 판단 로직 추출. 루프 자체는 옮기지 않음 — 판단
근거는 아래에 정리.

- src/gateway/chat/search-budget.ts (신규) — evaluateSearchBudget()가
  (이전 쿼리들, 현재 쿼리, 차단횟수) → (허용여부, 사유, 도구차단, 하드캡도달)
  을 반환하는 순수 함수. handle-chat.ts 2,936 → 2,908줄
- tests/search-budget.test.ts — 13개. 캡이 "반복에는 물리고 팬아웃에는 안
  물린다"는 양방향을 모두 고정

테스트가 찾은 약점 — 한 단어 우회:
"어떤 미등장 토큰이라도 있으면 새 주제로 본다"는 규칙이 팬아웃(도시 8곳 날씨
확인 등)을 보호하려고 일부러 느슨한데, 그 탓에 "우크라이나 전황" → "우크라이나
전황 다시"가 완전히 새로운 주제로 통과해 재검색 루프를 그대로 허용했음. 이
모듈이 막으려던 바로 그 패턴. 규칙 자체를 조이면 정당한 쿼리까지 막히므로,
변별력 없는 단어(다시/추가/자세히/더/또/again/more/detail)를 불용어에 추가하는
쪽으로 해결.

라운드 루프를 통째로 옮기지 않은 이유:
루프는 1,938줄이면서 외부 가변 변수 22개(allThinking, toolSkipForcedRetries,
toolsDisabledForRestOfTurn, fileOpOwner 등)를 변경하고, 본질적으로 I/O
(모델 스트리밍·도구 실행·SSE 전송)라 순수 함수가 될 수 없음. 옮기려면 22개를
상태 객체로 묶어 넘겨야 하는데 회귀 위험만 크고 테스트 가능성은 늘지 않음 —
줄 수만 다른 파일로 이동할 뿐임. 오늘 tool-scope/system-prompt 추출이 값어치
있었던 건 순수 함수가 되어 테스트가 붙었기 때문이지 줄 수가 줄어서가 아니므로,
같은 기준으로 루프 안의 '판단'만 뽑는 방향을 택함.

전체 79개 테스트 통과. 배포 후 뉴스 요청에서 news_search 8회 팬아웃 정상 확인.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-29 16:16:59 +09:00
co-authored by Claude Opus 5
parent 4f31af0593
commit 5eb47ca08b
3 changed files with 227 additions and 37 deletions
+9 -37
View File
@@ -72,6 +72,7 @@ import { createBrowserDesktopAdvisors, type BrowserDesktopAdvisorCtx } from './b
import type { SkillsManager } from '../skills-manager';
import { selectToolsForTurn, bootAllowedTools, codeAiBlockedTools } from './tool-scope';
import { buildChatSystemPrompt } from './system-prompt';
import { evaluateSearchBudget } from './search-budget';
const MAX_TOOL_ROUNDS = 50;
type ExecutionMode = 'interactive' | 'background_task' | 'heartbeat' | 'cron';
@@ -414,17 +415,6 @@ async function handleChat(
// web_search/web_fetch query this turn. This lets legitimate fan-out (e.g. checking
// weather for 8 different cities) extend past the soft cap, while still blocking
// redundant re-searches of the same topic (the original runaway-loop failure mode).
const SEARCH_BUDGET_STOPWORDS = new Set([
'오늘', '현재', '지금', '최고', '최저', '기온', '날씨', '뉴스', '헤드라인',
'국내', '국제', '알려줘', '검색', '정보', '오늘의', 'today', 'news', 'weather',
]);
const extractSearchQueryTokens = (q: string): Set<string> => new Set(
String(q || '')
.toLowerCase()
.replace(/[^\p{L}\p{N}\s]/gu, ' ')
.split(/\s+/)
.filter((w) => w.length >= 2 && !SEARCH_BUDGET_STOPWORDS.has(w)),
);
const orchestrationState = new OrchestrationTriggerState();
const orchestrationLog: string[] = [];
const orchestrationStats = getOrchestrationSessionStats(sessionId);
@@ -2656,34 +2646,16 @@ async function handleChat(
// the same "today's news" or "GPT-5.6 pricing" query 7-50+ times) is blocked as before.
// A hard cap still applies regardless of distinctness, so extension can't run away either.
if (toolName === 'web_search' || toolName === 'web_fetch') {
const SEARCH_SOFT_CAP = 5;
const SEARCH_HARD_CAP = 12;
const priorSearchCalls = allToolResults.filter((r) => r.name === 'web_search' || r.name === 'web_fetch');
const searchFetchCallsSoFar = priorSearchCalls.length;
let allowAsNewTopic = false;
if (searchFetchCallsSoFar >= SEARCH_SOFT_CAP && searchFetchCallsSoFar < SEARCH_HARD_CAP) {
const currentTokens = extractSearchQueryTokens(String(toolArgs?.query || toolArgs?.url || ''));
const seenTokens = new Set<string>();
for (const r of priorSearchCalls) {
for (const t of extractSearchQueryTokens(String((r.args as any)?.query || (r.args as any)?.url || ''))) seenTokens.add(t);
}
allowAsNewTopic = [...currentTokens].some((t) => !seenTokens.has(t));
}
if (searchFetchCallsSoFar >= SEARCH_SOFT_CAP && !allowAsNewTopic) {
const _budget = evaluateSearchBudget({
priorQueries: priorSearchCalls.map((r) => String((r.args as any)?.query || (r.args as any)?.url || '')),
currentQuery: String(toolArgs?.query || toolArgs?.url || ''),
priorBlockedCount: searchBudgetBlockedCount,
});
if (!_budget.allowed) {
searchBudgetBlockedCount++;
const secondPlusStrike = searchBudgetBlockedCount >= 2;
if (secondPlusStrike) toolsDisabledForRestOfTurn = true;
const hitHardCap = searchFetchCallsSoFar >= SEARCH_HARD_CAP;
const _blockedSearch: ToolResult = {
name: toolName,
args: toolArgs,
result: hitHardCap
? `[BLOCKED] Search hard cap reached (${searchFetchCallsSoFar} calls, max ${SEARCH_HARD_CAP}). Write your final text answer now using whatever results you already have; if that's not enough, say so explicitly.`
: secondPlusStrike
? `[BLOCKED] Search budget exhausted (${searchFetchCallsSoFar} calls) and you already ignored one stop instruction. Tools are now disabled for the rest of this turn — no more calls of any kind will run. Write your final text answer now using whatever results you already have; if that's not enough, say so explicitly.`
: `[BLOCKED] Search budget exhausted (${searchFetchCallsSoFar} web_search/web_fetch calls already made this turn, soft cap ${SEARCH_SOFT_CAP}) and this query repeats a topic you already searched. Stop re-searching the same thing and answer now with what you already have. (New, genuinely distinct topics — e.g. a different city or subject — are still allowed up to ${SEARCH_HARD_CAP} calls total.)`,
error: true,
};
if (_budget.disableToolsForRestOfTurn) toolsDisabledForRestOfTurn = true;
const _blockedSearch: ToolResult = { name: toolName, args: toolArgs, result: _budget.blockedReason, error: true };
allToolResults.push(_blockedSearch);
sendSSE('tool_result', { action: toolName, result: _blockedSearch.result, error: true, stepNum: allToolResults.length });
messages.push({ role: 'tool', name: toolName, tool_name: toolName, tool_call_id: toolCallId || undefined, content: _blockedSearch.result });
+98
View File
@@ -0,0 +1,98 @@
/**
* search-budget.ts
*
* Decides whether one more web_search / web_fetch call is allowed this turn.
*
* Exists because of a real runaway: kimi-k2.6 was observed issuing 7+ (once 50+) searches for a
* single broad request like "오늘 국제 뉴스 정리", walking topic by topic (Ukraine, Gaza,
* tariffs, …) instead of doing a couple of broad searches. Each call costs latency and tokens
* and the answer does not improve.
*
* The tension this encodes: a naive "max N calls" cap also kills legitimate fan-out — checking
* weather for 8 different cities is 8 genuinely distinct searches, not a loop. So past the soft
* cap a call is still allowed if its query introduces a topic token nobody has searched yet,
* with a hard cap as the backstop so extension cannot run away either.
*
* Extracted from handleChat() on 2026-07-29. The loop it lived in is I/O-bound and cannot be
* made pure, but this decision can — and a decision is where the bugs actually are.
*/
/**
* Words too generic to distinguish one search from another. Without these, "오늘 서울 날씨" and
* "오늘 부산 날씨" would look distinct only by city (correct), but "오늘 뉴스" and "현재 뉴스"
* would also look distinct (wrong) and the cap would never bite.
*/
export const SEARCH_BUDGET_STOPWORDS = new Set([
'오늘', '현재', '지금', '최고', '최저', '기온', '날씨', '뉴스', '헤드라인',
'국내', '국제', '알려줘', '검색', '정보', '오늘의', 'today', 'news', 'weather',
// 2026-07-29: added when the extraction's tests showed a one-word escape hatch. The
// "any unseen token means a new topic" rule is deliberately permissive to protect fan-out,
// but that also meant "우크라이나 전황" → "우크라이나 전황 다시" read as a brand-new topic and
// sailed past the cap — the exact re-search loop this module exists to stop. These words
// never distinguish one search from another, so they belong here rather than tightening the
// rule itself (which would start blocking legitimate distinct queries).
'다시', '추가', '자세히', '더', '또', 'again', 'more', 'detail', 'details',
]);
export const SEARCH_SOFT_CAP = 5;
export const SEARCH_HARD_CAP = 12;
/** Content words of a query, lowercased, punctuation stripped, stopwords removed. */
export function extractSearchQueryTokens(q: string): Set<string> {
return new Set(
String(q || '')
.toLowerCase()
.replace(/[^\p{L}\p{N}\s]/gu, ' ')
.split(/\s+/)
.filter((w) => w.length >= 2 && !SEARCH_BUDGET_STOPWORDS.has(w)),
);
}
export interface SearchBudgetDecision {
/** false → the call must be replaced with a [BLOCKED] tool result. */
allowed: boolean;
/** Message handed back to the model in place of results. Empty when allowed. */
blockedReason: string;
/** True once the model has ignored a stop instruction — caller disables tools for the turn. */
disableToolsForRestOfTurn: boolean;
/** Whether the hard cap (not just the soft cap) was the thing that stopped it. */
hitHardCap: boolean;
}
export interface SearchBudgetInput {
/** Queries/urls of the web_search + web_fetch calls already made THIS turn, in order. */
priorQueries: string[];
/** query (or url) of the call being attempted now. */
currentQuery: string;
/** How many times this turn a search has already been blocked. 2+ means tools get cut off. */
priorBlockedCount: number;
}
export function evaluateSearchBudget(input: SearchBudgetInput): SearchBudgetDecision {
const callsSoFar = input.priorQueries.length;
const ok: SearchBudgetDecision = { allowed: true, blockedReason: '', disableToolsForRestOfTurn: false, hitHardCap: false };
if (callsSoFar < SEARCH_SOFT_CAP) return ok;
// Between soft and hard cap a genuinely new topic still gets through.
if (callsSoFar < SEARCH_HARD_CAP) {
const seen = new Set<string>();
for (const q of input.priorQueries) {
for (const t of extractSearchQueryTokens(q)) seen.add(t);
}
const introducesNewTopic = [...extractSearchQueryTokens(input.currentQuery)].some(t => !seen.has(t));
if (introducesNewTopic) return ok;
}
const blockedCount = input.priorBlockedCount + 1;
const secondPlusStrike = blockedCount >= 2;
const hitHardCap = callsSoFar >= SEARCH_HARD_CAP;
const blockedReason = hitHardCap
? `[BLOCKED] Search hard cap reached (${callsSoFar} calls, max ${SEARCH_HARD_CAP}). Write your final text answer now using whatever results you already have; if that's not enough, say so explicitly.`
: secondPlusStrike
? `[BLOCKED] Search budget exhausted (${callsSoFar} calls) and you already ignored one stop instruction. Tools are now disabled for the rest of this turn — no more calls of any kind will run. Write your final text answer now using whatever results you already have; if that's not enough, say so explicitly.`
: `[BLOCKED] Search budget exhausted (${callsSoFar} web_search/web_fetch calls already made this turn, soft cap ${SEARCH_SOFT_CAP}) and this query repeats a topic you already searched. Stop re-searching the same thing and answer now with what you already have. (New, genuinely distinct topics — e.g. a different city or subject — are still allowed up to ${SEARCH_HARD_CAP} calls total.)`;
return { allowed: false, blockedReason, disableToolsForRestOfTurn: secondPlusStrike, hitHardCap };
}
+120
View File
@@ -0,0 +1,120 @@
/**
* search-budget.test.ts
*
* The runaway this guards against was real: kimi-k2.6 issuing 7+ (once 50+) searches for one
* broad request, walking topic by topic instead of doing a couple of broad searches.
*
* The hard part is that the obvious fix — a flat call cap — breaks legitimate fan-out. Checking
* weather for 8 cities is 8 distinct searches, not a loop. So both directions need pinning:
* the cap must bite on repetition AND must not bite on genuine breadth. A regression in either
* direction is silent (either searches quietly stop working, or the runaway comes back).
*/
import { test, describe } from 'node:test';
import assert from 'node:assert/strict';
import {
evaluateSearchBudget,
extractSearchQueryTokens,
SEARCH_SOFT_CAP,
SEARCH_HARD_CAP,
} from '../src/gateway/chat/search-budget';
const evalWith = (priorQueries: string[], currentQuery: string, priorBlockedCount = 0) =>
evaluateSearchBudget({ priorQueries, currentQuery, priorBlockedCount });
describe('extractSearchQueryTokens', () => {
test('불용어를 걸러 실제 주제어만 남긴다', () => {
const t = extractSearchQueryTokens('오늘 서울 날씨 알려줘');
assert.ok(t.has('서울'));
for (const stop of ['오늘', '날씨', '알려줘']) assert.ok(!t.has(stop), `${stop} 가 남음`);
});
test('대소문자/문장부호를 정규화한다', () => {
const t = extractSearchQueryTokens('RTX 5090, PCIe 5.0!');
assert.ok(t.has('rtx'));
assert.ok(t.has('5090'));
});
test('한 글자 토큰은 버린다 — 변별력이 없다', () => {
assert.ok(!extractSearchQueryTokens('a b 서 울').has('a'));
});
});
describe('소프트캡 이하 — 자유롭게 허용', () => {
test('0회~4회까지는 무조건 통과', () => {
for (let n = 0; n < SEARCH_SOFT_CAP; n++) {
const prior = Array.from({ length: n }, (_, i) => `쿼리${i}`);
assert.equal(evalWith(prior, '쿼리0').allowed, true, `${n}회째에서 막힘`);
}
});
});
describe('소프트캡 초과 — 반복은 막고 새 주제는 통과', () => {
const five = ['우크라이나 전황', '가자 지구', '관세 협상', '금리 인상', '유가 동향'];
test('이미 검색한 주제를 또 검색하면 막힌다 (원래의 폭주 패턴)', () => {
const d = evalWith(five, '우크라이나 전황 다시');
assert.equal(d.allowed, false);
assert.match(d.blockedReason, /\[BLOCKED\]/);
assert.match(d.blockedReason, /repeats a topic/);
});
test('새로운 주제는 소프트캡을 넘겨도 통과한다 — 정당한 팬아웃 보호', () => {
// 도시 8곳 날씨 확인 같은 경우가 여기 해당한다.
assert.equal(evalWith(five, '삿포로 적설량').allowed, true);
});
test('불용어만 다른 쿼리는 새 주제가 아니다', () => {
// "오늘 뉴스" vs "현재 뉴스" — 이게 통과하면 캡이 사실상 무력화된다.
const prior = ['오늘 뉴스', '현재 뉴스', '지금 뉴스', '오늘의 뉴스', '뉴스 알려줘'];
assert.equal(evalWith(prior, '지금 뉴스 알려줘').allowed, false);
});
});
describe('하드캡 — 새 주제라도 결국 멈춘다', () => {
const twelve = Array.from({ length: SEARCH_HARD_CAP }, (_, i) => `고유주제${i}`);
test('12회에 도달하면 완전히 새로운 주제도 막힌다', () => {
const d = evalWith(twelve, '완전히새로운주제');
assert.equal(d.allowed, false);
assert.equal(d.hitHardCap, true);
assert.match(d.blockedReason, /hard cap/i);
});
test('하드캡 직전(11회)에는 새 주제가 아직 통과한다', () => {
assert.equal(evalWith(twelve.slice(0, SEARCH_HARD_CAP - 1), '완전히새로운주제').allowed, true);
});
});
describe('반복 위반 — 두 번째부터 도구 자체를 끈다', () => {
const five = ['가', '나', '다', '라', '마'].map(s => `주제${s}`);
test('첫 차단에서는 검색만 막고 도구는 살려둔다', () => {
const d = evalWith(five, '주제가', 0);
assert.equal(d.allowed, false);
assert.equal(d.disableToolsForRestOfTurn, false);
});
test('이미 한 번 무시했다면 남은 턴 도구를 전부 끈다', () => {
const d = evalWith(five, '주제가', 1);
assert.equal(d.disableToolsForRestOfTurn, true);
assert.match(d.blockedReason, /already ignored one stop instruction/i);
});
});
describe('차단 메시지 — 모델이 다음에 뭘 해야 하는지 알려준다', () => {
test('세 경우 모두 "지금 답을 쓰라"고 지시한다', () => {
const five = ['주제1', '주제2', '주제3', '주제4', '주제5'];
const twelve = Array.from({ length: SEARCH_HARD_CAP }, (_, i) => `t${i}`);
for (const d of [evalWith(five, '주제1', 0), evalWith(five, '주제1', 1), evalWith(twelve, 'x', 0)]) {
assert.match(d.blockedReason, /answer now|final text answer/i);
}
});
test('허용될 때는 사유가 비어 있다', () => {
const d = evalWith([], '아무거나');
assert.equal(d.blockedReason, '');
assert.equal(d.disableToolsForRestOfTurn, false);
assert.equal(d.hitHardCap, false);
});
});