fix: "A vs B" 비교 검색을 코드에서 대상별로 쪼개 실행

어제(08-10) "A vs B로 합쳐 검색하지 말고 각각 따로 검색하라"를 web_search
설명에 넣었는데, 오늘 프로덕션 로그에서 모델이 세 턴 연속(3149/3287/3303)
정확히 금지된 형태를 그대로 날렸다:

  web_search({"query":"RTX 4080 vs RTX 5060 performance comparison specs"})

매번 technical.city 링크가 죽어 있었고(link-validator 2/5 dead), 모델
손에 남은 건 제목뿐이라 답변은 수치 0건의 일반론이 됐다. 사용자 평가는
"자료가 부실하고 요점도 안 맞는다".

오늘 하루 네 번째 같은 패턴이다 — news_search country 파라미터, 카테고리
단어, 다도시 날씨 배치, 그리고 이번 건. 도구 설명으로 쿼리 구성을 강제할
수 없다는 게 반복 확인됐으므로 코드에서 쪼갠다.

- splitComparisonQuery: 비교 동사(comparison/vs/비교/차이)는 버리고
  속성어(performance/specs/대역폭/성능)는 양쪽에 붙여 둘로 나눈다
- 합친 쿼리도 그대로 실행한다. 비교 페이지가 살아있을 땐 그게 최선의
  출처라서, 빼면 실패 모드를 다른 실패 모드로 바꾸는 것에 불과하다.
  쪼갠 검색은 추가분이고 결과 수를 3개로 더 조인다
- 오탐 쪽을 비싸게 본다: 속성어도 모델번호 꼴도 없으면 쪼개지 않는다
  ("Lakers vs Celtics"를 두 검색으로 만들면 질문 자체가 사라진다).
  "검색결과 스펙"처럼 과/와로 끝나는 단어에 걸리는 것도 길이로 막았다

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-08-11 13:13:05 +09:00
co-authored by Claude Opus 5
parent 23cd4886af
commit 9f31d76edc
2 changed files with 169 additions and 1 deletions
+96 -1
View File
@@ -843,8 +843,70 @@ function isLowValueNewsResult(url: string): boolean {
}
}
// ── Comparison-query splitting ────────────────────────────────────────────────
//
// "A vs B <spec>" queries reliably return review/overview pages that name both products and
// give the figures for neither, while searching each product separately returns the numbers
// immediately. Verified 2026-08-10: "RTX 3090 vs RTX 4080 Super memory bandwidth" produced a
// generic overview with no figure; the two separate searches produced 936 GB/s and 736 GB/s.
//
// That finding went into the web_search tool description as an instruction the same day, and the
// model ignored it — production log 2026-08-11 shows three consecutive turns issuing the exact
// combined form it was told not to use ("RTX 4080 vs RTX 5060 performance comparison specs"),
// each returning dead technical.city links, each producing an answer of pure generalities with no
// number in it. Same lesson as news_search's country/category params and weather's multi-city
// batching earlier that day: a tool description cannot enforce query construction on this model,
// so the split happens here in code instead ([[feedback_local_model_needs_code_backstop]]).
//
// The combined query still runs — when a comparison page IS alive it is genuinely the best source
// for this question, and dropping it would trade one failure mode for another. The split searches
// are additive and capped tighter so the extra grounding does not blow up the context.
const COMPARISON_SEPARATOR = /\s+(?:vs\.?|versus)\s+|\s*와\s+|\s*과\s+/i;
/** Words describing WHAT is being compared — kept, since they narrow each single-entity search. */
const COMPARISON_ATTRIBUTE = /\b(performance|specs?|specifications?|benchmarks?|review|memory\s*bandwidth|bandwidth|tflops|vram|price|speed)\b|성능|스펙|사양|벤치마크|대역폭|가격|속도/gi;
/** Words meaning "compare these" — dropped, since they are meaningless in a single-entity search. */
const COMPARISON_VERB = /\b(comparison|compare[ds]?|versus|vs\.?|difference|diff)\b|비교|차이/gi;
export interface ComparisonSplit { a: string; b: string; }
/**
* Split "RTX 4080 vs RTX 5060 performance comparison specs" into
* "RTX 4080 performance specs" + "RTX 5060 performance specs", or null when the query is not a
* spec comparison. Deliberately conservative: "Lakers vs Celtics" has no attribute word and no
* model numbers, so it is left alone rather than turned into two unrelated searches.
*/
export function splitComparisonQuery(query: string): ComparisonSplit | null {
const q = String(query || '').trim();
if (!q) return null;
const parts = q.split(COMPARISON_SEPARATOR);
if (parts.length !== 2) return null;
const attributes = Array.from(new Set(
(q.match(COMPARISON_ATTRIBUTE) || []).map(s => s.trim().toLowerCase()),
));
const strip = (s: string) => s
.replace(COMPARISON_ATTRIBUTE, ' ')
.replace(COMPARISON_VERB, ' ')
.replace(/\s+/g, ' ')
.trim();
const [entityA, entityB] = parts.map(strip);
if (entityA.length < 2 || entityB.length < 2) return null;
// Either an explicit attribute ("performance", "스펙") or two model-number-shaped entities.
// Without one of those this is not a spec comparison and splitting would just lose meaning.
const bothLookLikeModels = /\d/.test(entityA) && /\d/.test(entityB);
if (!attributes.length && !bothLookLikeModels) return null;
const tail = attributes.length ? attributes.join(' ') : 'specs';
return { a: `${entityA} ${tail}`.trim(), b: `${entityB} ${tail}`.trim() };
}
// ── Main web_search tool ──────────────────────────────────────────────────────
export async function executeWebSearch(args: { query: string; max_results?: number }): Promise<ToolResult> {
export async function executeWebSearch(args: { query: string; max_results?: number; _noSplit?: boolean }): Promise<ToolResult> {
if (!args._noSplit) {
const split = splitComparisonQuery(args.query || '');
if (split) return runComparisonSearch(args, split);
}
if (!args.query?.trim()) return { success: false, error: 'query is required' };
let limit = Math.min(args.max_results ?? 5, 10);
if (isPriceQuery(args.query)) limit = Math.max(limit, 5);
@@ -1031,6 +1093,39 @@ async function extractAnswerFromResults(query: string, rawResults: string): Prom
}
}
/**
* Runs the combined query plus one search per entity, and hands the model all three labelled.
* Partial failure is fine — any section that came back with results is still grounding the model
* did not have before, so this only ever falls back to whatever the combined query alone returned.
*/
async function runComparisonSearch(
args: { query: string; max_results?: number },
split: ComparisonSplit,
): Promise<ToolResult> {
const splitLimit = Math.min(args.max_results ?? 5, 3);
const [combined, a, b] = await Promise.all([
executeWebSearch({ ...args, _noSplit: true }),
executeWebSearch({ query: split.a, max_results: splitLimit, _noSplit: true }),
executeWebSearch({ query: split.b, max_results: splitLimit, _noSplit: true }),
]);
const sections: string[] = [];
const push = (label: string, r: ToolResult) => {
if (r.success && String(r.stdout || '').trim()) sections.push(`[${label}]\n${String(r.stdout).trim()}`);
};
push(`검색: ${args.query}`, combined);
push(`검색: ${split.a}`, a);
push(`검색: ${split.b}`, b);
if (!sections.length) return combined;
console.log(`[v2] web_search comparison split: "${args.query}" → "${split.a}" + "${split.b}"`);
return {
success: true,
data: { ...(combined.data as any || {}), comparison_split: [split.a, split.b] },
stdout: `비교 질문이라 각 대상을 따로 검색했습니다. 아래 세 검색 결과를 모두 근거로 쓰세요.\n\n${sections.join('\n\n')}`,
};
}
export async function executeWebSearchWithExtraction(args: { query: string; max_results?: number }): Promise<ToolResult> {
const res = await executeWebSearch(args);
if (!res.success || !res.stdout) return res;
+73
View File
@@ -0,0 +1,73 @@
/**
* splitComparisonQuery — 비교 질문을 대상별 검색으로 쪼개기
*
* 2026-08-10에 "A vs B로 합치지 말고 따로 검색하라"를 web_search 설명에 넣었는데, 08-11 프로덕션
* 로그에서 모델이 세 턴 연속으로 정확히 금지된 형태를 그대로 날렸다:
* web_search({"query":"RTX 4080 vs RTX 5060 performance comparison specs"})
* 매번 technical.city 링크가 죽어 있었고(2/5 dead) 답변은 수치 0건의 일반론이었다. 설명으로
* 강제가 안 되니 코드에서 쪼갠다.
*
* 이 테스트가 지키는 선: 쪼갤 만한 것만 쪼갠다. "Lakers vs Celtics"를 "Lakers"/"Celtics" 두
* 검색으로 만들면 원래 질문의 의미가 사라진다 — 오탐 쪽이 더 비싸다.
*/
import { test, describe } from 'node:test';
import assert from 'node:assert/strict';
import { splitComparisonQuery } from '../src/tools/web';
describe('쪼개야 하는 쿼리', () => {
test('실제 사고 쿼리 — 비교 동사는 버리고 속성어는 남긴다', () => {
const r = splitComparisonQuery('RTX 4080 vs RTX 5060 performance comparison specs');
assert.deepEqual(r, { a: 'RTX 4080 performance specs', b: 'RTX 5060 performance specs' });
});
test('속성어가 없어도 양쪽이 모델명 꼴이면 쪼갠다', () => {
const r = splitComparisonQuery('RTX 4080 vs RTX 5060');
assert.deepEqual(r, { a: 'RTX 4080 specs', b: 'RTX 5060 specs' });
});
test('어제 검증한 대역폭 케이스', () => {
const r = splitComparisonQuery('RTX 3090 vs RTX 4080 Super memory bandwidth');
assert.ok(r);
assert.ok(r!.a.startsWith('RTX 3090'));
assert.ok(r!.b.startsWith('RTX 4080 Super'));
assert.ok(r!.a.includes('bandwidth') && r!.b.includes('bandwidth'));
});
test('한국어 "와/과 ... 비교"도 쪼갠다', () => {
assert.deepEqual(
splitComparisonQuery('RTX 4080과 5060 성능 비교'),
{ a: 'RTX 4080 성능', b: '5060 성능' },
);
});
test('vs. / versus 표기도 인식한다', () => {
assert.ok(splitComparisonQuery('M4 Max vs. M3 Ultra benchmark'));
assert.ok(splitComparisonQuery('A100 versus H100 tflops'));
});
});
describe('쪼개면 안 되는 쿼리', () => {
test('스포츠 대진 — 속성어도 모델번호도 없다', () => {
assert.equal(splitComparisonQuery('Lakers vs Celtics'), null);
});
test('일반 검색은 건드리지 않는다', () => {
assert.equal(splitComparisonQuery('오늘 서울 날씨'), null);
assert.equal(splitComparisonQuery('RTX 4080 memory bandwidth'), null);
});
test('비교 대상이 셋이면 판단하지 않는다', () => {
assert.equal(splitComparisonQuery('A vs B vs C specs'), null);
});
test('"결과"처럼 과/와로 끝나는 단어에 잘못 걸리지 않는다', () => {
assert.equal(splitComparisonQuery('검색결과 스펙'), null);
assert.equal(splitComparisonQuery('효과 성능'), null);
});
test('빈 입력', () => {
assert.equal(splitComparisonQuery(''), null);
assert.equal(splitComparisonQuery(null as any), null);
});
});