diff --git a/src/tools/web.ts b/src/tools/web.ts index 17b4301..6a059fd 100644 --- a/src/tools/web.ts +++ b/src/tools/web.ts @@ -843,8 +843,70 @@ function isLowValueNewsResult(url: string): boolean { } } +// ── Comparison-query splitting ──────────────────────────────────────────────── +// +// "A vs B " queries reliably return review/overview pages that name both products and +// give the figures for neither, while searching each product separately returns the numbers +// immediately. Verified 2026-08-10: "RTX 3090 vs RTX 4080 Super memory bandwidth" produced a +// generic overview with no figure; the two separate searches produced 936 GB/s and 736 GB/s. +// +// That finding went into the web_search tool description as an instruction the same day, and the +// model ignored it — production log 2026-08-11 shows three consecutive turns issuing the exact +// combined form it was told not to use ("RTX 4080 vs RTX 5060 performance comparison specs"), +// each returning dead technical.city links, each producing an answer of pure generalities with no +// number in it. Same lesson as news_search's country/category params and weather's multi-city +// batching earlier that day: a tool description cannot enforce query construction on this model, +// so the split happens here in code instead ([[feedback_local_model_needs_code_backstop]]). +// +// The combined query still runs — when a comparison page IS alive it is genuinely the best source +// for this question, and dropping it would trade one failure mode for another. The split searches +// are additive and capped tighter so the extra grounding does not blow up the context. +const COMPARISON_SEPARATOR = /\s+(?:vs\.?|versus)\s+|\s*와\s+|\s*과\s+/i; +/** Words describing WHAT is being compared — kept, since they narrow each single-entity search. */ +const COMPARISON_ATTRIBUTE = /\b(performance|specs?|specifications?|benchmarks?|review|memory\s*bandwidth|bandwidth|tflops|vram|price|speed)\b|성능|스펙|사양|벤치마크|대역폭|가격|속도/gi; +/** Words meaning "compare these" — dropped, since they are meaningless in a single-entity search. */ +const COMPARISON_VERB = /\b(comparison|compare[ds]?|versus|vs\.?|difference|diff)\b|비교|차이/gi; + +export interface ComparisonSplit { a: string; b: string; } + +/** + * Split "RTX 4080 vs RTX 5060 performance comparison specs" into + * "RTX 4080 performance specs" + "RTX 5060 performance specs", or null when the query is not a + * spec comparison. Deliberately conservative: "Lakers vs Celtics" has no attribute word and no + * model numbers, so it is left alone rather than turned into two unrelated searches. + */ +export function splitComparisonQuery(query: string): ComparisonSplit | null { + const q = String(query || '').trim(); + if (!q) return null; + const parts = q.split(COMPARISON_SEPARATOR); + if (parts.length !== 2) return null; + + const attributes = Array.from(new Set( + (q.match(COMPARISON_ATTRIBUTE) || []).map(s => s.trim().toLowerCase()), + )); + const strip = (s: string) => s + .replace(COMPARISON_ATTRIBUTE, ' ') + .replace(COMPARISON_VERB, ' ') + .replace(/\s+/g, ' ') + .trim(); + const [entityA, entityB] = parts.map(strip); + if (entityA.length < 2 || entityB.length < 2) return null; + + // Either an explicit attribute ("performance", "스펙") or two model-number-shaped entities. + // Without one of those this is not a spec comparison and splitting would just lose meaning. + const bothLookLikeModels = /\d/.test(entityA) && /\d/.test(entityB); + if (!attributes.length && !bothLookLikeModels) return null; + + const tail = attributes.length ? attributes.join(' ') : 'specs'; + return { a: `${entityA} ${tail}`.trim(), b: `${entityB} ${tail}`.trim() }; +} + // ── Main web_search tool ────────────────────────────────────────────────────── -export async function executeWebSearch(args: { query: string; max_results?: number }): Promise { +export async function executeWebSearch(args: { query: string; max_results?: number; _noSplit?: boolean }): Promise { + if (!args._noSplit) { + const split = splitComparisonQuery(args.query || ''); + if (split) return runComparisonSearch(args, split); + } if (!args.query?.trim()) return { success: false, error: 'query is required' }; let limit = Math.min(args.max_results ?? 5, 10); if (isPriceQuery(args.query)) limit = Math.max(limit, 5); @@ -1031,6 +1093,39 @@ async function extractAnswerFromResults(query: string, rawResults: string): Prom } } +/** + * Runs the combined query plus one search per entity, and hands the model all three labelled. + * Partial failure is fine — any section that came back with results is still grounding the model + * did not have before, so this only ever falls back to whatever the combined query alone returned. + */ +async function runComparisonSearch( + args: { query: string; max_results?: number }, + split: ComparisonSplit, +): Promise { + const splitLimit = Math.min(args.max_results ?? 5, 3); + const [combined, a, b] = await Promise.all([ + executeWebSearch({ ...args, _noSplit: true }), + executeWebSearch({ query: split.a, max_results: splitLimit, _noSplit: true }), + executeWebSearch({ query: split.b, max_results: splitLimit, _noSplit: true }), + ]); + + const sections: string[] = []; + const push = (label: string, r: ToolResult) => { + if (r.success && String(r.stdout || '').trim()) sections.push(`[${label}]\n${String(r.stdout).trim()}`); + }; + push(`검색: ${args.query}`, combined); + push(`검색: ${split.a}`, a); + push(`검색: ${split.b}`, b); + + if (!sections.length) return combined; + console.log(`[v2] web_search comparison split: "${args.query}" → "${split.a}" + "${split.b}"`); + return { + success: true, + data: { ...(combined.data as any || {}), comparison_split: [split.a, split.b] }, + stdout: `비교 질문이라 각 대상을 따로 검색했습니다. 아래 세 검색 결과를 모두 근거로 쓰세요.\n\n${sections.join('\n\n')}`, + }; +} + export async function executeWebSearchWithExtraction(args: { query: string; max_results?: number }): Promise { const res = await executeWebSearch(args); if (!res.success || !res.stdout) return res; diff --git a/tests/web-comparison-split.test.ts b/tests/web-comparison-split.test.ts new file mode 100644 index 0000000..0e15999 --- /dev/null +++ b/tests/web-comparison-split.test.ts @@ -0,0 +1,73 @@ +/** + * splitComparisonQuery — 비교 질문을 대상별 검색으로 쪼개기 + * + * 2026-08-10에 "A vs B로 합치지 말고 따로 검색하라"를 web_search 설명에 넣었는데, 08-11 프로덕션 + * 로그에서 모델이 세 턴 연속으로 정확히 금지된 형태를 그대로 날렸다: + * web_search({"query":"RTX 4080 vs RTX 5060 performance comparison specs"}) + * 매번 technical.city 링크가 죽어 있었고(2/5 dead) 답변은 수치 0건의 일반론이었다. 설명으로 + * 강제가 안 되니 코드에서 쪼갠다. + * + * 이 테스트가 지키는 선: 쪼갤 만한 것만 쪼갠다. "Lakers vs Celtics"를 "Lakers"/"Celtics" 두 + * 검색으로 만들면 원래 질문의 의미가 사라진다 — 오탐 쪽이 더 비싸다. + */ + +import { test, describe } from 'node:test'; +import assert from 'node:assert/strict'; +import { splitComparisonQuery } from '../src/tools/web'; + +describe('쪼개야 하는 쿼리', () => { + test('실제 사고 쿼리 — 비교 동사는 버리고 속성어는 남긴다', () => { + const r = splitComparisonQuery('RTX 4080 vs RTX 5060 performance comparison specs'); + assert.deepEqual(r, { a: 'RTX 4080 performance specs', b: 'RTX 5060 performance specs' }); + }); + + test('속성어가 없어도 양쪽이 모델명 꼴이면 쪼갠다', () => { + const r = splitComparisonQuery('RTX 4080 vs RTX 5060'); + assert.deepEqual(r, { a: 'RTX 4080 specs', b: 'RTX 5060 specs' }); + }); + + test('어제 검증한 대역폭 케이스', () => { + const r = splitComparisonQuery('RTX 3090 vs RTX 4080 Super memory bandwidth'); + assert.ok(r); + assert.ok(r!.a.startsWith('RTX 3090')); + assert.ok(r!.b.startsWith('RTX 4080 Super')); + assert.ok(r!.a.includes('bandwidth') && r!.b.includes('bandwidth')); + }); + + test('한국어 "와/과 ... 비교"도 쪼갠다', () => { + assert.deepEqual( + splitComparisonQuery('RTX 4080과 5060 성능 비교'), + { a: 'RTX 4080 성능', b: '5060 성능' }, + ); + }); + + test('vs. / versus 표기도 인식한다', () => { + assert.ok(splitComparisonQuery('M4 Max vs. M3 Ultra benchmark')); + assert.ok(splitComparisonQuery('A100 versus H100 tflops')); + }); +}); + +describe('쪼개면 안 되는 쿼리', () => { + test('스포츠 대진 — 속성어도 모델번호도 없다', () => { + assert.equal(splitComparisonQuery('Lakers vs Celtics'), null); + }); + + test('일반 검색은 건드리지 않는다', () => { + assert.equal(splitComparisonQuery('오늘 서울 날씨'), null); + assert.equal(splitComparisonQuery('RTX 4080 memory bandwidth'), null); + }); + + test('비교 대상이 셋이면 판단하지 않는다', () => { + assert.equal(splitComparisonQuery('A vs B vs C specs'), null); + }); + + test('"결과"처럼 과/와로 끝나는 단어에 잘못 걸리지 않는다', () => { + assert.equal(splitComparisonQuery('검색결과 스펙'), null); + assert.equal(splitComparisonQuery('효과 성능'), null); + }); + + test('빈 입력', () => { + assert.equal(splitComparisonQuery(''), null); + assert.equal(splitComparisonQuery(null as any), null); + }); +});