From de5407b7014140f6371e6ad6c4f426eb45c5de80 Mon Sep 17 00:00:00 2001 From: kim Date: Tue, 11 Aug 2026 13:47:42 +0900 Subject: [PATCH] =?UTF-8?q?feat:=20=EB=B9=84=EA=B5=90=20=EA=B2=80=EC=83=89?= =?UTF-8?q?=EC=97=90=EC=84=9C=20=EC=8A=A4=ED=8E=99=EC=9D=B4=20=EC=8B=A4?= =?UTF-8?q?=EB=A6=B0=20=ED=8E=98=EC=9D=B4=EC=A7=80=20=EB=B3=B8=EB=AC=B8?= =?UTF-8?q?=EC=9D=84=20=EC=9E=90=EB=8F=99=EC=9C=BC=EB=A1=9C=20=EA=B0=80?= =?UTF-8?q?=EC=A0=B8=EC=98=A8=EB=8B=A4?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 쪼개기(9f31d76)로 좋은 출처가 결과 목록엔 들어왔는데도 답변이 여전히 일반론이었다. 원인은 모델이 결과를 안 펼쳐본다는 것이다 — 프로덕션 로그 3,300줄에서 web_fetch 호출은 딱 1번, 그것도 기상청 페이지였고 GPU 스펙 질문에선 0번이다. 도구 설명에 "결과 URL을 web_fetch로 읽어라"라고 써 있는데도 그렇다. 스니펫엔 제품 이름만 있고 수치는 본문에 있으니, 스니펫만 보고 쓰면 "80 라인업이 60 라인업보다 빠르다"에서 멈춘다. 사용자 질문("스펙을 제조사에서 찾기가 어려운가?")에 답하려고 상위 세 출처를 실제로 가져와 재보니, 제조사 공식 페이지가 셋 중 가장 나빴다: nvidia.com 6016자, 스펙 키워드 0개 — 표가 JS 렌더링이라 본문은 "This site requires Javascript" + "Game Changer" 홍보 문구 techpowerup 275자, 스펙 키워드 0개 — 봇 차단 nanoreview Cores/TMUs/Boost Clock/Bandwidth/TFLOPS 전부 있음 그래서 가져오되 거르는 구조로 했다. hasSpecContent가 통과시킨 첫 페이지 하나만 대상별로 붙인다 — 이게 없으면 "Game Changer" 6KB가 컨텍스트를 차지하고, 여러 장을 붙이면 모델이 읽고 지나가야 할 양만 늘어난다. 실측(같은 쿼리): nanoreview 4080·5060 본문 2장 자동 확보, 9.4초, RTX 4080 Cores 9728 / 게이밍 76 vs RTX 5060 Cores 3840 / 게이밍 43 — "얼마나 빠른지"에 실제로 답할 수 있는 수치가 처음으로 들어왔다. Co-Authored-By: Claude Opus 5 --- src/tools/web.ts | 69 +++++++++++++++++++++++++++++- tests/web-comparison-split.test.ts | 46 +++++++++++++++++++- 2 files changed, 112 insertions(+), 3 deletions(-) diff --git a/src/tools/web.ts b/src/tools/web.ts index 6a059fd..d046fc2 100644 --- a/src/tools/web.ts +++ b/src/tools/web.ts @@ -1093,6 +1093,50 @@ async function extractAnswerFromResults(query: string, rawResults: string): Prom } } +/** + * Text that actually contains specifications, as opposed to a page that merely talks about them. + * Measured 2026-08-11 against the three sources this path surfaces for a GPU comparison: + * nvidia.com product page → 6016 chars, ZERO matches (JS-rendered spec table; the fetched text + * is marketing copy plus "This site requires Javascript") + * techpowerup.com → 275 chars, ZERO matches (bot wall) + * nanoreview.net → matches CUDA/Boost Clock/Bandwidth/TFLOPS/GB-s/Base Clock + * So the manufacturer's own page is the worst of the three here, and this filter is what keeps + * 6KB of "Game Changer" out of the model's context. + */ +const SPEC_SIGNAL = /\d[\d,.]*\s*(GB\/s|GB|MB|MHz|GHz|TFLOPS|nm)\b|\b(cuda\s*cores?|boost\s*clock|base\s*clock|memory\s*(bus|interface|bandwidth)|cores|tmus)\b\s*[::]?\s*\d/i; + +/** Does this fetched page body actually carry specifications? See SPEC_SIGNAL for the measurements. */ +export function hasSpecContent(text: string): boolean { + return SPEC_SIGNAL.test(String(text || '')); +} + +/** + * Opens the top results until one of them actually contains specification text. + * + * web_search returns titles and snippets; the figures a spec question needs live in the page body. + * The tool description has told the model to web_fetch result URLs all along, and it does not: + * across a 3,300-line production log (2026-08-11) web_fetch was called exactly once, for a weather + * page, never for any of the GPU spec questions in that same log. So the answer was always written + * off snippets alone, which is why it stayed at the level of "80 라인업이 60 라인업보다 빠르다". + * + * Stops at the first page with real spec content: one good source answers the question, and each + * extra page is context the model has to read past. + */ +async function fetchFirstSpecPage(results: any[], maxTries = 2): Promise { + const urls = (Array.isArray(results) ? results : []) + .map(r => String(r?.url || '').trim()) + .filter(u => /^https?:\/\//i.test(u)) + .slice(0, maxTries); + for (const url of urls) { + try { + const r = await executeWebFetch({ url, max_chars: 3000 }); + const text = String(r.stdout || '').trim(); + if (r.success && hasSpecContent(text)) return `[본문: ${url}]\n${text}`; + } catch { /* a dead or walled URL is the normal case here, not an error worth surfacing */ } + } + return null; +} + /** * Runs the combined query plus one search per entity, and hands the model all three labelled. * Partial failure is fine — any section that came back with results is still grounding the model @@ -1119,10 +1163,31 @@ async function runComparisonSearch( if (!sections.length) return combined; console.log(`[v2] web_search comparison split: "${args.query}" → "${split.a}" + "${split.b}"`); + + // Snippets name the products; only the page body has the numbers. Fetch one real spec page per + // entity, in parallel, since the model will not do it itself. + const [pageA, pageB] = await Promise.all([ + fetchFirstSpecPage((a.data as any)?.results || []), + fetchFirstSpecPage((b.data as any)?.results || []), + ]); + const fetched = [pageA, pageB].filter(Boolean) as string[]; + if (fetched.length) console.log(`[v2] web_search comparison split: auto-fetched ${fetched.length} spec page(s)`); + + const body = fetched.length + ? `${sections.join('\n\n')}\n\n${fetched.join('\n\n')}` + : sections.join('\n\n'); + const lead = fetched.length + ? '비교 질문이라 각 대상을 따로 검색하고, 스펙이 실제로 실린 페이지 본문까지 가져왔습니다. 아래 [본문] 블록의 수치를 근거로 답하세요.' + : '비교 질문이라 각 대상을 따로 검색했습니다. 아래 세 검색 결과를 모두 근거로 쓰세요.'; + return { success: true, - data: { ...(combined.data as any || {}), comparison_split: [split.a, split.b] }, - stdout: `비교 질문이라 각 대상을 따로 검색했습니다. 아래 세 검색 결과를 모두 근거로 쓰세요.\n\n${sections.join('\n\n')}`, + data: { + ...(combined.data as any || {}), + comparison_split: [split.a, split.b], + auto_fetched_pages: fetched.length, + }, + stdout: `${lead}\n\n${body}`, }; } diff --git a/tests/web-comparison-split.test.ts b/tests/web-comparison-split.test.ts index 0e15999..0f26161 100644 --- a/tests/web-comparison-split.test.ts +++ b/tests/web-comparison-split.test.ts @@ -13,7 +13,7 @@ import { test, describe } from 'node:test'; import assert from 'node:assert/strict'; -import { splitComparisonQuery } from '../src/tools/web'; +import { splitComparisonQuery, hasSpecContent } from '../src/tools/web'; describe('쪼개야 하는 쿼리', () => { test('실제 사고 쿼리 — 비교 동사는 버리고 속성어는 남긴다', () => { @@ -71,3 +71,47 @@ describe('쪼개면 안 되는 쿼리', () => { assert.equal(splitComparisonQuery(null as any), null); }); }); + +/** + * hasSpecContent — 가져온 본문에 진짜 스펙이 실려 있는지 + * + * 이 판정이 필요한 이유는 2026-08-11 실측이다. 비교 검색이 물어오는 상위 세 출처를 실제로 + * 가져와 봤더니, 제조사 공식 페이지가 셋 중 가장 쓸모없었다: + * nvidia.com 6016자, 스펙 키워드 0개 (스펙 표가 JS 렌더링, 본문은 홍보 문구) + * techpowerup 275자, 스펙 키워드 0개 (봇 차단) + * nanoreview Cores/TMUs/Boost Clock/Bandwidth/TFLOPS 전부 있음 + * 이 필터가 없으면 "Game Changer" 6KB가 모델 컨텍스트를 차지한다. + */ +describe('hasSpecContent — 본문에 수치가 실렸는지 판정', () => { + test('NVIDIA 공식 페이지 본문은 통과시키지 않는다 (실측 발췌)', () => { + assert.equal(hasSpecContent( + 'GeForce RTX 5060 Family Graphics Cards | NVIDIA. This site requires Javascript in order to ' + + 'view all its content. Game Changer. The NVIDIA GeForce RTX 5060 Ti and RTX 5060, powered by ' + + 'the NVIDIA Blackwell architecture, enables game-changing AI capabilities.', + ), false); + }); + + test('봇 차단 페이지도 통과시키지 않는다 (실측 발췌)', () => { + assert.equal(hasSpecContent( + 'NVIDIA GeForce RTX 5060 Specs | TechPowerUp GPU Database. Automated bot check in progress. ' + + 'Your browser must support Javascript.', + ), false); + }); + + test('nanoreview 본문은 통과시킨다 (실측 발췌)', () => { + assert.equal(hasSpecContent( + 'GeForce RTX 5060. Graphics Processor: GB206-250. Cores: 3840. TMUs / ROPs: 120 / 48.', + ), true); + }); + + test('대역폭·클럭 표기도 스펙으로 본다', () => { + assert.equal(hasSpecContent('Memory Bandwidth: 936 GB/s'), true); + assert.equal(hasSpecContent('Boost Clock: 2505 MHz'), true); + assert.equal(hasSpecContent('CUDA Cores: 9728'), true); + }); + + test('숫자 없는 일반 문장은 스펙이 아니다', () => { + assert.equal(hasSpecContent('이 카드는 게이밍 성능이 매우 뛰어납니다.'), false); + assert.equal(hasSpecContent(''), false); + }); +});