fix: buildDirectPriceAnswer가 무관한 달러 값을 "Answer:"로 합성해 붙이던 문제
이 함수가 만드는 줄은 검색 결과 stdout 맨 앞에 붙는다. 시스템 프롬프트의 ANTI-HALLUCINATION은
"도구가 돌려준 것을 정확히 보고하라, 절대 무시하지 말라"고 지시하므로, 여기서 틀리면 모델에게
틀린 값을 신뢰하라고 시키는 셈이다. 게다가 합성된 숫자는 도구 출력 텍스트의 일부가 되므로,
모델이 그대로 옮기면 NUMERIC-GROUNDING이 "소스에 있다"며 통과시킨다 — 우리가 만든 숫자가
'근거 있음' 도장을 받아 나간다.
실제 사고(프로덕션 로그):
USER: 컨테이너선 급유비가 얼마나 될까?
TOOL: web_search("average fuel cost for 20,000 TEU container ship per voyage")
TOOL OK: Answer: The current price is approximately $38.00 USD.
$38은 스니펫에 있던 "신조선 7일 항해 시 TEU 1개당 추가 연료비"였고, 실제 답은 항해당 수백만
달러다. 그 턴은 모델이 결과 [1]을 직접 읽어 스스로 바로잡았지만 그건 운이다.
* isPriceQuery에 단어 경계가 없어 통화·가격 낱말이 다른 단어 **안에서** 걸렸다. "eur"가
"Europe"·"Neuralink"에, "cost"가 "costume"에, "value"가 "valuable"에 들어맞는다. 실사용
고유 검색어 195건 중 19건이 통과했고 그중 8건이 가격과 무관했다("Europe drought status
August 2026", "Neuralink Blindsight resolution pixels" 등). 경계를 넣어 19 → 11건.
* generic 자산 경로를 제거했다. 금·은·비트코인은 타당성 범위(온스당 300~10,000 등)와 자산명
대조가 성립해 판정이 의미가 있다. generic은 둘 다 없다 — 허용 범위가 $0.5~$5,000,000이라
사실상 모든 달러 표기를 받고, 문구도 무엇의 가격인지 말하지 못한 채 "The current price"라고만
쓴다. 이 로그 표본에서 generic 경로가 실제로 답을 낸 유일한 사례가 위 컨테이너선 건이다.
* 자산명 대조를 가점(+2)에서 **필수 조건**으로 올렸다. 가점일 때는 자산을 한 번도 언급하지
않은 스니펫의 숫자가 score 0으로 통과했다 — "금 시세" 질문에 다른 상품 가격이 답으로 나갈
수 있다는 뜻이다. 티커(XAU/XAG/BTC)도 함께 본다.
순효과(실사용 고유 검색어 195건): 발동 19 → 11건, 그중 실제로 "Answer:" 줄이 생길 수 있는
것 0건. 이 표본의 가격 질문은 전부 GPU·옵션·펌프 같은 generic 대상이라, 원래도 이 기능이
맞는 답을 낸 적이 없다.
테스트 544개 통과(+8).
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Rfme1WVEPkpNwnf5oVNXXc
This commit is contained in:
+41
-6
@@ -59,8 +59,15 @@ function isLowQualityGoogleUrl(url: string): boolean {
|
||||
return /google\.com\/share\.google\?/i.test(url);
|
||||
}
|
||||
|
||||
function isPriceQuery(query: string): boolean {
|
||||
return /price|cost|value|quote|trades?|usd|dollar|eur|gbp|jpy/i.test(query);
|
||||
// 2026-09-04 감사: 단어 경계가 없어서 통화·가격 낱말이 다른 단어 **안에서** 걸렸다.
|
||||
// "eur"가 "Europe"·"Neuralink"에, "cost"가 "costume"에, "value"가 "valuable"에 들어맞는다.
|
||||
// 실사용 고유 검색어 195건 중 19건이 이 함수를 통과했고 그중 상당수가 가격과 무관했다:
|
||||
// "Europe drought status August 2026" ← Eur(ope)
|
||||
// "Neuralink Blindsight resolution pixels" ← N(eur)alink
|
||||
// "European weather outlook August 2026 precipitation"
|
||||
// 통과하면 아래 buildDirectPriceAnswer가 검색 결과 맨 앞에 "Answer: …" 줄을 합성해 붙인다.
|
||||
export function isPriceQuery(query: string): boolean {
|
||||
return /\b(price|prices|pricing|cost|costs|value|quote|quotes|trade|trades|usd|dollar|dollars|eur|euro|euros|gbp|jpy)\b/i.test(query);
|
||||
}
|
||||
|
||||
function isBitcoinQuery(query: string): boolean {
|
||||
@@ -108,6 +115,14 @@ function hasFreshPriceCue(text: string): boolean {
|
||||
return /\b(current|today|live|latest|now|right now|spot)\b/.test(t);
|
||||
}
|
||||
|
||||
// 스니펫이 그 자산을 실제로 말하고 있는가. 티커도 함께 본다 — 시세 페이지는 "Bitcoin"보다
|
||||
// "BTC/USD", "XAU"를 쓰는 쪽이 흔하다.
|
||||
const ASSET_MENTION: Record<'silver' | 'gold' | 'bitcoin', RegExp> = {
|
||||
silver: /\b(silver|xag)\b/i,
|
||||
gold: /\b(gold|xau)\b/i,
|
||||
bitcoin: /\b(bitcoin|btc)\b/i,
|
||||
};
|
||||
|
||||
function detectPriceAsset(query: string): 'silver' | 'gold' | 'bitcoin' | 'generic' {
|
||||
const q = String(query || '').toLowerCase();
|
||||
if (/\b(silver|xag)\b/.test(q)) return 'silver';
|
||||
@@ -124,16 +139,38 @@ function isPlausibleUsdPrice(asset: 'silver' | 'gold' | 'bitcoin' | 'generic', v
|
||||
return valuePerOunceOrUnit >= 0.5 && valuePerOunceOrUnit <= 5_000_000;
|
||||
}
|
||||
|
||||
function buildDirectPriceAnswer(
|
||||
export function buildDirectPriceAnswer(
|
||||
query: string,
|
||||
results: SearchResultItem[]
|
||||
): string {
|
||||
if (!isPriceQuery(query)) return '';
|
||||
|
||||
const asset = detectPriceAsset(query);
|
||||
// 2026-09-04 감사 — generic 경로를 없앤다.
|
||||
//
|
||||
// 이 함수가 만드는 줄은 검색 결과 맨 앞에 붙고, 시스템 프롬프트는 ANTI-HALLUCINATION에서
|
||||
// "도구가 돌려준 것을 정확히 보고하라, 절대 무시하지 말라"고 지시한다. 즉 여기서 틀리면
|
||||
// 모델에게 틀린 값을 신뢰하라고 시키는 셈이다. 게다가 합성된 숫자는 도구 출력 텍스트의
|
||||
// 일부가 되므로, 모델이 그대로 옮기면 NUMERIC-GROUNDING이 "소스에 있다"며 통과시킨다 —
|
||||
// 우리가 만든 숫자가 '근거 있음' 도장을 받아 나간다.
|
||||
//
|
||||
// 금·은·비트코인은 타당성 범위(온스당 300~10,000 등)와 자산명 대조가 성립해서 판정이
|
||||
// 의미가 있다. generic은 둘 다 없다: 허용 범위가 $0.5~$5,000,000이라 사실상 모든 달러
|
||||
// 표기를 받고, 문구도 무엇의 가격인지 말하지 못한 채 "The current price"라고만 쓴다.
|
||||
// 실제 사고(2026-08, 프로덕션): "컨테이너선 급유비가 얼마나 될까?" →
|
||||
// web_search("average fuel cost for 20,000 TEU container ship per voyage")
|
||||
// TOOL OK: Answer: The current price is approximately $38.00 USD.
|
||||
// $38은 스니펫에 있던 "신조선 7일 항해 시 TEU 1개당 추가 연료비"였다. 실제 답은 항해당
|
||||
// 수백만 달러다. 그 턴은 모델이 결과 [1]을 직접 읽어 스스로 바로잡았지만, 그건 운이다.
|
||||
if (asset === 'generic') return '';
|
||||
|
||||
const candidates: Array<{ value: number; score: number; unit: 'ounce' | 'gram' | 'unknown' }> = [];
|
||||
for (const result of results) {
|
||||
const combined = `${result.title} ${result.snippet}`;
|
||||
// 자산명 대조를 가점이 아니라 **필수 조건**으로 올린다. 가점일 때는 자산을 한 번도
|
||||
// 언급하지 않은 스니펫의 숫자가 score 0으로 그대로 통과했다("금 시세" 질문에 다른
|
||||
// 상품 가격이 답으로 나갈 수 있다는 뜻이다).
|
||||
if (!ASSET_MENTION[asset].test(combined)) continue;
|
||||
const usdRaw = extractUsdPrice(combined);
|
||||
if (!usdRaw) continue;
|
||||
const usd = parseUsdNumber(usdRaw);
|
||||
@@ -141,12 +178,11 @@ function buildDirectPriceAnswer(
|
||||
const unit = detectPriceUnit(combined);
|
||||
const normalized = unit === 'gram' ? (usd * 31.1035) : usd;
|
||||
if (!isPlausibleUsdPrice(asset, normalized)) continue;
|
||||
let score = 0;
|
||||
let score = 2; // 자산명 일치(위에서 이미 강제)
|
||||
if (hasFreshPriceCue(combined)) score += 3;
|
||||
if (unit === 'ounce') score += 2;
|
||||
if (unit === 'gram') score += 1;
|
||||
if (hasHistoricalPriceCue(combined)) score -= 6;
|
||||
if (asset !== 'generic' && new RegExp(`\\b${asset}\\b`, 'i').test(combined)) score += 2;
|
||||
candidates.push({ value: normalized, score, unit });
|
||||
}
|
||||
|
||||
@@ -158,7 +194,6 @@ function buildDirectPriceAnswer(
|
||||
if (asset === 'bitcoin') return `Answer: The current Bitcoin price is approximately $${v} USD.`;
|
||||
if (asset === 'silver') return `Answer: The current silver price is approximately $${v} USD per ounce.`;
|
||||
if (asset === 'gold') return `Answer: The current gold price is approximately $${v} USD per ounce.`;
|
||||
return `Answer: The current price is approximately $${v} USD.`;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
/**
|
||||
* buildDirectPriceAnswer — 검색 결과 맨 앞에 붙는 합성 "Answer:" 줄.
|
||||
*
|
||||
* 이 줄이 왜 위험한지: 시스템 프롬프트의 ANTI-HALLUCINATION은 "도구가 돌려준 것을 정확히
|
||||
* 보고하라, 절대 무시하지 말라"고 지시한다. 여기서 틀리면 모델에게 틀린 값을 신뢰하라고
|
||||
* 시키는 셈이다. 게다가 합성된 숫자는 도구 출력 텍스트의 일부라, 모델이 그대로 옮기면
|
||||
* NUMERIC-GROUNDING이 "소스에 있다"며 통과시킨다 — 우리가 만든 숫자가 근거 도장을 받는다.
|
||||
*
|
||||
* 실제 사고(프로덕션 로그, 2026-08):
|
||||
* USER: 컨테이너선 급유비가 얼마나 될까?
|
||||
* TOOL: web_search("average fuel cost for 20,000 TEU container ship per voyage")
|
||||
* TOOL OK: Answer: The current price is approximately $38.00 USD.
|
||||
* $38은 "신조선 7일 항해 시 TEU 1개당 추가 연료비"였고, 실제 답은 항해당 수백만 달러다.
|
||||
*/
|
||||
|
||||
import { test, describe } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { buildDirectPriceAnswer, isPriceQuery } from '../src/tools/web';
|
||||
|
||||
describe('isPriceQuery — 통화·가격 낱말이 다른 단어 안에서 걸리면 안 된다', () => {
|
||||
test('단어 경계가 없어 오탐하던 실사용 검색어들', () => {
|
||||
// 실사용 고유 검색어 195건 중 19건이 통과했고, 아래는 그중 가격과 무관한 것들이다.
|
||||
for (const q of [
|
||||
'Europe drought status August 2026', // Eur(ope)
|
||||
'Neuralink Blindsight resolution pixels', // N(eur)alink
|
||||
'European weather outlook August 2026 precipitation',
|
||||
'costume design ideas', // (cost)ume
|
||||
'valuable lesson', // (value)able
|
||||
]) {
|
||||
assert.equal(isPriceQuery(q), false, q);
|
||||
}
|
||||
});
|
||||
|
||||
test('진짜 가격 질문은 계속 잡는다', () => {
|
||||
for (const q of ['gold price today', 'bitcoin usd quote', 'RTX 4090 current price', 'average fuel cost for container ship']) {
|
||||
assert.equal(isPriceQuery(q), true, q);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('buildDirectPriceAnswer — 무엇의 가격인지 모르면 답을 만들지 않는다', () => {
|
||||
const R = (title: string, snippet: string) => ({ title, url: 'https://example.invalid/x', snippet });
|
||||
|
||||
test('generic 자산에는 더 이상 "Answer:" 줄을 만들지 않는다 — 컨테이너선 사고의 직접 수정', () => {
|
||||
const out = buildDirectPriceAnswer(
|
||||
'average fuel cost for 20,000 TEU container ship per voyage',
|
||||
[R('Oil Prices and Container Shipping Costs', 'For a new ship, a 7-day voyage adds about $38 per TEU in fuel.')],
|
||||
);
|
||||
assert.equal(out, '');
|
||||
});
|
||||
|
||||
test('금 시세는 계속 만든다', () => {
|
||||
const out = buildDirectPriceAnswer(
|
||||
'gold price today',
|
||||
[R('Gold Price Today', 'Spot gold is currently trading at $2,410.50 per ounce.')],
|
||||
);
|
||||
assert.match(out, /Answer: The current gold price is approximately \$2,410\.50 USD per ounce\./);
|
||||
});
|
||||
|
||||
test('자산을 언급하지 않은 스니펫의 숫자는 쓰지 않는다 — 가점이 아니라 필수 조건', () => {
|
||||
const out = buildDirectPriceAnswer(
|
||||
'gold price today',
|
||||
[R('Weekly commodity roundup', 'Copper settled at $4,200.00 per tonne on light volume.')],
|
||||
);
|
||||
assert.equal(out, '');
|
||||
});
|
||||
|
||||
test('과거 시세 단서가 있으면 배제한다', () => {
|
||||
const out = buildDirectPriceAnswer(
|
||||
'gold price today',
|
||||
[R('Gold in 1980', 'Gold was worth $850.00 per ounce back in 1980, a historical peak.')],
|
||||
);
|
||||
assert.equal(out, '');
|
||||
});
|
||||
|
||||
test('가격 질문이 아니면 아무것도 만들지 않는다', () => {
|
||||
assert.equal(buildDirectPriceAnswer('Europe drought status', [R('Drought', 'Damage reached $38 billion.')]), '');
|
||||
});
|
||||
|
||||
test('결과가 없어도 터지지 않는다', () => {
|
||||
assert.equal(buildDirectPriceAnswer('gold price today', []), '');
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user