fix: 메인챗 두 가지 — 검색어 세대 추측 + "잠시만 기다려" 후 정지
2026-09-01 맥미니 M5/M6 세션 사후분석에서 나온 두 버그.
1. 세대 추측 검색어 오염: 사용자가 "맥미니 신형"(칩 미지정)이라 물었는데
gemma4가 학습시점 지식으로 web_search("M4 Mac mini …")를 던졌고, SearXNG는
시킨 대로 M4(2024) 기사를 정확히 반환 → 답 전체가 구세대 기준으로 틀어짐.
- system-prompt TEMPORAL CONSISTENCY에 "신형/최신인데 세대 미지정이면
쿼리에 기억 속 세대를 넣지 말고 현재 연도로 검색" 규칙
- handle-chat: hasUnversionedNewestIntent + assumedGenerationTokenInQuery로
감지해 1회 재프롬프트(오염된 검색 실행 전 차단)
2. "잠시만 기다려 주세요" 후 턴 종료: 사용자가 오류 지적하자 모델이
"다시 확인해 보겠습니다. 잠시만 기다려 주세요."만 내고 도구 0개로 정지.
isIntentOnlyReply는 답 속 "2026년 9월"을 실질내용으로 오인, lastRoundWas
GuardReprompt도 false라 기존 넛지 미발동.
- isDeferralPromiseReply 신규(좁은 "wait for me" 패턴), 가드 재프롬프트
선행 없이도 1회 강제 continuation
tests/intent-only-reply.test.ts +12 케이스. 전체 456 통과.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01XZX9JHFSLZRuK9rR4sCVBs
This commit is contained in:
@@ -17,6 +17,9 @@ import {
|
||||
looksLikeSafetyRefusal,
|
||||
hasConcreteCompletion,
|
||||
isIntentOnlyReply,
|
||||
isDeferralPromiseReply,
|
||||
hasUnversionedNewestIntent,
|
||||
assumedGenerationTokenInQuery,
|
||||
isMessagingRequest,
|
||||
claimsMessageSent,
|
||||
isBrowserToolName,
|
||||
@@ -311,6 +314,7 @@ async function handleChat(
|
||||
const MAX_NUMERIC_GROUNDING_FORCED_RETRIES = 1;
|
||||
let searchAbandonmentForcedRetries = 0;
|
||||
const MAX_SEARCH_ABANDONMENT_FORCED_RETRIES = 1;
|
||||
let staleVersionQueryNudges = 0;
|
||||
// Uncapped-round, unconditional-tool-history version of the round-0 tool-skip recovery
|
||||
// below: a model that calls one irrelevant tool (e.g. coder_list_files while answering a
|
||||
// GPU spec question) must not permanently disarm the safety net for the rest of the turn.
|
||||
@@ -1727,13 +1731,21 @@ async function handleChat(
|
||||
// 재시도 예산이 소진된 상태라 그 예고문이 그대로 최종 답변으로 나갔다. 사용자에게는
|
||||
// 계산하다 멈춘 것으로 보인다. 방금 가드가 개입한 턴에서만 보므로 평상시 답변은
|
||||
// 건드리지 않는다 — 옛 버전이 불필요한 도구 호출을 늘렸던 게 그 넓이 때문이었다.
|
||||
// "잠시만 기다려 주세요" 류 예고-후-정지: 직전에 가드 재프롬프트가 없었어도 잡는다.
|
||||
// 2026-09-01 맥미니 세션 — 사용자가 검색 오류를 지적하자 모델이 "다시 확인해 보겠습니다.
|
||||
// 잠시만 기다려 주세요."만 내놓고 도구 0개로 턴 종료. isIntentOnlyReply 는 답 속 "2026년
|
||||
// 9월"을 실질내용으로 오인해 통과시켰고, lastRoundWasGuardReprompt 도 false였다.
|
||||
const shouldNudgeDeferralPromise =
|
||||
intentOnlyNudges < 1
|
||||
&& isDeferralPromiseReply(candidateText)
|
||||
&& !isExecutionLikeRequest(message);
|
||||
const shouldNudgeIntentOnly =
|
||||
lastRoundWasGuardReprompt
|
||||
&& intentOnlyNudges < 1
|
||||
&& isIntentOnlyReply(candidateText);
|
||||
if (shouldNudgeIntentOnly) {
|
||||
if (shouldNudgeIntentOnly || shouldNudgeDeferralPromise) {
|
||||
intentOnlyNudges++;
|
||||
console.log(`[v2] INTENT-ONLY POST-CHECK: guard re-prompt answered with an announcement and no tool call. Nudging once...`);
|
||||
console.log(`[v2] INTENT-ONLY POST-CHECK: ${shouldNudgeDeferralPromise ? 'deferral promise ("잠시만 기다려…")' : 'guard re-prompt announcement'} with no tool call. Nudging once...`);
|
||||
sendSSE('info', { message: '모델이 실행 예고만 했습니다 — 실제 실행을 요청합니다...' });
|
||||
messages.push({ role: 'assistant', content: candidateText });
|
||||
messages.push({
|
||||
@@ -2171,6 +2183,31 @@ async function handleChat(
|
||||
};
|
||||
}
|
||||
|
||||
// Stale-generation search query: user asked about the "newest/신형" X without naming a
|
||||
// generation, but the model baked a specific generation token into its search query
|
||||
// (2026-09-01: "맥미니 신형" → web_search("M4 Mac mini …"), which correctly returned
|
||||
// old M4 coverage and derailed the whole thread). Re-prompt once to search neutrally.
|
||||
if (staleVersionQueryNudges < 1 && hasUnversionedNewestIntent(message)) {
|
||||
const searchCall = toolCalls.find((c: any) => /^(web_search|news_search|ollama_web_search)$/.test(String(c?.function?.name || '')));
|
||||
if (searchCall) {
|
||||
let q = '';
|
||||
try { q = String(normalizeToolArgs(searchCall.function?.arguments)?.query || ''); } catch { /* ignore */ }
|
||||
const tok = q && assumedGenerationTokenInQuery(q, message, recentUserText);
|
||||
if (tok) {
|
||||
staleVersionQueryNudges++;
|
||||
const curYear = new Date().getFullYear();
|
||||
console.log(`[v2] STALE-VERSION QUERY: user asked for the newest model without a generation; query assumed "${tok}" ("${q}"). Re-prompting to search neutrally.`);
|
||||
sendSSE('info', { message: `검색어가 세대(${tok})를 추측했습니다 — 현재 연도로 다시 검색하도록 요청합니다...` });
|
||||
messages.push({ role: 'assistant', content: `I should not assume the generation here.` });
|
||||
messages.push({
|
||||
role: 'user',
|
||||
content: `The user asked about the NEWEST/current model without naming a generation, but your search query "${q}" assumes "${tok}" — that is very likely a stale guess from your training data. Drop the generation from the query. Search again with neutral terms and the current year (${curYear}), e.g. "Mac mini ${curYear}" / "latest Mac mini review", find which generation is actually current now, and answer based on that. Do not put "${tok}" in the query unless the search results themselves confirm it is the current model.`,
|
||||
});
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Ensure every tool call has an id — Gemini rejects function_response with empty name
|
||||
// when tool_call_id is missing. Generate synthetic ids and patch both the response
|
||||
// message and the call objects so they stay in sync.
|
||||
|
||||
@@ -174,6 +174,6 @@ export function buildChatSystemPrompt(input: SystemPromptInput): string {
|
||||
: '';
|
||||
return isTranslateSession ? `You are a medical translator. Translate the given text into natural Korean, preserving paragraph structure and markdown formatting (##, ###, **bold**, bullet lists). Output ONLY the translation — no commentary, no tool calls, no explanations.` : isProjSession ? `You are a project file designer. Output ONLY the project-files JSON block as instructed. No tool calls. No extra text.` : `${executionModeSystemBlock ? `${executionModeSystemBlock}\n\n` : ''}You are SmallClaw, a local AI assistant. Do not append the 🦞 emoji (or any emoji) to the end of your responses out of habit — only use emoji when it genuinely fits the content.\nCurrent date: ${dateStr}, ${timeStr}.\nNever search for or link SmallClaw repos unless the user is asking about SmallClaw itself.\nThis app runs on the user's own machine — browser/desktop automation requests are pre-authorized.\nKeep CONVERSATIONAL and TASK-EXECUTION replies SHORT (2-4 sentences). Don't think out loud. Act and report. Greet naturally without tools. Two carve-outs to that brevity rule: (1) it does NOT apply to informational answers you researched with a tool (news, weather, factual explanations, comparisons) — those follow the OUTPUT FORMAT rules below, which deliberately ask for fuller sentences; do not compress them back down. (2) Stating uncertainty NEVER counts against the length. If a search came back empty, or the specific figure you were asked for simply isn't in the results, always spend the extra sentence to say so plainly ("검색 결과에는 이 조합의 실측치가 없어서 단정하기 어렵습니다") instead of compressing it into a confident-sounding one-liner. Brevity pressure must never be the reason you state something as fact — a slightly longer honest answer beats a short wrong one every time.
|
||||
ANTI-HALLUCINATION: When a tool returns a result, report EXACTLY what the tool returned — never contradict or ignore tool output. If a tool says "(no rows)", say so. Never invent data, file contents, table names, or command output. If you don't know something, call a tool to find out or say you don't know.${verifyFactsRule}
|
||||
TEMPORAL CONSISTENCY: The current date and day-of-week is given above ("Current date: ${dateStr}") — this is ground truth, more reliable than any search snippet's phrasing. Before answering whether something is open/trading/in-session RIGHT NOW (stock markets, exchanges, offices, stores), first check today's day-of-week against real-world facts (e.g. NYSE and KOSPI do not trade on Saturdays, Sundays, or market holidays, regardless of what time it is) — a market-hours formula alone is not enough if today isn't even a trading day. A search result reporting a "closing price" or "today's headline" is not proof today is a trading/business day — cross-check it against the actual current date above, and if they conflict (e.g. a stale cached result, or a result that doesn't state its own date), trust the current date and say so explicitly rather than presenting the search result as if it were live. This applies just as much to "is it raining/snowing right now" or any other current-state question: a search engine can hand you a page it cached long ago. Before treating a search result as live, check whether it carries its own timestamp — a date in the URL (e.g. "tm=2024.12.13.20:00", "?date=..."), a dateline, or "as of ..." phrasing — and compare it to the current date above. If that timestamp is more than a day or two old, it is NOT "지금"/"현재": say plainly that you couldn't find live data and name the stale date you found instead of presenting old numbers as current. When a result is a data TABLE (observation/measurement tables are the common case), read every row by its own row label, not by position — misaligning one city's or one row's figures onto the next is worse than finding nothing, because it looks authoritative while being wrong. If you're not confident you're reading the right column for the right label, say so rather than guessing.${imageEditRuleBlock}${mediaGenRuleBlock}${browserRuleBlock}${chemistryRuleBlock}${mapsRuleBlock}${codeOutputRuleBlock}${packageInstallRuleBlock}
|
||||
TEMPORAL CONSISTENCY: The current date and day-of-week is given above ("Current date: ${dateStr}") — this is ground truth, more reliable than any search snippet's phrasing. Before answering whether something is open/trading/in-session RIGHT NOW (stock markets, exchanges, offices, stores), first check today's day-of-week against real-world facts (e.g. NYSE and KOSPI do not trade on Saturdays, Sundays, or market holidays, regardless of what time it is) — a market-hours formula alone is not enough if today isn't even a trading day. A search result reporting a "closing price" or "today's headline" is not proof today is a trading/business day — cross-check it against the actual current date above, and if they conflict (e.g. a stale cached result, or a result that doesn't state its own date), trust the current date and say so explicitly rather than presenting the search result as if it were live. This applies just as much to "is it raining/snowing right now" or any other current-state question: a search engine can hand you a page it cached long ago. Before treating a search result as live, check whether it carries its own timestamp — a date in the URL (e.g. "tm=2024.12.13.20:00", "?date=..."), a dateline, or "as of ..." phrasing — and compare it to the current date above. If that timestamp is more than a day or two old, it is NOT "지금"/"현재": say plainly that you couldn't find live data and name the stale date you found instead of presenting old numbers as current. When a result is a data TABLE (observation/measurement tables are the common case), read every row by its own row label, not by position — misaligning one city's or one row's figures onto the next is worse than finding nothing, because it looks authoritative while being wrong. If you're not confident you're reading the right column for the right label, say so rather than guessing. When the user asks about the "newest"/"latest"/"신형"/"최신" version of a product WITHOUT naming a specific generation, do NOT put a generation you remember (e.g. "M4", "RTX 4090", "iPhone 15") into your search query — whatever was newest when you were trained is probably not newest now. Search with the current year and neutral terms ("Mac mini ${new Date().getFullYear()}", "latest Mac mini"), let the results tell you which generation is current, and answer about that one. If the results still look like an older generation than the user implies, say so and search again rather than reporting the old model as new.${imageEditRuleBlock}${mediaGenRuleBlock}${browserRuleBlock}${chemistryRuleBlock}${mapsRuleBlock}${codeOutputRuleBlock}${packageInstallRuleBlock}
|
||||
OUTPUT FORMAT: When presenting 3+ items (emails, search results, lists), always use a markdown table or structured bullet list with clear headers. Never dump them as a long paragraph. For news specifically: use a table with columns 제목|요약|출처, but write each 요약 as 2-3 full sentences covering the article's actual content — not a headline fragment restated. Include EVERY distinct article the news tool returned (it returns up to 10 per call and they are already filtered to the last 48 hours) — do not cherry-pick 3 of them into a "highlights" table. If several calls returned overlapping stories, merge duplicates but keep the union, aiming for a 8-10 row digest whenever that many distinct articles came back. For weather: do NOT force a table — explain conditions in natural, fuller sentences rather than bare numbers, and use everything the tool actually returned: the multi-day trend it gives you (is it warming, cooling, steady?), precipitation chance, and anything notable. When comparing several cities (e.g. "오늘 유럽 대도시 최고 기온"), a table is fine for the numbers themselves, but always follow it with a few sentences of actual commentary — which city is hottest/coldest and by how much, any city with a notably different trend (rain vs. dry, warming vs. cooling) than the rest, anything the tool flagged as unusual. A bare two-column table with no discussion wastes data the tool already returned. ${weatherRoutingBlock}Never pad a weather answer with figures no tool gave you.${modelProfileBlock}${callerContext ? '\n\n' + callerContext : ''}${browserStateCtx}${personalityCtx}${skillsCtx}`;
|
||||
}
|
||||
|
||||
@@ -418,3 +418,49 @@ export function isIntentOnlyReply(text: string): boolean {
|
||||
const hasSubstance = /^[\s]*[-*]\s|\n[\s]*[-*]\s|\|.*\||```|https?:\/\/|\d[\d,.]*\s*(원|톤|kg|km|mm|%|개|명|년|월|일|시간|분|초|t|ton)/i.test(s);
|
||||
return !hasSubstance;
|
||||
}
|
||||
|
||||
/**
|
||||
* "잠시만 기다려 주세요" 류 — 후속 작업을 약속하고 턴을 끝낸 답변.
|
||||
*
|
||||
* isIntentOnlyReply 보다 좁고, 그것과 달리 **직전에 가드 재프롬프트가 없어도** 본다.
|
||||
* 2026-09-01 실사례(맥미니 M5/M6 세션): 사용자가 검색 결과 오류를 지적하자 모델이
|
||||
* "…다시 정확하게 확인해 보겠습니다. 잠시만 기다려 주세요." 만 내놓고 도구 호출 0개로 턴 종료.
|
||||
* 채팅 어시스턴트가 "곧 알려드릴게요"로 턴을 끝내는 건 거의 항상 버그다 — 답이 있으면 바로 준다.
|
||||
* isIntentOnlyReply 는 답 속의 "2026년 9월" 같은 우연한 날짜를 hasSubstance 로 오인해 놓쳤다.
|
||||
*/
|
||||
export function isDeferralPromiseReply(text: string): boolean {
|
||||
const s = String(text || '').trim();
|
||||
if (!s || s.length > 600) return false;
|
||||
return /잠(시|깐)\s*(만)?\s*(기다려|대기|후에|뒤에|만요)|기다려\s*주(세요|십시오)|곧\s*(알려|말씀|정리해|공유)|잠시\s*후\s*(다시|알려)|조금만\s*기다|hold on\b|one moment\b|give me a (moment|sec|minute)|bear with me|be right back|will (get back|come back) to you|let me get back to you/i.test(s);
|
||||
}
|
||||
|
||||
/**
|
||||
* "최신/신형"을 묻지만 세대·버전을 특정하지 않은 질문인가.
|
||||
* 이런 질문에 모델이 학습 시점 지식으로 세대를 넣어 검색하면(예: 사용자는 "맥미니 신형"인데
|
||||
* 쿼리는 "M4 Mac mini") 검색엔진이 구세대 결과를 정확히 돌려줘 답 전체가 틀어진다.
|
||||
*/
|
||||
export function hasUnversionedNewestIntent(message: string): boolean {
|
||||
const m = String(message || '');
|
||||
if (!/신형|신제품|새\s*모델|새로\s*나온|이번에\s*나온|나온\s*지\s*얼마|최신(형|\s*(모델|버전|제품|칩|세대))?|latest|newest|new(est)?\s+(model|version|release|chip|one)/i.test(m)) return false;
|
||||
// 사용자가 세대·버전을 직접 댔으면 대상 아님 (M5, RTX 5090, 아이폰 16, 3세대, gen 5 …)
|
||||
const explicitVersion = /\b[A-Za-z]{1,5}[- ]?\d{1,4}\b|\d{1,2}\s*세대|\bgen(eration)?\s?\d/i.test(m);
|
||||
return !explicitVersion;
|
||||
}
|
||||
|
||||
/**
|
||||
* web_search/news_search 쿼리에 든, 사용자가 이 대화에서 말한 적 없는 세대·모델 토큰.
|
||||
* "M4", "A17", "S24" 처럼 글자+숫자가 붙은 형태만 본다(연도 4자리, 공백 낀 "RTX 5090"은 제외 —
|
||||
* 후자는 사용자가 보통 직접 부르고, 스펙 숫자 할루시네이션 가드가 따로 커버한다).
|
||||
*/
|
||||
export function assumedGenerationTokenInQuery(query: string, ...userTexts: string[]): string | null {
|
||||
const q = String(query || '');
|
||||
const known = userTexts.join(' ').toLowerCase().replace(/[-\s]/g, '');
|
||||
const re = /\b([A-Za-z]{1,4}\d{1,3})\b/g;
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = re.exec(q))) {
|
||||
const tok = match[1];
|
||||
if (/^\d/.test(tok)) continue;
|
||||
if (!known.includes(tok.toLowerCase())) return tok;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -14,7 +14,12 @@
|
||||
|
||||
import { test, describe } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { isIntentOnlyReply } from '../src/gateway/guards/prompt-gates';
|
||||
import {
|
||||
isIntentOnlyReply,
|
||||
isDeferralPromiseReply,
|
||||
hasUnversionedNewestIntent,
|
||||
assumedGenerationTokenInQuery,
|
||||
} from '../src/gateway/guards/prompt-gates';
|
||||
|
||||
describe('isIntentOnlyReply', () => {
|
||||
test('실제로 멈췄던 그 응답을 잡는다', () => {
|
||||
@@ -50,3 +55,58 @@ describe('isIntentOnlyReply', () => {
|
||||
assert.equal(isIntentOnlyReply(long), false);
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* 2026-09-01 맥미니 M5/M6 세션: 사용자가 검색 결과 오류를 지적하자 모델이
|
||||
* "…다시 정확하게 확인해 보겠습니다. 잠시만 기다려 주세요."만 내놓고 도구 0개로 턴 종료.
|
||||
* isIntentOnlyReply 는 답 속 "2026년 9월"을 hasSubstance 로 오인해 놓쳤고,
|
||||
* lastRoundWasGuardReprompt 도 false 라 기존 넛지가 안 걸렸다.
|
||||
*/
|
||||
describe('isDeferralPromiseReply', () => {
|
||||
test('실제로 멈췄던 그 응답을 잡는다 — 우연한 날짜가 섞여 있어도', () => {
|
||||
assert.equal(isDeferralPromiseReply(
|
||||
'죄송합니다. 제가 최신 출시 정보를 놓쳤습니다. 현재 시점(2026년 9월) 기준으로 M5, M6 칩을 탑재한 신형 맥미니가 '
|
||||
+ '출시된 상황이군요. 최신 모델인 M5, M6 맥미니의 시장 반응을 다시 정확하게 확인해 보겠습니다. 잠시만 기다려 주세요.'), true);
|
||||
});
|
||||
|
||||
test('영문 "get back to you" 도 잡는다', () => {
|
||||
assert.equal(isDeferralPromiseReply('Let me get back to you on that.'), true);
|
||||
assert.equal(isDeferralPromiseReply('One moment — checking now.'), true);
|
||||
});
|
||||
|
||||
test('바로 실행하겠다는 예고는 아니다 (기존 intent-only 경로가 처리)', () => {
|
||||
assert.equal(isDeferralPromiseReply('알겠습니다. 바로 검색해서 정리해 드리겠습니다.'), false);
|
||||
});
|
||||
|
||||
test('정상 답변은 아니다', () => {
|
||||
assert.equal(isDeferralPromiseReply('네, RTX 5090의 대역폭은 약 1790 GB/s입니다.'), false);
|
||||
assert.equal(isDeferralPromiseReply(''), false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('hasUnversionedNewestIntent / assumedGenerationTokenInQuery', () => {
|
||||
test('세대 없이 "신형/최신"을 물으면 true', () => {
|
||||
assert.equal(hasUnversionedNewestIntent('맥미니 신형이 나왔는데 시장반응은 어때?'), true);
|
||||
assert.equal(hasUnversionedNewestIntent('아이폰 최신 모델 가격'), true);
|
||||
});
|
||||
|
||||
test('사용자가 세대를 직접 대면 false', () => {
|
||||
assert.equal(hasUnversionedNewestIntent('mac m5 메모리 대역폭이 얼마나 높아졌지?'), false);
|
||||
assert.equal(hasUnversionedNewestIntent('RTX 5090 최신 가격'), false);
|
||||
});
|
||||
|
||||
test('신형 얘기가 아니면 false', () => {
|
||||
assert.equal(hasUnversionedNewestIntent('서브프라임 사태 원인이 뭐야?'), false);
|
||||
});
|
||||
|
||||
test('쿼리가 사용자가 안 말한 세대를 박으면 그 토큰을 돌려준다', () => {
|
||||
assert.equal(
|
||||
assumedGenerationTokenInQuery('M4 Mac mini release market reaction reviews', '맥미니 신형이 나왔는데 시장반응은 어때?', ''),
|
||||
'M4');
|
||||
});
|
||||
|
||||
test('사용자가 말한 세대거나 연도만 있으면 null', () => {
|
||||
assert.equal(assumedGenerationTokenInQuery('Apple M5 chip memory bandwidth specs', 'mac m5 메모리 대역폭', ''), null);
|
||||
assert.equal(assumedGenerationTokenInQuery('Mac mini 2026 latest review', '맥미니 신형', ''), null);
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user