From a5fb92d477dd8b75722b2503286585458c39c0cf Mon Sep 17 00:00:00 2001 From: kim Date: Thu, 16 Jul 2026 12:58:57 +0900 Subject: [PATCH] =?UTF-8?q?v4.1.10:=20=EC=9B=B9=EA=B2=80=EC=83=89/?= =?UTF-8?q?=EB=8F=84=EA=B5=AC=ED=98=B8=EC=B6=9C=20=EC=8B=A0=EB=A2=B0?= =?UTF-8?q?=EC=84=B1=20=EB=8C=80=EA=B0=9C=EC=84=A0=20=E2=80=94=20=EB=AA=A8?= =?UTF-8?q?=EB=8D=B8=EB=B3=84=20=ED=94=84=EB=A1=9C=ED=95=84=20=EC=8B=9C?= =?UTF-8?q?=EC=8A=A4=ED=85=9C=20=EB=8F=84=EC=9E=85?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - AUTO-RECOVER 구조적 허점 수정: "라운드 0 & 도구 0회 호출"일 때만 발동하던 걸 "이 요청에 필요한 도구가 아직 안 불렸으면" 언제든 발동하도록 변경 — 엉뚱한 도구 하나만 불러도 안전장치가 통째로 무장해제되던 버그 수정 - 뉴스/날씨/스펙·가격·비교 등 팩트성 질문에 도구 호출 강제(isLiveDataRequest/isFactualInfoRequest) — 확신에 차서 지어내는 것(mistral-large-3에서 다발, 없는 "2026 파리 올림픽" 등)을 사전 넛지+사후 강제재시도로 이중 차단 - web_search/web_fetch 턴당 5회 하드캡 추가 — kimi-k2.6이 단순 뉴스요청에 최대 50회까지 검색하던 문제 차단 - 모델이 함수 호출 대신 텍스트로 tool call을 흉내내는 케이스(`web_search{...}`, 반복시 토큰 열화로 `웍웍웍_search`, 가비지 토큰 삽입 `web_search Räikk{...}`) 복구 로직을 브레이스 앞 텍스트에서 실제 도구명을 찾는 방식으로 일반화 - 뉴스 검색: 대화 이력 주제로 새는 것 방지, 국제뉴스는 영어 쿼리 사용, 위키/홈페이지뿐인 저품질 결과는 다음 provider(tavily 등)로 폴백, 기사 실제 게재일 라벨링(오늘 날짜로 잘못 표기 금지) - email_send/kakao_send_message 실제 호출 없이 "보냈습니다"라고 답하면 강제 재시도 - browser_open: image_edit와 동일하게 명시 요청 없으면 코드 레벨 차단(기존엔 프롬프트 텍스트뿐) - config.json models.profiles로 모델별 시스템프롬프트 추가지침 주입 가능(코드 재배포 없이 모델별 이상행동 교정) - ollama-client 모델 fallback 재시도를 quota 초과 외 retired/deprecated 에러까지 확장, models.fallback 값 유지 - 각종 인텐트 판별 함수(브라우저/데스크톱/실행형 요청, 백그라운드 작업 재개/취소/상태질문)에 한국어 키워드 커버리지 추가 - shell 툴 스키마 예시와 PACKAGE INSTALL 프롬프트 규칙의 모순(pip install 예시 vs 금지 규칙) 제거 Co-Authored-By: Claude Sonnet 5 --- package.json | 2 +- src/agents/ollama-client.ts | 21 +-- src/gateway/server-v2.ts | 263 ++++++++++++++++++++++++++++++++---- src/tools/web.ts | 36 ++++- 4 files changed, 283 insertions(+), 39 deletions(-) diff --git a/package.json b/package.json index 7ad558b..6c780d9 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "smallclaw", - "version": "4.1.9", + "version": "4.1.10", "description": "Local AI agent framework powered by Ollama - OpenClaw alternative", "main": "dist/index.js", "bin": { diff --git a/src/agents/ollama-client.ts b/src/agents/ollama-client.ts index 32d56a3..3a38502 100644 --- a/src/agents/ollama-client.ts +++ b/src/agents/ollama-client.ts @@ -13,13 +13,16 @@ import { getProvider, getModelForRole, getPrimaryModel, getFallbackModel, resetP import type { LLMProvider, TokenUsage } from '../providers/LLMProvider'; import { AgentRole } from '../types'; -// Ollama Cloud's per-account session quota error. Matched case-insensitively -// against the thrown error message to decide whether a fallback retry makes -// sense (as opposed to e.g. a malformed request, which retrying won't fix). +// Errors where retrying with the fallback model can actually help: Ollama Cloud's +// per-account session quota error, and a cloud model being retired/renamed/removed +// out from under a hardcoded or configured model name (as opposed to e.g. a malformed +// request, which retrying won't fix). const QUOTA_ERROR_RE = /session usage limit/i; +const MODEL_UNAVAILABLE_RE = /\b(was retired|is deprecated|has been deprecated|model not found|does not exist|no such model|unknown model)\b/i; -function isQuotaError(err: any): boolean { - return QUOTA_ERROR_RE.test(String(err?.message || err || '')); +function isRetryableModelError(err: any): boolean { + const msg = String(err?.message || err || ''); + return QUOTA_ERROR_RE.test(msg) || MODEL_UNAVAILABLE_RE.test(msg); } export interface GenerateOutput { @@ -66,8 +69,8 @@ export class OllamaClient { return { message: result.message, thinking: result.thinking }; } catch (err: any) { const fallback = getFallbackModel(); - if (fallback && fallback !== model && isQuotaError(err)) { - console.warn(`[OllamaClient] ${model} hit quota limit, retrying with fallback model ${fallback}`); + if (fallback && fallback !== model && isRetryableModelError(err)) { + console.warn(`[OllamaClient] ${model} unavailable (quota or retirement), retrying with fallback model ${fallback}`); const result = await this.provider.chat(messages, fallback, chatOpts); return { message: result.message, thinking: result.thinking }; } @@ -119,8 +122,8 @@ export class OllamaClient { try { result = await this.provider.chat(messages, modelToUse, chatOpts); } catch (err: any) { - if (fallbackModel && fallbackModel !== modelToUse && isQuotaError(err)) { - console.warn(`[OllamaClient] ${modelToUse} hit quota limit, retrying with fallback model ${fallbackModel}`); + if (fallbackModel && fallbackModel !== modelToUse && isRetryableModelError(err)) { + console.warn(`[OllamaClient] ${modelToUse} unavailable (quota or retirement), retrying with fallback model ${fallbackModel}`); modelToUse = fallbackModel; result = await this.provider.chat(messages, modelToUse, chatOpts); } else { diff --git a/src/gateway/server-v2.ts b/src/gateway/server-v2.ts index 878661b..35f885a 100644 --- a/src/gateway/server-v2.ts +++ b/src/gateway/server-v2.ts @@ -1308,7 +1308,7 @@ function detectToolCategories(text: string, sessionId?: string): Set { // Tool rule blocks — compact, injected only when relevant const TOOL_BLOCKS: Record = { - web: `WEB TOOLS: web_search(query) → headlines+snippets. web_fetch(url) → full page text. Use web_search first to get URLs, then web_fetch to read. For Reddit: web_search with site:reddit.com "keyword", then web_fetch post URLs — never open browser for Reddit. IMPORTANT: NEVER use web_fetch or web_search for local URLs (localhost, 127.0.0.1, /api/files/...) — they are NOT web pages. Local files are already accessible in chat via ![alt](/api/files/name.png) or [name](/api/files/name.pptx).`, + web: `WEB TOOLS: web_search(query) → headlines+snippets. web_fetch(url) → full page text. Use web_search first to get URLs, then web_fetch to read. For Reddit: web_search with site:reddit.com "keyword", then web_fetch post URLs — never open browser for Reddit. IMPORTANT: NEVER use web_fetch or web_search for local URLs (localhost, 127.0.0.1, /api/files/...) — they are NOT web pages. Local files are already accessible in chat via ![alt](/api/files/name.png) or [name](/api/files/name.pptx). SEARCH BUDGET: for any single user request, use at most 5 web_search/web_fetch calls combined — pick the most relevant angles up front rather than searching topic-by-topic one at a time. Once you have enough to answer, stop searching and write the answer with what you have; do not keep searching for full/exhaustive coverage. NEWS QUERY STRATEGY: for Korean domestic news, a Korean-language query works well (e.g. "2026년 7월 16일 국내 뉴스 헤드라인"). For international/world news, the underlying search engines index English far better — search in English (e.g. "international news today", "world news headlines July 16 2026") rather than a Korean query, and do NOT bolt a single news outlet name onto the query (e.g. "... CNN", "... Reuters") — that tends to return the outlet's generic homepage instead of an actual headline. If your first search returns only a generic reference page (e.g. a Wikipedia year-overview) or a homepage instead of real headlines, that query failed — try a different phrasing rather than repeating close variants of the same failing query. NEWS DATING: search results carry their OWN publish date (visible in the snippet or URL, e.g. "2026.07.13" or a dated URL path like /2026/07/14/) — use THAT date in your header/summary, never default to today's date just because the user asked "오늘"/"today". If the freshest result you found is older than today, say so explicitly up front (e.g. "7월 13일 기준 뉴스이며, 이후 업데이트는 확인되지 않았습니다") instead of presenting it under today's date — do not make the user catch this themselves.`, browser: `BROWSER TOOLS: browser_open(url) → opens+returns snapshot. browser_snapshot() → refresh. browser_click(ref) → click by @ref. browser_fill(ref,text) → fill input. browser_press_key(key) → Enter/Tab/Escape. browser_wait(ms) → wait+snapshot. browser_close() → close tab. Chrome profile is persistent. NEVER use browser_open for local /api/files/ URLs — those are already inline in chat. SNAPSHOT RULE: browser_open/fill/wait/click all return a snapshot automatically — do NOT call browser_snapshot after them; only call it when you haven't received a fresh snapshot recently.`, @@ -1845,11 +1845,11 @@ function buildTools() { type: 'function', function: { name: 'shell', - description: 'Execute a terminal command and return its output. Use this for running scripts (python, node), CLI tools (pip, npm, git), and any shell command. The command runs in the workspace directory. NEVER use heredoc syntax (<]/g, ''); } +function isLiveDataRequest(message: string): boolean { + const m = String(message || ''); + return /(뉴스|속보|날씨|기온|예보|미세먼지|황사|환율|주가|금리|시세|장마)/.test(m) + || /\b(news|weather|forecast|stock price|exchange rate)\b/i.test(m); +} + +// Broader than isLiveDataRequest: catches factual/product-info questions (specs, prices, +// versions, comparisons, "what models exist") where the model tends to answer fluently from +// memorized training data instead of checking — those numbers go stale or are just wrong +// (e.g. confidently inventing GPU TFLOPS/VRAM figures). Deliberately excludes anything that +// looks like a coding/build request or a greeting, since those aren't "look this up" asks. +function isFactualInfoRequest(message: string): boolean { + const m = String(message || '').trim(); + if (!m) return false; + if (isExecutionLikeRequest(m) || isGreetingLikeMessage(m)) return false; + return /\b(vs\.?|versus)\b/i.test(m) + || /(비교|차이점?|스펙|사양|성능|가격|버전|몇\s*(세대|개|년|만원)|추천해|어떤\s*(게|것|모델|제품|카드)|뭐가\s*있|무엇이\s*있|종류가)/.test(m); +} + +// Per-model behavioral overrides. Some models (e.g. mistral-large-3) have observed quirks — +// skipping tool calls and confidently fabricating instead — that the generic system prompt +// doesn't fully correct. Rather than hardcoding model names in prompt text, config.json's +// models.profiles[modelName].extraSystemPrompt lets us attach model-specific reminders that +// only apply when that exact model is the one actually serving the turn, discovered/edited +// without a code deploy. +function getModelProfileExtraPrompt(modelName: string): string { + const m = String(modelName || '').trim(); + if (!m) return ''; + try { + const raw = getConfig().getConfig() as any; + const extra = String(raw?.models?.profiles?.[m]?.extraSystemPrompt || '').trim(); + return extra ? `\nMODEL-SPECIFIC NOTE: ${extra}` : ''; + } catch { + return ''; + } +} + function looksLikeSafetyRefusal(text: string): boolean { const s = String(text || '').trim().toLowerCase(); if (!s) return false; @@ -4792,6 +4835,19 @@ function hasConcreteCompletion(text: string): boolean { return /\b(done|completed|created|updated|fixed|implemented|finished|saved|wrote|executed|here(?:'s| is) (?:the|your)|success(?:fully)?)\b/i.test(s); } +function isMessagingRequest(message: string): boolean { + const m = String(message || ''); + return (/(이메일|메일)/.test(m) && /(보내|전송|발송)/.test(m)) + || /(카카오톡?|카톡)/.test(m) && /(보내|전송|발송)/.test(m) + || /\b(send)\b.*\b(email|mail|kakao)\b/i.test(m); +} + +function claimsMessageSent(text: string): boolean { + const s = String(text || ''); + return /(보냈습니다|보냈어요|전송했습니다|전송했어요|발송했습니다|발송했어요|보내드렸습니다|전송\s*완료|발송\s*완료)/.test(s) + || /\b(sent the (email|message)|email (has been|was) sent|message (has been|was) sent)\b/i.test(s); +} + function isBrowserToolName(name: string): boolean { return /^browser_(open|snapshot|click|fill|press_key|wait|scroll|close)$/i.test(String(name || '')); } @@ -5417,6 +5473,13 @@ async function handleChat( let browserAdvisorHintPreview = ''; let browserForcedRetries = 0; let browserAdvisorCallsThisTurn = 0; + let messagingForcedRetries = 0; + const MAX_MESSAGING_FORCED_RETRIES = 2; + // Uncapped-round, unconditional-tool-history version of the round-0 tool-skip recovery + // below: a model that calls one irrelevant tool (e.g. coder_list_files while answering a + // GPU spec question) must not permanently disarm the safety net for the rest of the turn. + let toolSkipForcedRetries = 0; + const MAX_TOOL_SKIP_FORCED_RETRIES = 2; // Scroll-before-act gate: tracks whether a fill/click has happened yet this turn. // Blocks PageDown/scroll calls on interactive pages until the model actually acts. let browserFillOrClickDoneThisTurn = false; @@ -5782,17 +5845,21 @@ async function handleChat( const browserRuleBlock = hasBrowserTools ? '\nBROWSER RULE: NEVER call browser_open, browser_snapshot, browser_click, browser_fill, browser_scroll, or any browser_* tool unless the user EXPLICITLY asks to use the browser (e.g. "브라우저로 열어줘", "Chrome으로 봐줘", "사이트 직접 들어가봐"). "열어줘" or "보여줘" alone is NOT a browser request — answer from knowledge, use web_fetch/web_search, or output the relevant URL/embed. Pasting a URL or asking a question is also NOT a browser request.' : ''; + // Resolved with the same call chatWithThinkingStream makes internally for its first + // attempt. If that call later falls back to models.fallback mid-turn (quota/retirement), + // this won't retroactively pick up the fallback's profile for the same turn. + const modelProfileBlock = getModelProfileExtraPrompt(getModelForRole('executor')); const messages: any[] = [ { role: 'system', content: isTranslateSession ? `You are a medical translator. Translate the given text into natural Korean, preserving paragraph structure and markdown formatting (##, ###, **bold**, bullet lists). Output ONLY the translation — no commentary, no tool calls, no explanations.` : isProjSession ? `You are a project file designer. Output ONLY the project-files JSON block as instructed. No tool calls. No extra text.` : `${executionModeSystemBlock ? `${executionModeSystemBlock}\n\n` : ''}You are SmallClaw 🦞, a local AI assistant.\nCurrent date: ${dateStr}, ${timeStr}.\nNever search for or link SmallClaw repos unless the user is asking about SmallClaw itself.\nThis app runs on the user's own machine — browser/desktop automation requests are pre-authorized.\nKeep responses SHORT (1-2 sentences). Don't think out loud. Act and report. Greet naturally without tools. -ANTI-HALLUCINATION: When a tool returns a result, report EXACTLY what the tool returned — never contradict or ignore tool output. If a tool says "(no rows)", say so. Never invent data, file contents, table names, or command output. If you don't know something, call a tool to find out or say you don't know. +ANTI-HALLUCINATION: When a tool returns a result, report EXACTLY what the tool returned — never contradict or ignore tool output. If a tool says "(no rows)", say so. Never invent data, file contents, table names, or command output. If you don't know something, call a tool to find out or say you don't know. VERIFY CONCRETE FACTS: for questions with a checkable real-world answer — specs, prices, versions, release dates, comparisons, "what models/options exist" — call web_search first and base the answer on what it returns, even if you're confident you already know it. Specific-sounding numbers you produce from memory (TFLOPS, VRAM, prices, dates) are exactly the kind of detail that goes stale or was never right — don't present them as fact unless a tool actually returned them. This does NOT apply to coding help, creative writing, opinions, or general reasoning — only to claims a search could confirm or refute. IMAGE EDITING RULE: NEVER call image_edit (or any editing tool) when a user uploads a photo without explicitly requesting edits. Uploading a photo is NOT a request to edit it. Only call image_edit when the user's message explicitly asks for an edit (e.g. "수채화로 바꿔줘", "회전해줘"). Violating this rule is a critical error. IMAGE/VIDEO GENERATION: When a user asks to create/draw/generate a NEW image from a description (no existing photo involved), call image_generate (local SDXL, 10-20 seconds by default). If the user explicitly asks for higher quality/detail/photorealism (e.g. "고품질로", "디테일 살려서"), pass quality="high" instead — this uses FLUX.1-schnell, noticeably better detail but 40-60 seconds total, so tell the user it'll take a bit before calling it. When they ask for a short video/clip/animation from a description, call video_generate (local LTX-Video, 30–90 seconds — tell the user it'll take a bit before calling it). All run entirely on local GPU hardware, no external API or cost. Do not confuse these with image_edit, which only modifies an existing uploaded/generated image.${browserRuleBlock} CHEMISTRY NOTATION: NEVER draw molecular structures as ASCII art (H/C/#/=/\\/| characters arranged to look like a diagram) — it always renders as garbled, misaligned text. Instead: for a formula or reaction, use LaTeX inside $...$ (e.g. $\\ce{C4H10}$, $\\ce{2H2 + O2 -> 2H2O}$ — mhchem extension is loaded). For an actual 2D structure with real bond lines (rings, branches), output a \`\`\`smiles\`\`\` code block containing the SMILES string (e.g. \`\`\`smiles\\nc1ccccc1\\n\`\`\` for benzene, \`\`\`smiles\\nCC(=O)Oc1ccccc1C(=O)O\\n\`\`\` for aspirin) — the client automatically renders it as a proper 2D diagram with bond lines. CODE OUTPUT: When writing code in a fenced code block, start with a filename comment on line 1: \`# filename: snake_game.py\` (Python), \`// filename: app.js\` (JS/C), \`\` (HTML). Never repeat code already written in this conversation. For modifications to existing files, use coder_overwrite_lines or coder_insert_lines (not coder_write_file — it only works for NEW files). All code changes are presented as diffs for the user to review before being applied. Write code directly — do not ask for permission. PACKAGE INSTALL: NEVER run pip install, npm install, apt-get, or any package installation command. If a package is missing, just write the code and mention the package name in a comment — let the user decide whether to install it. Do NOT attempt to install packages yourself. -OUTPUT FORMAT: When presenting 3+ items (news articles, emails, search results, lists), always use a markdown table or structured bullet list with clear headers. Never dump them as a long paragraph. Example: news → table with columns 제목|요약|출처.${callerContext ? '\n\n' + callerContext : ''}${browserStateCtx}${personalityCtx}${skillsManager.buildPromptContextForUser(username ? getUserWorkspace(username) : null, 16000, message)}`, +OUTPUT FORMAT: When presenting 3+ items (news articles, emails, search results, lists), always use a markdown table or structured bullet list with clear headers. Never dump them as a long paragraph. Example: news → table with columns 제목|요약|출처.${modelProfileBlock}${callerContext ? '\n\n' + callerContext : ''}${browserStateCtx}${personalityCtx}${skillsManager.buildPromptContextForUser(username ? getUserWorkspace(username) : null, 16000, message)}`, }, ]; // Emit the fixed-overhead size (system prompt + tool schemas) as soon as both are known — @@ -5824,6 +5891,33 @@ OUTPUT FORMAT: When presenting 3+ items (news articles, emails, search results, } messages.push({ role: 'user', content: resolveImageContent(message, workspacePath) }); + // News/weather/prices/etc. must come from a live tool call, never training data — and + // factual/product-info questions (specs, comparisons, "what models exist") should be + // verified against a real web search rather than answered from memorized (and often stale + // or simply wrong) training data. Nudge this up front (rather than only catching it after a + // wasted first generation — see the AUTO-RECOVER fallback below) so the model calls the + // tool on round 0 instead of burning a full generation on a fabricated answer that then + // gets discarded and retried. + if (isLiveDataRequest(message)) { + // "뉴스" alone means general current-events news — it must not get resolved into + // whatever topic happens to be freshest in the conversation history (observed: a bare + // "뉴스 다시 정리해줘" got interpreted as "updates on the GPUs/Ollama models we were just + // discussing" instead of actual world news, once the model was forced to search for + // something and grabbed the nearest available anchor). + const isNewsRequest = /(뉴스|속보)/.test(message) || /\bnews\b/i.test(message); + messages.push({ role: 'assistant', content: 'Got it — checking now.' }); + messages.push({ + role: 'user', + content: `Reminder: this needs live/current data. Call a tool first (web_search for news, weather_kma/weather_search for weather/forecast, etc.) before writing anything. Do NOT answer from memory or training data.${isNewsRequest ? ' If this message names a specific topic, search news about that topic; otherwise (a bare "뉴스"/"news" request) search today\'s general international/domestic headlines — do NOT default to unrelated topics from earlier in this conversation unless the user explicitly asks for updates on them.' : ''}`, + }); + } else if (isFactualInfoRequest(message)) { + messages.push({ role: 'assistant', content: 'Got it — let me verify that first.' }); + messages.push({ + role: 'user', + content: 'Reminder: this asks about concrete facts (specs, prices, versions, comparisons, or "what models/options exist"). Call web_search first to verify before answering — do not rely on memorized details, which are frequently stale or simply wrong. If search turns up nothing useful, say so explicitly rather than filling the gap from memory.', + }); + } + const replaceCurrentUserPromptWithAdvisorObjective = (objective: string): boolean => { const objectiveText = String(objective || '').trim(); if (!objectiveText) return false; @@ -7211,10 +7305,38 @@ RULES: let toolCalls = response.tool_calls; - // Auto-recover: if model wrote a tool call as text instead of using the tool mechanism + // Auto-recover: if model wrote a tool call as text instead of using the tool mechanism. + // Observed failure modes, both from garbled generation instead of a real function call: + // (a) bare `toolname{"arg":"val"}` repeated several times in one blob, degenerating on + // the repeats (e.g. "web_search" → "웍웍웍_search" on the 2nd/3rd/4th attempt) + // (b) a stray garbled token wedged between the name and the brace, e.g. + // `web_search Räikk{"query":"..."}` — the name itself is intact but not immediately + // adjacent to "{", so a strict "name directly followed by {" regex misses it too. + // Rather than pattern-matching every corruption shape, find each `{...}` JSON-looking blob + // and look at a short window of text right before it for any real tool name — closest + // occurrence wins. Validated against the actual available tool names so garbled filler + // ("Räikk", "웍웍웍_search") can never be mistaken for a real tool call. if ((!toolCalls || toolCalls.length === 0) && response.content) { - const textToolMatch = response.content.match(/"action"\s*:\s*"(\w+)"\s*,\s*"action_input"\s*:\s*(\{[^}]+\})/s) + const knownToolNames = tools.map((t: any) => String(t?.function?.name || '')).filter(Boolean); + let textToolMatch = response.content.match(/"action"\s*:\s*"(\w+)"\s*,\s*"action_input"\s*:\s*(\{[^}]+\})/s) || response.content.match(/"name"\s*:\s*"(\w+)"\s*,\s*"arguments"\s*:\s*(\{[^}]+\})/s); + if (!textToolMatch) { + for (const blobMatch of response.content.matchAll(/\{[^{}]*\}/g)) { + const blobStart = blobMatch.index ?? -1; + if (blobStart < 0) continue; + const precedingWindow = response.content.slice(Math.max(0, blobStart - 40), blobStart); + let bestName = ''; + let bestIdx = -1; + for (const name of knownToolNames) { + const idx = precedingWindow.lastIndexOf(name); + if (idx > bestIdx) { bestIdx = idx; bestName = name; } + } + if (bestName) { + textToolMatch = [blobMatch[0], bestName, blobMatch[0]] as any; + break; + } + } + } if (textToolMatch) { const toolName = textToolMatch[1]; try { @@ -7271,18 +7393,37 @@ RULES: } } - // Auto-recover: if model dumped pure reasoning without calling any tools on a - // question that clearly needs tools (search, file, browser), re-prompt once - if ((!toolCalls || toolCalls.length === 0) && response.content && round === 0 && allToolResults.length === 0) { + // Auto-recover: if model dumped pure reasoning without calling a RELEVANT tool on a + // question that clearly needs one (search, file, browser), re-prompt (bounded retries). + // NOTE: gate on "no relevant tool called yet", not "no tool called yet" / "round 0 only" — + // a model that calls one irrelevant/decoy tool (e.g. coder_list_files while answering a + // GPU spec question) must not permanently disarm this for the rest of the turn. + if ((!toolCalls || toolCalls.length === 0) && response.content && toolSkipForcedRetries < MAX_TOOL_SKIP_FORCED_RETRIES) { const content = response.content; const looksLikeReasoning = content.length > 300 && (/\b(let me|I need to|I should|the user|first,|wait,|hmm|the rules say)\b/i.test(content)); - const queryNeedsTools = /\b(search|find|look up|latest|news|info|open|browse|navigate|visit|click|type|fill|what happened|desktop|screen|window|vscode|vs code|codex|clipboard)\b/i.test(message); + const queryNeedsTools = /\b(search|find|look up|latest|news|info|open|browse|navigate|visit|click|type|fill|what happened|desktop|screen|window|vscode|vs code|codex|clipboard)\b/i.test(message) + || /(검색|찾아|찾아줘|알아봐|알아봐줘|열어|들어가|접속|방문|클릭|눌러|입력|무슨\s*일|바탕화면|화면|창|클립보드)/.test(message); const browserAutomationRequest = isBrowserAutomationRequest(message); const desktopAutomationRequest = isDesktopAutomationRequest(message); const looksLikeRefusal = looksLikeSafetyRefusal(content); - if (queryNeedsTools && (looksLikeReasoning || looksLikeRefusal)) { - console.log(`[v2] AUTO-RECOVER: Model dumped ${content.length} chars of reasoning instead of calling tools. Re-prompting...`); + // News/weather/prices — and now factual/product-info questions (specs, comparisons, + // "what models exist") — must never be answered from training data — force a tool call + // even when the model's reply looks like a confident final answer rather than stalled + // reasoning (a model that skips tools and answers fluently is more dangerous than one + // that visibly hesitates, since the fabrication reads as authoritative). + const liveDataRequest = isLiveDataRequest(message) || isFactualInfoRequest(message); + // Whether a tool that could actually ground THIS kind of request has already run — + // any other tool call (file listing, email check, etc.) doesn't count. + const groundingToolPattern = browserAutomationRequest + ? /^browser_/i + : desktopAutomationRequest + ? /^desktop_/i + : /^(web_search|web_fetch|weather_|ollama_web_)/i; + const hasGroundingToolCall = allToolResults.some((r) => groundingToolPattern.test(String(r?.name || ''))); + if (((queryNeedsTools && (looksLikeReasoning || looksLikeRefusal)) || liveDataRequest) && !hasGroundingToolCall) { + toolSkipForcedRetries++; + console.log(`[v2] AUTO-RECOVER (${toolSkipForcedRetries}/${MAX_TOOL_SKIP_FORCED_RETRIES}): Model dumped ${content.length} chars${liveDataRequest ? ' (live-data query answered without a grounding tool call)' : ' of reasoning instead of calling tools'}${allToolResults.length > 0 ? ` [${allToolResults.length} unrelated tool call(s) already made this turn]` : ''}. Re-prompting...`); allThinking += (allThinking ? '\n\n' : '') + content; sendSSE('thinking', { thinking: content.slice(0, 500) + '...' }); // Inject a forceful nudge and retry this round @@ -7315,8 +7456,14 @@ RULES: }); sendSSE('info', { message: 'Re-prompting model to execute desktop automation...' }); } else { - messages.push({ role: 'assistant', content: 'Let me search for that now.' }); - messages.push({ role: 'user', content: 'Yes, use the web_search tool right now. Do NOT think or plan — just call web_search.' }); + const isWeatherRequest = /(날씨|기온|예보|미세먼지|황사)/.test(message) || /\b(weather|forecast)\b/i.test(message); + const isNewsRequest = /(뉴스|속보)/.test(message) || /\bnews\b/i.test(message); + const toolHint = isWeatherRequest ? 'the weather tool (weather_kma / weather_search, etc. — NOT web_search)' : 'the web_search tool'; + const scopeHint = isNewsRequest + ? ' If this message names a specific topic, search news about that topic; otherwise (a bare "뉴스"/"news" request) search today\'s general international/domestic headlines — do NOT default to unrelated topics from earlier in this conversation unless the user explicitly asks for updates on them.' + : ''; + messages.push({ role: 'assistant', content: 'Let me check that now.' }); + messages.push({ role: 'user', content: `Yes, use ${toolHint} right now. Do NOT think or plan — just call it.${scopeHint}` }); sendSSE('info', { message: 'Re-prompting model to use tools...' }); } continue; // retry this round @@ -7353,6 +7500,29 @@ RULES: && continuationNudges < MAX_CONTINUATION_NUDGES && !hasConcreteCompletion(candidateText); + // A model claiming "sent!" without ever having called email_send/kakao_send_message is + // worse than a wrong info answer — the user believes a real message reached someone. + const sendToolCalled = allToolResults.some( + (r) => (r.name === 'email_send' || r.name === 'kakao_send_message') && !r.error, + ); + const shouldForceMessagingRetry = + isMessagingRequest(message) + && claimsMessageSent(candidateText) + && !sendToolCalled + && messagingForcedRetries < MAX_MESSAGING_FORCED_RETRIES; + + if (shouldForceMessagingRetry) { + messagingForcedRetries++; + console.log(`[v2] MESSAGING POST-CHECK: model claimed a message was sent without calling email_send/kakao_send_message (${messagingForcedRetries}/${MAX_MESSAGING_FORCED_RETRIES}). Re-prompting...`); + sendSSE('info', { message: 'Re-prompting model: verify the send with a real tool call...' }); + messages.push({ role: 'assistant', content: candidateText }); + messages.push({ + role: 'user', + content: 'You said the message was sent, but you never actually called email_send or kakao_send_message. Call the real tool now. If it fails, report the failure honestly instead of claiming success.', + }); + continue; + } + if (shouldForceBrowserRetry) { browserForcedRetries++; const reason = `browser advisor route=${browserAdvisorRoute || 'continue_browser'} requires continued execution`; @@ -8275,6 +8445,36 @@ RULES: sendSSE('tool_result', { action: toolName, result: _blocked.result, error: true, stepNum: allToolResults.length }); continue; } + // SEARCH BUDGET (system prompt WEB TOOLS text, ~line 1311) says at most 5 web_search/ + // web_fetch calls per turn — but that was prompt text only, with no code backstop, and + // some models (kimi-k2.6) reliably blow past it (observed 7 calls, once 50+, searching a + // broad "today's news" request topic-by-topic instead of synthesizing from a couple of + // broad searches). Hard-cap it here rather than trusting every model to self-limit. + if (toolName === 'web_search' || toolName === 'web_fetch') { + const searchFetchCallsSoFar = allToolResults.filter((r) => r.name === 'web_search' || r.name === 'web_fetch').length; + if (searchFetchCallsSoFar >= 5) { + const _blockedSearch: ToolResult = { + name: toolName, + args: toolArgs, + result: `[BLOCKED] Search budget exhausted (${searchFetchCallsSoFar} web_search/web_fetch calls already made this turn — max 5). Stop searching and answer now with what you already have. Do not keep searching topic-by-topic for exhaustive coverage.`, + error: true, + }; + allToolResults.push(_blockedSearch); + sendSSE('tool_result', { action: toolName, result: _blockedSearch.result, error: true, stepNum: allToolResults.length }); + continue; + } + } + // BROWSER RULE (system prompt, ~line 5789) says never call browser_open unless the user + // explicitly asked — but until now that was prompt text only, with no code backstop + // (unlike image_edit above). Block the actual first navigation of the turn when neither + // an explicit browser request nor an already-active browser session justifies it; once a + // session is legitimately open, later browser_open calls within it are left alone. + if (toolName === 'browser_open' && !isBrowserAutomationRequest(message) && !getBrowserSessionInfo(sessionId).active) { + const _blockedBrowser: ToolResult = { name: toolName, args: toolArgs, result: '[BLOCKED] browser_open was called without an explicit user request to use the browser. Do not open a browser unless the user explicitly asks (e.g. "브라우저로 열어줘", "Chrome으로 봐줘", "사이트 직접 들어가봐"). Answer from knowledge or use web_search/web_fetch instead.', error: true }; + allToolResults.push(_blockedBrowser); + sendSSE('tool_result', { action: toolName, result: _blockedBrowser.result, error: true, stepNum: allToolResults.length }); + continue; + } // Block browser_open for URLs the client renders as inline iframes (Maps, YouTube). if (toolName === 'browser_open' && /https?:\/\/(?:(?:www\.|m\.)?youtube\.com\/(?:watch|shorts|embed)|youtu\.be\/)/i.test(String(toolArgs?.url || ''))) { const _ytUrl = String(toolArgs.url || ''); @@ -8485,29 +8685,38 @@ function isResumeIntent(message: string): boolean { if (/\b(resume|rerun|re-run|run again|retry|restart)\b/i.test(text)) return true; if (/\b(continue|proceed)\b.*\b(task|it|that|this)\b/i.test(text)) return true; if (/\b(go ahead|do it|apply)\b.*\b(task|resume|rerun)\b/i.test(text)) return true; + if (/(재개해|다시\s*실행해|재시도해|재시작해)/.test(text)) return true; + if (/(계속해|이어서\s*해|이어서\s*진행)/.test(text) && /(작업|그거|그것|이거|이것)/.test(text)) return true; + if (/(진행해|해줘)/.test(text) && /(작업|재개)/.test(text)) return true; return false; } function isRerunIntent(message: string): boolean { - return /\b(rerun|re-run|run again|retry|restart|start again)\b/i.test(message); + return /\b(rerun|re-run|run again|retry|restart|start again)\b/i.test(message) + || /(다시\s*실행|재시도|재시작|다시\s*해봐)/.test(message); } function isCancelIntent(message: string): boolean { - return /\b(cancel|abort|stop( task)?|do not continue|don't continue)\b/i.test(message); + return /\b(cancel|abort|stop( task)?|do not continue|don't continue)\b/i.test(message) + || /(취소해|중단해|그만해|하지\s*마|멈춰)/.test(message); } function isStatusQuestion(message: string): boolean { return /\?|^\s*(what|why|how|status|did|where|when)\b/i.test(message) - || /\b(what happened|why did|status|stuck|failed|error|progress|what went wrong)\b/i.test(message); + || /\b(what happened|why did|status|stuck|failed|error|progress|what went wrong)\b/i.test(message) + || /(뭐야|왜\s|어떻게\s*됐|상태|어디까지|무슨\s*일|막혔|실패했|에러|진행\s*상황|뭐가\s*잘못)/.test(message); } function isTaskListIntent(message: string): boolean { return /\b(what|which|show|list)\b.*\b(background\s+)?tasks?\b/i.test(message) - || /\b(background\s+)?tasks?\b.*\b(do we have|running|active|current)\b/i.test(message); + || /\b(background\s+)?tasks?\b.*\b(do we have|running|active|current)\b/i.test(message) + || /(무슨|어떤|보여줘|목록).*(작업|백그라운드)/.test(message) + || /(작업|백그라운드).*(뭐\s*있|실행\s*중|진행\s*중|목록)/.test(message); } function isAdjustmentIntent(message: string): boolean { - return /\b(instead|change|adjust|update|only|skip|don't|do not|use|delete|remove|clear|keep|retry|try again)\b/i.test(message); + return /\b(instead|change|adjust|update|only|skip|don't|do not|use|delete|remove|clear|keep|retry|try again)\b/i.test(message) + || /(대신|바꿔|변경해|조정해|수정해|빼고|하지\s*말고|사용해|삭제해|제거해|지워|유지해|다시\s*시도)/.test(message); } function getLatestPauseContext(task: TaskRecord): { reason: string; detail: string } { @@ -13374,7 +13583,7 @@ app.post('/api/open-path', requireGatewayAuth, async (req, res) => { // Used by the Settings → Models tab to read/write provider config and // trigger the OpenAI OAuth flow. -import { getProvider, resetProvider, buildProviderForLLM } from '../providers/factory'; +import { getProvider, resetProvider, buildProviderForLLM, getModelForRole } from '../providers/factory'; import { buildWebhookRouter, resolveHookConfig } from './webhook-handler'; import { getMCPManager } from './mcp-manager'; import { startOAuthFlow, isConnected, clearTokens, loadTokens, exchangeManualCodeFromPending } from '../auth/openai-oauth'; diff --git a/src/tools/web.ts b/src/tools/web.ts index 700e943..c12d866 100644 --- a/src/tools/web.ts +++ b/src/tools/web.ts @@ -795,6 +795,30 @@ async function searchOllama(query: string, limit: number, endpoint: string, mode }; } +// A provider returning ANY result (even one) short-circuits the fallback chain below — +// which is right for most queries, but for news/headline queries a "result" that's actually +// just a bare homepage or a Wikipedia/Namu year-overview page isn't real news: it stops the +// chain from ever reaching Tavily/Google, which might have had actual headlines. Downgrade +// such results to "empty" (still kept as bestEmpty last-resort) so the chain keeps going — +// but only for queries that are actually asking for news, so a legitimate "무슨 사이트야" / +// "give me the Reuters homepage" style query isn't penalized for getting a homepage back. +function isNewsSeekingQuery(query: string): boolean { + return /(뉴스|헤드라인|속보)/.test(query) || /\b(news|headlines?)\b/i.test(query); +} + +function isLowValueNewsResult(url: string): boolean { + try { + const u = new URL(url); + const path = u.pathname.replace(/\/+$/, ''); + if (!path) return true; // bare homepage, e.g. https://www.reuters.com/ + if (/^\/wiki\/\d{4}$/i.test(path) && /wikipedia\.org$/i.test(u.hostname)) return true; // year-overview page + if (/^\/w\/\d{4}$/i.test(path) && /namu\.wiki$/i.test(u.hostname)) return true; + return false; + } catch { + return false; + } +} + // ── Main web_search tool ────────────────────────────────────────────────────── export async function executeWebSearch(args: { query: string; max_results?: number }): Promise { if (!args.query?.trim()) return { success: false, error: 'query is required' }; @@ -860,7 +884,11 @@ export async function executeWebSearch(args: { query: string; max_results?: numb const res = await runProvider(provider); await augmentEventContract(args.query, res); const resultCount = Array.isArray(res.data?.results) ? res.data.results.length : 0; - const hasResults = res.success && resultCount > 0; + let hasResults = res.success && resultCount > 0; + if (hasResults && isNewsSeekingQuery(args.query)) { + const results = res.data!.results as SearchResultItem[]; + if (results.every((r) => isLowValueNewsResult(r.url))) hasResults = false; + } diagnostics.attempted.push({ provider, @@ -895,7 +923,11 @@ export async function executeWebSearch(args: { query: string; max_results?: numb try { const res = await searchDDGHtml(args.query, limit); const resultCount = Array.isArray(res.data?.results) ? res.data.results.length : 0; - const hasResults = res.success && resultCount > 0; + let hasResults = res.success && resultCount > 0; + if (hasResults && isNewsSeekingQuery(args.query)) { + const results = res.data!.results as SearchResultItem[]; + if (results.every((r) => isLowValueNewsResult(r.url))) hasResults = false; + } diagnostics.attempted.push({ provider: 'ddg_html', status: hasResults ? 'success' : (res.success ? 'empty' : 'failed'),