v4.1.11: 뉴스 전용 도구 신설(NewsData.io) + 검색 루프/설정 버그 수정
- news_search 도구 신설(NewsData.io /latest API) — 과거 48시간 실시간 뉴스, 국가/언어/카테고리 필터, 실제 pubDate·출처 포함. 뉴스 요청엔 web_search보다 우선 사용하도록 시스템프롬프트·강제재시도 넛지 전부 갱신 — 위키피디아 연도페이지·날짜 오라벨링 문제가 구조적으로 해결됨 - 검색예산(5회) 차단 후에도 모델이 무시하고 계속 재시도하는 문제 수정 — "오늘 남미 주요 뉴스" 요청에서 실제 검색 6회 후 42라운드가 전부 헛되이 낭비된 사례 확인(TOOL[48]까지 감). 두 번째 차단부터는 해당 턴 나머지 동안 모델에게 tools를 빈 배열로 보내 물리적으로 더 이상 호출 못 하게 함 - OpenWeather 도시명 모호성 버그 수정 — "Rome, Italy"가 미국 조지아주 Rome으로 잘못 resolve되던 문제. 국가명→ISO코드 자동 정규화(40여개국) + 국가 불일치 시 에러로 재시도 유도 - SearXNG general 카테고리에 뉴스 아닌 위키피디아 엔진이 섞여있던 문제 수정 — 뉴스 쿼리는 categories=news로 daum_news/yahoo_news 등 뉴스 전용 엔진만 사용하도록 변경(사후 필터링 아닌 근본 차단) - 설정 화면에서 모델 전환 시 models.fallback/profiles가 매번 사라지던 버그 수정 — /api/settings/model 저장 로직이 models 객체를 통째로 새로 만들어서 덮어쓰던 것을, 기존 값을 먼저 펼친 뒤 필요한 필드만 덮어쓰도록 변경 - news.newsdata_api_key를 SECRET_FIELD_MAP에 추가해 vault 자동 암호화 대상에 포함(평문 미저장) Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
+15
-3
@@ -46,11 +46,20 @@
|
||||
},
|
||||
"models": {
|
||||
"primary": "kimi-k2.6:cloud",
|
||||
"fallback": "kimi-k2.6:cloud",
|
||||
"roles": {
|
||||
"manager": "kimi-k2.6:cloud",
|
||||
"executor": "kimi-k2.6:cloud",
|
||||
"verifier": "kimi-k2.6:cloud",
|
||||
"manager": "minimax-m3:cloud",
|
||||
"executor": "minimax-m3:cloud",
|
||||
"verifier": "minimax-m3:cloud",
|
||||
"background_task": ""
|
||||
},
|
||||
"profiles": {
|
||||
"mistral-large-3:675b-cloud": {
|
||||
"extraSystemPrompt": "You have a documented tendency to skip tool calls and answer fluently and confidently from memory instead — on news, weather, and factual/product questions (specs, prices, versions, comparisons) alike, sometimes inventing specific-sounding numbers or even nonexistent events. Before answering ANYTHING with a checkable real-world fact, call the relevant tool (web_search, weather_kma, etc.) first — do not trust your own confidence as a substitute for checking. If a tool call fails or returns nothing useful, say so plainly instead of filling the gap from memory."
|
||||
},
|
||||
"kimi-k2.6:cloud": {
|
||||
"extraSystemPrompt": "You have a documented tendency to massively over-search — observed doing 7+ (once 50+) web_search/web_fetch calls for a single broad request like \"오늘 국제 뉴스 정리\", searching topic-by-topic (Ukraine, Gaza, tariffs, ...) instead of a couple of broad searches. A hard cap now blocks you past 5 web_search/web_fetch calls per turn — but don't rely on the cap. Plan your search angles up front, pick the 1-3 most important ones, and write the answer once you have enough. Exhaustive topic-by-topic coverage is not the goal."
|
||||
}
|
||||
}
|
||||
},
|
||||
"tools": {
|
||||
@@ -282,6 +291,9 @@
|
||||
"pubmed": {
|
||||
"api_key": "vault:pubmed.api_key"
|
||||
},
|
||||
"news": {
|
||||
"newsdata_api_key": "vault:news.newsdata_api_key"
|
||||
},
|
||||
"email": {
|
||||
"accounts": [
|
||||
{
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "smallclaw",
|
||||
"version": "4.1.10",
|
||||
"version": "4.1.11",
|
||||
"description": "Local AI agent framework powered by Ollama - OpenClaw alternative",
|
||||
"main": "dist/index.js",
|
||||
"bin": {
|
||||
|
||||
@@ -322,6 +322,7 @@ const SECRET_FIELD_MAP: Array<[string[], string]> = [
|
||||
[['ppt', 'pixabay_key'], 'ppt.pixabay_key'],
|
||||
[['openweather', 'api_key'], 'openweather.api_key'],
|
||||
[['weather', 'api_key'], 'weather.api_key'],
|
||||
[['news', 'newsdata_api_key'], 'news.newsdata_api_key'],
|
||||
];
|
||||
|
||||
function deepGet(obj: any, keys: string[]): string | undefined {
|
||||
|
||||
@@ -165,7 +165,12 @@ export function registerSettingsRoutes(app: Express): void {
|
||||
Object.entries(roles || {}).filter(([, v]) => v !== '')
|
||||
);
|
||||
cm.updateConfig({
|
||||
// updateConfig() only shallow-merges at the top level, so writing `models` here
|
||||
// replaces the ENTIRE models object — spread current.models first or fields this
|
||||
// endpoint doesn't know about (fallback, profiles) get silently wiped on every
|
||||
// model switch from the Settings UI.
|
||||
models: {
|
||||
...current.models,
|
||||
primary: newPrimary,
|
||||
roles: { ...current.models.roles, ...filteredRoles },
|
||||
}
|
||||
|
||||
@@ -1308,7 +1308,7 @@ function detectToolCategories(text: string, sessionId?: string): Set<string> {
|
||||
|
||||
// Tool rule blocks — compact, injected only when relevant
|
||||
const TOOL_BLOCKS: Record<string, string> = {
|
||||
web: `WEB TOOLS: web_search(query) → headlines+snippets. web_fetch(url) → full page text. Use web_search first to get URLs, then web_fetch to read. For Reddit: web_search with site:reddit.com "keyword", then web_fetch post URLs — never open browser for Reddit. IMPORTANT: NEVER use web_fetch or web_search for local URLs (localhost, 127.0.0.1, /api/files/...) — they are NOT web pages. Local files are already accessible in chat via  or [name](/api/files/name.pptx). SEARCH BUDGET: for any single user request, use at most 5 web_search/web_fetch calls combined — pick the most relevant angles up front rather than searching topic-by-topic one at a time. Once you have enough to answer, stop searching and write the answer with what you have; do not keep searching for full/exhaustive coverage. NEWS QUERY STRATEGY: for Korean domestic news, a Korean-language query works well (e.g. "2026년 7월 16일 국내 뉴스 헤드라인"). For international/world news, the underlying search engines index English far better — search in English (e.g. "international news today", "world news headlines July 16 2026") rather than a Korean query, and do NOT bolt a single news outlet name onto the query (e.g. "... CNN", "... Reuters") — that tends to return the outlet's generic homepage instead of an actual headline. If your first search returns only a generic reference page (e.g. a Wikipedia year-overview) or a homepage instead of real headlines, that query failed — try a different phrasing rather than repeating close variants of the same failing query. NEWS DATING: search results carry their OWN publish date (visible in the snippet or URL, e.g. "2026.07.13" or a dated URL path like /2026/07/14/) — use THAT date in your header/summary, never default to today's date just because the user asked "오늘"/"today". If the freshest result you found is older than today, say so explicitly up front (e.g. "7월 13일 기준 뉴스이며, 이후 업데이트는 확인되지 않았습니다") instead of presenting it under today's date — do not make the user catch this themselves.`,
|
||||
web: `WEB TOOLS: web_search(query) → headlines+snippets. web_fetch(url) → full page text. Use web_search first to get URLs, then web_fetch to read. For Reddit: web_search with site:reddit.com "keyword", then web_fetch post URLs — never open browser for Reddit. IMPORTANT: NEVER use web_fetch or web_search for local URLs (localhost, 127.0.0.1, /api/files/...) — they are NOT web pages. Local files are already accessible in chat via  or [name](/api/files/name.pptx). SEARCH BUDGET: for any single user request, use at most 5 web_search/web_fetch calls combined — pick the most relevant angles up front rather than searching topic-by-topic one at a time. Once you have enough to answer, stop searching and write the answer with what you have; do not keep searching for full/exhaustive coverage. NEWS: use news_search FIRST for any "오늘 뉴스"/breaking-news request — it returns real articles with real publish dates, not generic web pages. Only fall back to web_search for news if news_search errors or returns nothing useful. NEWS QUERY STRATEGY (for web_search fallback / general search): for Korean domestic news, a Korean-language query works well (e.g. "2026년 7월 16일 국내 뉴스 헤드라인"). For international/world news, the underlying search engines index English far better — search in English (e.g. "international news today", "world news headlines July 16 2026") rather than a Korean query, and do NOT bolt a single news outlet name onto the query (e.g. "... CNN", "... Reuters") — that tends to return the outlet's generic homepage instead of an actual headline. If your first search returns only a generic reference page (e.g. a Wikipedia year-overview) or a homepage instead of real headlines, that query failed — try a different phrasing rather than repeating close variants of the same failing query. NEWS DATING: search results carry their OWN publish date (visible in the snippet, URL, or news_search's pubDate field) — use THAT date in your header/summary, never default to today's date just because the user asked "오늘"/"today". If the freshest result you found is older than today, say so explicitly up front (e.g. "7월 13일 기준 뉴스이며, 이후 업데이트는 확인되지 않았습니다") instead of presenting it under today's date — do not make the user catch this themselves.`,
|
||||
|
||||
browser: `BROWSER TOOLS: browser_open(url) → opens+returns snapshot. browser_snapshot() → refresh. browser_click(ref) → click by @ref. browser_fill(ref,text) → fill input. browser_press_key(key) → Enter/Tab/Escape. browser_wait(ms) → wait+snapshot. browser_close() → close tab. Chrome profile is persistent. NEVER use browser_open for local /api/files/ URLs — those are already inline in chat. SNAPSHOT RULE: browser_open/fill/wait/click all return a snapshot automatically — do NOT call browser_snapshot after them; only call it when you haven't received a fresh snapshot recently.`,
|
||||
|
||||
@@ -2441,6 +2441,22 @@ function buildTools() {
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
type: 'function',
|
||||
function: {
|
||||
name: 'news_search',
|
||||
description: 'Get real news headlines from the last 48 hours via NewsData.io. ALWAYS use this instead of web_search for "오늘 뉴스"/"today\'s news"/breaking-news requests — every result has a real publish date and source from the provider, not a generic scraped page (no Wikipedia year-overview pages, no stale articles mislabeled as today\'s).',
|
||||
parameters: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
query: { type: 'string', description: 'Optional keyword/topic filter (e.g. "economy", "South America"). Omit for general top headlines.' },
|
||||
country: { type: 'string', description: 'Optional ISO 3166-1 alpha-2 country code, e.g. "kr" (Korea), "us", "br" (Brazil), "jp". Omit for worldwide.' },
|
||||
language: { type: 'string', description: 'Optional ISO 639-1 language code, e.g. "ko", "en". Use "ko" for Korean-domestic requests, "en" for international.' },
|
||||
category: { type: 'string', description: '"business"|"entertainment"|"environment"|"food"|"health"|"politics"|"science"|"sports"|"technology"|"top"|"tourism"|"world"' },
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
type: 'function',
|
||||
function: {
|
||||
@@ -5480,6 +5496,14 @@ async function handleChat(
|
||||
// GPU spec question) must not permanently disarm the safety net for the rest of the turn.
|
||||
let toolSkipForcedRetries = 0;
|
||||
const MAX_TOOL_SKIP_FORCED_RETRIES = 2;
|
||||
// Observed failure: the search-budget block (below) stops execution but not the model —
|
||||
// it just keeps retrying with slightly reworded queries every round, burning up to
|
||||
// MAX_TOOL_ROUNDS (50) rounds of pure inference with zero real searches (once 42 wasted
|
||||
// rounds after only 6 real ones, for "오늘 남미 주요 뉴스"). After the block fires twice,
|
||||
// strip tool availability entirely for the rest of the turn — a text-only next call can't
|
||||
// retry a tool call no matter how insistent the model is.
|
||||
let searchBudgetBlockedCount = 0;
|
||||
let toolsDisabledForRestOfTurn = false;
|
||||
// Scroll-before-act gate: tracks whether a fill/click has happened yet this turn.
|
||||
// Blocks PageDown/scroll calls on interactive pages until the model actually acts.
|
||||
let browserFillOrClickDoneThisTurn = false;
|
||||
@@ -5908,7 +5932,7 @@ OUTPUT FORMAT: When presenting 3+ items (news articles, emails, search results,
|
||||
messages.push({ role: 'assistant', content: 'Got it — checking now.' });
|
||||
messages.push({
|
||||
role: 'user',
|
||||
content: `Reminder: this needs live/current data. Call a tool first (web_search for news, weather_kma/weather_search for weather/forecast, etc.) before writing anything. Do NOT answer from memory or training data.${isNewsRequest ? ' If this message names a specific topic, search news about that topic; otherwise (a bare "뉴스"/"news" request) search today\'s general international/domestic headlines — do NOT default to unrelated topics from earlier in this conversation unless the user explicitly asks for updates on them.' : ''}`,
|
||||
content: `Reminder: this needs live/current data. Call a tool first (${isNewsRequest ? 'news_search for news' : 'web_search for news'}, weather_kma/weather_search for weather/forecast, etc.) before writing anything. Do NOT answer from memory or training data.${isNewsRequest ? ' If this message names a specific topic, pass it as news_search\'s query param; otherwise (a bare "뉴스"/"news" request) call news_search with no query for general headlines — do NOT default to unrelated topics from earlier in this conversation unless the user explicitly asks for updates on them.' : ''}`,
|
||||
});
|
||||
} else if (isFactualInfoRequest(message)) {
|
||||
messages.push({ role: 'assistant', content: 'Got it — let me verify that first.' });
|
||||
@@ -7121,7 +7145,10 @@ RULES:
|
||||
const generationPromise = ollama.chatWithThinkingStream(messages, 'executor', (token: string) => {
|
||||
try { sendSSE('token', { text: token }); } catch {}
|
||||
}, {
|
||||
tools,
|
||||
// Empty (not omitted) once toolsDisabledForRestOfTurn trips — a model with no
|
||||
// functions available literally cannot retry a blocked tool call, which a text
|
||||
// instruction alone failed to stop (see search-budget block below).
|
||||
tools: toolsDisabledForRestOfTurn ? [] : tools,
|
||||
temperature: 0.3,
|
||||
// num_ctx omitted → adapter sizes it to the prompt (up to the model's native
|
||||
// context_length) instead of a fixed 8192 that silently truncated big contexts.
|
||||
@@ -7458,9 +7485,13 @@ RULES:
|
||||
} else {
|
||||
const isWeatherRequest = /(날씨|기온|예보|미세먼지|황사)/.test(message) || /\b(weather|forecast)\b/i.test(message);
|
||||
const isNewsRequest = /(뉴스|속보)/.test(message) || /\bnews\b/i.test(message);
|
||||
const toolHint = isWeatherRequest ? 'the weather tool (weather_kma / weather_search, etc. — NOT web_search)' : 'the web_search tool';
|
||||
const toolHint = isWeatherRequest
|
||||
? 'the weather tool (weather_kma / weather_search, etc. — NOT web_search)'
|
||||
: isNewsRequest
|
||||
? 'the news_search tool (NOT web_search)'
|
||||
: 'the web_search tool';
|
||||
const scopeHint = isNewsRequest
|
||||
? ' If this message names a specific topic, search news about that topic; otherwise (a bare "뉴스"/"news" request) search today\'s general international/domestic headlines — do NOT default to unrelated topics from earlier in this conversation unless the user explicitly asks for updates on them.'
|
||||
? ' If this message names a specific topic, pass it as news_search\'s query param; otherwise (a bare "뉴스"/"news" request) call news_search with no query for general headlines — do NOT default to unrelated topics from earlier in this conversation unless the user explicitly asks for updates on them.'
|
||||
: '';
|
||||
messages.push({ role: 'assistant', content: 'Let me check that now.' });
|
||||
messages.push({ role: 'user', content: `Yes, use ${toolHint} right now. Do NOT think or plan — just call it.${scopeHint}` });
|
||||
@@ -8453,10 +8484,15 @@ RULES:
|
||||
if (toolName === 'web_search' || toolName === 'web_fetch') {
|
||||
const searchFetchCallsSoFar = allToolResults.filter((r) => r.name === 'web_search' || r.name === 'web_fetch').length;
|
||||
if (searchFetchCallsSoFar >= 5) {
|
||||
searchBudgetBlockedCount++;
|
||||
const secondPlusStrike = searchBudgetBlockedCount >= 2;
|
||||
if (secondPlusStrike) toolsDisabledForRestOfTurn = true;
|
||||
const _blockedSearch: ToolResult = {
|
||||
name: toolName,
|
||||
args: toolArgs,
|
||||
result: `[BLOCKED] Search budget exhausted (${searchFetchCallsSoFar} web_search/web_fetch calls already made this turn — max 5). Stop searching and answer now with what you already have. Do not keep searching topic-by-topic for exhaustive coverage.`,
|
||||
result: secondPlusStrike
|
||||
? `[BLOCKED] Search budget exhausted (${searchFetchCallsSoFar} calls, max 5) and you already ignored one stop instruction. Tools are now disabled for the rest of this turn — no more calls of any kind will run. Write your final text answer now using whatever results you already have; if that's not enough, say so explicitly.`
|
||||
: `[BLOCKED] Search budget exhausted (${searchFetchCallsSoFar} web_search/web_fetch calls already made this turn — max 5). Stop searching and answer now with what you already have. Do not keep searching topic-by-topic for exhaustive coverage.`,
|
||||
error: true,
|
||||
};
|
||||
allToolResults.push(_blockedSearch);
|
||||
@@ -8464,6 +8500,17 @@ RULES:
|
||||
continue;
|
||||
}
|
||||
}
|
||||
if (toolsDisabledForRestOfTurn) {
|
||||
const _blockedAny: ToolResult = {
|
||||
name: toolName,
|
||||
args: toolArgs,
|
||||
result: '[BLOCKED] Tools are disabled for the rest of this turn (repeated search-budget violation). Write your final text answer now.',
|
||||
error: true,
|
||||
};
|
||||
allToolResults.push(_blockedAny);
|
||||
sendSSE('tool_result', { action: toolName, result: _blockedAny.result, error: true, stepNum: allToolResults.length });
|
||||
continue;
|
||||
}
|
||||
// BROWSER RULE (system prompt, ~line 5789) says never call browser_open unless the user
|
||||
// explicitly asked — but until now that was prompt text only, with no code backstop
|
||||
// (unlike image_edit above). Block the actual first navigation of the turn when neither
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
import { ToolResult } from '../types.js';
|
||||
import { getConfig } from '../config/config.js';
|
||||
import { getVault } from '../security/vault.js';
|
||||
|
||||
const NEWSDATA_BASE = 'https://newsdata.io/api/1/latest';
|
||||
|
||||
function getApiKey(): string | undefined {
|
||||
try {
|
||||
const cm = getConfig();
|
||||
const cfg = cm.getConfig();
|
||||
const resolved = cm.resolveSecret((cfg as any).news?.newsdata_api_key);
|
||||
if (resolved) return resolved;
|
||||
const vault = getVault(cm.getConfigDir());
|
||||
return vault.get('news.newsdata_api_key', 'news:getkey')?.expose();
|
||||
} catch {}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function formatArticles(results: any[]): string {
|
||||
if (!results.length) return '(no articles found)';
|
||||
return results.slice(0, 10).map((a, i) => {
|
||||
const title = a.title || '(no title)';
|
||||
const source = a.source_name || a.source_id || '';
|
||||
const pubDate = a.pubDate || '';
|
||||
const link = a.link || '';
|
||||
const desc = (a.description || '').slice(0, 200);
|
||||
return [
|
||||
`[${i + 1}] ${title}`,
|
||||
` Published: ${pubDate} | Source: ${source}`,
|
||||
` ${link}`,
|
||||
desc ? ` ${desc}` : '',
|
||||
].filter(Boolean).join('\n');
|
||||
}).join('\n\n');
|
||||
}
|
||||
|
||||
// NewsData.io's /latest endpoint only covers the past 48 hours (free tier) — exactly the
|
||||
// "today's news" window this tool exists for. Unlike web_search's generic scrape-and-rank,
|
||||
// every result here carries a real pubDate and source from the provider's own metadata, so
|
||||
// there's no risk of the "labeled as today but actually 3 days old" or "Wikipedia year page
|
||||
// instead of a headline" failure modes web_search hit for news queries.
|
||||
export const newsSearchTool = {
|
||||
name: 'news_search',
|
||||
description: 'Get real news headlines from the last 48 hours via NewsData.io — use this instead of web_search for any "오늘 뉴스"/"today\'s news"/breaking news request. Returns real articles with actual publish dates and sources, not generic web pages.',
|
||||
schema: {
|
||||
query: 'Optional keyword/topic to filter by (e.g. "economy", "South America"). Omit for general top headlines.',
|
||||
country: 'Optional ISO 3166-1 alpha-2 country code (e.g. "kr" for Korea, "us" for USA, "br" for Brazil). Omit for worldwide.',
|
||||
language: 'Optional ISO 639-1 language code (e.g. "ko", "en"). Defaults to "en" if country isn\'t Korea-related, "ko" otherwise — pass explicitly to override.',
|
||||
category: 'Optional: business, entertainment, environment, food, health, politics, science, sports, technology, top, tourism, or world.',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
query: { type: 'string', description: 'Optional keyword/topic filter. Omit for general top headlines.' },
|
||||
country: { type: 'string', description: 'Optional ISO 3166-1 alpha-2 country code, e.g. "kr", "us", "br", "jp".' },
|
||||
language: { type: 'string', description: 'Optional ISO 639-1 language code, e.g. "ko", "en".' },
|
||||
category: { type: 'string', description: '"business"|"entertainment"|"environment"|"food"|"health"|"politics"|"science"|"sports"|"technology"|"top"|"tourism"|"world"' },
|
||||
},
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const apiKey = getApiKey();
|
||||
if (!apiKey) return { success: false, error: 'NewsData.io API key not configured (set news.newsdata_api_key in .smallclaw/config.json)' };
|
||||
|
||||
const query = String(args?.query || '').trim();
|
||||
const country = String(args?.country || '').trim().toLowerCase();
|
||||
const language = String(args?.language || '').trim().toLowerCase();
|
||||
const category = String(args?.category || '').trim().toLowerCase();
|
||||
|
||||
const params = new URLSearchParams({ apikey: apiKey });
|
||||
if (query) params.set('q', query);
|
||||
if (country) params.set('country', country);
|
||||
if (language) params.set('language', language);
|
||||
if (category) params.set('category', category);
|
||||
|
||||
try {
|
||||
const res = await fetch(`${NEWSDATA_BASE}?${params}`, { signal: AbortSignal.timeout(15_000) });
|
||||
const data: any = await res.json();
|
||||
if (!res.ok || data.status !== 'success') {
|
||||
const msg = data?.results?.message || data?.message || `HTTP ${res.status}`;
|
||||
return { success: false, error: `NewsData.io error: ${msg}` };
|
||||
}
|
||||
const results: any[] = Array.isArray(data.results) ? data.results : [];
|
||||
return {
|
||||
success: true,
|
||||
stdout: formatArticles(results),
|
||||
data: { query, country, language, category, totalResults: data.totalResults, raw: results },
|
||||
};
|
||||
} catch (err: any) {
|
||||
return { success: false, error: err.message };
|
||||
}
|
||||
},
|
||||
};
|
||||
@@ -22,6 +22,7 @@ import { audioTranscribeTool } from './audio-transcribe.js';
|
||||
import { pythonEvalTool } from './python.js';
|
||||
import { sqliteTool } from './sqlite.js';
|
||||
import { weatherSearchTool, weatherAirPollutionTool, weatherOpenMeteoTool, weatherKmaTool, weatherAirKoreaTool, weatherNasaPowerTool, weatherEra5Tool, weatherCdsTool, weatherCmip6Tool, weatherMapScreenshotTool } from './weather.js';
|
||||
import { newsSearchTool } from './news.js';
|
||||
import { emailListTool, emailReadTool, emailSendTool, emailSearchTool, emailDeleteTool } from './email.js';
|
||||
import { kakaoSendTool } from './kakao.js';
|
||||
import { koreanLawSearchTool, koreanLawFetchTool, usCaseSearchTool } from './legal.js';
|
||||
@@ -222,6 +223,8 @@ class ToolRegistry {
|
||||
this.registerSafe(weatherCdsTool);
|
||||
this.registerSafe(weatherCmip6Tool);
|
||||
this.registerSafe(weatherMapScreenshotTool);
|
||||
// News (NewsData.io — real headlines with real publish dates, for news queries)
|
||||
this.registerSafe(newsSearchTool);
|
||||
// Email tools (IMAP + SMTP)
|
||||
this.registerSafe(emailListTool);
|
||||
this.registerSafe(emailReadTool);
|
||||
|
||||
+45
-1
@@ -327,6 +327,41 @@ function formatForecast(data: any, units: string): string {
|
||||
|
||||
// ── weather_search ────────────────────────────────────────────────────────────
|
||||
|
||||
// OpenWeather's `q` param disambiguates same-named cities by ISO-3166 country CODE
|
||||
// ("Rome,IT"), not by full country name — "Rome, Italy" silently resolves to whichever
|
||||
// "Rome" ranks first in its index (observed: Rome, Georgia, US instead of Rome, Italy).
|
||||
// The tool schema already documents the "City,CountryCode" format, but models often write
|
||||
// the full country name anyway — normalize the common ones rather than trust every caller
|
||||
// to follow the format, and verify the response actually landed in the right country.
|
||||
const COUNTRY_NAME_TO_ISO: Record<string, string> = {
|
||||
italy: 'IT', france: 'FR', germany: 'DE', spain: 'ES', portugal: 'PT',
|
||||
'united kingdom': 'GB', uk: 'GB', england: 'GB', netherlands: 'NL', belgium: 'BE',
|
||||
switzerland: 'CH', austria: 'AT', poland: 'PL', sweden: 'SE', norway: 'NO',
|
||||
denmark: 'DK', finland: 'FI', greece: 'GR', ireland: 'IE', 'czech republic': 'CZ',
|
||||
czechia: 'CZ', hungary: 'HU', russia: 'RU', china: 'CN', japan: 'JP',
|
||||
'south korea': 'KR', korea: 'KR', canada: 'CA', australia: 'AU', brazil: 'BR',
|
||||
mexico: 'MX', india: 'IN', turkey: 'TR', egypt: 'EG', 'south africa': 'ZA',
|
||||
vietnam: 'VN', thailand: 'TH', indonesia: 'ID', philippines: 'PH', singapore: 'SG',
|
||||
malaysia: 'MY', 'new zealand': 'NZ', ukraine: 'UA', romania: 'RO', bulgaria: 'BG',
|
||||
croatia: 'HR', slovakia: 'SK', slovenia: 'SI', iceland: 'IS', luxembourg: 'LU',
|
||||
'united states': 'US', usa: 'US', 'united states of america': 'US',
|
||||
};
|
||||
|
||||
// Returns the OpenWeather `q` string to send plus the ISO country code we expect back
|
||||
// (undefined if we couldn't determine one — no mismatch check in that case).
|
||||
function normalizeLocationForOpenWeather(location: string): { query: string; expectedCountry?: string } {
|
||||
const lastComma = location.lastIndexOf(',');
|
||||
if (lastComma < 0) return { query: location };
|
||||
const city = location.slice(0, lastComma).trim();
|
||||
const countryPart = location.slice(lastComma + 1).trim();
|
||||
if (/^[A-Za-z]{2}$/.test(countryPart)) {
|
||||
return { query: `${city},${countryPart.toUpperCase()}`, expectedCountry: countryPart.toUpperCase() };
|
||||
}
|
||||
const iso = COUNTRY_NAME_TO_ISO[countryPart.toLowerCase()];
|
||||
if (iso) return { query: `${city},${iso}`, expectedCountry: iso };
|
||||
return { query: location };
|
||||
}
|
||||
|
||||
export const weatherSearchTool = {
|
||||
name: 'weather_search',
|
||||
description: 'Get current weather or 5-day forecast for any city or coordinates using OpenWeather API. Returns temperature, humidity, wind, conditions, and more.',
|
||||
@@ -353,16 +388,25 @@ export const weatherSearchTool = {
|
||||
if (!location) return { success: false, error: 'location is required' };
|
||||
|
||||
const coordMatch = location.match(/^(-?\d+(?:\.\d+)?)\s*,\s*(-?\d+(?:\.\d+)?)$/);
|
||||
const normalized = coordMatch ? null : normalizeLocationForOpenWeather(location);
|
||||
const locationParams: Record<string, string> = coordMatch
|
||||
? { lat: coordMatch[1], lon: coordMatch[2] }
|
||||
: { q: location };
|
||||
: { q: normalized!.query };
|
||||
|
||||
try {
|
||||
if (type === 'forecast') {
|
||||
const data = await owFetch('forecast', { ...locationParams, units, cnt: '40' });
|
||||
const gotCountry = String(data.city?.country || '').toUpperCase();
|
||||
if (normalized?.expectedCountry && gotCountry && gotCountry !== normalized.expectedCountry) {
|
||||
return { success: false, error: `Ambiguous city name: "${location}" resolved to ${data.city?.name}, ${gotCountry} instead of ${normalized.expectedCountry}. Retry with "City,${normalized.expectedCountry}" — if that still doesn't land in the right country, try a nearby larger city or include a state/region qualifier.` };
|
||||
}
|
||||
return { success: true, stdout: formatForecast(data, units), data: { location, type, units, raw: data } };
|
||||
} else {
|
||||
const data = await owFetch('weather', { ...locationParams, units });
|
||||
const gotCountry = String(data.sys?.country || '').toUpperCase();
|
||||
if (normalized?.expectedCountry && gotCountry && gotCountry !== normalized.expectedCountry) {
|
||||
return { success: false, error: `Ambiguous city name: "${location}" resolved to ${data.name}, ${gotCountry} instead of ${normalized.expectedCountry}. Retry with "City,${normalized.expectedCountry}" — if that still doesn't land in the right country, try a nearby larger city or include a state/region qualifier.` };
|
||||
}
|
||||
return { success: true, stdout: formatCurrent(data, units), data: { location, type, units, raw: data } };
|
||||
}
|
||||
} catch (err: any) {
|
||||
|
||||
+9
-3
@@ -587,9 +587,15 @@ async function searchBrave(query: string, limit: number, apiKey: string): Promis
|
||||
}
|
||||
|
||||
// ── SearXNG (self-hosted/public metasearch, no key) ──────────────────────────
|
||||
async function searchSearXNG(query: string, limit: number, baseUrl: string): Promise<ToolResult> {
|
||||
async function searchSearXNG(query: string, limit: number, baseUrl: string, category?: string): Promise<ToolResult> {
|
||||
const base = baseUrl.replace(/\/+$/, '');
|
||||
const url = `${base}/search?q=${encodeURIComponent(query)}&format=json`;
|
||||
// Default (no categories param) hits SearXNG's "general" category, which includes engines
|
||||
// like Wikipedia — fine for most queries, but Wikipedia's static reference pages (e.g. a
|
||||
// "2026" year-overview article) are not news and shouldn't compete with actual dated
|
||||
// articles for a news-seeking query. Restricting to categories=news routes to the engines
|
||||
// actually tagged "news" (daum news, yahoo news, presearch's news variant) instead.
|
||||
const categoryParam = category ? `&categories=${encodeURIComponent(category)}` : '';
|
||||
const url = `${base}/search?q=${encodeURIComponent(query)}&format=json${categoryParam}`;
|
||||
const res = await fetch(url, {
|
||||
headers: { 'Accept': 'application/json', 'User-Agent': 'SmallClaw/1.0' },
|
||||
signal: AbortSignal.timeout(15_000),
|
||||
@@ -842,7 +848,7 @@ export async function executeWebSearch(args: { query: string; max_results?: numb
|
||||
const runProvider = (provider: SearchProvider): Promise<ToolResult> => {
|
||||
switch (provider) {
|
||||
case 'tavily': return searchTavily(args.query, limit, cfg.tavilyKey as string);
|
||||
case 'searxng': return searchSearXNG(args.query, limit, cfg.searxngUrl as string);
|
||||
case 'searxng': return searchSearXNG(args.query, limit, cfg.searxngUrl as string, isNewsSeekingQuery(args.query) ? 'news' : undefined);
|
||||
case 'google': return searchGoogle(args.query, limit, cfg.googleKey as string, cfg.googleCx as string);
|
||||
case 'brave': return searchBrave(args.query, limit, cfg.braveKey as string);
|
||||
case 'ollama_cloud': return searchOllamaCloud(args.query, limit, cfg.ollamaApiKey as string);
|
||||
|
||||
Reference in New Issue
Block a user