import { ToolResult } from '../types.js'; import { getConfig } from '../config/config.js'; type SearchResultItem = { title: string; url: string; snippet: string }; // Shared HTML → plain text stripper used by fetchCleanArticle and executeWebFetch. // preserveStructure=true keeps paragraph breaks (\s{3,}→\n\n); false collapses all whitespace. function stripHtml(html: string, preserveStructure = false): string { const text = html .replace(//gi, ' ') .replace(//gi, ' ') .replace(//gi, ' ') .replace(//gi, ' ') .replace(//gi, ' ') .replace(//g, ' ') .replace(/<[^>]+>/g, ' ') .replace(/ /g, ' ').replace(/&/g, '&').replace(/</g, '<') .replace(/>/g, '>').replace(/"/g, '"').replace(/'/g, "'"); return preserveStructure ? text.replace(/\s{3,}/g, '\n\n').trim() : text.replace(/\s+/g, ' ').trim(); } type StructuredSource = { id: number; tier: 'A' | 'B' | 'C'; title: string; url: string; snippet: string; score: number }; type StructuredEvidence = { id: number; source_id: number; excerpt: string; score: number }; type StructuredFact = { id: number; claim: string; evidence_ids: number[]; source_ids: number[]; confidence: number }; type SearchProvider = 'tavily' | 'google' | 'brave' | 'ddg' | 'ddg_html' | 'searxng' | 'ollama_cloud'; type SearchProviderAttempt = { provider: SearchProvider; status: 'success' | 'empty' | 'failed' | 'skipped'; reason?: string; duration_ms?: number; result_count?: number; }; type SearchDiagnostics = { query: string; preferred_provider: 'tavily' | 'google' | 'brave' | 'ddg' | 'searxng' | 'ollama_cloud'; provider_order: Array<'tavily' | 'google' | 'brave' | 'ddg' | 'searxng' | 'ollama_cloud'>; attempted: SearchProviderAttempt[]; selected_provider?: SearchProvider; }; function normalizeGoogleUrl(url: string): string { try { const u = new URL(url); // Standard Google redirect wrapper: /url?q= if ((u.hostname.includes('google.') || u.hostname === 'google.com') && u.pathname === '/url') { const q = u.searchParams.get('q'); if (q) return decodeURIComponent(q); } return url; } catch { return url; } } function isLowQualityGoogleUrl(url: string): boolean { return /google\.com\/share\.google\?/i.test(url); } function isPriceQuery(query: string): boolean { return /price|cost|value|quote|trades?|usd|dollar|eur|gbp|jpy/i.test(query); } function isBitcoinQuery(query: string): boolean { return /bitcoin|btc/i.test(query); } function isFreshQuery(query: string): boolean { return /\b(current|latest|today|now|right now|as of|recent)\b/i.test(query); } function extractUsdPrice(text: string): string | null { const patterns = [ /\$\s?([0-9]{1,3}(?:,[0-9]{3})+(?:\.[0-9]+)?)/, /\$\s?([0-9]+(?:\.[0-9]+)?)/, /\b([0-9]{1,3}(?:,[0-9]{3})+(?:\.[0-9]+)?)\s?USD\b/i, /\b([0-9]+(?:\.[0-9]+)?)\s?USD\b/i, ]; for (const pattern of patterns) { const match = text.match(pattern); if (match?.[1]) return match[1]; } return null; } function parseUsdNumber(raw: string): number | null { const n = Number(String(raw || '').replace(/,/g, '').trim()); return Number.isFinite(n) ? n : null; } function detectPriceUnit(text: string): 'ounce' | 'gram' | 'unknown' { const t = String(text || '').toLowerCase(); if (/\b(per\s*gram|\/g\b|1g\b|gram\b)\b/.test(t)) return 'gram'; if (/\b(per\s*ounce|\/oz\b|ounce\b|oz\b)\b/.test(t)) return 'ounce'; return 'unknown'; } function hasHistoricalPriceCue(text: string): boolean { const t = String(text || '').toLowerCase(); return /\b(around|circa|in|from)\s*(19|20)\d{2}\b/.test(t) || /\b(was worth|years? ago|historical|history)\b/.test(t); } function hasFreshPriceCue(text: string): boolean { const t = String(text || '').toLowerCase(); return /\b(current|today|live|latest|now|right now|spot)\b/.test(t); } function detectPriceAsset(query: string): 'silver' | 'gold' | 'bitcoin' | 'generic' { const q = String(query || '').toLowerCase(); if (/\b(silver|xag)\b/.test(q)) return 'silver'; if (/\b(gold|xau|comex gold)\b/.test(q)) return 'gold'; if (/\b(bitcoin|btc)\b/.test(q)) return 'bitcoin'; return 'generic'; } function isPlausibleUsdPrice(asset: 'silver' | 'gold' | 'bitcoin' | 'generic', valuePerOunceOrUnit: number): boolean { if (!Number.isFinite(valuePerOunceOrUnit) || valuePerOunceOrUnit <= 0) return false; if (asset === 'silver') return valuePerOunceOrUnit >= 5 && valuePerOunceOrUnit <= 200; if (asset === 'gold') return valuePerOunceOrUnit >= 300 && valuePerOunceOrUnit <= 10_000; if (asset === 'bitcoin') return valuePerOunceOrUnit >= 1_000 && valuePerOunceOrUnit <= 2_000_000; return valuePerOunceOrUnit >= 0.5 && valuePerOunceOrUnit <= 5_000_000; } function buildDirectPriceAnswer( query: string, results: SearchResultItem[] ): string { if (!isPriceQuery(query)) return ''; const asset = detectPriceAsset(query); const candidates: Array<{ value: number; score: number; unit: 'ounce' | 'gram' | 'unknown' }> = []; for (const result of results) { const combined = `${result.title} ${result.snippet}`; const usdRaw = extractUsdPrice(combined); if (!usdRaw) continue; const usd = parseUsdNumber(usdRaw); if (!usd) continue; const unit = detectPriceUnit(combined); const normalized = unit === 'gram' ? (usd * 31.1035) : usd; if (!isPlausibleUsdPrice(asset, normalized)) continue; let score = 0; if (hasFreshPriceCue(combined)) score += 3; if (unit === 'ounce') score += 2; if (unit === 'gram') score += 1; if (hasHistoricalPriceCue(combined)) score -= 6; if (asset !== 'generic' && new RegExp(`\\b${asset}\\b`, 'i').test(combined)) score += 2; candidates.push({ value: normalized, score, unit }); } if (candidates.length) { candidates.sort((a, b) => b.score - a.score); const best = candidates[0]; if (best.score >= 0) { const v = best.value.toLocaleString('en-US', { minimumFractionDigits: 2, maximumFractionDigits: 2 }); if (asset === 'bitcoin') return `Answer: The current Bitcoin price is approximately $${v} USD.`; if (asset === 'silver') return `Answer: The current silver price is approximately $${v} USD per ounce.`; if (asset === 'gold') return `Answer: The current gold price is approximately $${v} USD per ounce.`; return `Answer: The current price is approximately $${v} USD.`; } } // When snippets do not include live numeric quotes, still return a compact // actionable answer instead of only raw links. if (isBitcoinQuery(query)) { const financeResult = results.find(r => /google\.com\/finance\/quote\/BTC-USD/i.test(r.url)); if (financeResult) { return 'Answer: I found the live BTC-USD quote page on Google Finance. Open https://www.google.com/finance/quote/BTC-USD for the exact real-time value.'; } } return ''; } function isEventOutcomeQuery(query: string): boolean { const q = query.toLowerCase(); return /\b(what happened|outcome|key takeaways|takeaways|summary|recap|latest update|status)\b/.test(q) || (/\b(hearing|trial|case|investigation|lawsuit|court|testimony)\b/.test(q) && /\b(what|how|why|when|recent|latest)\b/.test(q)); } function isLowValueResult(r: SearchResultItem): boolean { const text = `${r.title} ${r.url} ${r.snippet}`.toLowerCase(); if (/youtube\.com|youtu\.be|podcast|opinion|editorial|letters to the editor|substack|reddit/.test(text)) return true; return false; } function sourceTier(r: SearchResultItem): 'A' | 'B' | 'C' { const text = `${r.title} ${r.url}`.toLowerCase(); if (/\.gov|\.mil|justice\.gov|congress\.gov|house\.gov|senate\.gov|courtlistener|supremecourt/.test(text)) return 'A'; if (/apnews|reuters|bloomberg|ft\.com|nytimes|wsj|bbc|pbs|politico|aljazeera|npr|washingtonpost/.test(text)) return 'B'; return 'C'; } function allowsTierCForQuery(query: string): boolean { const q = query.toLowerCase(); return /\b(opinion|podcast|youtube|video|commentary|analysis only|broader context)\b/.test(q); } function applySourceTierPolicy(query: string, ranked: SearchResultItem[]): SearchResultItem[] { if (!isEventOutcomeQuery(query)) return ranked; const enriched = ranked.map(r => ({ r, tier: sourceTier(r) })); const allowC = allowsTierCForQuery(query); const preferred = enriched.filter(x => x.tier === 'A' || x.tier === 'B' || allowC); return (preferred.length ? preferred : enriched.filter(x => x.tier !== 'C')).map(x => x.r); } function queryAnchorTokens(query: string): string[] { return query .toLowerCase() .replace(/[^a-z0-9\s]/g, ' ') .split(/\s+/) .filter(t => t.length >= 4 && !['what', 'when', 'where', 'which', 'latest', 'recent', 'about', 'during'].includes(t)) .slice(0, 10); } function relevanceScore(query: string, text: string): number { const q = query.toLowerCase(); const t = text.toLowerCase(); const anchors = queryAnchorTokens(q); let score = 0; for (const a of anchors) if (t.includes(a)) score += 1; if (/bondi/.test(t) && /epstein/.test(t)) score += 3; if (/hearing|trial|case|committee|judiciary|testif|lawmakers|congress/.test(t)) score += 2; return score; } function overlapScore(a: string, b: string): number { const at = new Set(a.toLowerCase().replace(/[^a-z0-9\s]/g, ' ').split(/\s+/).filter(t => t.length >= 4)); const bt = new Set(b.toLowerCase().replace(/[^a-z0-9\s]/g, ' ').split(/\s+/).filter(t => t.length >= 4)); if (!at.size || !bt.size) return 0; let both = 0; for (const t of at) if (bt.has(t)) both++; return both / Math.max(at.size, bt.size); } function selectDominantStoryCluster(query: string, ranked: SearchResultItem[]): SearchResultItem[] { if (!isEventOutcomeQuery(query) || ranked.length <= 2) return ranked; const clusters: SearchResultItem[][] = []; const threshold = 0.18; for (const r of ranked) { const text = `${r.title} ${r.snippet}`; let placed = false; for (const c of clusters) { const centroid = `${c[0].title} ${c[0].snippet}`; if (overlapScore(text, centroid) >= threshold) { c.push(r); placed = true; break; } } if (!placed) clusters.push([r]); } if (clusters.length <= 1) return ranked; clusters.sort((a, b) => { const sa = a.reduce((s, r) => s + relevanceScore(query, `${r.title} ${r.snippet}`), 0); const sb = b.reduce((s, r) => s + relevanceScore(query, `${r.title} ${r.snippet}`), 0); return sb - sa; }); return clusters[0]; } async function fetchCleanArticle(url: string, maxChars = 5000): Promise { const res = await fetch(url, { headers: { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) SmallClaw/1.0' }, signal: AbortSignal.timeout(15_000), redirect: 'follow', }); if (!res.ok) throw new Error(`HTTP ${res.status}`); const ct = String(res.headers.get('content-type') || ''); if (!/text|html|json/i.test(ct)) throw new Error(`Unsupported content-type: ${ct}`); return stripHtml(await res.text()).slice(0, maxChars); } function extractEvidenceSentences(query: string, text: string, max = 4): string[] { const sentences = text .split(/(?<=[.!?])\s+/) .map(s => s.trim()) .filter(s => s.length >= 40 && s.length <= 320); const verbs = /\b(said|stated|argued|clashed|pressed|refused|confirmed|announced|deflected|criticized|questioned|responded)\b/i; const scored = sentences.map(s => { let score = relevanceScore(query, s); if (verbs.test(s)) score += 2; if (/bondi|epstein|attorney general|committee|judiciary|lawmakers/i.test(s)) score += 1.5; return { s, score }; }).sort((a, b) => b.score - a.score); return scored.filter(x => x.score >= 2.5).slice(0, max).map(x => x.s); } function cleanClaimText(claim: string): string { return String(claim || '') .replace(/\[[0-9]+\]/g, '') .replace(/\(AP Photo[^)]*\)/gi, '') .replace(/\s+/g, ' ') .trim() .slice(0, 220); } async function buildEventOutcomeAnswer(query: string, ranked: SearchResultItem[]): Promise { const filtered = ranked.filter(r => !isLowValueResult(r)); const tiered = applySourceTierPolicy(query, filtered); const clustered = selectDominantStoryCluster(query, tiered); const gated = clustered.filter(r => relevanceScore(query, `${r.title} ${r.snippet}`) >= 2); const picked = (gated.length ? gated : clustered).slice(0, 4); if (!picked.length) return ''; const snippetEvidence: Array<{ claim: string; source: number }> = []; for (let i = 0; i < picked.length; i++) { const fromSnippet = extractEvidenceSentences(query, picked[i].snippet, 2); for (const c of fromSnippet) snippetEvidence.push({ claim: c, source: i + 1 }); } const pageTexts = snippetEvidence.length < 8 ? (await Promise.allSettled(picked.map(r => fetchCleanArticle(r.url, 4500)))).map(r => r.status === 'fulfilled' ? r.value : null) : picked.map(() => null); const evidence: Array<{ claim: string; source: number }> = [...snippetEvidence]; for (let i = 0; i < picked.length; i++) { if (!pageTexts[i]) continue; const fromPage = extractEvidenceSentences(query, pageTexts[i]!, 2); for (const c of fromPage) evidence.push({ claim: c, source: i + 1 }); } const dedup = new Set(); const top: Array<{ claim: string; source: number }> = []; for (const e of evidence) { const cleaned = cleanClaimText(e.claim); if (!cleaned || cleaned.length < 20) continue; const k = cleaned.toLowerCase().replace(/[^a-z0-9\s]/g, '').slice(0, 140); if (dedup.has(k)) continue; dedup.add(k); top.push({ claim: cleaned, source: e.source }); if (top.length >= 3) break; } if (!top.length) return ''; const first = top[0]; const summaryLine = `Answer: ${first.claim} [${first.source}]`; const bullets = top.slice(1).map(t => `- ${t.claim} [${t.source}]`).join('\n'); const sources = picked.slice(0, 3).map((r, i) => `[${i + 1}] ${r.url}`).join(' '); return `${summaryLine}${bullets ? `\n${bullets}` : ''}\nSources: ${sources}`; } async function buildStructuredEventBundle(query: string, ranked: SearchResultItem[]): Promise<{ answer: string; sources: StructuredSource[]; evidence: StructuredEvidence[]; facts: StructuredFact[]; } | null> { if (!isEventOutcomeQuery(query)) return null; const filtered = ranked.filter(r => !isLowValueResult(r)); const tiered = applySourceTierPolicy(query, filtered); const clustered = selectDominantStoryCluster(query, tiered); const pickedRaw = clustered.slice(0, 4); if (!pickedRaw.length) return null; const sources: StructuredSource[] = pickedRaw.map((r, i) => ({ id: i + 1, tier: sourceTier(r), title: r.title, url: r.url, snippet: r.snippet.slice(0, 500), score: relevanceScore(query, `${r.title} ${r.snippet}`), })); const snippetItems: Array<{ source_id: number; excerpt: string; score: number }> = []; for (const s of sources) { for (const ex of extractEvidenceSentences(query, s.snippet, 2)) { snippetItems.push({ source_id: s.id, excerpt: cleanClaimText(ex), score: relevanceScore(query, ex) + 1 }); } } const pageTexts = snippetItems.length < 14 ? (await Promise.allSettled(sources.map(s => fetchCleanArticle(s.url, 4500)))).map(r => r.status === 'fulfilled' ? r.value : null) : sources.map(() => null); let evidenceId = 1; const evidence: StructuredEvidence[] = []; for (let i = 0; i < sources.length; i++) { const s = sources[i]; for (const item of snippetItems.filter(e => e.source_id === s.id)) { evidence.push({ id: evidenceId++, source_id: s.id, excerpt: item.excerpt, score: item.score }); } const pageText = pageTexts[i]; if (pageText) { for (const ex of extractEvidenceSentences(query, pageText, 2)) { evidence.push({ id: evidenceId++, source_id: s.id, excerpt: cleanClaimText(ex), score: relevanceScore(query, ex) + 1.5 }); } } } const sortedEvidence = evidence .filter(e => e.excerpt.length >= 20) .sort((a, b) => b.score - a.score) .slice(0, 10); if (!sortedEvidence.length) return null; const seen = new Set(); const facts: StructuredFact[] = []; for (const e of sortedEvidence) { const key = e.excerpt.toLowerCase().replace(/[^a-z0-9\s]/g, '').slice(0, 140); if (seen.has(key)) continue; seen.add(key); facts.push({ id: facts.length + 1, claim: e.excerpt, evidence_ids: [e.id], source_ids: [e.source_id], confidence: Math.max(0.5, Math.min(0.95, e.score / 8)), }); if (facts.length >= 4) break; } if (!facts.length) return null; const lead = facts[0]; const bullets = facts.slice(1, 4).map(f => `- ${f.claim} [${f.source_ids[0]}]`).join('\n'); const sourceLine = sources.slice(0, 3).map(s => `[${s.id}] ${s.url}`).join(' '); const answer = `Answer: ${lead.claim} [${lead.source_ids[0]}]${bullets ? `\n${bullets}` : ''}\nSources: ${sourceLine}`; return { answer, sources, evidence: sortedEvidence, facts }; } async function augmentEventContract(query: string, res: ToolResult): Promise { const ranked = (res.data?.results || []) as SearchResultItem[]; if (!isEventOutcomeQuery(query) || !ranked.length) return res; const bundle = await buildStructuredEventBundle(query, ranked); if (!bundle) return res; const summaryText = ranked.map((r: SearchResultItem, i: number) => `[${i + 1}] ${r.title}\n ${r.url}\n ${r.snippet.slice(0, 400)}`).join('\n\n'); res.data = { ...(res.data || {}), answer: bundle.answer, sources: bundle.sources, evidence: bundle.evidence, facts: bundle.facts, }; res.stdout = `${bundle.answer}\n\n${summaryText}`; return res; } function domainTrustScore(url: string): number { try { const h = new URL(url).hostname.toLowerCase(); if (h.endsWith('.gov') || h.endsWith('.mil')) return 4; if (h.endsWith('.edu') || h.includes('justice.gov') || h.includes('sec.gov') || h.includes('federalreserve.gov')) return 3.5; if (h.includes('reuters.com') || h.includes('apnews.com') || h.includes('bloomberg.com') || h.includes('ft.com')) return 3; if (h.includes('wikipedia.org') || h.includes('ballotpedia.org')) return 2; if (h.includes('youtube.com') || h.includes('tiktok.com')) return 0.5; return 1.5; } catch { return 0; } } function rankResults(query: string, results: SearchResultItem[]) { const q = query.toLowerCase(); const freshness = /\b(current|latest|today|now|as of|recent)\b/.test(q); return [...results] .map(r => { const t = domainTrustScore(r.url); const text = `${r.title} ${r.snippet}`.toLowerCase(); let rel = 0; const tokens = q.replace(/[^a-z0-9\s]/g, ' ').split(/\s+/).filter(x => x.length >= 4); for (const tok of tokens) if (text.includes(tok)) rel += 1; return { r, score: t * (freshness ? 2 : 1) + rel * 0.4 }; }) .sort((a, b) => b.score - a.score) .map(x => x.r); } // ── Load optional API keys from config ───────────────────────────────────── // Cached with 5-minute TTL so config changes are picked up without restart. type SearchConfig = { preferred: 'tavily' | 'google' | 'brave' | 'ddg' | 'searxng' | 'ollama_cloud'; tavilyKey?: string; googleKey?: string; googleCx?: string; braveKey?: string; searxngUrl?: string; ollamaApiKey?: string }; let _searchConfigCache: { value: SearchConfig; expiresAt: number } | null = null; function getSearchConfig(): SearchConfig { const now = Date.now(); if (_searchConfigCache && now < _searchConfigCache.expiresAt) return _searchConfigCache.value; let value: SearchConfig = { preferred: 'ddg' }; try { const cm = getConfig(); const data = cm.getConfig(); const preferredRaw = String(data.search?.preferred_provider || 'ddg').toLowerCase(); const preferred = (['tavily', 'google', 'brave', 'ddg', 'searxng', 'ollama_cloud'].includes(preferredRaw) ? preferredRaw : 'ddg') as SearchConfig['preferred']; const searxngRaw = typeof data.search?.searxng_url === 'string' ? data.search.searxng_url.trim().replace(/\/+$/, '') : ''; value = { preferred, tavilyKey: cm.resolveSecret(data.search?.tavily_api_key), googleKey: cm.resolveSecret(data.search?.google_api_key), googleCx: data.search?.google_cx, braveKey: cm.resolveSecret(data.search?.brave_api_key), searxngUrl: searxngRaw || undefined, ollamaApiKey: cm.resolveSecret(data.search?.ollama_api_key), }; } catch {} _searchConfigCache = { value, expiresAt: now + 5 * 60_000 }; return value; } // ── Google Custom Search API ─────────────────────────────────────────────--- async function searchGoogle(query: string, limit: number, apiKey: string, cx: string): Promise { const url = `https://www.googleapis.com/customsearch/v1?q=${encodeURIComponent(query)}&key=${apiKey}&cx=${cx}&num=${limit}`; const res = await fetch(url, { signal: AbortSignal.timeout(15_000) }); if (!res.ok) throw new Error(`Google HTTP ${res.status}`); const data: any = await res.json(); const results = (data.items || []).map((r: any) => ({ title: r.title || '', url: normalizeGoogleUrl(r.link || ''), snippet: r.snippet || '', })); const ranked = rankResults(query, results); // Guard: some CSE configurations return mostly share.google wrappers that // are not reliable search hits for factual QA. Trigger provider fallback. if (results.length > 0) { const lowQuality = results.filter((r: { url: string }) => isLowQualityGoogleUrl(r.url)).length; if (lowQuality / results.length >= 0.5) { throw new Error('Google CSE returned mostly low-quality share links; falling back to other providers.'); } } const answer = buildDirectPriceAnswer(query, ranked); return { success: true, data: { query, results: ranked, answer: answer || undefined }, stdout: (answer ? `${answer}\n\n` : '') + ranked.map((r: any, i: number) => `[${i + 1}] ${r.title}\n ${r.url}\n ${r.snippet.slice(0, 400)}`).join('\n\n'), }; } // ── Tavily (best for AI agents, free 1k/mo) ─────────────────────────────────── async function searchTavily(query: string, limit: number, apiKey: string): Promise { const res = await fetch('https://api.tavily.com/search', { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ api_key: apiKey, query, max_results: limit, search_depth: 'basic', // Provider "answer" strings can be stale/inconsistent for freshness queries. // We synthesize from snippets instead of trusting this shortcut. include_answer: !isFreshQuery(query), }), signal: AbortSignal.timeout(15_000), }); if (!res.ok) throw new Error(`Tavily HTTP ${res.status}`); const data: any = await res.json(); const results = (data.results || []).map((r: any) => ({ title: r.title || '', url: r.url || '', snippet: r.content || '', })); const ranked = rankResults(query, results); // Use deterministic local extraction only (e.g., prices) to avoid stale provider summaries. const answer = buildDirectPriceAnswer(query, ranked); return { success: true, data: { query, results: ranked, answer: data.answer }, stdout: (answer ? `${answer}\n\n` : '') + ranked.map((r: any, i: number) => `[${i + 1}] ${r.title}\n ${r.url}\n ${r.snippet.slice(0, 400)}` ).join('\n\n'), }; } // ── Brave Search API (free 2k/mo) ───────────────────────────────────────────── async function searchBrave(query: string, limit: number, apiKey: string): Promise { const url = `https://api.search.brave.com/res/v1/web/search?q=${encodeURIComponent(query)}&count=${limit}`; const res = await fetch(url, { headers: { 'Accept': 'application/json', 'X-Subscription-Token': apiKey }, signal: AbortSignal.timeout(15_000), }); if (!res.ok) throw new Error(`Brave HTTP ${res.status}`); const data: any = await res.json(); const results = (data.web?.results || []).map((r: any) => ({ title: r.title || '', url: r.url || '', snippet: r.description || '', })); const ranked = rankResults(query, results); const answer = buildDirectPriceAnswer(query, ranked); return { success: true, data: { query, results: ranked, answer: answer || undefined }, stdout: (answer ? `${answer}\n\n` : '') + ranked.map((r: any, i: number) => `[${i + 1}] ${r.title}\n ${r.url}\n ${r.snippet}` ).join('\n\n'), }; } // Hangul-presence is a cheap, reliable enough signal for which language SearXNG's engines // should be told to prefer — matches the existing bilingual query-crafting convention // (Korean text for domestic queries, English for international) documented alongside // isNewsSeekingQuery, just expressed as an explicit API param instead of only query wording. function detectQueryLanguage(query: string): 'ko' | 'en' { return /[가-힣]/.test(query) ? 'ko' : 'en'; } // ── SearXNG (self-hosted/public metasearch, no key) ────────────────────────── async function searchSearXNG( query: string, limit: number, baseUrl: string, opts?: { category?: string; timeRange?: 'day' | 'week' | 'month' | 'year' }, ): Promise { const base = baseUrl.replace(/\/+$/, ''); // Default (no categories param) hits SearXNG's "general" category, which includes engines // like Wikipedia — fine for most queries, but Wikipedia's static reference pages (e.g. a // "2026" year-overview article) are not news and shouldn't compete with actual dated // articles for a news-seeking query. Restricting to categories=news routes to the engines // actually tagged "news" (daum news, yahoo news, presearch's news variant) instead. const categoryParam = opts?.category ? `&categories=${encodeURIComponent(opts.category)}` : ''; // time_range narrows results to engines' own recency metadata — a stronger filter than // categories=news alone against stale-but-still-"news-tagged" pages. 'week' rather than // 'day' to leave margin for engines with indexing lag instead of returning nothing. const timeRangeParam = opts?.timeRange ? `&time_range=${encodeURIComponent(opts.timeRange)}` : ''; const languageParam = `&language=${encodeURIComponent(detectQueryLanguage(query))}`; const url = `${base}/search?q=${encodeURIComponent(query)}&format=json${categoryParam}${timeRangeParam}${languageParam}`; const res = await fetch(url, { headers: { 'Accept': 'application/json', 'User-Agent': 'SmallClaw/1.0' }, signal: AbortSignal.timeout(15_000), }); if (!res.ok) throw new Error(`SearXNG HTTP ${res.status}`); const data: any = await res.json(); const raw: SearchResultItem[] = (data.results || []).slice(0, limit).map((r: any) => ({ title: r.title || '', url: r.url || '', snippet: r.content || '', })); const ranked = rankResults(query, raw); const answer = buildDirectPriceAnswer(query, ranked); return { success: true, data: { query, results: ranked, answer: answer || undefined }, stdout: (answer ? `${answer}\n\n` : '') + ranked.map((r, i) => `[${i + 1}] ${r.title}\n ${r.url}\n ${r.snippet.slice(0, 400)}` ).join('\n\n'), }; } // ── Ollama Cloud web search API (ollama.com, requires OLLAMA_API_KEY) ───────── async function searchOllamaCloud(query: string, limit: number, apiKey: string): Promise { const res = await fetch('https://ollama.com/api/web_search', { method: 'POST', headers: { 'Authorization': `Bearer ${apiKey}`, 'Content-Type': 'application/json' }, body: JSON.stringify({ query }), signal: AbortSignal.timeout(15_000), }); if (!res.ok) throw new Error(`Ollama web search HTTP ${res.status}`); const data: any = await res.json(); const raw: SearchResultItem[] = (data.results || []).slice(0, limit).map((r: any) => ({ title: r.title || '', url: r.url || '', snippet: r.content || '', })); const ranked = rankResults(query, raw); const answer = buildDirectPriceAnswer(query, ranked); return { success: true, data: { query, results: ranked, answer: answer || undefined }, stdout: (answer ? `${answer}\n\n` : '') + ranked.map((r, i) => `[${i + 1}] ${r.title}\n ${r.url}\n ${r.snippet.slice(0, 400)}` ).join('\n\n'), }; } // ── DuckDuckGo JSON endpoint (no key, more stable than HTML scrape) ─────────── async function searchDDG(query: string, limit: number): Promise { // DDG instant answer API — gives structured results without scraping HTML const url = `https://api.duckduckgo.com/?q=${encodeURIComponent(query)}&format=json&no_redirect=1&no_html=1&skip_disambig=1`; const res = await fetch(url, { headers: { 'User-Agent': 'SmallClaw/1.0' }, signal: AbortSignal.timeout(12_000), }); if (!res.ok) throw new Error(`DDG JSON HTTP ${res.status}`); const data: any = await res.json(); const results: Array<{ title: string; url: string; snippet: string }> = []; // Abstract (direct answer) if (data.AbstractText) { results.push({ title: data.Heading || query, url: data.AbstractURL || '', snippet: data.AbstractText, }); } // Related topics for (const topic of (data.RelatedTopics || [])) { if (results.length >= limit) break; if (topic.Text && topic.FirstURL) { results.push({ title: topic.Text.slice(0, 80), url: topic.FirstURL, snippet: topic.Text }); } else if (topic.Topics) { for (const sub of topic.Topics) { if (results.length >= limit) break; if (sub.Text && sub.FirstURL) { results.push({ title: sub.Text.slice(0, 80), url: sub.FirstURL, snippet: sub.Text }); } } } } // Results array for (const r of (data.Results || [])) { if (results.length >= limit) break; results.push({ title: r.Text || '', url: r.FirstURL || '', snippet: r.Text || '' }); } if (results.length === 0) { // Fall back to HTML scraper if JSON gave nothing return searchDDGHtml(query, limit); } const ranked = rankResults(query, results); const answer = buildDirectPriceAnswer(query, ranked); return { success: true, data: { query, results: ranked, answer: answer || undefined }, stdout: (answer ? `${answer}\n\n` : '') + ranked.map((r, i) => `[${i + 1}] ${r.title}\n ${r.url}\n ${r.snippet.slice(0, 400)}` ).join('\n\n'), }; } // ── DDG HTML scraper (last resort fallback) ─────────────────────────────────── async function searchDDGHtml(query: string, limit: number): Promise { const url = `https://html.duckduckgo.com/html/?q=${encodeURIComponent(query)}`; const res = await fetch(url, { headers: { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) SmallClaw/1.0' }, signal: AbortSignal.timeout(15_000), }); if (!res.ok) return { success: false, error: `DDG HTML HTTP ${res.status}` }; const html = await res.text(); const results: Array<{ title: string; url: string; snippet: string }> = []; const re = /]*>([^<]+)<\/a>[\s\S]*?]*>([\s\S]*?)<\/a>/g; let m; while ((m = re.exec(html)) !== null && results.length < limit) { const href = m[1]; const realUrl = href.startsWith('/l/?') || href.startsWith('//duckduckgo.com/l/?') ? decodeURIComponent(href.replace(/.*uddg=/, '')) : href; results.push({ title: m[2].trim(), url: realUrl, snippet: m[3].replace(/<[^>]+>/g, '').trim(), }); } if (results.length === 0) { return { success: false, error: 'No search results found. DDG may have changed its markup.' }; } const ranked = rankResults(query, results); const answer = buildDirectPriceAnswer(query, ranked); return { success: true, data: { query, results: ranked, answer: answer || undefined }, stdout: (answer ? `${answer}\n\n` : '') + ranked.map((r, i) => `[${i + 1}] ${r.title}\n ${r.url}\n ${r.snippet}`).join('\n\n'), }; } // ── Shared fallback chain for search-then-synthesize callers ───────────────── // Delegates to executeWebSearch so ollama_web_search uses the SAME provider chain // as web_search (searxng → ollama_cloud → tavily → google → brave → ddg → ddg_html), // including the empty-result guard (21367ea) and event-outcome enrichment. The two // paths previously diverged here — ollama_web_search skipped tavily/google/brave and // had no diagnostics — which stayed dormant only while those API keys were unset. // searchOllama only consumes res.stdout, so the diagnostics/price-answer metadata // executeWebSearch attaches to res.data is harmless here. async function searchForSynthesis(query: string, limit: number): Promise { return executeWebSearch({ query, max_results: limit }); } // ── Ollama web-search (model-assisted) ──────────────────────────────────────── // Search is executed deterministically in code FIRST, then the model only // summarizes the real results — it is never given the option to answer from // its own parametric memory instead of searching. An earlier version offered // the model a web_search tool and let it decide whether to call it; when it // skipped the tool (common with small local models), the code returned the // model's unsourced guess as if it were a search result (e.g. a fabricated // weather report with a date over a year stale). Never let the model choose // whether to search — only let it choose what to say about real results. async function searchOllama(query: string, limit: number, endpoint: string, model: string): Promise { const searchRes = await searchForSynthesis(query, limit); if (!searchRes.success) { return { success: false, error: `검색 실패: ${searchRes.error}` }; } if (!searchRes.stdout) { return { success: true, stdout: `"${query}"에 대한 검색 결과가 없습니다.`, data: { query, provider: 'ollama', searchQuery: query } }; } const synthRes = await fetch(`${endpoint}/api/chat`, { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ model, messages: [ { role: 'system', content: '아래 검색 결과만 근거로 질문에 답하세요. 검색 결과에 없는 내용은 추측하지 말고, 근거가 부족하면 그렇다고 말하세요.' }, { role: 'user', content: `질문: ${query}\n\n검색 결과:\n${searchRes.stdout}` }, ], stream: false, }), signal: AbortSignal.timeout(30_000), }); if (!synthRes.ok) throw new Error(`Ollama synthesis HTTP ${synthRes.status}`); const synthData: any = await synthRes.json(); const answer = String(synthData.message?.content || searchRes.stdout); return { success: true, stdout: answer, data: { query, provider: 'ollama', searchQuery: query, answer }, }; } // A provider returning ANY result (even one) short-circuits the fallback chain below — // which is right for most queries, but for news/headline queries a "result" that's actually // just a bare homepage or a Wikipedia/Namu year-overview page isn't real news: it stops the // chain from ever reaching Tavily/Google, which might have had actual headlines. Downgrade // such results to "empty" (still kept as bestEmpty last-resort) so the chain keeps going — // but only for queries that are actually asking for news, so a legitimate "무슨 사이트야" / // "give me the Reuters homepage" style query isn't penalized for getting a homepage back. function isNewsSeekingQuery(query: string): boolean { return /(뉴스|헤드라인|속보)/.test(query) || /\b(news|headlines?)\b/i.test(query); } function isLowValueNewsResult(url: string): boolean { try { const u = new URL(url); const path = u.pathname.replace(/\/+$/, ''); if (!path) return true; // bare homepage, e.g. https://www.reuters.com/ if (/^\/wiki\/\d{4}$/i.test(path) && /wikipedia\.org$/i.test(u.hostname)) return true; // year-overview page if (/^\/w\/\d{4}$/i.test(path) && /namu\.wiki$/i.test(u.hostname)) return true; return false; } catch { return false; } } // ── Comparison-query splitting ──────────────────────────────────────────────── // // "A vs B " queries reliably return review/overview pages that name both products and // give the figures for neither, while searching each product separately returns the numbers // immediately. Verified 2026-08-10: "RTX 3090 vs RTX 4080 Super memory bandwidth" produced a // generic overview with no figure; the two separate searches produced 936 GB/s and 736 GB/s. // // That finding went into the web_search tool description as an instruction the same day, and the // model ignored it — production log 2026-08-11 shows three consecutive turns issuing the exact // combined form it was told not to use ("RTX 4080 vs RTX 5060 performance comparison specs"), // each returning dead technical.city links, each producing an answer of pure generalities with no // number in it. Same lesson as news_search's country/category params and weather's multi-city // batching earlier that day: a tool description cannot enforce query construction on this model, // so the split happens here in code instead ([[feedback_local_model_needs_code_backstop]]). // // The combined query still runs — when a comparison page IS alive it is genuinely the best source // for this question, and dropping it would trade one failure mode for another. The split searches // are additive and capped tighter so the extra grounding does not blow up the context. const COMPARISON_SEPARATOR = /\s+(?:vs\.?|versus)\s+|\s*와\s+|\s*과\s+/i; /** Words describing WHAT is being compared — kept, since they narrow each single-entity search. */ const COMPARISON_ATTRIBUTE = /\b(performance|specs?|specifications?|benchmarks?|review|memory\s*bandwidth|bandwidth|tflops|vram|price|speed)\b|성능|스펙|사양|벤치마크|대역폭|가격|속도/gi; /** Words meaning "compare these" — dropped, since they are meaningless in a single-entity search. */ const COMPARISON_VERB = /\b(comparison|compare[ds]?|versus|vs\.?|difference|diff)\b|비교|차이/gi; export interface ComparisonSplit { a: string; b: string; } /** * Split "RTX 4080 vs RTX 5060 performance comparison specs" into * "RTX 4080 performance specs" + "RTX 5060 performance specs", or null when the query is not a * spec comparison. Deliberately conservative: "Lakers vs Celtics" has no attribute word and no * model numbers, so it is left alone rather than turned into two unrelated searches. */ export function splitComparisonQuery(query: string): ComparisonSplit | null { const q = String(query || '').trim(); if (!q) return null; const parts = q.split(COMPARISON_SEPARATOR); if (parts.length !== 2) return null; const attributes = Array.from(new Set( (q.match(COMPARISON_ATTRIBUTE) || []).map(s => s.trim().toLowerCase()), )); const strip = (s: string) => s .replace(COMPARISON_ATTRIBUTE, ' ') .replace(COMPARISON_VERB, ' ') .replace(/\s+/g, ' ') .trim(); const [entityA, entityB] = parts.map(strip); if (entityA.length < 2 || entityB.length < 2) return null; // Either an explicit attribute ("performance", "스펙") or two model-number-shaped entities. // Without one of those this is not a spec comparison and splitting would just lose meaning. const bothLookLikeModels = /\d/.test(entityA) && /\d/.test(entityB); if (!attributes.length && !bothLookLikeModels) return null; const tail = attributes.length ? attributes.join(' ') : 'specs'; return { a: `${entityA} ${tail}`.trim(), b: `${entityB} ${tail}`.trim() }; } // ── Main web_search tool ────────────────────────────────────────────────────── export async function executeWebSearch(args: { query: string; max_results?: number; _noSplit?: boolean }): Promise { if (!args._noSplit) { const split = splitComparisonQuery(args.query || ''); if (split) return runComparisonSearch(args, split); } if (!args.query?.trim()) return { success: false, error: 'query is required' }; let limit = Math.min(args.max_results ?? 5, 10); if (isPriceQuery(args.query)) limit = Math.max(limit, 5); const cfg = getSearchConfig(); // ddg sits last: its Instant Answer API only covers infobox-style queries and its // HTML scrape fallback is routinely bot-walled (see searchDDGHtml) — it rarely adds // value once searxng/ollama_cloud/tavily/google/brave have all had a shot. const candidates: Array<'tavily' | 'searxng' | 'google' | 'brave' | 'ollama_cloud' | 'ddg'> = ['searxng', 'ollama_cloud', 'tavily', 'google', 'brave', 'ddg']; const providerOrder = [cfg.preferred, ...candidates.filter(p => p !== cfg.preferred)]; const diagnostics: SearchDiagnostics = { query: args.query, preferred_provider: cfg.preferred, provider_order: providerOrder, attempted: [], }; const runProvider = (provider: SearchProvider): Promise => { switch (provider) { case 'tavily': return searchTavily(args.query, limit, cfg.tavilyKey as string); case 'searxng': return searchSearXNG(args.query, limit, cfg.searxngUrl as string, isNewsSeekingQuery(args.query) ? { category: 'news', timeRange: 'week' } : undefined); case 'google': return searchGoogle(args.query, limit, cfg.googleKey as string, cfg.googleCx as string); case 'brave': return searchBrave(args.query, limit, cfg.braveKey as string); case 'ollama_cloud': return searchOllamaCloud(args.query, limit, cfg.ollamaApiKey as string); case 'ddg': return searchDDG(args.query, limit); default: throw new Error(`unhandled provider: ${provider}`); } }; let lastErr = null; // A provider that returned HTTP-OK but zero results is kept as a last-resort answer — // better than a hard error if every provider ends up empty for this query. let bestEmpty: ToolResult | null = null; let bestEmptyProvider: SearchProvider | null = null; for (const provider of providerOrder) { if (provider === 'tavily' && !cfg.tavilyKey) { diagnostics.attempted.push({ provider, status: 'skipped', reason: 'missing_tavily_api_key' }); continue; } if (provider === 'google' && (!cfg.googleKey || !cfg.googleCx)) { diagnostics.attempted.push({ provider, status: 'skipped', reason: !cfg.googleKey ? 'missing_google_api_key' : 'missing_google_cx' }); continue; } if (provider === 'brave' && !cfg.braveKey) { diagnostics.attempted.push({ provider, status: 'skipped', reason: 'missing_brave_api_key' }); continue; } if (provider === 'searxng' && !cfg.searxngUrl) { diagnostics.attempted.push({ provider, status: 'skipped', reason: 'missing_searxng_url' }); continue; } if (provider === 'ollama_cloud' && !cfg.ollamaApiKey) { diagnostics.attempted.push({ provider, status: 'skipped', reason: 'missing_ollama_api_key' }); continue; } const started = Date.now(); try { const res = await runProvider(provider); await augmentEventContract(args.query, res); const resultCount = Array.isArray(res.data?.results) ? res.data.results.length : 0; let hasResults = res.success && resultCount > 0; if (hasResults && isNewsSeekingQuery(args.query)) { const results = res.data!.results as SearchResultItem[]; if (results.every((r) => isLowValueNewsResult(r.url))) hasResults = false; } diagnostics.attempted.push({ provider, status: hasResults ? 'success' : (res.success ? 'empty' : 'failed'), duration_ms: Date.now() - started, result_count: resultCount, ...(!res.success && { reason: res.error }), }); if (hasResults) { diagnostics.selected_provider = provider; res.data = { ...(res.data || {}), provider, search_diagnostics: diagnostics }; return res; } if (res.success && !bestEmpty) { bestEmpty = res; bestEmptyProvider = provider; } } catch (err) { lastErr = err; diagnostics.attempted.push({ provider, status: 'failed', reason: (err as any)?.message || String(err), duration_ms: Date.now() - started, }); } } // Final fallback: DDG HTML scrape (rarely succeeds — see comment on `candidates` above) const fallbackStarted = Date.now(); try { const res = await searchDDGHtml(args.query, limit); const resultCount = Array.isArray(res.data?.results) ? res.data.results.length : 0; let hasResults = res.success && resultCount > 0; if (hasResults && isNewsSeekingQuery(args.query)) { const results = res.data!.results as SearchResultItem[]; if (results.every((r) => isLowValueNewsResult(r.url))) hasResults = false; } diagnostics.attempted.push({ provider: 'ddg_html', status: hasResults ? 'success' : (res.success ? 'empty' : 'failed'), duration_ms: Date.now() - fallbackStarted, result_count: resultCount, ...(!res.success && { reason: res.error }), }); if (hasResults) { diagnostics.selected_provider = 'ddg_html'; res.data = { ...(res.data || {}), provider: 'ddg_html', search_diagnostics: diagnostics }; return res; } if (res.success && !bestEmpty) { bestEmpty = res; bestEmptyProvider = 'ddg_html'; } } catch (err) { lastErr = err; diagnostics.attempted.push({ provider: 'ddg_html', status: 'failed', reason: (err as any)?.message || String(err), duration_ms: Date.now() - fallbackStarted, }); } if (bestEmpty) { diagnostics.selected_provider = bestEmptyProvider as SearchProvider; bestEmpty.data = { ...(bestEmpty.data || {}), provider: bestEmptyProvider, search_diagnostics: diagnostics }; return bestEmpty; } let errMsg = 'unknown error'; if (lastErr) { if (typeof lastErr === 'object' && 'message' in lastErr) errMsg = (lastErr as any).message; else errMsg = String(lastErr); } return { success: false, error: `All search providers failed: ${errMsg}`, data: { query: args.query, search_diagnostics: diagnostics }, }; } // web_search results go straight to the primary chat model, which has to pull a specific // fact (e.g. a VRAM number) out of noisy snippets while also juggling a long system prompt, // conversation history and other tool results — measured 2026-08-08: it missed a figure that // was plainly present in the raw results, answering "검색결과에 수치가 없다" when it was there. // ollama_web_search already solves this by re-asking a model a single narrow question // ("answer only from this text, say so if it's not there") with nothing else in its context, // which reliably surfaces the fact. Reuse that same pattern here — prepend a focused extraction // on top of the raw results (not instead of: citations still need the original list/URLs). async function extractAnswerFromResults(query: string, rawResults: string): Promise { try { const { endpoint, model } = getOllamaConfig(); const res = await fetch(`${endpoint}/api/chat`, { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ model, messages: [ { role: 'system', content: '아래 검색 결과만 근거로 질문에 답하세요. 검색 결과에 없는 내용은 추측하지 말고, 근거가 부족하면 그렇다고 말하세요. 2~4문장으로 간결하게.' }, { role: 'user', content: `질문: ${query}\n\n검색 결과:\n${rawResults}` }, ], stream: false, }), signal: AbortSignal.timeout(15_000), }); if (!res.ok) return null; const data: any = await res.json(); const answer = String(data.message?.content || '').trim(); return answer || null; } catch { // Extraction is a best-effort add-on — any failure (timeout, endpoint down) must fall // back to the raw results silently rather than break web_search itself. return null; } } /** * Runs the combined query plus one search per entity, and hands the model all three labelled. * Partial failure is fine — any section that came back with results is still grounding the model * did not have before, so this only ever falls back to whatever the combined query alone returned. */ async function runComparisonSearch( args: { query: string; max_results?: number }, split: ComparisonSplit, ): Promise { const splitLimit = Math.min(args.max_results ?? 5, 3); const [combined, a, b] = await Promise.all([ executeWebSearch({ ...args, _noSplit: true }), executeWebSearch({ query: split.a, max_results: splitLimit, _noSplit: true }), executeWebSearch({ query: split.b, max_results: splitLimit, _noSplit: true }), ]); const sections: string[] = []; const push = (label: string, r: ToolResult) => { if (r.success && String(r.stdout || '').trim()) sections.push(`[${label}]\n${String(r.stdout).trim()}`); }; push(`검색: ${args.query}`, combined); push(`검색: ${split.a}`, a); push(`검색: ${split.b}`, b); if (!sections.length) return combined; console.log(`[v2] web_search comparison split: "${args.query}" → "${split.a}" + "${split.b}"`); return { success: true, data: { ...(combined.data as any || {}), comparison_split: [split.a, split.b] }, stdout: `비교 질문이라 각 대상을 따로 검색했습니다. 아래 세 검색 결과를 모두 근거로 쓰세요.\n\n${sections.join('\n\n')}`, }; } export async function executeWebSearchWithExtraction(args: { query: string; max_results?: number }): Promise { const res = await executeWebSearch(args); if (!res.success || !res.stdout) return res; const extracted = await extractAnswerFromResults(args.query, res.stdout); if (!extracted) return res; return { ...res, stdout: `[핵심 답변]\n${extracted}\n\n[검색 결과 원문]\n${res.stdout}`, }; } // ── web_fetch: fetch a URL and return clean text ────────────────────────────── export async function executeWebFetch(args: { url: string; max_chars?: number }): Promise { if (!args.url?.trim()) return { success: false, error: 'url is required' }; const maxChars = args.max_chars ?? 10_000; try { const res = await fetch(args.url, { headers: { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) SmallClaw/1.0' }, signal: AbortSignal.timeout(20_000), redirect: 'follow', }); if (!res.ok) return { success: false, error: `HTTP ${res.status} from ${args.url}` }; const contentType = res.headers.get('content-type') ?? ''; if (!contentType.includes('text') && !contentType.includes('json')) { return { success: false, error: `Non-text content-type: ${contentType}` }; } let text = stripHtml(await res.text(), true); if (text.length > maxChars) text = text.slice(0, maxChars) + '\n\n[...truncated]'; return { success: true, data: { url: args.url, length: text.length }, stdout: text, }; } catch (err: any) { return { success: false, error: `Fetch failed: ${err.message}` }; } } export const webSearchTool = { name: 'web_search', description: 'Search the web. Uses Tavily or Brave API if configured in ~/.smallclaw/config.json (search.tavily_api_key / search.brave_api_key), otherwise falls back to DuckDuckGo (no key needed).', execute: executeWebSearchWithExtraction, schema: { query: 'string (required) - Search query', max_results: 'number (optional, default 5) - Max results to return', }, }; export const webFetchTool = { name: 'web_fetch', description: 'Fetch and extract the text content of any URL. Good for reading articles, docs, or pages found via web_search.', execute: executeWebFetch, schema: { url: 'string (required) - Full URL to fetch (include https://)', max_chars: 'number (optional, default 10000) - Max characters to return', }, }; export function getOllamaConfig(): { endpoint: string; model: string } { try { const cm = getConfig(); const data = cm.getConfig(); const endpoint = String(data.ollama?.endpoint || 'http://localhost:11434'); const model = String(data.llm?.providers?.ollama?.model || data.models?.primary || 'llama3'); return { endpoint, model }; } catch {} return { endpoint: 'http://localhost:11434', model: 'llama3' }; } export const ollamaWebSearchTool = { name: 'ollama_web_search', description: '코드가 먼저 웹 검색을 수행하고(searxng→ollama_cloud→tavily→google→brave→ddg 폴백 체인, web_search와 동일) — 그 결과만 근거로 Ollama chat API(config의 ollama.endpoint+model, 변경 가능)가 자연어 답변을 요약·종합합니다. 모델은 검색을 직접 수행하거나 검색 여부를 결정하지 않음 — 코드가 무조건 검색한 결과 위에서만 답합니다(환각 방지). 참고: ollama_cloud provider(Ollama.com 웹서치 API)는 위 체인의 검색 후보 중 하나일 뿐, 이 툴과 별개.', schema: { query: 'string (required) - 검색 질문 또는 키워드', max_results: 'number (optional, default 5) - 최대 검색 결과 수', }, execute: async (args: { query: string; max_results?: number }): Promise => { if (!args.query?.trim()) return { success: false, error: 'query is required' }; const { endpoint, model } = getOllamaConfig(); const limit = Math.min(args.max_results ?? 5, 10); try { return await searchOllama(args.query, limit, endpoint, model); } catch (err: any) { return { success: false, error: `ollama_web_search failed: ${err.message}` }; } }, }; export const ollamaWebFetchTool = { name: 'ollama_web_fetch', description: 'URL을 가져온 후 Ollama 모델이 내용을 요약합니다. web_fetch로 가져온 원문 대신 모델이 핵심만 정리해서 반환합니다.', schema: { url: 'string (required) - 가져올 URL (https:// 포함)', instruction: 'string (optional) - 요약 지시 (예: "주요 수치만 뽑아줘", "3줄 요약")', }, execute: async (args: { url: string; instruction?: string }): Promise => { if (!args.url?.trim()) return { success: false, error: 'url is required' }; const { endpoint, model } = getOllamaConfig(); try { const fetchResult = await executeWebFetch({ url: args.url, max_chars: 8000 }); if (!fetchResult.success) return fetchResult; const pageText = fetchResult.stdout || ''; const prompt = args.instruction ? `다음 웹 페이지 내용을 읽고 "${args.instruction}":\n\n${pageText}` : `다음 웹 페이지 내용을 핵심 위주로 간결하게 요약해줘:\n\n${pageText}`; const res = await fetch(`${endpoint}/api/chat`, { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ model, messages: [{ role: 'user', content: prompt }], stream: false }), signal: AbortSignal.timeout(60_000), }); if (!res.ok) throw new Error(`Ollama HTTP ${res.status}`); const data: any = await res.json(); const answer = String(data.message?.content || ''); if (!answer) throw new Error('Ollama returned empty response'); return { success: true, stdout: answer, data: { url: args.url, provider: 'ollama', answer } }; } catch (err: any) { return { success: false, error: `ollama_web_fetch failed: ${err.message}` }; } }, };