This commit is contained in:
kim
2026-05-02 23:15:25 +09:00
parent ec53a112b4
commit 83eb31f121
62 changed files with 1465 additions and 453 deletions
+13 -8
View File
@@ -177,17 +177,22 @@ function getLocalConfigFilePath(): string {
return fsSync.existsSync(projectCfg) ? projectCfg : path.join(os.homedir(), '.smallclaw', 'config.json');
}
let _retrievalModeCache: { mode: RetrievalMode; expiresAt: number } | null = null;
function getRetrievalMode(): RetrievalMode {
const now = Date.now();
if (_retrievalModeCache && now < _retrievalModeCache.expiresAt) return _retrievalModeCache.mode;
let mode: RetrievalMode = 'standard';
try {
const p = getLocalConfigFilePath();
if (!fsSync.existsSync(p)) return 'standard';
const raw = JSON.parse(fsSync.readFileSync(p, 'utf-8'));
const mode = String(raw?.agent_policy?.retrieval_mode || 'standard').toLowerCase();
if (mode === 'fast' || mode === 'deep') return mode;
return 'standard';
} catch {
return 'standard';
}
if (fsSync.existsSync(p)) {
const raw = JSON.parse(fsSync.readFileSync(p, 'utf-8'));
const m = String(raw?.agent_policy?.retrieval_mode || 'standard').toLowerCase();
if (m === 'fast' || m === 'deep') mode = m;
}
} catch {}
_retrievalModeCache = { mode, expiresAt: now + 60_000 };
return mode;
}
function retrievalMaxLines(mode: RetrievalMode): number {
+126
View File
@@ -0,0 +1,126 @@
import { spawn } from 'child_process';
import path from 'path';
import fs from 'fs';
import { getConfig } from '../config/config.js';
import { ToolResult } from '../types.js';
function getWorkspacePath(): string {
try {
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
} catch {
return path.join(process.cwd(), 'workspace');
}
}
const IMAGE_EXTS = new Set(['.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.tiff', '.tif']);
// Runs OCR in a child process so the tesseract worker doesn't block the main event loop.
const OCR_SCRIPT = `
(async () => {
const imagePath = process.argv[2];
const lang = process.argv[3] || 'eng';
try {
const mod = await import('tesseract.js');
const createWorker = mod?.createWorker || mod?.default?.createWorker;
if (typeof createWorker !== 'function') {
process.stdout.write(JSON.stringify({ error: 'tesseract not available' }));
return;
}
const worker = await createWorker(lang);
if (worker && typeof worker.loadLanguage === 'function') {
await worker.loadLanguage(lang);
await worker.initialize(lang);
}
const out = await worker.recognize(imagePath);
if (worker && typeof worker.terminate === 'function') await worker.terminate();
const text = String(out?.data?.text || '')
.replace(/\\r/g, '')
.replace(/[ \\t]+\\n/g, '\\n')
.replace(/\\n{3,}/g, '\\n\\n')
.trim();
process.stdout.write(JSON.stringify({ text, confidence: Number(out?.data?.confidence || 0) }));
} catch (e) {
process.stdout.write(JSON.stringify({ error: String(e?.message || e) }));
}
})().catch(e => process.stdout.write(JSON.stringify({ error: String(e?.message || e) })));
`;
export const imageReadTool = {
name: 'image_read',
description: 'Read an image file and extract text via OCR. Returns filename, file size, and extracted text. Useful for screenshots, scanned documents, and diagrams containing text.',
schema: {
path: 'Path to the image file (PNG, JPG, WEBP, BMP, TIFF) — absolute or relative to workspace',
lang: 'OCR language: "eng" (English, default), "kor" (Korean), "kor+eng" (both)',
},
jsonSchema: {
type: 'object',
properties: {
path: { type: 'string', description: 'Path to the image file' },
lang: { type: 'string', description: 'OCR language code: "eng" | "kor" | "kor+eng" (default: "eng")' },
},
required: ['path'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = getWorkspacePath();
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!fs.existsSync(resolved)) {
return { success: false, error: `File not found: ${resolved}` };
}
const ext = path.extname(resolved).toLowerCase();
if (!IMAGE_EXTS.has(ext)) {
return { success: false, error: `Unsupported format "${ext}". Supported: ${[...IMAGE_EXTS].join(', ')}` };
}
const stat = fs.statSync(resolved);
const sizeMB = (stat.size / 1024 / 1024).toFixed(2);
const lang = String(args?.lang || 'eng').replace(/[^a-zA-Z+]/g, '');
const ocrResult = await new Promise<{ text?: string; confidence?: number; error?: string }>((resolve) => {
const child = spawn(process.execPath, ['--input-type=module', '-e', OCR_SCRIPT, resolved, lang], {
timeout: 90_000,
env: { ...process.env },
});
let out = '';
child.stdout.on('data', (d: Buffer) => { out += d.toString(); });
child.on('close', () => {
try { resolve(JSON.parse(out || '{}')); } catch { resolve({ error: 'Failed to parse OCR output' }); }
});
child.on('error', (e: Error) => resolve({ error: e.message }));
});
const lines: string[] = [
`File: ${path.basename(resolved)}`,
`Size: ${sizeMB} MB | Format: ${ext.slice(1).toUpperCase()} | Lang: ${lang}`,
];
if (ocrResult.error) {
lines.push(`OCR Error: ${ocrResult.error}`);
return { success: false, error: lines.join('\n') };
}
if (!ocrResult.text) {
lines.push('OCR: No text detected in image.');
} else {
lines.push(`OCR Confidence: ${(ocrResult.confidence ?? 0).toFixed(1)}%`, '');
lines.push('[Extracted Text]');
lines.push(ocrResult.text.slice(0, 15_000));
if (ocrResult.text.length > 15_000) lines.push('[... truncated]');
}
return {
success: true,
stdout: lines.join('\n'),
data: {
path: resolved,
size_bytes: stat.size,
ocr_text: ocrResult.text || '',
confidence: ocrResult.confidence ?? 0,
},
};
},
};
+86
View File
@@ -0,0 +1,86 @@
import { execFile } from 'child_process';
import { promisify } from 'util';
import path from 'path';
import fs from 'fs';
import { getConfig } from '../config/config.js';
import { ToolResult } from '../types.js';
const execFileAsync = promisify(execFile);
function getWorkspacePath(): string {
try {
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
} catch {
return path.join(process.cwd(), 'workspace');
}
}
export const pdfReadTool = {
name: 'pdf_read',
description: 'Extract text from a PDF file. Returns full text content, optionally limited to a page range. Ideal for reading research papers, reports, and documents.',
schema: {
path: 'Path to the PDF file (absolute, or relative to workspace)',
page_from: 'First page to extract, 1-indexed (default: 1)',
page_to: 'Last page to extract, inclusive (default: last page)',
max_chars: 'Maximum characters to return (default: 20000, max: 100000)',
},
jsonSchema: {
type: 'object',
properties: {
path: { type: 'string', description: 'Path to the PDF file' },
page_from: { type: 'number', description: 'First page (1-indexed)' },
page_to: { type: 'number', description: 'Last page (inclusive)' },
max_chars: { type: 'number', description: 'Max chars to return (default: 20000)' },
},
required: ['path'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = getWorkspacePath();
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!fs.existsSync(resolved)) {
return { success: false, error: `File not found: ${resolved}` };
}
if (path.extname(resolved).toLowerCase() !== '.pdf') {
return { success: false, error: 'File must have a .pdf extension' };
}
const maxChars = Math.min(100_000, Math.max(1_000, Number(args?.max_chars ?? 20_000)));
const cmdArgs: string[] = ['-layout', '-enc', 'UTF-8'];
if (args?.page_from) cmdArgs.push('-f', String(Math.max(1, Math.floor(Number(args.page_from)))));
if (args?.page_to) cmdArgs.push('-l', String(Math.max(1, Math.floor(Number(args.page_to)))));
cmdArgs.push(resolved, '-');
let stdout: string;
try {
const result = await execFileAsync('pdftotext', cmdArgs, { maxBuffer: 20 * 1024 * 1024, timeout: 30_000 });
stdout = result.stdout;
} catch (err: any) {
const msg = String(err.message || '');
if (msg.toLowerCase().includes('encrypt') || msg.toLowerCase().includes('password')) {
return { success: false, error: 'PDF is password-protected or encrypted' };
}
return { success: false, error: `pdftotext failed: ${msg}` };
}
const text = stdout.replace(/\r/g, '').trim();
if (!text) {
return { success: false, error: 'No text extracted. The PDF may be scanned (image-based). Try image_read with OCR instead.' };
}
const truncated = text.length > maxChars;
const output = truncated ? text.slice(0, maxChars) : text;
return {
success: true,
stdout: truncated
? `${output}\n\n[Truncated at ${maxChars.toLocaleString()} chars. Full length: ~${text.length.toLocaleString()} chars]`
: output,
data: { chars: text.length, truncated },
};
},
};
+2
View File
@@ -415,6 +415,7 @@ export const pptxTool: import('./registry.js').Tool = {
type: 'object',
properties: {
type: { type: 'string', description: 'Slide type: "title", "content", "section", "image", or "blank"' },
layout: { type: 'string', description: 'Column layout: "split-right" (default: small title top, text left, image right), "split-left" (small title top, image left, text right), "text" (full-width text only), "fullscreen" (image fills slide, title/text overlaid at bottom)' },
title: { type: 'string', description: 'Slide title text' },
subtitle: { type: 'string', description: 'Subtitle (for title slides)' },
bullets: { type: 'array', items: { type: 'string' }, description: 'Bullet point texts' },
@@ -548,6 +549,7 @@ export const editPptxTool: import('./registry.js').Tool = {
type: 'object',
properties: {
type: { type: 'string', description: 'Slide type: "title", "content", "section", "image", or "blank"' },
layout: { type: 'string', description: 'Column layout: "split-right" (default: small title top, text left, image right), "split-left" (small title top, image left, text right), "text" (full-width text only), "fullscreen" (image fills slide, title/text overlaid at bottom)' },
title: { type: 'string', description: 'Slide title text' },
subtitle: { type: 'string', description: 'Subtitle (for title slides)' },
bullets: { type: 'array', items: { type: 'string' }, description: 'Bullet point texts' },
+124
View File
@@ -0,0 +1,124 @@
import { spawn } from 'child_process';
import path from 'path';
import { getConfig } from '../config/config.js';
import { ToolResult } from '../types.js';
function getWorkspacePath(): string {
try {
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
} catch {
return path.join(process.cwd(), 'workspace');
}
}
// Patterns blocked to prevent filesystem damage and network abuse.
// os.system / subprocess allow arbitrary shell — not allowed.
// socket / urllib / requests would still work for reading; only server-creation
// and raw socket misuse would need further restrictions if desired.
const BLOCKED: RegExp[] = [
/\bos\s*\.\s*system\s*\(/,
/\bsubprocess\s*\.\s*(run|Popen|call|check_output|check_call|getoutput|getstatusoutput)\s*\(/,
/\bos\s*\.\s*(remove|unlink|rmdir|removedirs|rename|replace|makedirs|mkdir)\s*\(/,
/\bshutil\s*\.\s*(rmtree|move|copy2?|copytree)\s*\(/,
/\b__import__\s*\(/,
/\bimportlib\s*\.\s*import_module\s*\(/,
];
export const pythonEvalTool = {
name: 'python_eval',
description: 'Execute Python code and return stdout/stderr. Runs in the workspace directory. Useful for math, data analysis, CSV/JSON processing, and scripting tasks. Standard library and installed packages (numpy, pandas, etc.) are available.',
schema: {
code: 'Python code to execute',
timeout: 'Execution timeout in seconds (default: 15, max: 60)',
packages: 'Comma-separated pip packages to install if missing before running (e.g. "numpy,pandas")',
},
jsonSchema: {
type: 'object',
properties: {
code: { type: 'string', description: 'Python code to execute' },
timeout: { type: 'number', description: 'Timeout in seconds (default: 15, max: 60)' },
packages: { type: 'string', description: 'Comma-separated packages to pip install if missing' },
},
required: ['code'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const code = String(args?.code || '').trim();
if (!code) return { success: false, error: 'code is required' };
for (const pattern of BLOCKED) {
if (pattern.test(code)) {
return { success: false, error: `Blocked: ${pattern.source.replace(/\\b|\\s\*/g, '').replace(/\\\./g, '.')} is not allowed` };
}
}
const timeoutSec = Math.min(60, Math.max(1, Number(args?.timeout ?? 15)));
const workspacePath = getWorkspacePath();
if (args?.packages) {
const pkgs = String(args.packages).split(',').map((p: string) => p.trim()).filter((p: string) => /^[a-zA-Z0-9_\-\.]+$/.test(p));
for (const pkg of pkgs) {
await new Promise<void>((resolve) => {
const pip = spawn('python3', ['-m', 'pip', 'install', '--quiet', '--user', pkg], {
cwd: workspacePath,
timeout: 60_000,
});
pip.on('close', resolve);
pip.on('error', resolve);
});
}
}
const result = await new Promise<{ stdout: string; stderr: string; exitCode: number; timedOut: boolean }>((resolve) => {
let stdout = '';
let stderr = '';
let timedOut = false;
let settled = false;
const child = spawn('python3', ['-u', '-c', code], {
cwd: workspacePath,
env: { ...process.env, PYTHONDONTWRITEBYTECODE: '1', PYTHONUNBUFFERED: '1' },
});
const killTimer = setTimeout(() => {
timedOut = true;
child.kill('SIGKILL');
}, timeoutSec * 1000);
child.stdout.on('data', (d: Buffer) => {
stdout += d.toString();
if (stdout.length > 50_000) child.kill('SIGKILL');
});
child.stderr.on('data', (d: Buffer) => { stderr += d.toString(); });
child.on('close', (code: number | null) => {
if (settled) return;
settled = true;
clearTimeout(killTimer);
resolve({ stdout, stderr, exitCode: code ?? -1, timedOut });
});
child.on('error', (err: NodeJS.ErrnoException) => {
if (settled) return;
settled = true;
clearTimeout(killTimer);
resolve({ stdout, stderr, exitCode: -1, timedOut });
});
});
if (result.timedOut) {
return { success: false, error: `Execution timed out after ${timeoutSec}s` };
}
const parts: string[] = [];
if (result.stdout.trim()) parts.push(`[stdout]\n${result.stdout.trim().slice(0, 20_000)}`);
if (result.stderr.trim()) parts.push(`[stderr]\n${result.stderr.trim().slice(0, 3_000)}`);
if (!result.stdout.trim() && !result.stderr.trim()) parts.push('(no output)');
parts.push(`[exit ${result.exitCode}]`);
return {
success: result.exitCode === 0,
stdout: parts.join('\n\n'),
error: result.exitCode !== 0 ? `Python exited with code ${result.exitCode}` : undefined,
data: { exit_code: result.exitCode, stdout: result.stdout, stderr: result.stderr },
};
},
};
+21
View File
@@ -13,6 +13,11 @@ import { proposeRepairTool } from './self-repair.js';
import { personaReadTool, personaUpdateTool } from './persona.js';
import { pptxTool, editPptxTool } from './pptx.js';
import { pubmedSearchTool, pubmedFetchTool, pubmedFulltextTool } from './pubmed.js';
import { openalexSearchTool, semanticSearchTool } from './scholar.js';
import { pdfReadTool } from './pdf.js';
import { imageReadTool } from './image.js';
import { pythonEvalTool } from './python.js';
import { sqliteTool } from './sqlite.js';
export interface Tool {
name: string;
@@ -47,6 +52,10 @@ const TOOL_PROFILE_TOOL_NAMES: Record<Exclude<ToolProfile, 'full'>, ReadonlySet<
'apply_patch',
'memory_search',
'memory_write',
'pdf_read',
'image_read',
'python_eval',
'sqlite_query',
]),
web: new Set([
'web_search',
@@ -56,6 +65,10 @@ const TOOL_PROFILE_TOOL_NAMES: Record<Exclude<ToolProfile, 'full'>, ReadonlySet<
'pubmed_search',
'pubmed_fetch',
'pubmed_fulltext',
'openalex_search',
'semantic_search',
'pdf_read',
'image_read',
]),
};
@@ -161,6 +174,14 @@ class ToolRegistry {
this.registerSafe(pubmedSearchTool);
this.registerSafe(pubmedFetchTool);
this.registerSafe(pubmedFulltextTool);
// OpenAlex / Semantic Scholar tools
this.registerSafe(openalexSearchTool);
this.registerSafe(semanticSearchTool);
// Document & data tools
this.registerSafe(pdfReadTool);
this.registerSafe(imageReadTool);
this.registerSafe(pythonEvalTool);
this.registerSafe(sqliteTool);
// Multi-agent spawn tool
this.registerSafe(spawnAgentTool);
}
+256
View File
@@ -0,0 +1,256 @@
import fs from 'fs';
import path from 'path';
import os from 'os';
import { ToolResult } from '../types.js';
const OPENALEX_BASE = 'https://api.openalex.org';
const S2_BASE = 'https://api.semanticscholar.org/graph/v1';
const POLITE_EMAIL = 'research@smallclaw.local';
function getScholarConfig(): { s2Key?: string } {
try {
const projectCfg = path.join(process.cwd(), '.smallclaw', 'config.json');
const cfgPath = fs.existsSync(projectCfg)
? projectCfg
: path.join(os.homedir(), '.smallclaw', 'config.json');
if (fs.existsSync(cfgPath)) {
const data = JSON.parse(fs.readFileSync(cfgPath, 'utf-8'));
return { s2Key: data.semantic_scholar?.api_key || '' };
}
} catch {}
return {};
}
async function fetchJson(url: string, headers?: Record<string, string>): Promise<any> {
const res = await fetch(url, {
signal: AbortSignal.timeout(20_000),
headers: { 'Accept': 'application/json', 'User-Agent': 'SmallClaw/1.0', ...headers },
});
if (!res.ok) throw new Error(`HTTP ${res.status}: ${await res.text().catch(() => '')}`);
return res.json();
}
function reconstructAbstract(inv: Record<string, number[]> | null | undefined): string {
if (!inv) return '';
const words: Record<number, string> = {};
for (const [word, positions] of Object.entries(inv)) {
for (const pos of positions) words[pos] = word;
}
return Object.keys(words)
.sort((a, b) => Number(a) - Number(b))
.map(k => words[Number(k)])
.join(' ');
}
// ── openalex_search ───────────────────────────────────────────────────────────
export const openalexSearchTool = {
name: 'openalex_search',
description: 'Search OpenAlex for academic papers across all fields (200M+ works). Returns title, authors, year, journal, DOI, citation count, and open-access status. No API key required.',
schema: {
query: 'Search query (title words, author, topic, etc.)',
max_results: 'Number of results (1-10, default 5)',
min_year: 'Filter to papers from this year onwards (e.g. 2020)',
open_access_only:'Return only open-access papers (true/false)',
sort: 'Sort order: "relevance" (default), "cited_by_count", "publication_date"',
},
jsonSchema: {
type: 'object',
properties: {
query: { type: 'string', description: 'Search query' },
max_results: { type: 'number', description: 'Number of results (1–10, default 5)' },
min_year: { type: 'number', description: 'Minimum publication year (e.g. 2020)' },
open_access_only: { type: 'boolean', description: 'Only return open-access papers' },
sort: { type: 'string', description: '"relevance" | "cited_by_count" | "publication_date"' },
},
required: ['query'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const query = String(args?.query || '').trim();
if (!query) return { success: false, error: 'query is required' };
const limit = Math.min(Math.max(1, Number(args?.max_results ?? 5)), 10);
const sortMap: Record<string, string> = {
relevance: 'relevance_score:desc',
cited_by_count: 'cited_by_count:desc',
publication_date: 'publication_date:desc',
};
const sortKey = sortMap[String(args?.sort || 'relevance')] ?? 'relevance_score';
const filters: string[] = ['type:article'];
if (args?.min_year) filters.push(`publication_year:${args.min_year}-`);
if (args?.open_access_only) filters.push('is_oa:true');
const params = new URLSearchParams({
search: query,
'per-page': String(limit),
sort: sortKey,
filter: filters.join(','),
select: 'id,doi,display_name,publication_year,cited_by_count,open_access,authorships,abstract_inverted_index,primary_location',
mailto: POLITE_EMAIL,
});
let data: any;
try {
data = await fetchJson(`${OPENALEX_BASE}/works?${params}`);
} catch (err: any) {
return { success: false, error: `OpenAlex request failed: ${err.message}` };
}
const results: any[] = data?.results ?? [];
if (!results.length) {
return { success: false, error: `No results found on OpenAlex for: "${query}"` };
}
const lines: string[] = [`OpenAlex results for: "${query}" (${data?.meta?.count ?? '?'} total)\n`];
for (let i = 0; i < results.length; i++) {
const r = results[i];
const title = r.display_name || '(no title)';
const year = r.publication_year ?? '';
const doi = r.doi ? String(r.doi).replace('https://doi.org/', '') : '';
const journal = r.primary_location?.source?.display_name || '';
const authors = (r.authorships || [])
.slice(0, 6)
.map((a: any) => a?.author?.display_name || '')
.filter(Boolean)
.join(', ') + ((r.authorships || []).length > 6 ? ' et al.' : '');
const citations = r.cited_by_count ?? 0;
const isOa = r.open_access?.is_oa ?? false;
const oaUrl = r.open_access?.oa_url || '';
const abstract = reconstructAbstract(r.abstract_inverted_index).slice(0, 400);
lines.push(`[${i + 1}] ${title}`);
if (authors) lines.push(` Authors: ${authors}`);
lines.push(` Year: ${year}${journal ? ` | Journal: ${journal}` : ''}${citations ? ` | Citations: ${citations}` : ''}`);
if (doi) lines.push(` DOI: https://doi.org/${doi}`);
if (isOa) lines.push(` Open Access: YES${oaUrl ? ` — ${oaUrl}` : ''}`);
if (abstract) lines.push(` Abstract: ${abstract}${abstract.length >= 400 ? '...' : ''}`);
lines.push('');
}
return {
success: true,
stdout: lines.join('\n'),
data: { query, count: data?.meta?.count, results: results.map((r: any) => ({
title: r.display_name,
year: r.publication_year,
doi: r.doi ? String(r.doi).replace('https://doi.org/', '') : '',
citations: r.cited_by_count,
is_oa: r.open_access?.is_oa,
oa_url: r.open_access?.oa_url,
authors: (r.authorships || []).slice(0, 6).map((a: any) => a?.author?.display_name).filter(Boolean),
})) },
};
},
};
// ── semantic_search ───────────────────────────────────────────────────────────
export const semanticSearchTool = {
name: 'semantic_search',
description: 'Search Semantic Scholar for academic papers. Returns title, authors, abstract, year, citation count, influential citation count, and open-access PDF link if available. Recommended for CS, AI, and interdisciplinary research.',
schema: {
query: 'Search query',
max_results: 'Number of results (1-10, default 5)',
year_from: 'Filter to papers from this year onwards (e.g. 2020)',
fields: 'Extra fields to request (comma-separated), e.g. "references,tldr"',
},
jsonSchema: {
type: 'object',
properties: {
query: { type: 'string', description: 'Search query' },
max_results: { type: 'number', description: 'Number of results (1–10, default 5)' },
year_from: { type: 'number', description: 'Minimum publication year (e.g. 2020)' },
fields: { type: 'string', description: 'Extra fields, e.g. "tldr,references"' },
},
required: ['query'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const query = String(args?.query || '').trim();
if (!query) return { success: false, error: 'query is required' };
const limit = Math.min(Math.max(1, Number(args?.max_results ?? 5)), 10);
const cfg = getScholarConfig();
const headers: Record<string, string> = {};
if (cfg.s2Key) headers['x-api-key'] = cfg.s2Key;
const baseFields = 'title,authors,abstract,year,externalIds,openAccessPdf,citationCount,influentialCitationCount';
const extraFields = args?.fields ? `,${String(args.fields).replace(/\s/g, '')}` : '';
const fields = baseFields + extraFields;
const params = new URLSearchParams({
query,
limit: String(limit),
fields,
});
if (args?.year_from) params.set('year', `${args.year_from}-`);
let data: any;
try {
data = await fetchJson(`${S2_BASE}/paper/search?${params}`, headers);
} catch (err: any) {
const msg = String(err.message || '');
if (msg.includes('429') || msg.toLowerCase().includes('too many')) {
return {
success: false,
error: [
'Semantic Scholar rate limit exceeded.',
'Free tier allows ~1 req/sec. To get a higher limit:',
' 1. Apply at https://www.semanticscholar.org/product/api#api-key-form',
' 2. Add your key to .smallclaw/config.json: { "semantic_scholar": { "api_key": "<key>" } }',
'Alternatively, try openalex_search which has no rate limits.',
].join('\n'),
};
}
return { success: false, error: `Semantic Scholar request failed: ${msg}` };
}
const papers: any[] = data?.data ?? [];
if (!papers.length) {
return { success: false, error: `No results found on Semantic Scholar for: "${query}"` };
}
const lines: string[] = [`Semantic Scholar results for: "${query}" (${data?.total ?? '?'} total)\n`];
for (let i = 0; i < papers.length; i++) {
const p = papers[i];
const title = p.title || '(no title)';
const year = p.year ?? '';
const authors = (p.authors || []).slice(0, 6).map((a: any) => a.name || '').filter(Boolean).join(', ')
+ ((p.authors || []).length > 6 ? ' et al.' : '');
const doi = p.externalIds?.DOI || '';
const pmid = p.externalIds?.PubMed || '';
const oaPdf = p.openAccessPdf?.url || '';
const abstract = String(p.abstract || '').slice(0, 400);
const cites = p.citationCount ?? 0;
const inflCites= p.influentialCitationCount ?? 0;
const tldr = p.tldr?.text ? `TL;DR: ${p.tldr.text}` : '';
lines.push(`[${i + 1}] ${title}`);
if (authors) lines.push(` Authors: ${authors}`);
lines.push(` Year: ${year}${cites ? ` | Citations: ${cites}` : ''}${inflCites ? ` | Influential: ${inflCites}` : ''}`);
if (doi) lines.push(` DOI: https://doi.org/${doi}`);
if (pmid) lines.push(` PMID: ${pmid}`);
if (oaPdf) lines.push(` Open Access PDF: ${oaPdf}`);
if (tldr) lines.push(` ${tldr}`);
else if (abstract) lines.push(` Abstract: ${abstract}${abstract.length >= 400 ? '...' : ''}`);
lines.push('');
}
return {
success: true,
stdout: lines.join('\n'),
data: { query, total: data?.total, papers: papers.map((p: any) => ({
title: p.title,
year: p.year,
doi: p.externalIds?.DOI,
pmid: p.externalIds?.PubMed,
citations: p.citationCount,
inf_citations: p.influentialCitationCount,
oa_pdf: p.openAccessPdf?.url,
authors: (p.authors || []).slice(0, 6).map((a: any) => a.name),
})) },
};
},
};
+14 -14
View File
@@ -20,6 +20,19 @@ function isPathInsideDir(base: string, target: string): boolean {
return rel !== '' && !rel.startsWith('..') && !path.isAbsolute(rel);
}
// Compiled once at module load — never recreated per execution
const DANGEROUS_COMMANDS: Array<[RegExp, string]> = [
[/rm\s+-rf\s+\//, 'rm -rf /'],
[/mkfs/, 'filesystem format'],
[/dd\s+if=/, 'disk write'],
[/>\s*\/dev\//, 'device write'],
[/\bsudo\b/, 'privilege escalation'],
[/\bsu\s/, 'user switch'],
[/chmod\s+777/, 'world-writable permission'],
[/\bcurl\b.*\|.*\bbash\b/, 'curl-pipe-bash'],
[/\bwget\b.*-O.*\s*-\s*\|/, 'wget-pipe'],
];
// ── Absolute-path detector ────────────────────────────────────────────────────
// Catches commands that contain absolute paths outside the workspace even when
// cwd is inside it — e.g. `type C:\Windows\System32\config\SAM`
@@ -80,20 +93,7 @@ export async function executeShell(args: ShellToolArgs): Promise<ToolResult> {
}
}
// Hardcoded dangerous command patterns
const dangerousCommands: Array<[RegExp, string]> = [
[/rm\s+-rf\s+\//, 'rm -rf /'],
[/mkfs/, 'filesystem format'],
[/dd\s+if=/, 'disk write'],
[/>\s*\/dev\//, 'device write'],
[/\bsudo\b/, 'privilege escalation'],
[/\bsu\s/, 'user switch'],
[/chmod\s+777/, 'world-writable permission'],
[/\bcurl\b.*\|.*\bbash\b/, 'curl-pipe-bash'],
[/\bwget\b.*-O.*\s*-\s*\|/, 'wget-pipe'],
];
for (const [pattern, label] of dangerousCommands) {
for (const [pattern, label] of DANGEROUS_COMMANDS) {
if (pattern.test(args.command)) {
log.warn('[shell] Blocked dangerous command:', label);
return {
+127
View File
@@ -0,0 +1,127 @@
import path from 'path';
import fs from 'fs';
import { getConfig } from '../config/config.js';
import { ToolResult } from '../types.js';
function getWorkspacePath(): string {
try {
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
} catch {
return path.join(process.cwd(), 'workspace');
}
}
function isPathInsideDir(base: string, target: string): boolean {
const resolvedBase = path.resolve(base);
const resolvedTarget = path.resolve(target);
if (resolvedBase === resolvedTarget) return true;
const rel = path.relative(resolvedBase, resolvedTarget);
return rel !== '' && !rel.startsWith('..') && !path.isAbsolute(rel);
}
function formatTable(columns: string[], rows: any[][]): string {
if (!rows.length) return '(no rows)';
const widths = columns.map((c, i) =>
Math.min(40, Math.max(c.length, ...rows.map(r => String(r[i] ?? '').length)))
);
const pad = (s: string, w: number) => String(s ?? '').padEnd(w).slice(0, w);
const sep = widths.map(w => '-'.repeat(w)).join('-+-');
const header = columns.map((c, i) => pad(c, widths[i])).join(' | ');
return [header, sep, ...rows.map(r => r.map((cell, i) => pad(String(cell ?? ''), widths[i])).join(' | '))].join('\n');
}
const WRITE_RE = /^\s*(INSERT|UPDATE|DELETE|CREATE|DROP|ALTER|REPLACE|TRUNCATE|ATTACH|DETACH)\s/i;
export const sqliteTool = {
name: 'sqlite_query',
description: 'Execute SQL queries on a SQLite database file inside the workspace. SELECT queries are allowed by default. Set write=true to run INSERT/UPDATE/DELETE/CREATE/DROP.',
schema: {
db_path: 'Path to the .db file (relative to workspace, or absolute)',
query: 'SQL query to execute',
write: 'Allow write queries: INSERT/UPDATE/DELETE/CREATE/DROP (default: false)',
max_rows: 'Maximum rows to return for SELECT (default: 100)',
},
jsonSchema: {
type: 'object',
properties: {
db_path: { type: 'string', description: 'Path to the SQLite database file' },
query: { type: 'string', description: 'SQL query' },
write: { type: 'boolean', description: 'Allow write operations (default: false)' },
max_rows: { type: 'number', description: 'Max rows returned for SELECT (default: 100)' },
},
required: ['db_path', 'query'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const dbPath = String(args?.db_path || '').trim();
const query = String(args?.query || '').trim();
if (!dbPath) return { success: false, error: 'db_path is required' };
if (!query) return { success: false, error: 'query is required' };
const workspacePath = getWorkspacePath();
const resolved = path.isAbsolute(dbPath) ? dbPath : path.resolve(workspacePath, dbPath);
if (!isPathInsideDir(workspacePath, resolved)) {
return { success: false, error: 'db_path must be inside the workspace directory' };
}
const allowWrite = Boolean(args?.write);
const isWrite = WRITE_RE.test(query);
if (isWrite && !allowWrite) {
return { success: false, error: 'Write query requires write=true. Set write=true to allow INSERT/UPDATE/DELETE/CREATE.' };
}
let Database: any;
try {
const mod = await import('better-sqlite3');
Database = mod.default ?? mod;
} catch (err: any) {
return { success: false, error: `better-sqlite3 unavailable: ${err.message}` };
}
let db: any;
try {
db = new Database(resolved, { readonly: !allowWrite });
} catch (err: any) {
return { success: false, error: `Cannot open database: ${err.message}` };
}
try {
const maxRows = Math.min(1_000, Math.max(1, Number(args?.max_rows ?? 100)));
const stmt = db.prepare(query);
if (isWrite || !query.trim().toUpperCase().startsWith('SELECT')) {
const info = stmt.run();
return {
success: true,
stdout: `OK\nRows affected: ${info.changes} | Last insert rowid: ${info.lastInsertRowid}`,
data: { changes: info.changes, lastInsertRowid: info.lastInsertRowid },
};
}
const rows: any[] = stmt.all().slice(0, maxRows);
if (!rows.length) return { success: true, stdout: '(no rows returned)', data: { rows: [] } };
const columns = Object.keys(rows[0]);
const table = formatTable(columns, rows.map((r: any) => columns.map(c => r[c])));
let total = rows.length;
try {
// Wrap query in subquery to get total row count
const countRow = db.prepare(`SELECT COUNT(*) AS _cnt FROM (${query})`).get();
total = Number(countRow?._cnt ?? rows.length);
} catch {}
const note = rows.length < total ? `\n\n[Showing ${rows.length} of ${total} rows. Use max_rows to increase limit.]` : '';
return {
success: true,
stdout: table + note,
data: { columns, row_count: rows.length, total, rows },
};
} catch (err: any) {
return { success: false, error: `Query error: ${err.message}` };
} finally {
try { db.close(); } catch {}
}
},
};
+32 -43
View File
@@ -4,6 +4,24 @@ import fs from 'fs';
import path from 'path';
type SearchResultItem = { title: string; url: string; snippet: string };
// Shared HTML → plain text stripper used by fetchCleanArticle and executeWebFetch.
// preserveStructure=true keeps paragraph breaks (\s{3,}→\n\n); false collapses all whitespace.
function stripHtml(html: string, preserveStructure = false): string {
const text = html
.replace(/<script[\s\S]*?<\/script>/gi, ' ')
.replace(/<style[\s\S]*?<\/style>/gi, ' ')
.replace(/<nav[\s\S]*?<\/nav>/gi, ' ')
.replace(/<footer[\s\S]*?<\/footer>/gi, ' ')
.replace(/<header[\s\S]*?<\/header>/gi, ' ')
.replace(/<!--[\s\S]*?-->/g, ' ')
.replace(/<[^>]+>/g, ' ')
.replace(/&nbsp;/g, ' ').replace(/&amp;/g, '&').replace(/&lt;/g, '<')
.replace(/&gt;/g, '>').replace(/&quot;/g, '"').replace(/&#39;/g, "'");
return preserveStructure
? text.replace(/\s{3,}/g, '\n\n').trim()
: text.replace(/\s+/g, ' ').trim();
}
type StructuredSource = { id: number; tier: 'A' | 'B' | 'C'; title: string; url: string; snippet: string; score: number };
type StructuredEvidence = { id: number; source_id: number; excerpt: string; score: number };
type StructuredFact = { id: number; claim: string; evidence_ids: number[]; source_ids: number[]; confidence: number };
@@ -252,20 +270,7 @@ async function fetchCleanArticle(url: string, maxChars = 5000): Promise<string>
if (!res.ok) throw new Error(`HTTP ${res.status}`);
const ct = String(res.headers.get('content-type') || '');
if (!/text|html|json/i.test(ct)) throw new Error(`Unsupported content-type: ${ct}`);
const html = await res.text();
const text = html
.replace(/<script[\s\S]*?<\/script>/gi, ' ')
.replace(/<style[\s\S]*?<\/style>/gi, ' ')
.replace(/<nav[\s\S]*?<\/nav>/gi, ' ')
.replace(/<footer[\s\S]*?<\/footer>/gi, ' ')
.replace(/<header[\s\S]*?<\/header>/gi, ' ')
.replace(/<!--[\s\S]*?-->/g, ' ')
.replace(/<[^>]+>/g, ' ')
.replace(/&nbsp;/g, ' ').replace(/&amp;/g, '&').replace(/&lt;/g, '<')
.replace(/&gt;/g, '>').replace(/&quot;/g, '"').replace(/&#39;/g, "'")
.replace(/\s+/g, ' ')
.trim();
return text.slice(0, maxChars);
return stripHtml(await res.text()).slice(0, maxChars);
}
function extractEvidenceSentences(query: string, text: string, max = 4): string[] {
@@ -459,30 +464,26 @@ function rankResults(query: string, results: SearchResultItem[]) {
}
// ── Load optional API keys from ~/.smallclaw/config.json ─────────────────────
function getSearchConfig(): {
preferred: 'tavily' | 'google' | 'brave' | 'ddg';
tavilyKey?: string;
googleKey?: string;
googleCx?: string;
braveKey?: string;
} {
// Cached with 5-minute TTL so config changes are picked up without restart.
type SearchConfig = { preferred: 'tavily' | 'google' | 'brave' | 'ddg'; tavilyKey?: string; googleKey?: string; googleCx?: string; braveKey?: string };
let _searchConfigCache: { value: SearchConfig; expiresAt: number } | null = null;
function getSearchConfig(): SearchConfig {
const now = Date.now();
if (_searchConfigCache && now < _searchConfigCache.expiresAt) return _searchConfigCache.value;
let value: SearchConfig = { preferred: 'ddg' };
try {
const projectCfg = path.join(process.cwd(), '.smallclaw', 'config.json');
const cfg = fs.existsSync(projectCfg) ? projectCfg : path.join(os.homedir(), '.smallclaw', 'config.json');
if (fs.existsSync(cfg)) {
const data = JSON.parse(fs.readFileSync(cfg, 'utf-8'));
const preferredRaw = String(data.search?.preferred_provider || 'ddg').toLowerCase();
const preferred = (['tavily', 'google', 'brave', 'ddg'].includes(preferredRaw) ? preferredRaw : 'ddg') as 'tavily' | 'google' | 'brave' | 'ddg';
return {
preferred,
tavilyKey: data.search?.tavily_api_key,
googleKey: data.search?.google_api_key,
googleCx: data.search?.google_cx,
braveKey: data.search?.brave_api_key,
};
const preferred = (['tavily', 'google', 'brave', 'ddg'].includes(preferredRaw) ? preferredRaw : 'ddg') as SearchConfig['preferred'];
value = { preferred, tavilyKey: data.search?.tavily_api_key, googleKey: data.search?.google_api_key, googleCx: data.search?.google_cx, braveKey: data.search?.brave_api_key };
}
} catch {}
return { preferred: 'ddg' };
_searchConfigCache = { value, expiresAt: now + 5 * 60_000 };
return value;
}
// ── Google Custom Search API ─────────────────────────────────────────────---
async function searchGoogle(query: string, limit: number, apiKey: string, cx: string): Promise<ToolResult> {
@@ -833,19 +834,7 @@ export async function executeWebFetch(args: { url: string; max_chars?: number })
return { success: false, error: `Non-text content-type: ${contentType}` };
}
const html = await res.text();
let text = html
.replace(/<script[\s\S]*?<\/script>/gi, '')
.replace(/<style[\s\S]*?<\/style>/gi, '')
.replace(/<nav[\s\S]*?<\/nav>/gi, '')
.replace(/<footer[\s\S]*?<\/footer>/gi, '')
.replace(/<header[\s\S]*?<\/header>/gi, '')
.replace(/<!--[\s\S]*?-->/g, '')
.replace(/<[^>]+>/g, ' ')
.replace(/&nbsp;/g, ' ').replace(/&amp;/g, '&').replace(/&lt;/g, '<')
.replace(/&gt;/g, '>').replace(/&quot;/g, '"').replace(/&#39;/g, "'")
.replace(/\s{3,}/g, '\n\n')
.trim();
let text = stripHtml(await res.text(), true);
if (text.length > maxChars) text = text.slice(0, maxChars) + '\n\n[...truncated]';