v2.7.1 — image edit improvements, search config, bug fixes
image_edit: - speech_bubble operation with Korean font, rounded rect, tail - sketch quality: CLAHE pre-processing for better line contrast - output filename with timestamp to prevent overwrites - anime/painting stylize timeout 30s→600s (AnimeGANv2 model download) - remove_bg PNG output fix server-v2.ts: - resolveToolImageContent: embed edited image as base64 in tool results so vision models can see the output (3 execution paths covered) - _activeModelName: fix vision support detection for Ollama-hosted models (kimi/gemini run via Ollama — provider='ollama' was masking vision capability) - Synthetic tool call ID generation to prevent Gemini function_response empty name error - image_edit path hint format changed to English to prevent Gemini hallucination - userRequestedImageEdit: expanded keyword list (풍선, 달아, 붙여 등) - image_read OCR failure returns success:true with visual description hint - TOOL_BLOCKS photo: image_edit workflow documentation updated web-ui/index.html: - renderFileDownloads: skip download buttons for image file extensions Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
+364
-35
@@ -338,26 +338,95 @@ function resolveImageContent(content: any, workspacePath: string): any {
|
||||
|
||||
// Build ContentPart[] with text + image_url parts
|
||||
const parts: any[] = [];
|
||||
// Add text content without the image markdown
|
||||
// Add text content without the image markdown, but inject workspace-relative paths
|
||||
// so the AI knows the exact file path for image_edit / python_eval.
|
||||
const textOnly = content.replace(imageRe, '').trim();
|
||||
if (textOnly) parts.push({ type: 'text', text: textOnly });
|
||||
const pathHints = images.map(img => `[IMAGE PATH for image_edit/python_eval — use EXACTLY this string: ${decodeURIComponent(img.path)}]`).join('\n');
|
||||
const textWithPaths = [textOnly, pathHints].filter(Boolean).join('\n');
|
||||
if (textWithPaths) parts.push({ type: 'text', text: textWithPaths });
|
||||
|
||||
for (const img of images) {
|
||||
const filePath = path.resolve(workspacePath, img.path);
|
||||
if (fs.existsSync(filePath)) {
|
||||
const buf = fs.readFileSync(filePath);
|
||||
const b64 = buf.toString('base64');
|
||||
const ext = path.extname(filePath).toLowerCase().replace('.', '');
|
||||
const mime = ext === 'jpg' ? 'jpeg' : ext === 'svg' ? 'svg+xml' : ext || 'png';
|
||||
parts.push({
|
||||
type: 'image_url',
|
||||
image_url: { url: `data:image/${mime};base64,${b64}` },
|
||||
});
|
||||
const stat = fs.statSync(filePath);
|
||||
let buf: Buffer | null = null;
|
||||
if (stat.size <= 200_000) {
|
||||
buf = fs.readFileSync(filePath);
|
||||
} else {
|
||||
// Resize large images to 512px thumbnail so the model can still see them
|
||||
// without blowing up the token budget.
|
||||
try {
|
||||
const { execSync } = require('child_process');
|
||||
const script = `from PIL import Image; import sys; img=Image.open(sys.argv[1]); img.thumbnail((512,512)); img.save(sys.stdout.buffer, 'JPEG', quality=75)`;
|
||||
buf = execSync(`python3 -c "${script}" "${filePath}"`, { maxBuffer: 5 * 1024 * 1024 });
|
||||
} catch {
|
||||
buf = null;
|
||||
}
|
||||
}
|
||||
if (buf && buf.length > 0) {
|
||||
const b64 = buf.toString('base64');
|
||||
parts.push({
|
||||
type: 'image_url',
|
||||
image_url: { url: `data:image/jpeg;base64,${b64}` },
|
||||
});
|
||||
} else {
|
||||
parts.push({ type: 'text', text: `[Image: ${img.path}]` });
|
||||
}
|
||||
}
|
||||
}
|
||||
return parts.length > 0 ? parts : content;
|
||||
}
|
||||
|
||||
// Inject workspace-relative paths into tool result text so the AI knows the file location.
|
||||
// For image_edit / python_eval results: also embed a base64 thumbnail so the AI can SEE the result.
|
||||
function resolveToolImageContent(content: string, workspacePath: string, embedVision: boolean): any {
|
||||
if (!content || typeof content !== 'string') return content;
|
||||
const imageRe = /!\[([^\]]*)\]\(\/api\/files\/([^)]+)\)/g;
|
||||
if (!imageRe.test(content)) return content;
|
||||
imageRe.lastIndex = 0;
|
||||
|
||||
if (!embedVision) {
|
||||
// Path-hint only: inject [편집 결과: path] before each image markdown
|
||||
return content.replace(/!\[([^\]]*)\]\(\/api\/files\/([^)]+)\)/g, (m, alt, p) => {
|
||||
try { p = decodeURIComponent(p); } catch {}
|
||||
return `[편집 결과 경로: ${p}]\n${m}`;
|
||||
});
|
||||
}
|
||||
|
||||
// Full vision: convert to ContentPart[] with base64 thumbnails
|
||||
const parts: any[] = [];
|
||||
const textWithHints = content.replace(/!\[([^\]]*)\]\(\/api\/files\/([^)]+)\)/g, (m, alt, p) => {
|
||||
try { p = decodeURIComponent(p); } catch {}
|
||||
return `[편집 결과 경로: ${p}]`;
|
||||
});
|
||||
if (textWithHints.trim()) parts.push({ type: 'text', text: textWithHints });
|
||||
|
||||
let match: RegExpExecArray | null;
|
||||
imageRe.lastIndex = 0;
|
||||
while ((match = imageRe.exec(content)) !== null) {
|
||||
const imgPath = (() => { try { return decodeURIComponent(match[2]); } catch { return match[2]; } })();
|
||||
const filePath = path.resolve(workspacePath, imgPath);
|
||||
if (!fs.existsSync(filePath)) continue;
|
||||
try {
|
||||
const stat = fs.statSync(filePath);
|
||||
let buf: Buffer | null = null;
|
||||
if (stat.size <= 300_000) {
|
||||
buf = fs.readFileSync(filePath);
|
||||
} else {
|
||||
const { execSync } = require('child_process');
|
||||
const script = `from PIL import Image; import sys; img=Image.open(sys.argv[1]); img.thumbnail((600,600)); img.save(sys.stdout.buffer,'JPEG',quality=80)`;
|
||||
buf = execSync(`python3 -c "${script}" "${filePath}"`, { maxBuffer: 5 * 1024 * 1024 });
|
||||
}
|
||||
if (buf && buf.length > 0) {
|
||||
const ext = path.extname(filePath).toLowerCase();
|
||||
const mime = ext === '.png' ? 'image/png' : ext === '.webp' ? 'image/webp' : 'image/jpeg';
|
||||
parts.push({ type: 'image_url', image_url: { url: `data:${mime};base64,${buf.toString('base64')}` } });
|
||||
}
|
||||
} catch {}
|
||||
}
|
||||
return parts.length > 0 ? parts : content;
|
||||
}
|
||||
|
||||
// Path confinement check — immune to case/trailing-slash/"../" traversal
|
||||
function isPathInsideDir(base: string, target: string): boolean {
|
||||
const resolvedBase = path.resolve(base);
|
||||
@@ -1012,7 +1081,7 @@ function detectToolCategories(text: string): Set<string> {
|
||||
// 'why' removed — matches casual questions; keep specific error/debug terms
|
||||
const DEBUG = ['error', 'failed', 'how does', 'architecture', 'debug', 'caused', 'broke', 'not working', 'explain how', 'whats wrong', "what's wrong"];
|
||||
const WORKFLOW = ['workflow', 'agent builder', 'architect_workflow', 'deploy workflow', 'workflow template', 'build workflow'];
|
||||
const PHOTO = ['사진', '이미지', 'photo', 'image', 'picture', 'search_images', 'find photo', 'find image', 'show photo', 'show image'];
|
||||
const PHOTO = ['사진', '이미지', 'photo', 'image', 'picture', 'search_images', 'find photo', 'find image', 'show photo', 'show image', '이미지 편집', '사진 편집', '이미지 수정', '말풍선', '워터마크', '배경 제거', '크롭', '리사이즈', '애니메이션 스타일', '유화', '수채화', '스케치', '만화체', '그림체', '그림으로', '스타일 변환', 'stylize', 'anime style', 'cartoon style'];
|
||||
if (WEB.some(k => lower.includes(k))) cats.add('web');
|
||||
if (BROWSER.some(k => lower.includes(k))) cats.add('browser');
|
||||
if (DESKTOP.some(k => lower.includes(k))) cats.add('desktop');
|
||||
@@ -1024,7 +1093,9 @@ function detectToolCategories(text: string): Set<string> {
|
||||
if (PPTX.some(k => lower.includes(k))) cats.add('pptx');
|
||||
if (PUBMED.some(k => lower.includes(k)) || /PMC\d+|PMID\s*\d+/i.test(text)) cats.add('pubmed');
|
||||
if (SCHOLAR.some(k => lower.includes(k))) cats.add('scholar');
|
||||
const PDF = ['pdf', '.pdf', 'pdf_read', 'pdf_extract', '피디에프', '논문 파일', 'pdf 파일', '이미지 추출', '도표 추출', '그림 추출', '사진 추출'];
|
||||
const PDF = ['pdf', '.pdf', 'pdf_read', 'pdf_extract', '피디에프', '논문 파일', 'pdf 파일', '이미지 추출', '도표 추출', '그림 추출', '사진 추출', '논문'];
|
||||
// Also add pdf when pptx + any paper/doc cue — guarantees pdf tools are hinted for PDF→PPTX
|
||||
if (PPTX.some(k => lower.includes(k)) && ['논문', 'paper', 'document', '보고서', '문서'].some(k => lower.includes(k))) cats.add('pdf');
|
||||
const IMAGE_OCR = ['ocr', 'image_read', '이미지 읽기', '이미지 텍스트', '스크린샷 텍스트', 'extract text from image', 'read image'];
|
||||
const PYTHON = ['python', 'python_eval', '파이썬', 'pandas', 'numpy', 'matplotlib', 'scipy', 'csv 분석', 'csv 처리', 'data analysis', '데이터 분석', '수식 계산', 'calculate', 'compute'];
|
||||
const SQLITE = ['sqlite', 'sqlite_query', '.db', 'database query', 'sql query', 'sql 쿼리', '데이터베이스', '쿼리'];
|
||||
@@ -1081,15 +1152,54 @@ const TOOL_BLOCKS: Record<string, string> = {
|
||||
- Add to existing deck: edit_presentation(path, spec).
|
||||
- DOWNLOAD LINK: use EXACT tool output. Never rewrite URLs.
|
||||
|
||||
PDF→PPTX WORKFLOW — follow exactly:
|
||||
PDF→PPTX WORKFLOW — MANDATORY. When source is a PDF, ALL steps below are REQUIRED before create_presentation. NEVER skip image extraction:
|
||||
1. pdf_read(path) → get text content.
|
||||
2. pdf_extract_images(path, mode:"figures") → get figure_* file paths.
|
||||
3. In slide specs, use image_url = EXACT returned path (e.g. "uploads/xxx-images/figure_p2_1.png"). Never strip folders, never use image_path.
|
||||
4. ONLY use figure_* files. NEVER use img-* files.`,
|
||||
2. pdf_extract_images(path, mode:"figures") → get figure_* file paths. ALWAYS run this. Then immediately run python_eval to filter:
|
||||
from PIL import Image; import os
|
||||
figs = []
|
||||
for f in sorted(os.listdir("<abs_img_dir>")):
|
||||
if not f.startswith("figure_"): continue
|
||||
try:
|
||||
w,h = Image.open(os.path.join("<abs_img_dir>",f)).size
|
||||
if w < 250 or h < 200: continue # too small: title headers, tiny elements
|
||||
figs.append(f)
|
||||
except: pass
|
||||
print(figs)
|
||||
Use only the returned filenames as figures in slides.
|
||||
3. pdf_extract_tables(path) → if tables found, embed as markdown in slide content. Skip if no tables.
|
||||
4. pdf_extract_images(path, mode:"images") → get img-* file paths, then immediately run python_eval to filter embedded photos:
|
||||
from PIL import Image; import os
|
||||
out = []
|
||||
for f in sorted(os.listdir("<abs_img_dir>")):
|
||||
if not f.startswith("img-"): continue
|
||||
try:
|
||||
w,h = Image.open(os.path.join("<abs_img_dir>",f)).size; r=w/h
|
||||
if (w>1200 and h>1200 and 0.60<r<0.85): continue # full-page scan
|
||||
if w<150 or h<150: continue # icon/logo
|
||||
out.append(f)
|
||||
except: pass
|
||||
print(out)
|
||||
Replace <abs_img_dir> with the absolute path of the images output folder. Use only the returned filenames as embedded photos in slides.
|
||||
5. DEDUP RULE: figure_* = charts/graphs/diagrams/photos only. Tables detected by pdf_extract_tables → use markdown, NOT figure_* image.
|
||||
6. In slide specs, use image_url = EXACT workspace-relative path. Never strip folders, never use image_path.
|
||||
7. ONLY use figure_* and filtered img-* files. Raw unfiltered img-* files are FORBIDDEN.`,
|
||||
|
||||
workflow: `WORKFLOW TOOLS: ALWAYS call search_workflow_templates(intent) FIRST before creating anything — reuse existing workflows. If no match: (1) architect_workflow(desc) → (2) verify_workflow_credentials(wf_id) if creds needed → wait for user → (3) test_workflow(wf_id) → (4) deploy_workflow(wf_id, name, ...). Never call architect_workflow if search returns a match. Always relay tool user_message fields verbatim to the user.`,
|
||||
|
||||
photo: `PHOTO SEARCH: search_images(query, count?) → searches Pexels + Unsplash and returns embeddable image URLs. Display results as markdown images in your response. Use when the user asks to find, show, or search for photos or images.`,
|
||||
photo: `PHOTO SEARCH: search_images(query, count?) → searches Pexels + Unsplash and returns embeddable image URLs. Display results as markdown images in your response. Use when the user asks to find, show, or search for photos or images.
|
||||
|
||||
IMAGE EDIT TOOL: image_edit(path, operation, ...params) → 이미지 편집 및 변환.
|
||||
CRITICAL: 이미지 편집/변환 요청에는 반드시 image_edit 툴을 사용. shell/python_eval로 직접 코드 작성하거나 웹사이트/브라우저 사용 금지.
|
||||
STRICT: 사용자가 편집을 명시적으로 요청한 경우에만 image_edit 호출. 사진을 업로드했다고 해서 자동으로 회전·보정·분석하지 말 것. 요청 없이 먼저 편집하는 것은 금지.
|
||||
operations: crop | resize | rotate | flip | grayscale | brightness | contrast | sharpen | thumbnail | watermark | speech_bubble | stylize | remove_bg | convert
|
||||
stylize: style 파라미터로 스타일 지정.
|
||||
- 신경망(고품질): "anime"(기본, 애니/그림체), "painting"(유화), "celeba"(만화체), "anime_v1"
|
||||
- 전통필터: "sketch"(연필스케치), "sketch_color"(컬러스케치), "cartoon"(만화), "watercolor"(수채화)
|
||||
사용 예: image_edit({path:"uploads/photo.jpg", operation:"stylize", style:"anime"})
|
||||
remove_bg: AI 배경 제거, PNG 출력
|
||||
speech_bubble: text, position("top-left"|"top-right"|"bottom-left"|"bottom-right"), bg_color, text_color, font_size(자동)
|
||||
워크플로우: (1) [IMAGE PATH for image_edit/python_eval — use EXACTLY this string: ...] 힌트에서 입력 경로 읽기 → (2) image_edit 호출 → (3) 결과 자동 표시 → 재편집 가능.
|
||||
출력: uploads/ 폴더에 자동 저장.`,
|
||||
|
||||
weather: `WEATHER TOOL: weather_search(location, type?, units?) → real-time weather via OpenWeather API. location: city name (e.g. "Seoul"), "City,CountryCode" (e.g. "London,GB"), or "lat,lon". type: "current" (default) or "forecast" (5-day). units: "metric" °C (default), "imperial" °F. RULES: ALWAYS use weather_search for any weather / forecast / temperature query — NEVER use web_search for weather. Examples: weather_search({location:"Incheon"}) for current, weather_search({location:"Seoul",type:"forecast"}) for 5-day forecast.`,
|
||||
|
||||
@@ -1097,11 +1207,11 @@ PDF→PPTX WORKFLOW — follow exactly:
|
||||
|
||||
email: `EMAIL TOOLS (멀티 계정 지원): 모든 도구에 account? 파라미터로 계정 ID 지정 가능. 미지정 시 모든 계정 순회. 사용자별 설정에 등록된 계정만 접근 가능. email_list(folder?,limit?,account?) → 받은편지함 목록 (uid·제목·발신자·날짜·읽음여부). email_read(uid,folder?,account?) → 메일 전문 조회. email_send(to,subject,body,cc?,account?) → 메일 발송. email_search(query?,from?,subject?,since?,before?,folder?,limit?,account?) → 메일 검색. email_delete(uid,folder?,account?) → 메일 삭제. RULES: 이메일 관련 요청에는 항상 email_* 도구 사용 — web_search 사용 금지. email_list 결과는 계정별로 그룹핑된 원본 포맷(📬/📭 마커, 번호, UID 포함)을 그대로 사용자에게 전달할 것 — 자의적 재포맷팅 금지.`,
|
||||
|
||||
pdf: `PDF TOOLS: pdf_read(path, page_from?, page_to?, max_chars?) → extract text from a PDF. pdf_extract_images(path, mode?, out_dir?, page_from?, page_to?, dpi?) → extract images/figures from a PDF. mode: "figures" (PREFERRED — OpenCV auto-crop figures/tables, best for academic PDFs), "images" (embedded rasters — raw, often full-page scans), "both" = images+figures (default). out_dir: save to specific folder (e.g. PPTX project folder). Returns workspace-relative paths. NOTE: "pages" mode no longer exists — use "figures" instead.`,
|
||||
pdf: `PDF TOOLS: pdf_read(path, page_from?, page_to?, max_chars?) → extract text from a PDF. pdf_extract_images(path, mode?, out_dir?, page_from?, page_to?, dpi?) → extract images/figures from a PDF. mode: "figures" (PREFERRED — OpenCV auto-crop figures/tables, best for academic PDFs), "images" (embedded rasters — raw, often full-page scans), "both" = images+figures (default). out_dir: save to specific folder (e.g. PPTX project folder). Returns workspace-relative paths. NOTE: "pages" mode no longer exists — use "figures" instead. pdf_extract_tables(path, format?, engine?, page_from?, page_to?) → extract tables as structured data. format: "markdown" (default) | "csv" | "json". engine: "auto" (PyMuPDF first, pdfplumber fallback) | "pymupdf" | "pdfplumber". For scanned PDFs use pdf_extract_images mode "figures" instead.`,
|
||||
|
||||
image_ocr: `IMAGE OCR TOOL: image_read(path, lang?) → extract text from an image file via OCR. Supported formats: PNG, JPG, WEBP, BMP, TIFF. lang: "eng" (English, default), "kor" (Korean), "kor+eng" (both). Returns extracted text and confidence score. Useful for screenshots, scanned documents, and diagrams containing text.`,
|
||||
|
||||
python: `PYTHON TOOL: python_eval(code, timeout?, packages?) → execute Python code in the workspace directory. Returns stdout/stderr. timeout default 15s, max 60s. packages: comma-separated pip packages to install if missing (e.g. "numpy,pandas"). Standard library and pre-installed packages available. Use for math, data analysis, CSV/JSON processing, charting (save chart to workspace file).`,
|
||||
python: `PYTHON TOOL: python_eval(code, timeout?, packages?) → execute Python code in the workspace directory. Returns stdout/stderr. timeout default 15s, max 60s. packages: comma-separated pip packages to install if missing (e.g. "numpy,pandas"). Standard library and pre-installed packages available. Use for math, data analysis, CSV/JSON processing, charting (save chart to workspace file). IMAGE EDITING: input path = exact string from [IMAGE PATH for image_edit/python_eval — use EXACTLY this string: ...] hint in the message. Save output to uploads/<name>.jpg. FONT: For Korean/CJK text in Pillow (speech bubbles, watermarks, etc.) use ImageFont.truetype("/usr/share/fonts/truetype/nanum/NanumGothicBold.ttf", size) — DejaVuSans does NOT support Korean.`,
|
||||
|
||||
sqlite: `SQLITE TOOL: sqlite_query(db_path, query, write?, max_rows?) → query a SQLite .db file in the workspace. SELECT queries allowed by default. Set write=true for INSERT/UPDATE/DELETE/CREATE/DROP. Returns formatted table. db_path is relative to workspace or absolute (must be inside workspace).`,
|
||||
};
|
||||
@@ -2139,6 +2249,38 @@ function buildTools() {
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
type: 'function',
|
||||
function: {
|
||||
name: 'image_edit',
|
||||
description: 'Edit or artistically transform an image. ALWAYS use this tool for any image editing, filtering, or style conversion — do NOT use shell/python_eval/browser for image tasks. Supports: crop, resize, rotate, flip, grayscale, brightness, contrast, sharpen, thumbnail, watermark, speech_bubble, stylize (anime/painting/sketch/cartoon/watercolor), remove_bg, convert.',
|
||||
parameters: {
|
||||
type: 'object',
|
||||
required: ['path', 'operation'],
|
||||
properties: {
|
||||
path: { type: 'string', description: 'Input image path (relative to workspace or absolute)' },
|
||||
operation: { type: 'string', enum: ['crop','resize','rotate','flip','grayscale','brightness','contrast','sharpen','thumbnail','watermark','speech_bubble','stylize','remove_bg','convert'], description: 'Operation to perform' },
|
||||
output: { type: 'string', description: 'Output path (optional; auto-generated if omitted)' },
|
||||
style: { type: 'string', description: '[stylize] "anime"(default,AnimeGANv2) | "painting"(유화) | "celeba"(만화체) | "sketch" | "sketch_color" | "cartoon" | "watercolor"' },
|
||||
x: { type: 'number' }, y: { type: 'number' },
|
||||
width: { type: 'number' }, height: { type: 'number' },
|
||||
keep_aspect: { type: 'boolean' },
|
||||
degrees: { type: 'number' },
|
||||
direction: { type: 'string', enum: ['horizontal','vertical'], description: '[flip] "horizontal"=좌우반전(left-right mirror), "vertical"=상하반전(upside-down flip)' },
|
||||
value: { type: 'number', description: '[brightness/contrast/sharpen] factor (1.0=no change)' },
|
||||
text: { type: 'string', description: '[watermark/speech_bubble] text' },
|
||||
position: { type: 'string', description: '[watermark] center|bottom-right|bottom-left|top-right|top-left / [speech_bubble] top-left|top-right|bottom-left|bottom-right' },
|
||||
opacity: { type: 'number', description: '[watermark] 0.0–1.0' },
|
||||
bg_color: { type: 'string', description: '[speech_bubble] bubble background color' },
|
||||
text_color: { type: 'string', description: '[speech_bubble] text color' },
|
||||
border_color: { type: 'string', description: '[speech_bubble] border color' },
|
||||
font_size: { type: 'number', description: '[speech_bubble] font size (auto if omitted)' },
|
||||
quality: { type: 'number', description: '[jpeg/webp] quality 1-100 (default 85)' },
|
||||
},
|
||||
additionalProperties: false,
|
||||
},
|
||||
},
|
||||
},
|
||||
{
|
||||
type: 'function',
|
||||
function: {
|
||||
@@ -2622,6 +2764,20 @@ function repairJson(input: string): string {
|
||||
return s;
|
||||
}
|
||||
|
||||
// Returns true if the user message contains an explicit image editing request
|
||||
function userRequestedImageEdit(message: string): boolean {
|
||||
const stripped = message.replace(/!\[[^\]]*\]\([^)]*\)/g, '').trim();
|
||||
if (!stripped) return false;
|
||||
const keywords = [
|
||||
'편집','수정','바꿔','변환','회전','반전','자르','크롭','리사이즈','스케치','수채화','만화','애니',
|
||||
'밝게','어둡게','흑백','워터마크','말풍선','풍선','배경제거','remove','crop','resize','rotate','flip',
|
||||
'sketch','watercolor','cartoon','anime','stylize','brighten','darken','grayscale','convert','edit',
|
||||
'그림체','유화','스타일','필터','보정','합성','투명','다시붙','달아','붙여',
|
||||
];
|
||||
const lower = stripped.toLowerCase();
|
||||
return keywords.some(k => lower.includes(k));
|
||||
}
|
||||
|
||||
async function executeTool(name: string, args: any, workspacePath: string, sessionId: string = 'default', sendSSE?: (type: string, data: any) => void, username?: string): Promise<ToolResult> {
|
||||
// Filename inference: if the model forgot to pass filename, use the last one
|
||||
const needsFilename = ['read_file', 'create_file', 'replace_lines', 'insert_after', 'delete_lines', 'find_replace', 'delete_file'];
|
||||
@@ -3650,7 +3806,9 @@ async function executeTool(name: string, args: any, workspacePath: string, sessi
|
||||
const tool = getToolRegistry().get(name);
|
||||
if (!tool) return { name, args, result: `Unknown tool: ${name}`, error: true };
|
||||
const tr = await tool.execute({ ...args, _workspace: workspacePath, _workspacePath: workspacePath });
|
||||
return { name, args, result: tr.stdout || tr.error || '', error: !tr.success };
|
||||
const _resultText = tr.stdout || tr.error || '';
|
||||
const _hasImageMd = /!\[[^\]]*\]\(\/api\/files\/[^)]+\)/.test(_resultText);
|
||||
return { name, args, result: _resultText, error: !tr.success, ...(_hasImageMd ? { isImage: true } : {}) };
|
||||
}
|
||||
}
|
||||
} catch (err: any) {
|
||||
@@ -3797,12 +3955,15 @@ function sanitizeFinalReply(
|
||||
// Strip markdown links to PPTX/PDF files via the wrong /api/files/uploads/... path.
|
||||
// PPTX files are saved in {workspace}/{project-slug}/, not in uploads/.
|
||||
// The tool result already contains the correct link — the AI must not construct a second one.
|
||||
result = result.replace(/\[([^\]]*)\]\(\/api\/files\/uploads\/[^\)]*\.pptx[^\)]*\)/gi, '');
|
||||
result = result.replace(/\[([^\]]*)\]\(\/api\/pptx\/preview\?path=uploads[^\)]*\)/gi, '');
|
||||
// Also strip bare /api/files/uploads/...pptx URLs (unlinked)
|
||||
result = result.replace(/\/api\/files\/uploads\/[^\s)]+\.pptx/gi, '');
|
||||
// Fix wrong absolute localhost URLs — model sometimes rewrites the tool-returned relative
|
||||
// path as http://localhost:PORT/api/files/... — strip the host prefix to restore relative form.
|
||||
// Strip ALL AI-generated PPTX download links from text responses.
|
||||
// The tool result already contains the correct link — any link the AI writes is wrong.
|
||||
// 1. Markdown links containing /api/files/ + .pptx (catches project-folder paths and mangled URLs)
|
||||
result = result.replace(/\[[^\]]*\]\([^)]*api[^)]*files[^)]*\.pptx[^)]*\)/gi, '');
|
||||
// 2. Bare /api/files/...pptx URLs (not inside markdown parens)
|
||||
result = result.replace(/\/api\/files\/[^\s)]*\.pptx[^\s)]*/gi, '');
|
||||
// 3. /api/pptx/preview links
|
||||
result = result.replace(/\[([^\]]*)\]\(\/api\/pptx\/preview[^\)]*\)/gi, '');
|
||||
// 4. Absolute localhost URLs — strip host prefix to restore relative form
|
||||
result = result.replace(/\(https?:\/\/(?:localhost|127\.0\.0\.1)(?::\d+)?(\/api\/files\/[^\)]+\.pptx[^\)]*)\)/gi, '($1)');
|
||||
|
||||
return result.trim();
|
||||
@@ -4200,6 +4361,10 @@ async function handleChat(
|
||||
|
||||
const rawCfgForPreempt = (getConfig().getConfig() as any);
|
||||
const primaryProvider = rawCfgForPreempt.llm?.provider || 'ollama';
|
||||
// For vision support detection, check the active model name (not just provider).
|
||||
// kimi/gemini/qwen run through Ollama so primaryProvider is always 'ollama'.
|
||||
const _activeModelName: string = rawCfgForPreempt.llm?.providers?.[primaryProvider]?.model
|
||||
|| rawCfgForPreempt.models?.primary || '';
|
||||
const preemptCfg: {
|
||||
enabled: boolean;
|
||||
stallThresholdMs: number;
|
||||
@@ -4477,6 +4642,10 @@ async function handleChat(
|
||||
const toolName = String(call.tool || '').trim();
|
||||
const toolArgs = call.args || {};
|
||||
sendSSE('tool_call', { action: toolName, args: toolArgs, stepNum: allToolResults.length + 1, synthetic: true, actor: 'secondary' });
|
||||
if (toolName === 'image_edit' && !userRequestedImageEdit(message)) {
|
||||
allToolResults.push({ name: toolName, args: toolArgs, result: '[BLOCKED] image_edit was called without an explicit user edit request. Do not edit images unless the user explicitly asks.', error: true });
|
||||
continue;
|
||||
}
|
||||
const toolResult = await executeTool(toolName, toolArgs, workspacePath, sessionId, undefined, username);
|
||||
allToolResults.push(toolResult);
|
||||
logToolCall(workspacePath, toolName, toolArgs, toolResult.result, toolResult.error);
|
||||
@@ -4485,16 +4654,27 @@ async function handleChat(
|
||||
const secondarySliceLen = (toolName === 'create_presentation' || toolName === 'edit_presentation') ? 4000 : 500;
|
||||
sendSSE('tool_result', { action: toolName, result: toolResult.result.slice(0, secondarySliceLen), error: toolResult.error, stepNum: allToolResults.length, synthetic: true, actor: 'secondary' });
|
||||
// PPTX tools: when successful, signal completion instead of pushing the goal reminder
|
||||
const _isImageTool3 = (toolName === 'image_edit' || toolName === 'image_read' || toolName === 'image_info') && !toolResult.error && (toolResult as any).isImage;
|
||||
const goalReminder = ((toolName === 'create_presentation' || toolName === 'edit_presentation')
|
||||
&& !toolResult.error)
|
||||
? '\n\n[TASK COMPLETE: Presentation created. The download link is already in the tool result above — do NOT write any /api/files/... link in your response (the user already sees the correct link). Confirm completion in 1-2 sentences only. STOP — no more tool calls.]'
|
||||
: _isImageTool3
|
||||
? '\n\n[IMAGE COMPLETE: The result image is already displayed in chat. Do NOT include any image markdown () or /api/files/ link in your reply — just describe what was done in 1-2 sentences.]'
|
||||
: `\n\n[GOAL REMINDER: Your task is still: "${message.slice(0, 120)}". Stay focused on this goal only.]`;
|
||||
const isBrowserTool = isBrowserToolName(toolName);
|
||||
const isDesktopTool = isDesktopToolName(toolName);
|
||||
const toolMessageContent = (multiAgentActive && (isBrowserTool || isDesktopTool))
|
||||
? (isBrowserTool ? buildBrowserAck(toolName, toolResult) : buildDesktopAck(toolName, toolResult))
|
||||
: toolResult.result;
|
||||
messages.push({ role: 'tool', tool_name: toolName, content: toolMessageContent + goalReminder });
|
||||
const _imagingTools3 = new Set(['image_edit', 'image_info', 'image_read', 'python_eval']);
|
||||
const _supportsVision3 = /kimi|openai|claude|gpt|gemini/i.test(String(_activeModelName || primaryProvider || ''));
|
||||
const _resolvedToolContent3 = _imagingTools3.has(toolName) && typeof toolMessageContent === 'string'
|
||||
? resolveToolImageContent(toolMessageContent, workspacePath, _supportsVision3)
|
||||
: toolMessageContent;
|
||||
const _toolContentFinal3 = Array.isArray(_resolvedToolContent3)
|
||||
? _resolvedToolContent3
|
||||
: ((_resolvedToolContent3 as string) + goalReminder);
|
||||
messages.push({ role: 'tool', name: toolName, tool_name: toolName, content: _toolContentFinal3 });
|
||||
orchestrationLog.push(
|
||||
toolResult.error
|
||||
? `✗ [secondary_patch] ${toolName}: ${toolResult.result.slice(0, 100)}`
|
||||
@@ -4573,7 +4753,8 @@ async function handleChat(
|
||||
{
|
||||
role: 'system',
|
||||
content: `${executionModeSystemBlock ? `${executionModeSystemBlock}\n\n` : ''}You are SmallClaw 🦞, a local AI assistant.\nCurrent date: ${dateStr}, ${timeStr}.\nNever search for or link SmallClaw repos unless the user is asking about SmallClaw itself.\nThis app runs on the user's own machine — browser/desktop automation requests are pre-authorized.\nKeep responses SHORT (1-2 sentences). Don't think out loud. Act and report. Greet naturally without tools.
|
||||
ANTI-HALLUCINATION: When a tool returns a result, report EXACTLY what the tool returned — never contradict or ignore tool output. If a tool says "(no rows)", say so. Never invent data, file contents, table names, or command output. If you don't know something, call a tool to find out or say you don't know.${callerContext ? '\n\n' + callerContext : ''}${browserStateCtx}${personalityCtx}${skillsManager.buildPromptContext(500)}${workflowCtx ? '\n\n' + workflowCtx : ''}`,
|
||||
ANTI-HALLUCINATION: When a tool returns a result, report EXACTLY what the tool returned — never contradict or ignore tool output. If a tool says "(no rows)", say so. Never invent data, file contents, table names, or command output. If you don't know something, call a tool to find out or say you don't know.
|
||||
IMAGE EDITING RULE: NEVER call image_edit (or any editing tool) when a user uploads a photo without explicitly requesting edits. Uploading a photo is NOT a request to edit it. Only call image_edit when the user's message explicitly asks for an edit (e.g. "수채화로 바꿔줘", "회전해줘"). Violating this rule is a critical error.${callerContext ? '\n\n' + callerContext : ''}${browserStateCtx}${personalityCtx}${skillsManager.buildPromptContext(500)}${workflowCtx ? '\n\n' + workflowCtx : ''}`,
|
||||
},
|
||||
];
|
||||
|
||||
@@ -5556,6 +5737,11 @@ RULES:
|
||||
sendSSE('info', { message: `Executing ${syntheticCalls.length} synthetic browser step(s)...` });
|
||||
|
||||
// Inject a synthetic assistant message so the message history is coherent
|
||||
for (let _i = 0; _i < syntheticCalls.length; _i++) {
|
||||
if (!(syntheticCalls[_i] as any).id) {
|
||||
(syntheticCalls[_i] as any).id = `call_${(syntheticCalls[_i].function?.name || 'tool').replace(/\W/g, '_')}_${_i}_${Date.now()}`;
|
||||
}
|
||||
}
|
||||
const syntheticAssistant = {
|
||||
role: 'assistant',
|
||||
content: null,
|
||||
@@ -5570,7 +5756,10 @@ RULES:
|
||||
const toolArgs = normalizeToolArgs(call.function?.arguments);
|
||||
console.log(`[v2] SYNTHETIC TOOL: ${toolName}(${JSON.stringify(toolArgs).slice(0, 100)})`);
|
||||
sendSSE('tool_call', { action: toolName, args: toolArgs, stepNum: allToolResults.length + 1, synthetic: true });
|
||||
|
||||
if (toolName === 'image_edit' && !userRequestedImageEdit(message)) {
|
||||
allToolResults.push({ name: toolName, args: toolArgs, result: '[BLOCKED] image_edit was called without an explicit user edit request. Do not edit images unless the user explicitly asks.', error: true });
|
||||
continue;
|
||||
}
|
||||
const toolResult = await executeTool(toolName, toolArgs, workspacePath, sessionId, undefined, username);
|
||||
allToolResults.push(toolResult);
|
||||
logToolCall(workspacePath, toolName, toolArgs, toolResult.result, toolResult.error);
|
||||
@@ -5585,9 +5774,12 @@ RULES:
|
||||
sendSSE('tool_result', { action: toolName, result: toolResult.result.slice(0, 300), error: toolResult.error, stepNum: allToolResults.length, synthetic: true });
|
||||
|
||||
// PPTX tools: when successful, signal completion instead of pushing the goal reminder
|
||||
const _isImageTool2 = (toolName === 'image_edit' || toolName === 'image_read' || toolName === 'image_info') && !toolResult.error && (toolResult as any).isImage;
|
||||
const goalReminder = ((toolName === 'create_presentation' || toolName === 'edit_presentation')
|
||||
&& !toolResult.error)
|
||||
? '\n\n[TASK COMPLETE: Presentation created. The download link is already in the tool result above — do NOT write any /api/files/... link in your response (the user already sees the correct link). Confirm completion in 1-2 sentences only. STOP — no more tool calls.]'
|
||||
: _isImageTool2
|
||||
? '\n\n[IMAGE COMPLETE: The result image is already displayed in chat. Do NOT include any image markdown () or /api/files/ link in your reply — just describe what was done in 1-2 sentences.]'
|
||||
: `\n\n[GOAL REMINDER: Your task is still: "${message.slice(0, 120)}". Stay focused on this goal only.]`;
|
||||
const isBrowserTool = isBrowserToolName(toolName);
|
||||
const isDesktopTool = isDesktopToolName(toolName);
|
||||
@@ -5595,11 +5787,20 @@ RULES:
|
||||
const toolMessageContent = (multiAgentActive && (isBrowserTool || isDesktopTool))
|
||||
? (isBrowserTool ? buildBrowserAck(toolName, toolResult) : buildDesktopAck(toolName, toolResult))
|
||||
: toolResult.result;
|
||||
const _imagingTools2 = new Set(['image_edit', 'image_info', 'image_read', 'python_eval']);
|
||||
const _supportsVision2 = /kimi|openai|claude|gpt|gemini/i.test(String(_activeModelName || primaryProvider || ''));
|
||||
const _resolvedContent2 = _imagingTools2.has(toolName) && typeof toolMessageContent === 'string'
|
||||
? resolveToolImageContent(toolMessageContent, workspacePath, _supportsVision2)
|
||||
: toolMessageContent;
|
||||
const _contentFinal2 = Array.isArray(_resolvedContent2)
|
||||
? _resolvedContent2
|
||||
: ((_resolvedContent2 as string) + goalReminder);
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: toolMessageContent + goalReminder,
|
||||
content: _contentFinal2,
|
||||
});
|
||||
|
||||
if (isBrowserTool && !toolResult.error) {
|
||||
@@ -6304,6 +6505,20 @@ RULES:
|
||||
};
|
||||
}
|
||||
|
||||
// Ensure every tool call has an id — Gemini rejects function_response with empty name
|
||||
// when tool_call_id is missing. Generate synthetic ids and patch both the response
|
||||
// message and the call objects so they stay in sync.
|
||||
for (let _i = 0; _i < toolCalls.length; _i++) {
|
||||
if (!(toolCalls[_i] as any).id) {
|
||||
const _synId = `call_${(toolCalls[_i].function?.name || 'tool').replace(/\W/g, '_')}_${_i}_${Date.now()}`;
|
||||
(toolCalls[_i] as any).id = _synId;
|
||||
if (Array.isArray((response as any).tool_calls)) {
|
||||
const _rc = (response as any).tool_calls[_i];
|
||||
if (_rc && !_rc.id) _rc.id = _synId;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
messages.push(response);
|
||||
|
||||
const batchCreatedFiles = new Set<string>();
|
||||
@@ -6347,7 +6562,7 @@ RULES:
|
||||
logToolCall(workspacePath, toolName, toolArgs, blockMsg, true);
|
||||
sendSSE('tool_call', { action: toolName, args: toolArgs, stepNum: allToolResults.length });
|
||||
sendSSE('tool_result', { action: toolName, result: blockMsg.slice(0, 300), error: true, stepNum: allToolResults.length });
|
||||
messages.push({ role: 'tool', tool_name: toolName, tool_call_id: toolCallId || undefined, content: blockMsg });
|
||||
messages.push({ role: 'tool', name: toolName, tool_name: toolName, tool_call_id: toolCallId || undefined, content: blockMsg });
|
||||
messages.push({ role: 'user', content: `Scroll blocked. You must call browser_fill or browser_click on a @ref from the snapshot above before scrolling. Stop planning and act now.` });
|
||||
continue;
|
||||
}
|
||||
@@ -6382,6 +6597,7 @@ RULES:
|
||||
});
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: blockMsg,
|
||||
@@ -6428,6 +6644,7 @@ RULES:
|
||||
});
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: blockMsg,
|
||||
@@ -6441,6 +6658,7 @@ RULES:
|
||||
console.log(`[v2] SKIP: duplicate create_file("${fn}") in same batch`);
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: `${fn} already created in this batch. Use replace_lines to edit.`,
|
||||
@@ -6459,6 +6677,7 @@ RULES:
|
||||
console.log(`[v2] SKIP: duplicate create_presentation("${title}") in same batch`);
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: `Presentation "${title}" already created in this batch. Use the previous result.`,
|
||||
@@ -6491,6 +6710,7 @@ RULES:
|
||||
});
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: blockedResult.result,
|
||||
@@ -6522,6 +6742,7 @@ RULES:
|
||||
});
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: blockedResult.result,
|
||||
@@ -6557,6 +6778,7 @@ RULES:
|
||||
});
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: replayedResult.result,
|
||||
@@ -6566,6 +6788,7 @@ RULES:
|
||||
console.log(`[v2] SKIP: duplicate tool call ${toolName}(${JSON.stringify(toolArgs).slice(0, 80)})`);
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: 'Already ran this exact call. Use the previous result and move on.',
|
||||
@@ -6609,6 +6832,7 @@ RULES:
|
||||
fileOpHadToolFailure = true;
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: 'FILE_OP v2: secondary planner produced no replacement calls.',
|
||||
@@ -6670,6 +6894,7 @@ RULES:
|
||||
const failText = 'FILE_OP v2: secondary patch planner returned no executable calls.';
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: failText,
|
||||
@@ -6731,6 +6956,7 @@ RULES:
|
||||
if (!subPrompt) {
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: 'Sub-agent spawn failed: no task_prompt or input provided.',
|
||||
@@ -6800,6 +7026,7 @@ RULES:
|
||||
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: ackMsg,
|
||||
@@ -6871,6 +7098,7 @@ RULES:
|
||||
}
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
name: toolName,
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: 'Multi-agent orchestration is not enabled.',
|
||||
@@ -6878,6 +7106,12 @@ RULES:
|
||||
continue;
|
||||
}
|
||||
|
||||
if (toolName === 'image_edit' && !userRequestedImageEdit(message)) {
|
||||
const _blocked: ToolResult = { name: toolName, args: toolArgs, result: '[BLOCKED] image_edit was called without an explicit user edit request. Do not edit images unless the user explicitly asks.', error: true };
|
||||
allToolResults.push(_blocked);
|
||||
sendSSE('tool_result', { action: toolName, result: _blocked.result, error: true, stepNum: allToolResults.length });
|
||||
continue;
|
||||
}
|
||||
const toolResult = await executeTool(toolName, toolArgs, workspacePath, sessionId, sendSSE, username);
|
||||
if (canReplayReadOnlyCall(toolName)) cachedReadOnlyToolResults.set(callKey, toolResult);
|
||||
allToolResults.push(toolResult);
|
||||
@@ -6902,9 +7136,12 @@ RULES:
|
||||
}
|
||||
}
|
||||
// PPTX tools: when successful, signal completion instead of pushing the goal reminder
|
||||
const _isImageTool = (toolName === 'image_edit' || toolName === 'image_read' || toolName === 'image_info') && !toolResult.error && (toolResult as any).isImage;
|
||||
const goalReminder = ((toolName === 'create_presentation' || toolName === 'edit_presentation')
|
||||
&& !toolResult.error)
|
||||
? '\n\n[TASK COMPLETE: Presentation created. The download link is already in the tool result above — do NOT write any /api/files/... link in your response (the user already sees the correct link). Confirm completion in 1-2 sentences only. STOP — no more tool calls.]'
|
||||
: _isImageTool
|
||||
? '\n\n[IMAGE COMPLETE: The result image is already displayed in chat. Do NOT include any image markdown () or /api/files/ link in your reply — just describe what was done in 1-2 sentences.]'
|
||||
: `\n\n[GOAL REMINDER: Your task is still: "${message.slice(0, 120)}". Stay focused on this goal only.]`;
|
||||
// When orchestrator is active, LLM never sees raw snapshot/browser data.
|
||||
// Full data still flows to advisor via getBrowserAdvisorPacket().
|
||||
@@ -6913,11 +7150,19 @@ RULES:
|
||||
const toolMessageContent = (multiAgentActive && (isBrowserTool || isDesktopTool))
|
||||
? (isBrowserTool ? buildBrowserAck(toolName, toolResult) : buildDesktopAck(toolName, toolResult))
|
||||
: toolResult.result;
|
||||
const _imagingTools = new Set(['image_edit', 'image_info', 'image_read', 'python_eval']);
|
||||
const _supportsVision = /kimi|openai|claude|gpt|gemini/i.test(String(_activeModelName || primaryProvider || ''));
|
||||
const _resolvedToolContent = _imagingTools.has(toolName) && typeof toolMessageContent === 'string'
|
||||
? resolveToolImageContent(toolMessageContent, workspacePath, _supportsVision)
|
||||
: toolMessageContent;
|
||||
const _toolContentFinal = Array.isArray(_resolvedToolContent)
|
||||
? _resolvedToolContent
|
||||
: ((_resolvedToolContent as string) + goalReminder);
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
tool_name: toolName,
|
||||
tool_call_id: toolCallId || undefined,
|
||||
content: toolMessageContent + goalReminder,
|
||||
content: _toolContentFinal,
|
||||
});
|
||||
|
||||
if (isBrowserTool && !toolResult.error) {
|
||||
@@ -8090,7 +8335,37 @@ app.get('/api/files/{*filePath}', (req: express.Request, res: express.Response)
|
||||
console.log('[files] found in attachments:', attMatch);
|
||||
filePath = attMatch;
|
||||
} else {
|
||||
// Fallback 3: basename in workspace root (covers files saved by tools outside uploads/)
|
||||
// Fallback 3: prefix match in same directory ("figure_p05" → "figure_p05_1.png")
|
||||
const stem = basename.replace(/\.[^.]+$/, '');
|
||||
const parentDir = path.dirname(resolved);
|
||||
let prefixMatch: string | null = null;
|
||||
if (stem.length > 3 && fs.existsSync(parentDir) && fs.statSync(parentDir).isDirectory()) {
|
||||
const hit = fs.readdirSync(parentDir).find(f =>
|
||||
f.startsWith(stem + '_') || f.startsWith(stem + '.')
|
||||
);
|
||||
if (hit) prefixMatch = path.join(parentDir, hit);
|
||||
}
|
||||
// Also scan uploads/ subdirs when path is fully truncated
|
||||
if (!prefixMatch && stem.length > 5) {
|
||||
for (const ws of [user?.workspace, globalWorkspace].filter(Boolean) as string[]) {
|
||||
const uploadsDir = path.join(ws, 'uploads');
|
||||
if (!fs.existsSync(uploadsDir)) continue;
|
||||
for (const sub of fs.readdirSync(uploadsDir)) {
|
||||
const subDir = path.join(uploadsDir, sub);
|
||||
if (!fs.existsSync(subDir) || !fs.statSync(subDir).isDirectory()) continue;
|
||||
const hit = fs.readdirSync(subDir).find(f =>
|
||||
f.startsWith(stem + '_') || f.startsWith(stem + '.')
|
||||
);
|
||||
if (hit) { prefixMatch = path.join(subDir, hit); break; }
|
||||
}
|
||||
if (prefixMatch) break;
|
||||
}
|
||||
}
|
||||
if (prefixMatch) {
|
||||
console.log('[files] prefix match found:', prefixMatch);
|
||||
filePath = prefixMatch;
|
||||
} else {
|
||||
// Fallback 4: basename in workspace root (covers files saved by tools outside uploads/)
|
||||
const userRootMatch = user?.workspace ? path.join(user.workspace, basename) : null;
|
||||
const globalRootMatch = path.join(globalWorkspace, basename);
|
||||
if (userRootMatch && fs.existsSync(userRootMatch) && fs.statSync(userRootMatch).isFile()) {
|
||||
@@ -8103,6 +8378,7 @@ app.get('/api/files/{*filePath}', (req: express.Request, res: express.Response)
|
||||
console.log('[files] not found:', resolved, 'or:', globalResolved);
|
||||
res.status(404).json({ error: 'File not found' }); return;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -8317,6 +8593,29 @@ app.post('/api/upload/image', (req: express.Request, res: express.Response) => {
|
||||
const finalName = resolveUploadName(uploadsDir, filename);
|
||||
const filePath = path.join(uploadsDir, finalName);
|
||||
fs.writeFileSync(filePath, fileData);
|
||||
|
||||
// Auto-correct EXIF orientation for JPEG/HEIC uploads
|
||||
// Only fix pure rotations (3=180°, 6=90°CW, 8=90°CCW) — skip mirror flags (2,4,5,7)
|
||||
// which are rare on phone photos and often incorrectly set
|
||||
const _uploadExt = path.extname(finalName).toLowerCase();
|
||||
if (['.jpg', '.jpeg', '.heic', '.heif'].includes(_uploadExt)) {
|
||||
try {
|
||||
const { spawnSync } = require('child_process');
|
||||
const _pyScript = [
|
||||
'from PIL import Image',
|
||||
'import sys',
|
||||
'img = Image.open(sys.argv[1])',
|
||||
'exif = img._getexif() or {}',
|
||||
'orientation = exif.get(274, 1)',
|
||||
'rotation_map = {3: Image.Transpose.ROTATE_180, 6: Image.Transpose.ROTATE_270, 8: Image.Transpose.ROTATE_90}',
|
||||
'if orientation in rotation_map:',
|
||||
' img = img.transpose(rotation_map[orientation])',
|
||||
' img.save(sys.argv[1], quality=95)',
|
||||
].join('\n');
|
||||
spawnSync('python3', ['-c', _pyScript, filePath], { timeout: 10_000 });
|
||||
} catch {}
|
||||
}
|
||||
|
||||
const relativePath = `uploads/${finalName}`;
|
||||
res.json({ success: true, path: relativePath, url: `/api/files/${relativePath}`, size: fileData.length, type: filetype });
|
||||
});
|
||||
@@ -11258,8 +11557,38 @@ wss.on('connection', (ws: WebSocket, req: http.IncomingMessage) => {
|
||||
ws.on('pong', () => { (ws as any).isAlive = true; });
|
||||
console.log(`[v2] WS connected${user ? ` (user: ${user.username})` : ''}`);
|
||||
|
||||
// Deliver pending boot greeting if it hasn't expired
|
||||
if (_pendingBootGreeting && Date.now() < _pendingBootGreeting.expiresAt) {
|
||||
// ── Per-user boot greeting ──
|
||||
if (user?.workspace) {
|
||||
(async () => {
|
||||
try {
|
||||
const userBootSessionId = `boot-${user.username}`;
|
||||
setWorkspace(userBootSessionId, user.workspace);
|
||||
clearHistory(userBootSessionId);
|
||||
const userSnapshot = buildBootStartupSnapshot(user.workspace);
|
||||
const bootResult = await runBootMd(user.workspace, async (message, sessionId, sendSSE) => {
|
||||
const bootContext = [
|
||||
'CONTEXT: Internal startup BOOT.md turn. All data has been pre-fetched and is in the snapshot below.',
|
||||
'Do NOT call any tools. Read the snapshot and write a 2-3 sentence startup summary IN KOREAN. MUST include: (1) the active_model name from the snapshot, (2) recent activity if any — not a generic greeting.',
|
||||
'[BOOT STARTUP SNAPSHOT - pre-fetched runtime data, no tools needed]',
|
||||
userSnapshot,
|
||||
'[/BOOT STARTUP SNAPSHOT]',
|
||||
].join('\n\n');
|
||||
const effectiveSessionId = sessionId || userBootSessionId;
|
||||
setWorkspace(effectiveSessionId, user.workspace);
|
||||
const result = await handleChat(message, effectiveSessionId, sendSSE, undefined, undefined, bootContext);
|
||||
return { text: result.text };
|
||||
});
|
||||
if (bootResult.status === 'ran' && bootResult.reply) {
|
||||
try {
|
||||
ws.send(JSON.stringify({ type: 'boot_greeting', text: bootResult.reply, sessionId: userBootSessionId }));
|
||||
} catch {}
|
||||
}
|
||||
} catch (err: any) {
|
||||
console.warn(`[boot-md] Per-user boot failed for ${user.username}:`, err?.message || err);
|
||||
}
|
||||
})();
|
||||
} else if (_pendingBootGreeting && Date.now() < _pendingBootGreeting.expiresAt) {
|
||||
// Anonymous / legacy user: deliver the global boot greeting
|
||||
try {
|
||||
ws.send(JSON.stringify({ type: 'boot_greeting', text: _pendingBootGreeting.text, sessionId: _pendingBootGreeting.sessionId }));
|
||||
} catch {}
|
||||
|
||||
@@ -279,7 +279,8 @@ export class TelegramChannel {
|
||||
}
|
||||
} catch (err: any) {
|
||||
if (err.name === 'AbortError') break;
|
||||
console.error('[Telegram] Poll error:', err.message);
|
||||
const detail = err.cause ? ` (${err.cause.message || err.cause})` : '';
|
||||
console.error('[Telegram] Poll error:', err.message + detail);
|
||||
// Wait before retrying on error
|
||||
await new Promise(r => setTimeout(r, 5000));
|
||||
}
|
||||
|
||||
@@ -269,6 +269,9 @@ export function shouldRunPreflight(userMessage: string, mode: PreflightMode): bo
|
||||
const lower = text.toLowerCase();
|
||||
// File upload + wait-for-instruction: always primary_direct, no routing needed
|
||||
if (text.includes('이 파일로 무엇을 도와드릴까요') || text.includes('작업을 직접 알려주기 전까지')) return false;
|
||||
// Pure image upload (no actual text question): skip preflight, let primary handle it directly
|
||||
const textWithoutImages = text.replace(/!\[([^\]]*)\]\([^)]+\)/g, '').replace(/\[([^\]]+)\]\([^)]+\)/g, '').trim();
|
||||
if (textWithoutImages.length < 5 && /\.(jpg|jpeg|png|gif|webp|bmp|tiff|tif)/i.test(lower)) return false;
|
||||
if (text.includes('\n')) return true;
|
||||
|
||||
return /\b(plan|spec|checklist|refactor|debug|fix|error|stack|search|web|browse|tool|edit|file|code|implement|oauth|api|endpoint|config|settings|migration)\b/i.test(lower);
|
||||
|
||||
@@ -26,7 +26,7 @@ export const audioTranscribeTool = {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = getConfig().getConfig()?.workspace?.path || process.cwd();
|
||||
const workspacePath = args?._workspacePath || args?._workspace || getConfig().getConfig()?.workspace?.path || process.cwd();
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) {
|
||||
|
||||
+51
-18
@@ -11,13 +11,16 @@ const execFileAsync = promisify(execFile);
|
||||
const PATCH_OUTPUT_MAX_CHARS = 8000;
|
||||
|
||||
// Helper function to check if path is allowed
|
||||
function resolveWorkspacePath(targetPath: string): string {
|
||||
const config = getConfig().getConfig();
|
||||
const workspace = config.workspace.path;
|
||||
function resolveWorkspacePath(targetPath: string, workspacePath?: string): string {
|
||||
const workspace = workspacePath || getConfig().getConfig().workspace.path;
|
||||
if (path.isAbsolute(targetPath)) return targetPath;
|
||||
return path.join(workspace, targetPath);
|
||||
}
|
||||
|
||||
function getSessionWorkspace(args: any): string {
|
||||
return args?._workspacePath || args?._workspace || getConfig().getConfig().workspace.path;
|
||||
}
|
||||
|
||||
function normalizePathForCompare(p: string): string {
|
||||
const resolved = path.resolve(String(p || ''));
|
||||
if (process.platform === 'win32') return resolved.toLowerCase();
|
||||
@@ -168,6 +171,8 @@ export interface ReadToolArgs {
|
||||
path: string;
|
||||
start_line?: number;
|
||||
num_lines?: number;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
type RetrievalMode = 'fast' | 'standard' | 'deep';
|
||||
@@ -203,7 +208,8 @@ function retrievalMaxLines(mode: RetrievalMode): number {
|
||||
|
||||
export async function executeRead(args: ReadToolArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const absPath = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const absPath = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(absPath);
|
||||
if (!pathCheck.allowed) {
|
||||
return {
|
||||
@@ -252,6 +258,8 @@ export async function executeRead(args: ReadToolArgs): Promise<ToolResult> {
|
||||
export interface WriteToolArgs {
|
||||
path: string;
|
||||
content: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
export async function executeWrite(args: WriteToolArgs): Promise<ToolResult> {
|
||||
@@ -268,7 +276,8 @@ export async function executeWrite(args: WriteToolArgs): Promise<ToolResult> {
|
||||
error: 'content must be a string'
|
||||
};
|
||||
}
|
||||
const absPath = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const absPath = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(absPath);
|
||||
if (!pathCheck.allowed) {
|
||||
return {
|
||||
@@ -302,11 +311,14 @@ export interface EditToolArgs {
|
||||
path: string;
|
||||
old_str: string;
|
||||
new_str: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
export async function executeEdit(args: EditToolArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const absPath = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const absPath = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(absPath);
|
||||
if (!pathCheck.allowed) {
|
||||
return {
|
||||
@@ -356,11 +368,14 @@ export async function executeEdit(args: EditToolArgs): Promise<ToolResult> {
|
||||
// LIST DIRECTORY TOOL
|
||||
export interface ListToolArgs {
|
||||
path: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
export async function executeList(args: ListToolArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const absPath = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const absPath = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(absPath);
|
||||
if (!pathCheck.allowed) {
|
||||
return {
|
||||
@@ -441,9 +456,10 @@ export const listTool = {
|
||||
// ── DELETE ────────────────────────────────────────────────────────────────────
|
||||
import { rmSync, existsSync } from 'fs';
|
||||
|
||||
async function executeDelete(args: { path: string; recursive?: boolean }): Promise<ToolResult> {
|
||||
async function executeDelete(args: { path: string; recursive?: boolean; _workspacePath?: string; _workspace?: string }): Promise<ToolResult> {
|
||||
if (!args.path?.trim()) return { success: false, error: 'path is required' };
|
||||
const absPath = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const absPath = resolveWorkspacePath(args.path, workspacePath);
|
||||
if (!existsSync(absPath)) return { success: false, error: `Path does not exist: ${absPath}` };
|
||||
try {
|
||||
rmSync(absPath, { recursive: args.recursive ?? false, force: true });
|
||||
@@ -467,11 +483,14 @@ export const deleteTool = {
|
||||
export interface RenameArgs {
|
||||
path: string;
|
||||
new_path: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
export async function executeRename(args: RenameArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const src = resolveWorkspacePath(args.path);
|
||||
const dest = resolveWorkspacePath(args.new_path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const src = resolveWorkspacePath(args.path, workspacePath);
|
||||
const dest = resolveWorkspacePath(args.new_path, workspacePath);
|
||||
const srcCheck = isPathAllowed(src);
|
||||
const destCheck = isPathAllowed(dest);
|
||||
if (!srcCheck.allowed) return { success: false, error: srcCheck.reason };
|
||||
@@ -503,11 +522,14 @@ export const renameTool = {
|
||||
export interface CopyArgs {
|
||||
path: string;
|
||||
dest: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
export async function executeCopy(args: CopyArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const src = resolveWorkspacePath(args.path);
|
||||
const dest = resolveWorkspacePath(args.dest);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const src = resolveWorkspacePath(args.path, workspacePath);
|
||||
const dest = resolveWorkspacePath(args.dest, workspacePath);
|
||||
const srcCheck = isPathAllowed(src);
|
||||
const destCheck = isPathAllowed(dest);
|
||||
if (!srcCheck.allowed) return { success: false, error: srcCheck.reason };
|
||||
@@ -534,10 +556,13 @@ export const copyTool = {
|
||||
export interface MkdirArgs {
|
||||
path: string;
|
||||
recursive?: boolean;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
export async function executeMkdir(args: MkdirArgs): Promise<ToolResult> {
|
||||
export async function executeMkdir(args: MkdirArgs & { _workspacePath?: string; _workspace?: string }): Promise<ToolResult> {
|
||||
try {
|
||||
const abs = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const abs = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(abs);
|
||||
if (!pathCheck.allowed) return { success: false, error: pathCheck.reason };
|
||||
await fs.mkdir(abs, { recursive: args.recursive ?? true });
|
||||
@@ -560,10 +585,13 @@ export const mkdirTool = {
|
||||
// ── STAT / INFO ──────────────────────────────────────────────────────────────
|
||||
export interface StatArgs {
|
||||
path: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
export async function executeStat(args: StatArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const abs = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const abs = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(abs);
|
||||
if (!pathCheck.allowed) return { success: false, error: pathCheck.reason };
|
||||
const st = await fs.stat(abs);
|
||||
@@ -586,10 +614,13 @@ export const statTool = {
|
||||
export interface AppendArgs {
|
||||
path: string;
|
||||
content: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
export async function executeAppend(args: AppendArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const abs = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const abs = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(abs);
|
||||
if (!pathCheck.allowed) return { success: false, error: pathCheck.reason };
|
||||
await fs.mkdir(path.dirname(abs), { recursive: true });
|
||||
@@ -614,6 +645,8 @@ export const appendTool = {
|
||||
export interface ApplyPatchArgs {
|
||||
patch: string;
|
||||
check?: boolean;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
export async function executeApplyPatch(args: ApplyPatchArgs): Promise<ToolResult> {
|
||||
@@ -626,7 +659,7 @@ export async function executeApplyPatch(args: ApplyPatchArgs): Promise<ToolResul
|
||||
const validation = validatePatchPaths(targetPaths);
|
||||
if (!validation.ok) return { success: false, error: validation.error };
|
||||
|
||||
const workspacePath = getConfig().getConfig().workspace.path;
|
||||
const workspacePath = args._workspacePath || args._workspace || getConfig().getConfig().workspace.path;
|
||||
const tempPatchPath = path.join(
|
||||
os.tmpdir(),
|
||||
`smallclaw-apply-${Date.now()}-${Math.random().toString(36).slice(2)}.patch`
|
||||
|
||||
+542
-5
@@ -4,7 +4,9 @@ import fs from 'fs';
|
||||
import { getConfig } from '../config/config.js';
|
||||
import { ToolResult } from '../types.js';
|
||||
|
||||
function getWorkspacePath(): string {
|
||||
function getWorkspacePath(args?: any): string {
|
||||
const sessionPath = args?._workspacePath || args?._workspace;
|
||||
if (sessionPath) return sessionPath;
|
||||
try {
|
||||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
} catch {
|
||||
@@ -12,7 +14,29 @@ function getWorkspacePath(): string {
|
||||
}
|
||||
}
|
||||
|
||||
const IMAGE_EXTS = new Set(['.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.tiff', '.tif']);
|
||||
const IMAGE_EXTS = new Set(['.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.tiff', '.tif', '.heic', '.heif', '.avif']);
|
||||
|
||||
function runPython(script: string, timeoutMs = 60_000): Promise<any> {
|
||||
return new Promise((resolve) => {
|
||||
const child = spawn('python3', ['-c', script], {
|
||||
timeout: timeoutMs,
|
||||
env: { ...process.env },
|
||||
});
|
||||
let out = '';
|
||||
child.stdout.on('data', (d: Buffer) => { out += d.toString(); });
|
||||
child.stderr.on('data', () => {});
|
||||
child.on('close', () => {
|
||||
try { resolve(JSON.parse(out || '{}')); } catch { resolve({ error: 'Failed to parse output', raw: out.slice(0, 500) }); }
|
||||
});
|
||||
child.on('error', (e: Error) => resolve({ error: e.message }));
|
||||
});
|
||||
}
|
||||
|
||||
function buildImageMarkdown(filePath: string, workspacePath: string): string {
|
||||
const relPath = path.relative(workspacePath, filePath).replace(/\\/g, '/');
|
||||
const baseName = path.basename(filePath);
|
||||
return ``;
|
||||
}
|
||||
|
||||
// Runs OCR in a child process so the tesseract worker doesn't block the main event loop.
|
||||
const OCR_SCRIPT = `
|
||||
@@ -65,7 +89,7 @@ export const imageReadTool = {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = getWorkspacePath();
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) {
|
||||
@@ -96,11 +120,14 @@ export const imageReadTool = {
|
||||
const lines: string[] = [
|
||||
`File: ${path.basename(resolved)}`,
|
||||
`Size: ${sizeMB} MB | Format: ${ext.slice(1).toUpperCase()} | Lang: ${lang}`,
|
||||
'',
|
||||
buildImageMarkdown(resolved, workspacePath),
|
||||
];
|
||||
|
||||
if (ocrResult.error) {
|
||||
lines.push(`OCR Error: ${ocrResult.error}`);
|
||||
return { success: false, error: lines.join('\n') };
|
||||
lines.push(`OCR unavailable: ${ocrResult.error}`);
|
||||
lines.push('(Image is attached above — describe it visually.)');
|
||||
return { success: true, stdout: lines.join('\n') };
|
||||
}
|
||||
|
||||
if (!ocrResult.text) {
|
||||
@@ -124,3 +151,513 @@ export const imageReadTool = {
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
export const imagePreviewTool = {
|
||||
name: 'image_preview',
|
||||
description: 'Display an inline preview of any image file in the chat. Works with uploaded photos, screenshots, generated images, or diagrams. Returns a markdown image that renders directly in the conversation.',
|
||||
schema: {
|
||||
path: 'Path to the image file — absolute or relative to workspace',
|
||||
width: 'Max display width in pixels (optional)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
path: { type: 'string', description: 'Path to the image file' },
|
||||
width: { type: 'number', description: 'Max display width in pixels (optional)' },
|
||||
},
|
||||
required: ['path'],
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) {
|
||||
return { success: false, error: `File not found: ${resolved}` };
|
||||
}
|
||||
const ext = path.extname(resolved).toLowerCase();
|
||||
if (!IMAGE_EXTS.has(ext)) {
|
||||
return { success: false, error: `Not an image. Supported: ${[...IMAGE_EXTS].join(', ')}` };
|
||||
}
|
||||
|
||||
const stat = fs.statSync(resolved);
|
||||
const sizeMB = (stat.size / 1024 / 1024).toFixed(2);
|
||||
const widthAttr = args?.width ? ` width="${args.width}"` : '';
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stdout: `**${path.basename(resolved)}** (${sizeMB} MB)${widthAttr}\n\n${buildImageMarkdown(resolved, workspacePath)}`,
|
||||
data: { path: resolved, size: stat.size, rel_path: path.relative(workspacePath, resolved).replace(/\\/g, '/') },
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_info — metadata (dimensions, format, EXIF)
|
||||
// ---------------------------------------------------------------------------
|
||||
const INFO_SCRIPT = (imgPath: string) => `
|
||||
import json, sys
|
||||
try:
|
||||
from PIL import Image, ExifTags
|
||||
import os
|
||||
img = Image.open(${JSON.stringify(imgPath)})
|
||||
exif_data = {}
|
||||
try:
|
||||
raw = img._getexif()
|
||||
if raw:
|
||||
exif_data = {ExifTags.TAGS.get(k, str(k)): str(v) for k, v in raw.items() if k in ExifTags.TAGS}
|
||||
except Exception:
|
||||
pass
|
||||
stat = os.stat(${JSON.stringify(imgPath)})
|
||||
print(json.dumps({
|
||||
"width": img.width, "height": img.height,
|
||||
"format": img.format or "unknown", "mode": img.mode,
|
||||
"size_bytes": stat.st_size,
|
||||
"exif": exif_data,
|
||||
}))
|
||||
except Exception as e:
|
||||
print(json.dumps({"error": str(e)}))
|
||||
`;
|
||||
|
||||
export const imageInfoTool = {
|
||||
name: 'image_info',
|
||||
description: 'Get image metadata: dimensions (width × height), format, color mode, file size, and EXIF data (camera, GPS, date, etc.).',
|
||||
schema: {
|
||||
path: 'Path to the image file — absolute or relative to workspace',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
path: { type: 'string', description: 'Path to the image file' },
|
||||
},
|
||||
required: ['path'],
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
|
||||
|
||||
const result = await runPython(INFO_SCRIPT(resolved));
|
||||
if (result.error) return { success: false, error: result.error };
|
||||
|
||||
const lines: string[] = [
|
||||
`File: ${path.basename(resolved)}`,
|
||||
`Dimensions: ${result.width} × ${result.height} px`,
|
||||
`Format: ${result.format} | Mode: ${result.mode}`,
|
||||
`Size: ${(result.size_bytes / 1024).toFixed(1)} KB`,
|
||||
];
|
||||
if (result.exif && Object.keys(result.exif).length > 0) {
|
||||
lines.push('', 'EXIF:');
|
||||
const relevant = ['Make', 'Model', 'DateTime', 'DateTimeOriginal', 'GPSInfo', 'Orientation', 'Software'];
|
||||
for (const key of relevant) {
|
||||
if (result.exif[key]) lines.push(` ${key}: ${result.exif[key]}`);
|
||||
}
|
||||
}
|
||||
return { success: true, stdout: lines.join('\n'), data: result };
|
||||
},
|
||||
};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_edit — crop / resize / rotate / flip / convert / brightness /
|
||||
// contrast / grayscale / watermark / thumbnail
|
||||
// ---------------------------------------------------------------------------
|
||||
const EDIT_SCRIPT = (params: Record<string, any>) => `
|
||||
import json, sys, os
|
||||
try:
|
||||
from PIL import Image, ImageEnhance, ImageDraw, ImageFont
|
||||
import pi_heif; pi_heif.register_heif_opener()
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
from PIL import Image, ImageEnhance, ImageDraw, ImageFont
|
||||
import os, json
|
||||
|
||||
src = ${JSON.stringify(params.src)}
|
||||
dst = ${JSON.stringify(params.dst)}
|
||||
op = ${JSON.stringify(params.operation)}
|
||||
|
||||
img = Image.open(src)
|
||||
orig_format = img.format or "PNG"
|
||||
|
||||
if op == "crop":
|
||||
x, y, w, h = int(${params.x ?? 0}), int(${params.y ?? 0}), int(${params.width ?? 0}), int(${params.height ?? 0})
|
||||
img = img.crop((x, y, x + w, y + h))
|
||||
|
||||
elif op == "resize":
|
||||
w, h = int(${params.width ?? 0}), int(${params.height ?? 0})
|
||||
keep = ${params.keep_aspect ? 'True' : 'False'}
|
||||
if keep and w and h:
|
||||
img.thumbnail((w, h), Image.LANCZOS)
|
||||
elif w and h:
|
||||
img = img.resize((w, h), Image.LANCZOS)
|
||||
elif w:
|
||||
ratio = w / img.width; img = img.resize((w, int(img.height * ratio)), Image.LANCZOS)
|
||||
elif h:
|
||||
ratio = h / img.height; img = img.resize((int(img.width * ratio), h), Image.LANCZOS)
|
||||
|
||||
elif op == "rotate":
|
||||
deg = float(${params.degrees ?? 0})
|
||||
img = img.rotate(-deg, expand=True)
|
||||
|
||||
elif op == "flip":
|
||||
direction = ${JSON.stringify(params.direction ?? 'horizontal')}
|
||||
img = img.transpose(Image.FLIP_LEFT_RIGHT if direction == "horizontal" else Image.FLIP_TOP_BOTTOM)
|
||||
|
||||
elif op == "grayscale":
|
||||
img = img.convert("L").convert("RGB")
|
||||
|
||||
elif op == "brightness":
|
||||
factor = float(${params.value ?? 1.0})
|
||||
img = ImageEnhance.Brightness(img).enhance(factor)
|
||||
|
||||
elif op == "contrast":
|
||||
factor = float(${params.value ?? 1.0})
|
||||
img = ImageEnhance.Contrast(img).enhance(factor)
|
||||
|
||||
elif op == "sharpen":
|
||||
factor = float(${params.value ?? 2.0})
|
||||
img = ImageEnhance.Sharpness(img).enhance(factor)
|
||||
|
||||
elif op == "thumbnail":
|
||||
w, h = int(${params.width ?? 256}), int(${params.height ?? 256})
|
||||
img.thumbnail((w, h), Image.LANCZOS)
|
||||
|
||||
elif op == "watermark":
|
||||
text = ${JSON.stringify(params.text ?? '')}
|
||||
pos_key = ${JSON.stringify(params.position ?? 'bottom-right')}
|
||||
opacity = int(float(${params.opacity ?? 0.5}) * 255)
|
||||
draw = ImageDraw.Draw(img, "RGBA")
|
||||
try:
|
||||
has_cjk = any('가' <= c <= '힣' or '一' <= c <= '鿿' for c in text)
|
||||
if has_cjk:
|
||||
import os as _os
|
||||
_kor_fonts = [
|
||||
"/usr/share/fonts/truetype/nanum/NanumGothicBold.ttf",
|
||||
"/usr/share/fonts/truetype/nanum/NanumGothic.ttf",
|
||||
"/usr/share/fonts/opentype/noto/NotoSansCJK-Regular.ttc",
|
||||
]
|
||||
_fpath = next((f for f in _kor_fonts if _os.path.exists(f)), None)
|
||||
font = ImageFont.truetype(_fpath, size=max(20, img.width // 30)) if _fpath else ImageFont.load_default()
|
||||
else:
|
||||
font = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf", size=max(20, img.width // 30))
|
||||
except Exception:
|
||||
font = ImageFont.load_default()
|
||||
bbox = draw.textbbox((0, 0), text, font=font)
|
||||
tw, th = bbox[2] - bbox[0], bbox[3] - bbox[1]
|
||||
margin = 20
|
||||
positions = {
|
||||
"center": ((img.width - tw) // 2, (img.height - th) // 2),
|
||||
"bottom-right": (img.width - tw - margin, img.height - th - margin),
|
||||
"bottom-left": (margin, img.height - th - margin),
|
||||
"top-right": (img.width - tw - margin, margin),
|
||||
"top-left": (margin, margin),
|
||||
}
|
||||
px, py = positions.get(pos_key, positions["bottom-right"])
|
||||
draw.rectangle([px - 4, py - 4, px + tw + 4, py + th + 4], fill=(0, 0, 0, opacity // 2))
|
||||
draw.text((px, py), text, font=font, fill=(255, 255, 255, opacity))
|
||||
|
||||
elif op == "speech_bubble":
|
||||
import os as _os
|
||||
_text = ${JSON.stringify(params.text ?? '')}
|
||||
_pos = ${JSON.stringify(params.position ?? 'top-left')}
|
||||
_bg = ${JSON.stringify(params.bg_color ?? 'white')}
|
||||
_fg = ${JSON.stringify(params.text_color ?? 'black')}
|
||||
_border = ${JSON.stringify(params.border_color ?? '#333333')}
|
||||
_fsize = int(${params.font_size ?? 0}) or max(24, min(int(img.width / 22), 160))
|
||||
_pad = max(12, _fsize // 2); _tail = max(20, _fsize // 1.5); _r = max(10, _fsize // 3); _mg = max(10, _fsize // 3)
|
||||
_kor_fonts = [
|
||||
"/usr/share/fonts/truetype/nanum/NanumGothicBold.ttf",
|
||||
"/usr/share/fonts/opentype/noto/NotoSansCJK-Regular.ttc",
|
||||
]
|
||||
_fp = next((f for f in _kor_fonts if _os.path.exists(f)), None)
|
||||
_font = ImageFont.truetype(_fp, _fsize) if _fp else ImageFont.load_default()
|
||||
_lines = _text.split("\\n")
|
||||
_lh = _fsize + 6
|
||||
_tmp_draw = ImageDraw.Draw(img)
|
||||
_bw = int(max(_tmp_draw.textlength(l, font=_font) for l in _lines)) + _pad * 2
|
||||
_bh = _lh * len(_lines) + _pad * 2
|
||||
_bx = (img.width - _bw - _mg) if "right" in _pos else _mg
|
||||
_by = (img.height - _bh - _tail - _mg) if "bottom" in _pos else (_tail + _mg)
|
||||
img = img.convert("RGBA")
|
||||
_ov = Image.new("RGBA", img.size, (0, 0, 0, 0))
|
||||
_d = ImageDraw.Draw(_ov)
|
||||
_d.rounded_rectangle([_bx, _by, _bx + _bw, _by + _bh],
|
||||
radius=_r, fill=_bg, outline=_border, width=2)
|
||||
_cx = _bx + _bw // 2
|
||||
if "bottom" in _pos:
|
||||
_ty = _by + _bh
|
||||
_pts = [(_cx - 14, _ty), (_cx + 14, _ty), (_cx, _ty + _tail)]
|
||||
else:
|
||||
_ty = _by
|
||||
_pts = [(_cx - 14, _ty), (_cx + 14, _ty), (_cx, _ty - _tail)]
|
||||
_d.polygon(_pts, fill=_bg, outline=_border)
|
||||
img = Image.alpha_composite(img, _ov)
|
||||
_d2 = ImageDraw.Draw(img)
|
||||
for _i, _line in enumerate(_lines):
|
||||
_lw = _d2.textlength(_line, font=_font)
|
||||
_d2.text((_bx + (_bw - int(_lw)) // 2, _by + _pad + _i * _lh),
|
||||
_line, font=_font, fill=_fg)
|
||||
img = img.convert("RGB")
|
||||
|
||||
elif op == "stylize":
|
||||
import cv2 as _cv2
|
||||
import numpy as _np
|
||||
_style = ${JSON.stringify(params.style ?? 'anime')}
|
||||
_arr = _np.array(img.convert("RGB"))
|
||||
_bgr = _cv2.cvtColor(_arr, _cv2.COLOR_RGB2BGR)
|
||||
if _style == "sketch":
|
||||
_gray = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
|
||||
# CLAHE로 소스 대비 강화 → 밝은 사진에서도 선이 진하게 나옴
|
||||
_clahe = _cv2.createCLAHE(clipLimit=2.5, tileGridSize=(8,8))
|
||||
_gray_e = _clahe.apply(_gray)
|
||||
_sk = _cv2.divide(_gray_e, _cv2.bitwise_not(_cv2.GaussianBlur(_cv2.bitwise_not(_gray_e),(15,15),0)), scale=256.0)
|
||||
_sk_f = _sk.astype(_np.float32) / 255.0
|
||||
_sk_f = _np.power(_sk_f, 3.0)
|
||||
_lo = float(_np.percentile(_sk_f, 3))
|
||||
_sk_f = _np.clip((_sk_f - _lo) / (1.0 - _lo + 1e-6), 0.0, 1.0)
|
||||
_sk = (_sk_f * 255).astype(_np.uint8)
|
||||
_result = _cv2.cvtColor(_cv2.cvtColor(_sk, _cv2.COLOR_GRAY2BGR), _cv2.COLOR_BGR2RGB)
|
||||
elif _style == "sketch_color":
|
||||
# overlay blend: vivid color + sketch lines
|
||||
_gray = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
|
||||
_sk = _cv2.divide(_gray, _cv2.bitwise_not(_cv2.GaussianBlur(_cv2.bitwise_not(_gray),(21,21),0)), scale=256.0)
|
||||
_sk = _np.clip(_sk.astype(_np.float32)*0.85, 0, 255).astype(_np.uint8)
|
||||
_hsv = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2HSV).astype(_np.float32)
|
||||
_hsv[:,:,1] = _np.clip(_hsv[:,:,1]*2.0, 0, 255)
|
||||
_vivid = _cv2.cvtColor(_hsv.astype(_np.uint8), _cv2.COLOR_HSV2BGR)
|
||||
_vivid = _cv2.bilateralFilter(_vivid, 9, 60, 60)
|
||||
_a = _sk.astype(_np.float32)/255.0
|
||||
_b = _vivid.astype(_np.float32)/255.0
|
||||
_ov = _np.where(_a[...,_np.newaxis] < 0.5, 2*_a[...,_np.newaxis]*_b, 1-2*(1-_a[...,_np.newaxis])*(1-_b))
|
||||
_ov = _np.clip(_ov*255, 0, 255).astype(_np.uint8)
|
||||
_blend = _cv2.addWeighted(_ov, 0.4, _bgr, 0.6, 0)
|
||||
_result = _cv2.cvtColor(_blend, _cv2.COLOR_BGR2RGB)
|
||||
elif _style == "cartoon":
|
||||
# bilateral×7 (stronger flatten) + high-threshold Canny (외곽선만)
|
||||
_blur = _bgr.copy()
|
||||
for _ in range(7): _blur = _cv2.bilateralFilter(_blur, 9, 75, 75)
|
||||
_hsvc = _cv2.cvtColor(_blur, _cv2.COLOR_BGR2HSV)
|
||||
_hsvc[:,:,1] = _np.clip(_hsvc[:,:,1].astype(_np.float32)*1.5, 0, 255).astype(_np.uint8)
|
||||
_blur = _cv2.cvtColor(_hsvc, _cv2.COLOR_HSV2BGR)
|
||||
_gray = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
|
||||
# 9×9 pre-blur + high threshold → textures/clothing patterns filtered out
|
||||
_edges = _cv2.Canny(_cv2.GaussianBlur(_gray,(9,9),0), 70, 180)
|
||||
_mask = _cv2.cvtColor(_cv2.bitwise_not(_edges), _cv2.COLOR_GRAY2BGR)
|
||||
_result = _cv2.cvtColor((_blur.astype(_np.float32)*(_mask.astype(_np.float32)/255.0)).astype(_np.uint8), _cv2.COLOR_BGR2RGB)
|
||||
elif _style == "watercolor":
|
||||
# bilateral×3 (faces preserved) + median + soft edge overlay
|
||||
_wc = _bgr.copy()
|
||||
for _ in range(3): _wc = _cv2.bilateralFilter(_wc, 9, 75, 75)
|
||||
_wc = _cv2.medianBlur(_wc, 5)
|
||||
_hsvw = _cv2.cvtColor(_wc, _cv2.COLOR_BGR2HSV).astype(_np.float32)
|
||||
_hsvw[:,:,1] = _np.clip(_hsvw[:,:,1]*1.5, 0, 255)
|
||||
_hsvw[:,:,2] = _np.clip(_hsvw[:,:,2]*1.08, 0, 255)
|
||||
_wc = _cv2.cvtColor(_hsvw.astype(_np.uint8), _cv2.COLOR_HSV2BGR)
|
||||
_g3 = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
|
||||
_e3 = _cv2.Canny(_cv2.GaussianBlur(_g3,(9,9),0), 30, 100)
|
||||
_e3 = _cv2.GaussianBlur(_e3,(13,13),0) # very soft edges
|
||||
_e3a = _np.stack([_e3.astype(_np.float32)/255.0*0.22]*3, axis=-1) # 22% opacity
|
||||
_result = _cv2.cvtColor(_np.clip(_wc.astype(_np.float32)*(1-_e3a),0,255).astype(_np.uint8), _cv2.COLOR_BGR2RGB)
|
||||
else: # anime / painting / default — AnimeGANv2
|
||||
import torch as _torch
|
||||
_preset_map = {"anime":"paprika","painting":"face_paint_512_v2","celeba":"celeba_distill","anime_v1":"face_paint_512_v1","face_paint":"face_paint_512_v2"}
|
||||
_preset = _preset_map.get(_style, "paprika")
|
||||
_gen = _torch.hub.load('bryandlee/animegan2-pytorch:main','generator',pretrained=_preset,trust_repo=True)
|
||||
_f2p = _torch.hub.load('bryandlee/animegan2-pytorch:main','face2paint',trust_repo=True)
|
||||
_gen.eval()
|
||||
# 비율 보존: 긴 변 기준 768, 정사각형 패딩 후 변환, 패딩 제거
|
||||
_orig_w, _orig_h = img.size
|
||||
_scale = 768 / max(_orig_w, _orig_h)
|
||||
_rw, _rh = int(_orig_w*_scale), int(_orig_h*_scale)
|
||||
_resized = img.convert("RGB").resize((_rw, _rh), Image.LANCZOS)
|
||||
_sq = max(_rw, _rh)
|
||||
_canvas = Image.new("RGB", (_sq, _sq), (255,255,255))
|
||||
_canvas.paste(_resized, ((_sq-_rw)//2, (_sq-_rh)//2))
|
||||
_out_sq = _f2p(_gen, _canvas, size=_sq)
|
||||
# 패딩 제거
|
||||
_px, _py = (_sq-_rw)//2, (_sq-_rh)//2
|
||||
img = _out_sq.crop((_px, _py, _px+_rw, _py+_rh))
|
||||
_result = None
|
||||
if _result is not None:
|
||||
img = Image.fromarray(_result)
|
||||
|
||||
elif op == "remove_bg":
|
||||
try:
|
||||
from rembg import remove as _rembg_remove
|
||||
except ImportError:
|
||||
print(json.dumps({"error": "rembg not installed. Run: pip install rembg[cpu]"})); sys.exit(0)
|
||||
_inp = img.convert("RGBA")
|
||||
img = _rembg_remove(_inp)
|
||||
# force PNG output (preserves transparency)
|
||||
if not dst.lower().endswith(".png"):
|
||||
dst = os.path.splitext(dst)[0] + ".png"
|
||||
|
||||
elif op == "convert":
|
||||
pass # format is applied at save time
|
||||
|
||||
else:
|
||||
print(json.dumps({"error": f"Unknown operation: {op}"})); sys.exit(0)
|
||||
|
||||
# determine save format
|
||||
ext = os.path.splitext(dst)[1].lower().lstrip(".")
|
||||
fmt_map = {"jpg": "JPEG", "jpeg": "JPEG", "png": "PNG", "webp": "WEBP",
|
||||
"gif": "GIF", "bmp": "BMP", "tiff": "TIFF", "tif": "TIFF"}
|
||||
save_fmt = fmt_map.get(ext, orig_format)
|
||||
quality = int(${params.quality ?? 85})
|
||||
|
||||
if save_fmt == "JPEG" and img.mode in ("RGBA", "LA", "P"):
|
||||
img = img.convert("RGB")
|
||||
|
||||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||||
save_args = {}
|
||||
if save_fmt in ("JPEG", "WEBP"):
|
||||
save_args["quality"] = quality
|
||||
img.save(dst, format=save_fmt, **save_args)
|
||||
|
||||
stat = os.stat(dst)
|
||||
print(json.dumps({"output": dst, "width": img.width, "height": img.height,
|
||||
"format": save_fmt, "size_bytes": stat.st_size}))
|
||||
except Exception as e:
|
||||
import traceback
|
||||
print(json.dumps({"error": str(e), "trace": traceback.format_exc()[-500:]}))
|
||||
`;
|
||||
|
||||
export const imageEditTool = {
|
||||
name: 'image_edit',
|
||||
description: [
|
||||
'Edit an image file. Supported operations:',
|
||||
' crop — cut a rectangular region (x, y, width, height)',
|
||||
' resize — scale to new dimensions (width, height, keep_aspect)',
|
||||
' rotate — rotate clockwise by degrees',
|
||||
' flip — mirror horizontally or vertically (direction: horizontal|vertical)',
|
||||
' grayscale — convert to black & white',
|
||||
' brightness — adjust brightness (value: 0.5 = half, 2.0 = double)',
|
||||
' contrast — adjust contrast (value: 0.5 = half, 2.0 = double)',
|
||||
' sharpen — sharpen edges (value: 1.0 = none, 3.0 = strong)',
|
||||
' thumbnail — resize to fit within a box (width, height)',
|
||||
' watermark — overlay text (text, position, opacity)',
|
||||
' speech_bubble — add a speech bubble with tail (text, position, bg_color, text_color, border_color, font_size)',
|
||||
' stylize — convert to artistic style. Neural: anime(default)|painting|celeba|anime_v1 (AnimeGANv2). Classic: sketch|sketch_color|cartoon|watercolor',
|
||||
' remove_bg — remove background using AI (rembg); output is PNG with transparency',
|
||||
' convert — change file format (output path determines format)',
|
||||
'Returns the output file path and new dimensions.',
|
||||
].join('\n'),
|
||||
schema: {
|
||||
path: 'Input image file path (absolute or relative to workspace)',
|
||||
operation: 'Operation name: crop | resize | rotate | flip | grayscale | brightness | contrast | sharpen | thumbnail | watermark | speech_bubble | stylize | remove_bg | convert',
|
||||
output: 'Output file path (optional; defaults to <input>_<op>.<ext>)',
|
||||
x: '[crop] Left edge in pixels',
|
||||
y: '[crop] Top edge in pixels',
|
||||
width: '[crop/resize/thumbnail] Width in pixels',
|
||||
height: '[crop/resize/thumbnail] Height in pixels',
|
||||
keep_aspect: '[resize] Preserve aspect ratio (true/false, default false)',
|
||||
degrees: '[rotate] Clockwise rotation degrees (e.g. 90, 180, -90)',
|
||||
direction: '[flip] "horizontal"=좌우반전(left-right mirror) | "vertical"=상하반전(upside-down)',
|
||||
value: '[brightness/contrast/sharpen] Adjustment factor (1.0 = no change)',
|
||||
text: '[watermark/speech_bubble] Text string to overlay',
|
||||
position: '[watermark] "center"|"bottom-right"|"bottom-left"|"top-right"|"top-left" / [speech_bubble] "top-left"|"top-right"|"bottom-left"|"bottom-right"',
|
||||
opacity: '[watermark] Text opacity 0.0–1.0 (default 0.5)',
|
||||
bg_color: '[speech_bubble] Bubble background color (default "white")',
|
||||
text_color: '[speech_bubble] Text color (default "black")',
|
||||
border_color: '[speech_bubble] Border color (default "#333333")',
|
||||
font_size: '[speech_bubble] Font size in pixels (auto if omitted: image_width/22, clamped 24–160)',
|
||||
style: '[stylize] "anime"(default,paprika — 단체/풍경에 좋음) | "painting"(face_paint_v2 — 개인 인물 일러스트) | "celeba"(만화체) | "sketch" | "sketch_color" | "cartoon" | "watercolor"',
|
||||
quality: '[jpeg/webp output] Quality 1–100 (default 85)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
path: { type: 'string' },
|
||||
operation: { type: 'string', enum: ['crop','resize','rotate','flip','grayscale','brightness','contrast','sharpen','thumbnail','watermark','speech_bubble','stylize','remove_bg','convert'] },
|
||||
output: { type: 'string' },
|
||||
x: { type: 'number' },
|
||||
y: { type: 'number' },
|
||||
width: { type: 'number' },
|
||||
height: { type: 'number' },
|
||||
keep_aspect: { type: 'boolean' },
|
||||
degrees: { type: 'number' },
|
||||
direction: { type: 'string', enum: ['horizontal', 'vertical'] },
|
||||
value: { type: 'number' },
|
||||
text: { type: 'string' },
|
||||
position: { type: 'string' },
|
||||
opacity: { type: 'number' },
|
||||
bg_color: { type: 'string' },
|
||||
text_color: { type: 'string' },
|
||||
border_color: { type: 'string' },
|
||||
font_size: { type: 'number' },
|
||||
quality: { type: 'number' },
|
||||
},
|
||||
required: ['path', 'operation'],
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
const operation = String(args?.operation || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
if (!operation) return { success: false, error: 'operation is required' };
|
||||
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
|
||||
|
||||
// determine output path
|
||||
let outPath = String(args?.output || '').trim();
|
||||
if (!outPath) {
|
||||
const ext = args?.format
|
||||
? `.${args.format}`
|
||||
: (operation === 'convert' ? '.png' : path.extname(resolved));
|
||||
const base = path.basename(resolved, path.extname(resolved));
|
||||
outPath = path.join(path.dirname(resolved), `${base}_${operation}_${Date.now()}${ext}`);
|
||||
} else if (!path.isAbsolute(outPath)) {
|
||||
outPath = path.resolve(workspacePath, outPath);
|
||||
}
|
||||
|
||||
const params: Record<string, any> = {
|
||||
src: resolved,
|
||||
dst: outPath,
|
||||
operation,
|
||||
x: args?.x ?? 0,
|
||||
y: args?.y ?? 0,
|
||||
width: args?.width ?? 0,
|
||||
height: args?.height ?? 0,
|
||||
keep_aspect: args?.keep_aspect ?? false,
|
||||
degrees: args?.degrees ?? 0,
|
||||
direction: args?.direction ?? 'horizontal',
|
||||
value: args?.value ?? 1.0,
|
||||
text: args?.text ?? '',
|
||||
position: args?.position ?? 'bottom-right',
|
||||
opacity: args?.opacity ?? 0.5,
|
||||
quality: args?.quality ?? 85,
|
||||
style: args?.style ?? 'painting',
|
||||
};
|
||||
|
||||
// Neural stylize (anime/painting/celeba) downloads PyTorch models on first run (~5 min).
|
||||
// Classic filter styles (sketch/cartoon/watercolor) finish in <5s.
|
||||
const neuralStyles = new Set(['anime', 'painting', 'celeba', 'anime_v1', 'face_paint']);
|
||||
const timeoutMs = (operation === 'stylize' && neuralStyles.has(params.style)) ? 600_000 : 60_000;
|
||||
const result = await runPython(EDIT_SCRIPT(params), timeoutMs);
|
||||
if (result.error) return { success: false, error: result.error };
|
||||
|
||||
const relOut = path.relative(workspacePath, result.output).replace(/\\/g, '/');
|
||||
const sizeMB = (result.size_bytes / 1024 / 1024).toFixed(2);
|
||||
const preview = buildImageMarkdown(result.output, workspacePath);
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stdout: [
|
||||
`Operation: ${operation}`,
|
||||
`Output: ${result.output}`,
|
||||
`Dimensions: ${result.width} × ${result.height} px | Format: ${result.format} | Size: ${sizeMB} MB`,
|
||||
'',
|
||||
preview,
|
||||
].join('\n'),
|
||||
data: { ...result, rel_path: relOut },
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
@@ -0,0 +1,639 @@
|
||||
import { execFile } from 'child_process';
|
||||
import { promisify } from 'util';
|
||||
import path from 'path';
|
||||
import fs from 'fs';
|
||||
import { getConfig } from '../config/config.js';
|
||||
import { ToolResult } from '../types.js';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
function getWorkspacePath(): string {
|
||||
try {
|
||||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
} catch {
|
||||
return path.join(process.cwd(), 'workspace');
|
||||
}
|
||||
}
|
||||
|
||||
// CCITT G4 (fax-compressed) images produced by pdfimages cannot be opened
|
||||
// directly by most viewers. Wrap them in a minimal TIFF header so Pillow can
|
||||
// decode and re-save as PNG. Returns the number of files converted.
|
||||
//
|
||||
// Two known issues with naive conversion:
|
||||
// 1. PIL ignores PhotometricInterpretation=0 (WhiteIsZero) for CCITT T.6,
|
||||
// treating bit-0 as black → image is inverted. Fix: invert after open.
|
||||
// 2. Height cannot be derived from compressed data size alone (sparse pages
|
||||
// compress to near-zero bytes). Fix: use pdfimages -list to get real dims.
|
||||
const CCITT_CONVERT_PY = `
|
||||
import os, sys, struct, io, subprocess, re
|
||||
from PIL import Image, ImageOps
|
||||
|
||||
outdir = sys.argv[1]
|
||||
pdf_path = sys.argv[2] if len(sys.argv) > 2 else ''
|
||||
|
||||
# Get real image dimensions from pdfimages -list
|
||||
dim_map = {} # index -> (width, height)
|
||||
if pdf_path and os.path.exists(pdf_path):
|
||||
try:
|
||||
out = subprocess.check_output(['pdfimages', '-list', pdf_path],
|
||||
stderr=subprocess.DEVNULL, timeout=30).decode()
|
||||
for line in out.splitlines():
|
||||
m = re.match(r'\\s*(\\d+)\\s+\\S+\\s+\\S+\\s+(\\d+)\\s+(\\d+)', line)
|
||||
if m:
|
||||
idx, w, h = int(m.group(1)), int(m.group(2)), int(m.group(3))
|
||||
dim_map[idx] = (w, h)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
converted = 0
|
||||
for f in sorted(os.listdir(outdir)):
|
||||
if not f.endswith('.ccitt'):
|
||||
continue
|
||||
base = f[:-6]
|
||||
params_path = os.path.join(outdir, base + '.params')
|
||||
ccitt_path = os.path.join(outdir, f)
|
||||
png_path = os.path.join(outdir, base + '.png')
|
||||
|
||||
# Parse width from params file
|
||||
params = open(params_path).read().strip().split() if os.path.exists(params_path) else []
|
||||
width = None
|
||||
for i, p in enumerate(params):
|
||||
if p == '-X' and i + 1 < len(params):
|
||||
width = int(params[i + 1])
|
||||
width = width or 2000
|
||||
|
||||
# Get accurate height from pdfimages -list; fall back to estimation
|
||||
idx_match = re.search(r'-(\\d+)\\.ccitt$', f)
|
||||
idx = (int(idx_match.group(1)) + 1) if idx_match else -1 # pdfimages -list is 1-based
|
||||
if idx in dim_map:
|
||||
height = dim_map[idx][1]
|
||||
else:
|
||||
data_size = os.path.getsize(ccitt_path)
|
||||
height = max(100, (data_size * 8 // width) + 100)
|
||||
|
||||
try:
|
||||
data = open(ccitt_path, 'rb').read()
|
||||
strip_offset = 8 + 2 + 12 * 8 + 4
|
||||
def ifd_entry(tag, typ, count, value):
|
||||
return struct.pack('<HHII', tag, typ, count, value)
|
||||
entries = b''.join([
|
||||
ifd_entry(256, 4, 1, width),
|
||||
ifd_entry(257, 4, 1, height),
|
||||
ifd_entry(258, 3, 1, 1),
|
||||
ifd_entry(259, 3, 1, 4), # CCITT T.6
|
||||
ifd_entry(262, 3, 1, 0), # WhiteIsZero
|
||||
ifd_entry(278, 4, 1, height),
|
||||
ifd_entry(279, 4, 1, len(data)),
|
||||
ifd_entry(273, 4, 1, strip_offset),
|
||||
])
|
||||
tiff = (b'II' + struct.pack('<H', 42) + struct.pack('<I', 8)
|
||||
+ struct.pack('<H', 8) + entries + struct.pack('<I', 0) + data)
|
||||
img = Image.open(io.BytesIO(tiff))
|
||||
# PIL ignores WhiteIsZero for CCITT → invert to get correct polarity
|
||||
img = ImageOps.invert(img.convert('L'))
|
||||
img.save(png_path, 'PNG')
|
||||
os.remove(ccitt_path)
|
||||
if os.path.exists(params_path):
|
||||
os.remove(params_path)
|
||||
converted += 1
|
||||
print(f'ok:{base}.png', flush=True)
|
||||
except Exception as e:
|
||||
print(f'err:{base}:{e}', flush=True)
|
||||
print(f'done:{converted}', flush=True)
|
||||
`;
|
||||
|
||||
const FIGURES_DETECT_PY = `
|
||||
import sys, os
|
||||
import numpy as np
|
||||
import cv2
|
||||
|
||||
def detect_figures(img_path):
|
||||
img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)
|
||||
if img is None:
|
||||
return []
|
||||
h, w = img.shape
|
||||
_, binary = cv2.threshold(img, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)
|
||||
|
||||
# ── Step 1: detect horizontal lines ──
|
||||
horiz_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (w//8, 1))
|
||||
horiz_lines = cv2.morphologyEx(binary, cv2.MORPH_OPEN, horiz_kernel)
|
||||
dilate_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (40, 10))
|
||||
dilated = cv2.dilate(horiz_lines, dilate_kernel, iterations=1)
|
||||
contours, _ = cv2.findContours(dilated, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
|
||||
raw = sorted([cv2.boundingRect(c) for c in contours], key=lambda r: r[1])
|
||||
raw = [(x, y, x+bw, y+bh) for x, y, bw, bh in raw if bw > w*0.15]
|
||||
groups = []
|
||||
used = [False] * len(raw)
|
||||
for i, r1 in enumerate(raw):
|
||||
if used[i]: continue
|
||||
used[i] = True
|
||||
x1, y1, x2, y2 = r1
|
||||
for j in range(i+1, len(raw)):
|
||||
if used[j]: continue
|
||||
rx1, ry1, rx2, ry2 = raw[j]
|
||||
if ry1 - y2 > 300: continue
|
||||
overlap = max(0, min(x2, rx2) - max(x1, rx1))
|
||||
min_w = min(x2-x1, rx2-rx1)
|
||||
x_gap = max(0, max(x1, rx1) - min(x2, rx2))
|
||||
if min_w > 0 and (overlap / min_w >= 0.5 or x_gap <= 80):
|
||||
used[j] = True
|
||||
x1 = min(x1, rx1); y1 = min(y1, ry1)
|
||||
x2 = max(x2, rx2); y2 = max(y2, ry2)
|
||||
groups.append((x1, y1, x2, y2))
|
||||
|
||||
# ── Step 2: expand vertically and validate ──
|
||||
results = []
|
||||
for gx1_raw, gy1, gx2_raw, gy2 in groups:
|
||||
# Re-compute x bounds using median of per-row line extents to avoid
|
||||
# full-width separator lines bleeding into adjacent text columns.
|
||||
row_x1s, row_x2s = [], []
|
||||
for row in range(gy1, gy2 + 1):
|
||||
nz = np.where(horiz_lines[row, :] > 0)[0]
|
||||
if len(nz) > w * 0.10:
|
||||
row_x1s.append(int(nz[0]))
|
||||
row_x2s.append(int(nz[-1]))
|
||||
if len(row_x1s) >= 2:
|
||||
# Filter out full-width separator lines (span >85% of page) before computing bounds
|
||||
pairs = [(lx1, lx2) for lx1, lx2 in zip(row_x1s, row_x2s) if (lx2 - lx1) < w * 0.85]
|
||||
if pairs:
|
||||
gx1 = min(p[0] for p in pairs)
|
||||
gx2 = max(p[1] for p in pairs)
|
||||
else:
|
||||
gx1, gx2 = gx1_raw, gx2_raw # all lines are separators — use raw
|
||||
else:
|
||||
gx1, gx2 = gx1_raw, gx2_raw
|
||||
col_width = gx2 - gx1
|
||||
col_horiz_in = horiz_lines[gy1:gy2+1, gx1:gx2]
|
||||
line_per_row = (col_horiz_in > 0).sum(axis=1)
|
||||
actual = np.where(line_per_row > col_width * 0.5)[0]
|
||||
if len(actual) == 0:
|
||||
continue
|
||||
# Reject if detected lines are too far apart (text separators, not table/chart)
|
||||
first_line = int(actual[0]) + gy1
|
||||
last_line = int(actual[-1]) + gy1
|
||||
# If lines are far apart, this is likely text separators not a table/chart:
|
||||
# suppress expansion and require taller minimum height.
|
||||
large_gap = len(actual) >= 2 and (int(actual[-1]) - int(actual[0])) > 60
|
||||
if large_gap:
|
||||
top = first_line - 10
|
||||
if len(actual) >= 3:
|
||||
# 3+ separators = multi-section table: expand to capture rows below last separator
|
||||
bottom = last_line
|
||||
prev_blank = 0
|
||||
for row in range(last_line + 1, min(h, last_line + 300)):
|
||||
if binary[row, gx1:gx2].sum() == 0:
|
||||
prev_blank += 1
|
||||
if prev_blank > 30: break
|
||||
else:
|
||||
prev_blank = 0
|
||||
bottom = row
|
||||
else:
|
||||
# 1-2 lines = figure border / panel divider: minimal expansion
|
||||
bottom = last_line + 10
|
||||
else:
|
||||
top = first_line
|
||||
prev_blank = 0
|
||||
for row in range(first_line - 1, max(0, first_line - 500), -1):
|
||||
if binary[row, gx1:gx2].sum() == 0:
|
||||
prev_blank += 1
|
||||
if prev_blank > 50: break
|
||||
else:
|
||||
prev_blank = 0
|
||||
top = row
|
||||
bottom = last_line
|
||||
prev_blank = 0
|
||||
for row in range(last_line + 1, min(h, last_line + 400)):
|
||||
if binary[row, gx1:gx2].sum() == 0:
|
||||
prev_blank += 1
|
||||
if prev_blank > 50: break
|
||||
else:
|
||||
prev_blank = 0
|
||||
bottom = row
|
||||
margin = 20
|
||||
x1 = max(0, gx1-margin)
|
||||
y1 = max(0, top-margin)
|
||||
x2 = min(w, gx2+margin)
|
||||
y2 = min(h, bottom+margin)
|
||||
box_w = x2 - x1
|
||||
box_h = y2 - y1
|
||||
|
||||
# ── Reject obvious non-figures ──
|
||||
min_h = 120 if large_gap else 80
|
||||
if box_h < min_h or box_w < 100:
|
||||
continue
|
||||
# 2. Too tall — likely grabbed a whole text column
|
||||
if box_h > h * 0.75:
|
||||
continue
|
||||
# 3. Text density: if >60% of pixels are black, it's probably a text block
|
||||
roi = binary[y1:y2, x1:x2]
|
||||
if roi.size == 0:
|
||||
continue
|
||||
density = roi.sum() / (roi.size * 255)
|
||||
if density > 0.60:
|
||||
continue
|
||||
# 4. Aspect ratio: extremely wide+short bars are headers
|
||||
if box_h < box_w * 0.10:
|
||||
continue
|
||||
# 5. Full-width shallow bar — page header/footer banner
|
||||
if box_w > w * 0.85 and box_h < 160:
|
||||
continue
|
||||
# 6. Thin separator line expanded into text: original group was tiny but grew huge
|
||||
original_h = gy2 - gy1
|
||||
if original_h < 30 and box_h > original_h * 5:
|
||||
continue
|
||||
|
||||
results.append([x1, y1, x2, y2])
|
||||
|
||||
# Merge overlapping boxes
|
||||
results.sort(key=lambda r: (r[1], r[0]))
|
||||
merged = []
|
||||
used = [False] * len(results)
|
||||
for i, r1 in enumerate(results):
|
||||
if used[i]: continue
|
||||
x1, y1, x2, y2 = r1
|
||||
for j in range(i+1, len(results)):
|
||||
if used[j]: continue
|
||||
rx1, ry1, rx2, ry2 = results[j]
|
||||
ix1, iy1 = max(x1, rx1), max(y1, ry1)
|
||||
ix2, iy2 = min(x2, rx2), min(y2, ry2)
|
||||
if ix2 > ix1 and iy2 > iy1:
|
||||
inter = (ix2-ix1) * (iy2-iy1)
|
||||
smaller = min((x2-x1)*(y2-y1), (rx2-rx1)*(ry2-ry1))
|
||||
if smaller > 0 and inter / smaller > 0.3:
|
||||
used[j] = True
|
||||
x1 = min(x1, rx1); y1 = min(y1, ry1)
|
||||
x2 = max(x2, rx2); y2 = max(y2, ry2)
|
||||
merged.append([x1, y1, x2, y2])
|
||||
return merged
|
||||
|
||||
page_dir = sys.argv[1]
|
||||
out_dir = sys.argv[2]
|
||||
os.makedirs(out_dir, exist_ok=True)
|
||||
page_files = sorted(f for f in os.listdir(page_dir) if f.startswith('page-') and f.endswith('.png'))
|
||||
total = 0
|
||||
for page_f in page_files:
|
||||
page_num = page_f[5:-4]
|
||||
img_path = os.path.join(page_dir, page_f)
|
||||
boxes = detect_figures(img_path)
|
||||
if not boxes:
|
||||
continue
|
||||
img_color = cv2.imread(img_path)
|
||||
if img_color is None:
|
||||
continue
|
||||
for i, (x1, y1, x2, y2) in enumerate(boxes, 1):
|
||||
crop = img_color[y1:y2, x1:x2]
|
||||
if crop.size == 0:
|
||||
continue
|
||||
out_name = f'figure_p{page_num}_{i}.png'
|
||||
cv2.imwrite(os.path.join(out_dir, out_name), crop)
|
||||
print(f'ok:{out_name}', flush=True)
|
||||
total += 1
|
||||
print(f'done:{total}', flush=True)
|
||||
`;
|
||||
|
||||
async function convertCcittToPng(outDir: string, pdfPath: string): Promise<{ converted: number; errors: string[] }> {
|
||||
const errors: string[] = [];
|
||||
let converted = 0;
|
||||
try {
|
||||
const { stdout } = await execFileAsync('python3', ['-c', CCITT_CONVERT_PY, outDir, pdfPath], { timeout: 60_000 });
|
||||
for (const line of stdout.split('\n')) {
|
||||
if (line.startsWith('done:')) converted = parseInt(line.slice(5), 10) || 0;
|
||||
else if (line.startsWith('err:')) errors.push(line.slice(4));
|
||||
}
|
||||
} catch (err: any) {
|
||||
errors.push(`ccitt_convert: ${String(err.message || err).slice(0, 200)}`);
|
||||
}
|
||||
return { converted, errors };
|
||||
}
|
||||
|
||||
async function extractFigures(pageDir: string, outDir: string): Promise<{ files: string[]; errors: string[] }> {
|
||||
const files: string[] = [];
|
||||
const errors: string[] = [];
|
||||
try {
|
||||
const { stdout } = await execFileAsync('python3', ['-c', FIGURES_DETECT_PY, pageDir, outDir], { timeout: 120_000 });
|
||||
for (const line of stdout.split('\n')) {
|
||||
if (line.startsWith('ok:')) files.push(line.slice(3).trim());
|
||||
else if (line.startsWith('err:')) errors.push(line.slice(4));
|
||||
}
|
||||
} catch (err: any) {
|
||||
errors.push(`figures_detect: ${String(err.message || err).slice(0, 200)}`);
|
||||
}
|
||||
return { files, errors };
|
||||
}
|
||||
|
||||
export const pdfExtractImagesTool = {
|
||||
name: 'pdf_extract_images',
|
||||
description: 'Extract images and diagrams from a PDF file. mode "images" extracts embedded raster images. mode "figures" auto-detects and crops figures/tables using OpenCV line detection. mode "both" does images+figures (default). Use out_dir to save directly into a PPTX project folder.',
|
||||
schema: {
|
||||
path: 'Path to the PDF file (relative to workspace or absolute)',
|
||||
mode: '"images" (embedded rasters), "figures" (auto-crop figures/tables via OpenCV), or "both" = images+figures (default)',
|
||||
out_dir: 'Output folder name relative to workspace (default: uploads/{basename}-images). Use the PPTX project folder name to save images there directly.',
|
||||
page_from: 'First page (1-indexed, default: 1)',
|
||||
page_to: 'Last page (inclusive, default: last page)',
|
||||
dpi: 'Resolution for page rendering in DPI (default: 150)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
required: ['path'],
|
||||
properties: {
|
||||
path: { type: 'string', description: 'Path to the PDF file (relative to workspace or absolute)' },
|
||||
mode: { type: 'string', enum: ['images', 'figures', 'both'], description: 'Extraction mode: images=embedded rasters, figures=OpenCV-cropped figures/tables, both=images+figures (default)' },
|
||||
out_dir: { type: 'string', description: 'Output folder relative to workspace (default: uploads/{basename}-images). Set to PPTX project folder name to save images there directly.' },
|
||||
page_from: { type: 'number', description: 'First page (1-indexed)' },
|
||||
page_to: { type: 'number', description: 'Last page (inclusive)' },
|
||||
dpi: { type: 'number', description: 'Rendering DPI for pages mode (default: 150)' },
|
||||
},
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
|
||||
if (path.extname(resolved).toLowerCase() !== '.pdf') return { success: false, error: 'File must have a .pdf extension' };
|
||||
|
||||
const mode = String(args?.mode || 'both');
|
||||
const dpi = Math.min(300, Math.max(72, Number(args?.dpi ?? 150)));
|
||||
const pageFrom = args?.page_from ? Math.max(1, Math.floor(Number(args.page_from))) : null;
|
||||
const pageTo = args?.page_to ? Math.max(1, Math.floor(Number(args.page_to))) : null;
|
||||
|
||||
const basename = path.basename(resolved, '.pdf').replace(/[^a-zA-Z0-9가-힣._-]/g, '_');
|
||||
const dirBasename = basename.slice(0, 35);
|
||||
const customOutDir = args?.out_dir ? path.normalize(String(args.out_dir).trim()).replace(/\/+$/, '') : null;
|
||||
const outDir = customOutDir
|
||||
? (path.isAbsolute(customOutDir) ? customOutDir : path.join(workspacePath, customOutDir))
|
||||
: path.join(workspacePath, 'uploads', `${dirBasename}-images`);
|
||||
const outDirRel = customOutDir
|
||||
? (path.isAbsolute(customOutDir) ? path.relative(workspacePath, customOutDir) : customOutDir)
|
||||
: `uploads/${dirBasename}-images`;
|
||||
fs.mkdirSync(outDir, { recursive: true });
|
||||
|
||||
const extracted: string[] = [];
|
||||
const errors: string[] = [];
|
||||
|
||||
// --- pdfimages: extract embedded raster images ---
|
||||
if (mode === 'images' || mode === 'both') {
|
||||
const imgArgs = ['-all'];
|
||||
if (pageFrom) imgArgs.push('-f', String(pageFrom));
|
||||
if (pageTo) imgArgs.push('-l', String(pageTo));
|
||||
imgArgs.push(resolved, path.join(outDir, 'img'));
|
||||
try {
|
||||
await execFileAsync('pdfimages', imgArgs, { timeout: 60_000 });
|
||||
// Convert any CCITT G4 files to PNG before collecting results
|
||||
const ccittFiles = fs.readdirSync(outDir).filter(f => /^img-\d+\.ccitt$/i.test(f));
|
||||
if (ccittFiles.length > 0) {
|
||||
const { errors: ccittErrors } = await convertCcittToPng(outDir, resolved);
|
||||
errors.push(...ccittErrors);
|
||||
}
|
||||
const files = fs.readdirSync(outDir)
|
||||
.filter(f => /^img-\d+\.(jpg|jpeg|png|ppm|pbm|tif|tiff)$/i.test(f))
|
||||
.sort();
|
||||
extracted.push(...files.map(f => `${outDirRel}/${f}`));
|
||||
} catch (err: any) {
|
||||
errors.push(`pdfimages: ${String(err.message || err).slice(0, 200)}`);
|
||||
}
|
||||
}
|
||||
|
||||
// --- figures mode: render pages then auto-crop figures/tables with OpenCV ---
|
||||
if (mode === 'figures' || mode === 'both') {
|
||||
const figDpi = Math.min(300, Math.max(100, dpi));
|
||||
const ppmArgs = ['-png', '-r', String(figDpi)];
|
||||
if (pageFrom) ppmArgs.push('-f', String(pageFrom));
|
||||
if (pageTo) ppmArgs.push('-l', String(pageTo));
|
||||
ppmArgs.push(resolved, path.join(outDir, 'page'));
|
||||
try {
|
||||
await execFileAsync('pdftoppm', ppmArgs, { timeout: 120_000 });
|
||||
} catch (err: any) {
|
||||
errors.push(`pdftoppm: ${String(err.message || err).slice(0, 200)}`);
|
||||
}
|
||||
const { files: figFiles, errors: figErrors } = await extractFigures(outDir, outDir);
|
||||
errors.push(...figErrors);
|
||||
for (const f of fs.readdirSync(outDir).filter(f => f.startsWith('page-') && f.endsWith('.png'))) {
|
||||
try { fs.unlinkSync(path.join(outDir, f)); } catch {}
|
||||
}
|
||||
extracted.push(...figFiles.map(f => `${outDirRel}/${f}`));
|
||||
}
|
||||
|
||||
if (extracted.length === 0) {
|
||||
const errMsg = errors.length ? errors.join('; ') : 'No images found in the PDF';
|
||||
return { success: false, error: errMsg };
|
||||
}
|
||||
|
||||
const links = extracted.map(f => `[${path.basename(f)}](/api/files/${f})`).join('\n');
|
||||
return {
|
||||
success: true,
|
||||
stdout: `${extracted.length}개 파일 추출 완료:\n${links}`,
|
||||
data: { files: extracted, outDir, errors: errors.length ? errors : undefined },
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// pdf_extract_tables
|
||||
// 1차: PyMuPDF find_tables() — 선 기반 테이블, 빠름
|
||||
// 2차: pdfplumber — 공백/선 혼합, 선 없는 테이블에도 강함
|
||||
// ---------------------------------------------------------------------------
|
||||
const TABLE_EXTRACT_PY = `
|
||||
import sys, json, os, io
|
||||
|
||||
# PyMuPDF 1.24+ 가 import 시점에 C 레벨 fd=1(stdout)로 직접
|
||||
# "Consider using pymupdf_layout..." 를 출력해 JSON 파싱을 깨뜨린다.
|
||||
# sys.stdout 교체로는 막을 수 없으므로 os.dup2 로 fd 1 자체를 /dev/null 로 리다이렉트한다.
|
||||
_devnull_fd = os.open(os.devnull, os.O_WRONLY)
|
||||
_saved_fd = os.dup(1)
|
||||
os.dup2(_devnull_fd, 1)
|
||||
os.close(_devnull_fd)
|
||||
try:
|
||||
import fitz as _fitz
|
||||
except ImportError:
|
||||
_fitz = None
|
||||
finally:
|
||||
os.dup2(_saved_fd, 1) # stdout 복구
|
||||
os.close(_saved_fd)
|
||||
|
||||
pdf_path = sys.argv[1]
|
||||
fmt = sys.argv[2] if len(sys.argv) > 2 else 'markdown'
|
||||
page_from = int(sys.argv[3]) - 1 if len(sys.argv) > 3 else 0 # 0-indexed
|
||||
page_to = int(sys.argv[4]) - 1 if len(sys.argv) > 4 else None # inclusive, 0-indexed
|
||||
engine = sys.argv[5] if len(sys.argv) > 5 else 'auto' # auto|pymupdf|pdfplumber
|
||||
|
||||
def cell(v):
|
||||
return str(v).replace('\\n', ' ').strip() if v is not None else ''
|
||||
|
||||
def to_markdown(rows):
|
||||
if not rows or not rows[0]:
|
||||
return ''
|
||||
widths = [max(len(cell(r[i])) for r in rows if i < len(r)) for i in range(len(rows[0]))]
|
||||
widths = [max(w, 3) for w in widths]
|
||||
def row_str(r):
|
||||
return '| ' + ' | '.join(cell(r[i]).ljust(widths[i]) if i < len(r) else ' ' * widths[i] for i in range(len(widths))) + ' |'
|
||||
sep = '| ' + ' | '.join('-' * w for w in widths) + ' |'
|
||||
lines = [row_str(rows[0]), sep] + [row_str(r) for r in rows[1:]]
|
||||
return '\\n'.join(lines)
|
||||
|
||||
def to_csv(rows):
|
||||
import csv, io
|
||||
buf = io.StringIO()
|
||||
w = csv.writer(buf)
|
||||
for r in rows:
|
||||
w.writerow([cell(v) for v in r])
|
||||
return buf.getvalue().rstrip()
|
||||
|
||||
def format_table(rows, fmt):
|
||||
if fmt == 'csv': return to_csv(rows)
|
||||
if fmt == 'json': return json.dumps([[cell(v) for v in r] for r in rows], ensure_ascii=False)
|
||||
return to_markdown(rows)
|
||||
|
||||
results = []
|
||||
errors = []
|
||||
|
||||
def try_pymupdf():
|
||||
if _fitz is None:
|
||||
raise ImportError('PyMuPDF(fitz) not available')
|
||||
doc = _fitz.open(pdf_path)
|
||||
end = page_to if page_to is not None else len(doc) - 1
|
||||
found = []
|
||||
for pno in range(page_from, min(end + 1, len(doc))):
|
||||
page = doc[pno]
|
||||
tabs = page.find_tables().tables # TableFinder → .tables 리스트
|
||||
for ti, tab in enumerate(tabs):
|
||||
rows = tab.extract()
|
||||
if not rows: continue
|
||||
found.append({'page': pno + 1, 'table': ti + 1, 'rows': len(rows),
|
||||
'cols': len(rows[0]) if rows else 0,
|
||||
'data': format_table(rows, fmt)})
|
||||
doc.close()
|
||||
return found
|
||||
|
||||
def try_pdfplumber():
|
||||
import pdfplumber
|
||||
found = []
|
||||
with pdfplumber.open(pdf_path) as pdf:
|
||||
end = page_to if page_to is not None else len(pdf.pages) - 1
|
||||
for pno in range(page_from, min(end + 1, len(pdf.pages))):
|
||||
page = pdf.pages[pno]
|
||||
tables = page.extract_tables({
|
||||
'vertical_strategy': 'lines_strict',
|
||||
'horizontal_strategy': 'lines_strict',
|
||||
})
|
||||
# 선 감지 실패 시 text 기반으로 재시도
|
||||
if not tables:
|
||||
tables = page.extract_tables({
|
||||
'vertical_strategy': 'text',
|
||||
'horizontal_strategy': 'text',
|
||||
'snap_tolerance': 3,
|
||||
'join_tolerance': 3,
|
||||
'edge_min_length': 10,
|
||||
})
|
||||
for ti, rows in enumerate(tables):
|
||||
if not rows: continue
|
||||
found.append({'page': pno + 1, 'table': ti + 1, 'rows': len(rows),
|
||||
'cols': len(rows[0]) if rows else 0,
|
||||
'data': format_table(rows, fmt)})
|
||||
return found
|
||||
|
||||
try:
|
||||
if engine == 'pymupdf':
|
||||
results = try_pymupdf()
|
||||
elif engine == 'pdfplumber':
|
||||
results = try_pdfplumber()
|
||||
else: # auto: pymupdf first, pdfplumber if no tables found
|
||||
results = try_pymupdf()
|
||||
if not results:
|
||||
results = try_pdfplumber()
|
||||
if results:
|
||||
for r in results: r['engine'] = 'pdfplumber'
|
||||
else:
|
||||
errors.append('no_tables')
|
||||
else:
|
||||
for r in results: r['engine'] = 'pymupdf'
|
||||
except Exception as e:
|
||||
import traceback
|
||||
errors.append(str(e))
|
||||
errors.append(traceback.format_exc()[-600:])
|
||||
|
||||
print(json.dumps({'tables': results, 'errors': errors}, ensure_ascii=False))
|
||||
`;
|
||||
|
||||
export const pdfExtractTablesTool = {
|
||||
name: 'pdf_extract_tables',
|
||||
description: [
|
||||
'Extract tables from a PDF file as structured data.',
|
||||
'Uses PyMuPDF find_tables() first (fast, line-based); falls back to pdfplumber (handles borderless tables too).',
|
||||
'format: "markdown" (default) | "csv" | "json".',
|
||||
'engine: "auto" (default) | "pymupdf" | "pdfplumber".',
|
||||
'For scanned/image-based PDFs use pdf_extract_images with mode "figures" instead.',
|
||||
].join(' '),
|
||||
schema: {
|
||||
path: 'Path to the PDF file (absolute or relative to workspace)',
|
||||
format: 'Output format: "markdown" (default) | "csv" | "json"',
|
||||
engine: 'Extraction engine: "auto" (default) | "pymupdf" | "pdfplumber"',
|
||||
page_from: 'First page to scan, 1-indexed (default: 1)',
|
||||
page_to: 'Last page to scan, inclusive (default: last page)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
path: { type: 'string' },
|
||||
format: { type: 'string', enum: ['markdown', 'csv', 'json'] },
|
||||
engine: { type: 'string', enum: ['auto', 'pymupdf', 'pdfplumber'] },
|
||||
page_from: { type: 'number' },
|
||||
page_to: { type: 'number' },
|
||||
},
|
||||
required: ['path'],
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
|
||||
if (path.extname(resolved).toLowerCase() !== '.pdf') return { success: false, error: 'File must have a .pdf extension' };
|
||||
|
||||
const fmt = ['markdown', 'csv', 'json'].includes(args?.format) ? String(args.format) : 'markdown';
|
||||
const engine = ['auto', 'pymupdf', 'pdfplumber'].includes(args?.engine) ? String(args.engine) : 'auto';
|
||||
const pageFrom = args?.page_from ? String(Math.max(1, Math.floor(Number(args.page_from)))) : '1';
|
||||
const pageTo = args?.page_to ? String(Math.max(1, Math.floor(Number(args.page_to)))) : '9999';
|
||||
|
||||
let raw: { tables: any[]; errors: string[] };
|
||||
try {
|
||||
const { stdout } = await execFileAsync(
|
||||
'python3', ['-c', TABLE_EXTRACT_PY, resolved, fmt, pageFrom, pageTo, engine],
|
||||
{ timeout: 60_000, maxBuffer: 20 * 1024 * 1024 },
|
||||
);
|
||||
// PyMuPDF가 JSON 앞에 경고 텍스트를 stdout으로 출력할 수 있으므로
|
||||
// 첫 번째 '{' 이후만 JSON으로 파싱한다.
|
||||
const jsonStart = stdout.indexOf('{');
|
||||
raw = JSON.parse(jsonStart >= 0 ? stdout.slice(jsonStart) : stdout);
|
||||
} catch (err: any) {
|
||||
return { success: false, error: `Table extraction failed: ${String(err.message || err).slice(0, 300)}` };
|
||||
}
|
||||
|
||||
if (!raw.tables || raw.tables.length === 0) {
|
||||
const hint = raw.errors?.includes('no_tables')
|
||||
? 'No tables detected. If this is a scanned PDF, try pdf_extract_images with mode "figures".'
|
||||
: `No tables found. Errors: ${raw.errors?.join('; ') || 'none'}`;
|
||||
return { success: false, error: hint };
|
||||
}
|
||||
|
||||
const lines: string[] = [];
|
||||
for (const t of raw.tables) {
|
||||
lines.push(`### Page ${t.page} — Table ${t.table} (${t.rows} rows × ${t.cols} cols, engine: ${t.engine ?? engine})`);
|
||||
lines.push('');
|
||||
lines.push(t.data);
|
||||
lines.push('');
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stdout: lines.join('\n').trimEnd(),
|
||||
data: { table_count: raw.tables.length, tables: raw.tables, errors: raw.errors?.length ? raw.errors : undefined },
|
||||
};
|
||||
},
|
||||
};
|
||||
+4
-2
@@ -7,7 +7,9 @@ import { ToolResult } from '../types.js';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
function getWorkspacePath(): string {
|
||||
function getWorkspacePath(args?: any): string {
|
||||
const sessionPath = args?._workspacePath || args?._workspace;
|
||||
if (sessionPath) return sessionPath;
|
||||
try {
|
||||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
} catch {
|
||||
@@ -39,7 +41,7 @@ export const pdfReadTool = {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = getWorkspacePath();
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) {
|
||||
|
||||
+12
-3
@@ -598,7 +598,7 @@ export const pptxTool: import('./registry.js').Tool = {
|
||||
|
||||
const configMgr = getConfig();
|
||||
const config = configMgr.getConfig() as any;
|
||||
const workspacePath = args._workspacePath || config.workspace?.path || process.cwd();
|
||||
const workspacePath = args?._workspacePath || args?._workspace || config.workspace?.path || process.cwd();
|
||||
|
||||
// Inject config defaults into spec so Python engine can use them
|
||||
if (!spec.template && config.ppt?.template) spec.template = config.ppt.template;
|
||||
@@ -676,8 +676,17 @@ export const pptxTool: import('./registry.js').Tool = {
|
||||
let found: string | null = null;
|
||||
if (fs.existsSync(uploadsDir)) {
|
||||
for (const sub of fs.readdirSync(uploadsDir)) {
|
||||
const candidate = path.join(uploadsDir, sub, basename);
|
||||
const subDir = path.join(uploadsDir, sub);
|
||||
if (!fs.existsSync(subDir) || !fs.statSync(subDir).isDirectory()) continue;
|
||||
// Exact match
|
||||
const candidate = path.join(subDir, basename);
|
||||
if (fs.existsSync(candidate)) { found = candidate; break; }
|
||||
// Prefix match: "figure_p05" → "figure_p05_1.png"
|
||||
const prefix = basename.replace(/\.[^.]+$/, '');
|
||||
const prefixMatch = fs.readdirSync(subDir).find(f =>
|
||||
f.startsWith(prefix + '_') || f.startsWith(prefix + '.')
|
||||
);
|
||||
if (prefixMatch) { found = path.join(subDir, prefixMatch); break; }
|
||||
}
|
||||
}
|
||||
if (found) {
|
||||
@@ -814,7 +823,7 @@ export const editPptxTool: import('./registry.js').Tool = {
|
||||
|
||||
const configMgr = getConfig();
|
||||
const config = configMgr.getConfig() as any;
|
||||
const workspacePath = args._workspacePath || config.workspace?.path || process.cwd();
|
||||
const workspacePath = args?._workspacePath || args?._workspace || config.workspace?.path || process.cwd();
|
||||
|
||||
// Resolve absolute path
|
||||
const absPath = path.isAbsolute(existingPath)
|
||||
|
||||
+2
-2
@@ -355,7 +355,7 @@ export const pubmedFulltextTool = {
|
||||
|
||||
// Download PDF
|
||||
const config = getConfig().getConfig();
|
||||
const workspaceDir = args?._workspacePath || config.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
const workspaceDir = args?._workspacePath || args?._workspace || config.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
const pdfDir = path.join(workspaceDir, 'pubmed');
|
||||
fs.mkdirSync(pdfDir, { recursive: true });
|
||||
|
||||
@@ -472,7 +472,7 @@ export const pubmedFulltextTool = {
|
||||
|
||||
// Save to workspace (prefer per-user path injected by v2 executeTool)
|
||||
const config = getConfig().getConfig();
|
||||
const workspaceDir = args?._workspacePath || config.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
const workspaceDir = args?._workspacePath || args?._workspace || config.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
const savePath = args?.save_path
|
||||
? path.join(workspaceDir, args.save_path)
|
||||
: path.join(workspaceDir, 'pubmed', `${pmcid}.txt`);
|
||||
|
||||
+47
-2
@@ -1,9 +1,13 @@
|
||||
import { spawn } from 'child_process';
|
||||
import path from 'path';
|
||||
import fs from 'fs';
|
||||
import { getConfig } from '../config/config.js';
|
||||
import { ToolResult } from '../types.js';
|
||||
|
||||
function getWorkspacePath(): string {
|
||||
function getWorkspacePath(args?: any): string {
|
||||
// Prefer the session-specific workspace passed by executeTool
|
||||
const sessionPath = args?._workspacePath || args?._workspace;
|
||||
if (sessionPath) return sessionPath;
|
||||
try {
|
||||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
} catch {
|
||||
@@ -53,7 +57,21 @@ export const pythonEvalTool = {
|
||||
}
|
||||
|
||||
const timeoutSec = Math.min(60, Math.max(1, Number(args?.timeout ?? 15)));
|
||||
const workspacePath = getWorkspacePath();
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
|
||||
// Snapshot files before execution
|
||||
const beforeFiles = new Set<string>();
|
||||
try {
|
||||
function scanDir(dir: string, prefix: string) {
|
||||
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
||||
const rel = prefix ? `${prefix}/${entry.name}` : entry.name;
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) { scanDir(full, rel); }
|
||||
else if (entry.isFile()) { try { beforeFiles.add(`${rel}:${fs.statSync(full).mtimeMs}`); } catch {} }
|
||||
}
|
||||
}
|
||||
scanDir(workspacePath, '');
|
||||
} catch {}
|
||||
|
||||
if (args?.packages) {
|
||||
const pkgs = String(args.packages).split(',').map((p: string) => p.trim()).filter((p: string) => /^[a-zA-Z0-9_\-\.]+$/.test(p));
|
||||
@@ -108,10 +126,37 @@ export const pythonEvalTool = {
|
||||
return { success: false, error: `Execution timed out after ${timeoutSec}s` };
|
||||
}
|
||||
|
||||
// Detect newly created files
|
||||
const newFiles: string[] = [];
|
||||
try {
|
||||
function scanDirAfter(dir: string, prefix: string) {
|
||||
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
||||
const rel = prefix ? `${prefix}/${entry.name}` : entry.name;
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) { scanDirAfter(full, rel); }
|
||||
else if (entry.isFile()) {
|
||||
const key = `${rel}:${fs.statSync(full).mtimeMs}`;
|
||||
if (!beforeFiles.has(key)) { newFiles.push(rel); }
|
||||
}
|
||||
}
|
||||
}
|
||||
scanDirAfter(workspacePath, '');
|
||||
} catch {}
|
||||
|
||||
const parts: string[] = [];
|
||||
if (result.stdout.trim()) parts.push(`[stdout]\n${result.stdout.trim().slice(0, 20_000)}`);
|
||||
if (result.stderr.trim()) parts.push(`[stderr]\n${result.stderr.trim().slice(0, 3_000)}`);
|
||||
if (!result.stdout.trim() && !result.stderr.trim()) parts.push('(no output)');
|
||||
if (newFiles.length > 0) {
|
||||
const links = newFiles.map(f => {
|
||||
const ext = path.extname(f).toLowerCase();
|
||||
const isImage = ['.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.tiff', '.tif'].includes(ext);
|
||||
return isImage
|
||||
? ``
|
||||
: `[${path.basename(f)}](/api/files/${f})`;
|
||||
}).join('\n');
|
||||
parts.push(`[생성된 파일]\n${links}`);
|
||||
}
|
||||
parts.push(`[exit ${result.exitCode}]`);
|
||||
|
||||
return {
|
||||
|
||||
+10
-2
@@ -15,8 +15,8 @@ import { pptxTool, editPptxTool } from './pptx.js';
|
||||
import { pubmedSearchTool, pubmedFetchTool, pubmedFulltextTool } from './pubmed.js';
|
||||
import { openalexSearchTool, semanticSearchTool } from './scholar.js';
|
||||
import { pdfReadTool } from './pdf.js';
|
||||
import { pdfExtractImagesTool } from './pdf-extract.js';
|
||||
import { imageReadTool } from './image.js';
|
||||
import { pdfExtractImagesTool, pdfExtractTablesTool } from './pdf-extract.js';
|
||||
import { imageReadTool, imagePreviewTool, imageInfoTool, imageEditTool } from './image.js';
|
||||
import { audioTranscribeTool } from './audio-transcribe.js';
|
||||
import { pythonEvalTool } from './python.js';
|
||||
import { sqliteTool } from './sqlite.js';
|
||||
@@ -58,6 +58,9 @@ const TOOL_PROFILE_TOOL_NAMES: Record<Exclude<ToolProfile, 'full'>, ReadonlySet<
|
||||
'memory_write',
|
||||
'pdf_read',
|
||||
'image_read',
|
||||
'image_preview',
|
||||
'image_info',
|
||||
'image_edit',
|
||||
'audio_transcribe',
|
||||
'python_eval',
|
||||
'sqlite_query',
|
||||
@@ -74,6 +77,7 @@ const TOOL_PROFILE_TOOL_NAMES: Record<Exclude<ToolProfile, 'full'>, ReadonlySet<
|
||||
'semantic_search',
|
||||
'pdf_read',
|
||||
'image_read',
|
||||
'image_preview',
|
||||
]),
|
||||
};
|
||||
|
||||
@@ -187,7 +191,11 @@ class ToolRegistry {
|
||||
// Document & data tools
|
||||
this.registerSafe(pdfReadTool);
|
||||
this.registerSafe(pdfExtractImagesTool);
|
||||
this.registerSafe(pdfExtractTablesTool);
|
||||
this.registerSafe(imageReadTool);
|
||||
this.registerSafe(imagePreviewTool);
|
||||
this.registerSafe(imageInfoTool);
|
||||
this.registerSafe(imageEditTool);
|
||||
this.registerSafe(audioTranscribeTool);
|
||||
this.registerSafe(pythonEvalTool);
|
||||
this.registerSafe(sqliteTool);
|
||||
|
||||
+3
-1
@@ -6,6 +6,8 @@ import { log } from '../security/log-scrubber.js';
|
||||
export interface ShellToolArgs {
|
||||
command: string;
|
||||
cwd?: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
// ── Path confinement helper ───────────────────────────────────────────────────
|
||||
@@ -56,7 +58,7 @@ function containsOutOfScopeAbsPath(command: string, workspacePath: string): bool
|
||||
export async function executeShell(args: ShellToolArgs): Promise<ToolResult> {
|
||||
const config = getConfig().getConfig();
|
||||
const permissions = config.tools.permissions.shell;
|
||||
const workspacePath = path.resolve(config.workspace.path);
|
||||
const workspacePath = path.resolve(args._workspacePath || args._workspace || config.workspace.path);
|
||||
|
||||
// Determine and resolve working directory
|
||||
const cwd = path.resolve(args.cwd ? args.cwd : workspacePath);
|
||||
|
||||
+11
-7
@@ -3,9 +3,9 @@ import fs from 'fs';
|
||||
import { getConfig } from '../config/config.js';
|
||||
import { ToolResult } from '../types.js';
|
||||
|
||||
function getWorkspacePath(): string {
|
||||
function getWorkspacePath(args?: any): string {
|
||||
try {
|
||||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
return args?._workspacePath || args?._workspace || getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
} catch {
|
||||
return path.join(process.cwd(), 'workspace');
|
||||
}
|
||||
@@ -58,11 +58,15 @@ export const sqliteTool = {
|
||||
if (!dbPath) return { success: false, error: 'db_path is required' };
|
||||
if (!query) return { success: false, error: 'query is required' };
|
||||
|
||||
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
|
||||
const resolved = path.isAbsolute(dbPath) ? dbPath : path.resolve(workspacePath, dbPath);
|
||||
|
||||
if (!isPathInsideDir(workspacePath, resolved)) {
|
||||
return { success: false, error: 'db_path must be inside the workspace directory' };
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
let resolved: string;
|
||||
if (path.isAbsolute(dbPath)) {
|
||||
resolved = dbPath;
|
||||
} else {
|
||||
resolved = path.resolve(workspacePath, dbPath);
|
||||
if (!isPathInsideDir(workspacePath, resolved)) {
|
||||
return { success: false, error: 'relative db_path must stay inside workspace (use absolute path for shared databases)' };
|
||||
}
|
||||
}
|
||||
|
||||
const allowWrite = Boolean(args?.write);
|
||||
|
||||
@@ -4604,10 +4604,12 @@ function renderFileDownloads(steps, processEntries) {
|
||||
}
|
||||
}
|
||||
|
||||
const _imgExts = /\.(jpg|jpeg|png|gif|webp|heic|heif|bmp|svg|avif|tiff|tif)(\?|$)/i;
|
||||
for (const rawUrl of allUrls) {
|
||||
if (seenUrls.has(rawUrl)) continue;
|
||||
seenUrls.add(rawUrl);
|
||||
const decoded = (() => { try { return decodeURIComponent(rawUrl); } catch { return rawUrl; } })();
|
||||
if (_imgExts.test(decoded)) continue; // images shown inline, no download button needed
|
||||
const filename = decoded.split('/').pop() || 'download';
|
||||
const isPptx = decoded.toLowerCase().indexOf('.pptx') !== -1;
|
||||
if (isPptx) {
|
||||
|
||||
Reference in New Issue
Block a user