v2.7.1 — image edit improvements, search config, bug fixes
image_edit: - speech_bubble operation with Korean font, rounded rect, tail - sketch quality: CLAHE pre-processing for better line contrast - output filename with timestamp to prevent overwrites - anime/painting stylize timeout 30s→600s (AnimeGANv2 model download) - remove_bg PNG output fix server-v2.ts: - resolveToolImageContent: embed edited image as base64 in tool results so vision models can see the output (3 execution paths covered) - _activeModelName: fix vision support detection for Ollama-hosted models (kimi/gemini run via Ollama — provider='ollama' was masking vision capability) - Synthetic tool call ID generation to prevent Gemini function_response empty name error - image_edit path hint format changed to English to prevent Gemini hallucination - userRequestedImageEdit: expanded keyword list (풍선, 달아, 붙여 등) - image_read OCR failure returns success:true with visual description hint - TOOL_BLOCKS photo: image_edit workflow documentation updated web-ui/index.html: - renderFileDownloads: skip download buttons for image file extensions Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -26,7 +26,7 @@ export const audioTranscribeTool = {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = getConfig().getConfig()?.workspace?.path || process.cwd();
|
||||
const workspacePath = args?._workspacePath || args?._workspace || getConfig().getConfig()?.workspace?.path || process.cwd();
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) {
|
||||
|
||||
+51
-18
@@ -11,13 +11,16 @@ const execFileAsync = promisify(execFile);
|
||||
const PATCH_OUTPUT_MAX_CHARS = 8000;
|
||||
|
||||
// Helper function to check if path is allowed
|
||||
function resolveWorkspacePath(targetPath: string): string {
|
||||
const config = getConfig().getConfig();
|
||||
const workspace = config.workspace.path;
|
||||
function resolveWorkspacePath(targetPath: string, workspacePath?: string): string {
|
||||
const workspace = workspacePath || getConfig().getConfig().workspace.path;
|
||||
if (path.isAbsolute(targetPath)) return targetPath;
|
||||
return path.join(workspace, targetPath);
|
||||
}
|
||||
|
||||
function getSessionWorkspace(args: any): string {
|
||||
return args?._workspacePath || args?._workspace || getConfig().getConfig().workspace.path;
|
||||
}
|
||||
|
||||
function normalizePathForCompare(p: string): string {
|
||||
const resolved = path.resolve(String(p || ''));
|
||||
if (process.platform === 'win32') return resolved.toLowerCase();
|
||||
@@ -168,6 +171,8 @@ export interface ReadToolArgs {
|
||||
path: string;
|
||||
start_line?: number;
|
||||
num_lines?: number;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
type RetrievalMode = 'fast' | 'standard' | 'deep';
|
||||
@@ -203,7 +208,8 @@ function retrievalMaxLines(mode: RetrievalMode): number {
|
||||
|
||||
export async function executeRead(args: ReadToolArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const absPath = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const absPath = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(absPath);
|
||||
if (!pathCheck.allowed) {
|
||||
return {
|
||||
@@ -252,6 +258,8 @@ export async function executeRead(args: ReadToolArgs): Promise<ToolResult> {
|
||||
export interface WriteToolArgs {
|
||||
path: string;
|
||||
content: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
export async function executeWrite(args: WriteToolArgs): Promise<ToolResult> {
|
||||
@@ -268,7 +276,8 @@ export async function executeWrite(args: WriteToolArgs): Promise<ToolResult> {
|
||||
error: 'content must be a string'
|
||||
};
|
||||
}
|
||||
const absPath = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const absPath = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(absPath);
|
||||
if (!pathCheck.allowed) {
|
||||
return {
|
||||
@@ -302,11 +311,14 @@ export interface EditToolArgs {
|
||||
path: string;
|
||||
old_str: string;
|
||||
new_str: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
export async function executeEdit(args: EditToolArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const absPath = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const absPath = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(absPath);
|
||||
if (!pathCheck.allowed) {
|
||||
return {
|
||||
@@ -356,11 +368,14 @@ export async function executeEdit(args: EditToolArgs): Promise<ToolResult> {
|
||||
// LIST DIRECTORY TOOL
|
||||
export interface ListToolArgs {
|
||||
path: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
export async function executeList(args: ListToolArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const absPath = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const absPath = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(absPath);
|
||||
if (!pathCheck.allowed) {
|
||||
return {
|
||||
@@ -441,9 +456,10 @@ export const listTool = {
|
||||
// ── DELETE ────────────────────────────────────────────────────────────────────
|
||||
import { rmSync, existsSync } from 'fs';
|
||||
|
||||
async function executeDelete(args: { path: string; recursive?: boolean }): Promise<ToolResult> {
|
||||
async function executeDelete(args: { path: string; recursive?: boolean; _workspacePath?: string; _workspace?: string }): Promise<ToolResult> {
|
||||
if (!args.path?.trim()) return { success: false, error: 'path is required' };
|
||||
const absPath = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const absPath = resolveWorkspacePath(args.path, workspacePath);
|
||||
if (!existsSync(absPath)) return { success: false, error: `Path does not exist: ${absPath}` };
|
||||
try {
|
||||
rmSync(absPath, { recursive: args.recursive ?? false, force: true });
|
||||
@@ -467,11 +483,14 @@ export const deleteTool = {
|
||||
export interface RenameArgs {
|
||||
path: string;
|
||||
new_path: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
export async function executeRename(args: RenameArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const src = resolveWorkspacePath(args.path);
|
||||
const dest = resolveWorkspacePath(args.new_path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const src = resolveWorkspacePath(args.path, workspacePath);
|
||||
const dest = resolveWorkspacePath(args.new_path, workspacePath);
|
||||
const srcCheck = isPathAllowed(src);
|
||||
const destCheck = isPathAllowed(dest);
|
||||
if (!srcCheck.allowed) return { success: false, error: srcCheck.reason };
|
||||
@@ -503,11 +522,14 @@ export const renameTool = {
|
||||
export interface CopyArgs {
|
||||
path: string;
|
||||
dest: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
export async function executeCopy(args: CopyArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const src = resolveWorkspacePath(args.path);
|
||||
const dest = resolveWorkspacePath(args.dest);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const src = resolveWorkspacePath(args.path, workspacePath);
|
||||
const dest = resolveWorkspacePath(args.dest, workspacePath);
|
||||
const srcCheck = isPathAllowed(src);
|
||||
const destCheck = isPathAllowed(dest);
|
||||
if (!srcCheck.allowed) return { success: false, error: srcCheck.reason };
|
||||
@@ -534,10 +556,13 @@ export const copyTool = {
|
||||
export interface MkdirArgs {
|
||||
path: string;
|
||||
recursive?: boolean;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
export async function executeMkdir(args: MkdirArgs): Promise<ToolResult> {
|
||||
export async function executeMkdir(args: MkdirArgs & { _workspacePath?: string; _workspace?: string }): Promise<ToolResult> {
|
||||
try {
|
||||
const abs = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const abs = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(abs);
|
||||
if (!pathCheck.allowed) return { success: false, error: pathCheck.reason };
|
||||
await fs.mkdir(abs, { recursive: args.recursive ?? true });
|
||||
@@ -560,10 +585,13 @@ export const mkdirTool = {
|
||||
// ── STAT / INFO ──────────────────────────────────────────────────────────────
|
||||
export interface StatArgs {
|
||||
path: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
export async function executeStat(args: StatArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const abs = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const abs = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(abs);
|
||||
if (!pathCheck.allowed) return { success: false, error: pathCheck.reason };
|
||||
const st = await fs.stat(abs);
|
||||
@@ -586,10 +614,13 @@ export const statTool = {
|
||||
export interface AppendArgs {
|
||||
path: string;
|
||||
content: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
export async function executeAppend(args: AppendArgs): Promise<ToolResult> {
|
||||
try {
|
||||
const abs = resolveWorkspacePath(args.path);
|
||||
const workspacePath = getSessionWorkspace(args);
|
||||
const abs = resolveWorkspacePath(args.path, workspacePath);
|
||||
const pathCheck = isPathAllowed(abs);
|
||||
if (!pathCheck.allowed) return { success: false, error: pathCheck.reason };
|
||||
await fs.mkdir(path.dirname(abs), { recursive: true });
|
||||
@@ -614,6 +645,8 @@ export const appendTool = {
|
||||
export interface ApplyPatchArgs {
|
||||
patch: string;
|
||||
check?: boolean;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
export async function executeApplyPatch(args: ApplyPatchArgs): Promise<ToolResult> {
|
||||
@@ -626,7 +659,7 @@ export async function executeApplyPatch(args: ApplyPatchArgs): Promise<ToolResul
|
||||
const validation = validatePatchPaths(targetPaths);
|
||||
if (!validation.ok) return { success: false, error: validation.error };
|
||||
|
||||
const workspacePath = getConfig().getConfig().workspace.path;
|
||||
const workspacePath = args._workspacePath || args._workspace || getConfig().getConfig().workspace.path;
|
||||
const tempPatchPath = path.join(
|
||||
os.tmpdir(),
|
||||
`smallclaw-apply-${Date.now()}-${Math.random().toString(36).slice(2)}.patch`
|
||||
|
||||
+542
-5
@@ -4,7 +4,9 @@ import fs from 'fs';
|
||||
import { getConfig } from '../config/config.js';
|
||||
import { ToolResult } from '../types.js';
|
||||
|
||||
function getWorkspacePath(): string {
|
||||
function getWorkspacePath(args?: any): string {
|
||||
const sessionPath = args?._workspacePath || args?._workspace;
|
||||
if (sessionPath) return sessionPath;
|
||||
try {
|
||||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
} catch {
|
||||
@@ -12,7 +14,29 @@ function getWorkspacePath(): string {
|
||||
}
|
||||
}
|
||||
|
||||
const IMAGE_EXTS = new Set(['.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.tiff', '.tif']);
|
||||
const IMAGE_EXTS = new Set(['.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.tiff', '.tif', '.heic', '.heif', '.avif']);
|
||||
|
||||
function runPython(script: string, timeoutMs = 60_000): Promise<any> {
|
||||
return new Promise((resolve) => {
|
||||
const child = spawn('python3', ['-c', script], {
|
||||
timeout: timeoutMs,
|
||||
env: { ...process.env },
|
||||
});
|
||||
let out = '';
|
||||
child.stdout.on('data', (d: Buffer) => { out += d.toString(); });
|
||||
child.stderr.on('data', () => {});
|
||||
child.on('close', () => {
|
||||
try { resolve(JSON.parse(out || '{}')); } catch { resolve({ error: 'Failed to parse output', raw: out.slice(0, 500) }); }
|
||||
});
|
||||
child.on('error', (e: Error) => resolve({ error: e.message }));
|
||||
});
|
||||
}
|
||||
|
||||
function buildImageMarkdown(filePath: string, workspacePath: string): string {
|
||||
const relPath = path.relative(workspacePath, filePath).replace(/\\/g, '/');
|
||||
const baseName = path.basename(filePath);
|
||||
return ``;
|
||||
}
|
||||
|
||||
// Runs OCR in a child process so the tesseract worker doesn't block the main event loop.
|
||||
const OCR_SCRIPT = `
|
||||
@@ -65,7 +89,7 @@ export const imageReadTool = {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = getWorkspacePath();
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) {
|
||||
@@ -96,11 +120,14 @@ export const imageReadTool = {
|
||||
const lines: string[] = [
|
||||
`File: ${path.basename(resolved)}`,
|
||||
`Size: ${sizeMB} MB | Format: ${ext.slice(1).toUpperCase()} | Lang: ${lang}`,
|
||||
'',
|
||||
buildImageMarkdown(resolved, workspacePath),
|
||||
];
|
||||
|
||||
if (ocrResult.error) {
|
||||
lines.push(`OCR Error: ${ocrResult.error}`);
|
||||
return { success: false, error: lines.join('\n') };
|
||||
lines.push(`OCR unavailable: ${ocrResult.error}`);
|
||||
lines.push('(Image is attached above — describe it visually.)');
|
||||
return { success: true, stdout: lines.join('\n') };
|
||||
}
|
||||
|
||||
if (!ocrResult.text) {
|
||||
@@ -124,3 +151,513 @@ export const imageReadTool = {
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
export const imagePreviewTool = {
|
||||
name: 'image_preview',
|
||||
description: 'Display an inline preview of any image file in the chat. Works with uploaded photos, screenshots, generated images, or diagrams. Returns a markdown image that renders directly in the conversation.',
|
||||
schema: {
|
||||
path: 'Path to the image file — absolute or relative to workspace',
|
||||
width: 'Max display width in pixels (optional)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
path: { type: 'string', description: 'Path to the image file' },
|
||||
width: { type: 'number', description: 'Max display width in pixels (optional)' },
|
||||
},
|
||||
required: ['path'],
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) {
|
||||
return { success: false, error: `File not found: ${resolved}` };
|
||||
}
|
||||
const ext = path.extname(resolved).toLowerCase();
|
||||
if (!IMAGE_EXTS.has(ext)) {
|
||||
return { success: false, error: `Not an image. Supported: ${[...IMAGE_EXTS].join(', ')}` };
|
||||
}
|
||||
|
||||
const stat = fs.statSync(resolved);
|
||||
const sizeMB = (stat.size / 1024 / 1024).toFixed(2);
|
||||
const widthAttr = args?.width ? ` width="${args.width}"` : '';
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stdout: `**${path.basename(resolved)}** (${sizeMB} MB)${widthAttr}\n\n${buildImageMarkdown(resolved, workspacePath)}`,
|
||||
data: { path: resolved, size: stat.size, rel_path: path.relative(workspacePath, resolved).replace(/\\/g, '/') },
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_info — metadata (dimensions, format, EXIF)
|
||||
// ---------------------------------------------------------------------------
|
||||
const INFO_SCRIPT = (imgPath: string) => `
|
||||
import json, sys
|
||||
try:
|
||||
from PIL import Image, ExifTags
|
||||
import os
|
||||
img = Image.open(${JSON.stringify(imgPath)})
|
||||
exif_data = {}
|
||||
try:
|
||||
raw = img._getexif()
|
||||
if raw:
|
||||
exif_data = {ExifTags.TAGS.get(k, str(k)): str(v) for k, v in raw.items() if k in ExifTags.TAGS}
|
||||
except Exception:
|
||||
pass
|
||||
stat = os.stat(${JSON.stringify(imgPath)})
|
||||
print(json.dumps({
|
||||
"width": img.width, "height": img.height,
|
||||
"format": img.format or "unknown", "mode": img.mode,
|
||||
"size_bytes": stat.st_size,
|
||||
"exif": exif_data,
|
||||
}))
|
||||
except Exception as e:
|
||||
print(json.dumps({"error": str(e)}))
|
||||
`;
|
||||
|
||||
export const imageInfoTool = {
|
||||
name: 'image_info',
|
||||
description: 'Get image metadata: dimensions (width × height), format, color mode, file size, and EXIF data (camera, GPS, date, etc.).',
|
||||
schema: {
|
||||
path: 'Path to the image file — absolute or relative to workspace',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
path: { type: 'string', description: 'Path to the image file' },
|
||||
},
|
||||
required: ['path'],
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
|
||||
|
||||
const result = await runPython(INFO_SCRIPT(resolved));
|
||||
if (result.error) return { success: false, error: result.error };
|
||||
|
||||
const lines: string[] = [
|
||||
`File: ${path.basename(resolved)}`,
|
||||
`Dimensions: ${result.width} × ${result.height} px`,
|
||||
`Format: ${result.format} | Mode: ${result.mode}`,
|
||||
`Size: ${(result.size_bytes / 1024).toFixed(1)} KB`,
|
||||
];
|
||||
if (result.exif && Object.keys(result.exif).length > 0) {
|
||||
lines.push('', 'EXIF:');
|
||||
const relevant = ['Make', 'Model', 'DateTime', 'DateTimeOriginal', 'GPSInfo', 'Orientation', 'Software'];
|
||||
for (const key of relevant) {
|
||||
if (result.exif[key]) lines.push(` ${key}: ${result.exif[key]}`);
|
||||
}
|
||||
}
|
||||
return { success: true, stdout: lines.join('\n'), data: result };
|
||||
},
|
||||
};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_edit — crop / resize / rotate / flip / convert / brightness /
|
||||
// contrast / grayscale / watermark / thumbnail
|
||||
// ---------------------------------------------------------------------------
|
||||
const EDIT_SCRIPT = (params: Record<string, any>) => `
|
||||
import json, sys, os
|
||||
try:
|
||||
from PIL import Image, ImageEnhance, ImageDraw, ImageFont
|
||||
import pi_heif; pi_heif.register_heif_opener()
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
from PIL import Image, ImageEnhance, ImageDraw, ImageFont
|
||||
import os, json
|
||||
|
||||
src = ${JSON.stringify(params.src)}
|
||||
dst = ${JSON.stringify(params.dst)}
|
||||
op = ${JSON.stringify(params.operation)}
|
||||
|
||||
img = Image.open(src)
|
||||
orig_format = img.format or "PNG"
|
||||
|
||||
if op == "crop":
|
||||
x, y, w, h = int(${params.x ?? 0}), int(${params.y ?? 0}), int(${params.width ?? 0}), int(${params.height ?? 0})
|
||||
img = img.crop((x, y, x + w, y + h))
|
||||
|
||||
elif op == "resize":
|
||||
w, h = int(${params.width ?? 0}), int(${params.height ?? 0})
|
||||
keep = ${params.keep_aspect ? 'True' : 'False'}
|
||||
if keep and w and h:
|
||||
img.thumbnail((w, h), Image.LANCZOS)
|
||||
elif w and h:
|
||||
img = img.resize((w, h), Image.LANCZOS)
|
||||
elif w:
|
||||
ratio = w / img.width; img = img.resize((w, int(img.height * ratio)), Image.LANCZOS)
|
||||
elif h:
|
||||
ratio = h / img.height; img = img.resize((int(img.width * ratio), h), Image.LANCZOS)
|
||||
|
||||
elif op == "rotate":
|
||||
deg = float(${params.degrees ?? 0})
|
||||
img = img.rotate(-deg, expand=True)
|
||||
|
||||
elif op == "flip":
|
||||
direction = ${JSON.stringify(params.direction ?? 'horizontal')}
|
||||
img = img.transpose(Image.FLIP_LEFT_RIGHT if direction == "horizontal" else Image.FLIP_TOP_BOTTOM)
|
||||
|
||||
elif op == "grayscale":
|
||||
img = img.convert("L").convert("RGB")
|
||||
|
||||
elif op == "brightness":
|
||||
factor = float(${params.value ?? 1.0})
|
||||
img = ImageEnhance.Brightness(img).enhance(factor)
|
||||
|
||||
elif op == "contrast":
|
||||
factor = float(${params.value ?? 1.0})
|
||||
img = ImageEnhance.Contrast(img).enhance(factor)
|
||||
|
||||
elif op == "sharpen":
|
||||
factor = float(${params.value ?? 2.0})
|
||||
img = ImageEnhance.Sharpness(img).enhance(factor)
|
||||
|
||||
elif op == "thumbnail":
|
||||
w, h = int(${params.width ?? 256}), int(${params.height ?? 256})
|
||||
img.thumbnail((w, h), Image.LANCZOS)
|
||||
|
||||
elif op == "watermark":
|
||||
text = ${JSON.stringify(params.text ?? '')}
|
||||
pos_key = ${JSON.stringify(params.position ?? 'bottom-right')}
|
||||
opacity = int(float(${params.opacity ?? 0.5}) * 255)
|
||||
draw = ImageDraw.Draw(img, "RGBA")
|
||||
try:
|
||||
has_cjk = any('가' <= c <= '힣' or '一' <= c <= '鿿' for c in text)
|
||||
if has_cjk:
|
||||
import os as _os
|
||||
_kor_fonts = [
|
||||
"/usr/share/fonts/truetype/nanum/NanumGothicBold.ttf",
|
||||
"/usr/share/fonts/truetype/nanum/NanumGothic.ttf",
|
||||
"/usr/share/fonts/opentype/noto/NotoSansCJK-Regular.ttc",
|
||||
]
|
||||
_fpath = next((f for f in _kor_fonts if _os.path.exists(f)), None)
|
||||
font = ImageFont.truetype(_fpath, size=max(20, img.width // 30)) if _fpath else ImageFont.load_default()
|
||||
else:
|
||||
font = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf", size=max(20, img.width // 30))
|
||||
except Exception:
|
||||
font = ImageFont.load_default()
|
||||
bbox = draw.textbbox((0, 0), text, font=font)
|
||||
tw, th = bbox[2] - bbox[0], bbox[3] - bbox[1]
|
||||
margin = 20
|
||||
positions = {
|
||||
"center": ((img.width - tw) // 2, (img.height - th) // 2),
|
||||
"bottom-right": (img.width - tw - margin, img.height - th - margin),
|
||||
"bottom-left": (margin, img.height - th - margin),
|
||||
"top-right": (img.width - tw - margin, margin),
|
||||
"top-left": (margin, margin),
|
||||
}
|
||||
px, py = positions.get(pos_key, positions["bottom-right"])
|
||||
draw.rectangle([px - 4, py - 4, px + tw + 4, py + th + 4], fill=(0, 0, 0, opacity // 2))
|
||||
draw.text((px, py), text, font=font, fill=(255, 255, 255, opacity))
|
||||
|
||||
elif op == "speech_bubble":
|
||||
import os as _os
|
||||
_text = ${JSON.stringify(params.text ?? '')}
|
||||
_pos = ${JSON.stringify(params.position ?? 'top-left')}
|
||||
_bg = ${JSON.stringify(params.bg_color ?? 'white')}
|
||||
_fg = ${JSON.stringify(params.text_color ?? 'black')}
|
||||
_border = ${JSON.stringify(params.border_color ?? '#333333')}
|
||||
_fsize = int(${params.font_size ?? 0}) or max(24, min(int(img.width / 22), 160))
|
||||
_pad = max(12, _fsize // 2); _tail = max(20, _fsize // 1.5); _r = max(10, _fsize // 3); _mg = max(10, _fsize // 3)
|
||||
_kor_fonts = [
|
||||
"/usr/share/fonts/truetype/nanum/NanumGothicBold.ttf",
|
||||
"/usr/share/fonts/opentype/noto/NotoSansCJK-Regular.ttc",
|
||||
]
|
||||
_fp = next((f for f in _kor_fonts if _os.path.exists(f)), None)
|
||||
_font = ImageFont.truetype(_fp, _fsize) if _fp else ImageFont.load_default()
|
||||
_lines = _text.split("\\n")
|
||||
_lh = _fsize + 6
|
||||
_tmp_draw = ImageDraw.Draw(img)
|
||||
_bw = int(max(_tmp_draw.textlength(l, font=_font) for l in _lines)) + _pad * 2
|
||||
_bh = _lh * len(_lines) + _pad * 2
|
||||
_bx = (img.width - _bw - _mg) if "right" in _pos else _mg
|
||||
_by = (img.height - _bh - _tail - _mg) if "bottom" in _pos else (_tail + _mg)
|
||||
img = img.convert("RGBA")
|
||||
_ov = Image.new("RGBA", img.size, (0, 0, 0, 0))
|
||||
_d = ImageDraw.Draw(_ov)
|
||||
_d.rounded_rectangle([_bx, _by, _bx + _bw, _by + _bh],
|
||||
radius=_r, fill=_bg, outline=_border, width=2)
|
||||
_cx = _bx + _bw // 2
|
||||
if "bottom" in _pos:
|
||||
_ty = _by + _bh
|
||||
_pts = [(_cx - 14, _ty), (_cx + 14, _ty), (_cx, _ty + _tail)]
|
||||
else:
|
||||
_ty = _by
|
||||
_pts = [(_cx - 14, _ty), (_cx + 14, _ty), (_cx, _ty - _tail)]
|
||||
_d.polygon(_pts, fill=_bg, outline=_border)
|
||||
img = Image.alpha_composite(img, _ov)
|
||||
_d2 = ImageDraw.Draw(img)
|
||||
for _i, _line in enumerate(_lines):
|
||||
_lw = _d2.textlength(_line, font=_font)
|
||||
_d2.text((_bx + (_bw - int(_lw)) // 2, _by + _pad + _i * _lh),
|
||||
_line, font=_font, fill=_fg)
|
||||
img = img.convert("RGB")
|
||||
|
||||
elif op == "stylize":
|
||||
import cv2 as _cv2
|
||||
import numpy as _np
|
||||
_style = ${JSON.stringify(params.style ?? 'anime')}
|
||||
_arr = _np.array(img.convert("RGB"))
|
||||
_bgr = _cv2.cvtColor(_arr, _cv2.COLOR_RGB2BGR)
|
||||
if _style == "sketch":
|
||||
_gray = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
|
||||
# CLAHE로 소스 대비 강화 → 밝은 사진에서도 선이 진하게 나옴
|
||||
_clahe = _cv2.createCLAHE(clipLimit=2.5, tileGridSize=(8,8))
|
||||
_gray_e = _clahe.apply(_gray)
|
||||
_sk = _cv2.divide(_gray_e, _cv2.bitwise_not(_cv2.GaussianBlur(_cv2.bitwise_not(_gray_e),(15,15),0)), scale=256.0)
|
||||
_sk_f = _sk.astype(_np.float32) / 255.0
|
||||
_sk_f = _np.power(_sk_f, 3.0)
|
||||
_lo = float(_np.percentile(_sk_f, 3))
|
||||
_sk_f = _np.clip((_sk_f - _lo) / (1.0 - _lo + 1e-6), 0.0, 1.0)
|
||||
_sk = (_sk_f * 255).astype(_np.uint8)
|
||||
_result = _cv2.cvtColor(_cv2.cvtColor(_sk, _cv2.COLOR_GRAY2BGR), _cv2.COLOR_BGR2RGB)
|
||||
elif _style == "sketch_color":
|
||||
# overlay blend: vivid color + sketch lines
|
||||
_gray = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
|
||||
_sk = _cv2.divide(_gray, _cv2.bitwise_not(_cv2.GaussianBlur(_cv2.bitwise_not(_gray),(21,21),0)), scale=256.0)
|
||||
_sk = _np.clip(_sk.astype(_np.float32)*0.85, 0, 255).astype(_np.uint8)
|
||||
_hsv = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2HSV).astype(_np.float32)
|
||||
_hsv[:,:,1] = _np.clip(_hsv[:,:,1]*2.0, 0, 255)
|
||||
_vivid = _cv2.cvtColor(_hsv.astype(_np.uint8), _cv2.COLOR_HSV2BGR)
|
||||
_vivid = _cv2.bilateralFilter(_vivid, 9, 60, 60)
|
||||
_a = _sk.astype(_np.float32)/255.0
|
||||
_b = _vivid.astype(_np.float32)/255.0
|
||||
_ov = _np.where(_a[...,_np.newaxis] < 0.5, 2*_a[...,_np.newaxis]*_b, 1-2*(1-_a[...,_np.newaxis])*(1-_b))
|
||||
_ov = _np.clip(_ov*255, 0, 255).astype(_np.uint8)
|
||||
_blend = _cv2.addWeighted(_ov, 0.4, _bgr, 0.6, 0)
|
||||
_result = _cv2.cvtColor(_blend, _cv2.COLOR_BGR2RGB)
|
||||
elif _style == "cartoon":
|
||||
# bilateral×7 (stronger flatten) + high-threshold Canny (외곽선만)
|
||||
_blur = _bgr.copy()
|
||||
for _ in range(7): _blur = _cv2.bilateralFilter(_blur, 9, 75, 75)
|
||||
_hsvc = _cv2.cvtColor(_blur, _cv2.COLOR_BGR2HSV)
|
||||
_hsvc[:,:,1] = _np.clip(_hsvc[:,:,1].astype(_np.float32)*1.5, 0, 255).astype(_np.uint8)
|
||||
_blur = _cv2.cvtColor(_hsvc, _cv2.COLOR_HSV2BGR)
|
||||
_gray = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
|
||||
# 9×9 pre-blur + high threshold → textures/clothing patterns filtered out
|
||||
_edges = _cv2.Canny(_cv2.GaussianBlur(_gray,(9,9),0), 70, 180)
|
||||
_mask = _cv2.cvtColor(_cv2.bitwise_not(_edges), _cv2.COLOR_GRAY2BGR)
|
||||
_result = _cv2.cvtColor((_blur.astype(_np.float32)*(_mask.astype(_np.float32)/255.0)).astype(_np.uint8), _cv2.COLOR_BGR2RGB)
|
||||
elif _style == "watercolor":
|
||||
# bilateral×3 (faces preserved) + median + soft edge overlay
|
||||
_wc = _bgr.copy()
|
||||
for _ in range(3): _wc = _cv2.bilateralFilter(_wc, 9, 75, 75)
|
||||
_wc = _cv2.medianBlur(_wc, 5)
|
||||
_hsvw = _cv2.cvtColor(_wc, _cv2.COLOR_BGR2HSV).astype(_np.float32)
|
||||
_hsvw[:,:,1] = _np.clip(_hsvw[:,:,1]*1.5, 0, 255)
|
||||
_hsvw[:,:,2] = _np.clip(_hsvw[:,:,2]*1.08, 0, 255)
|
||||
_wc = _cv2.cvtColor(_hsvw.astype(_np.uint8), _cv2.COLOR_HSV2BGR)
|
||||
_g3 = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
|
||||
_e3 = _cv2.Canny(_cv2.GaussianBlur(_g3,(9,9),0), 30, 100)
|
||||
_e3 = _cv2.GaussianBlur(_e3,(13,13),0) # very soft edges
|
||||
_e3a = _np.stack([_e3.astype(_np.float32)/255.0*0.22]*3, axis=-1) # 22% opacity
|
||||
_result = _cv2.cvtColor(_np.clip(_wc.astype(_np.float32)*(1-_e3a),0,255).astype(_np.uint8), _cv2.COLOR_BGR2RGB)
|
||||
else: # anime / painting / default — AnimeGANv2
|
||||
import torch as _torch
|
||||
_preset_map = {"anime":"paprika","painting":"face_paint_512_v2","celeba":"celeba_distill","anime_v1":"face_paint_512_v1","face_paint":"face_paint_512_v2"}
|
||||
_preset = _preset_map.get(_style, "paprika")
|
||||
_gen = _torch.hub.load('bryandlee/animegan2-pytorch:main','generator',pretrained=_preset,trust_repo=True)
|
||||
_f2p = _torch.hub.load('bryandlee/animegan2-pytorch:main','face2paint',trust_repo=True)
|
||||
_gen.eval()
|
||||
# 비율 보존: 긴 변 기준 768, 정사각형 패딩 후 변환, 패딩 제거
|
||||
_orig_w, _orig_h = img.size
|
||||
_scale = 768 / max(_orig_w, _orig_h)
|
||||
_rw, _rh = int(_orig_w*_scale), int(_orig_h*_scale)
|
||||
_resized = img.convert("RGB").resize((_rw, _rh), Image.LANCZOS)
|
||||
_sq = max(_rw, _rh)
|
||||
_canvas = Image.new("RGB", (_sq, _sq), (255,255,255))
|
||||
_canvas.paste(_resized, ((_sq-_rw)//2, (_sq-_rh)//2))
|
||||
_out_sq = _f2p(_gen, _canvas, size=_sq)
|
||||
# 패딩 제거
|
||||
_px, _py = (_sq-_rw)//2, (_sq-_rh)//2
|
||||
img = _out_sq.crop((_px, _py, _px+_rw, _py+_rh))
|
||||
_result = None
|
||||
if _result is not None:
|
||||
img = Image.fromarray(_result)
|
||||
|
||||
elif op == "remove_bg":
|
||||
try:
|
||||
from rembg import remove as _rembg_remove
|
||||
except ImportError:
|
||||
print(json.dumps({"error": "rembg not installed. Run: pip install rembg[cpu]"})); sys.exit(0)
|
||||
_inp = img.convert("RGBA")
|
||||
img = _rembg_remove(_inp)
|
||||
# force PNG output (preserves transparency)
|
||||
if not dst.lower().endswith(".png"):
|
||||
dst = os.path.splitext(dst)[0] + ".png"
|
||||
|
||||
elif op == "convert":
|
||||
pass # format is applied at save time
|
||||
|
||||
else:
|
||||
print(json.dumps({"error": f"Unknown operation: {op}"})); sys.exit(0)
|
||||
|
||||
# determine save format
|
||||
ext = os.path.splitext(dst)[1].lower().lstrip(".")
|
||||
fmt_map = {"jpg": "JPEG", "jpeg": "JPEG", "png": "PNG", "webp": "WEBP",
|
||||
"gif": "GIF", "bmp": "BMP", "tiff": "TIFF", "tif": "TIFF"}
|
||||
save_fmt = fmt_map.get(ext, orig_format)
|
||||
quality = int(${params.quality ?? 85})
|
||||
|
||||
if save_fmt == "JPEG" and img.mode in ("RGBA", "LA", "P"):
|
||||
img = img.convert("RGB")
|
||||
|
||||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||||
save_args = {}
|
||||
if save_fmt in ("JPEG", "WEBP"):
|
||||
save_args["quality"] = quality
|
||||
img.save(dst, format=save_fmt, **save_args)
|
||||
|
||||
stat = os.stat(dst)
|
||||
print(json.dumps({"output": dst, "width": img.width, "height": img.height,
|
||||
"format": save_fmt, "size_bytes": stat.st_size}))
|
||||
except Exception as e:
|
||||
import traceback
|
||||
print(json.dumps({"error": str(e), "trace": traceback.format_exc()[-500:]}))
|
||||
`;
|
||||
|
||||
export const imageEditTool = {
|
||||
name: 'image_edit',
|
||||
description: [
|
||||
'Edit an image file. Supported operations:',
|
||||
' crop — cut a rectangular region (x, y, width, height)',
|
||||
' resize — scale to new dimensions (width, height, keep_aspect)',
|
||||
' rotate — rotate clockwise by degrees',
|
||||
' flip — mirror horizontally or vertically (direction: horizontal|vertical)',
|
||||
' grayscale — convert to black & white',
|
||||
' brightness — adjust brightness (value: 0.5 = half, 2.0 = double)',
|
||||
' contrast — adjust contrast (value: 0.5 = half, 2.0 = double)',
|
||||
' sharpen — sharpen edges (value: 1.0 = none, 3.0 = strong)',
|
||||
' thumbnail — resize to fit within a box (width, height)',
|
||||
' watermark — overlay text (text, position, opacity)',
|
||||
' speech_bubble — add a speech bubble with tail (text, position, bg_color, text_color, border_color, font_size)',
|
||||
' stylize — convert to artistic style. Neural: anime(default)|painting|celeba|anime_v1 (AnimeGANv2). Classic: sketch|sketch_color|cartoon|watercolor',
|
||||
' remove_bg — remove background using AI (rembg); output is PNG with transparency',
|
||||
' convert — change file format (output path determines format)',
|
||||
'Returns the output file path and new dimensions.',
|
||||
].join('\n'),
|
||||
schema: {
|
||||
path: 'Input image file path (absolute or relative to workspace)',
|
||||
operation: 'Operation name: crop | resize | rotate | flip | grayscale | brightness | contrast | sharpen | thumbnail | watermark | speech_bubble | stylize | remove_bg | convert',
|
||||
output: 'Output file path (optional; defaults to <input>_<op>.<ext>)',
|
||||
x: '[crop] Left edge in pixels',
|
||||
y: '[crop] Top edge in pixels',
|
||||
width: '[crop/resize/thumbnail] Width in pixels',
|
||||
height: '[crop/resize/thumbnail] Height in pixels',
|
||||
keep_aspect: '[resize] Preserve aspect ratio (true/false, default false)',
|
||||
degrees: '[rotate] Clockwise rotation degrees (e.g. 90, 180, -90)',
|
||||
direction: '[flip] "horizontal"=좌우반전(left-right mirror) | "vertical"=상하반전(upside-down)',
|
||||
value: '[brightness/contrast/sharpen] Adjustment factor (1.0 = no change)',
|
||||
text: '[watermark/speech_bubble] Text string to overlay',
|
||||
position: '[watermark] "center"|"bottom-right"|"bottom-left"|"top-right"|"top-left" / [speech_bubble] "top-left"|"top-right"|"bottom-left"|"bottom-right"',
|
||||
opacity: '[watermark] Text opacity 0.0–1.0 (default 0.5)',
|
||||
bg_color: '[speech_bubble] Bubble background color (default "white")',
|
||||
text_color: '[speech_bubble] Text color (default "black")',
|
||||
border_color: '[speech_bubble] Border color (default "#333333")',
|
||||
font_size: '[speech_bubble] Font size in pixels (auto if omitted: image_width/22, clamped 24–160)',
|
||||
style: '[stylize] "anime"(default,paprika — 단체/풍경에 좋음) | "painting"(face_paint_v2 — 개인 인물 일러스트) | "celeba"(만화체) | "sketch" | "sketch_color" | "cartoon" | "watercolor"',
|
||||
quality: '[jpeg/webp output] Quality 1–100 (default 85)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
path: { type: 'string' },
|
||||
operation: { type: 'string', enum: ['crop','resize','rotate','flip','grayscale','brightness','contrast','sharpen','thumbnail','watermark','speech_bubble','stylize','remove_bg','convert'] },
|
||||
output: { type: 'string' },
|
||||
x: { type: 'number' },
|
||||
y: { type: 'number' },
|
||||
width: { type: 'number' },
|
||||
height: { type: 'number' },
|
||||
keep_aspect: { type: 'boolean' },
|
||||
degrees: { type: 'number' },
|
||||
direction: { type: 'string', enum: ['horizontal', 'vertical'] },
|
||||
value: { type: 'number' },
|
||||
text: { type: 'string' },
|
||||
position: { type: 'string' },
|
||||
opacity: { type: 'number' },
|
||||
bg_color: { type: 'string' },
|
||||
text_color: { type: 'string' },
|
||||
border_color: { type: 'string' },
|
||||
font_size: { type: 'number' },
|
||||
quality: { type: 'number' },
|
||||
},
|
||||
required: ['path', 'operation'],
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
const operation = String(args?.operation || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
if (!operation) return { success: false, error: 'operation is required' };
|
||||
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
|
||||
|
||||
// determine output path
|
||||
let outPath = String(args?.output || '').trim();
|
||||
if (!outPath) {
|
||||
const ext = args?.format
|
||||
? `.${args.format}`
|
||||
: (operation === 'convert' ? '.png' : path.extname(resolved));
|
||||
const base = path.basename(resolved, path.extname(resolved));
|
||||
outPath = path.join(path.dirname(resolved), `${base}_${operation}_${Date.now()}${ext}`);
|
||||
} else if (!path.isAbsolute(outPath)) {
|
||||
outPath = path.resolve(workspacePath, outPath);
|
||||
}
|
||||
|
||||
const params: Record<string, any> = {
|
||||
src: resolved,
|
||||
dst: outPath,
|
||||
operation,
|
||||
x: args?.x ?? 0,
|
||||
y: args?.y ?? 0,
|
||||
width: args?.width ?? 0,
|
||||
height: args?.height ?? 0,
|
||||
keep_aspect: args?.keep_aspect ?? false,
|
||||
degrees: args?.degrees ?? 0,
|
||||
direction: args?.direction ?? 'horizontal',
|
||||
value: args?.value ?? 1.0,
|
||||
text: args?.text ?? '',
|
||||
position: args?.position ?? 'bottom-right',
|
||||
opacity: args?.opacity ?? 0.5,
|
||||
quality: args?.quality ?? 85,
|
||||
style: args?.style ?? 'painting',
|
||||
};
|
||||
|
||||
// Neural stylize (anime/painting/celeba) downloads PyTorch models on first run (~5 min).
|
||||
// Classic filter styles (sketch/cartoon/watercolor) finish in <5s.
|
||||
const neuralStyles = new Set(['anime', 'painting', 'celeba', 'anime_v1', 'face_paint']);
|
||||
const timeoutMs = (operation === 'stylize' && neuralStyles.has(params.style)) ? 600_000 : 60_000;
|
||||
const result = await runPython(EDIT_SCRIPT(params), timeoutMs);
|
||||
if (result.error) return { success: false, error: result.error };
|
||||
|
||||
const relOut = path.relative(workspacePath, result.output).replace(/\\/g, '/');
|
||||
const sizeMB = (result.size_bytes / 1024 / 1024).toFixed(2);
|
||||
const preview = buildImageMarkdown(result.output, workspacePath);
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stdout: [
|
||||
`Operation: ${operation}`,
|
||||
`Output: ${result.output}`,
|
||||
`Dimensions: ${result.width} × ${result.height} px | Format: ${result.format} | Size: ${sizeMB} MB`,
|
||||
'',
|
||||
preview,
|
||||
].join('\n'),
|
||||
data: { ...result, rel_path: relOut },
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
@@ -0,0 +1,639 @@
|
||||
import { execFile } from 'child_process';
|
||||
import { promisify } from 'util';
|
||||
import path from 'path';
|
||||
import fs from 'fs';
|
||||
import { getConfig } from '../config/config.js';
|
||||
import { ToolResult } from '../types.js';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
function getWorkspacePath(): string {
|
||||
try {
|
||||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
} catch {
|
||||
return path.join(process.cwd(), 'workspace');
|
||||
}
|
||||
}
|
||||
|
||||
// CCITT G4 (fax-compressed) images produced by pdfimages cannot be opened
|
||||
// directly by most viewers. Wrap them in a minimal TIFF header so Pillow can
|
||||
// decode and re-save as PNG. Returns the number of files converted.
|
||||
//
|
||||
// Two known issues with naive conversion:
|
||||
// 1. PIL ignores PhotometricInterpretation=0 (WhiteIsZero) for CCITT T.6,
|
||||
// treating bit-0 as black → image is inverted. Fix: invert after open.
|
||||
// 2. Height cannot be derived from compressed data size alone (sparse pages
|
||||
// compress to near-zero bytes). Fix: use pdfimages -list to get real dims.
|
||||
const CCITT_CONVERT_PY = `
|
||||
import os, sys, struct, io, subprocess, re
|
||||
from PIL import Image, ImageOps
|
||||
|
||||
outdir = sys.argv[1]
|
||||
pdf_path = sys.argv[2] if len(sys.argv) > 2 else ''
|
||||
|
||||
# Get real image dimensions from pdfimages -list
|
||||
dim_map = {} # index -> (width, height)
|
||||
if pdf_path and os.path.exists(pdf_path):
|
||||
try:
|
||||
out = subprocess.check_output(['pdfimages', '-list', pdf_path],
|
||||
stderr=subprocess.DEVNULL, timeout=30).decode()
|
||||
for line in out.splitlines():
|
||||
m = re.match(r'\\s*(\\d+)\\s+\\S+\\s+\\S+\\s+(\\d+)\\s+(\\d+)', line)
|
||||
if m:
|
||||
idx, w, h = int(m.group(1)), int(m.group(2)), int(m.group(3))
|
||||
dim_map[idx] = (w, h)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
converted = 0
|
||||
for f in sorted(os.listdir(outdir)):
|
||||
if not f.endswith('.ccitt'):
|
||||
continue
|
||||
base = f[:-6]
|
||||
params_path = os.path.join(outdir, base + '.params')
|
||||
ccitt_path = os.path.join(outdir, f)
|
||||
png_path = os.path.join(outdir, base + '.png')
|
||||
|
||||
# Parse width from params file
|
||||
params = open(params_path).read().strip().split() if os.path.exists(params_path) else []
|
||||
width = None
|
||||
for i, p in enumerate(params):
|
||||
if p == '-X' and i + 1 < len(params):
|
||||
width = int(params[i + 1])
|
||||
width = width or 2000
|
||||
|
||||
# Get accurate height from pdfimages -list; fall back to estimation
|
||||
idx_match = re.search(r'-(\\d+)\\.ccitt$', f)
|
||||
idx = (int(idx_match.group(1)) + 1) if idx_match else -1 # pdfimages -list is 1-based
|
||||
if idx in dim_map:
|
||||
height = dim_map[idx][1]
|
||||
else:
|
||||
data_size = os.path.getsize(ccitt_path)
|
||||
height = max(100, (data_size * 8 // width) + 100)
|
||||
|
||||
try:
|
||||
data = open(ccitt_path, 'rb').read()
|
||||
strip_offset = 8 + 2 + 12 * 8 + 4
|
||||
def ifd_entry(tag, typ, count, value):
|
||||
return struct.pack('<HHII', tag, typ, count, value)
|
||||
entries = b''.join([
|
||||
ifd_entry(256, 4, 1, width),
|
||||
ifd_entry(257, 4, 1, height),
|
||||
ifd_entry(258, 3, 1, 1),
|
||||
ifd_entry(259, 3, 1, 4), # CCITT T.6
|
||||
ifd_entry(262, 3, 1, 0), # WhiteIsZero
|
||||
ifd_entry(278, 4, 1, height),
|
||||
ifd_entry(279, 4, 1, len(data)),
|
||||
ifd_entry(273, 4, 1, strip_offset),
|
||||
])
|
||||
tiff = (b'II' + struct.pack('<H', 42) + struct.pack('<I', 8)
|
||||
+ struct.pack('<H', 8) + entries + struct.pack('<I', 0) + data)
|
||||
img = Image.open(io.BytesIO(tiff))
|
||||
# PIL ignores WhiteIsZero for CCITT → invert to get correct polarity
|
||||
img = ImageOps.invert(img.convert('L'))
|
||||
img.save(png_path, 'PNG')
|
||||
os.remove(ccitt_path)
|
||||
if os.path.exists(params_path):
|
||||
os.remove(params_path)
|
||||
converted += 1
|
||||
print(f'ok:{base}.png', flush=True)
|
||||
except Exception as e:
|
||||
print(f'err:{base}:{e}', flush=True)
|
||||
print(f'done:{converted}', flush=True)
|
||||
`;
|
||||
|
||||
const FIGURES_DETECT_PY = `
|
||||
import sys, os
|
||||
import numpy as np
|
||||
import cv2
|
||||
|
||||
def detect_figures(img_path):
|
||||
img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)
|
||||
if img is None:
|
||||
return []
|
||||
h, w = img.shape
|
||||
_, binary = cv2.threshold(img, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)
|
||||
|
||||
# ── Step 1: detect horizontal lines ──
|
||||
horiz_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (w//8, 1))
|
||||
horiz_lines = cv2.morphologyEx(binary, cv2.MORPH_OPEN, horiz_kernel)
|
||||
dilate_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (40, 10))
|
||||
dilated = cv2.dilate(horiz_lines, dilate_kernel, iterations=1)
|
||||
contours, _ = cv2.findContours(dilated, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
|
||||
raw = sorted([cv2.boundingRect(c) for c in contours], key=lambda r: r[1])
|
||||
raw = [(x, y, x+bw, y+bh) for x, y, bw, bh in raw if bw > w*0.15]
|
||||
groups = []
|
||||
used = [False] * len(raw)
|
||||
for i, r1 in enumerate(raw):
|
||||
if used[i]: continue
|
||||
used[i] = True
|
||||
x1, y1, x2, y2 = r1
|
||||
for j in range(i+1, len(raw)):
|
||||
if used[j]: continue
|
||||
rx1, ry1, rx2, ry2 = raw[j]
|
||||
if ry1 - y2 > 300: continue
|
||||
overlap = max(0, min(x2, rx2) - max(x1, rx1))
|
||||
min_w = min(x2-x1, rx2-rx1)
|
||||
x_gap = max(0, max(x1, rx1) - min(x2, rx2))
|
||||
if min_w > 0 and (overlap / min_w >= 0.5 or x_gap <= 80):
|
||||
used[j] = True
|
||||
x1 = min(x1, rx1); y1 = min(y1, ry1)
|
||||
x2 = max(x2, rx2); y2 = max(y2, ry2)
|
||||
groups.append((x1, y1, x2, y2))
|
||||
|
||||
# ── Step 2: expand vertically and validate ──
|
||||
results = []
|
||||
for gx1_raw, gy1, gx2_raw, gy2 in groups:
|
||||
# Re-compute x bounds using median of per-row line extents to avoid
|
||||
# full-width separator lines bleeding into adjacent text columns.
|
||||
row_x1s, row_x2s = [], []
|
||||
for row in range(gy1, gy2 + 1):
|
||||
nz = np.where(horiz_lines[row, :] > 0)[0]
|
||||
if len(nz) > w * 0.10:
|
||||
row_x1s.append(int(nz[0]))
|
||||
row_x2s.append(int(nz[-1]))
|
||||
if len(row_x1s) >= 2:
|
||||
# Filter out full-width separator lines (span >85% of page) before computing bounds
|
||||
pairs = [(lx1, lx2) for lx1, lx2 in zip(row_x1s, row_x2s) if (lx2 - lx1) < w * 0.85]
|
||||
if pairs:
|
||||
gx1 = min(p[0] for p in pairs)
|
||||
gx2 = max(p[1] for p in pairs)
|
||||
else:
|
||||
gx1, gx2 = gx1_raw, gx2_raw # all lines are separators — use raw
|
||||
else:
|
||||
gx1, gx2 = gx1_raw, gx2_raw
|
||||
col_width = gx2 - gx1
|
||||
col_horiz_in = horiz_lines[gy1:gy2+1, gx1:gx2]
|
||||
line_per_row = (col_horiz_in > 0).sum(axis=1)
|
||||
actual = np.where(line_per_row > col_width * 0.5)[0]
|
||||
if len(actual) == 0:
|
||||
continue
|
||||
# Reject if detected lines are too far apart (text separators, not table/chart)
|
||||
first_line = int(actual[0]) + gy1
|
||||
last_line = int(actual[-1]) + gy1
|
||||
# If lines are far apart, this is likely text separators not a table/chart:
|
||||
# suppress expansion and require taller minimum height.
|
||||
large_gap = len(actual) >= 2 and (int(actual[-1]) - int(actual[0])) > 60
|
||||
if large_gap:
|
||||
top = first_line - 10
|
||||
if len(actual) >= 3:
|
||||
# 3+ separators = multi-section table: expand to capture rows below last separator
|
||||
bottom = last_line
|
||||
prev_blank = 0
|
||||
for row in range(last_line + 1, min(h, last_line + 300)):
|
||||
if binary[row, gx1:gx2].sum() == 0:
|
||||
prev_blank += 1
|
||||
if prev_blank > 30: break
|
||||
else:
|
||||
prev_blank = 0
|
||||
bottom = row
|
||||
else:
|
||||
# 1-2 lines = figure border / panel divider: minimal expansion
|
||||
bottom = last_line + 10
|
||||
else:
|
||||
top = first_line
|
||||
prev_blank = 0
|
||||
for row in range(first_line - 1, max(0, first_line - 500), -1):
|
||||
if binary[row, gx1:gx2].sum() == 0:
|
||||
prev_blank += 1
|
||||
if prev_blank > 50: break
|
||||
else:
|
||||
prev_blank = 0
|
||||
top = row
|
||||
bottom = last_line
|
||||
prev_blank = 0
|
||||
for row in range(last_line + 1, min(h, last_line + 400)):
|
||||
if binary[row, gx1:gx2].sum() == 0:
|
||||
prev_blank += 1
|
||||
if prev_blank > 50: break
|
||||
else:
|
||||
prev_blank = 0
|
||||
bottom = row
|
||||
margin = 20
|
||||
x1 = max(0, gx1-margin)
|
||||
y1 = max(0, top-margin)
|
||||
x2 = min(w, gx2+margin)
|
||||
y2 = min(h, bottom+margin)
|
||||
box_w = x2 - x1
|
||||
box_h = y2 - y1
|
||||
|
||||
# ── Reject obvious non-figures ──
|
||||
min_h = 120 if large_gap else 80
|
||||
if box_h < min_h or box_w < 100:
|
||||
continue
|
||||
# 2. Too tall — likely grabbed a whole text column
|
||||
if box_h > h * 0.75:
|
||||
continue
|
||||
# 3. Text density: if >60% of pixels are black, it's probably a text block
|
||||
roi = binary[y1:y2, x1:x2]
|
||||
if roi.size == 0:
|
||||
continue
|
||||
density = roi.sum() / (roi.size * 255)
|
||||
if density > 0.60:
|
||||
continue
|
||||
# 4. Aspect ratio: extremely wide+short bars are headers
|
||||
if box_h < box_w * 0.10:
|
||||
continue
|
||||
# 5. Full-width shallow bar — page header/footer banner
|
||||
if box_w > w * 0.85 and box_h < 160:
|
||||
continue
|
||||
# 6. Thin separator line expanded into text: original group was tiny but grew huge
|
||||
original_h = gy2 - gy1
|
||||
if original_h < 30 and box_h > original_h * 5:
|
||||
continue
|
||||
|
||||
results.append([x1, y1, x2, y2])
|
||||
|
||||
# Merge overlapping boxes
|
||||
results.sort(key=lambda r: (r[1], r[0]))
|
||||
merged = []
|
||||
used = [False] * len(results)
|
||||
for i, r1 in enumerate(results):
|
||||
if used[i]: continue
|
||||
x1, y1, x2, y2 = r1
|
||||
for j in range(i+1, len(results)):
|
||||
if used[j]: continue
|
||||
rx1, ry1, rx2, ry2 = results[j]
|
||||
ix1, iy1 = max(x1, rx1), max(y1, ry1)
|
||||
ix2, iy2 = min(x2, rx2), min(y2, ry2)
|
||||
if ix2 > ix1 and iy2 > iy1:
|
||||
inter = (ix2-ix1) * (iy2-iy1)
|
||||
smaller = min((x2-x1)*(y2-y1), (rx2-rx1)*(ry2-ry1))
|
||||
if smaller > 0 and inter / smaller > 0.3:
|
||||
used[j] = True
|
||||
x1 = min(x1, rx1); y1 = min(y1, ry1)
|
||||
x2 = max(x2, rx2); y2 = max(y2, ry2)
|
||||
merged.append([x1, y1, x2, y2])
|
||||
return merged
|
||||
|
||||
page_dir = sys.argv[1]
|
||||
out_dir = sys.argv[2]
|
||||
os.makedirs(out_dir, exist_ok=True)
|
||||
page_files = sorted(f for f in os.listdir(page_dir) if f.startswith('page-') and f.endswith('.png'))
|
||||
total = 0
|
||||
for page_f in page_files:
|
||||
page_num = page_f[5:-4]
|
||||
img_path = os.path.join(page_dir, page_f)
|
||||
boxes = detect_figures(img_path)
|
||||
if not boxes:
|
||||
continue
|
||||
img_color = cv2.imread(img_path)
|
||||
if img_color is None:
|
||||
continue
|
||||
for i, (x1, y1, x2, y2) in enumerate(boxes, 1):
|
||||
crop = img_color[y1:y2, x1:x2]
|
||||
if crop.size == 0:
|
||||
continue
|
||||
out_name = f'figure_p{page_num}_{i}.png'
|
||||
cv2.imwrite(os.path.join(out_dir, out_name), crop)
|
||||
print(f'ok:{out_name}', flush=True)
|
||||
total += 1
|
||||
print(f'done:{total}', flush=True)
|
||||
`;
|
||||
|
||||
async function convertCcittToPng(outDir: string, pdfPath: string): Promise<{ converted: number; errors: string[] }> {
|
||||
const errors: string[] = [];
|
||||
let converted = 0;
|
||||
try {
|
||||
const { stdout } = await execFileAsync('python3', ['-c', CCITT_CONVERT_PY, outDir, pdfPath], { timeout: 60_000 });
|
||||
for (const line of stdout.split('\n')) {
|
||||
if (line.startsWith('done:')) converted = parseInt(line.slice(5), 10) || 0;
|
||||
else if (line.startsWith('err:')) errors.push(line.slice(4));
|
||||
}
|
||||
} catch (err: any) {
|
||||
errors.push(`ccitt_convert: ${String(err.message || err).slice(0, 200)}`);
|
||||
}
|
||||
return { converted, errors };
|
||||
}
|
||||
|
||||
async function extractFigures(pageDir: string, outDir: string): Promise<{ files: string[]; errors: string[] }> {
|
||||
const files: string[] = [];
|
||||
const errors: string[] = [];
|
||||
try {
|
||||
const { stdout } = await execFileAsync('python3', ['-c', FIGURES_DETECT_PY, pageDir, outDir], { timeout: 120_000 });
|
||||
for (const line of stdout.split('\n')) {
|
||||
if (line.startsWith('ok:')) files.push(line.slice(3).trim());
|
||||
else if (line.startsWith('err:')) errors.push(line.slice(4));
|
||||
}
|
||||
} catch (err: any) {
|
||||
errors.push(`figures_detect: ${String(err.message || err).slice(0, 200)}`);
|
||||
}
|
||||
return { files, errors };
|
||||
}
|
||||
|
||||
export const pdfExtractImagesTool = {
|
||||
name: 'pdf_extract_images',
|
||||
description: 'Extract images and diagrams from a PDF file. mode "images" extracts embedded raster images. mode "figures" auto-detects and crops figures/tables using OpenCV line detection. mode "both" does images+figures (default). Use out_dir to save directly into a PPTX project folder.',
|
||||
schema: {
|
||||
path: 'Path to the PDF file (relative to workspace or absolute)',
|
||||
mode: '"images" (embedded rasters), "figures" (auto-crop figures/tables via OpenCV), or "both" = images+figures (default)',
|
||||
out_dir: 'Output folder name relative to workspace (default: uploads/{basename}-images). Use the PPTX project folder name to save images there directly.',
|
||||
page_from: 'First page (1-indexed, default: 1)',
|
||||
page_to: 'Last page (inclusive, default: last page)',
|
||||
dpi: 'Resolution for page rendering in DPI (default: 150)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
required: ['path'],
|
||||
properties: {
|
||||
path: { type: 'string', description: 'Path to the PDF file (relative to workspace or absolute)' },
|
||||
mode: { type: 'string', enum: ['images', 'figures', 'both'], description: 'Extraction mode: images=embedded rasters, figures=OpenCV-cropped figures/tables, both=images+figures (default)' },
|
||||
out_dir: { type: 'string', description: 'Output folder relative to workspace (default: uploads/{basename}-images). Set to PPTX project folder name to save images there directly.' },
|
||||
page_from: { type: 'number', description: 'First page (1-indexed)' },
|
||||
page_to: { type: 'number', description: 'Last page (inclusive)' },
|
||||
dpi: { type: 'number', description: 'Rendering DPI for pages mode (default: 150)' },
|
||||
},
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
|
||||
if (path.extname(resolved).toLowerCase() !== '.pdf') return { success: false, error: 'File must have a .pdf extension' };
|
||||
|
||||
const mode = String(args?.mode || 'both');
|
||||
const dpi = Math.min(300, Math.max(72, Number(args?.dpi ?? 150)));
|
||||
const pageFrom = args?.page_from ? Math.max(1, Math.floor(Number(args.page_from))) : null;
|
||||
const pageTo = args?.page_to ? Math.max(1, Math.floor(Number(args.page_to))) : null;
|
||||
|
||||
const basename = path.basename(resolved, '.pdf').replace(/[^a-zA-Z0-9가-힣._-]/g, '_');
|
||||
const dirBasename = basename.slice(0, 35);
|
||||
const customOutDir = args?.out_dir ? path.normalize(String(args.out_dir).trim()).replace(/\/+$/, '') : null;
|
||||
const outDir = customOutDir
|
||||
? (path.isAbsolute(customOutDir) ? customOutDir : path.join(workspacePath, customOutDir))
|
||||
: path.join(workspacePath, 'uploads', `${dirBasename}-images`);
|
||||
const outDirRel = customOutDir
|
||||
? (path.isAbsolute(customOutDir) ? path.relative(workspacePath, customOutDir) : customOutDir)
|
||||
: `uploads/${dirBasename}-images`;
|
||||
fs.mkdirSync(outDir, { recursive: true });
|
||||
|
||||
const extracted: string[] = [];
|
||||
const errors: string[] = [];
|
||||
|
||||
// --- pdfimages: extract embedded raster images ---
|
||||
if (mode === 'images' || mode === 'both') {
|
||||
const imgArgs = ['-all'];
|
||||
if (pageFrom) imgArgs.push('-f', String(pageFrom));
|
||||
if (pageTo) imgArgs.push('-l', String(pageTo));
|
||||
imgArgs.push(resolved, path.join(outDir, 'img'));
|
||||
try {
|
||||
await execFileAsync('pdfimages', imgArgs, { timeout: 60_000 });
|
||||
// Convert any CCITT G4 files to PNG before collecting results
|
||||
const ccittFiles = fs.readdirSync(outDir).filter(f => /^img-\d+\.ccitt$/i.test(f));
|
||||
if (ccittFiles.length > 0) {
|
||||
const { errors: ccittErrors } = await convertCcittToPng(outDir, resolved);
|
||||
errors.push(...ccittErrors);
|
||||
}
|
||||
const files = fs.readdirSync(outDir)
|
||||
.filter(f => /^img-\d+\.(jpg|jpeg|png|ppm|pbm|tif|tiff)$/i.test(f))
|
||||
.sort();
|
||||
extracted.push(...files.map(f => `${outDirRel}/${f}`));
|
||||
} catch (err: any) {
|
||||
errors.push(`pdfimages: ${String(err.message || err).slice(0, 200)}`);
|
||||
}
|
||||
}
|
||||
|
||||
// --- figures mode: render pages then auto-crop figures/tables with OpenCV ---
|
||||
if (mode === 'figures' || mode === 'both') {
|
||||
const figDpi = Math.min(300, Math.max(100, dpi));
|
||||
const ppmArgs = ['-png', '-r', String(figDpi)];
|
||||
if (pageFrom) ppmArgs.push('-f', String(pageFrom));
|
||||
if (pageTo) ppmArgs.push('-l', String(pageTo));
|
||||
ppmArgs.push(resolved, path.join(outDir, 'page'));
|
||||
try {
|
||||
await execFileAsync('pdftoppm', ppmArgs, { timeout: 120_000 });
|
||||
} catch (err: any) {
|
||||
errors.push(`pdftoppm: ${String(err.message || err).slice(0, 200)}`);
|
||||
}
|
||||
const { files: figFiles, errors: figErrors } = await extractFigures(outDir, outDir);
|
||||
errors.push(...figErrors);
|
||||
for (const f of fs.readdirSync(outDir).filter(f => f.startsWith('page-') && f.endsWith('.png'))) {
|
||||
try { fs.unlinkSync(path.join(outDir, f)); } catch {}
|
||||
}
|
||||
extracted.push(...figFiles.map(f => `${outDirRel}/${f}`));
|
||||
}
|
||||
|
||||
if (extracted.length === 0) {
|
||||
const errMsg = errors.length ? errors.join('; ') : 'No images found in the PDF';
|
||||
return { success: false, error: errMsg };
|
||||
}
|
||||
|
||||
const links = extracted.map(f => `[${path.basename(f)}](/api/files/${f})`).join('\n');
|
||||
return {
|
||||
success: true,
|
||||
stdout: `${extracted.length}개 파일 추출 완료:\n${links}`,
|
||||
data: { files: extracted, outDir, errors: errors.length ? errors : undefined },
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// pdf_extract_tables
|
||||
// 1차: PyMuPDF find_tables() — 선 기반 테이블, 빠름
|
||||
// 2차: pdfplumber — 공백/선 혼합, 선 없는 테이블에도 강함
|
||||
// ---------------------------------------------------------------------------
|
||||
const TABLE_EXTRACT_PY = `
|
||||
import sys, json, os, io
|
||||
|
||||
# PyMuPDF 1.24+ 가 import 시점에 C 레벨 fd=1(stdout)로 직접
|
||||
# "Consider using pymupdf_layout..." 를 출력해 JSON 파싱을 깨뜨린다.
|
||||
# sys.stdout 교체로는 막을 수 없으므로 os.dup2 로 fd 1 자체를 /dev/null 로 리다이렉트한다.
|
||||
_devnull_fd = os.open(os.devnull, os.O_WRONLY)
|
||||
_saved_fd = os.dup(1)
|
||||
os.dup2(_devnull_fd, 1)
|
||||
os.close(_devnull_fd)
|
||||
try:
|
||||
import fitz as _fitz
|
||||
except ImportError:
|
||||
_fitz = None
|
||||
finally:
|
||||
os.dup2(_saved_fd, 1) # stdout 복구
|
||||
os.close(_saved_fd)
|
||||
|
||||
pdf_path = sys.argv[1]
|
||||
fmt = sys.argv[2] if len(sys.argv) > 2 else 'markdown'
|
||||
page_from = int(sys.argv[3]) - 1 if len(sys.argv) > 3 else 0 # 0-indexed
|
||||
page_to = int(sys.argv[4]) - 1 if len(sys.argv) > 4 else None # inclusive, 0-indexed
|
||||
engine = sys.argv[5] if len(sys.argv) > 5 else 'auto' # auto|pymupdf|pdfplumber
|
||||
|
||||
def cell(v):
|
||||
return str(v).replace('\\n', ' ').strip() if v is not None else ''
|
||||
|
||||
def to_markdown(rows):
|
||||
if not rows or not rows[0]:
|
||||
return ''
|
||||
widths = [max(len(cell(r[i])) for r in rows if i < len(r)) for i in range(len(rows[0]))]
|
||||
widths = [max(w, 3) for w in widths]
|
||||
def row_str(r):
|
||||
return '| ' + ' | '.join(cell(r[i]).ljust(widths[i]) if i < len(r) else ' ' * widths[i] for i in range(len(widths))) + ' |'
|
||||
sep = '| ' + ' | '.join('-' * w for w in widths) + ' |'
|
||||
lines = [row_str(rows[0]), sep] + [row_str(r) for r in rows[1:]]
|
||||
return '\\n'.join(lines)
|
||||
|
||||
def to_csv(rows):
|
||||
import csv, io
|
||||
buf = io.StringIO()
|
||||
w = csv.writer(buf)
|
||||
for r in rows:
|
||||
w.writerow([cell(v) for v in r])
|
||||
return buf.getvalue().rstrip()
|
||||
|
||||
def format_table(rows, fmt):
|
||||
if fmt == 'csv': return to_csv(rows)
|
||||
if fmt == 'json': return json.dumps([[cell(v) for v in r] for r in rows], ensure_ascii=False)
|
||||
return to_markdown(rows)
|
||||
|
||||
results = []
|
||||
errors = []
|
||||
|
||||
def try_pymupdf():
|
||||
if _fitz is None:
|
||||
raise ImportError('PyMuPDF(fitz) not available')
|
||||
doc = _fitz.open(pdf_path)
|
||||
end = page_to if page_to is not None else len(doc) - 1
|
||||
found = []
|
||||
for pno in range(page_from, min(end + 1, len(doc))):
|
||||
page = doc[pno]
|
||||
tabs = page.find_tables().tables # TableFinder → .tables 리스트
|
||||
for ti, tab in enumerate(tabs):
|
||||
rows = tab.extract()
|
||||
if not rows: continue
|
||||
found.append({'page': pno + 1, 'table': ti + 1, 'rows': len(rows),
|
||||
'cols': len(rows[0]) if rows else 0,
|
||||
'data': format_table(rows, fmt)})
|
||||
doc.close()
|
||||
return found
|
||||
|
||||
def try_pdfplumber():
|
||||
import pdfplumber
|
||||
found = []
|
||||
with pdfplumber.open(pdf_path) as pdf:
|
||||
end = page_to if page_to is not None else len(pdf.pages) - 1
|
||||
for pno in range(page_from, min(end + 1, len(pdf.pages))):
|
||||
page = pdf.pages[pno]
|
||||
tables = page.extract_tables({
|
||||
'vertical_strategy': 'lines_strict',
|
||||
'horizontal_strategy': 'lines_strict',
|
||||
})
|
||||
# 선 감지 실패 시 text 기반으로 재시도
|
||||
if not tables:
|
||||
tables = page.extract_tables({
|
||||
'vertical_strategy': 'text',
|
||||
'horizontal_strategy': 'text',
|
||||
'snap_tolerance': 3,
|
||||
'join_tolerance': 3,
|
||||
'edge_min_length': 10,
|
||||
})
|
||||
for ti, rows in enumerate(tables):
|
||||
if not rows: continue
|
||||
found.append({'page': pno + 1, 'table': ti + 1, 'rows': len(rows),
|
||||
'cols': len(rows[0]) if rows else 0,
|
||||
'data': format_table(rows, fmt)})
|
||||
return found
|
||||
|
||||
try:
|
||||
if engine == 'pymupdf':
|
||||
results = try_pymupdf()
|
||||
elif engine == 'pdfplumber':
|
||||
results = try_pdfplumber()
|
||||
else: # auto: pymupdf first, pdfplumber if no tables found
|
||||
results = try_pymupdf()
|
||||
if not results:
|
||||
results = try_pdfplumber()
|
||||
if results:
|
||||
for r in results: r['engine'] = 'pdfplumber'
|
||||
else:
|
||||
errors.append('no_tables')
|
||||
else:
|
||||
for r in results: r['engine'] = 'pymupdf'
|
||||
except Exception as e:
|
||||
import traceback
|
||||
errors.append(str(e))
|
||||
errors.append(traceback.format_exc()[-600:])
|
||||
|
||||
print(json.dumps({'tables': results, 'errors': errors}, ensure_ascii=False))
|
||||
`;
|
||||
|
||||
export const pdfExtractTablesTool = {
|
||||
name: 'pdf_extract_tables',
|
||||
description: [
|
||||
'Extract tables from a PDF file as structured data.',
|
||||
'Uses PyMuPDF find_tables() first (fast, line-based); falls back to pdfplumber (handles borderless tables too).',
|
||||
'format: "markdown" (default) | "csv" | "json".',
|
||||
'engine: "auto" (default) | "pymupdf" | "pdfplumber".',
|
||||
'For scanned/image-based PDFs use pdf_extract_images with mode "figures" instead.',
|
||||
].join(' '),
|
||||
schema: {
|
||||
path: 'Path to the PDF file (absolute or relative to workspace)',
|
||||
format: 'Output format: "markdown" (default) | "csv" | "json"',
|
||||
engine: 'Extraction engine: "auto" (default) | "pymupdf" | "pdfplumber"',
|
||||
page_from: 'First page to scan, 1-indexed (default: 1)',
|
||||
page_to: 'Last page to scan, inclusive (default: last page)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
path: { type: 'string' },
|
||||
format: { type: 'string', enum: ['markdown', 'csv', 'json'] },
|
||||
engine: { type: 'string', enum: ['auto', 'pymupdf', 'pdfplumber'] },
|
||||
page_from: { type: 'number' },
|
||||
page_to: { type: 'number' },
|
||||
},
|
||||
required: ['path'],
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
|
||||
if (path.extname(resolved).toLowerCase() !== '.pdf') return { success: false, error: 'File must have a .pdf extension' };
|
||||
|
||||
const fmt = ['markdown', 'csv', 'json'].includes(args?.format) ? String(args.format) : 'markdown';
|
||||
const engine = ['auto', 'pymupdf', 'pdfplumber'].includes(args?.engine) ? String(args.engine) : 'auto';
|
||||
const pageFrom = args?.page_from ? String(Math.max(1, Math.floor(Number(args.page_from)))) : '1';
|
||||
const pageTo = args?.page_to ? String(Math.max(1, Math.floor(Number(args.page_to)))) : '9999';
|
||||
|
||||
let raw: { tables: any[]; errors: string[] };
|
||||
try {
|
||||
const { stdout } = await execFileAsync(
|
||||
'python3', ['-c', TABLE_EXTRACT_PY, resolved, fmt, pageFrom, pageTo, engine],
|
||||
{ timeout: 60_000, maxBuffer: 20 * 1024 * 1024 },
|
||||
);
|
||||
// PyMuPDF가 JSON 앞에 경고 텍스트를 stdout으로 출력할 수 있으므로
|
||||
// 첫 번째 '{' 이후만 JSON으로 파싱한다.
|
||||
const jsonStart = stdout.indexOf('{');
|
||||
raw = JSON.parse(jsonStart >= 0 ? stdout.slice(jsonStart) : stdout);
|
||||
} catch (err: any) {
|
||||
return { success: false, error: `Table extraction failed: ${String(err.message || err).slice(0, 300)}` };
|
||||
}
|
||||
|
||||
if (!raw.tables || raw.tables.length === 0) {
|
||||
const hint = raw.errors?.includes('no_tables')
|
||||
? 'No tables detected. If this is a scanned PDF, try pdf_extract_images with mode "figures".'
|
||||
: `No tables found. Errors: ${raw.errors?.join('; ') || 'none'}`;
|
||||
return { success: false, error: hint };
|
||||
}
|
||||
|
||||
const lines: string[] = [];
|
||||
for (const t of raw.tables) {
|
||||
lines.push(`### Page ${t.page} — Table ${t.table} (${t.rows} rows × ${t.cols} cols, engine: ${t.engine ?? engine})`);
|
||||
lines.push('');
|
||||
lines.push(t.data);
|
||||
lines.push('');
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stdout: lines.join('\n').trimEnd(),
|
||||
data: { table_count: raw.tables.length, tables: raw.tables, errors: raw.errors?.length ? raw.errors : undefined },
|
||||
};
|
||||
},
|
||||
};
|
||||
+4
-2
@@ -7,7 +7,9 @@ import { ToolResult } from '../types.js';
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
function getWorkspacePath(): string {
|
||||
function getWorkspacePath(args?: any): string {
|
||||
const sessionPath = args?._workspacePath || args?._workspace;
|
||||
if (sessionPath) return sessionPath;
|
||||
try {
|
||||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
} catch {
|
||||
@@ -39,7 +41,7 @@ export const pdfReadTool = {
|
||||
const filePath = String(args?.path || '').trim();
|
||||
if (!filePath) return { success: false, error: 'path is required' };
|
||||
|
||||
const workspacePath = getWorkspacePath();
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||||
|
||||
if (!fs.existsSync(resolved)) {
|
||||
|
||||
+12
-3
@@ -598,7 +598,7 @@ export const pptxTool: import('./registry.js').Tool = {
|
||||
|
||||
const configMgr = getConfig();
|
||||
const config = configMgr.getConfig() as any;
|
||||
const workspacePath = args._workspacePath || config.workspace?.path || process.cwd();
|
||||
const workspacePath = args?._workspacePath || args?._workspace || config.workspace?.path || process.cwd();
|
||||
|
||||
// Inject config defaults into spec so Python engine can use them
|
||||
if (!spec.template && config.ppt?.template) spec.template = config.ppt.template;
|
||||
@@ -676,8 +676,17 @@ export const pptxTool: import('./registry.js').Tool = {
|
||||
let found: string | null = null;
|
||||
if (fs.existsSync(uploadsDir)) {
|
||||
for (const sub of fs.readdirSync(uploadsDir)) {
|
||||
const candidate = path.join(uploadsDir, sub, basename);
|
||||
const subDir = path.join(uploadsDir, sub);
|
||||
if (!fs.existsSync(subDir) || !fs.statSync(subDir).isDirectory()) continue;
|
||||
// Exact match
|
||||
const candidate = path.join(subDir, basename);
|
||||
if (fs.existsSync(candidate)) { found = candidate; break; }
|
||||
// Prefix match: "figure_p05" → "figure_p05_1.png"
|
||||
const prefix = basename.replace(/\.[^.]+$/, '');
|
||||
const prefixMatch = fs.readdirSync(subDir).find(f =>
|
||||
f.startsWith(prefix + '_') || f.startsWith(prefix + '.')
|
||||
);
|
||||
if (prefixMatch) { found = path.join(subDir, prefixMatch); break; }
|
||||
}
|
||||
}
|
||||
if (found) {
|
||||
@@ -814,7 +823,7 @@ export const editPptxTool: import('./registry.js').Tool = {
|
||||
|
||||
const configMgr = getConfig();
|
||||
const config = configMgr.getConfig() as any;
|
||||
const workspacePath = args._workspacePath || config.workspace?.path || process.cwd();
|
||||
const workspacePath = args?._workspacePath || args?._workspace || config.workspace?.path || process.cwd();
|
||||
|
||||
// Resolve absolute path
|
||||
const absPath = path.isAbsolute(existingPath)
|
||||
|
||||
+2
-2
@@ -355,7 +355,7 @@ export const pubmedFulltextTool = {
|
||||
|
||||
// Download PDF
|
||||
const config = getConfig().getConfig();
|
||||
const workspaceDir = args?._workspacePath || config.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
const workspaceDir = args?._workspacePath || args?._workspace || config.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
const pdfDir = path.join(workspaceDir, 'pubmed');
|
||||
fs.mkdirSync(pdfDir, { recursive: true });
|
||||
|
||||
@@ -472,7 +472,7 @@ export const pubmedFulltextTool = {
|
||||
|
||||
// Save to workspace (prefer per-user path injected by v2 executeTool)
|
||||
const config = getConfig().getConfig();
|
||||
const workspaceDir = args?._workspacePath || config.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
const workspaceDir = args?._workspacePath || args?._workspace || config.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
const savePath = args?.save_path
|
||||
? path.join(workspaceDir, args.save_path)
|
||||
: path.join(workspaceDir, 'pubmed', `${pmcid}.txt`);
|
||||
|
||||
+47
-2
@@ -1,9 +1,13 @@
|
||||
import { spawn } from 'child_process';
|
||||
import path from 'path';
|
||||
import fs from 'fs';
|
||||
import { getConfig } from '../config/config.js';
|
||||
import { ToolResult } from '../types.js';
|
||||
|
||||
function getWorkspacePath(): string {
|
||||
function getWorkspacePath(args?: any): string {
|
||||
// Prefer the session-specific workspace passed by executeTool
|
||||
const sessionPath = args?._workspacePath || args?._workspace;
|
||||
if (sessionPath) return sessionPath;
|
||||
try {
|
||||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
} catch {
|
||||
@@ -53,7 +57,21 @@ export const pythonEvalTool = {
|
||||
}
|
||||
|
||||
const timeoutSec = Math.min(60, Math.max(1, Number(args?.timeout ?? 15)));
|
||||
const workspacePath = getWorkspacePath();
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
|
||||
// Snapshot files before execution
|
||||
const beforeFiles = new Set<string>();
|
||||
try {
|
||||
function scanDir(dir: string, prefix: string) {
|
||||
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
||||
const rel = prefix ? `${prefix}/${entry.name}` : entry.name;
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) { scanDir(full, rel); }
|
||||
else if (entry.isFile()) { try { beforeFiles.add(`${rel}:${fs.statSync(full).mtimeMs}`); } catch {} }
|
||||
}
|
||||
}
|
||||
scanDir(workspacePath, '');
|
||||
} catch {}
|
||||
|
||||
if (args?.packages) {
|
||||
const pkgs = String(args.packages).split(',').map((p: string) => p.trim()).filter((p: string) => /^[a-zA-Z0-9_\-\.]+$/.test(p));
|
||||
@@ -108,10 +126,37 @@ export const pythonEvalTool = {
|
||||
return { success: false, error: `Execution timed out after ${timeoutSec}s` };
|
||||
}
|
||||
|
||||
// Detect newly created files
|
||||
const newFiles: string[] = [];
|
||||
try {
|
||||
function scanDirAfter(dir: string, prefix: string) {
|
||||
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
||||
const rel = prefix ? `${prefix}/${entry.name}` : entry.name;
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) { scanDirAfter(full, rel); }
|
||||
else if (entry.isFile()) {
|
||||
const key = `${rel}:${fs.statSync(full).mtimeMs}`;
|
||||
if (!beforeFiles.has(key)) { newFiles.push(rel); }
|
||||
}
|
||||
}
|
||||
}
|
||||
scanDirAfter(workspacePath, '');
|
||||
} catch {}
|
||||
|
||||
const parts: string[] = [];
|
||||
if (result.stdout.trim()) parts.push(`[stdout]\n${result.stdout.trim().slice(0, 20_000)}`);
|
||||
if (result.stderr.trim()) parts.push(`[stderr]\n${result.stderr.trim().slice(0, 3_000)}`);
|
||||
if (!result.stdout.trim() && !result.stderr.trim()) parts.push('(no output)');
|
||||
if (newFiles.length > 0) {
|
||||
const links = newFiles.map(f => {
|
||||
const ext = path.extname(f).toLowerCase();
|
||||
const isImage = ['.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.tiff', '.tif'].includes(ext);
|
||||
return isImage
|
||||
? ``
|
||||
: `[${path.basename(f)}](/api/files/${f})`;
|
||||
}).join('\n');
|
||||
parts.push(`[생성된 파일]\n${links}`);
|
||||
}
|
||||
parts.push(`[exit ${result.exitCode}]`);
|
||||
|
||||
return {
|
||||
|
||||
+10
-2
@@ -15,8 +15,8 @@ import { pptxTool, editPptxTool } from './pptx.js';
|
||||
import { pubmedSearchTool, pubmedFetchTool, pubmedFulltextTool } from './pubmed.js';
|
||||
import { openalexSearchTool, semanticSearchTool } from './scholar.js';
|
||||
import { pdfReadTool } from './pdf.js';
|
||||
import { pdfExtractImagesTool } from './pdf-extract.js';
|
||||
import { imageReadTool } from './image.js';
|
||||
import { pdfExtractImagesTool, pdfExtractTablesTool } from './pdf-extract.js';
|
||||
import { imageReadTool, imagePreviewTool, imageInfoTool, imageEditTool } from './image.js';
|
||||
import { audioTranscribeTool } from './audio-transcribe.js';
|
||||
import { pythonEvalTool } from './python.js';
|
||||
import { sqliteTool } from './sqlite.js';
|
||||
@@ -58,6 +58,9 @@ const TOOL_PROFILE_TOOL_NAMES: Record<Exclude<ToolProfile, 'full'>, ReadonlySet<
|
||||
'memory_write',
|
||||
'pdf_read',
|
||||
'image_read',
|
||||
'image_preview',
|
||||
'image_info',
|
||||
'image_edit',
|
||||
'audio_transcribe',
|
||||
'python_eval',
|
||||
'sqlite_query',
|
||||
@@ -74,6 +77,7 @@ const TOOL_PROFILE_TOOL_NAMES: Record<Exclude<ToolProfile, 'full'>, ReadonlySet<
|
||||
'semantic_search',
|
||||
'pdf_read',
|
||||
'image_read',
|
||||
'image_preview',
|
||||
]),
|
||||
};
|
||||
|
||||
@@ -187,7 +191,11 @@ class ToolRegistry {
|
||||
// Document & data tools
|
||||
this.registerSafe(pdfReadTool);
|
||||
this.registerSafe(pdfExtractImagesTool);
|
||||
this.registerSafe(pdfExtractTablesTool);
|
||||
this.registerSafe(imageReadTool);
|
||||
this.registerSafe(imagePreviewTool);
|
||||
this.registerSafe(imageInfoTool);
|
||||
this.registerSafe(imageEditTool);
|
||||
this.registerSafe(audioTranscribeTool);
|
||||
this.registerSafe(pythonEvalTool);
|
||||
this.registerSafe(sqliteTool);
|
||||
|
||||
+3
-1
@@ -6,6 +6,8 @@ import { log } from '../security/log-scrubber.js';
|
||||
export interface ShellToolArgs {
|
||||
command: string;
|
||||
cwd?: string;
|
||||
_workspacePath?: string;
|
||||
_workspace?: string;
|
||||
}
|
||||
|
||||
// ── Path confinement helper ───────────────────────────────────────────────────
|
||||
@@ -56,7 +58,7 @@ function containsOutOfScopeAbsPath(command: string, workspacePath: string): bool
|
||||
export async function executeShell(args: ShellToolArgs): Promise<ToolResult> {
|
||||
const config = getConfig().getConfig();
|
||||
const permissions = config.tools.permissions.shell;
|
||||
const workspacePath = path.resolve(config.workspace.path);
|
||||
const workspacePath = path.resolve(args._workspacePath || args._workspace || config.workspace.path);
|
||||
|
||||
// Determine and resolve working directory
|
||||
const cwd = path.resolve(args.cwd ? args.cwd : workspacePath);
|
||||
|
||||
+11
-7
@@ -3,9 +3,9 @@ import fs from 'fs';
|
||||
import { getConfig } from '../config/config.js';
|
||||
import { ToolResult } from '../types.js';
|
||||
|
||||
function getWorkspacePath(): string {
|
||||
function getWorkspacePath(args?: any): string {
|
||||
try {
|
||||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
return args?._workspacePath || args?._workspace || getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||||
} catch {
|
||||
return path.join(process.cwd(), 'workspace');
|
||||
}
|
||||
@@ -58,11 +58,15 @@ export const sqliteTool = {
|
||||
if (!dbPath) return { success: false, error: 'db_path is required' };
|
||||
if (!query) return { success: false, error: 'query is required' };
|
||||
|
||||
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
|
||||
const resolved = path.isAbsolute(dbPath) ? dbPath : path.resolve(workspacePath, dbPath);
|
||||
|
||||
if (!isPathInsideDir(workspacePath, resolved)) {
|
||||
return { success: false, error: 'db_path must be inside the workspace directory' };
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
let resolved: string;
|
||||
if (path.isAbsolute(dbPath)) {
|
||||
resolved = dbPath;
|
||||
} else {
|
||||
resolved = path.resolve(workspacePath, dbPath);
|
||||
if (!isPathInsideDir(workspacePath, resolved)) {
|
||||
return { success: false, error: 'relative db_path must stay inside workspace (use absolute path for shared databases)' };
|
||||
}
|
||||
}
|
||||
|
||||
const allowWrite = Boolean(args?.write);
|
||||
|
||||
Reference in New Issue
Block a user