v2.7.1 — image edit improvements, search config, bug fixes

image_edit:
- speech_bubble operation with Korean font, rounded rect, tail
- sketch quality: CLAHE pre-processing for better line contrast
- output filename with timestamp to prevent overwrites
- anime/painting stylize timeout 30s→600s (AnimeGANv2 model download)
- remove_bg PNG output fix

server-v2.ts:
- resolveToolImageContent: embed edited image as base64 in tool results
  so vision models can see the output (3 execution paths covered)
- _activeModelName: fix vision support detection for Ollama-hosted models
  (kimi/gemini run via Ollama — provider='ollama' was masking vision capability)
- Synthetic tool call ID generation to prevent Gemini function_response empty name error
- image_edit path hint format changed to English to prevent Gemini hallucination
- userRequestedImageEdit: expanded keyword list (풍선, 달아, 붙여 등)
- image_read OCR failure returns success:true with visual description hint
- TOOL_BLOCKS photo: image_edit workflow documentation updated

web-ui/index.html:
- renderFileDownloads: skip download buttons for image file extensions

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
kim
2026-05-18 22:03:32 +09:00
co-authored by Claude Sonnet 4.6
parent 34932ac50a
commit 245606252e
15 changed files with 1693 additions and 79 deletions
+1 -1
View File
@@ -26,7 +26,7 @@ export const audioTranscribeTool = {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = getConfig().getConfig()?.workspace?.path || process.cwd();
const workspacePath = args?._workspacePath || args?._workspace || getConfig().getConfig()?.workspace?.path || process.cwd();
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!fs.existsSync(resolved)) {
+51 -18
View File
@@ -11,13 +11,16 @@ const execFileAsync = promisify(execFile);
const PATCH_OUTPUT_MAX_CHARS = 8000;
// Helper function to check if path is allowed
function resolveWorkspacePath(targetPath: string): string {
const config = getConfig().getConfig();
const workspace = config.workspace.path;
function resolveWorkspacePath(targetPath: string, workspacePath?: string): string {
const workspace = workspacePath || getConfig().getConfig().workspace.path;
if (path.isAbsolute(targetPath)) return targetPath;
return path.join(workspace, targetPath);
}
function getSessionWorkspace(args: any): string {
return args?._workspacePath || args?._workspace || getConfig().getConfig().workspace.path;
}
function normalizePathForCompare(p: string): string {
const resolved = path.resolve(String(p || ''));
if (process.platform === 'win32') return resolved.toLowerCase();
@@ -168,6 +171,8 @@ export interface ReadToolArgs {
path: string;
start_line?: number;
num_lines?: number;
_workspacePath?: string;
_workspace?: string;
}
type RetrievalMode = 'fast' | 'standard' | 'deep';
@@ -203,7 +208,8 @@ function retrievalMaxLines(mode: RetrievalMode): number {
export async function executeRead(args: ReadToolArgs): Promise<ToolResult> {
try {
const absPath = resolveWorkspacePath(args.path);
const workspacePath = getSessionWorkspace(args);
const absPath = resolveWorkspacePath(args.path, workspacePath);
const pathCheck = isPathAllowed(absPath);
if (!pathCheck.allowed) {
return {
@@ -252,6 +258,8 @@ export async function executeRead(args: ReadToolArgs): Promise<ToolResult> {
export interface WriteToolArgs {
path: string;
content: string;
_workspacePath?: string;
_workspace?: string;
}
export async function executeWrite(args: WriteToolArgs): Promise<ToolResult> {
@@ -268,7 +276,8 @@ export async function executeWrite(args: WriteToolArgs): Promise<ToolResult> {
error: 'content must be a string'
};
}
const absPath = resolveWorkspacePath(args.path);
const workspacePath = getSessionWorkspace(args);
const absPath = resolveWorkspacePath(args.path, workspacePath);
const pathCheck = isPathAllowed(absPath);
if (!pathCheck.allowed) {
return {
@@ -302,11 +311,14 @@ export interface EditToolArgs {
path: string;
old_str: string;
new_str: string;
_workspacePath?: string;
_workspace?: string;
}
export async function executeEdit(args: EditToolArgs): Promise<ToolResult> {
try {
const absPath = resolveWorkspacePath(args.path);
const workspacePath = getSessionWorkspace(args);
const absPath = resolveWorkspacePath(args.path, workspacePath);
const pathCheck = isPathAllowed(absPath);
if (!pathCheck.allowed) {
return {
@@ -356,11 +368,14 @@ export async function executeEdit(args: EditToolArgs): Promise<ToolResult> {
// LIST DIRECTORY TOOL
export interface ListToolArgs {
path: string;
_workspacePath?: string;
_workspace?: string;
}
export async function executeList(args: ListToolArgs): Promise<ToolResult> {
try {
const absPath = resolveWorkspacePath(args.path);
const workspacePath = getSessionWorkspace(args);
const absPath = resolveWorkspacePath(args.path, workspacePath);
const pathCheck = isPathAllowed(absPath);
if (!pathCheck.allowed) {
return {
@@ -441,9 +456,10 @@ export const listTool = {
// ── DELETE ────────────────────────────────────────────────────────────────────
import { rmSync, existsSync } from 'fs';
async function executeDelete(args: { path: string; recursive?: boolean }): Promise<ToolResult> {
async function executeDelete(args: { path: string; recursive?: boolean; _workspacePath?: string; _workspace?: string }): Promise<ToolResult> {
if (!args.path?.trim()) return { success: false, error: 'path is required' };
const absPath = resolveWorkspacePath(args.path);
const workspacePath = getSessionWorkspace(args);
const absPath = resolveWorkspacePath(args.path, workspacePath);
if (!existsSync(absPath)) return { success: false, error: `Path does not exist: ${absPath}` };
try {
rmSync(absPath, { recursive: args.recursive ?? false, force: true });
@@ -467,11 +483,14 @@ export const deleteTool = {
export interface RenameArgs {
path: string;
new_path: string;
_workspacePath?: string;
_workspace?: string;
}
export async function executeRename(args: RenameArgs): Promise<ToolResult> {
try {
const src = resolveWorkspacePath(args.path);
const dest = resolveWorkspacePath(args.new_path);
const workspacePath = getSessionWorkspace(args);
const src = resolveWorkspacePath(args.path, workspacePath);
const dest = resolveWorkspacePath(args.new_path, workspacePath);
const srcCheck = isPathAllowed(src);
const destCheck = isPathAllowed(dest);
if (!srcCheck.allowed) return { success: false, error: srcCheck.reason };
@@ -503,11 +522,14 @@ export const renameTool = {
export interface CopyArgs {
path: string;
dest: string;
_workspacePath?: string;
_workspace?: string;
}
export async function executeCopy(args: CopyArgs): Promise<ToolResult> {
try {
const src = resolveWorkspacePath(args.path);
const dest = resolveWorkspacePath(args.dest);
const workspacePath = getSessionWorkspace(args);
const src = resolveWorkspacePath(args.path, workspacePath);
const dest = resolveWorkspacePath(args.dest, workspacePath);
const srcCheck = isPathAllowed(src);
const destCheck = isPathAllowed(dest);
if (!srcCheck.allowed) return { success: false, error: srcCheck.reason };
@@ -534,10 +556,13 @@ export const copyTool = {
export interface MkdirArgs {
path: string;
recursive?: boolean;
_workspacePath?: string;
_workspace?: string;
}
export async function executeMkdir(args: MkdirArgs): Promise<ToolResult> {
export async function executeMkdir(args: MkdirArgs & { _workspacePath?: string; _workspace?: string }): Promise<ToolResult> {
try {
const abs = resolveWorkspacePath(args.path);
const workspacePath = getSessionWorkspace(args);
const abs = resolveWorkspacePath(args.path, workspacePath);
const pathCheck = isPathAllowed(abs);
if (!pathCheck.allowed) return { success: false, error: pathCheck.reason };
await fs.mkdir(abs, { recursive: args.recursive ?? true });
@@ -560,10 +585,13 @@ export const mkdirTool = {
// ── STAT / INFO ──────────────────────────────────────────────────────────────
export interface StatArgs {
path: string;
_workspacePath?: string;
_workspace?: string;
}
export async function executeStat(args: StatArgs): Promise<ToolResult> {
try {
const abs = resolveWorkspacePath(args.path);
const workspacePath = getSessionWorkspace(args);
const abs = resolveWorkspacePath(args.path, workspacePath);
const pathCheck = isPathAllowed(abs);
if (!pathCheck.allowed) return { success: false, error: pathCheck.reason };
const st = await fs.stat(abs);
@@ -586,10 +614,13 @@ export const statTool = {
export interface AppendArgs {
path: string;
content: string;
_workspacePath?: string;
_workspace?: string;
}
export async function executeAppend(args: AppendArgs): Promise<ToolResult> {
try {
const abs = resolveWorkspacePath(args.path);
const workspacePath = getSessionWorkspace(args);
const abs = resolveWorkspacePath(args.path, workspacePath);
const pathCheck = isPathAllowed(abs);
if (!pathCheck.allowed) return { success: false, error: pathCheck.reason };
await fs.mkdir(path.dirname(abs), { recursive: true });
@@ -614,6 +645,8 @@ export const appendTool = {
export interface ApplyPatchArgs {
patch: string;
check?: boolean;
_workspacePath?: string;
_workspace?: string;
}
export async function executeApplyPatch(args: ApplyPatchArgs): Promise<ToolResult> {
@@ -626,7 +659,7 @@ export async function executeApplyPatch(args: ApplyPatchArgs): Promise<ToolResul
const validation = validatePatchPaths(targetPaths);
if (!validation.ok) return { success: false, error: validation.error };
const workspacePath = getConfig().getConfig().workspace.path;
const workspacePath = args._workspacePath || args._workspace || getConfig().getConfig().workspace.path;
const tempPatchPath = path.join(
os.tmpdir(),
`smallclaw-apply-${Date.now()}-${Math.random().toString(36).slice(2)}.patch`
+542 -5
View File
@@ -4,7 +4,9 @@ import fs from 'fs';
import { getConfig } from '../config/config.js';
import { ToolResult } from '../types.js';
function getWorkspacePath(): string {
function getWorkspacePath(args?: any): string {
const sessionPath = args?._workspacePath || args?._workspace;
if (sessionPath) return sessionPath;
try {
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
} catch {
@@ -12,7 +14,29 @@ function getWorkspacePath(): string {
}
}
const IMAGE_EXTS = new Set(['.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.tiff', '.tif']);
const IMAGE_EXTS = new Set(['.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.tiff', '.tif', '.heic', '.heif', '.avif']);
function runPython(script: string, timeoutMs = 60_000): Promise<any> {
return new Promise((resolve) => {
const child = spawn('python3', ['-c', script], {
timeout: timeoutMs,
env: { ...process.env },
});
let out = '';
child.stdout.on('data', (d: Buffer) => { out += d.toString(); });
child.stderr.on('data', () => {});
child.on('close', () => {
try { resolve(JSON.parse(out || '{}')); } catch { resolve({ error: 'Failed to parse output', raw: out.slice(0, 500) }); }
});
child.on('error', (e: Error) => resolve({ error: e.message }));
});
}
function buildImageMarkdown(filePath: string, workspacePath: string): string {
const relPath = path.relative(workspacePath, filePath).replace(/\\/g, '/');
const baseName = path.basename(filePath);
return `![${baseName}](/api/files/${relPath})`;
}
// Runs OCR in a child process so the tesseract worker doesn't block the main event loop.
const OCR_SCRIPT = `
@@ -65,7 +89,7 @@ export const imageReadTool = {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = getWorkspacePath();
const workspacePath = getWorkspacePath(args);
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!fs.existsSync(resolved)) {
@@ -96,11 +120,14 @@ export const imageReadTool = {
const lines: string[] = [
`File: ${path.basename(resolved)}`,
`Size: ${sizeMB} MB | Format: ${ext.slice(1).toUpperCase()} | Lang: ${lang}`,
'',
buildImageMarkdown(resolved, workspacePath),
];
if (ocrResult.error) {
lines.push(`OCR Error: ${ocrResult.error}`);
return { success: false, error: lines.join('\n') };
lines.push(`OCR unavailable: ${ocrResult.error}`);
lines.push('(Image is attached above — describe it visually.)');
return { success: true, stdout: lines.join('\n') };
}
if (!ocrResult.text) {
@@ -124,3 +151,513 @@ export const imageReadTool = {
};
},
};
export const imagePreviewTool = {
name: 'image_preview',
description: 'Display an inline preview of any image file in the chat. Works with uploaded photos, screenshots, generated images, or diagrams. Returns a markdown image that renders directly in the conversation.',
schema: {
path: 'Path to the image file — absolute or relative to workspace',
width: 'Max display width in pixels (optional)',
},
jsonSchema: {
type: 'object',
properties: {
path: { type: 'string', description: 'Path to the image file' },
width: { type: 'number', description: 'Max display width in pixels (optional)' },
},
required: ['path'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = getWorkspacePath(args);
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!fs.existsSync(resolved)) {
return { success: false, error: `File not found: ${resolved}` };
}
const ext = path.extname(resolved).toLowerCase();
if (!IMAGE_EXTS.has(ext)) {
return { success: false, error: `Not an image. Supported: ${[...IMAGE_EXTS].join(', ')}` };
}
const stat = fs.statSync(resolved);
const sizeMB = (stat.size / 1024 / 1024).toFixed(2);
const widthAttr = args?.width ? ` width="${args.width}"` : '';
return {
success: true,
stdout: `**${path.basename(resolved)}** (${sizeMB} MB)${widthAttr}\n\n${buildImageMarkdown(resolved, workspacePath)}`,
data: { path: resolved, size: stat.size, rel_path: path.relative(workspacePath, resolved).replace(/\\/g, '/') },
};
},
};
// ---------------------------------------------------------------------------
// image_info — metadata (dimensions, format, EXIF)
// ---------------------------------------------------------------------------
const INFO_SCRIPT = (imgPath: string) => `
import json, sys
try:
from PIL import Image, ExifTags
import os
img = Image.open(${JSON.stringify(imgPath)})
exif_data = {}
try:
raw = img._getexif()
if raw:
exif_data = {ExifTags.TAGS.get(k, str(k)): str(v) for k, v in raw.items() if k in ExifTags.TAGS}
except Exception:
pass
stat = os.stat(${JSON.stringify(imgPath)})
print(json.dumps({
"width": img.width, "height": img.height,
"format": img.format or "unknown", "mode": img.mode,
"size_bytes": stat.st_size,
"exif": exif_data,
}))
except Exception as e:
print(json.dumps({"error": str(e)}))
`;
export const imageInfoTool = {
name: 'image_info',
description: 'Get image metadata: dimensions (width × height), format, color mode, file size, and EXIF data (camera, GPS, date, etc.).',
schema: {
path: 'Path to the image file — absolute or relative to workspace',
},
jsonSchema: {
type: 'object',
properties: {
path: { type: 'string', description: 'Path to the image file' },
},
required: ['path'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = getWorkspacePath(args);
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
const result = await runPython(INFO_SCRIPT(resolved));
if (result.error) return { success: false, error: result.error };
const lines: string[] = [
`File: ${path.basename(resolved)}`,
`Dimensions: ${result.width} × ${result.height} px`,
`Format: ${result.format} | Mode: ${result.mode}`,
`Size: ${(result.size_bytes / 1024).toFixed(1)} KB`,
];
if (result.exif && Object.keys(result.exif).length > 0) {
lines.push('', 'EXIF:');
const relevant = ['Make', 'Model', 'DateTime', 'DateTimeOriginal', 'GPSInfo', 'Orientation', 'Software'];
for (const key of relevant) {
if (result.exif[key]) lines.push(` ${key}: ${result.exif[key]}`);
}
}
return { success: true, stdout: lines.join('\n'), data: result };
},
};
// ---------------------------------------------------------------------------
// image_edit — crop / resize / rotate / flip / convert / brightness /
// contrast / grayscale / watermark / thumbnail
// ---------------------------------------------------------------------------
const EDIT_SCRIPT = (params: Record<string, any>) => `
import json, sys, os
try:
from PIL import Image, ImageEnhance, ImageDraw, ImageFont
import pi_heif; pi_heif.register_heif_opener()
except Exception:
pass
try:
from PIL import Image, ImageEnhance, ImageDraw, ImageFont
import os, json
src = ${JSON.stringify(params.src)}
dst = ${JSON.stringify(params.dst)}
op = ${JSON.stringify(params.operation)}
img = Image.open(src)
orig_format = img.format or "PNG"
if op == "crop":
x, y, w, h = int(${params.x ?? 0}), int(${params.y ?? 0}), int(${params.width ?? 0}), int(${params.height ?? 0})
img = img.crop((x, y, x + w, y + h))
elif op == "resize":
w, h = int(${params.width ?? 0}), int(${params.height ?? 0})
keep = ${params.keep_aspect ? 'True' : 'False'}
if keep and w and h:
img.thumbnail((w, h), Image.LANCZOS)
elif w and h:
img = img.resize((w, h), Image.LANCZOS)
elif w:
ratio = w / img.width; img = img.resize((w, int(img.height * ratio)), Image.LANCZOS)
elif h:
ratio = h / img.height; img = img.resize((int(img.width * ratio), h), Image.LANCZOS)
elif op == "rotate":
deg = float(${params.degrees ?? 0})
img = img.rotate(-deg, expand=True)
elif op == "flip":
direction = ${JSON.stringify(params.direction ?? 'horizontal')}
img = img.transpose(Image.FLIP_LEFT_RIGHT if direction == "horizontal" else Image.FLIP_TOP_BOTTOM)
elif op == "grayscale":
img = img.convert("L").convert("RGB")
elif op == "brightness":
factor = float(${params.value ?? 1.0})
img = ImageEnhance.Brightness(img).enhance(factor)
elif op == "contrast":
factor = float(${params.value ?? 1.0})
img = ImageEnhance.Contrast(img).enhance(factor)
elif op == "sharpen":
factor = float(${params.value ?? 2.0})
img = ImageEnhance.Sharpness(img).enhance(factor)
elif op == "thumbnail":
w, h = int(${params.width ?? 256}), int(${params.height ?? 256})
img.thumbnail((w, h), Image.LANCZOS)
elif op == "watermark":
text = ${JSON.stringify(params.text ?? '')}
pos_key = ${JSON.stringify(params.position ?? 'bottom-right')}
opacity = int(float(${params.opacity ?? 0.5}) * 255)
draw = ImageDraw.Draw(img, "RGBA")
try:
has_cjk = any('가' <= c <= '힣' or '一' <= c <= '鿿' for c in text)
if has_cjk:
import os as _os
_kor_fonts = [
"/usr/share/fonts/truetype/nanum/NanumGothicBold.ttf",
"/usr/share/fonts/truetype/nanum/NanumGothic.ttf",
"/usr/share/fonts/opentype/noto/NotoSansCJK-Regular.ttc",
]
_fpath = next((f for f in _kor_fonts if _os.path.exists(f)), None)
font = ImageFont.truetype(_fpath, size=max(20, img.width // 30)) if _fpath else ImageFont.load_default()
else:
font = ImageFont.truetype("/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf", size=max(20, img.width // 30))
except Exception:
font = ImageFont.load_default()
bbox = draw.textbbox((0, 0), text, font=font)
tw, th = bbox[2] - bbox[0], bbox[3] - bbox[1]
margin = 20
positions = {
"center": ((img.width - tw) // 2, (img.height - th) // 2),
"bottom-right": (img.width - tw - margin, img.height - th - margin),
"bottom-left": (margin, img.height - th - margin),
"top-right": (img.width - tw - margin, margin),
"top-left": (margin, margin),
}
px, py = positions.get(pos_key, positions["bottom-right"])
draw.rectangle([px - 4, py - 4, px + tw + 4, py + th + 4], fill=(0, 0, 0, opacity // 2))
draw.text((px, py), text, font=font, fill=(255, 255, 255, opacity))
elif op == "speech_bubble":
import os as _os
_text = ${JSON.stringify(params.text ?? '')}
_pos = ${JSON.stringify(params.position ?? 'top-left')}
_bg = ${JSON.stringify(params.bg_color ?? 'white')}
_fg = ${JSON.stringify(params.text_color ?? 'black')}
_border = ${JSON.stringify(params.border_color ?? '#333333')}
_fsize = int(${params.font_size ?? 0}) or max(24, min(int(img.width / 22), 160))
_pad = max(12, _fsize // 2); _tail = max(20, _fsize // 1.5); _r = max(10, _fsize // 3); _mg = max(10, _fsize // 3)
_kor_fonts = [
"/usr/share/fonts/truetype/nanum/NanumGothicBold.ttf",
"/usr/share/fonts/opentype/noto/NotoSansCJK-Regular.ttc",
]
_fp = next((f for f in _kor_fonts if _os.path.exists(f)), None)
_font = ImageFont.truetype(_fp, _fsize) if _fp else ImageFont.load_default()
_lines = _text.split("\\n")
_lh = _fsize + 6
_tmp_draw = ImageDraw.Draw(img)
_bw = int(max(_tmp_draw.textlength(l, font=_font) for l in _lines)) + _pad * 2
_bh = _lh * len(_lines) + _pad * 2
_bx = (img.width - _bw - _mg) if "right" in _pos else _mg
_by = (img.height - _bh - _tail - _mg) if "bottom" in _pos else (_tail + _mg)
img = img.convert("RGBA")
_ov = Image.new("RGBA", img.size, (0, 0, 0, 0))
_d = ImageDraw.Draw(_ov)
_d.rounded_rectangle([_bx, _by, _bx + _bw, _by + _bh],
radius=_r, fill=_bg, outline=_border, width=2)
_cx = _bx + _bw // 2
if "bottom" in _pos:
_ty = _by + _bh
_pts = [(_cx - 14, _ty), (_cx + 14, _ty), (_cx, _ty + _tail)]
else:
_ty = _by
_pts = [(_cx - 14, _ty), (_cx + 14, _ty), (_cx, _ty - _tail)]
_d.polygon(_pts, fill=_bg, outline=_border)
img = Image.alpha_composite(img, _ov)
_d2 = ImageDraw.Draw(img)
for _i, _line in enumerate(_lines):
_lw = _d2.textlength(_line, font=_font)
_d2.text((_bx + (_bw - int(_lw)) // 2, _by + _pad + _i * _lh),
_line, font=_font, fill=_fg)
img = img.convert("RGB")
elif op == "stylize":
import cv2 as _cv2
import numpy as _np
_style = ${JSON.stringify(params.style ?? 'anime')}
_arr = _np.array(img.convert("RGB"))
_bgr = _cv2.cvtColor(_arr, _cv2.COLOR_RGB2BGR)
if _style == "sketch":
_gray = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
# CLAHE로 소스 대비 강화 → 밝은 사진에서도 선이 진하게 나옴
_clahe = _cv2.createCLAHE(clipLimit=2.5, tileGridSize=(8,8))
_gray_e = _clahe.apply(_gray)
_sk = _cv2.divide(_gray_e, _cv2.bitwise_not(_cv2.GaussianBlur(_cv2.bitwise_not(_gray_e),(15,15),0)), scale=256.0)
_sk_f = _sk.astype(_np.float32) / 255.0
_sk_f = _np.power(_sk_f, 3.0)
_lo = float(_np.percentile(_sk_f, 3))
_sk_f = _np.clip((_sk_f - _lo) / (1.0 - _lo + 1e-6), 0.0, 1.0)
_sk = (_sk_f * 255).astype(_np.uint8)
_result = _cv2.cvtColor(_cv2.cvtColor(_sk, _cv2.COLOR_GRAY2BGR), _cv2.COLOR_BGR2RGB)
elif _style == "sketch_color":
# overlay blend: vivid color + sketch lines
_gray = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
_sk = _cv2.divide(_gray, _cv2.bitwise_not(_cv2.GaussianBlur(_cv2.bitwise_not(_gray),(21,21),0)), scale=256.0)
_sk = _np.clip(_sk.astype(_np.float32)*0.85, 0, 255).astype(_np.uint8)
_hsv = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2HSV).astype(_np.float32)
_hsv[:,:,1] = _np.clip(_hsv[:,:,1]*2.0, 0, 255)
_vivid = _cv2.cvtColor(_hsv.astype(_np.uint8), _cv2.COLOR_HSV2BGR)
_vivid = _cv2.bilateralFilter(_vivid, 9, 60, 60)
_a = _sk.astype(_np.float32)/255.0
_b = _vivid.astype(_np.float32)/255.0
_ov = _np.where(_a[...,_np.newaxis] < 0.5, 2*_a[...,_np.newaxis]*_b, 1-2*(1-_a[...,_np.newaxis])*(1-_b))
_ov = _np.clip(_ov*255, 0, 255).astype(_np.uint8)
_blend = _cv2.addWeighted(_ov, 0.4, _bgr, 0.6, 0)
_result = _cv2.cvtColor(_blend, _cv2.COLOR_BGR2RGB)
elif _style == "cartoon":
# bilateral×7 (stronger flatten) + high-threshold Canny (외곽선만)
_blur = _bgr.copy()
for _ in range(7): _blur = _cv2.bilateralFilter(_blur, 9, 75, 75)
_hsvc = _cv2.cvtColor(_blur, _cv2.COLOR_BGR2HSV)
_hsvc[:,:,1] = _np.clip(_hsvc[:,:,1].astype(_np.float32)*1.5, 0, 255).astype(_np.uint8)
_blur = _cv2.cvtColor(_hsvc, _cv2.COLOR_HSV2BGR)
_gray = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
# 9×9 pre-blur + high threshold → textures/clothing patterns filtered out
_edges = _cv2.Canny(_cv2.GaussianBlur(_gray,(9,9),0), 70, 180)
_mask = _cv2.cvtColor(_cv2.bitwise_not(_edges), _cv2.COLOR_GRAY2BGR)
_result = _cv2.cvtColor((_blur.astype(_np.float32)*(_mask.astype(_np.float32)/255.0)).astype(_np.uint8), _cv2.COLOR_BGR2RGB)
elif _style == "watercolor":
# bilateral×3 (faces preserved) + median + soft edge overlay
_wc = _bgr.copy()
for _ in range(3): _wc = _cv2.bilateralFilter(_wc, 9, 75, 75)
_wc = _cv2.medianBlur(_wc, 5)
_hsvw = _cv2.cvtColor(_wc, _cv2.COLOR_BGR2HSV).astype(_np.float32)
_hsvw[:,:,1] = _np.clip(_hsvw[:,:,1]*1.5, 0, 255)
_hsvw[:,:,2] = _np.clip(_hsvw[:,:,2]*1.08, 0, 255)
_wc = _cv2.cvtColor(_hsvw.astype(_np.uint8), _cv2.COLOR_HSV2BGR)
_g3 = _cv2.cvtColor(_bgr, _cv2.COLOR_BGR2GRAY)
_e3 = _cv2.Canny(_cv2.GaussianBlur(_g3,(9,9),0), 30, 100)
_e3 = _cv2.GaussianBlur(_e3,(13,13),0) # very soft edges
_e3a = _np.stack([_e3.astype(_np.float32)/255.0*0.22]*3, axis=-1) # 22% opacity
_result = _cv2.cvtColor(_np.clip(_wc.astype(_np.float32)*(1-_e3a),0,255).astype(_np.uint8), _cv2.COLOR_BGR2RGB)
else: # anime / painting / default — AnimeGANv2
import torch as _torch
_preset_map = {"anime":"paprika","painting":"face_paint_512_v2","celeba":"celeba_distill","anime_v1":"face_paint_512_v1","face_paint":"face_paint_512_v2"}
_preset = _preset_map.get(_style, "paprika")
_gen = _torch.hub.load('bryandlee/animegan2-pytorch:main','generator',pretrained=_preset,trust_repo=True)
_f2p = _torch.hub.load('bryandlee/animegan2-pytorch:main','face2paint',trust_repo=True)
_gen.eval()
# 비율 보존: 긴 변 기준 768, 정사각형 패딩 후 변환, 패딩 제거
_orig_w, _orig_h = img.size
_scale = 768 / max(_orig_w, _orig_h)
_rw, _rh = int(_orig_w*_scale), int(_orig_h*_scale)
_resized = img.convert("RGB").resize((_rw, _rh), Image.LANCZOS)
_sq = max(_rw, _rh)
_canvas = Image.new("RGB", (_sq, _sq), (255,255,255))
_canvas.paste(_resized, ((_sq-_rw)//2, (_sq-_rh)//2))
_out_sq = _f2p(_gen, _canvas, size=_sq)
# 패딩 제거
_px, _py = (_sq-_rw)//2, (_sq-_rh)//2
img = _out_sq.crop((_px, _py, _px+_rw, _py+_rh))
_result = None
if _result is not None:
img = Image.fromarray(_result)
elif op == "remove_bg":
try:
from rembg import remove as _rembg_remove
except ImportError:
print(json.dumps({"error": "rembg not installed. Run: pip install rembg[cpu]"})); sys.exit(0)
_inp = img.convert("RGBA")
img = _rembg_remove(_inp)
# force PNG output (preserves transparency)
if not dst.lower().endswith(".png"):
dst = os.path.splitext(dst)[0] + ".png"
elif op == "convert":
pass # format is applied at save time
else:
print(json.dumps({"error": f"Unknown operation: {op}"})); sys.exit(0)
# determine save format
ext = os.path.splitext(dst)[1].lower().lstrip(".")
fmt_map = {"jpg": "JPEG", "jpeg": "JPEG", "png": "PNG", "webp": "WEBP",
"gif": "GIF", "bmp": "BMP", "tiff": "TIFF", "tif": "TIFF"}
save_fmt = fmt_map.get(ext, orig_format)
quality = int(${params.quality ?? 85})
if save_fmt == "JPEG" and img.mode in ("RGBA", "LA", "P"):
img = img.convert("RGB")
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
save_args = {}
if save_fmt in ("JPEG", "WEBP"):
save_args["quality"] = quality
img.save(dst, format=save_fmt, **save_args)
stat = os.stat(dst)
print(json.dumps({"output": dst, "width": img.width, "height": img.height,
"format": save_fmt, "size_bytes": stat.st_size}))
except Exception as e:
import traceback
print(json.dumps({"error": str(e), "trace": traceback.format_exc()[-500:]}))
`;
export const imageEditTool = {
name: 'image_edit',
description: [
'Edit an image file. Supported operations:',
' crop — cut a rectangular region (x, y, width, height)',
' resize — scale to new dimensions (width, height, keep_aspect)',
' rotate — rotate clockwise by degrees',
' flip — mirror horizontally or vertically (direction: horizontal|vertical)',
' grayscale — convert to black & white',
' brightness — adjust brightness (value: 0.5 = half, 2.0 = double)',
' contrast — adjust contrast (value: 0.5 = half, 2.0 = double)',
' sharpen — sharpen edges (value: 1.0 = none, 3.0 = strong)',
' thumbnail — resize to fit within a box (width, height)',
' watermark — overlay text (text, position, opacity)',
' speech_bubble — add a speech bubble with tail (text, position, bg_color, text_color, border_color, font_size)',
' stylize — convert to artistic style. Neural: anime(default)|painting|celeba|anime_v1 (AnimeGANv2). Classic: sketch|sketch_color|cartoon|watercolor',
' remove_bg — remove background using AI (rembg); output is PNG with transparency',
' convert — change file format (output path determines format)',
'Returns the output file path and new dimensions.',
].join('\n'),
schema: {
path: 'Input image file path (absolute or relative to workspace)',
operation: 'Operation name: crop | resize | rotate | flip | grayscale | brightness | contrast | sharpen | thumbnail | watermark | speech_bubble | stylize | remove_bg | convert',
output: 'Output file path (optional; defaults to <input>_<op>.<ext>)',
x: '[crop] Left edge in pixels',
y: '[crop] Top edge in pixels',
width: '[crop/resize/thumbnail] Width in pixels',
height: '[crop/resize/thumbnail] Height in pixels',
keep_aspect: '[resize] Preserve aspect ratio (true/false, default false)',
degrees: '[rotate] Clockwise rotation degrees (e.g. 90, 180, -90)',
direction: '[flip] "horizontal"=좌우반전(left-right mirror) | "vertical"=상하반전(upside-down)',
value: '[brightness/contrast/sharpen] Adjustment factor (1.0 = no change)',
text: '[watermark/speech_bubble] Text string to overlay',
position: '[watermark] "center"|"bottom-right"|"bottom-left"|"top-right"|"top-left" / [speech_bubble] "top-left"|"top-right"|"bottom-left"|"bottom-right"',
opacity: '[watermark] Text opacity 0.0–1.0 (default 0.5)',
bg_color: '[speech_bubble] Bubble background color (default "white")',
text_color: '[speech_bubble] Text color (default "black")',
border_color: '[speech_bubble] Border color (default "#333333")',
font_size: '[speech_bubble] Font size in pixels (auto if omitted: image_width/22, clamped 24–160)',
style: '[stylize] "anime"(default,paprika — 단체/풍경에 좋음) | "painting"(face_paint_v2 — 개인 인물 일러스트) | "celeba"(만화체) | "sketch" | "sketch_color" | "cartoon" | "watercolor"',
quality: '[jpeg/webp output] Quality 1–100 (default 85)',
},
jsonSchema: {
type: 'object',
properties: {
path: { type: 'string' },
operation: { type: 'string', enum: ['crop','resize','rotate','flip','grayscale','brightness','contrast','sharpen','thumbnail','watermark','speech_bubble','stylize','remove_bg','convert'] },
output: { type: 'string' },
x: { type: 'number' },
y: { type: 'number' },
width: { type: 'number' },
height: { type: 'number' },
keep_aspect: { type: 'boolean' },
degrees: { type: 'number' },
direction: { type: 'string', enum: ['horizontal', 'vertical'] },
value: { type: 'number' },
text: { type: 'string' },
position: { type: 'string' },
opacity: { type: 'number' },
bg_color: { type: 'string' },
text_color: { type: 'string' },
border_color: { type: 'string' },
font_size: { type: 'number' },
quality: { type: 'number' },
},
required: ['path', 'operation'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const filePath = String(args?.path || '').trim();
const operation = String(args?.operation || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
if (!operation) return { success: false, error: 'operation is required' };
const workspacePath = getWorkspacePath(args);
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
// determine output path
let outPath = String(args?.output || '').trim();
if (!outPath) {
const ext = args?.format
? `.${args.format}`
: (operation === 'convert' ? '.png' : path.extname(resolved));
const base = path.basename(resolved, path.extname(resolved));
outPath = path.join(path.dirname(resolved), `${base}_${operation}_${Date.now()}${ext}`);
} else if (!path.isAbsolute(outPath)) {
outPath = path.resolve(workspacePath, outPath);
}
const params: Record<string, any> = {
src: resolved,
dst: outPath,
operation,
x: args?.x ?? 0,
y: args?.y ?? 0,
width: args?.width ?? 0,
height: args?.height ?? 0,
keep_aspect: args?.keep_aspect ?? false,
degrees: args?.degrees ?? 0,
direction: args?.direction ?? 'horizontal',
value: args?.value ?? 1.0,
text: args?.text ?? '',
position: args?.position ?? 'bottom-right',
opacity: args?.opacity ?? 0.5,
quality: args?.quality ?? 85,
style: args?.style ?? 'painting',
};
// Neural stylize (anime/painting/celeba) downloads PyTorch models on first run (~5 min).
// Classic filter styles (sketch/cartoon/watercolor) finish in <5s.
const neuralStyles = new Set(['anime', 'painting', 'celeba', 'anime_v1', 'face_paint']);
const timeoutMs = (operation === 'stylize' && neuralStyles.has(params.style)) ? 600_000 : 60_000;
const result = await runPython(EDIT_SCRIPT(params), timeoutMs);
if (result.error) return { success: false, error: result.error };
const relOut = path.relative(workspacePath, result.output).replace(/\\/g, '/');
const sizeMB = (result.size_bytes / 1024 / 1024).toFixed(2);
const preview = buildImageMarkdown(result.output, workspacePath);
return {
success: true,
stdout: [
`Operation: ${operation}`,
`Output: ${result.output}`,
`Dimensions: ${result.width} × ${result.height} px | Format: ${result.format} | Size: ${sizeMB} MB`,
'',
preview,
].join('\n'),
data: { ...result, rel_path: relOut },
};
},
};
+639
View File
@@ -0,0 +1,639 @@
import { execFile } from 'child_process';
import { promisify } from 'util';
import path from 'path';
import fs from 'fs';
import { getConfig } from '../config/config.js';
import { ToolResult } from '../types.js';
const execFileAsync = promisify(execFile);
function getWorkspacePath(): string {
try {
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
} catch {
return path.join(process.cwd(), 'workspace');
}
}
// CCITT G4 (fax-compressed) images produced by pdfimages cannot be opened
// directly by most viewers. Wrap them in a minimal TIFF header so Pillow can
// decode and re-save as PNG. Returns the number of files converted.
//
// Two known issues with naive conversion:
// 1. PIL ignores PhotometricInterpretation=0 (WhiteIsZero) for CCITT T.6,
// treating bit-0 as black → image is inverted. Fix: invert after open.
// 2. Height cannot be derived from compressed data size alone (sparse pages
// compress to near-zero bytes). Fix: use pdfimages -list to get real dims.
const CCITT_CONVERT_PY = `
import os, sys, struct, io, subprocess, re
from PIL import Image, ImageOps
outdir = sys.argv[1]
pdf_path = sys.argv[2] if len(sys.argv) > 2 else ''
# Get real image dimensions from pdfimages -list
dim_map = {} # index -> (width, height)
if pdf_path and os.path.exists(pdf_path):
try:
out = subprocess.check_output(['pdfimages', '-list', pdf_path],
stderr=subprocess.DEVNULL, timeout=30).decode()
for line in out.splitlines():
m = re.match(r'\\s*(\\d+)\\s+\\S+\\s+\\S+\\s+(\\d+)\\s+(\\d+)', line)
if m:
idx, w, h = int(m.group(1)), int(m.group(2)), int(m.group(3))
dim_map[idx] = (w, h)
except Exception:
pass
converted = 0
for f in sorted(os.listdir(outdir)):
if not f.endswith('.ccitt'):
continue
base = f[:-6]
params_path = os.path.join(outdir, base + '.params')
ccitt_path = os.path.join(outdir, f)
png_path = os.path.join(outdir, base + '.png')
# Parse width from params file
params = open(params_path).read().strip().split() if os.path.exists(params_path) else []
width = None
for i, p in enumerate(params):
if p == '-X' and i + 1 < len(params):
width = int(params[i + 1])
width = width or 2000
# Get accurate height from pdfimages -list; fall back to estimation
idx_match = re.search(r'-(\\d+)\\.ccitt$', f)
idx = (int(idx_match.group(1)) + 1) if idx_match else -1 # pdfimages -list is 1-based
if idx in dim_map:
height = dim_map[idx][1]
else:
data_size = os.path.getsize(ccitt_path)
height = max(100, (data_size * 8 // width) + 100)
try:
data = open(ccitt_path, 'rb').read()
strip_offset = 8 + 2 + 12 * 8 + 4
def ifd_entry(tag, typ, count, value):
return struct.pack('<HHII', tag, typ, count, value)
entries = b''.join([
ifd_entry(256, 4, 1, width),
ifd_entry(257, 4, 1, height),
ifd_entry(258, 3, 1, 1),
ifd_entry(259, 3, 1, 4), # CCITT T.6
ifd_entry(262, 3, 1, 0), # WhiteIsZero
ifd_entry(278, 4, 1, height),
ifd_entry(279, 4, 1, len(data)),
ifd_entry(273, 4, 1, strip_offset),
])
tiff = (b'II' + struct.pack('<H', 42) + struct.pack('<I', 8)
+ struct.pack('<H', 8) + entries + struct.pack('<I', 0) + data)
img = Image.open(io.BytesIO(tiff))
# PIL ignores WhiteIsZero for CCITT → invert to get correct polarity
img = ImageOps.invert(img.convert('L'))
img.save(png_path, 'PNG')
os.remove(ccitt_path)
if os.path.exists(params_path):
os.remove(params_path)
converted += 1
print(f'ok:{base}.png', flush=True)
except Exception as e:
print(f'err:{base}:{e}', flush=True)
print(f'done:{converted}', flush=True)
`;
const FIGURES_DETECT_PY = `
import sys, os
import numpy as np
import cv2
def detect_figures(img_path):
img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)
if img is None:
return []
h, w = img.shape
_, binary = cv2.threshold(img, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)
# ── Step 1: detect horizontal lines ──
horiz_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (w//8, 1))
horiz_lines = cv2.morphologyEx(binary, cv2.MORPH_OPEN, horiz_kernel)
dilate_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (40, 10))
dilated = cv2.dilate(horiz_lines, dilate_kernel, iterations=1)
contours, _ = cv2.findContours(dilated, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
raw = sorted([cv2.boundingRect(c) for c in contours], key=lambda r: r[1])
raw = [(x, y, x+bw, y+bh) for x, y, bw, bh in raw if bw > w*0.15]
groups = []
used = [False] * len(raw)
for i, r1 in enumerate(raw):
if used[i]: continue
used[i] = True
x1, y1, x2, y2 = r1
for j in range(i+1, len(raw)):
if used[j]: continue
rx1, ry1, rx2, ry2 = raw[j]
if ry1 - y2 > 300: continue
overlap = max(0, min(x2, rx2) - max(x1, rx1))
min_w = min(x2-x1, rx2-rx1)
x_gap = max(0, max(x1, rx1) - min(x2, rx2))
if min_w > 0 and (overlap / min_w >= 0.5 or x_gap <= 80):
used[j] = True
x1 = min(x1, rx1); y1 = min(y1, ry1)
x2 = max(x2, rx2); y2 = max(y2, ry2)
groups.append((x1, y1, x2, y2))
# ── Step 2: expand vertically and validate ──
results = []
for gx1_raw, gy1, gx2_raw, gy2 in groups:
# Re-compute x bounds using median of per-row line extents to avoid
# full-width separator lines bleeding into adjacent text columns.
row_x1s, row_x2s = [], []
for row in range(gy1, gy2 + 1):
nz = np.where(horiz_lines[row, :] > 0)[0]
if len(nz) > w * 0.10:
row_x1s.append(int(nz[0]))
row_x2s.append(int(nz[-1]))
if len(row_x1s) >= 2:
# Filter out full-width separator lines (span >85% of page) before computing bounds
pairs = [(lx1, lx2) for lx1, lx2 in zip(row_x1s, row_x2s) if (lx2 - lx1) < w * 0.85]
if pairs:
gx1 = min(p[0] for p in pairs)
gx2 = max(p[1] for p in pairs)
else:
gx1, gx2 = gx1_raw, gx2_raw # all lines are separators — use raw
else:
gx1, gx2 = gx1_raw, gx2_raw
col_width = gx2 - gx1
col_horiz_in = horiz_lines[gy1:gy2+1, gx1:gx2]
line_per_row = (col_horiz_in > 0).sum(axis=1)
actual = np.where(line_per_row > col_width * 0.5)[0]
if len(actual) == 0:
continue
# Reject if detected lines are too far apart (text separators, not table/chart)
first_line = int(actual[0]) + gy1
last_line = int(actual[-1]) + gy1
# If lines are far apart, this is likely text separators not a table/chart:
# suppress expansion and require taller minimum height.
large_gap = len(actual) >= 2 and (int(actual[-1]) - int(actual[0])) > 60
if large_gap:
top = first_line - 10
if len(actual) >= 3:
# 3+ separators = multi-section table: expand to capture rows below last separator
bottom = last_line
prev_blank = 0
for row in range(last_line + 1, min(h, last_line + 300)):
if binary[row, gx1:gx2].sum() == 0:
prev_blank += 1
if prev_blank > 30: break
else:
prev_blank = 0
bottom = row
else:
# 1-2 lines = figure border / panel divider: minimal expansion
bottom = last_line + 10
else:
top = first_line
prev_blank = 0
for row in range(first_line - 1, max(0, first_line - 500), -1):
if binary[row, gx1:gx2].sum() == 0:
prev_blank += 1
if prev_blank > 50: break
else:
prev_blank = 0
top = row
bottom = last_line
prev_blank = 0
for row in range(last_line + 1, min(h, last_line + 400)):
if binary[row, gx1:gx2].sum() == 0:
prev_blank += 1
if prev_blank > 50: break
else:
prev_blank = 0
bottom = row
margin = 20
x1 = max(0, gx1-margin)
y1 = max(0, top-margin)
x2 = min(w, gx2+margin)
y2 = min(h, bottom+margin)
box_w = x2 - x1
box_h = y2 - y1
# ── Reject obvious non-figures ──
min_h = 120 if large_gap else 80
if box_h < min_h or box_w < 100:
continue
# 2. Too tall — likely grabbed a whole text column
if box_h > h * 0.75:
continue
# 3. Text density: if >60% of pixels are black, it's probably a text block
roi = binary[y1:y2, x1:x2]
if roi.size == 0:
continue
density = roi.sum() / (roi.size * 255)
if density > 0.60:
continue
# 4. Aspect ratio: extremely wide+short bars are headers
if box_h < box_w * 0.10:
continue
# 5. Full-width shallow bar — page header/footer banner
if box_w > w * 0.85 and box_h < 160:
continue
# 6. Thin separator line expanded into text: original group was tiny but grew huge
original_h = gy2 - gy1
if original_h < 30 and box_h > original_h * 5:
continue
results.append([x1, y1, x2, y2])
# Merge overlapping boxes
results.sort(key=lambda r: (r[1], r[0]))
merged = []
used = [False] * len(results)
for i, r1 in enumerate(results):
if used[i]: continue
x1, y1, x2, y2 = r1
for j in range(i+1, len(results)):
if used[j]: continue
rx1, ry1, rx2, ry2 = results[j]
ix1, iy1 = max(x1, rx1), max(y1, ry1)
ix2, iy2 = min(x2, rx2), min(y2, ry2)
if ix2 > ix1 and iy2 > iy1:
inter = (ix2-ix1) * (iy2-iy1)
smaller = min((x2-x1)*(y2-y1), (rx2-rx1)*(ry2-ry1))
if smaller > 0 and inter / smaller > 0.3:
used[j] = True
x1 = min(x1, rx1); y1 = min(y1, ry1)
x2 = max(x2, rx2); y2 = max(y2, ry2)
merged.append([x1, y1, x2, y2])
return merged
page_dir = sys.argv[1]
out_dir = sys.argv[2]
os.makedirs(out_dir, exist_ok=True)
page_files = sorted(f for f in os.listdir(page_dir) if f.startswith('page-') and f.endswith('.png'))
total = 0
for page_f in page_files:
page_num = page_f[5:-4]
img_path = os.path.join(page_dir, page_f)
boxes = detect_figures(img_path)
if not boxes:
continue
img_color = cv2.imread(img_path)
if img_color is None:
continue
for i, (x1, y1, x2, y2) in enumerate(boxes, 1):
crop = img_color[y1:y2, x1:x2]
if crop.size == 0:
continue
out_name = f'figure_p{page_num}_{i}.png'
cv2.imwrite(os.path.join(out_dir, out_name), crop)
print(f'ok:{out_name}', flush=True)
total += 1
print(f'done:{total}', flush=True)
`;
async function convertCcittToPng(outDir: string, pdfPath: string): Promise<{ converted: number; errors: string[] }> {
const errors: string[] = [];
let converted = 0;
try {
const { stdout } = await execFileAsync('python3', ['-c', CCITT_CONVERT_PY, outDir, pdfPath], { timeout: 60_000 });
for (const line of stdout.split('\n')) {
if (line.startsWith('done:')) converted = parseInt(line.slice(5), 10) || 0;
else if (line.startsWith('err:')) errors.push(line.slice(4));
}
} catch (err: any) {
errors.push(`ccitt_convert: ${String(err.message || err).slice(0, 200)}`);
}
return { converted, errors };
}
async function extractFigures(pageDir: string, outDir: string): Promise<{ files: string[]; errors: string[] }> {
const files: string[] = [];
const errors: string[] = [];
try {
const { stdout } = await execFileAsync('python3', ['-c', FIGURES_DETECT_PY, pageDir, outDir], { timeout: 120_000 });
for (const line of stdout.split('\n')) {
if (line.startsWith('ok:')) files.push(line.slice(3).trim());
else if (line.startsWith('err:')) errors.push(line.slice(4));
}
} catch (err: any) {
errors.push(`figures_detect: ${String(err.message || err).slice(0, 200)}`);
}
return { files, errors };
}
export const pdfExtractImagesTool = {
name: 'pdf_extract_images',
description: 'Extract images and diagrams from a PDF file. mode "images" extracts embedded raster images. mode "figures" auto-detects and crops figures/tables using OpenCV line detection. mode "both" does images+figures (default). Use out_dir to save directly into a PPTX project folder.',
schema: {
path: 'Path to the PDF file (relative to workspace or absolute)',
mode: '"images" (embedded rasters), "figures" (auto-crop figures/tables via OpenCV), or "both" = images+figures (default)',
out_dir: 'Output folder name relative to workspace (default: uploads/{basename}-images). Use the PPTX project folder name to save images there directly.',
page_from: 'First page (1-indexed, default: 1)',
page_to: 'Last page (inclusive, default: last page)',
dpi: 'Resolution for page rendering in DPI (default: 150)',
},
jsonSchema: {
type: 'object',
required: ['path'],
properties: {
path: { type: 'string', description: 'Path to the PDF file (relative to workspace or absolute)' },
mode: { type: 'string', enum: ['images', 'figures', 'both'], description: 'Extraction mode: images=embedded rasters, figures=OpenCV-cropped figures/tables, both=images+figures (default)' },
out_dir: { type: 'string', description: 'Output folder relative to workspace (default: uploads/{basename}-images). Set to PPTX project folder name to save images there directly.' },
page_from: { type: 'number', description: 'First page (1-indexed)' },
page_to: { type: 'number', description: 'Last page (inclusive)' },
dpi: { type: 'number', description: 'Rendering DPI for pages mode (default: 150)' },
},
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
if (path.extname(resolved).toLowerCase() !== '.pdf') return { success: false, error: 'File must have a .pdf extension' };
const mode = String(args?.mode || 'both');
const dpi = Math.min(300, Math.max(72, Number(args?.dpi ?? 150)));
const pageFrom = args?.page_from ? Math.max(1, Math.floor(Number(args.page_from))) : null;
const pageTo = args?.page_to ? Math.max(1, Math.floor(Number(args.page_to))) : null;
const basename = path.basename(resolved, '.pdf').replace(/[^a-zA-Z0-9가-힣._-]/g, '_');
const dirBasename = basename.slice(0, 35);
const customOutDir = args?.out_dir ? path.normalize(String(args.out_dir).trim()).replace(/\/+$/, '') : null;
const outDir = customOutDir
? (path.isAbsolute(customOutDir) ? customOutDir : path.join(workspacePath, customOutDir))
: path.join(workspacePath, 'uploads', `${dirBasename}-images`);
const outDirRel = customOutDir
? (path.isAbsolute(customOutDir) ? path.relative(workspacePath, customOutDir) : customOutDir)
: `uploads/${dirBasename}-images`;
fs.mkdirSync(outDir, { recursive: true });
const extracted: string[] = [];
const errors: string[] = [];
// --- pdfimages: extract embedded raster images ---
if (mode === 'images' || mode === 'both') {
const imgArgs = ['-all'];
if (pageFrom) imgArgs.push('-f', String(pageFrom));
if (pageTo) imgArgs.push('-l', String(pageTo));
imgArgs.push(resolved, path.join(outDir, 'img'));
try {
await execFileAsync('pdfimages', imgArgs, { timeout: 60_000 });
// Convert any CCITT G4 files to PNG before collecting results
const ccittFiles = fs.readdirSync(outDir).filter(f => /^img-\d+\.ccitt$/i.test(f));
if (ccittFiles.length > 0) {
const { errors: ccittErrors } = await convertCcittToPng(outDir, resolved);
errors.push(...ccittErrors);
}
const files = fs.readdirSync(outDir)
.filter(f => /^img-\d+\.(jpg|jpeg|png|ppm|pbm|tif|tiff)$/i.test(f))
.sort();
extracted.push(...files.map(f => `${outDirRel}/${f}`));
} catch (err: any) {
errors.push(`pdfimages: ${String(err.message || err).slice(0, 200)}`);
}
}
// --- figures mode: render pages then auto-crop figures/tables with OpenCV ---
if (mode === 'figures' || mode === 'both') {
const figDpi = Math.min(300, Math.max(100, dpi));
const ppmArgs = ['-png', '-r', String(figDpi)];
if (pageFrom) ppmArgs.push('-f', String(pageFrom));
if (pageTo) ppmArgs.push('-l', String(pageTo));
ppmArgs.push(resolved, path.join(outDir, 'page'));
try {
await execFileAsync('pdftoppm', ppmArgs, { timeout: 120_000 });
} catch (err: any) {
errors.push(`pdftoppm: ${String(err.message || err).slice(0, 200)}`);
}
const { files: figFiles, errors: figErrors } = await extractFigures(outDir, outDir);
errors.push(...figErrors);
for (const f of fs.readdirSync(outDir).filter(f => f.startsWith('page-') && f.endsWith('.png'))) {
try { fs.unlinkSync(path.join(outDir, f)); } catch {}
}
extracted.push(...figFiles.map(f => `${outDirRel}/${f}`));
}
if (extracted.length === 0) {
const errMsg = errors.length ? errors.join('; ') : 'No images found in the PDF';
return { success: false, error: errMsg };
}
const links = extracted.map(f => `[${path.basename(f)}](/api/files/${f})`).join('\n');
return {
success: true,
stdout: `${extracted.length}개 파일 추출 완료:\n${links}`,
data: { files: extracted, outDir, errors: errors.length ? errors : undefined },
};
},
};
// ---------------------------------------------------------------------------
// pdf_extract_tables
// 1차: PyMuPDF find_tables() — 선 기반 테이블, 빠름
// 2차: pdfplumber — 공백/선 혼합, 선 없는 테이블에도 강함
// ---------------------------------------------------------------------------
const TABLE_EXTRACT_PY = `
import sys, json, os, io
# PyMuPDF 1.24+ 가 import 시점에 C 레벨 fd=1(stdout)로 직접
# "Consider using pymupdf_layout..." 를 출력해 JSON 파싱을 깨뜨린다.
# sys.stdout 교체로는 막을 수 없으므로 os.dup2 로 fd 1 자체를 /dev/null 로 리다이렉트한다.
_devnull_fd = os.open(os.devnull, os.O_WRONLY)
_saved_fd = os.dup(1)
os.dup2(_devnull_fd, 1)
os.close(_devnull_fd)
try:
import fitz as _fitz
except ImportError:
_fitz = None
finally:
os.dup2(_saved_fd, 1) # stdout 복구
os.close(_saved_fd)
pdf_path = sys.argv[1]
fmt = sys.argv[2] if len(sys.argv) > 2 else 'markdown'
page_from = int(sys.argv[3]) - 1 if len(sys.argv) > 3 else 0 # 0-indexed
page_to = int(sys.argv[4]) - 1 if len(sys.argv) > 4 else None # inclusive, 0-indexed
engine = sys.argv[5] if len(sys.argv) > 5 else 'auto' # auto|pymupdf|pdfplumber
def cell(v):
return str(v).replace('\\n', ' ').strip() if v is not None else ''
def to_markdown(rows):
if not rows or not rows[0]:
return ''
widths = [max(len(cell(r[i])) for r in rows if i < len(r)) for i in range(len(rows[0]))]
widths = [max(w, 3) for w in widths]
def row_str(r):
return '| ' + ' | '.join(cell(r[i]).ljust(widths[i]) if i < len(r) else ' ' * widths[i] for i in range(len(widths))) + ' |'
sep = '| ' + ' | '.join('-' * w for w in widths) + ' |'
lines = [row_str(rows[0]), sep] + [row_str(r) for r in rows[1:]]
return '\\n'.join(lines)
def to_csv(rows):
import csv, io
buf = io.StringIO()
w = csv.writer(buf)
for r in rows:
w.writerow([cell(v) for v in r])
return buf.getvalue().rstrip()
def format_table(rows, fmt):
if fmt == 'csv': return to_csv(rows)
if fmt == 'json': return json.dumps([[cell(v) for v in r] for r in rows], ensure_ascii=False)
return to_markdown(rows)
results = []
errors = []
def try_pymupdf():
if _fitz is None:
raise ImportError('PyMuPDF(fitz) not available')
doc = _fitz.open(pdf_path)
end = page_to if page_to is not None else len(doc) - 1
found = []
for pno in range(page_from, min(end + 1, len(doc))):
page = doc[pno]
tabs = page.find_tables().tables # TableFinder → .tables 리스트
for ti, tab in enumerate(tabs):
rows = tab.extract()
if not rows: continue
found.append({'page': pno + 1, 'table': ti + 1, 'rows': len(rows),
'cols': len(rows[0]) if rows else 0,
'data': format_table(rows, fmt)})
doc.close()
return found
def try_pdfplumber():
import pdfplumber
found = []
with pdfplumber.open(pdf_path) as pdf:
end = page_to if page_to is not None else len(pdf.pages) - 1
for pno in range(page_from, min(end + 1, len(pdf.pages))):
page = pdf.pages[pno]
tables = page.extract_tables({
'vertical_strategy': 'lines_strict',
'horizontal_strategy': 'lines_strict',
})
# 선 감지 실패 시 text 기반으로 재시도
if not tables:
tables = page.extract_tables({
'vertical_strategy': 'text',
'horizontal_strategy': 'text',
'snap_tolerance': 3,
'join_tolerance': 3,
'edge_min_length': 10,
})
for ti, rows in enumerate(tables):
if not rows: continue
found.append({'page': pno + 1, 'table': ti + 1, 'rows': len(rows),
'cols': len(rows[0]) if rows else 0,
'data': format_table(rows, fmt)})
return found
try:
if engine == 'pymupdf':
results = try_pymupdf()
elif engine == 'pdfplumber':
results = try_pdfplumber()
else: # auto: pymupdf first, pdfplumber if no tables found
results = try_pymupdf()
if not results:
results = try_pdfplumber()
if results:
for r in results: r['engine'] = 'pdfplumber'
else:
errors.append('no_tables')
else:
for r in results: r['engine'] = 'pymupdf'
except Exception as e:
import traceback
errors.append(str(e))
errors.append(traceback.format_exc()[-600:])
print(json.dumps({'tables': results, 'errors': errors}, ensure_ascii=False))
`;
export const pdfExtractTablesTool = {
name: 'pdf_extract_tables',
description: [
'Extract tables from a PDF file as structured data.',
'Uses PyMuPDF find_tables() first (fast, line-based); falls back to pdfplumber (handles borderless tables too).',
'format: "markdown" (default) | "csv" | "json".',
'engine: "auto" (default) | "pymupdf" | "pdfplumber".',
'For scanned/image-based PDFs use pdf_extract_images with mode "figures" instead.',
].join(' '),
schema: {
path: 'Path to the PDF file (absolute or relative to workspace)',
format: 'Output format: "markdown" (default) | "csv" | "json"',
engine: 'Extraction engine: "auto" (default) | "pymupdf" | "pdfplumber"',
page_from: 'First page to scan, 1-indexed (default: 1)',
page_to: 'Last page to scan, inclusive (default: last page)',
},
jsonSchema: {
type: 'object',
properties: {
path: { type: 'string' },
format: { type: 'string', enum: ['markdown', 'csv', 'json'] },
engine: { type: 'string', enum: ['auto', 'pymupdf', 'pdfplumber'] },
page_from: { type: 'number' },
page_to: { type: 'number' },
},
required: ['path'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
if (path.extname(resolved).toLowerCase() !== '.pdf') return { success: false, error: 'File must have a .pdf extension' };
const fmt = ['markdown', 'csv', 'json'].includes(args?.format) ? String(args.format) : 'markdown';
const engine = ['auto', 'pymupdf', 'pdfplumber'].includes(args?.engine) ? String(args.engine) : 'auto';
const pageFrom = args?.page_from ? String(Math.max(1, Math.floor(Number(args.page_from)))) : '1';
const pageTo = args?.page_to ? String(Math.max(1, Math.floor(Number(args.page_to)))) : '9999';
let raw: { tables: any[]; errors: string[] };
try {
const { stdout } = await execFileAsync(
'python3', ['-c', TABLE_EXTRACT_PY, resolved, fmt, pageFrom, pageTo, engine],
{ timeout: 60_000, maxBuffer: 20 * 1024 * 1024 },
);
// PyMuPDF가 JSON 앞에 경고 텍스트를 stdout으로 출력할 수 있으므로
// 첫 번째 '{' 이후만 JSON으로 파싱한다.
const jsonStart = stdout.indexOf('{');
raw = JSON.parse(jsonStart >= 0 ? stdout.slice(jsonStart) : stdout);
} catch (err: any) {
return { success: false, error: `Table extraction failed: ${String(err.message || err).slice(0, 300)}` };
}
if (!raw.tables || raw.tables.length === 0) {
const hint = raw.errors?.includes('no_tables')
? 'No tables detected. If this is a scanned PDF, try pdf_extract_images with mode "figures".'
: `No tables found. Errors: ${raw.errors?.join('; ') || 'none'}`;
return { success: false, error: hint };
}
const lines: string[] = [];
for (const t of raw.tables) {
lines.push(`### Page ${t.page} — Table ${t.table} (${t.rows} rows × ${t.cols} cols, engine: ${t.engine ?? engine})`);
lines.push('');
lines.push(t.data);
lines.push('');
}
return {
success: true,
stdout: lines.join('\n').trimEnd(),
data: { table_count: raw.tables.length, tables: raw.tables, errors: raw.errors?.length ? raw.errors : undefined },
};
},
};
+4 -2
View File
@@ -7,7 +7,9 @@ import { ToolResult } from '../types.js';
const execFileAsync = promisify(execFile);
function getWorkspacePath(): string {
function getWorkspacePath(args?: any): string {
const sessionPath = args?._workspacePath || args?._workspace;
if (sessionPath) return sessionPath;
try {
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
} catch {
@@ -39,7 +41,7 @@ export const pdfReadTool = {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = getWorkspacePath();
const workspacePath = getWorkspacePath(args);
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!fs.existsSync(resolved)) {
+12 -3
View File
@@ -598,7 +598,7 @@ export const pptxTool: import('./registry.js').Tool = {
const configMgr = getConfig();
const config = configMgr.getConfig() as any;
const workspacePath = args._workspacePath || config.workspace?.path || process.cwd();
const workspacePath = args?._workspacePath || args?._workspace || config.workspace?.path || process.cwd();
// Inject config defaults into spec so Python engine can use them
if (!spec.template && config.ppt?.template) spec.template = config.ppt.template;
@@ -676,8 +676,17 @@ export const pptxTool: import('./registry.js').Tool = {
let found: string | null = null;
if (fs.existsSync(uploadsDir)) {
for (const sub of fs.readdirSync(uploadsDir)) {
const candidate = path.join(uploadsDir, sub, basename);
const subDir = path.join(uploadsDir, sub);
if (!fs.existsSync(subDir) || !fs.statSync(subDir).isDirectory()) continue;
// Exact match
const candidate = path.join(subDir, basename);
if (fs.existsSync(candidate)) { found = candidate; break; }
// Prefix match: "figure_p05" → "figure_p05_1.png"
const prefix = basename.replace(/\.[^.]+$/, '');
const prefixMatch = fs.readdirSync(subDir).find(f =>
f.startsWith(prefix + '_') || f.startsWith(prefix + '.')
);
if (prefixMatch) { found = path.join(subDir, prefixMatch); break; }
}
}
if (found) {
@@ -814,7 +823,7 @@ export const editPptxTool: import('./registry.js').Tool = {
const configMgr = getConfig();
const config = configMgr.getConfig() as any;
const workspacePath = args._workspacePath || config.workspace?.path || process.cwd();
const workspacePath = args?._workspacePath || args?._workspace || config.workspace?.path || process.cwd();
// Resolve absolute path
const absPath = path.isAbsolute(existingPath)
+2 -2
View File
@@ -355,7 +355,7 @@ export const pubmedFulltextTool = {
// Download PDF
const config = getConfig().getConfig();
const workspaceDir = args?._workspacePath || config.workspace?.path || path.join(process.cwd(), 'workspace');
const workspaceDir = args?._workspacePath || args?._workspace || config.workspace?.path || path.join(process.cwd(), 'workspace');
const pdfDir = path.join(workspaceDir, 'pubmed');
fs.mkdirSync(pdfDir, { recursive: true });
@@ -472,7 +472,7 @@ export const pubmedFulltextTool = {
// Save to workspace (prefer per-user path injected by v2 executeTool)
const config = getConfig().getConfig();
const workspaceDir = args?._workspacePath || config.workspace?.path || path.join(process.cwd(), 'workspace');
const workspaceDir = args?._workspacePath || args?._workspace || config.workspace?.path || path.join(process.cwd(), 'workspace');
const savePath = args?.save_path
? path.join(workspaceDir, args.save_path)
: path.join(workspaceDir, 'pubmed', `${pmcid}.txt`);
+47 -2
View File
@@ -1,9 +1,13 @@
import { spawn } from 'child_process';
import path from 'path';
import fs from 'fs';
import { getConfig } from '../config/config.js';
import { ToolResult } from '../types.js';
function getWorkspacePath(): string {
function getWorkspacePath(args?: any): string {
// Prefer the session-specific workspace passed by executeTool
const sessionPath = args?._workspacePath || args?._workspace;
if (sessionPath) return sessionPath;
try {
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
} catch {
@@ -53,7 +57,21 @@ export const pythonEvalTool = {
}
const timeoutSec = Math.min(60, Math.max(1, Number(args?.timeout ?? 15)));
const workspacePath = getWorkspacePath();
const workspacePath = getWorkspacePath(args);
// Snapshot files before execution
const beforeFiles = new Set<string>();
try {
function scanDir(dir: string, prefix: string) {
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
const rel = prefix ? `${prefix}/${entry.name}` : entry.name;
const full = path.join(dir, entry.name);
if (entry.isDirectory()) { scanDir(full, rel); }
else if (entry.isFile()) { try { beforeFiles.add(`${rel}:${fs.statSync(full).mtimeMs}`); } catch {} }
}
}
scanDir(workspacePath, '');
} catch {}
if (args?.packages) {
const pkgs = String(args.packages).split(',').map((p: string) => p.trim()).filter((p: string) => /^[a-zA-Z0-9_\-\.]+$/.test(p));
@@ -108,10 +126,37 @@ export const pythonEvalTool = {
return { success: false, error: `Execution timed out after ${timeoutSec}s` };
}
// Detect newly created files
const newFiles: string[] = [];
try {
function scanDirAfter(dir: string, prefix: string) {
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
const rel = prefix ? `${prefix}/${entry.name}` : entry.name;
const full = path.join(dir, entry.name);
if (entry.isDirectory()) { scanDirAfter(full, rel); }
else if (entry.isFile()) {
const key = `${rel}:${fs.statSync(full).mtimeMs}`;
if (!beforeFiles.has(key)) { newFiles.push(rel); }
}
}
}
scanDirAfter(workspacePath, '');
} catch {}
const parts: string[] = [];
if (result.stdout.trim()) parts.push(`[stdout]\n${result.stdout.trim().slice(0, 20_000)}`);
if (result.stderr.trim()) parts.push(`[stderr]\n${result.stderr.trim().slice(0, 3_000)}`);
if (!result.stdout.trim() && !result.stderr.trim()) parts.push('(no output)');
if (newFiles.length > 0) {
const links = newFiles.map(f => {
const ext = path.extname(f).toLowerCase();
const isImage = ['.png', '.jpg', '.jpeg', '.gif', '.webp', '.bmp', '.tiff', '.tif'].includes(ext);
return isImage
? `![${path.basename(f)}](/api/files/${f})`
: `[${path.basename(f)}](/api/files/${f})`;
}).join('\n');
parts.push(`[생성된 파일]\n${links}`);
}
parts.push(`[exit ${result.exitCode}]`);
return {
+10 -2
View File
@@ -15,8 +15,8 @@ import { pptxTool, editPptxTool } from './pptx.js';
import { pubmedSearchTool, pubmedFetchTool, pubmedFulltextTool } from './pubmed.js';
import { openalexSearchTool, semanticSearchTool } from './scholar.js';
import { pdfReadTool } from './pdf.js';
import { pdfExtractImagesTool } from './pdf-extract.js';
import { imageReadTool } from './image.js';
import { pdfExtractImagesTool, pdfExtractTablesTool } from './pdf-extract.js';
import { imageReadTool, imagePreviewTool, imageInfoTool, imageEditTool } from './image.js';
import { audioTranscribeTool } from './audio-transcribe.js';
import { pythonEvalTool } from './python.js';
import { sqliteTool } from './sqlite.js';
@@ -58,6 +58,9 @@ const TOOL_PROFILE_TOOL_NAMES: Record<Exclude<ToolProfile, 'full'>, ReadonlySet<
'memory_write',
'pdf_read',
'image_read',
'image_preview',
'image_info',
'image_edit',
'audio_transcribe',
'python_eval',
'sqlite_query',
@@ -74,6 +77,7 @@ const TOOL_PROFILE_TOOL_NAMES: Record<Exclude<ToolProfile, 'full'>, ReadonlySet<
'semantic_search',
'pdf_read',
'image_read',
'image_preview',
]),
};
@@ -187,7 +191,11 @@ class ToolRegistry {
// Document & data tools
this.registerSafe(pdfReadTool);
this.registerSafe(pdfExtractImagesTool);
this.registerSafe(pdfExtractTablesTool);
this.registerSafe(imageReadTool);
this.registerSafe(imagePreviewTool);
this.registerSafe(imageInfoTool);
this.registerSafe(imageEditTool);
this.registerSafe(audioTranscribeTool);
this.registerSafe(pythonEvalTool);
this.registerSafe(sqliteTool);
+3 -1
View File
@@ -6,6 +6,8 @@ import { log } from '../security/log-scrubber.js';
export interface ShellToolArgs {
command: string;
cwd?: string;
_workspacePath?: string;
_workspace?: string;
}
// ── Path confinement helper ───────────────────────────────────────────────────
@@ -56,7 +58,7 @@ function containsOutOfScopeAbsPath(command: string, workspacePath: string): bool
export async function executeShell(args: ShellToolArgs): Promise<ToolResult> {
const config = getConfig().getConfig();
const permissions = config.tools.permissions.shell;
const workspacePath = path.resolve(config.workspace.path);
const workspacePath = path.resolve(args._workspacePath || args._workspace || config.workspace.path);
// Determine and resolve working directory
const cwd = path.resolve(args.cwd ? args.cwd : workspacePath);
+11 -7
View File
@@ -3,9 +3,9 @@ import fs from 'fs';
import { getConfig } from '../config/config.js';
import { ToolResult } from '../types.js';
function getWorkspacePath(): string {
function getWorkspacePath(args?: any): string {
try {
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
return args?._workspacePath || args?._workspace || getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
} catch {
return path.join(process.cwd(), 'workspace');
}
@@ -58,11 +58,15 @@ export const sqliteTool = {
if (!dbPath) return { success: false, error: 'db_path is required' };
if (!query) return { success: false, error: 'query is required' };
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
const resolved = path.isAbsolute(dbPath) ? dbPath : path.resolve(workspacePath, dbPath);
if (!isPathInsideDir(workspacePath, resolved)) {
return { success: false, error: 'db_path must be inside the workspace directory' };
const workspacePath = getWorkspacePath(args);
let resolved: string;
if (path.isAbsolute(dbPath)) {
resolved = dbPath;
} else {
resolved = path.resolve(workspacePath, dbPath);
if (!isPathInsideDir(workspacePath, resolved)) {
return { success: false, error: 'relative db_path must stay inside workspace (use absolute path for shared databases)' };
}
}
const allowWrite = Boolean(args?.write);