Files
homeclaw/src/tools/pdf-extract.ts
T
kimandClaude Sonnet 4.6 613690da55 Release 2.9.6: PPTX wizard search + image extraction improvements
- Add Europe PMC search source (free API, life sciences, PMC full-text)
- Fix Europe PMC field mapping: journalInfo.journal.title, firstPublicationDate
- Add PMC-only filter button with live paper count
- Move sort selector to search row; shrink year input (flex:none, no spinners)
- Uniform source checkbox sizing with per-source accent colors
- Exclude preview/ directory from image grid (project-images endpoint)
- Numeric sort for project images (slide1→2→9→10→11)
- Skip PDF figure re-extraction if figure_* files already exist (pdf-extract.ts)
- Preserve img:done state across wizard prep re-runs (savedImgDone)

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-03 14:30:16 +09:00

658 lines
28 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { execFile } from 'child_process';
import { promisify } from 'util';
import path from 'path';
import fs from 'fs';
import { getConfig } from '../config/config.js';
import { ToolResult } from '../types.js';
const execFileAsync = promisify(execFile);
function getWorkspacePath(): string {
try {
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
} catch {
return path.join(process.cwd(), 'workspace');
}
}
function isPathInsideDir(base: string, target: string): boolean {
const resolvedBase = path.resolve(base);
const resolvedTarget = path.resolve(target);
if (resolvedBase === resolvedTarget) return true;
const rel = path.relative(resolvedBase, resolvedTarget);
return rel !== '' && !rel.startsWith('..') && !path.isAbsolute(rel);
}
// CCITT G4 (fax-compressed) images produced by pdfimages cannot be opened
// directly by most viewers. Wrap them in a minimal TIFF header so Pillow can
// decode and re-save as PNG. Returns the number of files converted.
//
// Two known issues with naive conversion:
// 1. PIL ignores PhotometricInterpretation=0 (WhiteIsZero) for CCITT T.6,
// treating bit-0 as black → image is inverted. Fix: invert after open.
// 2. Height cannot be derived from compressed data size alone (sparse pages
// compress to near-zero bytes). Fix: use pdfimages -list to get real dims.
const CCITT_CONVERT_PY = `
import os, sys, struct, io, subprocess, re
from PIL import Image, ImageOps
outdir = sys.argv[1]
pdf_path = sys.argv[2] if len(sys.argv) > 2 else ''
# Get real image dimensions from pdfimages -list
dim_map = {} # index -> (width, height)
if pdf_path and os.path.exists(pdf_path):
try:
out = subprocess.check_output(['pdfimages', '-list', pdf_path],
stderr=subprocess.DEVNULL, timeout=30).decode()
for line in out.splitlines():
m = re.match(r'\\s*(\\d+)\\s+\\S+\\s+\\S+\\s+(\\d+)\\s+(\\d+)', line)
if m:
idx, w, h = int(m.group(1)), int(m.group(2)), int(m.group(3))
dim_map[idx] = (w, h)
except Exception:
pass
converted = 0
for f in sorted(os.listdir(outdir)):
if not f.endswith('.ccitt'):
continue
base = f[:-6]
params_path = os.path.join(outdir, base + '.params')
ccitt_path = os.path.join(outdir, f)
png_path = os.path.join(outdir, base + '.png')
# Parse width from params file
params = open(params_path).read().strip().split() if os.path.exists(params_path) else []
width = None
for i, p in enumerate(params):
if p == '-X' and i + 1 < len(params):
width = int(params[i + 1])
width = width or 2000
# Get accurate height from pdfimages -list; fall back to estimation
idx_match = re.search(r'-(\\d+)\\.ccitt$', f)
idx = (int(idx_match.group(1)) + 1) if idx_match else -1 # pdfimages -list is 1-based
if idx in dim_map:
height = dim_map[idx][1]
else:
data_size = os.path.getsize(ccitt_path)
height = max(100, (data_size * 8 // width) + 100)
try:
data = open(ccitt_path, 'rb').read()
strip_offset = 8 + 2 + 12 * 8 + 4
def ifd_entry(tag, typ, count, value):
return struct.pack('<HHII', tag, typ, count, value)
entries = b''.join([
ifd_entry(256, 4, 1, width),
ifd_entry(257, 4, 1, height),
ifd_entry(258, 3, 1, 1),
ifd_entry(259, 3, 1, 4), # CCITT T.6
ifd_entry(262, 3, 1, 0), # WhiteIsZero
ifd_entry(278, 4, 1, height),
ifd_entry(279, 4, 1, len(data)),
ifd_entry(273, 4, 1, strip_offset),
])
tiff = (b'II' + struct.pack('<H', 42) + struct.pack('<I', 8)
+ struct.pack('<H', 8) + entries + struct.pack('<I', 0) + data)
img = Image.open(io.BytesIO(tiff))
# PIL ignores WhiteIsZero for CCITT → invert to get correct polarity
img = ImageOps.invert(img.convert('L'))
img.save(png_path, 'PNG')
os.remove(ccitt_path)
if os.path.exists(params_path):
os.remove(params_path)
converted += 1
print(f'ok:{base}.png', flush=True)
except Exception as e:
print(f'err:{base}:{e}', flush=True)
print(f'done:{converted}', flush=True)
`;
const FIGURES_DETECT_PY = `
import sys, os
import numpy as np
import cv2
def detect_figures(img_path):
img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)
if img is None:
return []
h, w = img.shape
_, binary = cv2.threshold(img, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)
# ── Step 1: detect horizontal lines ──
horiz_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (w//8, 1))
horiz_lines = cv2.morphologyEx(binary, cv2.MORPH_OPEN, horiz_kernel)
dilate_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (40, 10))
dilated = cv2.dilate(horiz_lines, dilate_kernel, iterations=1)
contours, _ = cv2.findContours(dilated, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
raw = sorted([cv2.boundingRect(c) for c in contours], key=lambda r: r[1])
raw = [(x, y, x+bw, y+bh) for x, y, bw, bh in raw if bw > w*0.15]
groups = []
used = [False] * len(raw)
for i, r1 in enumerate(raw):
if used[i]: continue
used[i] = True
x1, y1, x2, y2 = r1
for j in range(i+1, len(raw)):
if used[j]: continue
rx1, ry1, rx2, ry2 = raw[j]
if ry1 - y2 > 300: continue
overlap = max(0, min(x2, rx2) - max(x1, rx1))
min_w = min(x2-x1, rx2-rx1)
x_gap = max(0, max(x1, rx1) - min(x2, rx2))
if min_w > 0 and (overlap / min_w >= 0.5 or x_gap <= 80):
used[j] = True
x1 = min(x1, rx1); y1 = min(y1, ry1)
x2 = max(x2, rx2); y2 = max(y2, ry2)
groups.append((x1, y1, x2, y2))
# ── Step 2: expand vertically and validate ──
results = []
for gx1_raw, gy1, gx2_raw, gy2 in groups:
# Re-compute x bounds using median of per-row line extents to avoid
# full-width separator lines bleeding into adjacent text columns.
row_x1s, row_x2s = [], []
for row in range(gy1, gy2 + 1):
nz = np.where(horiz_lines[row, :] > 0)[0]
if len(nz) > w * 0.10:
row_x1s.append(int(nz[0]))
row_x2s.append(int(nz[-1]))
if len(row_x1s) >= 2:
# Filter out full-width separator lines (span >85% of page) before computing bounds
pairs = [(lx1, lx2) for lx1, lx2 in zip(row_x1s, row_x2s) if (lx2 - lx1) < w * 0.85]
if pairs:
gx1 = min(p[0] for p in pairs)
gx2 = max(p[1] for p in pairs)
else:
gx1, gx2 = gx1_raw, gx2_raw # all lines are separators — use raw
else:
gx1, gx2 = gx1_raw, gx2_raw
col_width = gx2 - gx1
col_horiz_in = horiz_lines[gy1:gy2+1, gx1:gx2]
line_per_row = (col_horiz_in > 0).sum(axis=1)
actual = np.where(line_per_row > col_width * 0.5)[0]
if len(actual) == 0:
continue
# Reject if detected lines are too far apart (text separators, not table/chart)
first_line = int(actual[0]) + gy1
last_line = int(actual[-1]) + gy1
# If lines are far apart, this is likely text separators not a table/chart:
# suppress expansion and require taller minimum height.
large_gap = len(actual) >= 2 and (int(actual[-1]) - int(actual[0])) > 60
if large_gap:
top = first_line - 10
if len(actual) >= 3:
# 3+ separators = multi-section table: expand to capture rows below last separator
bottom = last_line
prev_blank = 0
for row in range(last_line + 1, min(h, last_line + 300)):
if binary[row, gx1:gx2].sum() == 0:
prev_blank += 1
if prev_blank > 30: break
else:
prev_blank = 0
bottom = row
else:
# 1-2 lines = figure border / panel divider: minimal expansion
bottom = last_line + 10
else:
top = first_line
prev_blank = 0
for row in range(first_line - 1, max(0, first_line - 500), -1):
if binary[row, gx1:gx2].sum() == 0:
prev_blank += 1
if prev_blank > 50: break
else:
prev_blank = 0
top = row
bottom = last_line
prev_blank = 0
for row in range(last_line + 1, min(h, last_line + 400)):
if binary[row, gx1:gx2].sum() == 0:
prev_blank += 1
if prev_blank > 50: break
else:
prev_blank = 0
bottom = row
margin = 20
x1 = max(0, gx1-margin)
y1 = max(0, top-margin)
x2 = min(w, gx2+margin)
y2 = min(h, bottom+margin)
box_w = x2 - x1
box_h = y2 - y1
# ── Reject obvious non-figures ──
min_h = 120 if large_gap else 80
if box_h < min_h or box_w < 100:
continue
# 2. Too tall — likely grabbed a whole text column
if box_h > h * 0.75:
continue
# 3. Text density: if >60% of pixels are black, it's probably a text block
roi = binary[y1:y2, x1:x2]
if roi.size == 0:
continue
density = roi.sum() / (roi.size * 255)
if density > 0.60:
continue
# 4. Aspect ratio: extremely wide+short bars are headers
if box_h < box_w * 0.10:
continue
# 5. Full-width shallow bar — page header/footer banner
if box_w > w * 0.85 and box_h < 160:
continue
# 6. Thin separator line expanded into text: original group was tiny but grew huge
original_h = gy2 - gy1
if original_h < 30 and box_h > original_h * 5:
continue
results.append([x1, y1, x2, y2])
# Merge overlapping boxes
results.sort(key=lambda r: (r[1], r[0]))
merged = []
used = [False] * len(results)
for i, r1 in enumerate(results):
if used[i]: continue
x1, y1, x2, y2 = r1
for j in range(i+1, len(results)):
if used[j]: continue
rx1, ry1, rx2, ry2 = results[j]
ix1, iy1 = max(x1, rx1), max(y1, ry1)
ix2, iy2 = min(x2, rx2), min(y2, ry2)
if ix2 > ix1 and iy2 > iy1:
inter = (ix2-ix1) * (iy2-iy1)
smaller = min((x2-x1)*(y2-y1), (rx2-rx1)*(ry2-ry1))
if smaller > 0 and inter / smaller > 0.3:
used[j] = True
x1 = min(x1, rx1); y1 = min(y1, ry1)
x2 = max(x2, rx2); y2 = max(y2, ry2)
merged.append([x1, y1, x2, y2])
return merged
page_dir = sys.argv[1]
out_dir = sys.argv[2]
os.makedirs(out_dir, exist_ok=True)
page_files = sorted(f for f in os.listdir(page_dir) if f.startswith('page-') and f.endswith('.png'))
total = 0
for page_f in page_files:
page_num = page_f[5:-4]
img_path = os.path.join(page_dir, page_f)
boxes = detect_figures(img_path)
if not boxes:
continue
img_color = cv2.imread(img_path)
if img_color is None:
continue
for i, (x1, y1, x2, y2) in enumerate(boxes, 1):
crop = img_color[y1:y2, x1:x2]
if crop.size == 0:
continue
out_name = f'figure_p{page_num}_{i}.png'
cv2.imwrite(os.path.join(out_dir, out_name), crop)
print(f'ok:{out_name}', flush=True)
total += 1
print(f'done:{total}', flush=True)
`;
async function convertCcittToPng(outDir: string, pdfPath: string): Promise<{ converted: number; errors: string[] }> {
const errors: string[] = [];
let converted = 0;
try {
const { stdout } = await execFileAsync('python3', ['-c', CCITT_CONVERT_PY, outDir, pdfPath], { timeout: 60_000 });
for (const line of stdout.split('\n')) {
if (line.startsWith('done:')) converted = parseInt(line.slice(5), 10) || 0;
else if (line.startsWith('err:')) errors.push(line.slice(4));
}
} catch (err: any) {
errors.push(`ccitt_convert: ${String(err.message || err).slice(0, 200)}`);
}
return { converted, errors };
}
async function extractFigures(pageDir: string, outDir: string): Promise<{ files: string[]; errors: string[] }> {
const files: string[] = [];
const errors: string[] = [];
try {
const { stdout } = await execFileAsync('python3', ['-c', FIGURES_DETECT_PY, pageDir, outDir], { timeout: 120_000 });
for (const line of stdout.split('\n')) {
if (line.startsWith('ok:')) files.push(line.slice(3).trim());
else if (line.startsWith('err:')) errors.push(line.slice(4));
}
} catch (err: any) {
errors.push(`figures_detect: ${String(err.message || err).slice(0, 200)}`);
}
return { files, errors };
}
export const pdfExtractImagesTool = {
name: 'pdf_extract_images',
description: 'Extract images and diagrams from a PDF file. mode "images" extracts embedded raster images. mode "figures" auto-detects and crops figures/tables using OpenCV line detection. mode "both" does images+figures (default). Use out_dir to save directly into a PPTX project folder.',
schema: {
path: 'Path to the PDF file (relative to workspace or absolute)',
mode: '"images" (embedded rasters), "figures" (auto-crop figures/tables via OpenCV), or "both" = images+figures (default)',
out_dir: 'Output folder name relative to workspace (default: uploads/{basename}-images). Use the PPTX project folder name to save images there directly.',
page_from: 'First page (1-indexed, default: 1)',
page_to: 'Last page (inclusive, default: last page)',
dpi: 'Resolution for page rendering in DPI (default: 150)',
},
jsonSchema: {
type: 'object',
required: ['path'],
properties: {
path: { type: 'string', description: 'Path to the PDF file (relative to workspace or absolute)' },
mode: { type: 'string', enum: ['images', 'figures', 'both'], description: 'Extraction mode: images=embedded rasters, figures=OpenCV-cropped figures/tables, both=images+figures (default)' },
out_dir: { type: 'string', description: 'Output folder relative to workspace (default: uploads/{basename}-images). Set to PPTX project folder name to save images there directly.' },
page_from: { type: 'number', description: 'First page (1-indexed)' },
page_to: { type: 'number', description: 'Last page (inclusive)' },
dpi: { type: 'number', description: 'Rendering DPI for pages mode (default: 150)' },
},
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!isPathInsideDir(workspacePath, resolved)) return { success: false, error: 'Access denied: path escapes workspace' };
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
if (path.extname(resolved).toLowerCase() !== '.pdf') return { success: false, error: 'File must have a .pdf extension' };
const mode = String(args?.mode || 'both');
const dpi = Math.min(300, Math.max(72, Number(args?.dpi ?? 150)));
const pageFrom = args?.page_from ? Math.max(1, Math.floor(Number(args.page_from))) : null;
const pageTo = args?.page_to ? Math.max(1, Math.floor(Number(args.page_to))) : null;
const basename = path.basename(resolved, '.pdf').replace(/[^a-zA-Z0-9가-힣._-]/g, '_');
const dirBasename = basename.slice(0, 35);
const customOutDir = args?.out_dir ? path.normalize(String(args.out_dir).trim()).replace(/\/+$/, '') : null;
const outDir = customOutDir
? (path.isAbsolute(customOutDir) ? customOutDir : path.join(workspacePath, customOutDir))
: path.join(workspacePath, 'uploads', `${dirBasename}-images`);
const outDirRel = customOutDir
? (path.isAbsolute(customOutDir) ? path.relative(workspacePath, customOutDir) : customOutDir)
: `uploads/${dirBasename}-images`;
fs.mkdirSync(outDir, { recursive: true });
const extracted: string[] = [];
const errors: string[] = [];
// --- pdfimages: extract embedded raster images ---
if (mode === 'images' || mode === 'both') {
const imgArgs = ['-all'];
if (pageFrom) imgArgs.push('-f', String(pageFrom));
if (pageTo) imgArgs.push('-l', String(pageTo));
imgArgs.push(resolved, path.join(outDir, 'img'));
try {
await execFileAsync('pdfimages', imgArgs, { timeout: 60_000 });
// Convert any CCITT G4 files to PNG before collecting results
const ccittFiles = fs.readdirSync(outDir).filter(f => /^img-\d+\.ccitt$/i.test(f));
if (ccittFiles.length > 0) {
const { errors: ccittErrors } = await convertCcittToPng(outDir, resolved);
errors.push(...ccittErrors);
}
const files = fs.readdirSync(outDir)
.filter(f => /^img-\d+\.(jpg|jpeg|png|ppm|pbm|tif|tiff)$/i.test(f))
.sort();
extracted.push(...files.map(f => `${outDirRel}/${f}`));
} catch (err: any) {
errors.push(`pdfimages: ${String(err.message || err).slice(0, 200)}`);
}
}
// --- figures mode: render pages then auto-crop figures/tables with OpenCV ---
if (mode === 'figures' || mode === 'both') {
// Skip expensive re-extraction if figure files already exist in outDir
const existingFigs = fs.existsSync(outDir)
? fs.readdirSync(outDir).filter(f => f.startsWith('figure_') && f.endsWith('.png'))
: [];
if (existingFigs.length > 0 && !pageFrom && !pageTo) {
extracted.push(...existingFigs.sort().map(f => `${outDirRel}/${f}`));
} else {
const figDpi = Math.min(300, Math.max(100, dpi));
const ppmArgs = ['-png', '-r', String(figDpi)];
if (pageFrom) ppmArgs.push('-f', String(pageFrom));
if (pageTo) ppmArgs.push('-l', String(pageTo));
ppmArgs.push(resolved, path.join(outDir, 'page'));
try {
await execFileAsync('pdftoppm', ppmArgs, { timeout: 120_000 });
} catch (err: any) {
errors.push(`pdftoppm: ${String(err.message || err).slice(0, 200)}`);
}
const { files: figFiles, errors: figErrors } = await extractFigures(outDir, outDir);
errors.push(...figErrors);
for (const f of fs.readdirSync(outDir).filter(f => f.startsWith('page-') && f.endsWith('.png'))) {
try { fs.unlinkSync(path.join(outDir, f)); } catch {}
}
extracted.push(...figFiles.map(f => `${outDirRel}/${f}`));
}
}
if (extracted.length === 0) {
const errMsg = errors.length ? errors.join('; ') : 'No images found in the PDF';
return { success: false, error: errMsg };
}
const links = extracted.map(f => `[${path.basename(f)}](/api/files/${f})`).join('\n');
return {
success: true,
stdout: `${extracted.length}개 파일 추출 완료:\n${links}`,
data: { files: extracted, outDir, errors: errors.length ? errors : undefined },
};
},
};
// ---------------------------------------------------------------------------
// pdf_extract_tables
// 1차: PyMuPDF find_tables() — 선 기반 테이블, 빠름
// 2차: pdfplumber — 공백/선 혼합, 선 없는 테이블에도 강함
// ---------------------------------------------------------------------------
const TABLE_EXTRACT_PY = `
import sys, json, os, io
# PyMuPDF 1.24+ 가 import 시점에 C 레벨 fd=1(stdout)로 직접
# "Consider using pymupdf_layout..." 를 출력해 JSON 파싱을 깨뜨린다.
# sys.stdout 교체로는 막을 수 없으므로 os.dup2 로 fd 1 자체를 /dev/null 로 리다이렉트한다.
_devnull_fd = os.open(os.devnull, os.O_WRONLY)
_saved_fd = os.dup(1)
os.dup2(_devnull_fd, 1)
os.close(_devnull_fd)
try:
import fitz as _fitz
except ImportError:
_fitz = None
finally:
os.dup2(_saved_fd, 1) # stdout 복구
os.close(_saved_fd)
pdf_path = sys.argv[1]
fmt = sys.argv[2] if len(sys.argv) > 2 else 'markdown'
page_from = int(sys.argv[3]) - 1 if len(sys.argv) > 3 else 0 # 0-indexed
page_to = int(sys.argv[4]) - 1 if len(sys.argv) > 4 else None # inclusive, 0-indexed
engine = sys.argv[5] if len(sys.argv) > 5 else 'auto' # auto|pymupdf|pdfplumber
def cell(v):
return str(v).replace('\\n', ' ').strip() if v is not None else ''
def to_markdown(rows):
if not rows or not rows[0]:
return ''
widths = [max(len(cell(r[i])) for r in rows if i < len(r)) for i in range(len(rows[0]))]
widths = [max(w, 3) for w in widths]
def row_str(r):
return '| ' + ' | '.join(cell(r[i]).ljust(widths[i]) if i < len(r) else ' ' * widths[i] for i in range(len(widths))) + ' |'
sep = '| ' + ' | '.join('-' * w for w in widths) + ' |'
lines = [row_str(rows[0]), sep] + [row_str(r) for r in rows[1:]]
return '\\n'.join(lines)
def to_csv(rows):
import csv, io
buf = io.StringIO()
w = csv.writer(buf)
for r in rows:
w.writerow([cell(v) for v in r])
return buf.getvalue().rstrip()
def format_table(rows, fmt):
if fmt == 'csv': return to_csv(rows)
if fmt == 'json': return json.dumps([[cell(v) for v in r] for r in rows], ensure_ascii=False)
return to_markdown(rows)
results = []
errors = []
def try_pymupdf():
if _fitz is None:
raise ImportError('PyMuPDF(fitz) not available')
doc = _fitz.open(pdf_path)
end = page_to if page_to is not None else len(doc) - 1
found = []
for pno in range(page_from, min(end + 1, len(doc))):
page = doc[pno]
tabs = page.find_tables().tables # TableFinder → .tables 리스트
for ti, tab in enumerate(tabs):
rows = tab.extract()
if not rows: continue
found.append({'page': pno + 1, 'table': ti + 1, 'rows': len(rows),
'cols': len(rows[0]) if rows else 0,
'data': format_table(rows, fmt)})
doc.close()
return found
def try_pdfplumber():
import pdfplumber
found = []
with pdfplumber.open(pdf_path) as pdf:
end = page_to if page_to is not None else len(pdf.pages) - 1
for pno in range(page_from, min(end + 1, len(pdf.pages))):
page = pdf.pages[pno]
tables = page.extract_tables({
'vertical_strategy': 'lines_strict',
'horizontal_strategy': 'lines_strict',
})
# 선 감지 실패 시 text 기반으로 재시도
if not tables:
tables = page.extract_tables({
'vertical_strategy': 'text',
'horizontal_strategy': 'text',
'snap_tolerance': 3,
'join_tolerance': 3,
'edge_min_length': 10,
})
for ti, rows in enumerate(tables):
if not rows: continue
found.append({'page': pno + 1, 'table': ti + 1, 'rows': len(rows),
'cols': len(rows[0]) if rows else 0,
'data': format_table(rows, fmt)})
return found
try:
if engine == 'pymupdf':
results = try_pymupdf()
elif engine == 'pdfplumber':
results = try_pdfplumber()
else: # auto: pymupdf first, pdfplumber if no tables found
results = try_pymupdf()
if not results:
results = try_pdfplumber()
if results:
for r in results: r['engine'] = 'pdfplumber'
else:
errors.append('no_tables')
else:
for r in results: r['engine'] = 'pymupdf'
except Exception as e:
import traceback
errors.append(str(e))
errors.append(traceback.format_exc()[-600:])
print(json.dumps({'tables': results, 'errors': errors}, ensure_ascii=False))
`;
export const pdfExtractTablesTool = {
name: 'pdf_extract_tables',
description: [
'Extract tables from a PDF file as structured data.',
'Uses PyMuPDF find_tables() first (fast, line-based); falls back to pdfplumber (handles borderless tables too).',
'format: "markdown" (default) | "csv" | "json".',
'engine: "auto" (default) | "pymupdf" | "pdfplumber".',
'For scanned/image-based PDFs use pdf_extract_images with mode "figures" instead.',
].join(' '),
schema: {
path: 'Path to the PDF file (absolute or relative to workspace)',
format: 'Output format: "markdown" (default) | "csv" | "json"',
engine: 'Extraction engine: "auto" (default) | "pymupdf" | "pdfplumber"',
page_from: 'First page to scan, 1-indexed (default: 1)',
page_to: 'Last page to scan, inclusive (default: last page)',
},
jsonSchema: {
type: 'object',
properties: {
path: { type: 'string' },
format: { type: 'string', enum: ['markdown', 'csv', 'json'] },
engine: { type: 'string', enum: ['auto', 'pymupdf', 'pdfplumber'] },
page_from: { type: 'number' },
page_to: { type: 'number' },
},
required: ['path'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const filePath = String(args?.path || '').trim();
if (!filePath) return { success: false, error: 'path is required' };
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
if (!isPathInsideDir(workspacePath, resolved)) return { success: false, error: 'Access denied: path escapes workspace' };
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
if (path.extname(resolved).toLowerCase() !== '.pdf') return { success: false, error: 'File must have a .pdf extension' };
const fmt = ['markdown', 'csv', 'json'].includes(args?.format) ? String(args.format) : 'markdown';
const engine = ['auto', 'pymupdf', 'pdfplumber'].includes(args?.engine) ? String(args.engine) : 'auto';
const pageFrom = args?.page_from ? String(Math.max(1, Math.floor(Number(args.page_from)))) : '1';
const pageTo = args?.page_to ? String(Math.max(1, Math.floor(Number(args.page_to)))) : '9999';
let raw: { tables: any[]; errors: string[] };
try {
const { stdout } = await execFileAsync(
'python3', ['-c', TABLE_EXTRACT_PY, resolved, fmt, pageFrom, pageTo, engine],
{ timeout: 60_000, maxBuffer: 20 * 1024 * 1024 },
);
// PyMuPDF가 JSON 앞에 경고 텍스트를 stdout으로 출력할 수 있으므로
// 첫 번째 '{' 이후만 JSON으로 파싱한다.
const jsonStart = stdout.indexOf('{');
raw = JSON.parse(jsonStart >= 0 ? stdout.slice(jsonStart) : stdout);
} catch (err: any) {
return { success: false, error: `Table extraction failed: ${String(err.message || err).slice(0, 300)}` };
}
if (!raw.tables || raw.tables.length === 0) {
const hint = raw.errors?.includes('no_tables')
? 'No tables detected. If this is a scanned PDF, try pdf_extract_images with mode "figures".'
: `No tables found. Errors: ${raw.errors?.join('; ') || 'none'}`;
return { success: false, error: hint };
}
const lines: string[] = [];
for (const t of raw.tables) {
lines.push(`### Page ${t.page} — Table ${t.table} (${t.rows} rows × ${t.cols} cols, engine: ${t.engine ?? engine})`);
lines.push('');
lines.push(t.data);
lines.push('');
}
return {
success: true,
stdout: lines.join('\n').trimEnd(),
data: { table_count: raw.tables.length, tables: raw.tables, errors: raw.errors?.length ? raw.errors : undefined },
};
},
};