- Add Europe PMC search source (free API, life sciences, PMC full-text) - Fix Europe PMC field mapping: journalInfo.journal.title, firstPublicationDate - Add PMC-only filter button with live paper count - Move sort selector to search row; shrink year input (flex:none, no spinners) - Uniform source checkbox sizing with per-source accent colors - Exclude preview/ directory from image grid (project-images endpoint) - Numeric sort for project images (slide1→2→9→10→11) - Skip PDF figure re-extraction if figure_* files already exist (pdf-extract.ts) - Preserve img:done state across wizard prep re-runs (savedImgDone) Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
658 lines
28 KiB
TypeScript
658 lines
28 KiB
TypeScript
import { execFile } from 'child_process';
|
||
import { promisify } from 'util';
|
||
import path from 'path';
|
||
import fs from 'fs';
|
||
import { getConfig } from '../config/config.js';
|
||
import { ToolResult } from '../types.js';
|
||
|
||
const execFileAsync = promisify(execFile);
|
||
|
||
function getWorkspacePath(): string {
|
||
try {
|
||
return getConfig().getConfig()?.workspace?.path || path.join(process.cwd(), 'workspace');
|
||
} catch {
|
||
return path.join(process.cwd(), 'workspace');
|
||
}
|
||
}
|
||
|
||
function isPathInsideDir(base: string, target: string): boolean {
|
||
const resolvedBase = path.resolve(base);
|
||
const resolvedTarget = path.resolve(target);
|
||
if (resolvedBase === resolvedTarget) return true;
|
||
const rel = path.relative(resolvedBase, resolvedTarget);
|
||
return rel !== '' && !rel.startsWith('..') && !path.isAbsolute(rel);
|
||
}
|
||
|
||
// CCITT G4 (fax-compressed) images produced by pdfimages cannot be opened
|
||
// directly by most viewers. Wrap them in a minimal TIFF header so Pillow can
|
||
// decode and re-save as PNG. Returns the number of files converted.
|
||
//
|
||
// Two known issues with naive conversion:
|
||
// 1. PIL ignores PhotometricInterpretation=0 (WhiteIsZero) for CCITT T.6,
|
||
// treating bit-0 as black → image is inverted. Fix: invert after open.
|
||
// 2. Height cannot be derived from compressed data size alone (sparse pages
|
||
// compress to near-zero bytes). Fix: use pdfimages -list to get real dims.
|
||
const CCITT_CONVERT_PY = `
|
||
import os, sys, struct, io, subprocess, re
|
||
from PIL import Image, ImageOps
|
||
|
||
outdir = sys.argv[1]
|
||
pdf_path = sys.argv[2] if len(sys.argv) > 2 else ''
|
||
|
||
# Get real image dimensions from pdfimages -list
|
||
dim_map = {} # index -> (width, height)
|
||
if pdf_path and os.path.exists(pdf_path):
|
||
try:
|
||
out = subprocess.check_output(['pdfimages', '-list', pdf_path],
|
||
stderr=subprocess.DEVNULL, timeout=30).decode()
|
||
for line in out.splitlines():
|
||
m = re.match(r'\\s*(\\d+)\\s+\\S+\\s+\\S+\\s+(\\d+)\\s+(\\d+)', line)
|
||
if m:
|
||
idx, w, h = int(m.group(1)), int(m.group(2)), int(m.group(3))
|
||
dim_map[idx] = (w, h)
|
||
except Exception:
|
||
pass
|
||
|
||
converted = 0
|
||
for f in sorted(os.listdir(outdir)):
|
||
if not f.endswith('.ccitt'):
|
||
continue
|
||
base = f[:-6]
|
||
params_path = os.path.join(outdir, base + '.params')
|
||
ccitt_path = os.path.join(outdir, f)
|
||
png_path = os.path.join(outdir, base + '.png')
|
||
|
||
# Parse width from params file
|
||
params = open(params_path).read().strip().split() if os.path.exists(params_path) else []
|
||
width = None
|
||
for i, p in enumerate(params):
|
||
if p == '-X' and i + 1 < len(params):
|
||
width = int(params[i + 1])
|
||
width = width or 2000
|
||
|
||
# Get accurate height from pdfimages -list; fall back to estimation
|
||
idx_match = re.search(r'-(\\d+)\\.ccitt$', f)
|
||
idx = (int(idx_match.group(1)) + 1) if idx_match else -1 # pdfimages -list is 1-based
|
||
if idx in dim_map:
|
||
height = dim_map[idx][1]
|
||
else:
|
||
data_size = os.path.getsize(ccitt_path)
|
||
height = max(100, (data_size * 8 // width) + 100)
|
||
|
||
try:
|
||
data = open(ccitt_path, 'rb').read()
|
||
strip_offset = 8 + 2 + 12 * 8 + 4
|
||
def ifd_entry(tag, typ, count, value):
|
||
return struct.pack('<HHII', tag, typ, count, value)
|
||
entries = b''.join([
|
||
ifd_entry(256, 4, 1, width),
|
||
ifd_entry(257, 4, 1, height),
|
||
ifd_entry(258, 3, 1, 1),
|
||
ifd_entry(259, 3, 1, 4), # CCITT T.6
|
||
ifd_entry(262, 3, 1, 0), # WhiteIsZero
|
||
ifd_entry(278, 4, 1, height),
|
||
ifd_entry(279, 4, 1, len(data)),
|
||
ifd_entry(273, 4, 1, strip_offset),
|
||
])
|
||
tiff = (b'II' + struct.pack('<H', 42) + struct.pack('<I', 8)
|
||
+ struct.pack('<H', 8) + entries + struct.pack('<I', 0) + data)
|
||
img = Image.open(io.BytesIO(tiff))
|
||
# PIL ignores WhiteIsZero for CCITT → invert to get correct polarity
|
||
img = ImageOps.invert(img.convert('L'))
|
||
img.save(png_path, 'PNG')
|
||
os.remove(ccitt_path)
|
||
if os.path.exists(params_path):
|
||
os.remove(params_path)
|
||
converted += 1
|
||
print(f'ok:{base}.png', flush=True)
|
||
except Exception as e:
|
||
print(f'err:{base}:{e}', flush=True)
|
||
print(f'done:{converted}', flush=True)
|
||
`;
|
||
|
||
const FIGURES_DETECT_PY = `
|
||
import sys, os
|
||
import numpy as np
|
||
import cv2
|
||
|
||
def detect_figures(img_path):
|
||
img = cv2.imread(img_path, cv2.IMREAD_GRAYSCALE)
|
||
if img is None:
|
||
return []
|
||
h, w = img.shape
|
||
_, binary = cv2.threshold(img, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)
|
||
|
||
# ── Step 1: detect horizontal lines ──
|
||
horiz_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (w//8, 1))
|
||
horiz_lines = cv2.morphologyEx(binary, cv2.MORPH_OPEN, horiz_kernel)
|
||
dilate_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (40, 10))
|
||
dilated = cv2.dilate(horiz_lines, dilate_kernel, iterations=1)
|
||
contours, _ = cv2.findContours(dilated, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
|
||
raw = sorted([cv2.boundingRect(c) for c in contours], key=lambda r: r[1])
|
||
raw = [(x, y, x+bw, y+bh) for x, y, bw, bh in raw if bw > w*0.15]
|
||
groups = []
|
||
used = [False] * len(raw)
|
||
for i, r1 in enumerate(raw):
|
||
if used[i]: continue
|
||
used[i] = True
|
||
x1, y1, x2, y2 = r1
|
||
for j in range(i+1, len(raw)):
|
||
if used[j]: continue
|
||
rx1, ry1, rx2, ry2 = raw[j]
|
||
if ry1 - y2 > 300: continue
|
||
overlap = max(0, min(x2, rx2) - max(x1, rx1))
|
||
min_w = min(x2-x1, rx2-rx1)
|
||
x_gap = max(0, max(x1, rx1) - min(x2, rx2))
|
||
if min_w > 0 and (overlap / min_w >= 0.5 or x_gap <= 80):
|
||
used[j] = True
|
||
x1 = min(x1, rx1); y1 = min(y1, ry1)
|
||
x2 = max(x2, rx2); y2 = max(y2, ry2)
|
||
groups.append((x1, y1, x2, y2))
|
||
|
||
# ── Step 2: expand vertically and validate ──
|
||
results = []
|
||
for gx1_raw, gy1, gx2_raw, gy2 in groups:
|
||
# Re-compute x bounds using median of per-row line extents to avoid
|
||
# full-width separator lines bleeding into adjacent text columns.
|
||
row_x1s, row_x2s = [], []
|
||
for row in range(gy1, gy2 + 1):
|
||
nz = np.where(horiz_lines[row, :] > 0)[0]
|
||
if len(nz) > w * 0.10:
|
||
row_x1s.append(int(nz[0]))
|
||
row_x2s.append(int(nz[-1]))
|
||
if len(row_x1s) >= 2:
|
||
# Filter out full-width separator lines (span >85% of page) before computing bounds
|
||
pairs = [(lx1, lx2) for lx1, lx2 in zip(row_x1s, row_x2s) if (lx2 - lx1) < w * 0.85]
|
||
if pairs:
|
||
gx1 = min(p[0] for p in pairs)
|
||
gx2 = max(p[1] for p in pairs)
|
||
else:
|
||
gx1, gx2 = gx1_raw, gx2_raw # all lines are separators — use raw
|
||
else:
|
||
gx1, gx2 = gx1_raw, gx2_raw
|
||
col_width = gx2 - gx1
|
||
col_horiz_in = horiz_lines[gy1:gy2+1, gx1:gx2]
|
||
line_per_row = (col_horiz_in > 0).sum(axis=1)
|
||
actual = np.where(line_per_row > col_width * 0.5)[0]
|
||
if len(actual) == 0:
|
||
continue
|
||
# Reject if detected lines are too far apart (text separators, not table/chart)
|
||
first_line = int(actual[0]) + gy1
|
||
last_line = int(actual[-1]) + gy1
|
||
# If lines are far apart, this is likely text separators not a table/chart:
|
||
# suppress expansion and require taller minimum height.
|
||
large_gap = len(actual) >= 2 and (int(actual[-1]) - int(actual[0])) > 60
|
||
if large_gap:
|
||
top = first_line - 10
|
||
if len(actual) >= 3:
|
||
# 3+ separators = multi-section table: expand to capture rows below last separator
|
||
bottom = last_line
|
||
prev_blank = 0
|
||
for row in range(last_line + 1, min(h, last_line + 300)):
|
||
if binary[row, gx1:gx2].sum() == 0:
|
||
prev_blank += 1
|
||
if prev_blank > 30: break
|
||
else:
|
||
prev_blank = 0
|
||
bottom = row
|
||
else:
|
||
# 1-2 lines = figure border / panel divider: minimal expansion
|
||
bottom = last_line + 10
|
||
else:
|
||
top = first_line
|
||
prev_blank = 0
|
||
for row in range(first_line - 1, max(0, first_line - 500), -1):
|
||
if binary[row, gx1:gx2].sum() == 0:
|
||
prev_blank += 1
|
||
if prev_blank > 50: break
|
||
else:
|
||
prev_blank = 0
|
||
top = row
|
||
bottom = last_line
|
||
prev_blank = 0
|
||
for row in range(last_line + 1, min(h, last_line + 400)):
|
||
if binary[row, gx1:gx2].sum() == 0:
|
||
prev_blank += 1
|
||
if prev_blank > 50: break
|
||
else:
|
||
prev_blank = 0
|
||
bottom = row
|
||
margin = 20
|
||
x1 = max(0, gx1-margin)
|
||
y1 = max(0, top-margin)
|
||
x2 = min(w, gx2+margin)
|
||
y2 = min(h, bottom+margin)
|
||
box_w = x2 - x1
|
||
box_h = y2 - y1
|
||
|
||
# ── Reject obvious non-figures ──
|
||
min_h = 120 if large_gap else 80
|
||
if box_h < min_h or box_w < 100:
|
||
continue
|
||
# 2. Too tall — likely grabbed a whole text column
|
||
if box_h > h * 0.75:
|
||
continue
|
||
# 3. Text density: if >60% of pixels are black, it's probably a text block
|
||
roi = binary[y1:y2, x1:x2]
|
||
if roi.size == 0:
|
||
continue
|
||
density = roi.sum() / (roi.size * 255)
|
||
if density > 0.60:
|
||
continue
|
||
# 4. Aspect ratio: extremely wide+short bars are headers
|
||
if box_h < box_w * 0.10:
|
||
continue
|
||
# 5. Full-width shallow bar — page header/footer banner
|
||
if box_w > w * 0.85 and box_h < 160:
|
||
continue
|
||
# 6. Thin separator line expanded into text: original group was tiny but grew huge
|
||
original_h = gy2 - gy1
|
||
if original_h < 30 and box_h > original_h * 5:
|
||
continue
|
||
|
||
results.append([x1, y1, x2, y2])
|
||
|
||
# Merge overlapping boxes
|
||
results.sort(key=lambda r: (r[1], r[0]))
|
||
merged = []
|
||
used = [False] * len(results)
|
||
for i, r1 in enumerate(results):
|
||
if used[i]: continue
|
||
x1, y1, x2, y2 = r1
|
||
for j in range(i+1, len(results)):
|
||
if used[j]: continue
|
||
rx1, ry1, rx2, ry2 = results[j]
|
||
ix1, iy1 = max(x1, rx1), max(y1, ry1)
|
||
ix2, iy2 = min(x2, rx2), min(y2, ry2)
|
||
if ix2 > ix1 and iy2 > iy1:
|
||
inter = (ix2-ix1) * (iy2-iy1)
|
||
smaller = min((x2-x1)*(y2-y1), (rx2-rx1)*(ry2-ry1))
|
||
if smaller > 0 and inter / smaller > 0.3:
|
||
used[j] = True
|
||
x1 = min(x1, rx1); y1 = min(y1, ry1)
|
||
x2 = max(x2, rx2); y2 = max(y2, ry2)
|
||
merged.append([x1, y1, x2, y2])
|
||
return merged
|
||
|
||
page_dir = sys.argv[1]
|
||
out_dir = sys.argv[2]
|
||
os.makedirs(out_dir, exist_ok=True)
|
||
page_files = sorted(f for f in os.listdir(page_dir) if f.startswith('page-') and f.endswith('.png'))
|
||
total = 0
|
||
for page_f in page_files:
|
||
page_num = page_f[5:-4]
|
||
img_path = os.path.join(page_dir, page_f)
|
||
boxes = detect_figures(img_path)
|
||
if not boxes:
|
||
continue
|
||
img_color = cv2.imread(img_path)
|
||
if img_color is None:
|
||
continue
|
||
for i, (x1, y1, x2, y2) in enumerate(boxes, 1):
|
||
crop = img_color[y1:y2, x1:x2]
|
||
if crop.size == 0:
|
||
continue
|
||
out_name = f'figure_p{page_num}_{i}.png'
|
||
cv2.imwrite(os.path.join(out_dir, out_name), crop)
|
||
print(f'ok:{out_name}', flush=True)
|
||
total += 1
|
||
print(f'done:{total}', flush=True)
|
||
`;
|
||
|
||
async function convertCcittToPng(outDir: string, pdfPath: string): Promise<{ converted: number; errors: string[] }> {
|
||
const errors: string[] = [];
|
||
let converted = 0;
|
||
try {
|
||
const { stdout } = await execFileAsync('python3', ['-c', CCITT_CONVERT_PY, outDir, pdfPath], { timeout: 60_000 });
|
||
for (const line of stdout.split('\n')) {
|
||
if (line.startsWith('done:')) converted = parseInt(line.slice(5), 10) || 0;
|
||
else if (line.startsWith('err:')) errors.push(line.slice(4));
|
||
}
|
||
} catch (err: any) {
|
||
errors.push(`ccitt_convert: ${String(err.message || err).slice(0, 200)}`);
|
||
}
|
||
return { converted, errors };
|
||
}
|
||
|
||
async function extractFigures(pageDir: string, outDir: string): Promise<{ files: string[]; errors: string[] }> {
|
||
const files: string[] = [];
|
||
const errors: string[] = [];
|
||
try {
|
||
const { stdout } = await execFileAsync('python3', ['-c', FIGURES_DETECT_PY, pageDir, outDir], { timeout: 120_000 });
|
||
for (const line of stdout.split('\n')) {
|
||
if (line.startsWith('ok:')) files.push(line.slice(3).trim());
|
||
else if (line.startsWith('err:')) errors.push(line.slice(4));
|
||
}
|
||
} catch (err: any) {
|
||
errors.push(`figures_detect: ${String(err.message || err).slice(0, 200)}`);
|
||
}
|
||
return { files, errors };
|
||
}
|
||
|
||
export const pdfExtractImagesTool = {
|
||
name: 'pdf_extract_images',
|
||
description: 'Extract images and diagrams from a PDF file. mode "images" extracts embedded raster images. mode "figures" auto-detects and crops figures/tables using OpenCV line detection. mode "both" does images+figures (default). Use out_dir to save directly into a PPTX project folder.',
|
||
schema: {
|
||
path: 'Path to the PDF file (relative to workspace or absolute)',
|
||
mode: '"images" (embedded rasters), "figures" (auto-crop figures/tables via OpenCV), or "both" = images+figures (default)',
|
||
out_dir: 'Output folder name relative to workspace (default: uploads/{basename}-images). Use the PPTX project folder name to save images there directly.',
|
||
page_from: 'First page (1-indexed, default: 1)',
|
||
page_to: 'Last page (inclusive, default: last page)',
|
||
dpi: 'Resolution for page rendering in DPI (default: 150)',
|
||
},
|
||
jsonSchema: {
|
||
type: 'object',
|
||
required: ['path'],
|
||
properties: {
|
||
path: { type: 'string', description: 'Path to the PDF file (relative to workspace or absolute)' },
|
||
mode: { type: 'string', enum: ['images', 'figures', 'both'], description: 'Extraction mode: images=embedded rasters, figures=OpenCV-cropped figures/tables, both=images+figures (default)' },
|
||
out_dir: { type: 'string', description: 'Output folder relative to workspace (default: uploads/{basename}-images). Set to PPTX project folder name to save images there directly.' },
|
||
page_from: { type: 'number', description: 'First page (1-indexed)' },
|
||
page_to: { type: 'number', description: 'Last page (inclusive)' },
|
||
dpi: { type: 'number', description: 'Rendering DPI for pages mode (default: 150)' },
|
||
},
|
||
additionalProperties: false,
|
||
},
|
||
execute: async (args: any): Promise<ToolResult> => {
|
||
const filePath = String(args?.path || '').trim();
|
||
if (!filePath) return { success: false, error: 'path is required' };
|
||
|
||
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
|
||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||
|
||
if (!isPathInsideDir(workspacePath, resolved)) return { success: false, error: 'Access denied: path escapes workspace' };
|
||
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
|
||
if (path.extname(resolved).toLowerCase() !== '.pdf') return { success: false, error: 'File must have a .pdf extension' };
|
||
|
||
const mode = String(args?.mode || 'both');
|
||
const dpi = Math.min(300, Math.max(72, Number(args?.dpi ?? 150)));
|
||
const pageFrom = args?.page_from ? Math.max(1, Math.floor(Number(args.page_from))) : null;
|
||
const pageTo = args?.page_to ? Math.max(1, Math.floor(Number(args.page_to))) : null;
|
||
|
||
const basename = path.basename(resolved, '.pdf').replace(/[^a-zA-Z0-9가-힣._-]/g, '_');
|
||
const dirBasename = basename.slice(0, 35);
|
||
const customOutDir = args?.out_dir ? path.normalize(String(args.out_dir).trim()).replace(/\/+$/, '') : null;
|
||
const outDir = customOutDir
|
||
? (path.isAbsolute(customOutDir) ? customOutDir : path.join(workspacePath, customOutDir))
|
||
: path.join(workspacePath, 'uploads', `${dirBasename}-images`);
|
||
const outDirRel = customOutDir
|
||
? (path.isAbsolute(customOutDir) ? path.relative(workspacePath, customOutDir) : customOutDir)
|
||
: `uploads/${dirBasename}-images`;
|
||
fs.mkdirSync(outDir, { recursive: true });
|
||
|
||
const extracted: string[] = [];
|
||
const errors: string[] = [];
|
||
|
||
// --- pdfimages: extract embedded raster images ---
|
||
if (mode === 'images' || mode === 'both') {
|
||
const imgArgs = ['-all'];
|
||
if (pageFrom) imgArgs.push('-f', String(pageFrom));
|
||
if (pageTo) imgArgs.push('-l', String(pageTo));
|
||
imgArgs.push(resolved, path.join(outDir, 'img'));
|
||
try {
|
||
await execFileAsync('pdfimages', imgArgs, { timeout: 60_000 });
|
||
// Convert any CCITT G4 files to PNG before collecting results
|
||
const ccittFiles = fs.readdirSync(outDir).filter(f => /^img-\d+\.ccitt$/i.test(f));
|
||
if (ccittFiles.length > 0) {
|
||
const { errors: ccittErrors } = await convertCcittToPng(outDir, resolved);
|
||
errors.push(...ccittErrors);
|
||
}
|
||
const files = fs.readdirSync(outDir)
|
||
.filter(f => /^img-\d+\.(jpg|jpeg|png|ppm|pbm|tif|tiff)$/i.test(f))
|
||
.sort();
|
||
extracted.push(...files.map(f => `${outDirRel}/${f}`));
|
||
} catch (err: any) {
|
||
errors.push(`pdfimages: ${String(err.message || err).slice(0, 200)}`);
|
||
}
|
||
}
|
||
|
||
// --- figures mode: render pages then auto-crop figures/tables with OpenCV ---
|
||
if (mode === 'figures' || mode === 'both') {
|
||
// Skip expensive re-extraction if figure files already exist in outDir
|
||
const existingFigs = fs.existsSync(outDir)
|
||
? fs.readdirSync(outDir).filter(f => f.startsWith('figure_') && f.endsWith('.png'))
|
||
: [];
|
||
if (existingFigs.length > 0 && !pageFrom && !pageTo) {
|
||
extracted.push(...existingFigs.sort().map(f => `${outDirRel}/${f}`));
|
||
} else {
|
||
const figDpi = Math.min(300, Math.max(100, dpi));
|
||
const ppmArgs = ['-png', '-r', String(figDpi)];
|
||
if (pageFrom) ppmArgs.push('-f', String(pageFrom));
|
||
if (pageTo) ppmArgs.push('-l', String(pageTo));
|
||
ppmArgs.push(resolved, path.join(outDir, 'page'));
|
||
try {
|
||
await execFileAsync('pdftoppm', ppmArgs, { timeout: 120_000 });
|
||
} catch (err: any) {
|
||
errors.push(`pdftoppm: ${String(err.message || err).slice(0, 200)}`);
|
||
}
|
||
const { files: figFiles, errors: figErrors } = await extractFigures(outDir, outDir);
|
||
errors.push(...figErrors);
|
||
for (const f of fs.readdirSync(outDir).filter(f => f.startsWith('page-') && f.endsWith('.png'))) {
|
||
try { fs.unlinkSync(path.join(outDir, f)); } catch {}
|
||
}
|
||
extracted.push(...figFiles.map(f => `${outDirRel}/${f}`));
|
||
}
|
||
}
|
||
|
||
if (extracted.length === 0) {
|
||
const errMsg = errors.length ? errors.join('; ') : 'No images found in the PDF';
|
||
return { success: false, error: errMsg };
|
||
}
|
||
|
||
const links = extracted.map(f => `[${path.basename(f)}](/api/files/${f})`).join('\n');
|
||
return {
|
||
success: true,
|
||
stdout: `${extracted.length}개 파일 추출 완료:\n${links}`,
|
||
data: { files: extracted, outDir, errors: errors.length ? errors : undefined },
|
||
};
|
||
},
|
||
};
|
||
|
||
// ---------------------------------------------------------------------------
|
||
// pdf_extract_tables
|
||
// 1차: PyMuPDF find_tables() — 선 기반 테이블, 빠름
|
||
// 2차: pdfplumber — 공백/선 혼합, 선 없는 테이블에도 강함
|
||
// ---------------------------------------------------------------------------
|
||
const TABLE_EXTRACT_PY = `
|
||
import sys, json, os, io
|
||
|
||
# PyMuPDF 1.24+ 가 import 시점에 C 레벨 fd=1(stdout)로 직접
|
||
# "Consider using pymupdf_layout..." 를 출력해 JSON 파싱을 깨뜨린다.
|
||
# sys.stdout 교체로는 막을 수 없으므로 os.dup2 로 fd 1 자체를 /dev/null 로 리다이렉트한다.
|
||
_devnull_fd = os.open(os.devnull, os.O_WRONLY)
|
||
_saved_fd = os.dup(1)
|
||
os.dup2(_devnull_fd, 1)
|
||
os.close(_devnull_fd)
|
||
try:
|
||
import fitz as _fitz
|
||
except ImportError:
|
||
_fitz = None
|
||
finally:
|
||
os.dup2(_saved_fd, 1) # stdout 복구
|
||
os.close(_saved_fd)
|
||
|
||
pdf_path = sys.argv[1]
|
||
fmt = sys.argv[2] if len(sys.argv) > 2 else 'markdown'
|
||
page_from = int(sys.argv[3]) - 1 if len(sys.argv) > 3 else 0 # 0-indexed
|
||
page_to = int(sys.argv[4]) - 1 if len(sys.argv) > 4 else None # inclusive, 0-indexed
|
||
engine = sys.argv[5] if len(sys.argv) > 5 else 'auto' # auto|pymupdf|pdfplumber
|
||
|
||
def cell(v):
|
||
return str(v).replace('\\n', ' ').strip() if v is not None else ''
|
||
|
||
def to_markdown(rows):
|
||
if not rows or not rows[0]:
|
||
return ''
|
||
widths = [max(len(cell(r[i])) for r in rows if i < len(r)) for i in range(len(rows[0]))]
|
||
widths = [max(w, 3) for w in widths]
|
||
def row_str(r):
|
||
return '| ' + ' | '.join(cell(r[i]).ljust(widths[i]) if i < len(r) else ' ' * widths[i] for i in range(len(widths))) + ' |'
|
||
sep = '| ' + ' | '.join('-' * w for w in widths) + ' |'
|
||
lines = [row_str(rows[0]), sep] + [row_str(r) for r in rows[1:]]
|
||
return '\\n'.join(lines)
|
||
|
||
def to_csv(rows):
|
||
import csv, io
|
||
buf = io.StringIO()
|
||
w = csv.writer(buf)
|
||
for r in rows:
|
||
w.writerow([cell(v) for v in r])
|
||
return buf.getvalue().rstrip()
|
||
|
||
def format_table(rows, fmt):
|
||
if fmt == 'csv': return to_csv(rows)
|
||
if fmt == 'json': return json.dumps([[cell(v) for v in r] for r in rows], ensure_ascii=False)
|
||
return to_markdown(rows)
|
||
|
||
results = []
|
||
errors = []
|
||
|
||
def try_pymupdf():
|
||
if _fitz is None:
|
||
raise ImportError('PyMuPDF(fitz) not available')
|
||
doc = _fitz.open(pdf_path)
|
||
end = page_to if page_to is not None else len(doc) - 1
|
||
found = []
|
||
for pno in range(page_from, min(end + 1, len(doc))):
|
||
page = doc[pno]
|
||
tabs = page.find_tables().tables # TableFinder → .tables 리스트
|
||
for ti, tab in enumerate(tabs):
|
||
rows = tab.extract()
|
||
if not rows: continue
|
||
found.append({'page': pno + 1, 'table': ti + 1, 'rows': len(rows),
|
||
'cols': len(rows[0]) if rows else 0,
|
||
'data': format_table(rows, fmt)})
|
||
doc.close()
|
||
return found
|
||
|
||
def try_pdfplumber():
|
||
import pdfplumber
|
||
found = []
|
||
with pdfplumber.open(pdf_path) as pdf:
|
||
end = page_to if page_to is not None else len(pdf.pages) - 1
|
||
for pno in range(page_from, min(end + 1, len(pdf.pages))):
|
||
page = pdf.pages[pno]
|
||
tables = page.extract_tables({
|
||
'vertical_strategy': 'lines_strict',
|
||
'horizontal_strategy': 'lines_strict',
|
||
})
|
||
# 선 감지 실패 시 text 기반으로 재시도
|
||
if not tables:
|
||
tables = page.extract_tables({
|
||
'vertical_strategy': 'text',
|
||
'horizontal_strategy': 'text',
|
||
'snap_tolerance': 3,
|
||
'join_tolerance': 3,
|
||
'edge_min_length': 10,
|
||
})
|
||
for ti, rows in enumerate(tables):
|
||
if not rows: continue
|
||
found.append({'page': pno + 1, 'table': ti + 1, 'rows': len(rows),
|
||
'cols': len(rows[0]) if rows else 0,
|
||
'data': format_table(rows, fmt)})
|
||
return found
|
||
|
||
try:
|
||
if engine == 'pymupdf':
|
||
results = try_pymupdf()
|
||
elif engine == 'pdfplumber':
|
||
results = try_pdfplumber()
|
||
else: # auto: pymupdf first, pdfplumber if no tables found
|
||
results = try_pymupdf()
|
||
if not results:
|
||
results = try_pdfplumber()
|
||
if results:
|
||
for r in results: r['engine'] = 'pdfplumber'
|
||
else:
|
||
errors.append('no_tables')
|
||
else:
|
||
for r in results: r['engine'] = 'pymupdf'
|
||
except Exception as e:
|
||
import traceback
|
||
errors.append(str(e))
|
||
errors.append(traceback.format_exc()[-600:])
|
||
|
||
print(json.dumps({'tables': results, 'errors': errors}, ensure_ascii=False))
|
||
`;
|
||
|
||
export const pdfExtractTablesTool = {
|
||
name: 'pdf_extract_tables',
|
||
description: [
|
||
'Extract tables from a PDF file as structured data.',
|
||
'Uses PyMuPDF find_tables() first (fast, line-based); falls back to pdfplumber (handles borderless tables too).',
|
||
'format: "markdown" (default) | "csv" | "json".',
|
||
'engine: "auto" (default) | "pymupdf" | "pdfplumber".',
|
||
'For scanned/image-based PDFs use pdf_extract_images with mode "figures" instead.',
|
||
].join(' '),
|
||
schema: {
|
||
path: 'Path to the PDF file (absolute or relative to workspace)',
|
||
format: 'Output format: "markdown" (default) | "csv" | "json"',
|
||
engine: 'Extraction engine: "auto" (default) | "pymupdf" | "pdfplumber"',
|
||
page_from: 'First page to scan, 1-indexed (default: 1)',
|
||
page_to: 'Last page to scan, inclusive (default: last page)',
|
||
},
|
||
jsonSchema: {
|
||
type: 'object',
|
||
properties: {
|
||
path: { type: 'string' },
|
||
format: { type: 'string', enum: ['markdown', 'csv', 'json'] },
|
||
engine: { type: 'string', enum: ['auto', 'pymupdf', 'pdfplumber'] },
|
||
page_from: { type: 'number' },
|
||
page_to: { type: 'number' },
|
||
},
|
||
required: ['path'],
|
||
additionalProperties: false,
|
||
},
|
||
execute: async (args: any): Promise<ToolResult> => {
|
||
const filePath = String(args?.path || '').trim();
|
||
if (!filePath) return { success: false, error: 'path is required' };
|
||
|
||
const workspacePath = String(args?._workspacePath || args?._workspace || '') || getWorkspacePath();
|
||
const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(workspacePath, filePath);
|
||
|
||
if (!isPathInsideDir(workspacePath, resolved)) return { success: false, error: 'Access denied: path escapes workspace' };
|
||
if (!fs.existsSync(resolved)) return { success: false, error: `File not found: ${resolved}` };
|
||
if (path.extname(resolved).toLowerCase() !== '.pdf') return { success: false, error: 'File must have a .pdf extension' };
|
||
|
||
const fmt = ['markdown', 'csv', 'json'].includes(args?.format) ? String(args.format) : 'markdown';
|
||
const engine = ['auto', 'pymupdf', 'pdfplumber'].includes(args?.engine) ? String(args.engine) : 'auto';
|
||
const pageFrom = args?.page_from ? String(Math.max(1, Math.floor(Number(args.page_from)))) : '1';
|
||
const pageTo = args?.page_to ? String(Math.max(1, Math.floor(Number(args.page_to)))) : '9999';
|
||
|
||
let raw: { tables: any[]; errors: string[] };
|
||
try {
|
||
const { stdout } = await execFileAsync(
|
||
'python3', ['-c', TABLE_EXTRACT_PY, resolved, fmt, pageFrom, pageTo, engine],
|
||
{ timeout: 60_000, maxBuffer: 20 * 1024 * 1024 },
|
||
);
|
||
// PyMuPDF가 JSON 앞에 경고 텍스트를 stdout으로 출력할 수 있으므로
|
||
// 첫 번째 '{' 이후만 JSON으로 파싱한다.
|
||
const jsonStart = stdout.indexOf('{');
|
||
raw = JSON.parse(jsonStart >= 0 ? stdout.slice(jsonStart) : stdout);
|
||
} catch (err: any) {
|
||
return { success: false, error: `Table extraction failed: ${String(err.message || err).slice(0, 300)}` };
|
||
}
|
||
|
||
if (!raw.tables || raw.tables.length === 0) {
|
||
const hint = raw.errors?.includes('no_tables')
|
||
? 'No tables detected. If this is a scanned PDF, try pdf_extract_images with mode "figures".'
|
||
: `No tables found. Errors: ${raw.errors?.join('; ') || 'none'}`;
|
||
return { success: false, error: hint };
|
||
}
|
||
|
||
const lines: string[] = [];
|
||
for (const t of raw.tables) {
|
||
lines.push(`### Page ${t.page} — Table ${t.table} (${t.rows} rows × ${t.cols} cols, engine: ${t.engine ?? engine})`);
|
||
lines.push('');
|
||
lines.push(t.data);
|
||
lines.push('');
|
||
}
|
||
|
||
return {
|
||
success: true,
|
||
stdout: lines.join('\n').trimEnd(),
|
||
data: { table_count: raw.tables.length, tables: raw.tables, errors: raw.errors?.length ? raw.errors : undefined },
|
||
};
|
||
},
|
||
};
|