Release 2.9.5: PPTX wizard improvements + PDF extraction fixes

- PDF: skip references/bibliography section from text extraction
- PDF: add minimal system prompt for translate sessions (translation-only, no tool calls)
- Wizard: image download button per card + bulk download all
- Wizard: translation modal download button (saves as _ko.txt / _en.txt)
- Wizard: save papers.json immediately on upload, after text extraction, and before outline step
- Wizard: short image directory names (22 chars + timestamp suffix)
- Wizard: project-images scans workspace root dir alongside pptx/ folder
- Server: pptx/list scans only pptx/ folder (workspace root scanning removed after file migration)
- create_presentation: save output to workspace/pptx/{slug}/ instead of workspace root
- edit_presentation: fuzzy path search includes pptx/ subfolder

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
kim
2026-06-02 23:10:40 +09:00
co-authored by Claude Sonnet 4.6
parent bba06bab6f
commit b292061c71
4 changed files with 259 additions and 50 deletions
+133 -16
View File
@@ -17,6 +17,7 @@ function isPathInsideDir(base: string, target: string): boolean {
const COLUMN_EXTRACT_SCRIPT = `
import sys, fitz, re
from collections import Counter
pdf_path = sys.argv[1]
page_from = int(sys.argv[2]) if len(sys.argv) > 2 else 1
@@ -26,8 +27,9 @@ doc = fitz.open(pdf_path)
total = doc.page_count
end = min(page_to, total) if page_to > 0 else total
LINE_TOL = 4
MARGIN = 0.07
LINE_TOL = 4
MARGIN = 0.07
TABLE_GAP = 60 # min px gap between words -> skip as table row
FURNITURE_RE = re.compile(
r'@[\\w.]+\\.|https?://|doi\\.org|\\u00a9|All rights are reserved'
@@ -46,6 +48,77 @@ def is_furniture(text, y0, y1, ph):
if FURNITURE_RE.search(text): return True
return False
def get_body_size(doc, start, end_p):
sizes = []
for pi in range(start, min(end_p, start+4)):
page = doc[pi]
ph = page.rect.height
for b in page.get_text("dict")["blocks"]:
if b.get("type") != 0: continue
if b["bbox"][1] < ph*0.07 or b["bbox"][3] > ph*0.93: continue
for ln in b.get("lines", []):
for sp in ln.get("spans", []):
if len(sp["text"].strip()) > 3:
sizes.append(round(sp["size"] * 2) / 2)
if not sizes: return 10.0
return Counter(sizes).most_common(1)[0][0]
body_size = get_body_size(doc, page_from-1, end)
def build_font_map(page, ph):
fmap = {}
for b in page.get_text("dict")["blocks"]:
if b.get("type") != 0: continue
if b["bbox"][1] < ph*0.07 or b["bbox"][3] > ph*0.93: continue
for ln in b.get("lines", []):
spans = [s for s in ln.get("spans", []) if s["text"].strip()]
if not spans: continue
y = ln["bbox"][1]
max_size = max(s["size"] for s in spans)
all_bold = all(bool(s["flags"] & 16) for s in spans)
matched = next((k for k in fmap if abs(k-y) <= LINE_TOL), None)
if matched is not None:
prev = fmap[matched]
fmap[matched] = (max(prev[0], max_size), prev[1] and all_bold)
else:
fmap[y] = (max_size, all_bold)
return fmap
def get_font_info(y, fmap):
best, bd = None, float('inf')
for ky, v in fmap.items():
d = abs(ky - y)
if d < bd and d <= LINE_TOL*3:
bd, best = d, v
return best
def format_line(text, y, fmap):
info = get_font_info(y, fmap)
if not info: return text
max_size, all_bold = info
ratio = max_size / body_size if body_size > 0 else 1.0
t = text.strip()
if ratio >= 1.35:
return '## ' + t
if ratio >= 1.12 or (all_bold and 3 < len(t) < 100):
return '### ' + t
if all_bold:
return '**' + t + '**'
return text
def col_baseline(words):
if not words: return None
xs = sorted(w[0] for w in words)
return xs[max(0, len(xs)//10)]
def make_indent(x_start, base_x, col_w):
if base_x is None or col_w <= 0: return ''
off = x_start - base_x
if off < col_w * 0.04: return ''
if off < col_w * 0.12: return ' '
if off < col_w * 0.25: return ' '
return ' '
def words_to_lines(wlist):
wlist.sort(key=lambda w: (w[1], w[0]))
groups, cur = [], []
@@ -57,18 +130,30 @@ def words_to_lines(wlist):
if cur: groups.append(cur)
lines = []
for g in groups:
text = ' '.join(w[4] for w in sorted(g, key=lambda w: w[0]))
g_sorted = sorted(g, key=lambda w: w[0])
text = ' '.join(w[4] for w in g_sorted)
text = re.sub(r'^([A-Z]) ([a-z][a-z])', lambda m: m.group(1)+m.group(2), text)
lines.append((g[0][1], text))
gaps = [g_sorted[k+1][0] - g_sorted[k][2] for k in range(len(g_sorted)-1)]
max_gap = max(gaps) if gaps else 0
lines.append((g[0][1], g_sorted[0][0], text, max_gap))
return lines
BULLET_RE = re.compile(r'^[\\u2022\\u00b7]\\s*|^[\\-\\*]\\s+(?=\\S)')
REF_HEADING_RE = re.compile(
r'^\\s*(References|Bibliography|참고문헌|REFERENCES|BIBLIOGRAPHY|Literature Cited)\\s*$',
re.IGNORECASE)
pages_text = []
stop_extraction = False
for pi in range(page_from-1, end):
if stop_extraction: break
page = doc[pi]
ph, pw = page.rect.height, page.rect.width
mid = pw * 0.52
full_w_thr = pw * 0.55
fmap = build_font_map(page, ph)
skip_rects = []
for b in page.get_text("blocks"):
x0,y0,x1,y1,txt = b[0],b[1],b[2],b[3],b[4]
@@ -97,19 +182,52 @@ for pi in range(page_from-1, end):
elif col=='left': left_w.append((wx0,wy0,wx1,wy1,word))
else: right_w.append((wx0,wy0,wx1,wy1,word))
full_lines = words_to_lines(full_w)
left_lines = words_to_lines(left_w)
full_base = col_baseline(full_w)
left_base = col_baseline(left_w)
right_base = col_baseline(right_w)
full_lines = words_to_lines(full_w)
left_lines = words_to_lines(left_w)
right_lines = words_to_lines(right_w)
all_entries = [(y,'F',t) for y,t in full_lines] \
+ [(y,'L',t) for y,t in left_lines] \
+ [(y,'R',t) for y,t in right_lines]
all_entries.sort(key=lambda e: (e[0] if e[1] in ('F','L') else e[0]+10000))
pages_text.append('\\n'.join(t for _,_,t in all_entries))
all_ys = sorted(e[0] for e in full_lines + left_lines)
lh = None
if len(all_ys) >= 4:
diffs = [all_ys[k+1]-all_ys[k] for k in range(len(all_ys)-1) if 0 < all_ys[k+1]-all_ys[k] < 40]
if diffs: lh = sorted(diffs)[len(diffs)//2]
def render(line_list, base_x, col_w, col_type):
out = []
prev_y = None
for y, x_start, text, max_gap in line_list:
if max_gap >= TABLE_GAP:
prev_y = None; continue
if prev_y is not None and lh and (y - prev_y) > lh * 1.8:
out.append((y - 0.5, col_type, ''))
t = text.strip()
indent = make_indent(x_start, base_x, col_w)
if BULLET_RE.match(t):
t = BULLET_RE.sub('- ', t)
out.append((y, col_type, indent + t))
else:
out.append((y, col_type, indent + format_line(t, y, fmap)))
prev_y = y
return out
all_entries = render(full_lines, full_base, pw * 0.85, 'F')
all_entries += render(left_lines, left_base, pw * 0.45, 'L')
all_entries += render(right_lines, right_base, pw * 0.45, 'R')
all_entries.sort(key=lambda e: e[0] if e[1] in ('F','L') else e[0]+10000)
page_lines = [t for _,_,t in all_entries]
# Stop at References/Bibliography heading
for idx, ln in enumerate(page_lines):
if REF_HEADING_RE.match(ln.lstrip('# ').lstrip('*').strip()):
page_lines = page_lines[:idx]
stop_extraction = True
break
pages_text.append('\\n'.join(page_lines))
full = '\\n\\n'.join(pages_text)
# drop cap 후처리
lines = full.split('\\n')
result = []
i = 0
@@ -121,7 +239,6 @@ while i < len(lines):
else:
result.append(lines[i])
i += 1
print('\\n'.join(result))
`;
@@ -173,7 +290,7 @@ async function runOcr(resolved: string, pageFrom: number, pageTo: number): Promi
export const pdfReadTool = {
name: 'pdf_read',
description: 'Extract text from a PDF file. For text-based PDFs uses pdftotext; for scanned/image PDFs automatically falls back to Tesseract OCR (kor+eng). Returns full text, optionally limited to a page range.',
description: 'Extract text from a PDF file. Returns markdown-formatted text: headings (##/###), bold (**), indentation, and bullet lists. Multi-column academic PDFs are handled correctly. Complex tables are skipped (use pdf_extract_images for table figures). For scanned/image PDFs falls back to Tesseract OCR.',
schema: {
path: 'Path to the PDF file (absolute, or relative to workspace)',
page_from: 'First page to extract, 1-indexed (default: 1)',
@@ -219,7 +336,7 @@ export const pdfReadTool = {
let method = 'pymupdf';
if (!forceOcr) {
// Primary: PyMuPDF column-aware extraction (handles 2-column academic PDFs)
// Primary: PyMuPDF markdown-aware extraction (headings, bold, indentation, table skip)
try {
const { stdout } = await execFileAsync(
'python3', ['-c', COLUMN_EXTRACT_SCRIPT, resolved, String(pageFrom), String(pageTo)],
+38 -19
View File
@@ -703,7 +703,7 @@ export const pptxTool: import('./registry.js').Tool = {
.replace(/^_+|_+$/g, '')
.toLowerCase()
.slice(0, 60) || 'presentation';
const projectDir = path.join(workspacePath, projectSlug);
const projectDir = path.join(workspacePath, 'pptx', projectSlug);
if (!fs.existsSync(projectDir)) {
fs.mkdirSync(projectDir, { recursive: true });
}
@@ -751,10 +751,18 @@ export const pptxTool: import('./registry.js').Tool = {
const resolvedSrc = fs.existsSync(srcPath) ? srcPath : fs.existsSync(srcPathAlt) ? srcPathAlt : null;
if (resolvedSrc) {
const ext = path.extname(resolvedSrc) || '.png';
const fname = `slide${i + 1}_image${ext}`;
fs.copyFileSync(resolvedSrc, path.join(projectDir, fname));
slide.image_path = fname;
console.log(`[pptx] slide ${i + 1} image_url local copy -> ${fname}`);
const resolvedReal = fs.realpathSync(resolvedSrc);
const projectReal = fs.realpathSync(projectDir);
if (resolvedReal.startsWith(projectReal + path.sep) || resolvedReal.startsWith(projectReal + '/')) {
// Already inside project folder — use relative path, no copy needed
slide.image_path = path.relative(projectDir, resolvedReal);
console.log(`[pptx] slide ${i + 1} image_url already in project -> ${slide.image_path}`);
} else {
const fname = `slide${i + 1}_image${ext}`;
fs.copyFileSync(resolvedSrc, path.join(projectDir, fname));
slide.image_path = fname;
console.log(`[pptx] slide ${i + 1} image_url local copy -> ${fname}`);
}
} else {
// Last resort: search uploads/ subdirs for a file with the same basename
const basename = path.basename(url);
@@ -812,11 +820,18 @@ export const pptxTool: import('./registry.js').Tool = {
? slide.image_path
: path.join(workspacePath, slide.image_path);
if (fs.existsSync(inWorkspace)) {
const ext = path.extname(slide.image_path) || '.png';
const fname = `slide${i + 1}_image${ext}`;
fs.copyFileSync(inWorkspace, path.join(projectDir, fname));
slide.image_path = fname;
console.log(`[pptx] slide ${i + 1} image_path resolved from workspace -> ${fname}`);
const resolvedReal = fs.realpathSync(inWorkspace);
const projectReal = fs.realpathSync(projectDir);
if (resolvedReal.startsWith(projectReal + path.sep) || resolvedReal.startsWith(projectReal + '/')) {
slide.image_path = path.relative(projectDir, resolvedReal);
console.log(`[pptx] slide ${i + 1} image_path already in project -> ${slide.image_path}`);
} else {
const ext = path.extname(slide.image_path) || '.png';
const fname = `slide${i + 1}_image${ext}`;
fs.copyFileSync(inWorkspace, path.join(projectDir, fname));
slide.image_path = fname;
console.log(`[pptx] slide ${i + 1} image_path resolved from workspace -> ${fname}`);
}
} else {
// Last resort: search uploads/ subdirs for a file with the same basename
const basename = path.basename(slide.image_path);
@@ -972,15 +987,19 @@ export const editPptxTool: import('./registry.js').Tool = {
const givenFolder = path.basename(path.dirname(absPath));
let fuzzyMatch: string | null = null;
try {
for (const entry of fs.readdirSync(workspacePath)) {
if (!entry.startsWith(givenFolder) || entry === givenFolder) continue;
const candidate = path.join(workspacePath, entry);
if (!fs.statSync(candidate).isDirectory()) continue;
const pptxFiles = fs.readdirSync(candidate).filter(f => f.toLowerCase().endsWith('.pptx'));
if (pptxFiles.length > 0) {
fuzzyMatch = path.join(candidate, pptxFiles[0]);
console.log(`[pptx] edit_presentation: fuzzy match "${existingPath}" → "${fuzzyMatch}"`);
break;
const searchDirs = [workspacePath, path.join(workspacePath, 'pptx')];
outer: for (const searchDir of searchDirs) {
if (!fs.existsSync(searchDir)) continue;
for (const entry of fs.readdirSync(searchDir)) {
if (!entry.startsWith(givenFolder) || entry === givenFolder) continue;
const candidate = path.join(searchDir, entry);
if (!fs.statSync(candidate).isDirectory()) continue;
const pptxFiles = fs.readdirSync(candidate).filter(f => f.toLowerCase().endsWith('.pptx'));
if (pptxFiles.length > 0) {
fuzzyMatch = path.join(candidate, pptxFiles[0]);
console.log(`[pptx] edit_presentation: fuzzy match "${existingPath}" → "${fuzzyMatch}"`);
break outer;
}
}
}
} catch {}