Release 2.9.5: PPTX wizard improvements + PDF extraction fixes
- PDF: skip references/bibliography section from text extraction
- PDF: add minimal system prompt for translate sessions (translation-only, no tool calls)
- Wizard: image download button per card + bulk download all
- Wizard: translation modal download button (saves as _ko.txt / _en.txt)
- Wizard: save papers.json immediately on upload, after text extraction, and before outline step
- Wizard: short image directory names (22 chars + timestamp suffix)
- Wizard: project-images scans workspace root dir alongside pptx/ folder
- Server: pptx/list scans only pptx/ folder (workspace root scanning removed after file migration)
- create_presentation: save output to workspace/pptx/{slug}/ instead of workspace root
- edit_presentation: fuzzy path search includes pptx/ subfolder
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
+133
-16
@@ -17,6 +17,7 @@ function isPathInsideDir(base: string, target: string): boolean {
|
||||
|
||||
const COLUMN_EXTRACT_SCRIPT = `
|
||||
import sys, fitz, re
|
||||
from collections import Counter
|
||||
|
||||
pdf_path = sys.argv[1]
|
||||
page_from = int(sys.argv[2]) if len(sys.argv) > 2 else 1
|
||||
@@ -26,8 +27,9 @@ doc = fitz.open(pdf_path)
|
||||
total = doc.page_count
|
||||
end = min(page_to, total) if page_to > 0 else total
|
||||
|
||||
LINE_TOL = 4
|
||||
MARGIN = 0.07
|
||||
LINE_TOL = 4
|
||||
MARGIN = 0.07
|
||||
TABLE_GAP = 60 # min px gap between words -> skip as table row
|
||||
|
||||
FURNITURE_RE = re.compile(
|
||||
r'@[\\w.]+\\.|https?://|doi\\.org|\\u00a9|All rights are reserved'
|
||||
@@ -46,6 +48,77 @@ def is_furniture(text, y0, y1, ph):
|
||||
if FURNITURE_RE.search(text): return True
|
||||
return False
|
||||
|
||||
def get_body_size(doc, start, end_p):
|
||||
sizes = []
|
||||
for pi in range(start, min(end_p, start+4)):
|
||||
page = doc[pi]
|
||||
ph = page.rect.height
|
||||
for b in page.get_text("dict")["blocks"]:
|
||||
if b.get("type") != 0: continue
|
||||
if b["bbox"][1] < ph*0.07 or b["bbox"][3] > ph*0.93: continue
|
||||
for ln in b.get("lines", []):
|
||||
for sp in ln.get("spans", []):
|
||||
if len(sp["text"].strip()) > 3:
|
||||
sizes.append(round(sp["size"] * 2) / 2)
|
||||
if not sizes: return 10.0
|
||||
return Counter(sizes).most_common(1)[0][0]
|
||||
|
||||
body_size = get_body_size(doc, page_from-1, end)
|
||||
|
||||
def build_font_map(page, ph):
|
||||
fmap = {}
|
||||
for b in page.get_text("dict")["blocks"]:
|
||||
if b.get("type") != 0: continue
|
||||
if b["bbox"][1] < ph*0.07 or b["bbox"][3] > ph*0.93: continue
|
||||
for ln in b.get("lines", []):
|
||||
spans = [s for s in ln.get("spans", []) if s["text"].strip()]
|
||||
if not spans: continue
|
||||
y = ln["bbox"][1]
|
||||
max_size = max(s["size"] for s in spans)
|
||||
all_bold = all(bool(s["flags"] & 16) for s in spans)
|
||||
matched = next((k for k in fmap if abs(k-y) <= LINE_TOL), None)
|
||||
if matched is not None:
|
||||
prev = fmap[matched]
|
||||
fmap[matched] = (max(prev[0], max_size), prev[1] and all_bold)
|
||||
else:
|
||||
fmap[y] = (max_size, all_bold)
|
||||
return fmap
|
||||
|
||||
def get_font_info(y, fmap):
|
||||
best, bd = None, float('inf')
|
||||
for ky, v in fmap.items():
|
||||
d = abs(ky - y)
|
||||
if d < bd and d <= LINE_TOL*3:
|
||||
bd, best = d, v
|
||||
return best
|
||||
|
||||
def format_line(text, y, fmap):
|
||||
info = get_font_info(y, fmap)
|
||||
if not info: return text
|
||||
max_size, all_bold = info
|
||||
ratio = max_size / body_size if body_size > 0 else 1.0
|
||||
t = text.strip()
|
||||
if ratio >= 1.35:
|
||||
return '## ' + t
|
||||
if ratio >= 1.12 or (all_bold and 3 < len(t) < 100):
|
||||
return '### ' + t
|
||||
if all_bold:
|
||||
return '**' + t + '**'
|
||||
return text
|
||||
|
||||
def col_baseline(words):
|
||||
if not words: return None
|
||||
xs = sorted(w[0] for w in words)
|
||||
return xs[max(0, len(xs)//10)]
|
||||
|
||||
def make_indent(x_start, base_x, col_w):
|
||||
if base_x is None or col_w <= 0: return ''
|
||||
off = x_start - base_x
|
||||
if off < col_w * 0.04: return ''
|
||||
if off < col_w * 0.12: return ' '
|
||||
if off < col_w * 0.25: return ' '
|
||||
return ' '
|
||||
|
||||
def words_to_lines(wlist):
|
||||
wlist.sort(key=lambda w: (w[1], w[0]))
|
||||
groups, cur = [], []
|
||||
@@ -57,18 +130,30 @@ def words_to_lines(wlist):
|
||||
if cur: groups.append(cur)
|
||||
lines = []
|
||||
for g in groups:
|
||||
text = ' '.join(w[4] for w in sorted(g, key=lambda w: w[0]))
|
||||
g_sorted = sorted(g, key=lambda w: w[0])
|
||||
text = ' '.join(w[4] for w in g_sorted)
|
||||
text = re.sub(r'^([A-Z]) ([a-z][a-z])', lambda m: m.group(1)+m.group(2), text)
|
||||
lines.append((g[0][1], text))
|
||||
gaps = [g_sorted[k+1][0] - g_sorted[k][2] for k in range(len(g_sorted)-1)]
|
||||
max_gap = max(gaps) if gaps else 0
|
||||
lines.append((g[0][1], g_sorted[0][0], text, max_gap))
|
||||
return lines
|
||||
|
||||
BULLET_RE = re.compile(r'^[\\u2022\\u00b7]\\s*|^[\\-\\*]\\s+(?=\\S)')
|
||||
REF_HEADING_RE = re.compile(
|
||||
r'^\\s*(References|Bibliography|참고문헌|REFERENCES|BIBLIOGRAPHY|Literature Cited)\\s*$',
|
||||
re.IGNORECASE)
|
||||
|
||||
pages_text = []
|
||||
stop_extraction = False
|
||||
for pi in range(page_from-1, end):
|
||||
if stop_extraction: break
|
||||
page = doc[pi]
|
||||
ph, pw = page.rect.height, page.rect.width
|
||||
mid = pw * 0.52
|
||||
full_w_thr = pw * 0.55
|
||||
|
||||
fmap = build_font_map(page, ph)
|
||||
|
||||
skip_rects = []
|
||||
for b in page.get_text("blocks"):
|
||||
x0,y0,x1,y1,txt = b[0],b[1],b[2],b[3],b[4]
|
||||
@@ -97,19 +182,52 @@ for pi in range(page_from-1, end):
|
||||
elif col=='left': left_w.append((wx0,wy0,wx1,wy1,word))
|
||||
else: right_w.append((wx0,wy0,wx1,wy1,word))
|
||||
|
||||
full_lines = words_to_lines(full_w)
|
||||
left_lines = words_to_lines(left_w)
|
||||
full_base = col_baseline(full_w)
|
||||
left_base = col_baseline(left_w)
|
||||
right_base = col_baseline(right_w)
|
||||
|
||||
full_lines = words_to_lines(full_w)
|
||||
left_lines = words_to_lines(left_w)
|
||||
right_lines = words_to_lines(right_w)
|
||||
|
||||
all_entries = [(y,'F',t) for y,t in full_lines] \
|
||||
+ [(y,'L',t) for y,t in left_lines] \
|
||||
+ [(y,'R',t) for y,t in right_lines]
|
||||
all_entries.sort(key=lambda e: (e[0] if e[1] in ('F','L') else e[0]+10000))
|
||||
pages_text.append('\\n'.join(t for _,_,t in all_entries))
|
||||
all_ys = sorted(e[0] for e in full_lines + left_lines)
|
||||
lh = None
|
||||
if len(all_ys) >= 4:
|
||||
diffs = [all_ys[k+1]-all_ys[k] for k in range(len(all_ys)-1) if 0 < all_ys[k+1]-all_ys[k] < 40]
|
||||
if diffs: lh = sorted(diffs)[len(diffs)//2]
|
||||
|
||||
def render(line_list, base_x, col_w, col_type):
|
||||
out = []
|
||||
prev_y = None
|
||||
for y, x_start, text, max_gap in line_list:
|
||||
if max_gap >= TABLE_GAP:
|
||||
prev_y = None; continue
|
||||
if prev_y is not None and lh and (y - prev_y) > lh * 1.8:
|
||||
out.append((y - 0.5, col_type, ''))
|
||||
t = text.strip()
|
||||
indent = make_indent(x_start, base_x, col_w)
|
||||
if BULLET_RE.match(t):
|
||||
t = BULLET_RE.sub('- ', t)
|
||||
out.append((y, col_type, indent + t))
|
||||
else:
|
||||
out.append((y, col_type, indent + format_line(t, y, fmap)))
|
||||
prev_y = y
|
||||
return out
|
||||
|
||||
all_entries = render(full_lines, full_base, pw * 0.85, 'F')
|
||||
all_entries += render(left_lines, left_base, pw * 0.45, 'L')
|
||||
all_entries += render(right_lines, right_base, pw * 0.45, 'R')
|
||||
all_entries.sort(key=lambda e: e[0] if e[1] in ('F','L') else e[0]+10000)
|
||||
page_lines = [t for _,_,t in all_entries]
|
||||
# Stop at References/Bibliography heading
|
||||
for idx, ln in enumerate(page_lines):
|
||||
if REF_HEADING_RE.match(ln.lstrip('# ').lstrip('*').strip()):
|
||||
page_lines = page_lines[:idx]
|
||||
stop_extraction = True
|
||||
break
|
||||
pages_text.append('\\n'.join(page_lines))
|
||||
|
||||
full = '\\n\\n'.join(pages_text)
|
||||
|
||||
# drop cap 후처리
|
||||
lines = full.split('\\n')
|
||||
result = []
|
||||
i = 0
|
||||
@@ -121,7 +239,6 @@ while i < len(lines):
|
||||
else:
|
||||
result.append(lines[i])
|
||||
i += 1
|
||||
|
||||
print('\\n'.join(result))
|
||||
`;
|
||||
|
||||
@@ -173,7 +290,7 @@ async function runOcr(resolved: string, pageFrom: number, pageTo: number): Promi
|
||||
|
||||
export const pdfReadTool = {
|
||||
name: 'pdf_read',
|
||||
description: 'Extract text from a PDF file. For text-based PDFs uses pdftotext; for scanned/image PDFs automatically falls back to Tesseract OCR (kor+eng). Returns full text, optionally limited to a page range.',
|
||||
description: 'Extract text from a PDF file. Returns markdown-formatted text: headings (##/###), bold (**), indentation, and bullet lists. Multi-column academic PDFs are handled correctly. Complex tables are skipped (use pdf_extract_images for table figures). For scanned/image PDFs falls back to Tesseract OCR.',
|
||||
schema: {
|
||||
path: 'Path to the PDF file (absolute, or relative to workspace)',
|
||||
page_from: 'First page to extract, 1-indexed (default: 1)',
|
||||
@@ -219,7 +336,7 @@ export const pdfReadTool = {
|
||||
let method = 'pymupdf';
|
||||
|
||||
if (!forceOcr) {
|
||||
// Primary: PyMuPDF column-aware extraction (handles 2-column academic PDFs)
|
||||
// Primary: PyMuPDF markdown-aware extraction (headings, bold, indentation, table skip)
|
||||
try {
|
||||
const { stdout } = await execFileAsync(
|
||||
'python3', ['-c', COLUMN_EXTRACT_SCRIPT, resolved, String(pageFrom), String(pageTo)],
|
||||
|
||||
+38
-19
@@ -703,7 +703,7 @@ export const pptxTool: import('./registry.js').Tool = {
|
||||
.replace(/^_+|_+$/g, '')
|
||||
.toLowerCase()
|
||||
.slice(0, 60) || 'presentation';
|
||||
const projectDir = path.join(workspacePath, projectSlug);
|
||||
const projectDir = path.join(workspacePath, 'pptx', projectSlug);
|
||||
if (!fs.existsSync(projectDir)) {
|
||||
fs.mkdirSync(projectDir, { recursive: true });
|
||||
}
|
||||
@@ -751,10 +751,18 @@ export const pptxTool: import('./registry.js').Tool = {
|
||||
const resolvedSrc = fs.existsSync(srcPath) ? srcPath : fs.existsSync(srcPathAlt) ? srcPathAlt : null;
|
||||
if (resolvedSrc) {
|
||||
const ext = path.extname(resolvedSrc) || '.png';
|
||||
const fname = `slide${i + 1}_image${ext}`;
|
||||
fs.copyFileSync(resolvedSrc, path.join(projectDir, fname));
|
||||
slide.image_path = fname;
|
||||
console.log(`[pptx] slide ${i + 1} image_url local copy -> ${fname}`);
|
||||
const resolvedReal = fs.realpathSync(resolvedSrc);
|
||||
const projectReal = fs.realpathSync(projectDir);
|
||||
if (resolvedReal.startsWith(projectReal + path.sep) || resolvedReal.startsWith(projectReal + '/')) {
|
||||
// Already inside project folder — use relative path, no copy needed
|
||||
slide.image_path = path.relative(projectDir, resolvedReal);
|
||||
console.log(`[pptx] slide ${i + 1} image_url already in project -> ${slide.image_path}`);
|
||||
} else {
|
||||
const fname = `slide${i + 1}_image${ext}`;
|
||||
fs.copyFileSync(resolvedSrc, path.join(projectDir, fname));
|
||||
slide.image_path = fname;
|
||||
console.log(`[pptx] slide ${i + 1} image_url local copy -> ${fname}`);
|
||||
}
|
||||
} else {
|
||||
// Last resort: search uploads/ subdirs for a file with the same basename
|
||||
const basename = path.basename(url);
|
||||
@@ -812,11 +820,18 @@ export const pptxTool: import('./registry.js').Tool = {
|
||||
? slide.image_path
|
||||
: path.join(workspacePath, slide.image_path);
|
||||
if (fs.existsSync(inWorkspace)) {
|
||||
const ext = path.extname(slide.image_path) || '.png';
|
||||
const fname = `slide${i + 1}_image${ext}`;
|
||||
fs.copyFileSync(inWorkspace, path.join(projectDir, fname));
|
||||
slide.image_path = fname;
|
||||
console.log(`[pptx] slide ${i + 1} image_path resolved from workspace -> ${fname}`);
|
||||
const resolvedReal = fs.realpathSync(inWorkspace);
|
||||
const projectReal = fs.realpathSync(projectDir);
|
||||
if (resolvedReal.startsWith(projectReal + path.sep) || resolvedReal.startsWith(projectReal + '/')) {
|
||||
slide.image_path = path.relative(projectDir, resolvedReal);
|
||||
console.log(`[pptx] slide ${i + 1} image_path already in project -> ${slide.image_path}`);
|
||||
} else {
|
||||
const ext = path.extname(slide.image_path) || '.png';
|
||||
const fname = `slide${i + 1}_image${ext}`;
|
||||
fs.copyFileSync(inWorkspace, path.join(projectDir, fname));
|
||||
slide.image_path = fname;
|
||||
console.log(`[pptx] slide ${i + 1} image_path resolved from workspace -> ${fname}`);
|
||||
}
|
||||
} else {
|
||||
// Last resort: search uploads/ subdirs for a file with the same basename
|
||||
const basename = path.basename(slide.image_path);
|
||||
@@ -972,15 +987,19 @@ export const editPptxTool: import('./registry.js').Tool = {
|
||||
const givenFolder = path.basename(path.dirname(absPath));
|
||||
let fuzzyMatch: string | null = null;
|
||||
try {
|
||||
for (const entry of fs.readdirSync(workspacePath)) {
|
||||
if (!entry.startsWith(givenFolder) || entry === givenFolder) continue;
|
||||
const candidate = path.join(workspacePath, entry);
|
||||
if (!fs.statSync(candidate).isDirectory()) continue;
|
||||
const pptxFiles = fs.readdirSync(candidate).filter(f => f.toLowerCase().endsWith('.pptx'));
|
||||
if (pptxFiles.length > 0) {
|
||||
fuzzyMatch = path.join(candidate, pptxFiles[0]);
|
||||
console.log(`[pptx] edit_presentation: fuzzy match "${existingPath}" → "${fuzzyMatch}"`);
|
||||
break;
|
||||
const searchDirs = [workspacePath, path.join(workspacePath, 'pptx')];
|
||||
outer: for (const searchDir of searchDirs) {
|
||||
if (!fs.existsSync(searchDir)) continue;
|
||||
for (const entry of fs.readdirSync(searchDir)) {
|
||||
if (!entry.startsWith(givenFolder) || entry === givenFolder) continue;
|
||||
const candidate = path.join(searchDir, entry);
|
||||
if (!fs.statSync(candidate).isDirectory()) continue;
|
||||
const pptxFiles = fs.readdirSync(candidate).filter(f => f.toLowerCase().endsWith('.pptx'));
|
||||
if (pptxFiles.length > 0) {
|
||||
fuzzyMatch = path.join(candidate, pptxFiles[0]);
|
||||
console.log(`[pptx] edit_presentation: fuzzy match "${existingPath}" → "${fuzzyMatch}"`);
|
||||
break outer;
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {}
|
||||
|
||||
Reference in New Issue
Block a user