Release 2.9.1: PPTX wizard project persistence + PDF column extraction
- PDF: PyMuPDF 2-column aware extraction (left col → right col, drop cap merge, header/footer/footnote removal), reference stripping, 200k char limit - Wizard: chunked parallel translation (15k chars/chunk, ~4x faster) - Wizard: save UPL_ (upload) papers to papers.json (were silently skipped) - Wizard: restore UPL_ papers from manifest with _isUpload/_localPdf fields - Wizard: project-papers endpoint correctly resolves UPL_ ko.txt in project dir - Wizard: images loaded on manifest restore (renderImgGrid unconditional) - Wizard: outline.json auto-save after generation and slide edits - Wizard: outline.json fallback load when no PPTX exists - Wizard: loadProjList() refresh after savePapersToProject() - Code editor: default font size 14→12px Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
+134
-14
@@ -15,6 +15,116 @@ function isPathInsideDir(base: string, target: string): boolean {
|
||||
return rel !== '' && !rel.startsWith('..') && !path.isAbsolute(rel);
|
||||
}
|
||||
|
||||
const COLUMN_EXTRACT_SCRIPT = `
|
||||
import sys, fitz, re
|
||||
|
||||
pdf_path = sys.argv[1]
|
||||
page_from = int(sys.argv[2]) if len(sys.argv) > 2 else 1
|
||||
page_to = int(sys.argv[3]) if len(sys.argv) > 3 else 0
|
||||
|
||||
doc = fitz.open(pdf_path)
|
||||
total = doc.page_count
|
||||
end = min(page_to, total) if page_to > 0 else total
|
||||
|
||||
LINE_TOL = 4
|
||||
MARGIN = 0.07
|
||||
|
||||
FURNITURE_RE = re.compile(
|
||||
r'@[\\w.]+\\.|https?://|doi\\.org|\\u00a9|All rights are reserved'
|
||||
r'|Submitted,|Revised,|Accepted,|Published:'
|
||||
r'|Correspondence|Address correspondence'
|
||||
r'|Academic Editor|Licensee\\b|open access article'
|
||||
r'|ICMJE|Confl|\\$\\d+\\.\\d+|contributed equally|Potential Confl',
|
||||
re.IGNORECASE)
|
||||
HEADER_RE = re.compile(
|
||||
r'^(American Journal|Dentofacial Orthop|July \\d{4}|Vol \\d+|Issue \\d+|\\d{1,3}\\s*$)',
|
||||
re.IGNORECASE)
|
||||
|
||||
def is_furniture(text, y0, y1, ph):
|
||||
if y0 < ph*MARGIN or y1 > ph*(1-MARGIN): return True
|
||||
if HEADER_RE.search(text.strip()): return True
|
||||
if FURNITURE_RE.search(text): return True
|
||||
return False
|
||||
|
||||
def words_to_lines(wlist):
|
||||
wlist.sort(key=lambda w: (w[1], w[0]))
|
||||
groups, cur = [], []
|
||||
for w in wlist:
|
||||
if not cur or abs(w[1]-cur[-1][1]) <= LINE_TOL:
|
||||
cur.append(w)
|
||||
else:
|
||||
groups.append(cur); cur = [w]
|
||||
if cur: groups.append(cur)
|
||||
lines = []
|
||||
for g in groups:
|
||||
text = ' '.join(w[4] for w in sorted(g, key=lambda w: w[0]))
|
||||
text = re.sub(r'^([A-Z]) ([a-z][a-z])', lambda m: m.group(1)+m.group(2), text)
|
||||
lines.append((g[0][1], text))
|
||||
return lines
|
||||
|
||||
pages_text = []
|
||||
for pi in range(page_from-1, end):
|
||||
page = doc[pi]
|
||||
ph, pw = page.rect.height, page.rect.width
|
||||
mid = pw * 0.52
|
||||
full_w_thr = pw * 0.55
|
||||
|
||||
skip_rects = []
|
||||
for b in page.get_text("blocks"):
|
||||
x0,y0,x1,y1,txt = b[0],b[1],b[2],b[3],b[4]
|
||||
if is_furniture(txt, y0, y1, ph): skip_rects.append((x0,y0,x1,y1))
|
||||
|
||||
def in_skip(wx0,wy0,wx1,wy1):
|
||||
for sx0,sy0,sx1,sy1 in skip_rects:
|
||||
if wx0<sx1 and wx1>sx0 and wy0<sy1 and wy1>sy0: return True
|
||||
return False
|
||||
|
||||
block_info = {}
|
||||
for b in page.get_text("blocks"):
|
||||
bno=b[5]; bx0,by0,bx1,by1=b[0],b[1],b[2],b[3]
|
||||
bw=bx1-bx0
|
||||
block_info[bno]='full' if bw>=full_w_thr else ('left' if (bx0+bx1)/2<mid else 'right')
|
||||
|
||||
full_w, left_w, right_w = [], [], []
|
||||
for w in page.get_text("words"):
|
||||
wx0,wy0,wx1,wy1,word,bno=w[0],w[1],w[2],w[3],w[4],w[5]
|
||||
word=word.strip()
|
||||
if not word: continue
|
||||
if wy0<ph*MARGIN or wy1>ph*(1-MARGIN): continue
|
||||
if in_skip(wx0,wy0,wx1,wy1): continue
|
||||
col=block_info.get(bno,'left' if wx0<mid else 'right')
|
||||
if col=='full': full_w.append((wx0,wy0,wx1,wy1,word))
|
||||
elif col=='left': left_w.append((wx0,wy0,wx1,wy1,word))
|
||||
else: right_w.append((wx0,wy0,wx1,wy1,word))
|
||||
|
||||
full_lines = words_to_lines(full_w)
|
||||
left_lines = words_to_lines(left_w)
|
||||
right_lines = words_to_lines(right_w)
|
||||
|
||||
all_entries = [(y,'F',t) for y,t in full_lines] \
|
||||
+ [(y,'L',t) for y,t in left_lines] \
|
||||
+ [(y,'R',t) for y,t in right_lines]
|
||||
all_entries.sort(key=lambda e: (e[0] if e[1] in ('F','L') else e[0]+10000))
|
||||
pages_text.append('\\n'.join(t for _,_,t in all_entries))
|
||||
|
||||
full = '\\n\\n'.join(pages_text)
|
||||
|
||||
# drop cap 후처리
|
||||
lines = full.split('\\n')
|
||||
result = []
|
||||
i = 0
|
||||
while i < len(lines):
|
||||
ln = lines[i].strip()
|
||||
if re.match(r'^[A-Z]$', ln) and i+1 < len(lines) and lines[i+1] and lines[i+1][0].islower():
|
||||
result.append(ln + lines[i+1])
|
||||
i += 2
|
||||
else:
|
||||
result.append(lines[i])
|
||||
i += 1
|
||||
|
||||
print('\\n'.join(result))
|
||||
`;
|
||||
|
||||
const OCR_SCRIPT = `
|
||||
import sys, os, subprocess, tempfile
|
||||
try:
|
||||
@@ -100,29 +210,39 @@ export const pdfReadTool = {
|
||||
return { success: false, error: 'File must have a .pdf extension' };
|
||||
}
|
||||
|
||||
const maxChars = Math.min(100_000, Math.max(1_000, Number(args?.max_chars ?? 20_000)));
|
||||
const maxChars = Math.min(200_000, Math.max(1_000, Number(args?.max_chars ?? 20_000)));
|
||||
const pageFrom = args?.page_from ? Math.max(1, Math.floor(Number(args.page_from))) : 1;
|
||||
const pageTo = args?.page_to ? Math.max(1, Math.floor(Number(args.page_to))) : 0;
|
||||
const forceOcr = args?.ocr === true;
|
||||
|
||||
let text = '';
|
||||
let method = 'pdftotext';
|
||||
let method = 'pymupdf';
|
||||
|
||||
if (!forceOcr) {
|
||||
const cmdArgs: string[] = ['-layout', '-enc', 'UTF-8'];
|
||||
if (pageFrom > 1) cmdArgs.push('-f', String(pageFrom));
|
||||
if (pageTo > 0) cmdArgs.push('-l', String(pageTo));
|
||||
cmdArgs.push(resolved, '-');
|
||||
|
||||
// Primary: PyMuPDF column-aware extraction (handles 2-column academic PDFs)
|
||||
try {
|
||||
const result = await execFileAsync('pdftotext', cmdArgs, { maxBuffer: 20 * 1024 * 1024, timeout: 30_000 });
|
||||
text = result.stdout.replace(/\r/g, '').trim();
|
||||
} catch (err: any) {
|
||||
const msg = String(err.message || '');
|
||||
if (msg.toLowerCase().includes('encrypt') || msg.toLowerCase().includes('password')) {
|
||||
return { success: false, error: 'PDF is password-protected or encrypted' };
|
||||
const { stdout } = await execFileAsync(
|
||||
'python3', ['-c', COLUMN_EXTRACT_SCRIPT, resolved, String(pageFrom), String(pageTo)],
|
||||
{ maxBuffer: 20 * 1024 * 1024, timeout: 30_000 }
|
||||
);
|
||||
text = stdout.replace(/\r/g, '').trim();
|
||||
} catch (_pyErr) {
|
||||
// Fallback: pdftotext
|
||||
method = 'pdftotext';
|
||||
const cmdArgs: string[] = ['-enc', 'UTF-8'];
|
||||
if (pageFrom > 1) cmdArgs.push('-f', String(pageFrom));
|
||||
if (pageTo > 0) cmdArgs.push('-l', String(pageTo));
|
||||
cmdArgs.push(resolved, '-');
|
||||
try {
|
||||
const result = await execFileAsync('pdftotext', cmdArgs, { maxBuffer: 20 * 1024 * 1024, timeout: 30_000 });
|
||||
text = result.stdout.replace(/\r/g, '').trim();
|
||||
} catch (err: any) {
|
||||
const msg = String(err.message || '');
|
||||
if (msg.toLowerCase().includes('encrypt') || msg.toLowerCase().includes('password')) {
|
||||
return { success: false, error: 'PDF is password-protected or encrypted' };
|
||||
}
|
||||
return { success: false, error: `pdftotext failed: ${msg}` };
|
||||
}
|
||||
return { success: false, error: `pdftotext failed: ${msg}` };
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user