Release 2.9.6: PPTX wizard search + image extraction improvements

- Add Europe PMC search source (free API, life sciences, PMC full-text)
- Fix Europe PMC field mapping: journalInfo.journal.title, firstPublicationDate
- Add PMC-only filter button with live paper count
- Move sort selector to search row; shrink year input (flex:none, no spinners)
- Uniform source checkbox sizing with per-source accent colors
- Exclude preview/ directory from image grid (project-images endpoint)
- Numeric sort for project images (slide1→2→9→10→11)
- Skip PDF figure re-extraction if figure_* files already exist (pdf-extract.ts)
- Preserve img:done state across wizard prep re-runs (savedImgDone)

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
kim
2026-06-03 14:30:16 +09:00
co-authored by Claude Sonnet 4.6
parent b292061c71
commit 613690da55
4 changed files with 163 additions and 43 deletions
+23 -15
View File
@@ -408,22 +408,30 @@ export const pdfExtractImagesTool = {
// --- figures mode: render pages then auto-crop figures/tables with OpenCV ---
if (mode === 'figures' || mode === 'both') {
const figDpi = Math.min(300, Math.max(100, dpi));
const ppmArgs = ['-png', '-r', String(figDpi)];
if (pageFrom) ppmArgs.push('-f', String(pageFrom));
if (pageTo) ppmArgs.push('-l', String(pageTo));
ppmArgs.push(resolved, path.join(outDir, 'page'));
try {
await execFileAsync('pdftoppm', ppmArgs, { timeout: 120_000 });
} catch (err: any) {
errors.push(`pdftoppm: ${String(err.message || err).slice(0, 200)}`);
// Skip expensive re-extraction if figure files already exist in outDir
const existingFigs = fs.existsSync(outDir)
? fs.readdirSync(outDir).filter(f => f.startsWith('figure_') && f.endsWith('.png'))
: [];
if (existingFigs.length > 0 && !pageFrom && !pageTo) {
extracted.push(...existingFigs.sort().map(f => `${outDirRel}/${f}`));
} else {
const figDpi = Math.min(300, Math.max(100, dpi));
const ppmArgs = ['-png', '-r', String(figDpi)];
if (pageFrom) ppmArgs.push('-f', String(pageFrom));
if (pageTo) ppmArgs.push('-l', String(pageTo));
ppmArgs.push(resolved, path.join(outDir, 'page'));
try {
await execFileAsync('pdftoppm', ppmArgs, { timeout: 120_000 });
} catch (err: any) {
errors.push(`pdftoppm: ${String(err.message || err).slice(0, 200)}`);
}
const { files: figFiles, errors: figErrors } = await extractFigures(outDir, outDir);
errors.push(...figErrors);
for (const f of fs.readdirSync(outDir).filter(f => f.startsWith('page-') && f.endsWith('.png'))) {
try { fs.unlinkSync(path.join(outDir, f)); } catch {}
}
extracted.push(...figFiles.map(f => `${outDirRel}/${f}`));
}
const { files: figFiles, errors: figErrors } = await extractFigures(outDir, outDir);
errors.push(...figErrors);
for (const f of fs.readdirSync(outDir).filter(f => f.startsWith('page-') && f.endsWith('.png'))) {
try { fs.unlinkSync(path.join(outDir, f)); } catch {}
}
extracted.push(...figFiles.map(f => `${outDirRel}/${f}`));
}
if (extracted.length === 0) {