Release 2.9.6: PPTX wizard search + image extraction improvements
- Add Europe PMC search source (free API, life sciences, PMC full-text) - Fix Europe PMC field mapping: journalInfo.journal.title, firstPublicationDate - Add PMC-only filter button with live paper count - Move sort selector to search row; shrink year input (flex:none, no spinners) - Uniform source checkbox sizing with per-source accent colors - Exclude preview/ directory from image grid (project-images endpoint) - Numeric sort for project images (slide1→2→9→10→11) - Skip PDF figure re-extraction if figure_* files already exist (pdf-extract.ts) - Preserve img:done state across wizard prep re-runs (savedImgDone) Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
+23
-15
@@ -408,22 +408,30 @@ export const pdfExtractImagesTool = {
|
||||
|
||||
// --- figures mode: render pages then auto-crop figures/tables with OpenCV ---
|
||||
if (mode === 'figures' || mode === 'both') {
|
||||
const figDpi = Math.min(300, Math.max(100, dpi));
|
||||
const ppmArgs = ['-png', '-r', String(figDpi)];
|
||||
if (pageFrom) ppmArgs.push('-f', String(pageFrom));
|
||||
if (pageTo) ppmArgs.push('-l', String(pageTo));
|
||||
ppmArgs.push(resolved, path.join(outDir, 'page'));
|
||||
try {
|
||||
await execFileAsync('pdftoppm', ppmArgs, { timeout: 120_000 });
|
||||
} catch (err: any) {
|
||||
errors.push(`pdftoppm: ${String(err.message || err).slice(0, 200)}`);
|
||||
// Skip expensive re-extraction if figure files already exist in outDir
|
||||
const existingFigs = fs.existsSync(outDir)
|
||||
? fs.readdirSync(outDir).filter(f => f.startsWith('figure_') && f.endsWith('.png'))
|
||||
: [];
|
||||
if (existingFigs.length > 0 && !pageFrom && !pageTo) {
|
||||
extracted.push(...existingFigs.sort().map(f => `${outDirRel}/${f}`));
|
||||
} else {
|
||||
const figDpi = Math.min(300, Math.max(100, dpi));
|
||||
const ppmArgs = ['-png', '-r', String(figDpi)];
|
||||
if (pageFrom) ppmArgs.push('-f', String(pageFrom));
|
||||
if (pageTo) ppmArgs.push('-l', String(pageTo));
|
||||
ppmArgs.push(resolved, path.join(outDir, 'page'));
|
||||
try {
|
||||
await execFileAsync('pdftoppm', ppmArgs, { timeout: 120_000 });
|
||||
} catch (err: any) {
|
||||
errors.push(`pdftoppm: ${String(err.message || err).slice(0, 200)}`);
|
||||
}
|
||||
const { files: figFiles, errors: figErrors } = await extractFigures(outDir, outDir);
|
||||
errors.push(...figErrors);
|
||||
for (const f of fs.readdirSync(outDir).filter(f => f.startsWith('page-') && f.endsWith('.png'))) {
|
||||
try { fs.unlinkSync(path.join(outDir, f)); } catch {}
|
||||
}
|
||||
extracted.push(...figFiles.map(f => `${outDirRel}/${f}`));
|
||||
}
|
||||
const { files: figFiles, errors: figErrors } = await extractFigures(outDir, outDir);
|
||||
errors.push(...figErrors);
|
||||
for (const f of fs.readdirSync(outDir).filter(f => f.startsWith('page-') && f.endsWith('.png'))) {
|
||||
try { fs.unlinkSync(path.join(outDir, f)); } catch {}
|
||||
}
|
||||
extracted.push(...figFiles.map(f => `${outDirRel}/${f}`));
|
||||
}
|
||||
|
||||
if (extracted.length === 0) {
|
||||
|
||||
Reference in New Issue
Block a user