438 lines
16 KiB
Python
438 lines
16 KiB
Python
#!/usr/bin/env python3
|
|
"""pptx_preview.py — Convert PPTX slides to PNG images for preview.
|
|
|
|
Usage: python pptx_preview.py <pptx_path> <output_dir>
|
|
|
|
Output: PNG images saved to <output_dir>/slide_1.png, slide_2.png, ...
|
|
Stdout: JSON {"success":true,"images":["slide_1.png",...],"count":N}
|
|
or {"success":false,"error":"..."}
|
|
|
|
Strategy:
|
|
1. PPTX → PDF → PNGs via LibreOffice + PyMuPDF (best quality)
|
|
2. Fallback: direct image export via LibreOffice
|
|
3. Fallback: python-pptx + Pillow (text/card-based previews, no LibreOffice needed)
|
|
"""
|
|
|
|
import sys
|
|
import json
|
|
import os
|
|
import subprocess
|
|
import shutil
|
|
import time
|
|
import tempfile
|
|
|
|
# ─── Paths ──────────────────────────────────────────────────────────────────────
|
|
|
|
SOFFICE = r"C:\Program Files\LibreOffice\program\soffice.exe"
|
|
if not os.path.exists(SOFFICE):
|
|
for alt in [r"C:\Program Files (x86)\LibreOffice\program\soffice.exe"]:
|
|
if os.path.exists(alt):
|
|
SOFFICE = alt
|
|
break
|
|
|
|
HAS_LIBREOFFICE = os.path.exists(SOFFICE)
|
|
|
|
|
|
def log(msg):
|
|
"""Log diagnostic message to stderr."""
|
|
print(f"[preview] {msg}", file=sys.stderr, flush=True)
|
|
|
|
|
|
def wait_for_file(filepath, timeout=5, interval=0.2):
|
|
"""Wait until a file exists and is readable."""
|
|
start = time.time()
|
|
while time.time() - start < timeout:
|
|
if os.path.exists(filepath):
|
|
try:
|
|
with open(filepath, 'rb') as f:
|
|
f.read(1)
|
|
return True
|
|
except (IOError, OSError):
|
|
pass
|
|
time.sleep(interval)
|
|
return False
|
|
|
|
|
|
def convert_pptx_to_pdf(pptx_path: str, output_dir: str) -> str:
|
|
"""Convert PPTX to PDF using LibreOffice. Returns PDF path."""
|
|
# Use a unique profile dir to avoid LibreOffice lock contention
|
|
profile_dir = os.path.join(tempfile.gettempdir(), f"lo_preview_{os.getpid()}_{int(time.time())}")
|
|
os.makedirs(profile_dir, exist_ok=True)
|
|
try:
|
|
result = subprocess.run(
|
|
[
|
|
SOFFICE,
|
|
"--headless",
|
|
"--convert-to", "pdf",
|
|
"--outdir", output_dir,
|
|
"-env:UserInstallation=file:///" + profile_dir.replace("\\", "/"),
|
|
pptx_path,
|
|
],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=60,
|
|
)
|
|
if result.returncode != 0:
|
|
log(f"LibreOffice stderr: {result.stderr[:500]}")
|
|
log(f"LibreOffice stdout: {result.stdout[:500]}")
|
|
raise RuntimeError(f"LibreOffice exited with code {result.returncode}")
|
|
|
|
# LibreOffice names the PDF after the input file
|
|
base = os.path.splitext(os.path.basename(pptx_path))[0]
|
|
pdf_path = os.path.join(output_dir, base + ".pdf")
|
|
if not os.path.exists(pdf_path):
|
|
for f in os.listdir(output_dir):
|
|
if f.lower().endswith(".pdf"):
|
|
pdf_path = os.path.join(output_dir, f)
|
|
break
|
|
if not os.path.exists(pdf_path):
|
|
raise RuntimeError("LibreOffice did not create a PDF file")
|
|
return pdf_path
|
|
finally:
|
|
try:
|
|
shutil.rmtree(profile_dir, ignore_errors=True)
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def convert_pdf_to_images(pdf_path: str, output_dir: str, dpi: int = 150) -> list:
|
|
"""Convert PDF pages to PNG images using PyMuPDF. Returns list of filenames."""
|
|
import fitz
|
|
|
|
# Disable MuPDF error/warning messages (they print to stdout and corrupt JSON output)
|
|
try:
|
|
fitz.set_messages()
|
|
except (TypeError, AttributeError):
|
|
pass
|
|
|
|
try:
|
|
doc = fitz.open(pdf_path)
|
|
except Exception as e:
|
|
# Handle structure tree errors and other PDF issues by raising to trigger fallback
|
|
error_msg = str(e).lower()
|
|
if "structure tree" in error_msg or "no common ancestor" in error_msg:
|
|
raise RuntimeError(f"PDF structure tree error: {e}")
|
|
raise
|
|
log(f"PDF has {doc.page_count} pages")
|
|
images = []
|
|
for i in range(doc.page_count):
|
|
try:
|
|
page = doc[i]
|
|
pix = page.get_pixmap(dpi=dpi)
|
|
filename = f"slide_{i + 1}.png"
|
|
out_path = os.path.join(output_dir, filename)
|
|
pix.save(out_path)
|
|
images.append(filename)
|
|
log(f" Saved page {i+1}/{doc.page_count}")
|
|
except Exception as e:
|
|
log(f" Failed page {i+1}: {e}")
|
|
doc.close()
|
|
return images
|
|
|
|
|
|
def convert_pptx_to_images_direct(pptx_path: str, output_dir: str) -> list:
|
|
"""Fallback: try LibreOffice direct image export."""
|
|
profile_dir = os.path.join(tempfile.gettempdir(), f"lo_direct_{os.getpid()}_{int(time.time())}")
|
|
os.makedirs(profile_dir, exist_ok=True)
|
|
try:
|
|
result = subprocess.run(
|
|
[
|
|
SOFFICE,
|
|
"--headless",
|
|
"--convert-to", "png",
|
|
"--outdir", output_dir,
|
|
"-env:UserInstallation=file:///" + profile_dir.replace("\\", "/"),
|
|
pptx_path,
|
|
],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=60,
|
|
)
|
|
images = []
|
|
for f in sorted(os.listdir(output_dir)):
|
|
if f.lower().endswith(".png"):
|
|
images.append(f)
|
|
return images
|
|
finally:
|
|
try:
|
|
shutil.rmtree(profile_dir, ignore_errors=True)
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def _has_cjk(text: str) -> bool:
|
|
"""Detect CJK (Chinese/Japanese/Korean) characters in text."""
|
|
for ch in text:
|
|
cp = ord(ch)
|
|
if (0x4E00 <= cp <= 0x9FFF or # CJK Unified
|
|
0xAC00 <= cp <= 0xD7AF or # Hangul Syllables
|
|
0x3040 <= cp <= 0x309F or # Hiragana
|
|
0x30A0 <= cp <= 0x30FF or # Katakana
|
|
0xFF00 <= cp <= 0xFFEF): # Fullwidth Forms
|
|
return True
|
|
return False
|
|
|
|
|
|
def _get_font(size: int = 24, bold: bool = False, text: str = ""):
|
|
"""Try to load a system font, fall back to default.
|
|
|
|
If text contains CJK characters, prefer CJK-capable fonts
|
|
(malgun, msgothic, Yu Gothic) over Latin-only fonts (arial).
|
|
"""
|
|
from PIL import ImageFont
|
|
|
|
cjk = _has_cjk(text) if text else False
|
|
|
|
# CJK-capable fonts first when CJK text is detected
|
|
if cjk:
|
|
cjk_candidates = [
|
|
"malgunbd.ttf" if bold else "malgun.ttf",
|
|
"C:\\Windows\\Fonts\\malgunbd.ttf" if bold else "C:\\Windows\\Fonts\\malgun.ttf",
|
|
"malgunsl.ttf", "C:\\Windows\\Fonts\\malgunsl.ttf",
|
|
"msgothic.ttc", "C:\\Windows\\Fonts\\msgothic.ttc",
|
|
"YuGothB.ttc" if bold else "YuGothR.ttc",
|
|
"C:\\Windows\\Fonts\\YuGothB.ttc" if bold else "C:\\Windows\\Fonts\\YuGothR.ttc",
|
|
"meiryo.ttc", "C:\\Windows\\Fonts\\meiryo.ttc",
|
|
"NotoSansCJK-Regular.ttc",
|
|
]
|
|
for name in cjk_candidates:
|
|
try:
|
|
return ImageFont.truetype(name, size)
|
|
except (IOError, OSError):
|
|
continue
|
|
|
|
# Latin / general fonts
|
|
candidates = [
|
|
"arialbd.ttf" if bold else "arial.ttf",
|
|
"Arial Bold.ttf" if bold else "Arial.ttf",
|
|
"arial.ttf", "Arial.ttf",
|
|
"calibrib.ttf" if bold else "calibri.ttf",
|
|
"Calibri Bold.ttf" if bold else "Calibri.ttf",
|
|
"DejaVuSans-Bold.ttf" if bold else "DejaVuSans.ttf",
|
|
"consolab.ttf" if bold else "consola.ttf",
|
|
"C:\\Windows\\Fonts\\arialbd.ttf" if bold else "C:\\Windows\\Fonts\\arial.ttf",
|
|
"C:\\Windows\\Fonts\\calibri.ttf",
|
|
"C:\\Windows\\Fonts\\consola.ttf",
|
|
]
|
|
# Fallback to CJK fonts even for Latin if nothing else works
|
|
if not cjk:
|
|
candidates += [
|
|
"malgun.ttf", "C:\\Windows\\Fonts\\malgun.ttf",
|
|
]
|
|
for name in candidates:
|
|
try:
|
|
return ImageFont.truetype(name, size)
|
|
except (IOError, OSError):
|
|
continue
|
|
return ImageFont.load_default()
|
|
|
|
|
|
def _fit_dimensions(img_w: int, img_h: int, max_w: int, max_h: int):
|
|
"""Calculate display dimensions that fit within max_w x max_h."""
|
|
ratio = img_w / img_h
|
|
box_ratio = max_w / max_h
|
|
if ratio > box_ratio:
|
|
return int(max_w), int(max_w / ratio)
|
|
else:
|
|
return int(max_h * ratio), int(max_h)
|
|
|
|
|
|
def generate_card_preview(pptx_path: str, output_dir: str) -> list:
|
|
"""Generate card-based slide previews using python-pptx + Pillow.
|
|
This is the fallback when LibreOffice is not available.
|
|
Produces a simple preview card for each slide showing title, key content, and images.
|
|
"""
|
|
from pptx import Presentation
|
|
from PIL import Image as PILImage, ImageDraw, ImageFont
|
|
import io
|
|
|
|
prs = Presentation(pptx_path)
|
|
images = []
|
|
W, H = 960, 540 # 16:9 aspect ratio
|
|
|
|
# Color palette matching the PPTX template defaults
|
|
BG_LIGHT = (255, 255, 255)
|
|
BG_DARK = (31, 36, 45)
|
|
TITLE_COLOR = (26, 26, 46)
|
|
SUBTITLE_COLOR = (95, 111, 134)
|
|
BODY_COLOR = (45, 55, 72)
|
|
ACCENT_COLOR = (22, 104, 227)
|
|
SECTION_BG = (22, 104, 227)
|
|
SECTION_TITLE_COLOR = (255, 255, 255)
|
|
|
|
for i, slide in enumerate(prs.slides):
|
|
# Detect slide type from layout name
|
|
layout_name = (slide.slide_layout.name or "").lower() if slide.slide_layout else ""
|
|
is_title = "title" in layout_name
|
|
is_section = "section" in layout_name
|
|
|
|
# Extract text and images from slide
|
|
slide_w = prs.slide_width or 1
|
|
slide_h = prs.slide_height or 1
|
|
texts = []
|
|
slide_images = []
|
|
for shape in slide.shapes:
|
|
if shape.has_text_frame:
|
|
for para in shape.text_frame.paragraphs:
|
|
text = para.text.strip()
|
|
if text:
|
|
texts.append(text)
|
|
# Extract embedded images (PICTURE shape_type == 13)
|
|
# Skip full-slide background images (skin/wallpaper covers ≥90% of slide)
|
|
if shape.shape_type == 13:
|
|
is_bg = (shape.width >= slide_w * 0.9 and shape.height >= slide_h * 0.9)
|
|
if is_bg:
|
|
continue
|
|
try:
|
|
image = shape.image
|
|
pil_img = PILImage.open(io.BytesIO(image.blob))
|
|
slide_images.append(pil_img)
|
|
except Exception:
|
|
pass
|
|
|
|
title = texts[0] if texts else f"Slide {i + 1}"
|
|
body_lines = texts[1:] if len(texts) > 1 else []
|
|
|
|
# Determine colors
|
|
if is_section:
|
|
bg = SECTION_BG
|
|
title_color = SECTION_TITLE_COLOR
|
|
accent_color = SECTION_TITLE_COLOR
|
|
else:
|
|
bg = BG_LIGHT
|
|
title_color = TITLE_COLOR
|
|
accent_color = ACCENT_COLOR
|
|
|
|
# Determine layout: side-by-side if images present
|
|
has_images = len(slide_images) > 0
|
|
text_width = 520 if has_images else 864
|
|
img_x = 600 if has_images else W
|
|
|
|
img = PILImage.new("RGB", (W, H), bg)
|
|
draw = ImageDraw.Draw(img)
|
|
|
|
# Draw accent line (for non-section slides)
|
|
if not is_section:
|
|
draw.rectangle([48, 80, 160, 84], fill=accent_color)
|
|
|
|
# Title
|
|
display_title = title[:80] + ("..." if len(title) > 80 else "")
|
|
title_font = _get_font(32, bold=True, text=display_title)
|
|
draw.text((48, 24), display_title, fill=title_color, font=title_font)
|
|
|
|
# Body / bullet lines
|
|
y = 100
|
|
if body_lines:
|
|
body_font = _get_font(16, text=" ".join(body_lines[:3]))
|
|
bullet_font = _get_font(16)
|
|
max_lines = 6 if has_images else 10
|
|
for line_idx, line in enumerate(body_lines[:max_lines]):
|
|
display_line = line[:80] + ("..." if len(line) > 80 else "")
|
|
# Draw bullet
|
|
draw.text((72, y), "●", fill=accent_color, font=bullet_font)
|
|
draw.text((96, y), display_line, fill=BODY_COLOR if not is_section else SECTION_TITLE_COLOR, font=body_font)
|
|
y += 28
|
|
if len(body_lines) > max_lines:
|
|
draw.text((72, y), f"... +{len(body_lines) - max_lines} more", fill=SUBTITLE_COLOR, font=_get_font(14))
|
|
y += 28
|
|
|
|
# Draw images on the right side
|
|
if slide_images:
|
|
img_y = 80
|
|
remaining_h = H - img_y - 40
|
|
per_img_h = remaining_h // len(slide_images[:2])
|
|
for pil_img in slide_images[:2]:
|
|
fit_w, fit_h = _fit_dimensions(pil_img.width, pil_img.height, 340, per_img_h)
|
|
thumb = pil_img.convert("RGB").resize((fit_w, fit_h), PILImage.LANCZOS)
|
|
paste_x = img_x + (340 - fit_w) // 2
|
|
img.paste(thumb, (paste_x, img_y))
|
|
img_y += fit_h + 12
|
|
|
|
# Slide number
|
|
num_font = _get_font(12)
|
|
draw.text((W - 60, H - 30), str(i + 1), fill=SUBTITLE_COLOR, font=num_font)
|
|
|
|
filename = f"slide_{i + 1}.png"
|
|
img.save(os.path.join(output_dir, filename))
|
|
images.append(filename)
|
|
|
|
return images
|
|
|
|
|
|
def generate_preview(pptx_path: str, output_dir: str) -> dict:
|
|
"""Generate PNG previews of PPTX slides."""
|
|
if not os.path.exists(pptx_path):
|
|
return {"success": False, "error": f"File not found: {pptx_path}"}
|
|
|
|
os.makedirs(output_dir, exist_ok=True)
|
|
|
|
# Clean any existing preview images
|
|
for f in os.listdir(output_dir):
|
|
if f.lower().endswith((".png", ".pdf")):
|
|
os.remove(os.path.join(output_dir, f))
|
|
|
|
# Wait for PPTX file to be fully written (handles race condition)
|
|
if not wait_for_file(pptx_path, timeout=5):
|
|
return {"success": False, "error": f"PPTX file not ready: {pptx_path}"}
|
|
|
|
images = []
|
|
errors = []
|
|
|
|
# Strategy 1: PPTX → PDF → PNG (best quality, requires LibreOffice + PyMuPDF)
|
|
if HAS_LIBREOFFICE:
|
|
try:
|
|
log("Trying LibreOffice PDF route...")
|
|
pdf_path = convert_pptx_to_pdf(pptx_path, output_dir)
|
|
log(f"PDF created: {pdf_path}")
|
|
try:
|
|
images = convert_pdf_to_images(pdf_path, output_dir, dpi=150)
|
|
log(f"PDF → PNG success: {len(images)} images")
|
|
except ImportError:
|
|
log("PyMuPDF not available, will try direct export")
|
|
except Exception as e:
|
|
errors.append(f"PyMuPDF failed: {str(e)[:200]}")
|
|
log(f"PyMuPDF failed: {e}")
|
|
# Clean up intermediate PDF
|
|
try:
|
|
if os.path.exists(pdf_path):
|
|
os.remove(pdf_path)
|
|
except OSError:
|
|
pass
|
|
except Exception as e:
|
|
errors.append(f"PDF route failed: {str(e)[:200]}")
|
|
log(f"LibreOffice PDF failed: {e}")
|
|
|
|
# Strategy 2: SKIP — LibreOffice --convert-to png only exports the first slide,
|
|
# producing a single-image preview. Fall through to Strategy 3 instead.
|
|
|
|
# Strategy 3: python-pptx + Pillow card-based preview (no external deps)
|
|
if not images:
|
|
try:
|
|
log("Falling back to card-based preview...")
|
|
images = generate_card_preview(pptx_path, output_dir)
|
|
log(f"Card preview success: {len(images)} images")
|
|
except Exception as e:
|
|
errors.append(f"Card preview failed: {str(e)[:200]}")
|
|
log(f"Card preview failed: {e}")
|
|
|
|
if not images:
|
|
return {"success": False, "error": "; ".join(errors) if errors else "No preview method available"}
|
|
|
|
return {"success": True, "images": images, "count": len(images)}
|
|
|
|
|
|
def main():
|
|
if len(sys.argv) < 3:
|
|
print(json.dumps({"success": False, "error": "Usage: python pptx_preview.py <pptx_path> <output_dir>"}))
|
|
sys.exit(1)
|
|
|
|
pptx_path = sys.argv[1]
|
|
output_dir = sys.argv[2]
|
|
|
|
result = generate_preview(pptx_path, output_dir)
|
|
print(json.dumps(result, ensure_ascii=False))
|
|
sys.exit(0 if result["success"] else 1)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main() |