v1.0.1
This commit is contained in:
+274
-32
@@ -354,30 +354,59 @@ def fit_dimensions(img_w, img_h, max_w, max_h, mode="contain"):
|
||||
return w, h
|
||||
|
||||
|
||||
def _ensure_compatible_image(image_path: str) -> str:
|
||||
"""Convert WebP or other unsupported formats to PNG for python-pptx compatibility.
|
||||
|
||||
python-pptx relies on PIL for image analysis, and PowerPoint/OOXML does not
|
||||
reliably support WebP. Converting to PNG before embedding avoids silent failures.
|
||||
"""
|
||||
ext = os.path.splitext(image_path)[1].lower()
|
||||
supported = {'.png', '.jpg', '.jpeg', '.gif', '.bmp', '.tiff', '.emf', '.wmf'}
|
||||
if ext in supported:
|
||||
return image_path
|
||||
|
||||
# Attempt PIL conversion to PNG
|
||||
try:
|
||||
from PIL import Image as PILImage
|
||||
with PILImage.open(image_path) as img:
|
||||
# RGBA or P mode can cause issues with some PowerPoint viewers; convert to RGB
|
||||
if img.mode in ('RGBA', 'P', 'LA', 'L'):
|
||||
rgb_img = img.convert('RGB')
|
||||
else:
|
||||
rgb_img = img
|
||||
new_path = image_path + '.converted.png'
|
||||
rgb_img.save(new_path, 'PNG')
|
||||
return new_path
|
||||
except Exception:
|
||||
# If conversion fails, return original and let downstream handle the failure
|
||||
return image_path
|
||||
|
||||
|
||||
def add_image_or_placeholder(slide, image_path, left, top, max_width, max_height,
|
||||
warnings, color_hex="FF0000", fit_mode="contain"):
|
||||
"""Add an image preserving aspect ratio, or a red placeholder if not found.
|
||||
"""Add an image preserving aspect ratio, or a subtle placeholder if not found/unsupported.
|
||||
|
||||
fit_mode:
|
||||
'contain' (default) — entire image visible, may leave empty bars
|
||||
'cover' — fill the bounding box completely, cropping if needed
|
||||
"""
|
||||
if os.path.exists(image_path):
|
||||
converted_path = _ensure_compatible_image(image_path)
|
||||
try:
|
||||
# Read native dimensions and compute aspect-ratio-preserving size
|
||||
from PIL import Image as PILImage
|
||||
with PILImage.open(image_path) as img:
|
||||
with PILImage.open(converted_path) as img:
|
||||
img_w, img_h = img.size
|
||||
fit_w, fit_h = fit_dimensions(img_w, img_h, max_width, max_height, mode=fit_mode)
|
||||
# Center within the bounding box both horizontally and vertically
|
||||
center_x = left + (max_width - fit_w) / 2
|
||||
center_y = top + (max_height - fit_h) / 2
|
||||
slide.shapes.add_picture(image_path, center_x, center_y, fit_w, fit_h)
|
||||
slide.shapes.add_picture(converted_path, center_x, center_y, fit_w, fit_h)
|
||||
return
|
||||
except Exception:
|
||||
# If image dimension reading fails, try with native size (no stretching)
|
||||
try:
|
||||
pic = slide.shapes.add_picture(image_path, left, top)
|
||||
pic = slide.shapes.add_picture(converted_path, left, top)
|
||||
# Scale down if larger than max bounds
|
||||
native_w = pic.width
|
||||
native_h = pic.height
|
||||
@@ -397,11 +426,9 @@ def add_image_or_placeholder(slide, image_path, left, top, max_width, max_height
|
||||
return
|
||||
except Exception:
|
||||
pass
|
||||
add_text_box(slide, left, top, max_width, max_height,
|
||||
f"[Image not found: {image_path}]",
|
||||
font_size=Pt(14), color_hex=color_hex,
|
||||
alignment=PP_ALIGN.CENTER)
|
||||
warnings.append(f"Image not found: {image_path}")
|
||||
# Draw a subtle gray placeholder instead of red error text
|
||||
add_shape_rect(slide, left, top, max_width, max_height, "E8E8E8")
|
||||
warnings.append(f"Image missing or unsupported: {image_path}")
|
||||
|
||||
|
||||
def resolve_image_path(image_path: str, project_dir: str, workspace_path: str) -> str:
|
||||
@@ -549,6 +576,236 @@ def make_image_slide(prs, slide_spec, project_dir, workspace_path, colors, warni
|
||||
return slide
|
||||
|
||||
|
||||
# ─── Image Download ─────────────────────────────────────────────────────────────
|
||||
|
||||
# Known defunct/redirect-heavy image domains that need replacement
|
||||
_DEFUNCT_IMAGE_DOMAINS = {
|
||||
'source.unsplash.com', # Shut down — redirect to images.unsplash.com
|
||||
'unsplash.it', # Redirects to picsum.photos
|
||||
}
|
||||
|
||||
# Free image APIs that don't require auth and return actual image bytes
|
||||
_FREE_IMAGE_APIS = {
|
||||
'picsum.photos': 'https://picsum.photos/1200/800',
|
||||
'dummyimage.com': 'https://dummyimage.com/1200x800/cccccc/666666.png&text=Image',
|
||||
}
|
||||
|
||||
|
||||
# ─── Image Download Helpers ─────────────────────────────────────────────────────
|
||||
|
||||
# Byte-level magic numbers for image format detection
|
||||
_IMAGE_SIGNATURES = {
|
||||
b'\xff\xd8\xff': '.jpg',
|
||||
b'\x89PNG\r\n\x1a\n': '.png',
|
||||
b'GIF87a': '.gif',
|
||||
b'GIF89a': '.gif',
|
||||
b'RIFF': '.webp', # WebP starts with RIFF...WEBP
|
||||
b'BM': '.bmp',
|
||||
b'II\x2a\x00': '.tiff',
|
||||
b'MM\x00\x2a': '.tiff',
|
||||
}
|
||||
|
||||
def _detect_image_ext(data: bytes) -> str:
|
||||
"""Detect image format from magic bytes; returns extension or '.jpg' as fallback."""
|
||||
for sig, ext in _IMAGE_SIGNATURES.items():
|
||||
if data[:len(sig)] == sig:
|
||||
return ext
|
||||
return '.jpg'
|
||||
|
||||
def _detect_image_ext_from_headers(content_type: str) -> str:
|
||||
"""Guess extension from Content-Type header."""
|
||||
ct = (content_type or '').lower().strip()
|
||||
mapping = {
|
||||
'image/jpeg': '.jpg', 'image/jpg': '.jpg',
|
||||
'image/png': '.png', 'image/gif': '.gif',
|
||||
'image/webp': '.webp', 'image/bmp': '.bmp',
|
||||
'image/tiff': '.tiff',
|
||||
}
|
||||
return mapping.get(ct.split(';')[0].strip(), '')
|
||||
|
||||
|
||||
def _download_images(slides_spec, project_dir, warnings, max_retries=3):
|
||||
"""Download image_url slides into project_dir with retry, browser headers, and validation."""
|
||||
import time
|
||||
from urllib.parse import urlparse, urlencode, urlunparse
|
||||
try:
|
||||
import requests
|
||||
_HAS_REQUESTS = True
|
||||
except ImportError:
|
||||
_HAS_REQUESTS = False
|
||||
|
||||
# Pre-process: fix defunct URLs and clean up known-bad domains
|
||||
for slide_spec in slides_spec:
|
||||
url = slide_spec.get("image_url", "")
|
||||
if not url:
|
||||
continue
|
||||
parsed = urlparse(url)
|
||||
domain = (parsed.netloc or '').lower()
|
||||
|
||||
# source.unsplash.com is shut down — rewrite to images.unsplash.com
|
||||
if domain == 'source.unsplash.com':
|
||||
url = url.replace('source.unsplash.com', 'images.unsplash.com', 1)
|
||||
slide_spec["image_url"] = url
|
||||
domain = 'images.unsplash.com'
|
||||
|
||||
# Flag defunct domains as warnings
|
||||
if domain in _DEFUNCT_IMAGE_DOMAINS:
|
||||
warnings.append(f"Image URL uses a defunct service ({domain}). Slide may have a placeholder image.")
|
||||
slide_spec.pop("image_url", None)
|
||||
continue
|
||||
|
||||
# Build a browser-like session
|
||||
if _HAS_REQUESTS:
|
||||
session = requests.Session()
|
||||
session.headers.update({
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
|
||||
'Accept': 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8',
|
||||
'Accept-Language': 'en-US,en;q=0.9,ko;q=0.8',
|
||||
'Accept-Encoding': 'gzip, deflate, br',
|
||||
'Sec-Fetch-Dest': 'image',
|
||||
'Sec-Fetch-Mode': 'no-cors',
|
||||
'Sec-Fetch-Site': 'cross-site',
|
||||
'Cache-Control': 'no-cache',
|
||||
})
|
||||
else:
|
||||
from urllib.request import build_opener, Request
|
||||
from urllib.error import URLError, HTTPError
|
||||
_img_opener = build_opener()
|
||||
_ua = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36'
|
||||
_img_opener.addheaders = [('User-Agent', _ua)]
|
||||
|
||||
for i, slide_spec in enumerate(slides_spec):
|
||||
image_url = slide_spec.get("image_url")
|
||||
if not image_url:
|
||||
continue
|
||||
|
||||
# Unsplash-specific: add resize params for reliable download
|
||||
download_url = image_url
|
||||
if 'images.unsplash.com' in image_url or 'unsplash.com' in image_url:
|
||||
sep = '&' if '?' in image_url else '?'
|
||||
# Request a reasonable size with good quality; Unsplash respects these params
|
||||
download_url = f"{image_url}{sep}w=1200&q=80&auto=format"
|
||||
|
||||
# Determine file extension from URL, falling back to detection
|
||||
url_path = download_url.split("?")[0]
|
||||
ext = os.path.splitext(url_path.split("/")[-1])[1].lower()
|
||||
if ext not in ('.jpg', '.jpeg', '.png', '.gif', '.webp', '.bmp', '.tiff'):
|
||||
ext = '' # Will be determined from response
|
||||
|
||||
last_error = None
|
||||
for attempt in range(1, max_retries + 1):
|
||||
try:
|
||||
if _HAS_REQUESTS:
|
||||
resp = session.get(download_url, timeout=30, allow_redirects=True)
|
||||
if resp.status_code == 403 and 'images.unsplash.com' in download_url:
|
||||
# Unsplash might need a source param; retry with source identifier
|
||||
sep = '&' if '?' in download_url else '?'
|
||||
retry_url = f"{download_url}{sep}source=smallclaw"
|
||||
resp = session.get(retry_url, timeout=30, allow_redirects=True)
|
||||
if resp.status_code != 200:
|
||||
raise ValueError(f"HTTP {resp.status_code}")
|
||||
data = resp.content
|
||||
content_type = resp.headers.get('Content-Type', '')
|
||||
else:
|
||||
req = Request(download_url, headers={
|
||||
'User-Agent': _ua,
|
||||
'Accept': 'image/*,*/*;q=0.8',
|
||||
})
|
||||
with _img_opener.open(req, timeout=30) as resp_obj:
|
||||
if resp_obj.getcode() != 200:
|
||||
raise HTTPError(download_url, resp_obj.getcode(), f"HTTP {resp_obj.getcode()}", resp_obj.headers, None)
|
||||
data = resp_obj.read()
|
||||
content_type = resp_obj.headers.get('Content-Type', '')
|
||||
|
||||
# Validate: must have some content
|
||||
if not data or len(data) < 16:
|
||||
raise ValueError(f"Response too small ({len(data) if data else 0} bytes)")
|
||||
|
||||
# Validate: reject clearly non-image content (HTML pages, etc.)
|
||||
ct_lower = (content_type or '').lower().split(';')[0].strip()
|
||||
if ct_lower and ct_lower not in ('image/jpeg', 'image/jpg', 'image/png',
|
||||
'image/gif', 'image/webp', 'image/bmp', 'image/tiff',
|
||||
'application/octet-stream', 'binary/octet-stream', ''):
|
||||
# If content-type is text/html or similar, it's not an image
|
||||
if ct_lower.startswith('text/') or ct_lower in ('application/html', 'application/xml'):
|
||||
raise ValueError(f"Non-image Content-Type: {content_type}")
|
||||
# For other unknown types, check magic bytes instead of rejecting
|
||||
|
||||
# Determine extension from content or magic bytes
|
||||
if not ext:
|
||||
ext_from_ct = _detect_image_ext_from_headers(content_type)
|
||||
ext_from_magic = _detect_image_ext(data)
|
||||
ext = ext_from_ct or ext_from_magic or '.jpg'
|
||||
|
||||
fname = f"slide{i + 1}_image{ext}"
|
||||
dest = os.path.join(project_dir, fname)
|
||||
with open(dest, 'wb') as f:
|
||||
f.write(data)
|
||||
|
||||
slide_spec["image_path"] = fname
|
||||
last_error = None
|
||||
print(f"[pptx_gen] Downloaded image for slide {i+1}: {len(data)} bytes -> {fname}", file=sys.stderr, flush=True)
|
||||
break
|
||||
except Exception as e:
|
||||
last_error = e
|
||||
if attempt < max_retries:
|
||||
wait = 2 ** attempt
|
||||
print(f"[pptx_gen] Download attempt {attempt}/{max_retries} failed for slide {i+1}: {e}. Retrying in {wait}s...", file=sys.stderr, flush=True)
|
||||
time.sleep(wait)
|
||||
else:
|
||||
print(f"[pptx_gen] Download failed after {max_retries} attempts for slide {i+1}: {e}", file=sys.stderr, flush=True)
|
||||
|
||||
if last_error:
|
||||
domain = ''
|
||||
try:
|
||||
from urllib.parse import urlparse
|
||||
domain = urlparse(image_url).netloc
|
||||
except Exception:
|
||||
pass
|
||||
err_msg = f"Failed to download image for slide {i+1} from {domain or image_url}: {last_error}"
|
||||
warnings.append(err_msg)
|
||||
|
||||
slide_spec.pop("image_url", None)
|
||||
|
||||
|
||||
# ─── Download Page Generator ────────────────────────────────────────────────────
|
||||
|
||||
def _create_download_page(pptx_path: str, download_url: str, title: str, slide_count: int):
|
||||
"""Create a local download.html next to the PPTX so users can grab the file via browser."""
|
||||
project_dir = os.path.dirname(pptx_path)
|
||||
filename = os.path.basename(pptx_path)
|
||||
html_path = os.path.join(project_dir, "download.html")
|
||||
|
||||
# Build a relative link so the HTML works whether opened via file:// or served
|
||||
html = f"""<!DOCTYPE html>
|
||||
<html lang="ko">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<title>{title} - 다운로드</title>
|
||||
<style>
|
||||
body {{ font-family: 'Malgun Gothic', sans-serif; max-width: 600px; margin: 60px auto; padding: 20px; text-align: center; background: #f8f9fa; }}
|
||||
.card {{ background: white; border-radius: 12px; padding: 40px 30px; box-shadow: 0 4px 20px rgba(0,0,0,0.08); }}
|
||||
h1 {{ color: #1a1a2e; font-size: 24px; margin-bottom: 10px; }}
|
||||
p {{ color: #5f6f86; margin-bottom: 30px; }}
|
||||
.btn {{ display: inline-block; background: #1668e3; color: white; text-decoration: none; padding: 14px 36px; border-radius: 8px; font-size: 16px; font-weight: bold; transition: background 0.2s; }}
|
||||
.btn:hover {{ background: #1255bb; }}
|
||||
.meta {{ margin-top: 20px; font-size: 12px; color: #888; }}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="card">
|
||||
<h1>{title}</h1>
|
||||
<p>슬라이드 {slide_count}장이 준비되었습니다.</p>
|
||||
<a class="btn" href="{filename}" download>프레젠테이션 다운로드</a>
|
||||
<div class="meta">{pptx_path}</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
with open(html_path, "w", encoding="utf-8") as f:
|
||||
f.write(html)
|
||||
|
||||
|
||||
# ─── Main Generation ───────────────────────────────────────────────────────────
|
||||
|
||||
def generate(spec: dict, workspace_path: str) -> dict:
|
||||
@@ -577,28 +834,7 @@ def generate(spec: dict, workspace_path: str) -> dict:
|
||||
os.makedirs(project_dir, exist_ok=True)
|
||||
|
||||
# Download image_url slides into project_dir
|
||||
# Set a browser-like User-Agent so image hosts (Wikipedia, Unsplash, etc.) don't block us
|
||||
from urllib.request import build_opener, Request
|
||||
_img_opener = build_opener()
|
||||
_img_opener.addheaders = [('User-Agent', 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36')]
|
||||
|
||||
for i, slide_spec in enumerate(slides_spec):
|
||||
image_url = slide_spec.get("image_url")
|
||||
if image_url:
|
||||
try:
|
||||
ext = os.path.splitext(image_url.split("/")[-1].split("?")[0])[1] or ".jpg"
|
||||
fname = f"slide{i + 1}_image{ext}"
|
||||
dest = os.path.join(project_dir, fname)
|
||||
req = Request(image_url, headers={'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'})
|
||||
with _img_opener.open(req, timeout=30) as resp:
|
||||
with open(dest, 'wb') as f:
|
||||
f.write(resp.read())
|
||||
slide_spec["image_path"] = fname
|
||||
except Exception as e:
|
||||
err_msg = f"Failed to download image from {image_url}: {e}"
|
||||
warnings.append(err_msg)
|
||||
print(f"[pptx_gen] WARNING: {err_msg}", file=sys.stderr, flush=True)
|
||||
slide_spec.pop("image_url", None)
|
||||
_download_images(slides_spec, project_dir, warnings)
|
||||
|
||||
if is_edit:
|
||||
output_path = existing_path
|
||||
@@ -709,7 +945,13 @@ def generate(spec: dict, workspace_path: str) -> dict:
|
||||
if warnings:
|
||||
stdout_text += "\n\nWarnings:\n" + "\n".join(f"- {w}" for w in warnings)
|
||||
if any("download" in w.lower() or "image" in w.lower() for w in warnings):
|
||||
stdout_text += "\n\nNote: Some images appear as red placeholders. This is expected — do NOT retry."
|
||||
stdout_text += "\n\nNote: Some images could not be embedded and appear as light-gray placeholders. This is expected — do NOT retry."
|
||||
|
||||
# Create a local download.html page so users can download via browser even when the API gateway isn't rendering the link
|
||||
try:
|
||||
_create_download_page(output_path, download_url, title, total_slides)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return {
|
||||
"success": True,
|
||||
|
||||
Reference in New Issue
Block a user