This commit is contained in:
kim
2026-04-29 18:03:48 +09:00
parent 4324985508
commit 1106553b81
19 changed files with 74 additions and 274 deletions
+5 -194
View File
@@ -596,198 +596,6 @@ def make_image_slide(prs, slide_spec, project_dir, workspace_path, colors, warni
return slide
# ─── Image Download ─────────────────────────────────────────────────────────────
# Known defunct/redirect-heavy image domains that need replacement
_DEFUNCT_IMAGE_DOMAINS = {
'source.unsplash.com', # Shut down — redirect to images.unsplash.com
'unsplash.it', # Redirects to picsum.photos
}
# Free image APIs that don't require auth and return actual image bytes
_FREE_IMAGE_APIS = {
'picsum.photos': 'https://picsum.photos/1200/800',
'dummyimage.com': 'https://dummyimage.com/1200x800/cccccc/666666.png&text=Image',
}
# ─── Image Download Helpers ─────────────────────────────────────────────────────
# Byte-level magic numbers for image format detection
_IMAGE_SIGNATURES = {
b'\xff\xd8\xff': '.jpg',
b'\x89PNG\r\n\x1a\n': '.png',
b'GIF87a': '.gif',
b'GIF89a': '.gif',
b'RIFF': '.webp', # WebP starts with RIFF...WEBP
b'BM': '.bmp',
b'II\x2a\x00': '.tiff',
b'MM\x00\x2a': '.tiff',
}
def _detect_image_ext(data: bytes) -> str:
"""Detect image format from magic bytes; returns extension or '.jpg' as fallback."""
for sig, ext in _IMAGE_SIGNATURES.items():
if data[:len(sig)] == sig:
return ext
return '.jpg'
def _detect_image_ext_from_headers(content_type: str) -> str:
"""Guess extension from Content-Type header."""
ct = (content_type or '').lower().strip()
mapping = {
'image/jpeg': '.jpg', 'image/jpg': '.jpg',
'image/png': '.png', 'image/gif': '.gif',
'image/webp': '.webp', 'image/bmp': '.bmp',
'image/tiff': '.tiff',
}
return mapping.get(ct.split(';')[0].strip(), '')
def _download_images(slides_spec, project_dir, warnings, max_retries=3):
"""Download image_url slides into project_dir with retry, browser headers, and validation."""
import time
from urllib.parse import urlparse, urlencode, urlunparse
try:
import requests
_HAS_REQUESTS = True
except ImportError:
_HAS_REQUESTS = False
# Pre-process: fix defunct URLs and clean up known-bad domains
for slide_spec in slides_spec:
url = slide_spec.get("image_url", "")
if not url:
continue
parsed = urlparse(url)
domain = (parsed.netloc or '').lower()
# source.unsplash.com is shut down — rewrite to images.unsplash.com
if domain == 'source.unsplash.com':
url = url.replace('source.unsplash.com', 'images.unsplash.com', 1)
slide_spec["image_url"] = url
domain = 'images.unsplash.com'
# Flag defunct domains as warnings
if domain in _DEFUNCT_IMAGE_DOMAINS:
warnings.append(f"Image URL uses a defunct service ({domain}). Slide may have a placeholder image.")
slide_spec.pop("image_url", None)
continue
# Build a browser-like session
if _HAS_REQUESTS:
session = requests.Session()
session.headers.update({
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
'Accept': 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8',
'Accept-Language': 'en-US,en;q=0.9,ko;q=0.8',
'Accept-Encoding': 'gzip, deflate, br',
'Sec-Fetch-Dest': 'image',
'Sec-Fetch-Mode': 'no-cors',
'Sec-Fetch-Site': 'cross-site',
'Cache-Control': 'no-cache',
})
else:
from urllib.request import build_opener, Request
from urllib.error import URLError, HTTPError
_img_opener = build_opener()
_ua = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36'
_img_opener.addheaders = [('User-Agent', _ua)]
for i, slide_spec in enumerate(slides_spec):
image_url = slide_spec.get("image_url")
if not image_url:
continue
# Unsplash-specific: add resize params for reliable download
download_url = image_url
if 'images.unsplash.com' in image_url or 'unsplash.com' in image_url:
sep = '&' if '?' in image_url else '?'
# Request a reasonable size with good quality; Unsplash respects these params
download_url = f"{image_url}{sep}w=1200&q=80&auto=format"
# Determine file extension from URL, falling back to detection
url_path = download_url.split("?")[0]
ext = os.path.splitext(url_path.split("/")[-1])[1].lower()
if ext not in ('.jpg', '.jpeg', '.png', '.gif', '.webp', '.bmp', '.tiff'):
ext = '' # Will be determined from response
last_error = None
for attempt in range(1, max_retries + 1):
try:
if _HAS_REQUESTS:
resp = session.get(download_url, timeout=30, allow_redirects=True)
if resp.status_code == 403 and 'images.unsplash.com' in download_url:
# Unsplash might need a source param; retry with source identifier
sep = '&' if '?' in download_url else '?'
retry_url = f"{download_url}{sep}source=smallclaw"
resp = session.get(retry_url, timeout=30, allow_redirects=True)
if resp.status_code != 200:
raise ValueError(f"HTTP {resp.status_code}")
data = resp.content
content_type = resp.headers.get('Content-Type', '')
else:
req = Request(download_url, headers={
'User-Agent': _ua,
'Accept': 'image/*,*/*;q=0.8',
})
with _img_opener.open(req, timeout=30) as resp_obj:
if resp_obj.getcode() != 200:
raise HTTPError(download_url, resp_obj.getcode(), f"HTTP {resp_obj.getcode()}", resp_obj.headers, None)
data = resp_obj.read()
content_type = resp_obj.headers.get('Content-Type', '')
# Validate: must have some content
if not data or len(data) < 16:
raise ValueError(f"Response too small ({len(data) if data else 0} bytes)")
# Validate: reject clearly non-image content (HTML pages, etc.)
ct_lower = (content_type or '').lower().split(';')[0].strip()
if ct_lower and ct_lower not in ('image/jpeg', 'image/jpg', 'image/png',
'image/gif', 'image/webp', 'image/bmp', 'image/tiff',
'application/octet-stream', 'binary/octet-stream', ''):
# If content-type is text/html or similar, it's not an image
if ct_lower.startswith('text/') or ct_lower in ('application/html', 'application/xml'):
raise ValueError(f"Non-image Content-Type: {content_type}")
# For other unknown types, check magic bytes instead of rejecting
# Determine extension from content or magic bytes
if not ext:
ext_from_ct = _detect_image_ext_from_headers(content_type)
ext_from_magic = _detect_image_ext(data)
ext = ext_from_ct or ext_from_magic or '.jpg'
fname = f"slide{i + 1}_image{ext}"
dest = os.path.join(project_dir, fname)
with open(dest, 'wb') as f:
f.write(data)
slide_spec["image_path"] = fname
last_error = None
print(f"[pptx_gen] Downloaded image for slide {i+1}: {len(data)} bytes -> {fname}", file=sys.stderr, flush=True)
break
except Exception as e:
last_error = e
if attempt < max_retries:
wait = 2 ** attempt
print(f"[pptx_gen] Download attempt {attempt}/{max_retries} failed for slide {i+1}: {e}. Retrying in {wait}s...", file=sys.stderr, flush=True)
time.sleep(wait)
else:
print(f"[pptx_gen] Download failed after {max_retries} attempts for slide {i+1}: {e}", file=sys.stderr, flush=True)
if last_error:
domain = ''
try:
from urllib.parse import urlparse
domain = urlparse(image_url).netloc
except Exception:
pass
err_msg = f"Failed to download image for slide {i+1} from {domain or image_url}: {last_error}"
warnings.append(err_msg)
slide_spec.pop("image_url", None)
# ─── Download Page Generator ────────────────────────────────────────────────────
def _create_download_page(pptx_path: str, download_url: str, title: str, slide_count: int):
@@ -865,8 +673,11 @@ def generate(spec: dict, workspace_path: str) -> dict:
if ip or iu:
print(f"[pptx_gen] slide {i+1}: type={s.get('type','?')} image_path='{ip}' image_url='{iu[:60] if iu else ''}'", file=sys.stderr, flush=True)
# Download image_url slides into project_dir
_download_images(slides_spec, project_dir, warnings)
# Warn about any remaining image_url slides (downloading is now handled by TypeScript)
for s in slides_spec:
if s.get("image_url") and not s.get("image_path"):
warnings.append(f"image_url not downloaded (TypeScript should handle this): {s['image_url'][:80]}")
s.pop("image_url", None)
if is_edit:
output_path = existing_path