v1.0.5
This commit is contained in:
+5
-194
@@ -596,198 +596,6 @@ def make_image_slide(prs, slide_spec, project_dir, workspace_path, colors, warni
|
||||
return slide
|
||||
|
||||
|
||||
# ─── Image Download ─────────────────────────────────────────────────────────────
|
||||
|
||||
# Known defunct/redirect-heavy image domains that need replacement
|
||||
_DEFUNCT_IMAGE_DOMAINS = {
|
||||
'source.unsplash.com', # Shut down — redirect to images.unsplash.com
|
||||
'unsplash.it', # Redirects to picsum.photos
|
||||
}
|
||||
|
||||
# Free image APIs that don't require auth and return actual image bytes
|
||||
_FREE_IMAGE_APIS = {
|
||||
'picsum.photos': 'https://picsum.photos/1200/800',
|
||||
'dummyimage.com': 'https://dummyimage.com/1200x800/cccccc/666666.png&text=Image',
|
||||
}
|
||||
|
||||
|
||||
# ─── Image Download Helpers ─────────────────────────────────────────────────────
|
||||
|
||||
# Byte-level magic numbers for image format detection
|
||||
_IMAGE_SIGNATURES = {
|
||||
b'\xff\xd8\xff': '.jpg',
|
||||
b'\x89PNG\r\n\x1a\n': '.png',
|
||||
b'GIF87a': '.gif',
|
||||
b'GIF89a': '.gif',
|
||||
b'RIFF': '.webp', # WebP starts with RIFF...WEBP
|
||||
b'BM': '.bmp',
|
||||
b'II\x2a\x00': '.tiff',
|
||||
b'MM\x00\x2a': '.tiff',
|
||||
}
|
||||
|
||||
def _detect_image_ext(data: bytes) -> str:
|
||||
"""Detect image format from magic bytes; returns extension or '.jpg' as fallback."""
|
||||
for sig, ext in _IMAGE_SIGNATURES.items():
|
||||
if data[:len(sig)] == sig:
|
||||
return ext
|
||||
return '.jpg'
|
||||
|
||||
def _detect_image_ext_from_headers(content_type: str) -> str:
|
||||
"""Guess extension from Content-Type header."""
|
||||
ct = (content_type or '').lower().strip()
|
||||
mapping = {
|
||||
'image/jpeg': '.jpg', 'image/jpg': '.jpg',
|
||||
'image/png': '.png', 'image/gif': '.gif',
|
||||
'image/webp': '.webp', 'image/bmp': '.bmp',
|
||||
'image/tiff': '.tiff',
|
||||
}
|
||||
return mapping.get(ct.split(';')[0].strip(), '')
|
||||
|
||||
|
||||
def _download_images(slides_spec, project_dir, warnings, max_retries=3):
|
||||
"""Download image_url slides into project_dir with retry, browser headers, and validation."""
|
||||
import time
|
||||
from urllib.parse import urlparse, urlencode, urlunparse
|
||||
try:
|
||||
import requests
|
||||
_HAS_REQUESTS = True
|
||||
except ImportError:
|
||||
_HAS_REQUESTS = False
|
||||
|
||||
# Pre-process: fix defunct URLs and clean up known-bad domains
|
||||
for slide_spec in slides_spec:
|
||||
url = slide_spec.get("image_url", "")
|
||||
if not url:
|
||||
continue
|
||||
parsed = urlparse(url)
|
||||
domain = (parsed.netloc or '').lower()
|
||||
|
||||
# source.unsplash.com is shut down — rewrite to images.unsplash.com
|
||||
if domain == 'source.unsplash.com':
|
||||
url = url.replace('source.unsplash.com', 'images.unsplash.com', 1)
|
||||
slide_spec["image_url"] = url
|
||||
domain = 'images.unsplash.com'
|
||||
|
||||
# Flag defunct domains as warnings
|
||||
if domain in _DEFUNCT_IMAGE_DOMAINS:
|
||||
warnings.append(f"Image URL uses a defunct service ({domain}). Slide may have a placeholder image.")
|
||||
slide_spec.pop("image_url", None)
|
||||
continue
|
||||
|
||||
# Build a browser-like session
|
||||
if _HAS_REQUESTS:
|
||||
session = requests.Session()
|
||||
session.headers.update({
|
||||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
|
||||
'Accept': 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8',
|
||||
'Accept-Language': 'en-US,en;q=0.9,ko;q=0.8',
|
||||
'Accept-Encoding': 'gzip, deflate, br',
|
||||
'Sec-Fetch-Dest': 'image',
|
||||
'Sec-Fetch-Mode': 'no-cors',
|
||||
'Sec-Fetch-Site': 'cross-site',
|
||||
'Cache-Control': 'no-cache',
|
||||
})
|
||||
else:
|
||||
from urllib.request import build_opener, Request
|
||||
from urllib.error import URLError, HTTPError
|
||||
_img_opener = build_opener()
|
||||
_ua = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36'
|
||||
_img_opener.addheaders = [('User-Agent', _ua)]
|
||||
|
||||
for i, slide_spec in enumerate(slides_spec):
|
||||
image_url = slide_spec.get("image_url")
|
||||
if not image_url:
|
||||
continue
|
||||
|
||||
# Unsplash-specific: add resize params for reliable download
|
||||
download_url = image_url
|
||||
if 'images.unsplash.com' in image_url or 'unsplash.com' in image_url:
|
||||
sep = '&' if '?' in image_url else '?'
|
||||
# Request a reasonable size with good quality; Unsplash respects these params
|
||||
download_url = f"{image_url}{sep}w=1200&q=80&auto=format"
|
||||
|
||||
# Determine file extension from URL, falling back to detection
|
||||
url_path = download_url.split("?")[0]
|
||||
ext = os.path.splitext(url_path.split("/")[-1])[1].lower()
|
||||
if ext not in ('.jpg', '.jpeg', '.png', '.gif', '.webp', '.bmp', '.tiff'):
|
||||
ext = '' # Will be determined from response
|
||||
|
||||
last_error = None
|
||||
for attempt in range(1, max_retries + 1):
|
||||
try:
|
||||
if _HAS_REQUESTS:
|
||||
resp = session.get(download_url, timeout=30, allow_redirects=True)
|
||||
if resp.status_code == 403 and 'images.unsplash.com' in download_url:
|
||||
# Unsplash might need a source param; retry with source identifier
|
||||
sep = '&' if '?' in download_url else '?'
|
||||
retry_url = f"{download_url}{sep}source=smallclaw"
|
||||
resp = session.get(retry_url, timeout=30, allow_redirects=True)
|
||||
if resp.status_code != 200:
|
||||
raise ValueError(f"HTTP {resp.status_code}")
|
||||
data = resp.content
|
||||
content_type = resp.headers.get('Content-Type', '')
|
||||
else:
|
||||
req = Request(download_url, headers={
|
||||
'User-Agent': _ua,
|
||||
'Accept': 'image/*,*/*;q=0.8',
|
||||
})
|
||||
with _img_opener.open(req, timeout=30) as resp_obj:
|
||||
if resp_obj.getcode() != 200:
|
||||
raise HTTPError(download_url, resp_obj.getcode(), f"HTTP {resp_obj.getcode()}", resp_obj.headers, None)
|
||||
data = resp_obj.read()
|
||||
content_type = resp_obj.headers.get('Content-Type', '')
|
||||
|
||||
# Validate: must have some content
|
||||
if not data or len(data) < 16:
|
||||
raise ValueError(f"Response too small ({len(data) if data else 0} bytes)")
|
||||
|
||||
# Validate: reject clearly non-image content (HTML pages, etc.)
|
||||
ct_lower = (content_type or '').lower().split(';')[0].strip()
|
||||
if ct_lower and ct_lower not in ('image/jpeg', 'image/jpg', 'image/png',
|
||||
'image/gif', 'image/webp', 'image/bmp', 'image/tiff',
|
||||
'application/octet-stream', 'binary/octet-stream', ''):
|
||||
# If content-type is text/html or similar, it's not an image
|
||||
if ct_lower.startswith('text/') or ct_lower in ('application/html', 'application/xml'):
|
||||
raise ValueError(f"Non-image Content-Type: {content_type}")
|
||||
# For other unknown types, check magic bytes instead of rejecting
|
||||
|
||||
# Determine extension from content or magic bytes
|
||||
if not ext:
|
||||
ext_from_ct = _detect_image_ext_from_headers(content_type)
|
||||
ext_from_magic = _detect_image_ext(data)
|
||||
ext = ext_from_ct or ext_from_magic or '.jpg'
|
||||
|
||||
fname = f"slide{i + 1}_image{ext}"
|
||||
dest = os.path.join(project_dir, fname)
|
||||
with open(dest, 'wb') as f:
|
||||
f.write(data)
|
||||
|
||||
slide_spec["image_path"] = fname
|
||||
last_error = None
|
||||
print(f"[pptx_gen] Downloaded image for slide {i+1}: {len(data)} bytes -> {fname}", file=sys.stderr, flush=True)
|
||||
break
|
||||
except Exception as e:
|
||||
last_error = e
|
||||
if attempt < max_retries:
|
||||
wait = 2 ** attempt
|
||||
print(f"[pptx_gen] Download attempt {attempt}/{max_retries} failed for slide {i+1}: {e}. Retrying in {wait}s...", file=sys.stderr, flush=True)
|
||||
time.sleep(wait)
|
||||
else:
|
||||
print(f"[pptx_gen] Download failed after {max_retries} attempts for slide {i+1}: {e}", file=sys.stderr, flush=True)
|
||||
|
||||
if last_error:
|
||||
domain = ''
|
||||
try:
|
||||
from urllib.parse import urlparse
|
||||
domain = urlparse(image_url).netloc
|
||||
except Exception:
|
||||
pass
|
||||
err_msg = f"Failed to download image for slide {i+1} from {domain or image_url}: {last_error}"
|
||||
warnings.append(err_msg)
|
||||
|
||||
slide_spec.pop("image_url", None)
|
||||
|
||||
|
||||
# ─── Download Page Generator ────────────────────────────────────────────────────
|
||||
|
||||
def _create_download_page(pptx_path: str, download_url: str, title: str, slide_count: int):
|
||||
@@ -865,8 +673,11 @@ def generate(spec: dict, workspace_path: str) -> dict:
|
||||
if ip or iu:
|
||||
print(f"[pptx_gen] slide {i+1}: type={s.get('type','?')} image_path='{ip}' image_url='{iu[:60] if iu else ''}'", file=sys.stderr, flush=True)
|
||||
|
||||
# Download image_url slides into project_dir
|
||||
_download_images(slides_spec, project_dir, warnings)
|
||||
# Warn about any remaining image_url slides (downloading is now handled by TypeScript)
|
||||
for s in slides_spec:
|
||||
if s.get("image_url") and not s.get("image_path"):
|
||||
warnings.append(f"image_url not downloaded (TypeScript should handle this): {s['image_url'][:80]}")
|
||||
s.pop("image_url", None)
|
||||
|
||||
if is_edit:
|
||||
output_path = existing_path
|
||||
|
||||
Reference in New Issue
Block a user