This commit is contained in:
kim
2026-04-29 18:03:48 +09:00
parent 4324985508
commit 1106553b81
19 changed files with 74 additions and 274 deletions
+5 -194
View File
@@ -596,198 +596,6 @@ def make_image_slide(prs, slide_spec, project_dir, workspace_path, colors, warni
return slide
# ─── Image Download ─────────────────────────────────────────────────────────────
# Known defunct/redirect-heavy image domains that need replacement
_DEFUNCT_IMAGE_DOMAINS = {
'source.unsplash.com', # Shut down — redirect to images.unsplash.com
'unsplash.it', # Redirects to picsum.photos
}
# Free image APIs that don't require auth and return actual image bytes
_FREE_IMAGE_APIS = {
'picsum.photos': 'https://picsum.photos/1200/800',
'dummyimage.com': 'https://dummyimage.com/1200x800/cccccc/666666.png&text=Image',
}
# ─── Image Download Helpers ─────────────────────────────────────────────────────
# Byte-level magic numbers for image format detection
_IMAGE_SIGNATURES = {
b'\xff\xd8\xff': '.jpg',
b'\x89PNG\r\n\x1a\n': '.png',
b'GIF87a': '.gif',
b'GIF89a': '.gif',
b'RIFF': '.webp', # WebP starts with RIFF...WEBP
b'BM': '.bmp',
b'II\x2a\x00': '.tiff',
b'MM\x00\x2a': '.tiff',
}
def _detect_image_ext(data: bytes) -> str:
"""Detect image format from magic bytes; returns extension or '.jpg' as fallback."""
for sig, ext in _IMAGE_SIGNATURES.items():
if data[:len(sig)] == sig:
return ext
return '.jpg'
def _detect_image_ext_from_headers(content_type: str) -> str:
"""Guess extension from Content-Type header."""
ct = (content_type or '').lower().strip()
mapping = {
'image/jpeg': '.jpg', 'image/jpg': '.jpg',
'image/png': '.png', 'image/gif': '.gif',
'image/webp': '.webp', 'image/bmp': '.bmp',
'image/tiff': '.tiff',
}
return mapping.get(ct.split(';')[0].strip(), '')
def _download_images(slides_spec, project_dir, warnings, max_retries=3):
"""Download image_url slides into project_dir with retry, browser headers, and validation."""
import time
from urllib.parse import urlparse, urlencode, urlunparse
try:
import requests
_HAS_REQUESTS = True
except ImportError:
_HAS_REQUESTS = False
# Pre-process: fix defunct URLs and clean up known-bad domains
for slide_spec in slides_spec:
url = slide_spec.get("image_url", "")
if not url:
continue
parsed = urlparse(url)
domain = (parsed.netloc or '').lower()
# source.unsplash.com is shut down — rewrite to images.unsplash.com
if domain == 'source.unsplash.com':
url = url.replace('source.unsplash.com', 'images.unsplash.com', 1)
slide_spec["image_url"] = url
domain = 'images.unsplash.com'
# Flag defunct domains as warnings
if domain in _DEFUNCT_IMAGE_DOMAINS:
warnings.append(f"Image URL uses a defunct service ({domain}). Slide may have a placeholder image.")
slide_spec.pop("image_url", None)
continue
# Build a browser-like session
if _HAS_REQUESTS:
session = requests.Session()
session.headers.update({
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
'Accept': 'image/avif,image/webp,image/apng,image/svg+xml,image/*,*/*;q=0.8',
'Accept-Language': 'en-US,en;q=0.9,ko;q=0.8',
'Accept-Encoding': 'gzip, deflate, br',
'Sec-Fetch-Dest': 'image',
'Sec-Fetch-Mode': 'no-cors',
'Sec-Fetch-Site': 'cross-site',
'Cache-Control': 'no-cache',
})
else:
from urllib.request import build_opener, Request
from urllib.error import URLError, HTTPError
_img_opener = build_opener()
_ua = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36'
_img_opener.addheaders = [('User-Agent', _ua)]
for i, slide_spec in enumerate(slides_spec):
image_url = slide_spec.get("image_url")
if not image_url:
continue
# Unsplash-specific: add resize params for reliable download
download_url = image_url
if 'images.unsplash.com' in image_url or 'unsplash.com' in image_url:
sep = '&' if '?' in image_url else '?'
# Request a reasonable size with good quality; Unsplash respects these params
download_url = f"{image_url}{sep}w=1200&q=80&auto=format"
# Determine file extension from URL, falling back to detection
url_path = download_url.split("?")[0]
ext = os.path.splitext(url_path.split("/")[-1])[1].lower()
if ext not in ('.jpg', '.jpeg', '.png', '.gif', '.webp', '.bmp', '.tiff'):
ext = '' # Will be determined from response
last_error = None
for attempt in range(1, max_retries + 1):
try:
if _HAS_REQUESTS:
resp = session.get(download_url, timeout=30, allow_redirects=True)
if resp.status_code == 403 and 'images.unsplash.com' in download_url:
# Unsplash might need a source param; retry with source identifier
sep = '&' if '?' in download_url else '?'
retry_url = f"{download_url}{sep}source=smallclaw"
resp = session.get(retry_url, timeout=30, allow_redirects=True)
if resp.status_code != 200:
raise ValueError(f"HTTP {resp.status_code}")
data = resp.content
content_type = resp.headers.get('Content-Type', '')
else:
req = Request(download_url, headers={
'User-Agent': _ua,
'Accept': 'image/*,*/*;q=0.8',
})
with _img_opener.open(req, timeout=30) as resp_obj:
if resp_obj.getcode() != 200:
raise HTTPError(download_url, resp_obj.getcode(), f"HTTP {resp_obj.getcode()}", resp_obj.headers, None)
data = resp_obj.read()
content_type = resp_obj.headers.get('Content-Type', '')
# Validate: must have some content
if not data or len(data) < 16:
raise ValueError(f"Response too small ({len(data) if data else 0} bytes)")
# Validate: reject clearly non-image content (HTML pages, etc.)
ct_lower = (content_type or '').lower().split(';')[0].strip()
if ct_lower and ct_lower not in ('image/jpeg', 'image/jpg', 'image/png',
'image/gif', 'image/webp', 'image/bmp', 'image/tiff',
'application/octet-stream', 'binary/octet-stream', ''):
# If content-type is text/html or similar, it's not an image
if ct_lower.startswith('text/') or ct_lower in ('application/html', 'application/xml'):
raise ValueError(f"Non-image Content-Type: {content_type}")
# For other unknown types, check magic bytes instead of rejecting
# Determine extension from content or magic bytes
if not ext:
ext_from_ct = _detect_image_ext_from_headers(content_type)
ext_from_magic = _detect_image_ext(data)
ext = ext_from_ct or ext_from_magic or '.jpg'
fname = f"slide{i + 1}_image{ext}"
dest = os.path.join(project_dir, fname)
with open(dest, 'wb') as f:
f.write(data)
slide_spec["image_path"] = fname
last_error = None
print(f"[pptx_gen] Downloaded image for slide {i+1}: {len(data)} bytes -> {fname}", file=sys.stderr, flush=True)
break
except Exception as e:
last_error = e
if attempt < max_retries:
wait = 2 ** attempt
print(f"[pptx_gen] Download attempt {attempt}/{max_retries} failed for slide {i+1}: {e}. Retrying in {wait}s...", file=sys.stderr, flush=True)
time.sleep(wait)
else:
print(f"[pptx_gen] Download failed after {max_retries} attempts for slide {i+1}: {e}", file=sys.stderr, flush=True)
if last_error:
domain = ''
try:
from urllib.parse import urlparse
domain = urlparse(image_url).netloc
except Exception:
pass
err_msg = f"Failed to download image for slide {i+1} from {domain or image_url}: {last_error}"
warnings.append(err_msg)
slide_spec.pop("image_url", None)
# ─── Download Page Generator ────────────────────────────────────────────────────
def _create_download_page(pptx_path: str, download_url: str, title: str, slide_count: int):
@@ -865,8 +673,11 @@ def generate(spec: dict, workspace_path: str) -> dict:
if ip or iu:
print(f"[pptx_gen] slide {i+1}: type={s.get('type','?')} image_path='{ip}' image_url='{iu[:60] if iu else ''}'", file=sys.stderr, flush=True)
# Download image_url slides into project_dir
_download_images(slides_spec, project_dir, warnings)
# Warn about any remaining image_url slides (downloading is now handled by TypeScript)
for s in slides_spec:
if s.get("image_url") and not s.get("image_path"):
warnings.append(f"image_url not downloaded (TypeScript should handle this): {s['image_url'][:80]}")
s.pop("image_url", None)
if is_edit:
output_path = existing_path
+69 -32
View File
@@ -529,45 +529,66 @@ export const pptxTool: import('./registry.js').Tool = {
if (!spec.template && config.ppt?.template) spec.template = config.ppt.template;
if (!spec.default_skin && config.ppt?.skin) spec.default_skin = config.ppt.skin;
// Fix deprecated/defunct image URLs before sending to Python
// Fix deprecated/defunct image URLs before downloading
for (const slide of spec.slides) {
if (slide.image_url) {
// source.unsplash.com is shut down — redirect to images.unsplash.com
slide.image_url = slide.image_url.replace(/^https?:\/\/source\.unsplash\.com/i, 'https://images.unsplash.com');
}
}
// Resolve image_search: download images by keyword before creating slides
const slidesNeedingImages = spec.slides.filter(s => s.image_search && !s.image_path && !s.image_url);
if (slidesNeedingImages.length > 0) {
const projectSlug = (spec.title || 'presentation').replace(/[^a-zA-Z0-9가-힣_\-]/g, '_').toLowerCase().slice(0, 60);
const projectDir = path.join(workspacePath, projectSlug);
console.log(`[pptx] image_search: resolving ${slidesNeedingImages.length} slides, projectDir=${projectDir}`);
if (!fs.existsSync(projectDir)) {
fs.mkdirSync(projectDir, { recursive: true });
}
for (const slide of spec.slides) {
if (!slide.image_search || slide.image_path || slide.image_url) continue;
const idx = spec.slides.indexOf(slide);
const sendSSE = (args as any)?._sendSSE as SSESender | undefined;
const localPath = await searchAndDownloadImage(slide.image_search, projectDir, idx, sendSSE);
console.log(`[pptx] image_search: slide ${idx + 1} keyword="${slide.image_search}" -> localPath="${localPath}"`);
// ── Phase 1: Create project directory ──
const projectSlug = (spec.title || 'presentation').replace(/[^a-zA-Z0-9가-힣_\-]/g, '_').toLowerCase().slice(0, 60);
const projectDir = path.join(workspacePath, projectSlug);
if (!fs.existsSync(projectDir)) {
fs.mkdirSync(projectDir, { recursive: true });
}
console.log(`[pptx] project dir: ${projectDir}`);
// ── Phase 2: Download all images (image_search + image_url) ──
const sendSSE = (args as any)?._sendSSE as SSESender | undefined;
for (let i = 0; i < spec.slides.length; i++) {
const slide = spec.slides[i];
// image_search takes priority
if (slide.image_search && !slide.image_path) {
const localPath = await searchAndDownloadImage(slide.image_search, projectDir, i, sendSSE);
console.log(`[pptx] slide ${i + 1} image_search="${slide.image_search}" -> "${localPath}"`);
if (localPath) {
const fullPath = path.join(projectDir, localPath);
console.log(`[pptx] image_search: file exists=${fs.existsSync(fullPath)} size=${fs.statSync(fullPath).size} at ${fullPath}`);
slide.image_path = localPath;
}
delete slide.image_search;
// image_search overrides image_url
if (slide.image_url) delete slide.image_url;
continue;
}
// image_url as fallback
if (slide.image_url && !slide.image_path) {
const ext = slide.image_url.match(/\.(png|webp|gif)/i) ? '.png' : '.jpg';
const fname = `slide${i + 1}_image${ext}`;
const dest = path.join(projectDir, fname);
sendSSE?.('info', { message: `📥 Slide ${i + 1}: downloading image URL...` });
if (await downloadImageByUrl(slide.image_url, dest)) {
slide.image_path = fname;
console.log(`[pptx] slide ${i + 1} image_url downloaded -> ${fname}`);
} else {
console.log(`[pptx] slide ${i + 1} image_url failed: ${slide.image_url.slice(0, 80)}`);
sendSSE?.('info', { message: `⚠️ Slide ${i + 1}: image URL download failed` });
}
delete slide.image_url;
}
}
// Debug: log final slide specs for image slides
for (const [i, s] of spec.slides.entries()) {
if (s.image_path || s.image_url || s.image_search) {
console.log(`[pptx] slide ${i + 1}: type=${s.type} image_path="${s.image_path}" image_url="${s.image_url}"`);
if (s.image_path) {
const full = path.join(projectDir, s.image_path);
console.log(`[pptx] slide ${i + 1}: type=${s.type} image_path="${s.image_path}" exists=${fs.existsSync(full)}`);
}
}
// ── Phase 3: Generate PPTX ──
return await generateWithPython(spec, workspacePath);
},
};
@@ -646,27 +667,43 @@ export const editPptxTool: import('./registry.js').Tool = {
return { success: false, error: 'spec.slides must be a non-empty array' };
}
// Fix deprecated/defunct image URLs
// Fix deprecated/defunct image URLs before downloading
for (const slide of spec.slides) {
if (slide.image_url) {
slide.image_url = slide.image_url.replace(/^https?:\/\/source\.unsplash\.com/i, 'https://images.unsplash.com');
}
}
// Resolve image_search: download images by keyword before creating slides
const slidesNeedingImages = spec.slides.filter(s => s.image_search && !s.image_path && !s.image_url);
if (slidesNeedingImages.length > 0) {
// For edit, put downloads in the same directory as the existing PPTX
const editProjectDir = path.dirname(absPath);
for (const slide of spec.slides) {
if (!slide.image_search || slide.image_path || slide.image_url) continue;
const idx = spec.slides.indexOf(slide);
const sendSSE = (args as any)?._sendSSE as SSESender | undefined;
const localPath = await searchAndDownloadImage(slide.image_search, editProjectDir, idx, sendSSE);
// ── Download all images into the PPTX directory ──
const editProjectDir = path.dirname(absPath);
const sendSSE = (args as any)?._sendSSE as SSESender | undefined;
for (let i = 0; i < spec.slides.length; i++) {
const slide = spec.slides[i];
// image_search takes priority
if (slide.image_search && !slide.image_path) {
const localPath = await searchAndDownloadImage(slide.image_search, editProjectDir, i, sendSSE);
if (localPath) {
slide.image_path = localPath;
}
delete slide.image_search;
if (slide.image_url) delete slide.image_url;
continue;
}
// image_url as fallback
if (slide.image_url && !slide.image_path) {
const ext = slide.image_url.match(/\.(png|webp|gif)/i) ? '.png' : '.jpg';
const fname = `slide${i + 1}_image${ext}`;
const dest = path.join(editProjectDir, fname);
sendSSE?.('info', { message: `📥 Slide ${i + 1}: downloading image URL...` });
if (await downloadImageByUrl(slide.image_url, dest)) {
slide.image_path = fname;
} else {
sendSSE?.('info', { message: `⚠️ Slide ${i + 1}: image URL download failed` });
}
delete slide.image_url;
}
}
-24
View File
@@ -1,24 +0,0 @@
<!DOCTYPE html>
<html lang="ko">
<head>
<meta charset="UTF-8">
<title>가금류 사진 - 다운로드</title>
<style>
body { font-family: 'Malgun Gothic', sans-serif; max-width: 600px; margin: 60px auto; padding: 20px; text-align: center; background: #f8f9fa; }
.card { background: white; border-radius: 12px; padding: 40px 30px; box-shadow: 0 4px 20px rgba(0,0,0,0.08); }
h1 { color: #1a1a2e; font-size: 24px; margin-bottom: 10px; }
p { color: #5f6f86; margin-bottom: 30px; }
.btn { display: inline-block; background: #1668e3; color: white; text-decoration: none; padding: 14px 36px; border-radius: 8px; font-size: 16px; font-weight: bold; transition: background 0.2s; }
.btn:hover { background: #1255bb; }
.meta { margin-top: 20px; font-size: 12px; color: #888; }
</style>
</head>
<body>
<div class="card">
<h1>가금류 사진</h1>
<p>슬라이드 4장이 준비되었습니다.</p>
<a class="btn" href="가금류_사진.pptx" download>프레젠테이션 다운로드</a>
<div class="meta">D:\smallclaw\workspace\가금류_사진\가금류_사진.pptx</div>
</div>
</body>
</html>
Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.5 MiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.5 MiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.2 MiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.5 MiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 65 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 510 KiB

Binary file not shown.
-24
View File
@@ -1,24 +0,0 @@
<!DOCTYPE html>
<html lang="ko">
<head>
<meta charset="UTF-8">
<title>고양이 사진 - 다운로드</title>
<style>
body { font-family: 'Malgun Gothic', sans-serif; max-width: 600px; margin: 60px auto; padding: 20px; text-align: center; background: #f8f9fa; }
.card { background: white; border-radius: 12px; padding: 40px 30px; box-shadow: 0 4px 20px rgba(0,0,0,0.08); }
h1 { color: #1a1a2e; font-size: 24px; margin-bottom: 10px; }
p { color: #5f6f86; margin-bottom: 30px; }
.btn { display: inline-block; background: #1668e3; color: white; text-decoration: none; padding: 14px 36px; border-radius: 8px; font-size: 16px; font-weight: bold; transition: background 0.2s; }
.btn:hover { background: #1255bb; }
.meta { margin-top: 20px; font-size: 12px; color: #888; }
</style>
</head>
<body>
<div class="card">
<h1>고양이 사진</h1>
<p>슬라이드 4장이 준비되었습니다.</p>
<a class="btn" href="고양이_사진.pptx" download>프레젠테이션 다운로드</a>
<div class="meta">D:\smallclaw\workspace\고양이_사진\고양이_사진.pptx</div>
</div>
</body>
</html>
Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.5 MiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.2 MiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.1 MiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.9 MiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 402 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 37 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 21 KiB

Binary file not shown.