feat: 스튜디오 앱 사진→동영상(LTX Image-to-Video)

studio-app.html에 시작 이미지 업로드/제거 UI + strength 슬라이더 추가,
imagegen.ts에 ltxImageToVideoWorkflow (ComfyUI 공식 ltxv_image_to_video.json
템플릿 기반 LoadImage→LTXVImgToVideo→LTXVConditioning 그래프) 추가.
실제 생성 테스트로 프레임0(원본 이미지 일치)·프레임15(모션 발생) 확인 완료된
작업을 커밋만 안 하고 남겨뒀던 것.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-08-07 13:13:58 +09:00
co-authored by Claude Sonnet 5
parent 61885cb923
commit 02247e4ee3
2 changed files with 114 additions and 7 deletions
+62 -7
View File
@@ -239,6 +239,31 @@ function ltxVideoWorkflow(p: { prompt: string; negative_prompt: string; width: n
};
}
// Image-to-video variant — same LTX checkpoint/text-encoder as ltxVideoWorkflow above,
// but the source photo goes through LTXVImgToVideo *before* LTXVConditioning (wiring
// copied from ComfyUI's bundled ltxv_image_to_video.json template, node IDs kept
// matching that template for traceability). LTXVImgToVideo bakes the image into both
// the conditioning and the starting latent, replacing EmptyLTXVLatentVideo entirely —
// there's no plain "add an image on top of the txt2vid graph" path, the whole latent
// source changes.
function ltxImageToVideoWorkflow(p: { imageFilename: string; prompt: string; negative_prompt: string; width: number; height: number; length: number; fps: number; steps: number; guidance_scale: number; strength: number; seed: number; ckpt_name?: string }) {
return {
'38': { class_type: 'CLIPLoader', inputs: { clip_name: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'ltxv', device: 'default' } },
'44': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: p.ckpt_name || 'ltxv-2b-0.9.8-distilled-fp8.safetensors' } },
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.prompt } },
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.negative_prompt } },
'78': { class_type: 'LoadImage', inputs: { image: p.imageFilename } },
'77': { class_type: 'LTXVImgToVideo', inputs: { positive: ['6', 0], negative: ['7', 0], vae: ['44', 2], image: ['78', 0], width: p.width, height: p.height, length: p.length, batch_size: 1, strength: p.strength } },
'69': { class_type: 'LTXVConditioning', inputs: { positive: ['77', 0], negative: ['77', 1], frame_rate: p.fps } },
'71': { class_type: 'LTXVScheduler', inputs: { steps: p.steps, max_shift: 2.05, base_shift: 0.95, stretch: true, terminal: 0.1, latent: ['77', 2] } },
'73': { class_type: 'KSamplerSelect', inputs: { sampler_name: 'euler' } },
'72': { class_type: 'SamplerCustom', inputs: { model: ['44', 0], add_noise: true, noise_seed: p.seed, cfg: p.guidance_scale, positive: ['69', 0], negative: ['69', 1], sampler: ['73', 0], sigmas: ['71', 0], latent_image: ['77', 2] } },
'8': { class_type: 'VAEDecode', inputs: { samples: ['72', 0], vae: ['44', 2] } },
'80': { class_type: 'CreateVideo', inputs: { images: ['8', 0], fps: p.fps } },
'81': { class_type: 'SaveVideo', inputs: { video: ['80', 0], filename_prefix: 'ltx_i2v_api', format: 'mp4', codec: 'h264' } },
};
}
// Same submit/poll plumbing as comfyGenerateImage but the SaveVideo node's output file
// isn't a PNG — copy verbatim rather than assuming an image extension.
async function comfyGenerateVideo(promptGraph: Record<string, any>, dst: string, timeoutMs: number): Promise<{ output: string } | { error: string }> {
@@ -496,11 +521,14 @@ export const videoGenerateTool = {
description: [
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
'Three models available via the "model" param: "ltx" (default — LTX-Video 2B distilled via ComfyUI, ~25-40s), "ltx_hq" (LTX-Video 2B non-distilled/"dev" checkpoint via ComfyUI — noticeably better motion coherence and detail, ~40-90s, VRAM headroom is tight at ~10GB/12GB so avoid stacking a large image_generate call at the same time), and "cogvideox" (THUDM CogVideoX-2B — still the old venv-script path, slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try a couple and compare if unsure which fits the request.',
'Optional "image" param turns this into image-to-video: pass a path to an existing photo and the video will start from it instead of pure noise (only supported for "ltx"/"ltx_hq", not "cogvideox" — no image-conditioned CogVideoX checkpoint is installed).',
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
].join('\n'),
schema: {
prompt: 'Text description of the video/scene to generate (English works best)',
model: '"ltx" (default, fast, distilled), "ltx_hq" (LTX-Video non-distilled "dev" checkpoint, better quality, slower), or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
image: 'Path to a source photo to animate (image-to-video). Optional — omit for plain text-to-video. Only works with model "ltx"/"ltx_hq"; ignored (with an error) for "cogvideox".',
strength: 'How much the video is allowed to drift from the source image, 0-1 (only used with "image", default 0.5 — LTX\'s own example workflow uses 0.15 for near-static motion, higher values allow more change)',
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
height: 'Video height in pixels, multiple of 32 (default 480)',
@@ -515,6 +543,8 @@ export const videoGenerateTool = {
properties: {
prompt: { type: 'string' },
model: { type: 'string', enum: ['ltx', 'ltx_hq', 'cogvideox'] },
image: { type: 'string' },
strength: { type: 'number' },
negative_prompt: { type: 'string' },
width: { type: 'number' },
height: { type: 'number' },
@@ -543,6 +573,21 @@ export const videoGenerateTool = {
outPath = path.resolve(workspacePath, outPath);
}
const imageArg = String(args?.image || '').trim();
if (imageArg && isCogVideoX) {
return { success: false, error: 'image-to-video is not supported for model "cogvideox" — use "ltx" or "ltx_hq"' };
}
let inputImageFilename: string | undefined;
if (imageArg) {
const srcPath = path.isAbsolute(imageArg) ? imageArg : path.resolve(workspacePath, imageArg);
if (!isPathInsideDir(workspacePath, srcPath)) return { success: false, error: 'Access denied: path escapes workspace' };
if (!fs.existsSync(srcPath)) return { success: false, error: `Source image not found: ${imageArg}` };
fs.mkdirSync(COMFY_INPUT_DIR, { recursive: true });
inputImageFilename = `vid_src_${Date.now()}${path.extname(srcPath) || '.png'}`;
fs.copyFileSync(srcPath, path.join(COMFY_INPUT_DIR, inputImageFilename));
}
const strength = Math.min(1, Math.max(0, toFiniteNumber(args?.strength, 0.5)));
const negativePromptRaw = args?.negative_prompt || 'worst quality, blurry, distorted, deformed';
const [translatedPrompt, translatedNegative] = await Promise.all([
translatePromptToEnglish(prompt),
@@ -579,13 +624,23 @@ export const videoGenerateTool = {
// whatever was requested to the nearest valid value instead of erroring.
const requestedFrames = toFiniteNumber(args?.num_frames, 65);
num_frames = Math.max(9, Math.round((requestedFrames - 1) / 8) * 8 + 1);
const graph = ltxVideoWorkflow({
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, length: num_frames, fps, steps, guidance_scale,
seed: randomSeed(),
ckpt_name: isHQ ? 'ltxv-2b-0.9.6-dev-04-25.safetensors' : undefined,
});
const ckpt_name = isHQ ? 'ltxv-2b-0.9.6-dev-04-25.safetensors' : undefined;
const graph = inputImageFilename
? ltxImageToVideoWorkflow({
imageFilename: inputImageFilename,
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, length: num_frames, fps, steps, guidance_scale, strength,
seed: randomSeed(),
ckpt_name,
})
: ltxVideoWorkflow({
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, length: num_frames, fps, steps, guidance_scale,
seed: randomSeed(),
ckpt_name,
});
// The non-distilled "dev" checkpoint is ~1.4x the weight size of the distilled one and
// runs full CFG (two forward passes/step instead of one), so it's meaningfully slower —
// give it more headroom than the regular ltx path's 300s.
+52
View File
@@ -246,6 +246,21 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon
<input type="number" id="f-fps" value="24">
</div>
</div>
<div class="st-field" id="video-image-field">
<label>시작 이미지 (선택 — 사진에서 동영상 만들기)</label>
<input type="file" id="f-video-image" accept="image/*" style="display:none" onchange="onVideoImageSelected(this.files[0])">
<div class="st-dropzone" id="video-image-dropzone" onclick="document.getElementById('f-video-image').click()">사진을 선택하려면 탭하세요 (없으면 텍스트만으로 생성)</div>
<div id="video-image-preview" style="display:none;align-items:center;gap:8px;margin-top:4px">
<img id="video-image-thumb" style="width:48px;height:48px;object-fit:cover;border-radius:6px">
<button type="button" class="hdr-btn" style="font-size:11px" onclick="clearVideoImage()">✕ 제거</button>
</div>
<div class="st-hint" id="video-image-cogvideox-warn" style="display:none;color:#ef4444">CogVideoX는 이미지 시작을 지원하지 않아요 — LTX/LTX고화질로 바꾸세요.</div>
</div>
<div class="st-field" id="video-strength-field" style="display:none">
<label>이미지 반영 강도: <span id="video-strength-val">0.5</span></label>
<input type="range" id="f-video-strength" min="0" max="1" step="0.05" value="0.5" oninput="document.getElementById('video-strength-val').textContent=this.value">
<div class="st-hint" style="margin:0">낮음 = 시작 이미지에 거의 고정(움직임 적음) · 높음 = 프롬프트 쪽으로 더 많이 변형</div>
</div>
</div>
<div id="row-edit-only" style="display:none;flex-direction:column;gap:10px;">
<div class="st-field">
@@ -867,9 +882,42 @@ function onQualityChange(){
}
}
let uploadedVideoImagePath=null;
async function onVideoImageSelected(file){
if(!file) return;
const form=new FormData();
form.append('image', file);
uploadedVideoImagePath=null;
document.getElementById('video-image-dropzone').textContent='업로드 중...';
try{
const r=await fetch('/api/upload/image',{method:'POST',headers:authH(),body:form});
const d=await r.json();
if(!r.ok) throw new Error(d.error||'업로드 실패');
uploadedVideoImagePath=d.path;
document.getElementById('video-image-thumb').src=d.url || URL.createObjectURL(file);
document.getElementById('video-image-preview').style.display='flex';
document.getElementById('video-image-dropzone').style.display='none';
document.getElementById('video-strength-field').style.display='flex';
}catch(e){
document.getElementById('video-image-dropzone').textContent='업로드 실패: '+e.message;
}
}
function clearVideoImage(){
uploadedVideoImagePath=null;
document.getElementById('f-video-image').value='';
document.getElementById('video-image-preview').style.display='none';
document.getElementById('video-image-dropzone').style.display='block';
document.getElementById('video-image-dropzone').textContent='사진을 선택하려면 탭하세요 (없으면 텍스트만으로 생성)';
document.getElementById('video-strength-field').style.display='none';
}
function onVideoModelChange(){
const m=document.getElementById('f-video-model').value;
const guidanceEl=document.getElementById('f-guidance');
const isCogVideoX = m==='cogvideox';
document.getElementById('video-image-cogvideox-warn').style.display = isCogVideoX ? 'block' : 'none';
document.getElementById('video-image-dropzone').style.pointerEvents = isCogVideoX ? 'none' : 'auto';
document.getElementById('video-image-dropzone').style.opacity = isCogVideoX ? '0.5' : '1';
if(isCogVideoX) clearVideoImage();
if(m==='cogvideox'){
document.getElementById('f-frames').value=49;
document.getElementById('f-fps').value=8;
@@ -1109,6 +1157,10 @@ async function generate(){
body.model=document.getElementById('f-video-model').value;
body.num_frames=parseInt(document.getElementById('f-frames').value)||undefined;
body.fps=parseInt(document.getElementById('f-fps').value)||undefined;
if(uploadedVideoImagePath && body.model!=='cogvideox'){
body.image=uploadedVideoImagePath;
body.strength=parseFloat(document.getElementById('f-video-strength').value)||0.5;
}
}
try{