feat: 스튜디오 앱 사진→동영상(LTX Image-to-Video)
studio-app.html에 시작 이미지 업로드/제거 UI + strength 슬라이더 추가, imagegen.ts에 ltxImageToVideoWorkflow (ComfyUI 공식 ltxv_image_to_video.json 템플릿 기반 LoadImage→LTXVImgToVideo→LTXVConditioning 그래프) 추가. 실제 생성 테스트로 프레임0(원본 이미지 일치)·프레임15(모션 발생) 확인 완료된 작업을 커밋만 안 하고 남겨뒀던 것. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
+62
-7
@@ -239,6 +239,31 @@ function ltxVideoWorkflow(p: { prompt: string; negative_prompt: string; width: n
|
||||
};
|
||||
}
|
||||
|
||||
// Image-to-video variant — same LTX checkpoint/text-encoder as ltxVideoWorkflow above,
|
||||
// but the source photo goes through LTXVImgToVideo *before* LTXVConditioning (wiring
|
||||
// copied from ComfyUI's bundled ltxv_image_to_video.json template, node IDs kept
|
||||
// matching that template for traceability). LTXVImgToVideo bakes the image into both
|
||||
// the conditioning and the starting latent, replacing EmptyLTXVLatentVideo entirely —
|
||||
// there's no plain "add an image on top of the txt2vid graph" path, the whole latent
|
||||
// source changes.
|
||||
function ltxImageToVideoWorkflow(p: { imageFilename: string; prompt: string; negative_prompt: string; width: number; height: number; length: number; fps: number; steps: number; guidance_scale: number; strength: number; seed: number; ckpt_name?: string }) {
|
||||
return {
|
||||
'38': { class_type: 'CLIPLoader', inputs: { clip_name: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'ltxv', device: 'default' } },
|
||||
'44': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: p.ckpt_name || 'ltxv-2b-0.9.8-distilled-fp8.safetensors' } },
|
||||
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.prompt } },
|
||||
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.negative_prompt } },
|
||||
'78': { class_type: 'LoadImage', inputs: { image: p.imageFilename } },
|
||||
'77': { class_type: 'LTXVImgToVideo', inputs: { positive: ['6', 0], negative: ['7', 0], vae: ['44', 2], image: ['78', 0], width: p.width, height: p.height, length: p.length, batch_size: 1, strength: p.strength } },
|
||||
'69': { class_type: 'LTXVConditioning', inputs: { positive: ['77', 0], negative: ['77', 1], frame_rate: p.fps } },
|
||||
'71': { class_type: 'LTXVScheduler', inputs: { steps: p.steps, max_shift: 2.05, base_shift: 0.95, stretch: true, terminal: 0.1, latent: ['77', 2] } },
|
||||
'73': { class_type: 'KSamplerSelect', inputs: { sampler_name: 'euler' } },
|
||||
'72': { class_type: 'SamplerCustom', inputs: { model: ['44', 0], add_noise: true, noise_seed: p.seed, cfg: p.guidance_scale, positive: ['69', 0], negative: ['69', 1], sampler: ['73', 0], sigmas: ['71', 0], latent_image: ['77', 2] } },
|
||||
'8': { class_type: 'VAEDecode', inputs: { samples: ['72', 0], vae: ['44', 2] } },
|
||||
'80': { class_type: 'CreateVideo', inputs: { images: ['8', 0], fps: p.fps } },
|
||||
'81': { class_type: 'SaveVideo', inputs: { video: ['80', 0], filename_prefix: 'ltx_i2v_api', format: 'mp4', codec: 'h264' } },
|
||||
};
|
||||
}
|
||||
|
||||
// Same submit/poll plumbing as comfyGenerateImage but the SaveVideo node's output file
|
||||
// isn't a PNG — copy verbatim rather than assuming an image extension.
|
||||
async function comfyGenerateVideo(promptGraph: Record<string, any>, dst: string, timeoutMs: number): Promise<{ output: string } | { error: string }> {
|
||||
@@ -496,11 +521,14 @@ export const videoGenerateTool = {
|
||||
description: [
|
||||
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
|
||||
'Three models available via the "model" param: "ltx" (default — LTX-Video 2B distilled via ComfyUI, ~25-40s), "ltx_hq" (LTX-Video 2B non-distilled/"dev" checkpoint via ComfyUI — noticeably better motion coherence and detail, ~40-90s, VRAM headroom is tight at ~10GB/12GB so avoid stacking a large image_generate call at the same time), and "cogvideox" (THUDM CogVideoX-2B — still the old venv-script path, slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try a couple and compare if unsure which fits the request.',
|
||||
'Optional "image" param turns this into image-to-video: pass a path to an existing photo and the video will start from it instead of pure noise (only supported for "ltx"/"ltx_hq", not "cogvideox" — no image-conditioned CogVideoX checkpoint is installed).',
|
||||
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
|
||||
].join('\n'),
|
||||
schema: {
|
||||
prompt: 'Text description of the video/scene to generate (English works best)',
|
||||
model: '"ltx" (default, fast, distilled), "ltx_hq" (LTX-Video non-distilled "dev" checkpoint, better quality, slower), or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
|
||||
image: 'Path to a source photo to animate (image-to-video). Optional — omit for plain text-to-video. Only works with model "ltx"/"ltx_hq"; ignored (with an error) for "cogvideox".',
|
||||
strength: 'How much the video is allowed to drift from the source image, 0-1 (only used with "image", default 0.5 — LTX\'s own example workflow uses 0.15 for near-static motion, higher values allow more change)',
|
||||
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
|
||||
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
|
||||
height: 'Video height in pixels, multiple of 32 (default 480)',
|
||||
@@ -515,6 +543,8 @@ export const videoGenerateTool = {
|
||||
properties: {
|
||||
prompt: { type: 'string' },
|
||||
model: { type: 'string', enum: ['ltx', 'ltx_hq', 'cogvideox'] },
|
||||
image: { type: 'string' },
|
||||
strength: { type: 'number' },
|
||||
negative_prompt: { type: 'string' },
|
||||
width: { type: 'number' },
|
||||
height: { type: 'number' },
|
||||
@@ -543,6 +573,21 @@ export const videoGenerateTool = {
|
||||
outPath = path.resolve(workspacePath, outPath);
|
||||
}
|
||||
|
||||
const imageArg = String(args?.image || '').trim();
|
||||
if (imageArg && isCogVideoX) {
|
||||
return { success: false, error: 'image-to-video is not supported for model "cogvideox" — use "ltx" or "ltx_hq"' };
|
||||
}
|
||||
let inputImageFilename: string | undefined;
|
||||
if (imageArg) {
|
||||
const srcPath = path.isAbsolute(imageArg) ? imageArg : path.resolve(workspacePath, imageArg);
|
||||
if (!isPathInsideDir(workspacePath, srcPath)) return { success: false, error: 'Access denied: path escapes workspace' };
|
||||
if (!fs.existsSync(srcPath)) return { success: false, error: `Source image not found: ${imageArg}` };
|
||||
fs.mkdirSync(COMFY_INPUT_DIR, { recursive: true });
|
||||
inputImageFilename = `vid_src_${Date.now()}${path.extname(srcPath) || '.png'}`;
|
||||
fs.copyFileSync(srcPath, path.join(COMFY_INPUT_DIR, inputImageFilename));
|
||||
}
|
||||
const strength = Math.min(1, Math.max(0, toFiniteNumber(args?.strength, 0.5)));
|
||||
|
||||
const negativePromptRaw = args?.negative_prompt || 'worst quality, blurry, distorted, deformed';
|
||||
const [translatedPrompt, translatedNegative] = await Promise.all([
|
||||
translatePromptToEnglish(prompt),
|
||||
@@ -579,13 +624,23 @@ export const videoGenerateTool = {
|
||||
// whatever was requested to the nearest valid value instead of erroring.
|
||||
const requestedFrames = toFiniteNumber(args?.num_frames, 65);
|
||||
num_frames = Math.max(9, Math.round((requestedFrames - 1) / 8) * 8 + 1);
|
||||
const graph = ltxVideoWorkflow({
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width, height, length: num_frames, fps, steps, guidance_scale,
|
||||
seed: randomSeed(),
|
||||
ckpt_name: isHQ ? 'ltxv-2b-0.9.6-dev-04-25.safetensors' : undefined,
|
||||
});
|
||||
const ckpt_name = isHQ ? 'ltxv-2b-0.9.6-dev-04-25.safetensors' : undefined;
|
||||
const graph = inputImageFilename
|
||||
? ltxImageToVideoWorkflow({
|
||||
imageFilename: inputImageFilename,
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width, height, length: num_frames, fps, steps, guidance_scale, strength,
|
||||
seed: randomSeed(),
|
||||
ckpt_name,
|
||||
})
|
||||
: ltxVideoWorkflow({
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width, height, length: num_frames, fps, steps, guidance_scale,
|
||||
seed: randomSeed(),
|
||||
ckpt_name,
|
||||
});
|
||||
// The non-distilled "dev" checkpoint is ~1.4x the weight size of the distilled one and
|
||||
// runs full CFG (two forward passes/step instead of one), so it's meaningfully slower —
|
||||
// give it more headroom than the regular ltx path's 300s.
|
||||
|
||||
@@ -246,6 +246,21 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon
|
||||
<input type="number" id="f-fps" value="24">
|
||||
</div>
|
||||
</div>
|
||||
<div class="st-field" id="video-image-field">
|
||||
<label>시작 이미지 (선택 — 사진에서 동영상 만들기)</label>
|
||||
<input type="file" id="f-video-image" accept="image/*" style="display:none" onchange="onVideoImageSelected(this.files[0])">
|
||||
<div class="st-dropzone" id="video-image-dropzone" onclick="document.getElementById('f-video-image').click()">사진을 선택하려면 탭하세요 (없으면 텍스트만으로 생성)</div>
|
||||
<div id="video-image-preview" style="display:none;align-items:center;gap:8px;margin-top:4px">
|
||||
<img id="video-image-thumb" style="width:48px;height:48px;object-fit:cover;border-radius:6px">
|
||||
<button type="button" class="hdr-btn" style="font-size:11px" onclick="clearVideoImage()">✕ 제거</button>
|
||||
</div>
|
||||
<div class="st-hint" id="video-image-cogvideox-warn" style="display:none;color:#ef4444">CogVideoX는 이미지 시작을 지원하지 않아요 — LTX/LTX고화질로 바꾸세요.</div>
|
||||
</div>
|
||||
<div class="st-field" id="video-strength-field" style="display:none">
|
||||
<label>이미지 반영 강도: <span id="video-strength-val">0.5</span></label>
|
||||
<input type="range" id="f-video-strength" min="0" max="1" step="0.05" value="0.5" oninput="document.getElementById('video-strength-val').textContent=this.value">
|
||||
<div class="st-hint" style="margin:0">낮음 = 시작 이미지에 거의 고정(움직임 적음) · 높음 = 프롬프트 쪽으로 더 많이 변형</div>
|
||||
</div>
|
||||
</div>
|
||||
<div id="row-edit-only" style="display:none;flex-direction:column;gap:10px;">
|
||||
<div class="st-field">
|
||||
@@ -867,9 +882,42 @@ function onQualityChange(){
|
||||
}
|
||||
}
|
||||
|
||||
let uploadedVideoImagePath=null;
|
||||
async function onVideoImageSelected(file){
|
||||
if(!file) return;
|
||||
const form=new FormData();
|
||||
form.append('image', file);
|
||||
uploadedVideoImagePath=null;
|
||||
document.getElementById('video-image-dropzone').textContent='업로드 중...';
|
||||
try{
|
||||
const r=await fetch('/api/upload/image',{method:'POST',headers:authH(),body:form});
|
||||
const d=await r.json();
|
||||
if(!r.ok) throw new Error(d.error||'업로드 실패');
|
||||
uploadedVideoImagePath=d.path;
|
||||
document.getElementById('video-image-thumb').src=d.url || URL.createObjectURL(file);
|
||||
document.getElementById('video-image-preview').style.display='flex';
|
||||
document.getElementById('video-image-dropzone').style.display='none';
|
||||
document.getElementById('video-strength-field').style.display='flex';
|
||||
}catch(e){
|
||||
document.getElementById('video-image-dropzone').textContent='업로드 실패: '+e.message;
|
||||
}
|
||||
}
|
||||
function clearVideoImage(){
|
||||
uploadedVideoImagePath=null;
|
||||
document.getElementById('f-video-image').value='';
|
||||
document.getElementById('video-image-preview').style.display='none';
|
||||
document.getElementById('video-image-dropzone').style.display='block';
|
||||
document.getElementById('video-image-dropzone').textContent='사진을 선택하려면 탭하세요 (없으면 텍스트만으로 생성)';
|
||||
document.getElementById('video-strength-field').style.display='none';
|
||||
}
|
||||
function onVideoModelChange(){
|
||||
const m=document.getElementById('f-video-model').value;
|
||||
const guidanceEl=document.getElementById('f-guidance');
|
||||
const isCogVideoX = m==='cogvideox';
|
||||
document.getElementById('video-image-cogvideox-warn').style.display = isCogVideoX ? 'block' : 'none';
|
||||
document.getElementById('video-image-dropzone').style.pointerEvents = isCogVideoX ? 'none' : 'auto';
|
||||
document.getElementById('video-image-dropzone').style.opacity = isCogVideoX ? '0.5' : '1';
|
||||
if(isCogVideoX) clearVideoImage();
|
||||
if(m==='cogvideox'){
|
||||
document.getElementById('f-frames').value=49;
|
||||
document.getElementById('f-fps').value=8;
|
||||
@@ -1109,6 +1157,10 @@ async function generate(){
|
||||
body.model=document.getElementById('f-video-model').value;
|
||||
body.num_frames=parseInt(document.getElementById('f-frames').value)||undefined;
|
||||
body.fps=parseInt(document.getElementById('f-fps').value)||undefined;
|
||||
if(uploadedVideoImagePath && body.model!=='cogvideox'){
|
||||
body.image=uploadedVideoImagePath;
|
||||
body.strength=parseFloat(document.getElementById('f-video-strength').value)||0.5;
|
||||
}
|
||||
}
|
||||
|
||||
try{
|
||||
|
||||
Reference in New Issue
Block a user