feat: 스튜디오 앱 사진→동영상(LTX Image-to-Video)

studio-app.html에 시작 이미지 업로드/제거 UI + strength 슬라이더 추가,
imagegen.ts에 ltxImageToVideoWorkflow (ComfyUI 공식 ltxv_image_to_video.json
템플릿 기반 LoadImage→LTXVImgToVideo→LTXVConditioning 그래프) 추가.
실제 생성 테스트로 프레임0(원본 이미지 일치)·프레임15(모션 발생) 확인 완료된
작업을 커밋만 안 하고 남겨뒀던 것.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-08-07 13:13:58 +09:00
co-authored by Claude Sonnet 5
parent 61885cb923
commit 02247e4ee3
2 changed files with 114 additions and 7 deletions
+62 -7
View File
@@ -239,6 +239,31 @@ function ltxVideoWorkflow(p: { prompt: string; negative_prompt: string; width: n
};
}
// Image-to-video variant — same LTX checkpoint/text-encoder as ltxVideoWorkflow above,
// but the source photo goes through LTXVImgToVideo *before* LTXVConditioning (wiring
// copied from ComfyUI's bundled ltxv_image_to_video.json template, node IDs kept
// matching that template for traceability). LTXVImgToVideo bakes the image into both
// the conditioning and the starting latent, replacing EmptyLTXVLatentVideo entirely —
// there's no plain "add an image on top of the txt2vid graph" path, the whole latent
// source changes.
function ltxImageToVideoWorkflow(p: { imageFilename: string; prompt: string; negative_prompt: string; width: number; height: number; length: number; fps: number; steps: number; guidance_scale: number; strength: number; seed: number; ckpt_name?: string }) {
return {
'38': { class_type: 'CLIPLoader', inputs: { clip_name: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'ltxv', device: 'default' } },
'44': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: p.ckpt_name || 'ltxv-2b-0.9.8-distilled-fp8.safetensors' } },
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.prompt } },
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.negative_prompt } },
'78': { class_type: 'LoadImage', inputs: { image: p.imageFilename } },
'77': { class_type: 'LTXVImgToVideo', inputs: { positive: ['6', 0], negative: ['7', 0], vae: ['44', 2], image: ['78', 0], width: p.width, height: p.height, length: p.length, batch_size: 1, strength: p.strength } },
'69': { class_type: 'LTXVConditioning', inputs: { positive: ['77', 0], negative: ['77', 1], frame_rate: p.fps } },
'71': { class_type: 'LTXVScheduler', inputs: { steps: p.steps, max_shift: 2.05, base_shift: 0.95, stretch: true, terminal: 0.1, latent: ['77', 2] } },
'73': { class_type: 'KSamplerSelect', inputs: { sampler_name: 'euler' } },
'72': { class_type: 'SamplerCustom', inputs: { model: ['44', 0], add_noise: true, noise_seed: p.seed, cfg: p.guidance_scale, positive: ['69', 0], negative: ['69', 1], sampler: ['73', 0], sigmas: ['71', 0], latent_image: ['77', 2] } },
'8': { class_type: 'VAEDecode', inputs: { samples: ['72', 0], vae: ['44', 2] } },
'80': { class_type: 'CreateVideo', inputs: { images: ['8', 0], fps: p.fps } },
'81': { class_type: 'SaveVideo', inputs: { video: ['80', 0], filename_prefix: 'ltx_i2v_api', format: 'mp4', codec: 'h264' } },
};
}
// Same submit/poll plumbing as comfyGenerateImage but the SaveVideo node's output file
// isn't a PNG — copy verbatim rather than assuming an image extension.
async function comfyGenerateVideo(promptGraph: Record<string, any>, dst: string, timeoutMs: number): Promise<{ output: string } | { error: string }> {
@@ -496,11 +521,14 @@ export const videoGenerateTool = {
description: [
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
'Three models available via the "model" param: "ltx" (default — LTX-Video 2B distilled via ComfyUI, ~25-40s), "ltx_hq" (LTX-Video 2B non-distilled/"dev" checkpoint via ComfyUI — noticeably better motion coherence and detail, ~40-90s, VRAM headroom is tight at ~10GB/12GB so avoid stacking a large image_generate call at the same time), and "cogvideox" (THUDM CogVideoX-2B — still the old venv-script path, slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try a couple and compare if unsure which fits the request.',
'Optional "image" param turns this into image-to-video: pass a path to an existing photo and the video will start from it instead of pure noise (only supported for "ltx"/"ltx_hq", not "cogvideox" — no image-conditioned CogVideoX checkpoint is installed).',
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
].join('\n'),
schema: {
prompt: 'Text description of the video/scene to generate (English works best)',
model: '"ltx" (default, fast, distilled), "ltx_hq" (LTX-Video non-distilled "dev" checkpoint, better quality, slower), or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
image: 'Path to a source photo to animate (image-to-video). Optional — omit for plain text-to-video. Only works with model "ltx"/"ltx_hq"; ignored (with an error) for "cogvideox".',
strength: 'How much the video is allowed to drift from the source image, 0-1 (only used with "image", default 0.5 — LTX\'s own example workflow uses 0.15 for near-static motion, higher values allow more change)',
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
height: 'Video height in pixels, multiple of 32 (default 480)',
@@ -515,6 +543,8 @@ export const videoGenerateTool = {
properties: {
prompt: { type: 'string' },
model: { type: 'string', enum: ['ltx', 'ltx_hq', 'cogvideox'] },
image: { type: 'string' },
strength: { type: 'number' },
negative_prompt: { type: 'string' },
width: { type: 'number' },
height: { type: 'number' },
@@ -543,6 +573,21 @@ export const videoGenerateTool = {
outPath = path.resolve(workspacePath, outPath);
}
const imageArg = String(args?.image || '').trim();
if (imageArg && isCogVideoX) {
return { success: false, error: 'image-to-video is not supported for model "cogvideox" — use "ltx" or "ltx_hq"' };
}
let inputImageFilename: string | undefined;
if (imageArg) {
const srcPath = path.isAbsolute(imageArg) ? imageArg : path.resolve(workspacePath, imageArg);
if (!isPathInsideDir(workspacePath, srcPath)) return { success: false, error: 'Access denied: path escapes workspace' };
if (!fs.existsSync(srcPath)) return { success: false, error: `Source image not found: ${imageArg}` };
fs.mkdirSync(COMFY_INPUT_DIR, { recursive: true });
inputImageFilename = `vid_src_${Date.now()}${path.extname(srcPath) || '.png'}`;
fs.copyFileSync(srcPath, path.join(COMFY_INPUT_DIR, inputImageFilename));
}
const strength = Math.min(1, Math.max(0, toFiniteNumber(args?.strength, 0.5)));
const negativePromptRaw = args?.negative_prompt || 'worst quality, blurry, distorted, deformed';
const [translatedPrompt, translatedNegative] = await Promise.all([
translatePromptToEnglish(prompt),
@@ -579,13 +624,23 @@ export const videoGenerateTool = {
// whatever was requested to the nearest valid value instead of erroring.
const requestedFrames = toFiniteNumber(args?.num_frames, 65);
num_frames = Math.max(9, Math.round((requestedFrames - 1) / 8) * 8 + 1);
const graph = ltxVideoWorkflow({
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, length: num_frames, fps, steps, guidance_scale,
seed: randomSeed(),
ckpt_name: isHQ ? 'ltxv-2b-0.9.6-dev-04-25.safetensors' : undefined,
});
const ckpt_name = isHQ ? 'ltxv-2b-0.9.6-dev-04-25.safetensors' : undefined;
const graph = inputImageFilename
? ltxImageToVideoWorkflow({
imageFilename: inputImageFilename,
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, length: num_frames, fps, steps, guidance_scale, strength,
seed: randomSeed(),
ckpt_name,
})
: ltxVideoWorkflow({
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, length: num_frames, fps, steps, guidance_scale,
seed: randomSeed(),
ckpt_name,
});
// The non-distilled "dev" checkpoint is ~1.4x the weight size of the distilled one and
// runs full CFG (two forward passes/step instead of one), so it's meaningfully slower —
// give it more headroom than the regular ltx path's 300s.