v4.1.18: 동영상 생성에 CogVideoX-2B 옵션 추가 + 스튜디오 UI 레이아웃 수정

video_generate에 model 파라미터 추가 — 기존 LTX-Video(빠름, 기본값)에
CogVideoX-2B(THUDM, 느리지만 화질 비교용)를 선택지로 추가. CogVideoX는
12GB VRAM 한계에 거의 딱 맞아(enable_model_cpu_offload에도 peak ~11GB)
해상도/프레임수 기본값 이상으로 올리면 여유가 없음.

스튜디오앱 동영상 탭에서 모델/프레임수/FPS 세 필드가 320px 폭 폼에
한 줄로 욱여넣어져 프레임수부터 우측 패널에 가려 안 보이던 레이아웃
버그 수정 — 모델 드롭다운을 단독 줄로 분리하고 프레임수/FPS는 그 아래
별도 행으로 이동.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-18 13:07:51 +09:00
co-authored by Claude Sonnet 5
parent 16c3d0e4ae
commit 20ee4fc45e
3 changed files with 121 additions and 34 deletions
+68 -15
View File
@@ -510,28 +510,73 @@ except Exception as e:
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
// CogVideoX-2B — added 2026-07-17 as a second option alongside LTX-Video for quality
// comparison. Barely fits: peak VRAM measured at ~11GB of the 12GB card even with
// enable_model_cpu_offload(), so there's little headroom for pushing resolution/frames
// higher than the defaults below (unlike LTX-Video, which has real slack). Noticeably
// slower than LTX-Video too. fp16 per the model card's own recommendation for the 2B
// checkpoint (the 5B checkpoint recommends bf16, but 5B doesn't fit here at all — OOMs
// even offloaded, its active-component compute footprint alone exceeds 12GB).
const COGVIDEOX_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch
from diffusers import CogVideoXPipeline
from diffusers.utils import export_to_video
pipe = CogVideoXPipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16)
pipe.enable_model_cpu_offload()
pipe.vae.enable_tiling()
video = pipe(
prompt=${JSON.stringify(p.prompt)},
negative_prompt=${JSON.stringify(p.negative_prompt)},
width=int(${p.width}), height=int(${p.height}),
num_frames=int(${p.num_frames}),
num_inference_steps=int(${p.steps}),
guidance_scale=float(${p.guidance_scale}),
).frames[0]
dst = ${JSON.stringify(p.dst)}
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
export_to_video(video, dst, fps=int(${p.fps}))
stat = os.stat(dst)
print("###RESULT###" + json.dumps({
"output": dst, "width": int(${p.width}), "height": int(${p.height}),
"num_frames": int(${p.num_frames}), "fps": int(${p.fps}),
"size_bytes": stat.st_size,
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
}))
except Exception as e:
import traceback
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
export const videoGenerateTool = {
name: 'video_generate',
description: [
'Generate a short video clip from a text prompt using a local LTX-Video model (runs on-machine GPU, no external API).',
'Takes roughly 30–90 seconds depending on resolution/steps/frame count. Output is an h264 mp4.',
'Returns a download link — there is no inline video preview in chat yet.',
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
'Two models available via the "model" param: "ltx" (default — fast, ~30-90s, more headroom for higher resolution/frame count) and "cogvideox" (THUDM CogVideoX-2B — slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try both and compare if unsure which fits the request.',
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
].join('\n'),
schema: {
prompt: 'Text description of the video/scene to generate (English works best)',
model: '"ltx" (default, fast) or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
width: 'Video width in pixels, multiple of 32 (default 704)',
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
height: 'Video height in pixels, multiple of 32 (default 480)',
num_frames: 'Number of frames — duration = num_frames / fps (default 65, ~2.7s at 24fps)',
fps: 'Output frame rate (default 24)',
steps: 'Denoising steps — more = higher quality but slower (default 40, range 15–50)',
guidance_scale: 'How closely to follow the prompt (default 3.0, range 1–10 — LTX-Video responds best to a narrower range than SDXL)',
num_frames: 'Number of frames — duration = num_frames / fps (default 65 for ltx, 49 for cogvideox)',
fps: 'Output frame rate (default 24 for ltx, 8 for cogvideox)',
steps: 'Denoising steps — more = higher quality but slower (default 40 for ltx / 25 for cogvideox — cogvideox is much slower per step, higher values risk exceeding reverse-proxy timeouts, range 15–50)',
guidance_scale: 'How closely to follow the prompt (default 3.0 for ltx / 6.0 for cogvideox, range 1–10)',
output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)',
},
jsonSchema: {
type: 'object',
properties: {
prompt: { type: 'string' },
model: { type: 'string', enum: ['ltx', 'cogvideox'] },
negative_prompt: { type: 'string' },
width: { type: 'number' },
height: { type: 'number' },
@@ -548,10 +593,13 @@ export const videoGenerateTool = {
const prompt = String(args?.prompt || '').trim();
if (!prompt) return { success: false, error: 'prompt is required' };
const model = args?.model === 'cogvideox' ? 'cogvideox' : 'ltx';
const isCogVideoX = model === 'cogvideox';
const workspacePath = getWorkspacePath(args);
let outPath = String(args?.output || '').trim();
if (!outPath) {
outPath = path.join(workspacePath, `ltx_${Date.now()}.mp4`);
outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : 'ltx'}_${Date.now()}.mp4`);
} else if (!path.isAbsolute(outPath)) {
outPath = path.resolve(workspacePath, outPath);
}
@@ -565,16 +613,21 @@ export const videoGenerateTool = {
const params = {
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width: Math.round((args?.width ?? 704) / 32) * 32,
width: Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32,
height: Math.round((args?.height ?? 480) / 32) * 32,
num_frames: toFiniteNumber(args?.num_frames, 65),
fps: toFiniteNumber(args?.fps, 24),
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? 3.0)),
num_frames: toFiniteNumber(args?.num_frames, isCogVideoX ? 49 : 65),
fps: toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24),
// CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50
// (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any
// reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it
// lower to keep typical runs under ~4 minutes; still overridable via the steps param.
steps: Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40))),
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0))),
dst: outPath,
};
const result = await runVenvPython(LTX_SCRIPT(params), 600_000);
const script = isCogVideoX ? COGVIDEOX_SCRIPT(params) : LTX_SCRIPT(params);
const result = await runVenvPython(script, 600_000);
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
}