v4.1.18: 동영상 생성에 CogVideoX-2B 옵션 추가 + 스튜디오 UI 레이아웃 수정
video_generate에 model 파라미터 추가 — 기존 LTX-Video(빠름, 기본값)에 CogVideoX-2B(THUDM, 느리지만 화질 비교용)를 선택지로 추가. CogVideoX는 12GB VRAM 한계에 거의 딱 맞아(enable_model_cpu_offload에도 peak ~11GB) 해상도/프레임수 기본값 이상으로 올리면 여유가 없음. 스튜디오앱 동영상 탭에서 모델/프레임수/FPS 세 필드가 320px 폭 폼에 한 줄로 욱여넣어져 프레임수부터 우측 패널에 가려 안 보이던 레이아웃 버그 수정 — 모델 드롭다운을 단독 줄로 분리하고 프레임수/FPS는 그 아래 별도 행으로 이동. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
+68
-15
@@ -510,28 +510,73 @@ except Exception as e:
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
|
||||
// CogVideoX-2B — added 2026-07-17 as a second option alongside LTX-Video for quality
|
||||
// comparison. Barely fits: peak VRAM measured at ~11GB of the 12GB card even with
|
||||
// enable_model_cpu_offload(), so there's little headroom for pushing resolution/frames
|
||||
// higher than the defaults below (unlike LTX-Video, which has real slack). Noticeably
|
||||
// slower than LTX-Video too. fp16 per the model card's own recommendation for the 2B
|
||||
// checkpoint (the 5B checkpoint recommends bf16, but 5B doesn't fit here at all — OOMs
|
||||
// even offloaded, its active-component compute footprint alone exceeds 12GB).
|
||||
const COGVIDEOX_SCRIPT = (p: Record<string, any>) => `
|
||||
import os, json, sys
|
||||
try:
|
||||
import torch
|
||||
from diffusers import CogVideoXPipeline
|
||||
from diffusers.utils import export_to_video
|
||||
|
||||
pipe = CogVideoXPipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16)
|
||||
pipe.enable_model_cpu_offload()
|
||||
pipe.vae.enable_tiling()
|
||||
|
||||
video = pipe(
|
||||
prompt=${JSON.stringify(p.prompt)},
|
||||
negative_prompt=${JSON.stringify(p.negative_prompt)},
|
||||
width=int(${p.width}), height=int(${p.height}),
|
||||
num_frames=int(${p.num_frames}),
|
||||
num_inference_steps=int(${p.steps}),
|
||||
guidance_scale=float(${p.guidance_scale}),
|
||||
).frames[0]
|
||||
|
||||
dst = ${JSON.stringify(p.dst)}
|
||||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||||
export_to_video(video, dst, fps=int(${p.fps}))
|
||||
|
||||
stat = os.stat(dst)
|
||||
print("###RESULT###" + json.dumps({
|
||||
"output": dst, "width": int(${p.width}), "height": int(${p.height}),
|
||||
"num_frames": int(${p.num_frames}), "fps": int(${p.fps}),
|
||||
"size_bytes": stat.st_size,
|
||||
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
|
||||
}))
|
||||
except Exception as e:
|
||||
import traceback
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
|
||||
export const videoGenerateTool = {
|
||||
name: 'video_generate',
|
||||
description: [
|
||||
'Generate a short video clip from a text prompt using a local LTX-Video model (runs on-machine GPU, no external API).',
|
||||
'Takes roughly 30–90 seconds depending on resolution/steps/frame count. Output is an h264 mp4.',
|
||||
'Returns a download link — there is no inline video preview in chat yet.',
|
||||
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
|
||||
'Two models available via the "model" param: "ltx" (default — fast, ~30-90s, more headroom for higher resolution/frame count) and "cogvideox" (THUDM CogVideoX-2B — slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try both and compare if unsure which fits the request.',
|
||||
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
|
||||
].join('\n'),
|
||||
schema: {
|
||||
prompt: 'Text description of the video/scene to generate (English works best)',
|
||||
model: '"ltx" (default, fast) or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
|
||||
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
|
||||
width: 'Video width in pixels, multiple of 32 (default 704)',
|
||||
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
|
||||
height: 'Video height in pixels, multiple of 32 (default 480)',
|
||||
num_frames: 'Number of frames — duration = num_frames / fps (default 65, ~2.7s at 24fps)',
|
||||
fps: 'Output frame rate (default 24)',
|
||||
steps: 'Denoising steps — more = higher quality but slower (default 40, range 15–50)',
|
||||
guidance_scale: 'How closely to follow the prompt (default 3.0, range 1–10 — LTX-Video responds best to a narrower range than SDXL)',
|
||||
num_frames: 'Number of frames — duration = num_frames / fps (default 65 for ltx, 49 for cogvideox)',
|
||||
fps: 'Output frame rate (default 24 for ltx, 8 for cogvideox)',
|
||||
steps: 'Denoising steps — more = higher quality but slower (default 40 for ltx / 25 for cogvideox — cogvideox is much slower per step, higher values risk exceeding reverse-proxy timeouts, range 15–50)',
|
||||
guidance_scale: 'How closely to follow the prompt (default 3.0 for ltx / 6.0 for cogvideox, range 1–10)',
|
||||
output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
prompt: { type: 'string' },
|
||||
model: { type: 'string', enum: ['ltx', 'cogvideox'] },
|
||||
negative_prompt: { type: 'string' },
|
||||
width: { type: 'number' },
|
||||
height: { type: 'number' },
|
||||
@@ -548,10 +593,13 @@ export const videoGenerateTool = {
|
||||
const prompt = String(args?.prompt || '').trim();
|
||||
if (!prompt) return { success: false, error: 'prompt is required' };
|
||||
|
||||
const model = args?.model === 'cogvideox' ? 'cogvideox' : 'ltx';
|
||||
const isCogVideoX = model === 'cogvideox';
|
||||
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
let outPath = String(args?.output || '').trim();
|
||||
if (!outPath) {
|
||||
outPath = path.join(workspacePath, `ltx_${Date.now()}.mp4`);
|
||||
outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : 'ltx'}_${Date.now()}.mp4`);
|
||||
} else if (!path.isAbsolute(outPath)) {
|
||||
outPath = path.resolve(workspacePath, outPath);
|
||||
}
|
||||
@@ -565,16 +613,21 @@ export const videoGenerateTool = {
|
||||
const params = {
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width: Math.round((args?.width ?? 704) / 32) * 32,
|
||||
width: Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32,
|
||||
height: Math.round((args?.height ?? 480) / 32) * 32,
|
||||
num_frames: toFiniteNumber(args?.num_frames, 65),
|
||||
fps: toFiniteNumber(args?.fps, 24),
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
|
||||
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? 3.0)),
|
||||
num_frames: toFiniteNumber(args?.num_frames, isCogVideoX ? 49 : 65),
|
||||
fps: toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24),
|
||||
// CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50
|
||||
// (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any
|
||||
// reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it
|
||||
// lower to keep typical runs under ~4 minutes; still overridable via the steps param.
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40))),
|
||||
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0))),
|
||||
dst: outPath,
|
||||
};
|
||||
|
||||
const result = await runVenvPython(LTX_SCRIPT(params), 600_000);
|
||||
const script = isCogVideoX ? COGVIDEOX_SCRIPT(params) : LTX_SCRIPT(params);
|
||||
const result = await runVenvPython(script, 600_000);
|
||||
if (result.error || !result.output) {
|
||||
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user