From 20ee4fc45ec4817f52a4aacb4e991b95a4e1c49d Mon Sep 17 00:00:00 2001 From: kim Date: Sat, 18 Jul 2026 13:07:51 +0900 Subject: [PATCH] =?UTF-8?q?v4.1.18:=20=EB=8F=99=EC=98=81=EC=83=81=20?= =?UTF-8?q?=EC=83=9D=EC=84=B1=EC=97=90=20CogVideoX-2B=20=EC=98=B5=EC=85=98?= =?UTF-8?q?=20=EC=B6=94=EA=B0=80=20+=20=EC=8A=A4=ED=8A=9C=EB=94=94?= =?UTF-8?q?=EC=98=A4=20UI=20=EB=A0=88=EC=9D=B4=EC=95=84=EC=9B=83=20?= =?UTF-8?q?=EC=88=98=EC=A0=95?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit video_generate에 model 파라미터 추가 — 기존 LTX-Video(빠름, 기본값)에 CogVideoX-2B(THUDM, 느리지만 화질 비교용)를 선택지로 추가. CogVideoX는 12GB VRAM 한계에 거의 딱 맞아(enable_model_cpu_offload에도 peak ~11GB) 해상도/프레임수 기본값 이상으로 올리면 여유가 없음. 스튜디오앱 동영상 탭에서 모델/프레임수/FPS 세 필드가 320px 폭 폼에 한 줄로 욱여넣어져 프레임수부터 우측 패널에 가려 안 보이던 레이아웃 버그 수정 — 모델 드롭다운을 단독 줄로 분리하고 프레임수/FPS는 그 아래 별도 행으로 이동. Co-Authored-By: Claude Sonnet 5 --- src/gateway/routes/imagegen.ts | 2 +- src/tools/imagegen.ts | 83 ++++++++++++++++++++++++++++------ web-ui/studio-app.html | 70 ++++++++++++++++++++-------- 3 files changed, 121 insertions(+), 34 deletions(-) diff --git a/src/gateway/routes/imagegen.ts b/src/gateway/routes/imagegen.ts index 39bf215..9b9bbb0 100644 --- a/src/gateway/routes/imagegen.ts +++ b/src/gateway/routes/imagegen.ts @@ -96,7 +96,7 @@ export function registerImagegenRoutes(app: Express): void { const user = (req as any).user; if (!user) return res.status(401).json({ error: 'Unauthorized' }); - console.log(`[imagegen] generate-video start user=${user.username} prompt="${String(req.body?.prompt || '').slice(0, 120)}"`); + console.log(`[imagegen] generate-video start user=${user.username} model=${req.body?.model === 'cogvideox' ? 'cogvideox' : 'ltx'} prompt="${String(req.body?.prompt || '').slice(0, 120)}"`); try { const outDir = galleryDir(user.workspace); fs.mkdirSync(outDir, { recursive: true }); diff --git a/src/tools/imagegen.ts b/src/tools/imagegen.ts index 7454abe..3c55ba7 100644 --- a/src/tools/imagegen.ts +++ b/src/tools/imagegen.ts @@ -510,28 +510,73 @@ except Exception as e: print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]})) `; +// CogVideoX-2B — added 2026-07-17 as a second option alongside LTX-Video for quality +// comparison. Barely fits: peak VRAM measured at ~11GB of the 12GB card even with +// enable_model_cpu_offload(), so there's little headroom for pushing resolution/frames +// higher than the defaults below (unlike LTX-Video, which has real slack). Noticeably +// slower than LTX-Video too. fp16 per the model card's own recommendation for the 2B +// checkpoint (the 5B checkpoint recommends bf16, but 5B doesn't fit here at all — OOMs +// even offloaded, its active-component compute footprint alone exceeds 12GB). +const COGVIDEOX_SCRIPT = (p: Record) => ` +import os, json, sys +try: + import torch + from diffusers import CogVideoXPipeline + from diffusers.utils import export_to_video + + pipe = CogVideoXPipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16) + pipe.enable_model_cpu_offload() + pipe.vae.enable_tiling() + + video = pipe( + prompt=${JSON.stringify(p.prompt)}, + negative_prompt=${JSON.stringify(p.negative_prompt)}, + width=int(${p.width}), height=int(${p.height}), + num_frames=int(${p.num_frames}), + num_inference_steps=int(${p.steps}), + guidance_scale=float(${p.guidance_scale}), + ).frames[0] + + dst = ${JSON.stringify(p.dst)} + os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True) + export_to_video(video, dst, fps=int(${p.fps})) + + stat = os.stat(dst) + print("###RESULT###" + json.dumps({ + "output": dst, "width": int(${p.width}), "height": int(${p.height}), + "num_frames": int(${p.num_frames}), "fps": int(${p.fps}), + "size_bytes": stat.st_size, + "vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2, + })) +except Exception as e: + import traceback + print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]})) +`; + export const videoGenerateTool = { name: 'video_generate', description: [ - 'Generate a short video clip from a text prompt using a local LTX-Video model (runs on-machine GPU, no external API).', - 'Takes roughly 30–90 seconds depending on resolution/steps/frame count. Output is an h264 mp4.', - 'Returns a download link — there is no inline video preview in chat yet.', + 'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).', + 'Two models available via the "model" param: "ltx" (default — fast, ~30-90s, more headroom for higher resolution/frame count) and "cogvideox" (THUDM CogVideoX-2B — slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try both and compare if unsure which fits the request.', + 'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.', ].join('\n'), schema: { prompt: 'Text description of the video/scene to generate (English works best)', + model: '"ltx" (default, fast) or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)', negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")', - width: 'Video width in pixels, multiple of 32 (default 704)', + width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)', height: 'Video height in pixels, multiple of 32 (default 480)', - num_frames: 'Number of frames — duration = num_frames / fps (default 65, ~2.7s at 24fps)', - fps: 'Output frame rate (default 24)', - steps: 'Denoising steps — more = higher quality but slower (default 40, range 15–50)', - guidance_scale: 'How closely to follow the prompt (default 3.0, range 1–10 — LTX-Video responds best to a narrower range than SDXL)', + num_frames: 'Number of frames — duration = num_frames / fps (default 65 for ltx, 49 for cogvideox)', + fps: 'Output frame rate (default 24 for ltx, 8 for cogvideox)', + steps: 'Denoising steps — more = higher quality but slower (default 40 for ltx / 25 for cogvideox — cogvideox is much slower per step, higher values risk exceeding reverse-proxy timeouts, range 15–50)', + guidance_scale: 'How closely to follow the prompt (default 3.0 for ltx / 6.0 for cogvideox, range 1–10)', output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)', }, jsonSchema: { type: 'object', properties: { prompt: { type: 'string' }, + model: { type: 'string', enum: ['ltx', 'cogvideox'] }, negative_prompt: { type: 'string' }, width: { type: 'number' }, height: { type: 'number' }, @@ -548,10 +593,13 @@ export const videoGenerateTool = { const prompt = String(args?.prompt || '').trim(); if (!prompt) return { success: false, error: 'prompt is required' }; + const model = args?.model === 'cogvideox' ? 'cogvideox' : 'ltx'; + const isCogVideoX = model === 'cogvideox'; + const workspacePath = getWorkspacePath(args); let outPath = String(args?.output || '').trim(); if (!outPath) { - outPath = path.join(workspacePath, `ltx_${Date.now()}.mp4`); + outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : 'ltx'}_${Date.now()}.mp4`); } else if (!path.isAbsolute(outPath)) { outPath = path.resolve(workspacePath, outPath); } @@ -565,16 +613,21 @@ export const videoGenerateTool = { const params = { prompt: translatedPrompt || prompt, negative_prompt: translatedNegative || negativePromptRaw, - width: Math.round((args?.width ?? 704) / 32) * 32, + width: Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32, height: Math.round((args?.height ?? 480) / 32) * 32, - num_frames: toFiniteNumber(args?.num_frames, 65), - fps: toFiniteNumber(args?.fps, 24), - steps: Math.min(50, Math.max(15, args?.steps ?? 40)), - guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? 3.0)), + num_frames: toFiniteNumber(args?.num_frames, isCogVideoX ? 49 : 65), + fps: toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24), + // CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50 + // (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any + // reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it + // lower to keep typical runs under ~4 minutes; still overridable via the steps param. + steps: Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40))), + guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0))), dst: outPath, }; - const result = await runVenvPython(LTX_SCRIPT(params), 600_000); + const script = isCogVideoX ? COGVIDEOX_SCRIPT(params) : LTX_SCRIPT(params); + const result = await runVenvPython(script, 600_000); if (result.error || !result.output) { return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw }; } diff --git a/web-ui/studio-app.html b/web-ui/studio-app.html index 6298277..bdd5a08 100644 --- a/web-ui/studio-app.html +++ b/web-ui/studio-app.html @@ -87,6 +87,7 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon .st-result-wrap{position:relative;width:100%;height:100%;display:flex;align-items:center;justify-content:center;} .st-result-dl{position:absolute;bottom:10px;right:10px;background:rgba(0,0,0,.65);color:#fff;padding:6px 12px;border-radius:20px;font-size:12px;text-decoration:none;backdrop-filter:blur(4px);} .st-result-dl:hover{background:rgba(0,0,0,.85);} +.st-video-meta{position:absolute;top:10px;left:10px;max-width:calc(100% - 70px);background:rgba(0,0,0,.65);color:#fff;padding:6px 12px;border-radius:12px;font-size:12px;line-height:1.4;white-space:normal;backdrop-filter:blur(4px);z-index:5;} .st-zoom-ctl{position:absolute;top:10px;right:10px;display:flex;align-items:center;gap:2px;background:rgba(0,0,0,.6);border-radius:20px;padding:3px;backdrop-filter:blur(4px);z-index:5;} .st-zoom-ctl button{background:none;border:none;color:#fff;font-size:14px;font-family:inherit;cursor:pointer;width:24px;height:24px;border-radius:50%;display:flex;align-items:center;justify-content:center;} @@ -94,10 +95,14 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon .st-zoom-ctl span{color:#fff;font-size:10px;min-width:34px;text-align:center;user-select:none;} .st-edit-ctl{position:absolute;top:10px;left:10px;display:flex;flex-wrap:wrap;align-items:center;gap:2px;background:rgba(0,0,0,.6);border-radius:16px;padding:3px;backdrop-filter:blur(4px);max-width:calc(100% - 20px);z-index:5;} -.st-edit-ctl button{background:none;border:none;color:#fff;font-size:15px;font-family:inherit;cursor:pointer;width:26px;height:26px;border-radius:50%;display:flex;align-items:center;justify-content:center;} +.st-edit-ctl button{position:relative;background:none;border:none;color:#fff;font-size:15px;font-family:inherit;cursor:pointer;width:26px;height:26px;border-radius:50%;display:flex;align-items:center;justify-content:center;} .st-edit-ctl button:hover{background:rgba(255,255,255,.18);} .st-edit-ctl button.active{background:var(--brand);color:#1e1b2e;} .st-edit-ctl button:disabled{opacity:.35;cursor:wait;} +/* Native title-attribute tooltips render just below the cursor and get hidden behind it on + these small, closely-packed icon buttons — custom tooltip above the button instead. */ +.st-edit-ctl button[data-tip]::after{content:attr(data-tip);position:absolute;bottom:calc(100% + 6px);left:50%;transform:translateX(-50%);background:rgba(0,0,0,.9);color:#fff;font-size:11px;line-height:1;padding:4px 8px;border-radius:5px;white-space:nowrap;opacity:0;pointer-events:none;transition:opacity .12s;z-index:20;} +.st-edit-ctl button[data-tip]:hover::after{opacity:1;} .st-edit-panel{position:absolute;top:44px;left:10px;background:rgba(20,20,24,.92);border-radius:10px;padding:10px;backdrop-filter:blur(4px);display:none;flex-direction:column;gap:8px;width:220px;z-index:6;color:#fff;font-size:11px;} .st-edit-panel.show{display:flex;} @@ -188,14 +193,23 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon -