v4.1.18: 동영상 생성에 CogVideoX-2B 옵션 추가 + 스튜디오 UI 레이아웃 수정

video_generate에 model 파라미터 추가 — 기존 LTX-Video(빠름, 기본값)에
CogVideoX-2B(THUDM, 느리지만 화질 비교용)를 선택지로 추가. CogVideoX는
12GB VRAM 한계에 거의 딱 맞아(enable_model_cpu_offload에도 peak ~11GB)
해상도/프레임수 기본값 이상으로 올리면 여유가 없음.

스튜디오앱 동영상 탭에서 모델/프레임수/FPS 세 필드가 320px 폭 폼에
한 줄로 욱여넣어져 프레임수부터 우측 패널에 가려 안 보이던 레이아웃
버그 수정 — 모델 드롭다운을 단독 줄로 분리하고 프레임수/FPS는 그 아래
별도 행으로 이동.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-18 13:07:51 +09:00
co-authored by Claude Sonnet 5
parent 16c3d0e4ae
commit 20ee4fc45e
3 changed files with 121 additions and 34 deletions
+1 -1
View File
@@ -96,7 +96,7 @@ export function registerImagegenRoutes(app: Express): void {
const user = (req as any).user;
if (!user) return res.status(401).json({ error: 'Unauthorized' });
console.log(`[imagegen] generate-video start user=${user.username} prompt="${String(req.body?.prompt || '').slice(0, 120)}"`);
console.log(`[imagegen] generate-video start user=${user.username} model=${req.body?.model === 'cogvideox' ? 'cogvideox' : 'ltx'} prompt="${String(req.body?.prompt || '').slice(0, 120)}"`);
try {
const outDir = galleryDir(user.workspace);
fs.mkdirSync(outDir, { recursive: true });
+68 -15
View File
@@ -510,28 +510,73 @@ except Exception as e:
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
// CogVideoX-2B — added 2026-07-17 as a second option alongside LTX-Video for quality
// comparison. Barely fits: peak VRAM measured at ~11GB of the 12GB card even with
// enable_model_cpu_offload(), so there's little headroom for pushing resolution/frames
// higher than the defaults below (unlike LTX-Video, which has real slack). Noticeably
// slower than LTX-Video too. fp16 per the model card's own recommendation for the 2B
// checkpoint (the 5B checkpoint recommends bf16, but 5B doesn't fit here at all — OOMs
// even offloaded, its active-component compute footprint alone exceeds 12GB).
const COGVIDEOX_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch
from diffusers import CogVideoXPipeline
from diffusers.utils import export_to_video
pipe = CogVideoXPipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16)
pipe.enable_model_cpu_offload()
pipe.vae.enable_tiling()
video = pipe(
prompt=${JSON.stringify(p.prompt)},
negative_prompt=${JSON.stringify(p.negative_prompt)},
width=int(${p.width}), height=int(${p.height}),
num_frames=int(${p.num_frames}),
num_inference_steps=int(${p.steps}),
guidance_scale=float(${p.guidance_scale}),
).frames[0]
dst = ${JSON.stringify(p.dst)}
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
export_to_video(video, dst, fps=int(${p.fps}))
stat = os.stat(dst)
print("###RESULT###" + json.dumps({
"output": dst, "width": int(${p.width}), "height": int(${p.height}),
"num_frames": int(${p.num_frames}), "fps": int(${p.fps}),
"size_bytes": stat.st_size,
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
}))
except Exception as e:
import traceback
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
export const videoGenerateTool = {
name: 'video_generate',
description: [
'Generate a short video clip from a text prompt using a local LTX-Video model (runs on-machine GPU, no external API).',
'Takes roughly 30–90 seconds depending on resolution/steps/frame count. Output is an h264 mp4.',
'Returns a download link — there is no inline video preview in chat yet.',
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
'Two models available via the "model" param: "ltx" (default — fast, ~30-90s, more headroom for higher resolution/frame count) and "cogvideox" (THUDM CogVideoX-2B — slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try both and compare if unsure which fits the request.',
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
].join('\n'),
schema: {
prompt: 'Text description of the video/scene to generate (English works best)',
model: '"ltx" (default, fast) or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
width: 'Video width in pixels, multiple of 32 (default 704)',
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
height: 'Video height in pixels, multiple of 32 (default 480)',
num_frames: 'Number of frames — duration = num_frames / fps (default 65, ~2.7s at 24fps)',
fps: 'Output frame rate (default 24)',
steps: 'Denoising steps — more = higher quality but slower (default 40, range 15–50)',
guidance_scale: 'How closely to follow the prompt (default 3.0, range 1–10 — LTX-Video responds best to a narrower range than SDXL)',
num_frames: 'Number of frames — duration = num_frames / fps (default 65 for ltx, 49 for cogvideox)',
fps: 'Output frame rate (default 24 for ltx, 8 for cogvideox)',
steps: 'Denoising steps — more = higher quality but slower (default 40 for ltx / 25 for cogvideox — cogvideox is much slower per step, higher values risk exceeding reverse-proxy timeouts, range 15–50)',
guidance_scale: 'How closely to follow the prompt (default 3.0 for ltx / 6.0 for cogvideox, range 1–10)',
output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)',
},
jsonSchema: {
type: 'object',
properties: {
prompt: { type: 'string' },
model: { type: 'string', enum: ['ltx', 'cogvideox'] },
negative_prompt: { type: 'string' },
width: { type: 'number' },
height: { type: 'number' },
@@ -548,10 +593,13 @@ export const videoGenerateTool = {
const prompt = String(args?.prompt || '').trim();
if (!prompt) return { success: false, error: 'prompt is required' };
const model = args?.model === 'cogvideox' ? 'cogvideox' : 'ltx';
const isCogVideoX = model === 'cogvideox';
const workspacePath = getWorkspacePath(args);
let outPath = String(args?.output || '').trim();
if (!outPath) {
outPath = path.join(workspacePath, `ltx_${Date.now()}.mp4`);
outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : 'ltx'}_${Date.now()}.mp4`);
} else if (!path.isAbsolute(outPath)) {
outPath = path.resolve(workspacePath, outPath);
}
@@ -565,16 +613,21 @@ export const videoGenerateTool = {
const params = {
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width: Math.round((args?.width ?? 704) / 32) * 32,
width: Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32,
height: Math.round((args?.height ?? 480) / 32) * 32,
num_frames: toFiniteNumber(args?.num_frames, 65),
fps: toFiniteNumber(args?.fps, 24),
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? 3.0)),
num_frames: toFiniteNumber(args?.num_frames, isCogVideoX ? 49 : 65),
fps: toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24),
// CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50
// (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any
// reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it
// lower to keep typical runs under ~4 minutes; still overridable via the steps param.
steps: Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40))),
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0))),
dst: outPath,
};
const result = await runVenvPython(LTX_SCRIPT(params), 600_000);
const script = isCogVideoX ? COGVIDEOX_SCRIPT(params) : LTX_SCRIPT(params);
const result = await runVenvPython(script, 600_000);
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
}
+47 -13
View File
@@ -87,6 +87,7 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon
.st-result-wrap{position:relative;width:100%;height:100%;display:flex;align-items:center;justify-content:center;}
.st-result-dl{position:absolute;bottom:10px;right:10px;background:rgba(0,0,0,.65);color:#fff;padding:6px 12px;border-radius:20px;font-size:12px;text-decoration:none;backdrop-filter:blur(4px);}
.st-result-dl:hover{background:rgba(0,0,0,.85);}
.st-video-meta{position:absolute;top:10px;left:10px;max-width:calc(100% - 70px);background:rgba(0,0,0,.65);color:#fff;padding:6px 12px;border-radius:12px;font-size:12px;line-height:1.4;white-space:normal;backdrop-filter:blur(4px);z-index:5;}
.st-zoom-ctl{position:absolute;top:10px;right:10px;display:flex;align-items:center;gap:2px;background:rgba(0,0,0,.6);border-radius:20px;padding:3px;backdrop-filter:blur(4px);z-index:5;}
.st-zoom-ctl button{background:none;border:none;color:#fff;font-size:14px;font-family:inherit;cursor:pointer;width:24px;height:24px;border-radius:50%;display:flex;align-items:center;justify-content:center;}
@@ -94,10 +95,14 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon
.st-zoom-ctl span{color:#fff;font-size:10px;min-width:34px;text-align:center;user-select:none;}
.st-edit-ctl{position:absolute;top:10px;left:10px;display:flex;flex-wrap:wrap;align-items:center;gap:2px;background:rgba(0,0,0,.6);border-radius:16px;padding:3px;backdrop-filter:blur(4px);max-width:calc(100% - 20px);z-index:5;}
.st-edit-ctl button{background:none;border:none;color:#fff;font-size:15px;font-family:inherit;cursor:pointer;width:26px;height:26px;border-radius:50%;display:flex;align-items:center;justify-content:center;}
.st-edit-ctl button{position:relative;background:none;border:none;color:#fff;font-size:15px;font-family:inherit;cursor:pointer;width:26px;height:26px;border-radius:50%;display:flex;align-items:center;justify-content:center;}
.st-edit-ctl button:hover{background:rgba(255,255,255,.18);}
.st-edit-ctl button.active{background:var(--brand);color:#1e1b2e;}
.st-edit-ctl button:disabled{opacity:.35;cursor:wait;}
/* Native title-attribute tooltips render just below the cursor and get hidden behind it on
these small, closely-packed icon buttons — custom tooltip above the button instead. */
.st-edit-ctl button[data-tip]::after{content:attr(data-tip);position:absolute;bottom:calc(100% + 6px);left:50%;transform:translateX(-50%);background:rgba(0,0,0,.9);color:#fff;font-size:11px;line-height:1;padding:4px 8px;border-radius:5px;white-space:nowrap;opacity:0;pointer-events:none;transition:opacity .12s;z-index:20;}
.st-edit-ctl button[data-tip]:hover::after{opacity:1;}
.st-edit-panel{position:absolute;top:44px;left:10px;background:rgba(20,20,24,.92);border-radius:10px;padding:10px;backdrop-filter:blur(4px);display:none;flex-direction:column;gap:8px;width:220px;z-index:6;color:#fff;font-size:11px;}
.st-edit-panel.show{display:flex;}
@@ -188,7 +193,15 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon
<input type="number" id="f-seed" placeholder="랜덤">
</div>
</div>
<div class="st-row" id="row-video-only" style="display:none">
<div id="row-video-only" style="display:none;flex-direction:column;gap:10px;">
<div class="st-field">
<label>모델</label>
<select id="f-video-model" onchange="onVideoModelChange()">
<option value="ltx">LTX-Video (빠름 · 30~90초)</option>
<option value="cogvideox">CogVideoX-2B (느림 · 1~3분, VRAM 여유 적음)</option>
</select>
</div>
<div class="st-row">
<div class="st-field">
<label>프레임수</label>
<input type="number" id="f-frames" value="65" step="8">
@@ -198,6 +211,7 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon
<input type="number" id="f-fps" value="24">
</div>
</div>
</div>
<div id="row-edit-only" style="display:none;flex-direction:column;gap:10px;">
<div class="st-field">
<label>사진 업로드</label>
@@ -356,7 +370,7 @@ async function deleteSelected(selectedSet, deleteUrlFn, reloadFn, selectAllId){
reloadFn();
}
function resultMediaHtml(url, type, filename, isChainedEdit){
function resultMediaHtml(url, type, filename, isChainedEdit, meta){
resultZoom=1;
cancelCropUI();
currentResultPath = type==='video' ? null : url.replace(/^\/api\/files\//, '');
@@ -366,15 +380,15 @@ function resultMediaHtml(url, type, filename, isChainedEdit){
: '<img src="'+url+'" alt="result">';
const editToolbar = type==='video' ? '' :
'<div class="st-edit-ctl" onclick="event.stopPropagation()">'
+'<button type="button" id="edit-crop-btn" onclick="toggleCropMode()" title="자르기">✂️</button>'
+'<button type="button" onclick="quickRotate()" title="회전">🔄</button>'
+'<button type="button" onclick="quickFlip()" title="좌우반전">↔️</button>'
+'<button type="button" onclick="quickAutoEnhance()" title="자동보정">✨</button>'
+'<button type="button" onclick="quickRemoveBg()" title="배경투명화">🪄</button>'
+'<button type="button" onclick="togglePanel(\'filter-panel\')" title="필터">🎨</button>'
+'<button type="button" onclick="togglePanel(\'adjust-panel\')" title="보정">🎚️</button>'
+'<button type="button" onclick="togglePanel(\'text-panel\')" title="텍스트">📝</button>'
+'<button type="button" id="save-result-btn" onclick="saveResult()" title="저장">💾</button>'
+'<button type="button" id="edit-crop-btn" onclick="toggleCropMode()" data-tip="자르기">✂️</button>'
+'<button type="button" onclick="quickRotate()" data-tip="회전">🔄</button>'
+'<button type="button" onclick="quickFlip()" data-tip="좌우반전">↔️</button>'
+'<button type="button" onclick="quickAutoEnhance()" data-tip="자동보정">✨</button>'
+'<button type="button" onclick="quickRemoveBg()" data-tip="배경투명화">🪄</button>'
+'<button type="button" onclick="togglePanel(\'filter-panel\')" data-tip="필터">🎨</button>'
+'<button type="button" onclick="togglePanel(\'adjust-panel\')" data-tip="보정">🎚️</button>'
+'<button type="button" onclick="togglePanel(\'text-panel\')" data-tip="텍스트">📝</button>'
+'<button type="button" id="save-result-btn" onclick="saveResult()" data-tip="저장">💾</button>'
+'</div>'
+'<div class="st-edit-panel" id="filter-panel" onclick="event.stopPropagation()">'
+'<div class="st-filter-grid">'
@@ -421,6 +435,7 @@ function resultMediaHtml(url, type, filename, isChainedEdit){
+'<button type="button" onclick="resetResultZoom()" title="원래크기">⟲</button>'
+'</div>'
+editToolbar
+(meta && type==='video' ? '<div class="st-video-meta">'+meta.width+'×'+meta.height+' · '+meta.num_frames+'프레임 · '+meta.fps+'fps ('+(meta.num_frames/meta.fps).toFixed(1)+'초)</div>' : '')
+'<a class="st-result-dl" href="'+url+'" download="'+(filename||'')+'" title="다운로드" onclick="event.stopPropagation()">⬇ 다운로드</a></div>';
}
@@ -685,6 +700,24 @@ function onQualityChange(){
}
}
function onVideoModelChange(){
const m=document.getElementById('f-video-model').value;
const guidanceEl=document.getElementById('f-guidance');
if(m==='cogvideox'){
document.getElementById('f-frames').value=49;
document.getElementById('f-fps').value=8;
document.getElementById('f-steps').value=25;
guidanceEl.value=6.0;
document.getElementById('hint-text').textContent='CogVideoX-2B 로컬 생성 · 보통 3~5분, VRAM 여유가 적어서 해상도/프레임수/스텝수를 기본값보다 많이 올리면 실패하거나 너무 오래 걸릴 수 있어요';
} else {
document.getElementById('f-frames').value=65;
document.getElementById('f-fps').value=24;
document.getElementById('f-steps').value=40;
guidanceEl.value=3.0;
document.getElementById('hint-text').textContent='LTX-Video 로컬 생성 · 보통 30~90초, 시간이 걸립니다';
}
}
let genStartTime=0, genTimer=null;
function startTimer(){
genStartTime=Date.now();
@@ -880,6 +913,7 @@ async function generate(){
const seed=document.getElementById('f-seed').value;
if(seed) body.seed=parseInt(seed);
} else {
body.model=document.getElementById('f-video-model').value;
body.num_frames=parseInt(document.getElementById('f-frames').value)||undefined;
body.fps=parseInt(document.getElementById('f-fps').value)||undefined;
}
@@ -896,7 +930,7 @@ async function generate(){
if(currentTab==='image'){
resultArea.innerHTML=resultMediaHtml(d.url, 'image', 'generated_'+Date.now()+'.png');
} else {
resultArea.innerHTML=resultMediaHtml(d.url, 'video', 'generated_'+Date.now()+'.mp4');
resultArea.innerHTML=resultMediaHtml(d.url, 'video', 'generated_'+Date.now()+'.mp4', false, {width:d.width, height:d.height, num_frames:d.num_frames, fps:d.fps});
}
loadGallery();
}catch(e){