v4.1.18: 동영상 생성에 CogVideoX-2B 옵션 추가 + 스튜디오 UI 레이아웃 수정
video_generate에 model 파라미터 추가 — 기존 LTX-Video(빠름, 기본값)에 CogVideoX-2B(THUDM, 느리지만 화질 비교용)를 선택지로 추가. CogVideoX는 12GB VRAM 한계에 거의 딱 맞아(enable_model_cpu_offload에도 peak ~11GB) 해상도/프레임수 기본값 이상으로 올리면 여유가 없음. 스튜디오앱 동영상 탭에서 모델/프레임수/FPS 세 필드가 320px 폭 폼에 한 줄로 욱여넣어져 프레임수부터 우측 패널에 가려 안 보이던 레이아웃 버그 수정 — 모델 드롭다운을 단독 줄로 분리하고 프레임수/FPS는 그 아래 별도 행으로 이동. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -96,7 +96,7 @@ export function registerImagegenRoutes(app: Express): void {
|
||||
const user = (req as any).user;
|
||||
if (!user) return res.status(401).json({ error: 'Unauthorized' });
|
||||
|
||||
console.log(`[imagegen] generate-video start user=${user.username} prompt="${String(req.body?.prompt || '').slice(0, 120)}"`);
|
||||
console.log(`[imagegen] generate-video start user=${user.username} model=${req.body?.model === 'cogvideox' ? 'cogvideox' : 'ltx'} prompt="${String(req.body?.prompt || '').slice(0, 120)}"`);
|
||||
try {
|
||||
const outDir = galleryDir(user.workspace);
|
||||
fs.mkdirSync(outDir, { recursive: true });
|
||||
|
||||
+68
-15
@@ -510,28 +510,73 @@ except Exception as e:
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
|
||||
// CogVideoX-2B — added 2026-07-17 as a second option alongside LTX-Video for quality
|
||||
// comparison. Barely fits: peak VRAM measured at ~11GB of the 12GB card even with
|
||||
// enable_model_cpu_offload(), so there's little headroom for pushing resolution/frames
|
||||
// higher than the defaults below (unlike LTX-Video, which has real slack). Noticeably
|
||||
// slower than LTX-Video too. fp16 per the model card's own recommendation for the 2B
|
||||
// checkpoint (the 5B checkpoint recommends bf16, but 5B doesn't fit here at all — OOMs
|
||||
// even offloaded, its active-component compute footprint alone exceeds 12GB).
|
||||
const COGVIDEOX_SCRIPT = (p: Record<string, any>) => `
|
||||
import os, json, sys
|
||||
try:
|
||||
import torch
|
||||
from diffusers import CogVideoXPipeline
|
||||
from diffusers.utils import export_to_video
|
||||
|
||||
pipe = CogVideoXPipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16)
|
||||
pipe.enable_model_cpu_offload()
|
||||
pipe.vae.enable_tiling()
|
||||
|
||||
video = pipe(
|
||||
prompt=${JSON.stringify(p.prompt)},
|
||||
negative_prompt=${JSON.stringify(p.negative_prompt)},
|
||||
width=int(${p.width}), height=int(${p.height}),
|
||||
num_frames=int(${p.num_frames}),
|
||||
num_inference_steps=int(${p.steps}),
|
||||
guidance_scale=float(${p.guidance_scale}),
|
||||
).frames[0]
|
||||
|
||||
dst = ${JSON.stringify(p.dst)}
|
||||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||||
export_to_video(video, dst, fps=int(${p.fps}))
|
||||
|
||||
stat = os.stat(dst)
|
||||
print("###RESULT###" + json.dumps({
|
||||
"output": dst, "width": int(${p.width}), "height": int(${p.height}),
|
||||
"num_frames": int(${p.num_frames}), "fps": int(${p.fps}),
|
||||
"size_bytes": stat.st_size,
|
||||
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
|
||||
}))
|
||||
except Exception as e:
|
||||
import traceback
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
|
||||
export const videoGenerateTool = {
|
||||
name: 'video_generate',
|
||||
description: [
|
||||
'Generate a short video clip from a text prompt using a local LTX-Video model (runs on-machine GPU, no external API).',
|
||||
'Takes roughly 30–90 seconds depending on resolution/steps/frame count. Output is an h264 mp4.',
|
||||
'Returns a download link — there is no inline video preview in chat yet.',
|
||||
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
|
||||
'Two models available via the "model" param: "ltx" (default — fast, ~30-90s, more headroom for higher resolution/frame count) and "cogvideox" (THUDM CogVideoX-2B — slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try both and compare if unsure which fits the request.',
|
||||
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
|
||||
].join('\n'),
|
||||
schema: {
|
||||
prompt: 'Text description of the video/scene to generate (English works best)',
|
||||
model: '"ltx" (default, fast) or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
|
||||
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
|
||||
width: 'Video width in pixels, multiple of 32 (default 704)',
|
||||
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
|
||||
height: 'Video height in pixels, multiple of 32 (default 480)',
|
||||
num_frames: 'Number of frames — duration = num_frames / fps (default 65, ~2.7s at 24fps)',
|
||||
fps: 'Output frame rate (default 24)',
|
||||
steps: 'Denoising steps — more = higher quality but slower (default 40, range 15–50)',
|
||||
guidance_scale: 'How closely to follow the prompt (default 3.0, range 1–10 — LTX-Video responds best to a narrower range than SDXL)',
|
||||
num_frames: 'Number of frames — duration = num_frames / fps (default 65 for ltx, 49 for cogvideox)',
|
||||
fps: 'Output frame rate (default 24 for ltx, 8 for cogvideox)',
|
||||
steps: 'Denoising steps — more = higher quality but slower (default 40 for ltx / 25 for cogvideox — cogvideox is much slower per step, higher values risk exceeding reverse-proxy timeouts, range 15–50)',
|
||||
guidance_scale: 'How closely to follow the prompt (default 3.0 for ltx / 6.0 for cogvideox, range 1–10)',
|
||||
output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
prompt: { type: 'string' },
|
||||
model: { type: 'string', enum: ['ltx', 'cogvideox'] },
|
||||
negative_prompt: { type: 'string' },
|
||||
width: { type: 'number' },
|
||||
height: { type: 'number' },
|
||||
@@ -548,10 +593,13 @@ export const videoGenerateTool = {
|
||||
const prompt = String(args?.prompt || '').trim();
|
||||
if (!prompt) return { success: false, error: 'prompt is required' };
|
||||
|
||||
const model = args?.model === 'cogvideox' ? 'cogvideox' : 'ltx';
|
||||
const isCogVideoX = model === 'cogvideox';
|
||||
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
let outPath = String(args?.output || '').trim();
|
||||
if (!outPath) {
|
||||
outPath = path.join(workspacePath, `ltx_${Date.now()}.mp4`);
|
||||
outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : 'ltx'}_${Date.now()}.mp4`);
|
||||
} else if (!path.isAbsolute(outPath)) {
|
||||
outPath = path.resolve(workspacePath, outPath);
|
||||
}
|
||||
@@ -565,16 +613,21 @@ export const videoGenerateTool = {
|
||||
const params = {
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width: Math.round((args?.width ?? 704) / 32) * 32,
|
||||
width: Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32,
|
||||
height: Math.round((args?.height ?? 480) / 32) * 32,
|
||||
num_frames: toFiniteNumber(args?.num_frames, 65),
|
||||
fps: toFiniteNumber(args?.fps, 24),
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
|
||||
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? 3.0)),
|
||||
num_frames: toFiniteNumber(args?.num_frames, isCogVideoX ? 49 : 65),
|
||||
fps: toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24),
|
||||
// CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50
|
||||
// (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any
|
||||
// reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it
|
||||
// lower to keep typical runs under ~4 minutes; still overridable via the steps param.
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40))),
|
||||
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0))),
|
||||
dst: outPath,
|
||||
};
|
||||
|
||||
const result = await runVenvPython(LTX_SCRIPT(params), 600_000);
|
||||
const script = isCogVideoX ? COGVIDEOX_SCRIPT(params) : LTX_SCRIPT(params);
|
||||
const result = await runVenvPython(script, 600_000);
|
||||
if (result.error || !result.output) {
|
||||
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
|
||||
}
|
||||
|
||||
+47
-13
@@ -87,6 +87,7 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon
|
||||
.st-result-wrap{position:relative;width:100%;height:100%;display:flex;align-items:center;justify-content:center;}
|
||||
.st-result-dl{position:absolute;bottom:10px;right:10px;background:rgba(0,0,0,.65);color:#fff;padding:6px 12px;border-radius:20px;font-size:12px;text-decoration:none;backdrop-filter:blur(4px);}
|
||||
.st-result-dl:hover{background:rgba(0,0,0,.85);}
|
||||
.st-video-meta{position:absolute;top:10px;left:10px;max-width:calc(100% - 70px);background:rgba(0,0,0,.65);color:#fff;padding:6px 12px;border-radius:12px;font-size:12px;line-height:1.4;white-space:normal;backdrop-filter:blur(4px);z-index:5;}
|
||||
|
||||
.st-zoom-ctl{position:absolute;top:10px;right:10px;display:flex;align-items:center;gap:2px;background:rgba(0,0,0,.6);border-radius:20px;padding:3px;backdrop-filter:blur(4px);z-index:5;}
|
||||
.st-zoom-ctl button{background:none;border:none;color:#fff;font-size:14px;font-family:inherit;cursor:pointer;width:24px;height:24px;border-radius:50%;display:flex;align-items:center;justify-content:center;}
|
||||
@@ -94,10 +95,14 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon
|
||||
.st-zoom-ctl span{color:#fff;font-size:10px;min-width:34px;text-align:center;user-select:none;}
|
||||
|
||||
.st-edit-ctl{position:absolute;top:10px;left:10px;display:flex;flex-wrap:wrap;align-items:center;gap:2px;background:rgba(0,0,0,.6);border-radius:16px;padding:3px;backdrop-filter:blur(4px);max-width:calc(100% - 20px);z-index:5;}
|
||||
.st-edit-ctl button{background:none;border:none;color:#fff;font-size:15px;font-family:inherit;cursor:pointer;width:26px;height:26px;border-radius:50%;display:flex;align-items:center;justify-content:center;}
|
||||
.st-edit-ctl button{position:relative;background:none;border:none;color:#fff;font-size:15px;font-family:inherit;cursor:pointer;width:26px;height:26px;border-radius:50%;display:flex;align-items:center;justify-content:center;}
|
||||
.st-edit-ctl button:hover{background:rgba(255,255,255,.18);}
|
||||
.st-edit-ctl button.active{background:var(--brand);color:#1e1b2e;}
|
||||
.st-edit-ctl button:disabled{opacity:.35;cursor:wait;}
|
||||
/* Native title-attribute tooltips render just below the cursor and get hidden behind it on
|
||||
these small, closely-packed icon buttons — custom tooltip above the button instead. */
|
||||
.st-edit-ctl button[data-tip]::after{content:attr(data-tip);position:absolute;bottom:calc(100% + 6px);left:50%;transform:translateX(-50%);background:rgba(0,0,0,.9);color:#fff;font-size:11px;line-height:1;padding:4px 8px;border-radius:5px;white-space:nowrap;opacity:0;pointer-events:none;transition:opacity .12s;z-index:20;}
|
||||
.st-edit-ctl button[data-tip]:hover::after{opacity:1;}
|
||||
|
||||
.st-edit-panel{position:absolute;top:44px;left:10px;background:rgba(20,20,24,.92);border-radius:10px;padding:10px;backdrop-filter:blur(4px);display:none;flex-direction:column;gap:8px;width:220px;z-index:6;color:#fff;font-size:11px;}
|
||||
.st-edit-panel.show{display:flex;}
|
||||
@@ -188,7 +193,15 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon
|
||||
<input type="number" id="f-seed" placeholder="랜덤">
|
||||
</div>
|
||||
</div>
|
||||
<div class="st-row" id="row-video-only" style="display:none">
|
||||
<div id="row-video-only" style="display:none;flex-direction:column;gap:10px;">
|
||||
<div class="st-field">
|
||||
<label>모델</label>
|
||||
<select id="f-video-model" onchange="onVideoModelChange()">
|
||||
<option value="ltx">LTX-Video (빠름 · 30~90초)</option>
|
||||
<option value="cogvideox">CogVideoX-2B (느림 · 1~3분, VRAM 여유 적음)</option>
|
||||
</select>
|
||||
</div>
|
||||
<div class="st-row">
|
||||
<div class="st-field">
|
||||
<label>프레임수</label>
|
||||
<input type="number" id="f-frames" value="65" step="8">
|
||||
@@ -198,6 +211,7 @@ body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;fon
|
||||
<input type="number" id="f-fps" value="24">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div id="row-edit-only" style="display:none;flex-direction:column;gap:10px;">
|
||||
<div class="st-field">
|
||||
<label>사진 업로드</label>
|
||||
@@ -356,7 +370,7 @@ async function deleteSelected(selectedSet, deleteUrlFn, reloadFn, selectAllId){
|
||||
reloadFn();
|
||||
}
|
||||
|
||||
function resultMediaHtml(url, type, filename, isChainedEdit){
|
||||
function resultMediaHtml(url, type, filename, isChainedEdit, meta){
|
||||
resultZoom=1;
|
||||
cancelCropUI();
|
||||
currentResultPath = type==='video' ? null : url.replace(/^\/api\/files\//, '');
|
||||
@@ -366,15 +380,15 @@ function resultMediaHtml(url, type, filename, isChainedEdit){
|
||||
: '<img src="'+url+'" alt="result">';
|
||||
const editToolbar = type==='video' ? '' :
|
||||
'<div class="st-edit-ctl" onclick="event.stopPropagation()">'
|
||||
+'<button type="button" id="edit-crop-btn" onclick="toggleCropMode()" title="자르기">✂️</button>'
|
||||
+'<button type="button" onclick="quickRotate()" title="회전">🔄</button>'
|
||||
+'<button type="button" onclick="quickFlip()" title="좌우반전">↔️</button>'
|
||||
+'<button type="button" onclick="quickAutoEnhance()" title="자동보정">✨</button>'
|
||||
+'<button type="button" onclick="quickRemoveBg()" title="배경투명화">🪄</button>'
|
||||
+'<button type="button" onclick="togglePanel(\'filter-panel\')" title="필터">🎨</button>'
|
||||
+'<button type="button" onclick="togglePanel(\'adjust-panel\')" title="보정">🎚️</button>'
|
||||
+'<button type="button" onclick="togglePanel(\'text-panel\')" title="텍스트">📝</button>'
|
||||
+'<button type="button" id="save-result-btn" onclick="saveResult()" title="저장">💾</button>'
|
||||
+'<button type="button" id="edit-crop-btn" onclick="toggleCropMode()" data-tip="자르기">✂️</button>'
|
||||
+'<button type="button" onclick="quickRotate()" data-tip="회전">🔄</button>'
|
||||
+'<button type="button" onclick="quickFlip()" data-tip="좌우반전">↔️</button>'
|
||||
+'<button type="button" onclick="quickAutoEnhance()" data-tip="자동보정">✨</button>'
|
||||
+'<button type="button" onclick="quickRemoveBg()" data-tip="배경투명화">🪄</button>'
|
||||
+'<button type="button" onclick="togglePanel(\'filter-panel\')" data-tip="필터">🎨</button>'
|
||||
+'<button type="button" onclick="togglePanel(\'adjust-panel\')" data-tip="보정">🎚️</button>'
|
||||
+'<button type="button" onclick="togglePanel(\'text-panel\')" data-tip="텍스트">📝</button>'
|
||||
+'<button type="button" id="save-result-btn" onclick="saveResult()" data-tip="저장">💾</button>'
|
||||
+'</div>'
|
||||
+'<div class="st-edit-panel" id="filter-panel" onclick="event.stopPropagation()">'
|
||||
+'<div class="st-filter-grid">'
|
||||
@@ -421,6 +435,7 @@ function resultMediaHtml(url, type, filename, isChainedEdit){
|
||||
+'<button type="button" onclick="resetResultZoom()" title="원래크기">⟲</button>'
|
||||
+'</div>'
|
||||
+editToolbar
|
||||
+(meta && type==='video' ? '<div class="st-video-meta">'+meta.width+'×'+meta.height+' · '+meta.num_frames+'프레임 · '+meta.fps+'fps ('+(meta.num_frames/meta.fps).toFixed(1)+'초)</div>' : '')
|
||||
+'<a class="st-result-dl" href="'+url+'" download="'+(filename||'')+'" title="다운로드" onclick="event.stopPropagation()">⬇ 다운로드</a></div>';
|
||||
}
|
||||
|
||||
@@ -685,6 +700,24 @@ function onQualityChange(){
|
||||
}
|
||||
}
|
||||
|
||||
function onVideoModelChange(){
|
||||
const m=document.getElementById('f-video-model').value;
|
||||
const guidanceEl=document.getElementById('f-guidance');
|
||||
if(m==='cogvideox'){
|
||||
document.getElementById('f-frames').value=49;
|
||||
document.getElementById('f-fps').value=8;
|
||||
document.getElementById('f-steps').value=25;
|
||||
guidanceEl.value=6.0;
|
||||
document.getElementById('hint-text').textContent='CogVideoX-2B 로컬 생성 · 보통 3~5분, VRAM 여유가 적어서 해상도/프레임수/스텝수를 기본값보다 많이 올리면 실패하거나 너무 오래 걸릴 수 있어요';
|
||||
} else {
|
||||
document.getElementById('f-frames').value=65;
|
||||
document.getElementById('f-fps').value=24;
|
||||
document.getElementById('f-steps').value=40;
|
||||
guidanceEl.value=3.0;
|
||||
document.getElementById('hint-text').textContent='LTX-Video 로컬 생성 · 보통 30~90초, 시간이 걸립니다';
|
||||
}
|
||||
}
|
||||
|
||||
let genStartTime=0, genTimer=null;
|
||||
function startTimer(){
|
||||
genStartTime=Date.now();
|
||||
@@ -880,6 +913,7 @@ async function generate(){
|
||||
const seed=document.getElementById('f-seed').value;
|
||||
if(seed) body.seed=parseInt(seed);
|
||||
} else {
|
||||
body.model=document.getElementById('f-video-model').value;
|
||||
body.num_frames=parseInt(document.getElementById('f-frames').value)||undefined;
|
||||
body.fps=parseInt(document.getElementById('f-fps').value)||undefined;
|
||||
}
|
||||
@@ -896,7 +930,7 @@ async function generate(){
|
||||
if(currentTab==='image'){
|
||||
resultArea.innerHTML=resultMediaHtml(d.url, 'image', 'generated_'+Date.now()+'.png');
|
||||
} else {
|
||||
resultArea.innerHTML=resultMediaHtml(d.url, 'video', 'generated_'+Date.now()+'.mp4');
|
||||
resultArea.innerHTML=resultMediaHtml(d.url, 'video', 'generated_'+Date.now()+'.mp4', false, {width:d.width, height:d.height, num_frames:d.num_frames, fps:d.fps});
|
||||
}
|
||||
loadGallery();
|
||||
}catch(e){
|
||||
|
||||
Reference in New Issue
Block a user