이미지스튜디오를 ComfyUI 백엔드로 전환, 원본 UI 임베드, LTX 비디오 고화질 옵션 추가
diffusers venv 스크립트 방식 대신 상시구동 ComfyUI(systemd, GPU1)를 통해 SDXL/FLUX/PuLID 스타일변환/LTX 비디오를 생성하도록 전환. ComfyUI 자체 웹 UI를 리버스 프록시로 스튜디오 앱에 같은 도메인으로 임베드(routes-comfyui.ts) 하고, LTX-Video에 non-distilled "dev" 체크포인트 기반 고화질 옵션을 추가. 날씨 앱에는 Windy 제트기류/태풍 마커, 게이트웨이에는 wol-gate 웨이크 로그 조회 API도 함께 반영. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
+274
-310
@@ -82,231 +82,206 @@ function runVenvPython(script: string, timeoutMs: number): Promise<any> {
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_generate — local SDXL text-to-image
|
||||
// ---------------------------------------------------------------------------
|
||||
const SDXL_SCRIPT = (p: Record<string, any>) => `
|
||||
import os, json, sys
|
||||
try:
|
||||
import torch
|
||||
from diffusers import StableDiffusionXLPipeline
|
||||
// ComfyUI backend — replaces the old per-request venv-script spawning for
|
||||
// image_generate and image_style_transform (2026-08-05). ComfyUI runs as a
|
||||
// persistent systemd --user service (~/.config/systemd/user/comfyui.service,
|
||||
// GPU1, port 8188) with models pre-loaded in models/checkpoints|diffusion_models|
|
||||
// clip|vae|pulid|insightface|facexlib under /home/kim/comfyui. See
|
||||
// project_comfyui_evaluation / project_windy_jetstream_feature memory for how
|
||||
// these were chosen and benchmarked — FLUX.1-schnell (fp8) beat the old 4-bit
|
||||
// bitsandbytes path 2-3x on the same GPU; SDXL was a wash so it moved over too
|
||||
// for one consistent backend. video_generate (LTX/CogVideoX) is untouched —
|
||||
// still runs through the venv-script path below, no ComfyUI workflow for those.
|
||||
const COMFY_URL = 'http://127.0.0.1:8188';
|
||||
const COMFY_DIR = '/home/kim/comfyui';
|
||||
const COMFY_OUTPUT_DIR = path.join(COMFY_DIR, 'output');
|
||||
const COMFY_INPUT_DIR = path.join(COMFY_DIR, 'input');
|
||||
|
||||
pipe = StableDiffusionXLPipeline.from_pretrained(
|
||||
"stabilityai/stable-diffusion-xl-base-1.0",
|
||||
torch_dtype=torch.float16, variant="fp16", use_safetensors=True,
|
||||
)
|
||||
pipe = pipe.to("cuda")
|
||||
pipe.enable_vae_slicing()
|
||||
function randomSeed(): number {
|
||||
return Math.floor(Math.random() * 0xFFFFFFFF);
|
||||
}
|
||||
|
||||
kwargs = dict(
|
||||
prompt=${JSON.stringify(p.prompt)},
|
||||
negative_prompt=${JSON.stringify(p.negative_prompt || '')} or None,
|
||||
width=int(${p.width}), height=int(${p.height}),
|
||||
num_inference_steps=int(${p.steps}),
|
||||
guidance_scale=float(${p.guidance_scale}),
|
||||
)
|
||||
${p.seed != null ? `kwargs["generator"] = torch.Generator("cuda").manual_seed(int(${p.seed}))` : ''}
|
||||
async function comfySubmit(promptGraph: Record<string, any>): Promise<string> {
|
||||
const res = await fetch(`${COMFY_URL}/prompt`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ prompt: promptGraph }),
|
||||
});
|
||||
const data: any = await res.json().catch(() => ({}));
|
||||
if (data?.node_errors && Object.keys(data.node_errors).length) {
|
||||
throw new Error('ComfyUI rejected the workflow: ' + JSON.stringify(data.node_errors));
|
||||
}
|
||||
if (!data?.prompt_id) throw new Error(data?.error ? JSON.stringify(data.error) : 'ComfyUI did not return a prompt_id');
|
||||
return data.prompt_id;
|
||||
}
|
||||
|
||||
image = pipe(**kwargs).images[0]
|
||||
dst = ${JSON.stringify(p.dst)}
|
||||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||||
image.save(dst)
|
||||
print("###RESULT###" + json.dumps({
|
||||
"output": dst, "width": image.width, "height": image.height,
|
||||
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
|
||||
}))
|
||||
except Exception as e:
|
||||
import traceback
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
// No websocket/progress push used here (keeps this dependency-free) — just poll
|
||||
// /history, same as the manual testing that validated every workflow below.
|
||||
async function comfyPollResult(promptId: string, timeoutMs: number): Promise<{ filename: string; subfolder: string }> {
|
||||
const start = Date.now();
|
||||
while (Date.now() - start < timeoutMs) {
|
||||
const res = await fetch(`${COMFY_URL}/history/${promptId}`);
|
||||
const data: any = await res.json().catch(() => ({}));
|
||||
const entry = data?.[promptId];
|
||||
if (entry) {
|
||||
if (entry.status?.status_str === 'error') {
|
||||
const errMsg = entry.status?.messages?.find((m: any) => m[0] === 'execution_error')?.[1]?.exception_message;
|
||||
throw new Error(errMsg || 'ComfyUI execution failed');
|
||||
}
|
||||
for (const nodeOut of Object.values<any>(entry.outputs || {})) {
|
||||
if (nodeOut?.images?.length) return { filename: nodeOut.images[0].filename, subfolder: nodeOut.images[0].subfolder || '' };
|
||||
}
|
||||
}
|
||||
await new Promise((r) => setTimeout(r, 2000));
|
||||
}
|
||||
throw new Error('Timed out waiting for ComfyUI generation');
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_generate (quality: 'high') — local FLUX.1-schnell, 4-bit quantized
|
||||
// ---------------------------------------------------------------------------
|
||||
// Community mirror of the (gated) official repo — same Apache-2.0 weights,
|
||||
// just rehosted without the HF license-gate. Needed because neither FLUX.1-dev
|
||||
// nor -schnell can be pulled from black-forest-labs/* without an HF token tied
|
||||
// to an account that has clicked through the license on huggingface.co.
|
||||
const FLUX_MODEL_ID = 'Niansuh/FLUX.1-schnell';
|
||||
const FLUX_SCRIPT = (p: Record<string, any>) => `
|
||||
import os, json, sys
|
||||
try:
|
||||
import torch, shutil
|
||||
from diffusers import FluxPipeline, FluxTransformer2DModel, BitsAndBytesConfig as DBnBConfig
|
||||
from transformers import T5EncoderModel, BitsAndBytesConfig as TBnBConfig
|
||||
from huggingface_hub import snapshot_download
|
||||
async function comfyGenerateImage(
|
||||
promptGraph: Record<string, any>,
|
||||
dst: string,
|
||||
width: number,
|
||||
height: number,
|
||||
timeoutMs: number,
|
||||
): Promise<{ output: string; width: number; height: number } | { error: string }> {
|
||||
try {
|
||||
const promptId = await comfySubmit(promptGraph);
|
||||
const { filename, subfolder } = await comfyPollResult(promptId, timeoutMs);
|
||||
const srcFile = path.join(COMFY_OUTPUT_DIR, subfolder, filename);
|
||||
fs.mkdirSync(path.dirname(dst), { recursive: true });
|
||||
fs.copyFileSync(srcFile, dst);
|
||||
return { output: dst, width, height };
|
||||
} catch (e: any) {
|
||||
return { error: e?.message || String(e) };
|
||||
}
|
||||
}
|
||||
|
||||
MODEL_ID = "${FLUX_MODEL_ID}"
|
||||
function sdxlWorkflow(p: { prompt: string; negative_prompt: string; width: number; height: number; steps: number; guidance_scale: number; seed: number }) {
|
||||
return {
|
||||
'3': { class_type: 'KSampler', inputs: { cfg: p.guidance_scale, denoise: 1.0, latent_image: ['5', 0], model: ['4', 0], negative: ['7', 0], positive: ['6', 0], sampler_name: 'euler', scheduler: 'normal', seed: p.seed, steps: p.steps } },
|
||||
'4': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: 'sd_xl_base_1.0.safetensors' } },
|
||||
'5': { class_type: 'EmptyLatentImage', inputs: { batch_size: 1, height: p.height, width: p.width } },
|
||||
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['4', 1], text: p.prompt } },
|
||||
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['4', 1], text: p.negative_prompt || '' } },
|
||||
'8': { class_type: 'VAEDecode', inputs: { samples: ['3', 0], vae: ['4', 2] } },
|
||||
'9': { class_type: 'SaveImage', inputs: { filename_prefix: 'sdxl_api', images: ['8', 0] } },
|
||||
};
|
||||
}
|
||||
|
||||
# This mirror ships scheduler/config.json instead of the scheduler_config.json
|
||||
# filename diffusers expects — patch it once per cache (idempotent).
|
||||
snap_dir = snapshot_download(MODEL_ID, allow_patterns=["scheduler/config.json"])
|
||||
sched_cfg = os.path.join(snap_dir, "scheduler", "scheduler_config.json")
|
||||
if not os.path.exists(sched_cfg):
|
||||
shutil.copy(os.path.join(snap_dir, "scheduler", "config.json"), sched_cfg)
|
||||
function fluxSchnellWorkflow(p: { prompt: string; width: number; height: number; steps: number; seed: number }) {
|
||||
return {
|
||||
'1': { class_type: 'UNETLoader', inputs: { unet_name: 'flux1-schnell-fp8.safetensors', weight_dtype: 'default' } },
|
||||
'2': { class_type: 'DualCLIPLoader', inputs: { clip_name1: 'clip_l.safetensors', clip_name2: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'flux' } },
|
||||
'3': { class_type: 'VAELoader', inputs: { vae_name: 'ae.safetensors' } },
|
||||
'4': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.prompt } },
|
||||
'5': { class_type: 'EmptySD3LatentImage', inputs: { batch_size: 1, height: p.height, width: p.width } },
|
||||
'6': { class_type: 'KSampler', inputs: { cfg: 1.0, denoise: 1.0, latent_image: ['5', 0], model: ['1', 0], negative: ['4', 0], positive: ['4', 0], sampler_name: 'euler', scheduler: 'simple', seed: p.seed, steps: p.steps } },
|
||||
'7': { class_type: 'VAEDecode', inputs: { samples: ['6', 0], vae: ['3', 0] } },
|
||||
'8': { class_type: 'SaveImage', inputs: { filename_prefix: 'flux_api', images: ['7', 0] } },
|
||||
};
|
||||
}
|
||||
|
||||
transformer_4bit = FluxTransformer2DModel.from_pretrained(
|
||||
MODEL_ID, subfolder="transformer",
|
||||
quantization_config=DBnBConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16),
|
||||
torch_dtype=torch.bfloat16,
|
||||
)
|
||||
text_encoder_2_4bit = T5EncoderModel.from_pretrained(
|
||||
MODEL_ID, subfolder="text_encoder_2",
|
||||
quantization_config=TBnBConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16),
|
||||
torch_dtype=torch.bfloat16,
|
||||
)
|
||||
pipe = FluxPipeline.from_pretrained(
|
||||
MODEL_ID,
|
||||
transformer=transformer_4bit,
|
||||
text_encoder_2=text_encoder_2_4bit,
|
||||
torch_dtype=torch.bfloat16,
|
||||
)
|
||||
pipe.enable_model_cpu_offload()
|
||||
// FLUX.1-dev + PuLID (identity-locked img2img). Unlike the txt2img workflows
|
||||
// above, guidance=3.5/denoise~0.45 (the old SDXL-tuned defaults) produced almost
|
||||
// no visible style change at all when this was benchmarked — FLUX's flow-matching
|
||||
// sampler needs guidance~8 and denoise~0.85 before the caricature prompt actually
|
||||
// overrides the source photo. See IMG2IMG_STYLE_PRESETS below for the tuned values.
|
||||
function fluxPulidStyleWorkflow(p: { imageFilename: string; prompt: string; negative_prompt: string; denoise: number; guidance: number; steps: number; seed: number; pulid_weight: number }) {
|
||||
const graph: Record<string, any> = {
|
||||
'1': { class_type: 'UNETLoader', inputs: { unet_name: 'flux1-dev-fp8-e4m3fn.safetensors', weight_dtype: 'default' } },
|
||||
'2': { class_type: 'DualCLIPLoader', inputs: { clip_name1: 'clip_l.safetensors', clip_name2: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'flux' } },
|
||||
'3': { class_type: 'VAELoader', inputs: { vae_name: 'ae.safetensors' } },
|
||||
'4': { class_type: 'LoadImage', inputs: { image: p.imageFilename } },
|
||||
'4b': { class_type: 'ImageScaleToTotalPixels', inputs: { image: ['4', 0], upscale_method: 'lanczos', megapixels: 1.0, resolution_steps: 16 } },
|
||||
'5': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.prompt } },
|
||||
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.negative_prompt || '' } },
|
||||
'7': { class_type: 'FluxGuidance', inputs: { conditioning: ['5', 0], guidance: p.guidance } },
|
||||
'12': { class_type: 'VAEEncode', inputs: { pixels: ['4b', 0], vae: ['3', 0] } },
|
||||
'13': { class_type: 'KSampler', inputs: { model: ['1', 0], positive: ['7', 0], negative: ['6', 0], latent_image: ['12', 0], seed: p.seed, steps: p.steps, cfg: 1.0, sampler_name: 'euler', scheduler: 'simple', denoise: p.denoise } },
|
||||
'14': { class_type: 'VAEDecode', inputs: { samples: ['13', 0], vae: ['3', 0] } },
|
||||
'15': { class_type: 'SaveImage', inputs: { filename_prefix: 'flux_pulid_api', images: ['14', 0] } },
|
||||
};
|
||||
// face_lock=false (or no face found upstream): skip PuLID nodes entirely rather
|
||||
// than including a disconnected node — ComfyUI validates the graph as a DAG, so
|
||||
// a "no-op" node with missing required inputs would just fail to queue.
|
||||
if (p.pulid_weight > 0) {
|
||||
graph['8'] = { class_type: 'PulidFluxModelLoader', inputs: { pulid_file: 'pulid_flux_v0.9.1.safetensors' } };
|
||||
graph['9'] = { class_type: 'PulidFluxInsightFaceLoader', inputs: { provider: 'CUDA' } };
|
||||
graph['10'] = { class_type: 'PulidFluxEvaClipLoader', inputs: {} };
|
||||
graph['11'] = { class_type: 'ApplyPulidFlux', inputs: { model: ['1', 0], pulid_flux: ['8', 0], eva_clip: ['10', 0], face_analysis: ['9', 0], image: ['4b', 0], weight: p.pulid_weight, start_at: 0.0, end_at: 1.0 } };
|
||||
graph['13'].inputs.model = ['11', 0];
|
||||
}
|
||||
return graph;
|
||||
}
|
||||
|
||||
kwargs = dict(
|
||||
prompt=${JSON.stringify(p.prompt)},
|
||||
guidance_scale=0.0,
|
||||
num_inference_steps=int(${p.steps}),
|
||||
max_sequence_length=256,
|
||||
width=int(${p.width}), height=int(${p.height}),
|
||||
)
|
||||
${p.seed != null ? `kwargs["generator"] = torch.Generator("cpu").manual_seed(int(${p.seed}))` : ''}
|
||||
// LTX-Video text-to-video, ported to ComfyUI 2026-08-05 (see project_comfyui_evaluation
|
||||
// memory) — the old venv-script path left GPU1 fully free between requests, but once
|
||||
// ComfyUI started staying resident there for images, the two independent processes
|
||||
// started fighting over the same VRAM (one real request took >200s and had to be
|
||||
// killed). Moving LTX into ComfyUI too puts all of GPU1 under one memory manager.
|
||||
// Node graph copied from ComfyUI's own bundled ltxv_text_to_video.json template —
|
||||
// SamplerCustom+LTXVScheduler+KSamplerSelect, not plain KSampler, is how LTX is meant
|
||||
// to be driven. The T5-XXL text encoder is the same file already downloaded for FLUX.
|
||||
function ltxVideoWorkflow(p: { prompt: string; negative_prompt: string; width: number; height: number; length: number; fps: number; steps: number; guidance_scale: number; seed: number; ckpt_name?: string }) {
|
||||
return {
|
||||
'38': { class_type: 'CLIPLoader', inputs: { clip_name: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'ltxv', device: 'default' } },
|
||||
'44': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: p.ckpt_name || 'ltxv-2b-0.9.8-distilled-fp8.safetensors' } },
|
||||
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.prompt } },
|
||||
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.negative_prompt } },
|
||||
'70': { class_type: 'EmptyLTXVLatentVideo', inputs: { width: p.width, height: p.height, length: p.length, batch_size: 1 } },
|
||||
'69': { class_type: 'LTXVConditioning', inputs: { positive: ['6', 0], negative: ['7', 0], frame_rate: p.fps } },
|
||||
'71': { class_type: 'LTXVScheduler', inputs: { steps: p.steps, max_shift: 2.05, base_shift: 0.95, stretch: true, terminal: 0.1, latent: ['70', 0] } },
|
||||
'73': { class_type: 'KSamplerSelect', inputs: { sampler_name: 'euler' } },
|
||||
'72': { class_type: 'SamplerCustom', inputs: { model: ['44', 0], add_noise: true, noise_seed: p.seed, cfg: p.guidance_scale, positive: ['69', 0], negative: ['69', 1], sampler: ['73', 0], sigmas: ['71', 0], latent_image: ['70', 0] } },
|
||||
'8': { class_type: 'VAEDecode', inputs: { samples: ['72', 0], vae: ['44', 2] } },
|
||||
'78': { class_type: 'CreateVideo', inputs: { images: ['8', 0], fps: p.fps } },
|
||||
'79': { class_type: 'SaveVideo', inputs: { video: ['78', 0], filename_prefix: 'ltx_api', format: 'mp4', codec: 'h264' } },
|
||||
};
|
||||
}
|
||||
|
||||
image = pipe(**kwargs).images[0]
|
||||
dst = ${JSON.stringify(p.dst)}
|
||||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||||
image.save(dst)
|
||||
print("###RESULT###" + json.dumps({
|
||||
"output": dst, "width": image.width, "height": image.height,
|
||||
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
|
||||
}))
|
||||
except Exception as e:
|
||||
import traceback
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_style_transform — local SDXL img2img (cartoon/watercolor only)
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_edit's stylize op (AnimeGANv2 + OpenCV) already covers anime/sketch/bw
|
||||
// well — those are identity-preserving and near-instant. But its "cartoon" and
|
||||
// "watercolor" presets are mechanical edge/color filters that read as a mild
|
||||
// photo filter rather than real artistic reinterpretation. SDXL img2img trades
|
||||
// speed and some facial-identity fidelity for a genuinely more painterly result
|
||||
// on those two styles specifically.
|
||||
const IMG2IMG_SCRIPT = (p: Record<string, any>) => `
|
||||
import os, json, sys
|
||||
try:
|
||||
import torch
|
||||
from diffusers import StableDiffusionXLImg2ImgPipeline
|
||||
from PIL import Image, ImageOps
|
||||
|
||||
pipe = StableDiffusionXLImg2ImgPipeline.from_pretrained(
|
||||
"stabilityai/stable-diffusion-xl-base-1.0",
|
||||
torch_dtype=torch.float16, variant="fp16", use_safetensors=True,
|
||||
)
|
||||
pipe = pipe.to("cuda")
|
||||
pipe.enable_vae_slicing()
|
||||
|
||||
init_image = ImageOps.exif_transpose(Image.open(${JSON.stringify(p.src)})).convert("RGB")
|
||||
max_side = int(${p.max_side})
|
||||
w, h = init_image.size
|
||||
scale = min(1.0, max_side / max(w, h))
|
||||
w, h = max(8, int(w * scale) // 8 * 8), max(8, int(h * scale) // 8 * 8)
|
||||
init_image = init_image.resize((w, h))
|
||||
|
||||
# Face-ID lock (IP-Adapter-FaceID/SDXL): without this, img2img alone lets the
|
||||
# style prompt pull the face toward SDXL's own learned priors — this is what
|
||||
# caused the watercolor identity-drift problem. Extracting a face embedding and
|
||||
# conditioning generation on it keeps the result recognizably the same person.
|
||||
# Falls back to plain img2img if no face is detected (e.g. non-portrait photo).
|
||||
face_embeds = None
|
||||
if ${p.face_lock ? 'True' : 'False'}:
|
||||
try:
|
||||
import cv2
|
||||
from insightface.app import FaceAnalysis
|
||||
_face_app = FaceAnalysis(name="buffalo_l", providers=["CPUExecutionProvider"])
|
||||
_face_app.prepare(ctx_id=0, det_size=(640, 640))
|
||||
_cv_img = cv2.imread(${JSON.stringify(p.src)})
|
||||
_faces = _face_app.get(_cv_img)
|
||||
if _faces:
|
||||
_emb = torch.from_numpy(_faces[0].normed_embedding).unsqueeze(0).unsqueeze(0)
|
||||
face_embeds = torch.cat([torch.zeros_like(_emb), _emb], dim=0).to(dtype=torch.float16, device="cuda")
|
||||
except Exception:
|
||||
face_embeds = None
|
||||
|
||||
kwargs = dict(
|
||||
prompt=${JSON.stringify(p.prompt)},
|
||||
negative_prompt=${JSON.stringify(p.negative_prompt)},
|
||||
image=init_image,
|
||||
strength=float(${p.strength}),
|
||||
num_inference_steps=int(${p.steps}),
|
||||
guidance_scale=float(${p.guidance_scale}),
|
||||
)
|
||||
${p.seed != null ? `kwargs["generator"] = torch.Generator("cuda").manual_seed(int(${p.seed}))` : ''}
|
||||
|
||||
if face_embeds is not None:
|
||||
pipe.load_ip_adapter("h94/IP-Adapter-FaceID", subfolder=None, weight_name="ip-adapter-faceid_sdxl.bin", image_encoder_folder=None)
|
||||
pipe.set_ip_adapter_scale(0.8)
|
||||
kwargs["ip_adapter_image_embeds"] = [face_embeds]
|
||||
|
||||
image = pipe(**kwargs).images[0]
|
||||
dst = ${JSON.stringify(p.dst)}
|
||||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||||
image.save(dst)
|
||||
print("###RESULT###" + json.dumps({
|
||||
"output": dst, "width": image.width, "height": image.height,
|
||||
"face_locked": face_embeds is not None,
|
||||
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
|
||||
}))
|
||||
except Exception as e:
|
||||
import traceback
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
// Same submit/poll plumbing as comfyGenerateImage but the SaveVideo node's output file
|
||||
// isn't a PNG — copy verbatim rather than assuming an image extension.
|
||||
async function comfyGenerateVideo(promptGraph: Record<string, any>, dst: string, timeoutMs: number): Promise<{ output: string } | { error: string }> {
|
||||
try {
|
||||
const promptId = await comfySubmit(promptGraph);
|
||||
const { filename, subfolder } = await comfyPollResult(promptId, timeoutMs);
|
||||
const srcFile = path.join(COMFY_OUTPUT_DIR, subfolder, filename);
|
||||
fs.mkdirSync(path.dirname(dst), { recursive: true });
|
||||
fs.copyFileSync(srcFile, dst);
|
||||
return { output: dst };
|
||||
} catch (e: any) {
|
||||
return { error: e?.message || String(e) };
|
||||
}
|
||||
}
|
||||
|
||||
const IMG2IMG_STYLE_PRESETS: Record<string, { prompt: string; negative: string; strength: number; guidance_scale: number }> = {
|
||||
cartoon: {
|
||||
// strength was 0.6 — looked like a genuine caricature but the pose/expression
|
||||
// ("인상") drifted too far from the source photo even with face-ID identity
|
||||
// locked, since face-ID only conditions identity, not expression/composition.
|
||||
// 0.45 keeps the photo's actual expression and framing intact while still
|
||||
// applying visible cel-shaded/cartoon coloring and clean outlines.
|
||||
// Tuned on FLUX.1-dev + PuLID (2026-08-05) — see fluxPulidStyleWorkflow comment.
|
||||
// The old SDXL-era numbers (strength 0.45 / guidance 7.0) are no longer used.
|
||||
prompt: 'caricature portrait illustration, exaggerated facial features, bold clean outlines, vibrant flat colors, humorous comic art style, digital illustration',
|
||||
negative: 'photorealistic, blurry, deformed hands, extra limbs, low quality, watermark, text',
|
||||
strength: 0.45,
|
||||
guidance_scale: 7.0,
|
||||
strength: 0.85,
|
||||
guidance_scale: 8.0,
|
||||
},
|
||||
// "watercolor" was tried here too but dropped: the "watercolor painting portrait"
|
||||
// prompt pulled SDXL toward its own learned face priors much harder than cartoon
|
||||
// does — even down to strength 0.25-0.3 it reliably drifted the subject toward a
|
||||
// different-looking (often different-gender-presenting) face, with the watercolor
|
||||
// effect barely visible at the strengths low enough to keep identity intact.
|
||||
// image_edit's classic OpenCV watercolor filter (mild but identity-safe) is used
|
||||
// instead — see EDIT_STYLE_MAP in gateway/routes/imagegen.ts.
|
||||
//
|
||||
// "sketch" was tried too and also dropped: going from a color photo to monochrome
|
||||
// graphite linework needed strength/guidance high enough (0.8/9.0) that, with no
|
||||
// seed pinned, run-to-run variance sometimes drifted the face into a different-
|
||||
// looking person (same failure shape as watercolor). Backing off to 0.6/8.0 for
|
||||
// more consistent identity made the result look worse than image_edit's classic
|
||||
// sketch filter — flat/muddy rather than either a clean sketch or a good likeness.
|
||||
// Reverted to image_edit's OpenCV sketch filter (see EDIT_STYLE_MAP).
|
||||
// watercolor/sketch: still dropped for the same identity-drift reasons as before
|
||||
// (see git history) — image_edit's OpenCV filters remain the fallback for those.
|
||||
};
|
||||
|
||||
export const imageStyleTransformTool = {
|
||||
name: 'image_style_transform',
|
||||
description: [
|
||||
'Transform an existing photo (e.g. a portrait) into a new artistic style using local SDXL img2img with IP-Adapter-FaceID identity locking (runs on-machine GPU, no external API).',
|
||||
'Supports style="cartoon" (caricature/comic illustration) — more artistically interpretive than image_edit\'s classic OpenCV cartoon filter, at the cost of speed (~15-40s vs ~2-3s). When a face is detected in the source photo, a face embedding (insightface/buffalo_l) conditions the generation via IP-Adapter-FaceID so the result stays recognizably the same person even under a strong style prompt — falls back to plain img2img if no face is found.',
|
||||
'For anime/sketch/watercolor/black-and-white, use image_edit\'s stylize/filter operations instead — those are faster (both "watercolor" and "sketch" SDXL presets were tried and dropped: the style jump needed enough strength that, with no seed pinned, results sometimes drifted the face into a different-looking person).',
|
||||
'Transform an existing photo (e.g. a portrait) into a new artistic style using local FLUX.1-dev img2img with PuLID identity locking, via ComfyUI (runs on-machine GPU, no external API).',
|
||||
'Supports style="cartoon" (caricature/comic illustration) — more artistically interpretive than image_edit\'s classic OpenCV cartoon filter, at the cost of speed (~60-80s vs ~2-3s). A face embedding (insightface + EVA-CLIP) conditions the generation via PuLID so the result stays recognizably the same person even under a strong style prompt.',
|
||||
'For anime/sketch/watercolor/black-and-white, use image_edit\'s stylize/filter operations instead — those are faster (both "watercolor" and "sketch" presets were tried and dropped: the style jump needed enough strength that results sometimes drifted the face into a different-looking person).',
|
||||
'Returns the transformed image inline in the chat.',
|
||||
].join('\n'),
|
||||
schema: {
|
||||
image: 'Path to the source image to transform (required)',
|
||||
style: '"cartoon" (caricature/comic illustration) — currently the only supported style',
|
||||
face_lock: 'Whether to lock facial identity via IP-Adapter-FaceID when a face is detected (optional, default true). Turn off to compare plain img2img — face-ID conditioning sometimes reads as slightly uncanny/over-smoothed; plain img2img gives looser but more natural-looking results.',
|
||||
strength: 'How strongly to restyle, 0.05-1.0 (optional; default 0.45). Lower preserves the original photo\'s pose/expression more, higher restyles more aggressively but can drift away from them.',
|
||||
steps: 'Denoising steps (default 40, range 15-50)',
|
||||
guidance_scale: 'How closely to follow the style prompt (default 7.0)',
|
||||
face_lock: 'Whether to lock facial identity via PuLID (optional, default true). Turn off to compare plain img2img — face-lock conditioning sometimes reads as slightly uncanny/over-smoothed; plain img2img gives looser but more natural-looking results.',
|
||||
strength: 'How strongly to restyle, 0.05-1.0 (optional; default 0.85 — FLUX\'s flow-matching sampler needs much higher values than the old SDXL default to produce a visible style change at all). Lower preserves the original photo\'s pose/expression more, higher restyles more aggressively but can drift away from them.',
|
||||
steps: 'Denoising steps (default 20, range 10-40)',
|
||||
guidance_scale: 'How closely to follow the style prompt (default 8.0)',
|
||||
seed: 'Random seed for reproducibility (optional)',
|
||||
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
|
||||
},
|
||||
@@ -344,34 +319,40 @@ export const imageStyleTransformTool = {
|
||||
outPath = path.resolve(workspacePath, outPath);
|
||||
}
|
||||
|
||||
const params = {
|
||||
src: srcPath,
|
||||
const faceLock = args?.face_lock ?? true;
|
||||
const denoise = Math.min(1, Math.max(0.05, args?.strength ?? preset.strength));
|
||||
const guidance = toFiniteNumber(args?.guidance_scale, preset.guidance_scale);
|
||||
const steps = Math.min(40, Math.max(10, args?.steps ?? 20));
|
||||
const seed = args?.seed != null ? toFiniteNumber(args.seed, 0) : randomSeed();
|
||||
|
||||
// ComfyUI reads source images from its own input/ dir — copy in under a unique
|
||||
// name (same host, so a filesystem copy, no HTTP upload round-trip needed).
|
||||
fs.mkdirSync(COMFY_INPUT_DIR, { recursive: true });
|
||||
const inputFilename = `style_src_${Date.now()}${path.extname(srcPath) || '.png'}`;
|
||||
fs.copyFileSync(srcPath, path.join(COMFY_INPUT_DIR, inputFilename));
|
||||
|
||||
const graph = fluxPulidStyleWorkflow({
|
||||
imageFilename: inputFilename,
|
||||
prompt: preset.prompt,
|
||||
negative_prompt: preset.negative,
|
||||
strength: Math.min(1, Math.max(0.05, args?.strength ?? preset.strength)),
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
|
||||
guidance_scale: toFiniteNumber(args?.guidance_scale, preset.guidance_scale),
|
||||
seed: args?.seed != null ? toFiniteNumber(args.seed, 0) : undefined,
|
||||
face_lock: args?.face_lock ?? true,
|
||||
max_side: 1024,
|
||||
dst: outPath,
|
||||
};
|
||||
// 600s (not 180s like the other scripts): the face-ID adapter weights (~1.7GB)
|
||||
// download from HF Hub on first run and can take several minutes.
|
||||
const result = await runVenvPython(IMG2IMG_SCRIPT(params), 600_000);
|
||||
if (result.error || !result.output) {
|
||||
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
|
||||
denoise, guidance, steps, seed,
|
||||
pulid_weight: faceLock ? 1.0 : 0,
|
||||
});
|
||||
// 600s (not the 180s the txt2img workflows use): first request after a
|
||||
// service restart also has to load PuLID/EVA-CLIP/insightface on top of FLUX.1-dev.
|
||||
const result = await comfyGenerateImage(graph, outPath, 0, 0, 600_000);
|
||||
if ('error' in result || !result.output) {
|
||||
return { success: false, error: 'error' in result ? result.error : 'Generator returned no output file' };
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stdout: [
|
||||
`Style: ${style}`,
|
||||
`Transformed: ${result.width} × ${result.height} px`,
|
||||
`Style: ${style}${faceLock ? ' (face-locked via PuLID)' : ''}`,
|
||||
'',
|
||||
buildImageMarkdown(result.output, workspacePath),
|
||||
].join('\n'),
|
||||
data: { ...result, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
|
||||
data: { output: result.output, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
|
||||
};
|
||||
},
|
||||
};
|
||||
@@ -379,9 +360,9 @@ export const imageStyleTransformTool = {
|
||||
export const imageGenerateTool = {
|
||||
name: 'image_generate',
|
||||
description: [
|
||||
'Generate an image from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
|
||||
'quality="fast" (default): SDXL, ~10-20 seconds, default resolution 1024×1024, good for casual/quick illustrations.',
|
||||
'quality="high": FLUX.1-schnell (4-bit quantized), ~40-60 seconds total (model load + generation), noticeably more photorealistic detail and prompt accuracy. Use only when the user explicitly asks for higher quality/detail/photorealism, or for a "고품질" request — otherwise default to fast.',
|
||||
'Generate an image from a text prompt using a local diffusion model via ComfyUI (runs on-machine GPU, no external API).',
|
||||
'quality="fast" (default): SDXL, ~15-20 seconds, default resolution 1024×1024, good for casual/quick illustrations.',
|
||||
'quality="high": FLUX.1-schnell (fp8), ~15-25 seconds, noticeably more photorealistic detail and prompt accuracy — and no longer much slower than fast mode (moved off the old 4-bit bitsandbytes path, which needed ~40-60s), so lean toward "high" more readily than before. Still default to fast for quick/casual requests.',
|
||||
'Returns the generated image inline in the chat.',
|
||||
].join('\n'),
|
||||
schema: {
|
||||
@@ -433,30 +414,20 @@ export const imageGenerateTool = {
|
||||
const width = Math.round((args?.width ?? 1024) / 16) * 16;
|
||||
const height = Math.round((args?.height ?? 1024) / 16) * 16;
|
||||
|
||||
let result: any;
|
||||
if (isHighQuality) {
|
||||
const params = {
|
||||
prompt: translatedPrompt || prompt,
|
||||
width, height,
|
||||
steps: Math.min(8, Math.max(1, args?.steps ?? 4)),
|
||||
seed: args?.seed != null ? toFiniteNumber(args.seed, 0) : undefined,
|
||||
dst: outPath,
|
||||
};
|
||||
result = await runVenvPython(FLUX_SCRIPT(params), 180_000);
|
||||
} else {
|
||||
const params = {
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width, height,
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
|
||||
guidance_scale: toFiniteNumber(args?.guidance_scale, 7.0),
|
||||
seed: args?.seed != null ? toFiniteNumber(args.seed, 0) : undefined,
|
||||
dst: outPath,
|
||||
};
|
||||
result = await runVenvPython(SDXL_SCRIPT(params), 180_000);
|
||||
}
|
||||
if (result.error || !result.output) {
|
||||
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
|
||||
const seed = args?.seed != null ? toFiniteNumber(args.seed, 0) : randomSeed();
|
||||
const graph = isHighQuality
|
||||
? fluxSchnellWorkflow({ prompt: translatedPrompt || prompt, width, height, steps: Math.min(8, Math.max(1, args?.steps ?? 4)), seed })
|
||||
: sdxlWorkflow({
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width, height,
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
|
||||
guidance_scale: toFiniteNumber(args?.guidance_scale, 7.0),
|
||||
seed,
|
||||
});
|
||||
const result = await comfyGenerateImage(graph, outPath, width, height, 180_000);
|
||||
if ('error' in result || !result.output) {
|
||||
return { success: false, error: 'error' in result ? result.error : 'Generator returned no output file' };
|
||||
}
|
||||
|
||||
return {
|
||||
@@ -473,42 +444,9 @@ export const imageGenerateTool = {
|
||||
};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// video_generate — local LTX-Video text-to-video
|
||||
// video_generate — LTX-Video moved to ComfyUI (see ltxVideoWorkflow above);
|
||||
// CogVideoX-2B still runs the old venv-script path below.
|
||||
// ---------------------------------------------------------------------------
|
||||
const LTX_SCRIPT = (p: Record<string, any>) => `
|
||||
import os, json, sys
|
||||
try:
|
||||
import torch
|
||||
from diffusers import LTXPipeline
|
||||
from diffusers.utils import export_to_video
|
||||
|
||||
pipe = LTXPipeline.from_pretrained("Lightricks/LTX-Video", torch_dtype=torch.bfloat16)
|
||||
pipe.enable_model_cpu_offload()
|
||||
|
||||
video = pipe(
|
||||
prompt=${JSON.stringify(p.prompt)},
|
||||
negative_prompt=${JSON.stringify(p.negative_prompt)},
|
||||
width=int(${p.width}), height=int(${p.height}),
|
||||
num_frames=int(${p.num_frames}),
|
||||
num_inference_steps=int(${p.steps}),
|
||||
guidance_scale=float(${p.guidance_scale}),
|
||||
).frames[0]
|
||||
|
||||
dst = ${JSON.stringify(p.dst)}
|
||||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||||
export_to_video(video, dst, fps=int(${p.fps}))
|
||||
|
||||
stat = os.stat(dst)
|
||||
print("###RESULT###" + json.dumps({
|
||||
"output": dst, "width": int(${p.width}), "height": int(${p.height}),
|
||||
"num_frames": int(${p.num_frames}), "fps": int(${p.fps}),
|
||||
"size_bytes": stat.st_size,
|
||||
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
|
||||
}))
|
||||
except Exception as e:
|
||||
import traceback
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
|
||||
// CogVideoX-2B — added 2026-07-17 as a second option alongside LTX-Video for quality
|
||||
// comparison. Barely fits: peak VRAM measured at ~11GB of the 12GB card even with
|
||||
@@ -557,12 +495,12 @@ export const videoGenerateTool = {
|
||||
name: 'video_generate',
|
||||
description: [
|
||||
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
|
||||
'Two models available via the "model" param: "ltx" (default — fast, ~30-90s, more headroom for higher resolution/frame count) and "cogvideox" (THUDM CogVideoX-2B — slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try both and compare if unsure which fits the request.',
|
||||
'Three models available via the "model" param: "ltx" (default — LTX-Video 2B distilled via ComfyUI, ~25-40s), "ltx_hq" (LTX-Video 2B non-distilled/"dev" checkpoint via ComfyUI — noticeably better motion coherence and detail, ~40-90s, VRAM headroom is tight at ~10GB/12GB so avoid stacking a large image_generate call at the same time), and "cogvideox" (THUDM CogVideoX-2B — still the old venv-script path, slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try a couple and compare if unsure which fits the request.',
|
||||
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
|
||||
].join('\n'),
|
||||
schema: {
|
||||
prompt: 'Text description of the video/scene to generate (English works best)',
|
||||
model: '"ltx" (default, fast) or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
|
||||
model: '"ltx" (default, fast, distilled), "ltx_hq" (LTX-Video non-distilled "dev" checkpoint, better quality, slower), or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
|
||||
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
|
||||
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
|
||||
height: 'Video height in pixels, multiple of 32 (default 480)',
|
||||
@@ -576,7 +514,7 @@ export const videoGenerateTool = {
|
||||
type: 'object',
|
||||
properties: {
|
||||
prompt: { type: 'string' },
|
||||
model: { type: 'string', enum: ['ltx', 'cogvideox'] },
|
||||
model: { type: 'string', enum: ['ltx', 'ltx_hq', 'cogvideox'] },
|
||||
negative_prompt: { type: 'string' },
|
||||
width: { type: 'number' },
|
||||
height: { type: 'number' },
|
||||
@@ -593,13 +531,14 @@ export const videoGenerateTool = {
|
||||
const prompt = String(args?.prompt || '').trim();
|
||||
if (!prompt) return { success: false, error: 'prompt is required' };
|
||||
|
||||
const model = args?.model === 'cogvideox' ? 'cogvideox' : 'ltx';
|
||||
const model = args?.model === 'cogvideox' ? 'cogvideox' : args?.model === 'ltx_hq' ? 'ltx_hq' : 'ltx';
|
||||
const isCogVideoX = model === 'cogvideox';
|
||||
const isHQ = model === 'ltx_hq';
|
||||
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
let outPath = String(args?.output || '').trim();
|
||||
if (!outPath) {
|
||||
outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : 'ltx'}_${Date.now()}.mp4`);
|
||||
outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : isHQ ? 'ltx_hq' : 'ltx'}_${Date.now()}.mp4`);
|
||||
} else if (!path.isAbsolute(outPath)) {
|
||||
outPath = path.resolve(workspacePath, outPath);
|
||||
}
|
||||
@@ -610,41 +549,66 @@ export const videoGenerateTool = {
|
||||
translatePromptToEnglish(negativePromptRaw),
|
||||
]);
|
||||
|
||||
const params = {
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width: Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32,
|
||||
height: Math.round((args?.height ?? 480) / 32) * 32,
|
||||
num_frames: toFiniteNumber(args?.num_frames, isCogVideoX ? 49 : 65),
|
||||
fps: toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24),
|
||||
// CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50
|
||||
// (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any
|
||||
// reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it
|
||||
// lower to keep typical runs under ~4 minutes; still overridable via the steps param.
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40))),
|
||||
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0))),
|
||||
dst: outPath,
|
||||
};
|
||||
const width = Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32;
|
||||
const height = Math.round((args?.height ?? 480) / 32) * 32;
|
||||
const fps = toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24);
|
||||
const guidance_scale = Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0)));
|
||||
// CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50
|
||||
// (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any
|
||||
// reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it
|
||||
// lower to keep typical runs under ~4 minutes; still overridable via the steps param.
|
||||
const steps = Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40)));
|
||||
|
||||
const script = isCogVideoX ? COGVIDEOX_SCRIPT(params) : LTX_SCRIPT(params);
|
||||
const result = await runVenvPython(script, 600_000);
|
||||
if (result.error || !result.output) {
|
||||
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
|
||||
let num_frames: number;
|
||||
let outputMeta: { output: string };
|
||||
if (isCogVideoX) {
|
||||
num_frames = toFiniteNumber(args?.num_frames, 49);
|
||||
const params = {
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width, height, num_frames, fps, steps, guidance_scale,
|
||||
dst: outPath,
|
||||
};
|
||||
const result = await runVenvPython(COGVIDEOX_SCRIPT(params), 600_000);
|
||||
if (result.error || !result.output) {
|
||||
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
|
||||
}
|
||||
outputMeta = { output: result.output };
|
||||
} else {
|
||||
// LTX-Video wants length as 8k+1 (its temporal VAE downsamples by 8) — snap
|
||||
// whatever was requested to the nearest valid value instead of erroring.
|
||||
const requestedFrames = toFiniteNumber(args?.num_frames, 65);
|
||||
num_frames = Math.max(9, Math.round((requestedFrames - 1) / 8) * 8 + 1);
|
||||
const graph = ltxVideoWorkflow({
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width, height, length: num_frames, fps, steps, guidance_scale,
|
||||
seed: randomSeed(),
|
||||
ckpt_name: isHQ ? 'ltxv-2b-0.9.6-dev-04-25.safetensors' : undefined,
|
||||
});
|
||||
// The non-distilled "dev" checkpoint is ~1.4x the weight size of the distilled one and
|
||||
// runs full CFG (two forward passes/step instead of one), so it's meaningfully slower —
|
||||
// give it more headroom than the regular ltx path's 300s.
|
||||
const result = await comfyGenerateVideo(graph, outPath, isHQ ? 420_000 : 300_000);
|
||||
if ('error' in result) {
|
||||
return { success: false, error: result.error };
|
||||
}
|
||||
outputMeta = result;
|
||||
}
|
||||
|
||||
const relOut = path.relative(workspacePath, result.output).replace(/\\/g, '/');
|
||||
const sizeMB = (result.size_bytes / 1024 / 1024).toFixed(2);
|
||||
const durationSec = (result.num_frames / result.fps).toFixed(1);
|
||||
const relOut = path.relative(workspacePath, outputMeta.output).replace(/\\/g, '/');
|
||||
const sizeMB = (fs.statSync(outputMeta.output).size / 1024 / 1024).toFixed(2);
|
||||
const durationSec = (num_frames / fps).toFixed(1);
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stdout: [
|
||||
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
|
||||
`Generated: ${result.width} × ${result.height} px | ${durationSec}s (${result.num_frames}f @ ${result.fps}fps) | ${sizeMB} MB`,
|
||||
`Generated: ${width} × ${height} px | ${durationSec}s (${num_frames}f @ ${fps}fps) | ${sizeMB} MB`,
|
||||
'',
|
||||
`[${path.basename(result.output)}](/api/files/${relOut})`,
|
||||
`[${path.basename(outputMeta.output)}](/api/files/${relOut})`,
|
||||
].filter((line): line is string => line !== null).join('\n'),
|
||||
data: { ...result, rel_path: relOut },
|
||||
data: { output: outputMeta.output, width, height, num_frames, fps, rel_path: relOut },
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user