이미지스튜디오를 ComfyUI 백엔드로 전환, 원본 UI 임베드, LTX 비디오 고화질 옵션 추가

diffusers venv 스크립트 방식 대신 상시구동 ComfyUI(systemd, GPU1)를 통해
SDXL/FLUX/PuLID 스타일변환/LTX 비디오를 생성하도록 전환. ComfyUI 자체 웹
UI를 리버스 프록시로 스튜디오 앱에 같은 도메인으로 임베드(routes-comfyui.ts)
하고, LTX-Video에 non-distilled "dev" 체크포인트 기반 고화질 옵션을 추가.
날씨 앱에는 Windy 제트기류/태풍 마커, 게이트웨이에는 wol-gate 웨이크 로그
조회 API도 함께 반영.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-08-05 17:21:36 +09:00
co-authored by Claude Sonnet 5
parent 675ee75de5
commit 6014e51ba3
7 changed files with 566 additions and 329 deletions
+274 -310
View File
@@ -82,231 +82,206 @@ function runVenvPython(script: string, timeoutMs: number): Promise<any> {
}
// ---------------------------------------------------------------------------
// image_generate — local SDXL text-to-image
// ---------------------------------------------------------------------------
const SDXL_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch
from diffusers import StableDiffusionXLPipeline
// ComfyUI backend — replaces the old per-request venv-script spawning for
// image_generate and image_style_transform (2026-08-05). ComfyUI runs as a
// persistent systemd --user service (~/.config/systemd/user/comfyui.service,
// GPU1, port 8188) with models pre-loaded in models/checkpoints|diffusion_models|
// clip|vae|pulid|insightface|facexlib under /home/kim/comfyui. See
// project_comfyui_evaluation / project_windy_jetstream_feature memory for how
// these were chosen and benchmarked — FLUX.1-schnell (fp8) beat the old 4-bit
// bitsandbytes path 2-3x on the same GPU; SDXL was a wash so it moved over too
// for one consistent backend. video_generate (LTX/CogVideoX) is untouched —
// still runs through the venv-script path below, no ComfyUI workflow for those.
const COMFY_URL = 'http://127.0.0.1:8188';
const COMFY_DIR = '/home/kim/comfyui';
const COMFY_OUTPUT_DIR = path.join(COMFY_DIR, 'output');
const COMFY_INPUT_DIR = path.join(COMFY_DIR, 'input');
pipe = StableDiffusionXLPipeline.from_pretrained(
"stabilityai/stable-diffusion-xl-base-1.0",
torch_dtype=torch.float16, variant="fp16", use_safetensors=True,
)
pipe = pipe.to("cuda")
pipe.enable_vae_slicing()
function randomSeed(): number {
return Math.floor(Math.random() * 0xFFFFFFFF);
}
kwargs = dict(
prompt=${JSON.stringify(p.prompt)},
negative_prompt=${JSON.stringify(p.negative_prompt || '')} or None,
width=int(${p.width}), height=int(${p.height}),
num_inference_steps=int(${p.steps}),
guidance_scale=float(${p.guidance_scale}),
)
${p.seed != null ? `kwargs["generator"] = torch.Generator("cuda").manual_seed(int(${p.seed}))` : ''}
async function comfySubmit(promptGraph: Record<string, any>): Promise<string> {
const res = await fetch(`${COMFY_URL}/prompt`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ prompt: promptGraph }),
});
const data: any = await res.json().catch(() => ({}));
if (data?.node_errors && Object.keys(data.node_errors).length) {
throw new Error('ComfyUI rejected the workflow: ' + JSON.stringify(data.node_errors));
}
if (!data?.prompt_id) throw new Error(data?.error ? JSON.stringify(data.error) : 'ComfyUI did not return a prompt_id');
return data.prompt_id;
}
image = pipe(**kwargs).images[0]
dst = ${JSON.stringify(p.dst)}
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
image.save(dst)
print("###RESULT###" + json.dumps({
"output": dst, "width": image.width, "height": image.height,
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
}))
except Exception as e:
import traceback
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
// No websocket/progress push used here (keeps this dependency-free) — just poll
// /history, same as the manual testing that validated every workflow below.
async function comfyPollResult(promptId: string, timeoutMs: number): Promise<{ filename: string; subfolder: string }> {
const start = Date.now();
while (Date.now() - start < timeoutMs) {
const res = await fetch(`${COMFY_URL}/history/${promptId}`);
const data: any = await res.json().catch(() => ({}));
const entry = data?.[promptId];
if (entry) {
if (entry.status?.status_str === 'error') {
const errMsg = entry.status?.messages?.find((m: any) => m[0] === 'execution_error')?.[1]?.exception_message;
throw new Error(errMsg || 'ComfyUI execution failed');
}
for (const nodeOut of Object.values<any>(entry.outputs || {})) {
if (nodeOut?.images?.length) return { filename: nodeOut.images[0].filename, subfolder: nodeOut.images[0].subfolder || '' };
}
}
await new Promise((r) => setTimeout(r, 2000));
}
throw new Error('Timed out waiting for ComfyUI generation');
}
// ---------------------------------------------------------------------------
// image_generate (quality: 'high') — local FLUX.1-schnell, 4-bit quantized
// ---------------------------------------------------------------------------
// Community mirror of the (gated) official repo — same Apache-2.0 weights,
// just rehosted without the HF license-gate. Needed because neither FLUX.1-dev
// nor -schnell can be pulled from black-forest-labs/* without an HF token tied
// to an account that has clicked through the license on huggingface.co.
const FLUX_MODEL_ID = 'Niansuh/FLUX.1-schnell';
const FLUX_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch, shutil
from diffusers import FluxPipeline, FluxTransformer2DModel, BitsAndBytesConfig as DBnBConfig
from transformers import T5EncoderModel, BitsAndBytesConfig as TBnBConfig
from huggingface_hub import snapshot_download
async function comfyGenerateImage(
promptGraph: Record<string, any>,
dst: string,
width: number,
height: number,
timeoutMs: number,
): Promise<{ output: string; width: number; height: number } | { error: string }> {
try {
const promptId = await comfySubmit(promptGraph);
const { filename, subfolder } = await comfyPollResult(promptId, timeoutMs);
const srcFile = path.join(COMFY_OUTPUT_DIR, subfolder, filename);
fs.mkdirSync(path.dirname(dst), { recursive: true });
fs.copyFileSync(srcFile, dst);
return { output: dst, width, height };
} catch (e: any) {
return { error: e?.message || String(e) };
}
}
MODEL_ID = "${FLUX_MODEL_ID}"
function sdxlWorkflow(p: { prompt: string; negative_prompt: string; width: number; height: number; steps: number; guidance_scale: number; seed: number }) {
return {
'3': { class_type: 'KSampler', inputs: { cfg: p.guidance_scale, denoise: 1.0, latent_image: ['5', 0], model: ['4', 0], negative: ['7', 0], positive: ['6', 0], sampler_name: 'euler', scheduler: 'normal', seed: p.seed, steps: p.steps } },
'4': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: 'sd_xl_base_1.0.safetensors' } },
'5': { class_type: 'EmptyLatentImage', inputs: { batch_size: 1, height: p.height, width: p.width } },
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['4', 1], text: p.prompt } },
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['4', 1], text: p.negative_prompt || '' } },
'8': { class_type: 'VAEDecode', inputs: { samples: ['3', 0], vae: ['4', 2] } },
'9': { class_type: 'SaveImage', inputs: { filename_prefix: 'sdxl_api', images: ['8', 0] } },
};
}
# This mirror ships scheduler/config.json instead of the scheduler_config.json
# filename diffusers expects — patch it once per cache (idempotent).
snap_dir = snapshot_download(MODEL_ID, allow_patterns=["scheduler/config.json"])
sched_cfg = os.path.join(snap_dir, "scheduler", "scheduler_config.json")
if not os.path.exists(sched_cfg):
shutil.copy(os.path.join(snap_dir, "scheduler", "config.json"), sched_cfg)
function fluxSchnellWorkflow(p: { prompt: string; width: number; height: number; steps: number; seed: number }) {
return {
'1': { class_type: 'UNETLoader', inputs: { unet_name: 'flux1-schnell-fp8.safetensors', weight_dtype: 'default' } },
'2': { class_type: 'DualCLIPLoader', inputs: { clip_name1: 'clip_l.safetensors', clip_name2: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'flux' } },
'3': { class_type: 'VAELoader', inputs: { vae_name: 'ae.safetensors' } },
'4': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.prompt } },
'5': { class_type: 'EmptySD3LatentImage', inputs: { batch_size: 1, height: p.height, width: p.width } },
'6': { class_type: 'KSampler', inputs: { cfg: 1.0, denoise: 1.0, latent_image: ['5', 0], model: ['1', 0], negative: ['4', 0], positive: ['4', 0], sampler_name: 'euler', scheduler: 'simple', seed: p.seed, steps: p.steps } },
'7': { class_type: 'VAEDecode', inputs: { samples: ['6', 0], vae: ['3', 0] } },
'8': { class_type: 'SaveImage', inputs: { filename_prefix: 'flux_api', images: ['7', 0] } },
};
}
transformer_4bit = FluxTransformer2DModel.from_pretrained(
MODEL_ID, subfolder="transformer",
quantization_config=DBnBConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16),
torch_dtype=torch.bfloat16,
)
text_encoder_2_4bit = T5EncoderModel.from_pretrained(
MODEL_ID, subfolder="text_encoder_2",
quantization_config=TBnBConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16),
torch_dtype=torch.bfloat16,
)
pipe = FluxPipeline.from_pretrained(
MODEL_ID,
transformer=transformer_4bit,
text_encoder_2=text_encoder_2_4bit,
torch_dtype=torch.bfloat16,
)
pipe.enable_model_cpu_offload()
// FLUX.1-dev + PuLID (identity-locked img2img). Unlike the txt2img workflows
// above, guidance=3.5/denoise~0.45 (the old SDXL-tuned defaults) produced almost
// no visible style change at all when this was benchmarked — FLUX's flow-matching
// sampler needs guidance~8 and denoise~0.85 before the caricature prompt actually
// overrides the source photo. See IMG2IMG_STYLE_PRESETS below for the tuned values.
function fluxPulidStyleWorkflow(p: { imageFilename: string; prompt: string; negative_prompt: string; denoise: number; guidance: number; steps: number; seed: number; pulid_weight: number }) {
const graph: Record<string, any> = {
'1': { class_type: 'UNETLoader', inputs: { unet_name: 'flux1-dev-fp8-e4m3fn.safetensors', weight_dtype: 'default' } },
'2': { class_type: 'DualCLIPLoader', inputs: { clip_name1: 'clip_l.safetensors', clip_name2: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'flux' } },
'3': { class_type: 'VAELoader', inputs: { vae_name: 'ae.safetensors' } },
'4': { class_type: 'LoadImage', inputs: { image: p.imageFilename } },
'4b': { class_type: 'ImageScaleToTotalPixels', inputs: { image: ['4', 0], upscale_method: 'lanczos', megapixels: 1.0, resolution_steps: 16 } },
'5': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.prompt } },
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.negative_prompt || '' } },
'7': { class_type: 'FluxGuidance', inputs: { conditioning: ['5', 0], guidance: p.guidance } },
'12': { class_type: 'VAEEncode', inputs: { pixels: ['4b', 0], vae: ['3', 0] } },
'13': { class_type: 'KSampler', inputs: { model: ['1', 0], positive: ['7', 0], negative: ['6', 0], latent_image: ['12', 0], seed: p.seed, steps: p.steps, cfg: 1.0, sampler_name: 'euler', scheduler: 'simple', denoise: p.denoise } },
'14': { class_type: 'VAEDecode', inputs: { samples: ['13', 0], vae: ['3', 0] } },
'15': { class_type: 'SaveImage', inputs: { filename_prefix: 'flux_pulid_api', images: ['14', 0] } },
};
// face_lock=false (or no face found upstream): skip PuLID nodes entirely rather
// than including a disconnected node — ComfyUI validates the graph as a DAG, so
// a "no-op" node with missing required inputs would just fail to queue.
if (p.pulid_weight > 0) {
graph['8'] = { class_type: 'PulidFluxModelLoader', inputs: { pulid_file: 'pulid_flux_v0.9.1.safetensors' } };
graph['9'] = { class_type: 'PulidFluxInsightFaceLoader', inputs: { provider: 'CUDA' } };
graph['10'] = { class_type: 'PulidFluxEvaClipLoader', inputs: {} };
graph['11'] = { class_type: 'ApplyPulidFlux', inputs: { model: ['1', 0], pulid_flux: ['8', 0], eva_clip: ['10', 0], face_analysis: ['9', 0], image: ['4b', 0], weight: p.pulid_weight, start_at: 0.0, end_at: 1.0 } };
graph['13'].inputs.model = ['11', 0];
}
return graph;
}
kwargs = dict(
prompt=${JSON.stringify(p.prompt)},
guidance_scale=0.0,
num_inference_steps=int(${p.steps}),
max_sequence_length=256,
width=int(${p.width}), height=int(${p.height}),
)
${p.seed != null ? `kwargs["generator"] = torch.Generator("cpu").manual_seed(int(${p.seed}))` : ''}
// LTX-Video text-to-video, ported to ComfyUI 2026-08-05 (see project_comfyui_evaluation
// memory) — the old venv-script path left GPU1 fully free between requests, but once
// ComfyUI started staying resident there for images, the two independent processes
// started fighting over the same VRAM (one real request took >200s and had to be
// killed). Moving LTX into ComfyUI too puts all of GPU1 under one memory manager.
// Node graph copied from ComfyUI's own bundled ltxv_text_to_video.json template —
// SamplerCustom+LTXVScheduler+KSamplerSelect, not plain KSampler, is how LTX is meant
// to be driven. The T5-XXL text encoder is the same file already downloaded for FLUX.
function ltxVideoWorkflow(p: { prompt: string; negative_prompt: string; width: number; height: number; length: number; fps: number; steps: number; guidance_scale: number; seed: number; ckpt_name?: string }) {
return {
'38': { class_type: 'CLIPLoader', inputs: { clip_name: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'ltxv', device: 'default' } },
'44': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: p.ckpt_name || 'ltxv-2b-0.9.8-distilled-fp8.safetensors' } },
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.prompt } },
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.negative_prompt } },
'70': { class_type: 'EmptyLTXVLatentVideo', inputs: { width: p.width, height: p.height, length: p.length, batch_size: 1 } },
'69': { class_type: 'LTXVConditioning', inputs: { positive: ['6', 0], negative: ['7', 0], frame_rate: p.fps } },
'71': { class_type: 'LTXVScheduler', inputs: { steps: p.steps, max_shift: 2.05, base_shift: 0.95, stretch: true, terminal: 0.1, latent: ['70', 0] } },
'73': { class_type: 'KSamplerSelect', inputs: { sampler_name: 'euler' } },
'72': { class_type: 'SamplerCustom', inputs: { model: ['44', 0], add_noise: true, noise_seed: p.seed, cfg: p.guidance_scale, positive: ['69', 0], negative: ['69', 1], sampler: ['73', 0], sigmas: ['71', 0], latent_image: ['70', 0] } },
'8': { class_type: 'VAEDecode', inputs: { samples: ['72', 0], vae: ['44', 2] } },
'78': { class_type: 'CreateVideo', inputs: { images: ['8', 0], fps: p.fps } },
'79': { class_type: 'SaveVideo', inputs: { video: ['78', 0], filename_prefix: 'ltx_api', format: 'mp4', codec: 'h264' } },
};
}
image = pipe(**kwargs).images[0]
dst = ${JSON.stringify(p.dst)}
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
image.save(dst)
print("###RESULT###" + json.dumps({
"output": dst, "width": image.width, "height": image.height,
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
}))
except Exception as e:
import traceback
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
// ---------------------------------------------------------------------------
// image_style_transform — local SDXL img2img (cartoon/watercolor only)
// ---------------------------------------------------------------------------
// image_edit's stylize op (AnimeGANv2 + OpenCV) already covers anime/sketch/bw
// well — those are identity-preserving and near-instant. But its "cartoon" and
// "watercolor" presets are mechanical edge/color filters that read as a mild
// photo filter rather than real artistic reinterpretation. SDXL img2img trades
// speed and some facial-identity fidelity for a genuinely more painterly result
// on those two styles specifically.
const IMG2IMG_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch
from diffusers import StableDiffusionXLImg2ImgPipeline
from PIL import Image, ImageOps
pipe = StableDiffusionXLImg2ImgPipeline.from_pretrained(
"stabilityai/stable-diffusion-xl-base-1.0",
torch_dtype=torch.float16, variant="fp16", use_safetensors=True,
)
pipe = pipe.to("cuda")
pipe.enable_vae_slicing()
init_image = ImageOps.exif_transpose(Image.open(${JSON.stringify(p.src)})).convert("RGB")
max_side = int(${p.max_side})
w, h = init_image.size
scale = min(1.0, max_side / max(w, h))
w, h = max(8, int(w * scale) // 8 * 8), max(8, int(h * scale) // 8 * 8)
init_image = init_image.resize((w, h))
# Face-ID lock (IP-Adapter-FaceID/SDXL): without this, img2img alone lets the
# style prompt pull the face toward SDXL's own learned priors — this is what
# caused the watercolor identity-drift problem. Extracting a face embedding and
# conditioning generation on it keeps the result recognizably the same person.
# Falls back to plain img2img if no face is detected (e.g. non-portrait photo).
face_embeds = None
if ${p.face_lock ? 'True' : 'False'}:
try:
import cv2
from insightface.app import FaceAnalysis
_face_app = FaceAnalysis(name="buffalo_l", providers=["CPUExecutionProvider"])
_face_app.prepare(ctx_id=0, det_size=(640, 640))
_cv_img = cv2.imread(${JSON.stringify(p.src)})
_faces = _face_app.get(_cv_img)
if _faces:
_emb = torch.from_numpy(_faces[0].normed_embedding).unsqueeze(0).unsqueeze(0)
face_embeds = torch.cat([torch.zeros_like(_emb), _emb], dim=0).to(dtype=torch.float16, device="cuda")
except Exception:
face_embeds = None
kwargs = dict(
prompt=${JSON.stringify(p.prompt)},
negative_prompt=${JSON.stringify(p.negative_prompt)},
image=init_image,
strength=float(${p.strength}),
num_inference_steps=int(${p.steps}),
guidance_scale=float(${p.guidance_scale}),
)
${p.seed != null ? `kwargs["generator"] = torch.Generator("cuda").manual_seed(int(${p.seed}))` : ''}
if face_embeds is not None:
pipe.load_ip_adapter("h94/IP-Adapter-FaceID", subfolder=None, weight_name="ip-adapter-faceid_sdxl.bin", image_encoder_folder=None)
pipe.set_ip_adapter_scale(0.8)
kwargs["ip_adapter_image_embeds"] = [face_embeds]
image = pipe(**kwargs).images[0]
dst = ${JSON.stringify(p.dst)}
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
image.save(dst)
print("###RESULT###" + json.dumps({
"output": dst, "width": image.width, "height": image.height,
"face_locked": face_embeds is not None,
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
}))
except Exception as e:
import traceback
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
// Same submit/poll plumbing as comfyGenerateImage but the SaveVideo node's output file
// isn't a PNG — copy verbatim rather than assuming an image extension.
async function comfyGenerateVideo(promptGraph: Record<string, any>, dst: string, timeoutMs: number): Promise<{ output: string } | { error: string }> {
try {
const promptId = await comfySubmit(promptGraph);
const { filename, subfolder } = await comfyPollResult(promptId, timeoutMs);
const srcFile = path.join(COMFY_OUTPUT_DIR, subfolder, filename);
fs.mkdirSync(path.dirname(dst), { recursive: true });
fs.copyFileSync(srcFile, dst);
return { output: dst };
} catch (e: any) {
return { error: e?.message || String(e) };
}
}
const IMG2IMG_STYLE_PRESETS: Record<string, { prompt: string; negative: string; strength: number; guidance_scale: number }> = {
cartoon: {
// strength was 0.6 — looked like a genuine caricature but the pose/expression
// ("인상") drifted too far from the source photo even with face-ID identity
// locked, since face-ID only conditions identity, not expression/composition.
// 0.45 keeps the photo's actual expression and framing intact while still
// applying visible cel-shaded/cartoon coloring and clean outlines.
// Tuned on FLUX.1-dev + PuLID (2026-08-05) — see fluxPulidStyleWorkflow comment.
// The old SDXL-era numbers (strength 0.45 / guidance 7.0) are no longer used.
prompt: 'caricature portrait illustration, exaggerated facial features, bold clean outlines, vibrant flat colors, humorous comic art style, digital illustration',
negative: 'photorealistic, blurry, deformed hands, extra limbs, low quality, watermark, text',
strength: 0.45,
guidance_scale: 7.0,
strength: 0.85,
guidance_scale: 8.0,
},
// "watercolor" was tried here too but dropped: the "watercolor painting portrait"
// prompt pulled SDXL toward its own learned face priors much harder than cartoon
// does — even down to strength 0.25-0.3 it reliably drifted the subject toward a
// different-looking (often different-gender-presenting) face, with the watercolor
// effect barely visible at the strengths low enough to keep identity intact.
// image_edit's classic OpenCV watercolor filter (mild but identity-safe) is used
// instead — see EDIT_STYLE_MAP in gateway/routes/imagegen.ts.
//
// "sketch" was tried too and also dropped: going from a color photo to monochrome
// graphite linework needed strength/guidance high enough (0.8/9.0) that, with no
// seed pinned, run-to-run variance sometimes drifted the face into a different-
// looking person (same failure shape as watercolor). Backing off to 0.6/8.0 for
// more consistent identity made the result look worse than image_edit's classic
// sketch filter — flat/muddy rather than either a clean sketch or a good likeness.
// Reverted to image_edit's OpenCV sketch filter (see EDIT_STYLE_MAP).
// watercolor/sketch: still dropped for the same identity-drift reasons as before
// (see git history) — image_edit's OpenCV filters remain the fallback for those.
};
export const imageStyleTransformTool = {
name: 'image_style_transform',
description: [
'Transform an existing photo (e.g. a portrait) into a new artistic style using local SDXL img2img with IP-Adapter-FaceID identity locking (runs on-machine GPU, no external API).',
'Supports style="cartoon" (caricature/comic illustration) — more artistically interpretive than image_edit\'s classic OpenCV cartoon filter, at the cost of speed (~15-40s vs ~2-3s). When a face is detected in the source photo, a face embedding (insightface/buffalo_l) conditions the generation via IP-Adapter-FaceID so the result stays recognizably the same person even under a strong style prompt — falls back to plain img2img if no face is found.',
'For anime/sketch/watercolor/black-and-white, use image_edit\'s stylize/filter operations instead — those are faster (both "watercolor" and "sketch" SDXL presets were tried and dropped: the style jump needed enough strength that, with no seed pinned, results sometimes drifted the face into a different-looking person).',
'Transform an existing photo (e.g. a portrait) into a new artistic style using local FLUX.1-dev img2img with PuLID identity locking, via ComfyUI (runs on-machine GPU, no external API).',
'Supports style="cartoon" (caricature/comic illustration) — more artistically interpretive than image_edit\'s classic OpenCV cartoon filter, at the cost of speed (~60-80s vs ~2-3s). A face embedding (insightface + EVA-CLIP) conditions the generation via PuLID so the result stays recognizably the same person even under a strong style prompt.',
'For anime/sketch/watercolor/black-and-white, use image_edit\'s stylize/filter operations instead — those are faster (both "watercolor" and "sketch" presets were tried and dropped: the style jump needed enough strength that results sometimes drifted the face into a different-looking person).',
'Returns the transformed image inline in the chat.',
].join('\n'),
schema: {
image: 'Path to the source image to transform (required)',
style: '"cartoon" (caricature/comic illustration) — currently the only supported style',
face_lock: 'Whether to lock facial identity via IP-Adapter-FaceID when a face is detected (optional, default true). Turn off to compare plain img2img — face-ID conditioning sometimes reads as slightly uncanny/over-smoothed; plain img2img gives looser but more natural-looking results.',
strength: 'How strongly to restyle, 0.05-1.0 (optional; default 0.45). Lower preserves the original photo\'s pose/expression more, higher restyles more aggressively but can drift away from them.',
steps: 'Denoising steps (default 40, range 15-50)',
guidance_scale: 'How closely to follow the style prompt (default 7.0)',
face_lock: 'Whether to lock facial identity via PuLID (optional, default true). Turn off to compare plain img2img — face-lock conditioning sometimes reads as slightly uncanny/over-smoothed; plain img2img gives looser but more natural-looking results.',
strength: 'How strongly to restyle, 0.05-1.0 (optional; default 0.85 — FLUX\'s flow-matching sampler needs much higher values than the old SDXL default to produce a visible style change at all). Lower preserves the original photo\'s pose/expression more, higher restyles more aggressively but can drift away from them.',
steps: 'Denoising steps (default 20, range 10-40)',
guidance_scale: 'How closely to follow the style prompt (default 8.0)',
seed: 'Random seed for reproducibility (optional)',
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
},
@@ -344,34 +319,40 @@ export const imageStyleTransformTool = {
outPath = path.resolve(workspacePath, outPath);
}
const params = {
src: srcPath,
const faceLock = args?.face_lock ?? true;
const denoise = Math.min(1, Math.max(0.05, args?.strength ?? preset.strength));
const guidance = toFiniteNumber(args?.guidance_scale, preset.guidance_scale);
const steps = Math.min(40, Math.max(10, args?.steps ?? 20));
const seed = args?.seed != null ? toFiniteNumber(args.seed, 0) : randomSeed();
// ComfyUI reads source images from its own input/ dir — copy in under a unique
// name (same host, so a filesystem copy, no HTTP upload round-trip needed).
fs.mkdirSync(COMFY_INPUT_DIR, { recursive: true });
const inputFilename = `style_src_${Date.now()}${path.extname(srcPath) || '.png'}`;
fs.copyFileSync(srcPath, path.join(COMFY_INPUT_DIR, inputFilename));
const graph = fluxPulidStyleWorkflow({
imageFilename: inputFilename,
prompt: preset.prompt,
negative_prompt: preset.negative,
strength: Math.min(1, Math.max(0.05, args?.strength ?? preset.strength)),
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
guidance_scale: toFiniteNumber(args?.guidance_scale, preset.guidance_scale),
seed: args?.seed != null ? toFiniteNumber(args.seed, 0) : undefined,
face_lock: args?.face_lock ?? true,
max_side: 1024,
dst: outPath,
};
// 600s (not 180s like the other scripts): the face-ID adapter weights (~1.7GB)
// download from HF Hub on first run and can take several minutes.
const result = await runVenvPython(IMG2IMG_SCRIPT(params), 600_000);
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
denoise, guidance, steps, seed,
pulid_weight: faceLock ? 1.0 : 0,
});
// 600s (not the 180s the txt2img workflows use): first request after a
// service restart also has to load PuLID/EVA-CLIP/insightface on top of FLUX.1-dev.
const result = await comfyGenerateImage(graph, outPath, 0, 0, 600_000);
if ('error' in result || !result.output) {
return { success: false, error: 'error' in result ? result.error : 'Generator returned no output file' };
}
return {
success: true,
stdout: [
`Style: ${style}`,
`Transformed: ${result.width} × ${result.height} px`,
`Style: ${style}${faceLock ? ' (face-locked via PuLID)' : ''}`,
'',
buildImageMarkdown(result.output, workspacePath),
].join('\n'),
data: { ...result, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
data: { output: result.output, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
};
},
};
@@ -379,9 +360,9 @@ export const imageStyleTransformTool = {
export const imageGenerateTool = {
name: 'image_generate',
description: [
'Generate an image from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
'quality="fast" (default): SDXL, ~10-20 seconds, default resolution 1024×1024, good for casual/quick illustrations.',
'quality="high": FLUX.1-schnell (4-bit quantized), ~40-60 seconds total (model load + generation), noticeably more photorealistic detail and prompt accuracy. Use only when the user explicitly asks for higher quality/detail/photorealism, or for a "고품질" request — otherwise default to fast.',
'Generate an image from a text prompt using a local diffusion model via ComfyUI (runs on-machine GPU, no external API).',
'quality="fast" (default): SDXL, ~15-20 seconds, default resolution 1024×1024, good for casual/quick illustrations.',
'quality="high": FLUX.1-schnell (fp8), ~15-25 seconds, noticeably more photorealistic detail and prompt accuracy — and no longer much slower than fast mode (moved off the old 4-bit bitsandbytes path, which needed ~40-60s), so lean toward "high" more readily than before. Still default to fast for quick/casual requests.',
'Returns the generated image inline in the chat.',
].join('\n'),
schema: {
@@ -433,30 +414,20 @@ export const imageGenerateTool = {
const width = Math.round((args?.width ?? 1024) / 16) * 16;
const height = Math.round((args?.height ?? 1024) / 16) * 16;
let result: any;
if (isHighQuality) {
const params = {
prompt: translatedPrompt || prompt,
width, height,
steps: Math.min(8, Math.max(1, args?.steps ?? 4)),
seed: args?.seed != null ? toFiniteNumber(args.seed, 0) : undefined,
dst: outPath,
};
result = await runVenvPython(FLUX_SCRIPT(params), 180_000);
} else {
const params = {
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height,
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
guidance_scale: toFiniteNumber(args?.guidance_scale, 7.0),
seed: args?.seed != null ? toFiniteNumber(args.seed, 0) : undefined,
dst: outPath,
};
result = await runVenvPython(SDXL_SCRIPT(params), 180_000);
}
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
const seed = args?.seed != null ? toFiniteNumber(args.seed, 0) : randomSeed();
const graph = isHighQuality
? fluxSchnellWorkflow({ prompt: translatedPrompt || prompt, width, height, steps: Math.min(8, Math.max(1, args?.steps ?? 4)), seed })
: sdxlWorkflow({
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height,
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
guidance_scale: toFiniteNumber(args?.guidance_scale, 7.0),
seed,
});
const result = await comfyGenerateImage(graph, outPath, width, height, 180_000);
if ('error' in result || !result.output) {
return { success: false, error: 'error' in result ? result.error : 'Generator returned no output file' };
}
return {
@@ -473,42 +444,9 @@ export const imageGenerateTool = {
};
// ---------------------------------------------------------------------------
// video_generate — local LTX-Video text-to-video
// video_generate — LTX-Video moved to ComfyUI (see ltxVideoWorkflow above);
// CogVideoX-2B still runs the old venv-script path below.
// ---------------------------------------------------------------------------
const LTX_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch
from diffusers import LTXPipeline
from diffusers.utils import export_to_video
pipe = LTXPipeline.from_pretrained("Lightricks/LTX-Video", torch_dtype=torch.bfloat16)
pipe.enable_model_cpu_offload()
video = pipe(
prompt=${JSON.stringify(p.prompt)},
negative_prompt=${JSON.stringify(p.negative_prompt)},
width=int(${p.width}), height=int(${p.height}),
num_frames=int(${p.num_frames}),
num_inference_steps=int(${p.steps}),
guidance_scale=float(${p.guidance_scale}),
).frames[0]
dst = ${JSON.stringify(p.dst)}
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
export_to_video(video, dst, fps=int(${p.fps}))
stat = os.stat(dst)
print("###RESULT###" + json.dumps({
"output": dst, "width": int(${p.width}), "height": int(${p.height}),
"num_frames": int(${p.num_frames}), "fps": int(${p.fps}),
"size_bytes": stat.st_size,
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
}))
except Exception as e:
import traceback
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
// CogVideoX-2B — added 2026-07-17 as a second option alongside LTX-Video for quality
// comparison. Barely fits: peak VRAM measured at ~11GB of the 12GB card even with
@@ -557,12 +495,12 @@ export const videoGenerateTool = {
name: 'video_generate',
description: [
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
'Two models available via the "model" param: "ltx" (default — fast, ~30-90s, more headroom for higher resolution/frame count) and "cogvideox" (THUDM CogVideoX-2B — slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try both and compare if unsure which fits the request.',
'Three models available via the "model" param: "ltx" (default — LTX-Video 2B distilled via ComfyUI, ~25-40s), "ltx_hq" (LTX-Video 2B non-distilled/"dev" checkpoint via ComfyUI — noticeably better motion coherence and detail, ~40-90s, VRAM headroom is tight at ~10GB/12GB so avoid stacking a large image_generate call at the same time), and "cogvideox" (THUDM CogVideoX-2B — still the old venv-script path, slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try a couple and compare if unsure which fits the request.',
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
].join('\n'),
schema: {
prompt: 'Text description of the video/scene to generate (English works best)',
model: '"ltx" (default, fast) or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
model: '"ltx" (default, fast, distilled), "ltx_hq" (LTX-Video non-distilled "dev" checkpoint, better quality, slower), or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
height: 'Video height in pixels, multiple of 32 (default 480)',
@@ -576,7 +514,7 @@ export const videoGenerateTool = {
type: 'object',
properties: {
prompt: { type: 'string' },
model: { type: 'string', enum: ['ltx', 'cogvideox'] },
model: { type: 'string', enum: ['ltx', 'ltx_hq', 'cogvideox'] },
negative_prompt: { type: 'string' },
width: { type: 'number' },
height: { type: 'number' },
@@ -593,13 +531,14 @@ export const videoGenerateTool = {
const prompt = String(args?.prompt || '').trim();
if (!prompt) return { success: false, error: 'prompt is required' };
const model = args?.model === 'cogvideox' ? 'cogvideox' : 'ltx';
const model = args?.model === 'cogvideox' ? 'cogvideox' : args?.model === 'ltx_hq' ? 'ltx_hq' : 'ltx';
const isCogVideoX = model === 'cogvideox';
const isHQ = model === 'ltx_hq';
const workspacePath = getWorkspacePath(args);
let outPath = String(args?.output || '').trim();
if (!outPath) {
outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : 'ltx'}_${Date.now()}.mp4`);
outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : isHQ ? 'ltx_hq' : 'ltx'}_${Date.now()}.mp4`);
} else if (!path.isAbsolute(outPath)) {
outPath = path.resolve(workspacePath, outPath);
}
@@ -610,41 +549,66 @@ export const videoGenerateTool = {
translatePromptToEnglish(negativePromptRaw),
]);
const params = {
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width: Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32,
height: Math.round((args?.height ?? 480) / 32) * 32,
num_frames: toFiniteNumber(args?.num_frames, isCogVideoX ? 49 : 65),
fps: toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24),
// CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50
// (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any
// reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it
// lower to keep typical runs under ~4 minutes; still overridable via the steps param.
steps: Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40))),
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0))),
dst: outPath,
};
const width = Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32;
const height = Math.round((args?.height ?? 480) / 32) * 32;
const fps = toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24);
const guidance_scale = Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0)));
// CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50
// (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any
// reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it
// lower to keep typical runs under ~4 minutes; still overridable via the steps param.
const steps = Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40)));
const script = isCogVideoX ? COGVIDEOX_SCRIPT(params) : LTX_SCRIPT(params);
const result = await runVenvPython(script, 600_000);
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
let num_frames: number;
let outputMeta: { output: string };
if (isCogVideoX) {
num_frames = toFiniteNumber(args?.num_frames, 49);
const params = {
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, num_frames, fps, steps, guidance_scale,
dst: outPath,
};
const result = await runVenvPython(COGVIDEOX_SCRIPT(params), 600_000);
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
}
outputMeta = { output: result.output };
} else {
// LTX-Video wants length as 8k+1 (its temporal VAE downsamples by 8) — snap
// whatever was requested to the nearest valid value instead of erroring.
const requestedFrames = toFiniteNumber(args?.num_frames, 65);
num_frames = Math.max(9, Math.round((requestedFrames - 1) / 8) * 8 + 1);
const graph = ltxVideoWorkflow({
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, length: num_frames, fps, steps, guidance_scale,
seed: randomSeed(),
ckpt_name: isHQ ? 'ltxv-2b-0.9.6-dev-04-25.safetensors' : undefined,
});
// The non-distilled "dev" checkpoint is ~1.4x the weight size of the distilled one and
// runs full CFG (two forward passes/step instead of one), so it's meaningfully slower —
// give it more headroom than the regular ltx path's 300s.
const result = await comfyGenerateVideo(graph, outPath, isHQ ? 420_000 : 300_000);
if ('error' in result) {
return { success: false, error: result.error };
}
outputMeta = result;
}
const relOut = path.relative(workspacePath, result.output).replace(/\\/g, '/');
const sizeMB = (result.size_bytes / 1024 / 1024).toFixed(2);
const durationSec = (result.num_frames / result.fps).toFixed(1);
const relOut = path.relative(workspacePath, outputMeta.output).replace(/\\/g, '/');
const sizeMB = (fs.statSync(outputMeta.output).size / 1024 / 1024).toFixed(2);
const durationSec = (num_frames / fps).toFixed(1);
return {
success: true,
stdout: [
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
`Generated: ${result.width} × ${result.height} px | ${durationSec}s (${result.num_frames}f @ ${result.fps}fps) | ${sizeMB} MB`,
`Generated: ${width} × ${height} px | ${durationSec}s (${num_frames}f @ ${fps}fps) | ${sizeMB} MB`,
'',
`[${path.basename(result.output)}](/api/files/${relOut})`,
`[${path.basename(outputMeta.output)}](/api/files/${relOut})`,
].filter((line): line is string => line !== null).join('\n'),
data: { ...result, rel_path: relOut },
data: { output: outputMeta.output, width, height, num_frames, fps, rel_path: relOut },
};
},
};