Files
homeclaw/src/tools/imagegen.ts
T
kimandClaude Sonnet 5 02247e4ee3 feat: 스튜디오 앱 사진→동영상(LTX Image-to-Video)
studio-app.html에 시작 이미지 업로드/제거 UI + strength 슬라이더 추가,
imagegen.ts에 ltxImageToVideoWorkflow (ComfyUI 공식 ltxv_image_to_video.json
템플릿 기반 LoadImage→LTXVImgToVideo→LTXVConditioning 그래프) 추가.
실제 생성 테스트로 프레임0(원본 이미지 일치)·프레임15(모션 발생) 확인 완료된
작업을 커밋만 안 하고 남겨뒀던 것.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-07 13:13:58 +09:00

670 lines
38 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { spawn } from 'child_process';
import path from 'path';
import fs from 'fs';
import { ToolResult } from '../types.js';
import { getWorkspacePath } from '../config/paths.js';
import { buildImageMarkdown, toFiniteNumber, isPathInsideDir } from './image.js';
import { getOllamaConfig } from './web.js';
// SDXL/LTX-Video's text encoders (CLIP/T5) are trained overwhelmingly on English
// captions, so non-English prompts (e.g. Korean) produce poor prompt adherence.
// Route non-ASCII prompts through the configured Ollama chat model for a quick
// English translation before handing them to the diffusion pipeline.
async function translatePromptToEnglish(text: string): Promise<string | null> {
if (!text || !/[^\x00-\x7F]/.test(text)) return null;
try {
const { endpoint, model } = getOllamaConfig();
const res = await fetch(`${endpoint}/api/chat`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
model,
messages: [{
role: 'user',
content: `Translate the following image/video generation prompt into natural, descriptive English. Output ONLY the translated English text — no quotes, no explanation:\n\n${text}`,
}],
stream: false,
}),
signal: AbortSignal.timeout(30_000),
});
if (!res.ok) return null;
const data: any = await res.json();
const translated = String(data.message?.content || '').trim();
return translated || null;
} catch {
return null;
}
}
// Local diffusion models (SDXL, LTX-Video) run in a dedicated venv with their
// own torch/diffusers stack, pinned to the second GPU (04:00.0 — kept free of
// the voice engine that permanently resides on GPU0). See
// /srv/homeclaw/.smallclaw/imagegen-venv.
const VENV_PYTHON = '/srv/homeclaw/.smallclaw/imagegen-venv/bin/python3';
const HF_HOME = '/srv/homeclaw/.smallclaw/imagegen-venv/hf-cache';
const GEN_GPU = '1';
function runVenvPython(script: string, timeoutMs: number): Promise<any> {
return new Promise((resolve) => {
const child = spawn(VENV_PYTHON, ['-c', script], {
timeout: timeoutMs,
// BNB_CUDA_VERSION: bitsandbytes (used by the FLUX high-quality path) ships no
// prebuilt binary for our CUDA 13.2 torch build yet — pin it to the newest
// available (13.0) binary, which is ABI-compatible. No-op for SDXL/LTX, which
// don't use bitsandbytes.
env: { ...process.env, HF_HOME, CUDA_VISIBLE_DEVICES: GEN_GPU, BNB_CUDA_VERSION: '130' },
});
let out = '';
let err = '';
child.stdout.on('data', (d: Buffer) => { out += d.toString('utf8'); });
child.stderr.on('data', (d: Buffer) => { err += d.toString('utf8'); });
child.on('close', (code: number | null, signal: string | null) => {
const marker = out.lastIndexOf('###RESULT###');
if (marker === -1) {
// Process exited (crashed, OOM-killed, or timed out) without ever printing a
// result marker — never silently fall through to a success-shaped {}, since
// that leaves result.output undefined and crashes the caller downstream.
resolve({
error: `Generator process exited without output (code=${code}, signal=${signal})`,
trace: (err || out).slice(-1500),
});
return;
}
const jsonPart = out.slice(marker + '###RESULT###'.length);
try {
resolve(JSON.parse(jsonPart.trim()));
} catch {
resolve({ error: 'Failed to parse generator output', raw: (jsonPart || err).slice(-1500) });
}
});
child.on('error', (e: Error) => resolve({ error: e.message }));
});
}
// ---------------------------------------------------------------------------
// ComfyUI backend — replaces the old per-request venv-script spawning for
// image_generate and image_style_transform (2026-08-05). ComfyUI runs as a
// persistent systemd --user service (~/.config/systemd/user/comfyui.service,
// GPU1, port 8188) with models pre-loaded in models/checkpoints|diffusion_models|
// clip|vae|pulid|insightface|facexlib under /home/kim/comfyui. See
// project_comfyui_evaluation / project_windy_jetstream_feature memory for how
// these were chosen and benchmarked — FLUX.1-schnell (fp8) beat the old 4-bit
// bitsandbytes path 2-3x on the same GPU; SDXL was a wash so it moved over too
// for one consistent backend. video_generate (LTX/CogVideoX) is untouched —
// still runs through the venv-script path below, no ComfyUI workflow for those.
const COMFY_URL = 'http://127.0.0.1:8188';
const COMFY_DIR = '/home/kim/comfyui';
const COMFY_OUTPUT_DIR = path.join(COMFY_DIR, 'output');
const COMFY_INPUT_DIR = path.join(COMFY_DIR, 'input');
function randomSeed(): number {
return Math.floor(Math.random() * 0xFFFFFFFF);
}
async function comfySubmit(promptGraph: Record<string, any>): Promise<string> {
const res = await fetch(`${COMFY_URL}/prompt`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ prompt: promptGraph }),
});
const data: any = await res.json().catch(() => ({}));
if (data?.node_errors && Object.keys(data.node_errors).length) {
throw new Error('ComfyUI rejected the workflow: ' + JSON.stringify(data.node_errors));
}
if (!data?.prompt_id) throw new Error(data?.error ? JSON.stringify(data.error) : 'ComfyUI did not return a prompt_id');
return data.prompt_id;
}
// No websocket/progress push used here (keeps this dependency-free) — just poll
// /history, same as the manual testing that validated every workflow below.
async function comfyPollResult(promptId: string, timeoutMs: number): Promise<{ filename: string; subfolder: string }> {
const start = Date.now();
while (Date.now() - start < timeoutMs) {
const res = await fetch(`${COMFY_URL}/history/${promptId}`);
const data: any = await res.json().catch(() => ({}));
const entry = data?.[promptId];
if (entry) {
if (entry.status?.status_str === 'error') {
const errMsg = entry.status?.messages?.find((m: any) => m[0] === 'execution_error')?.[1]?.exception_message;
throw new Error(errMsg || 'ComfyUI execution failed');
}
for (const nodeOut of Object.values<any>(entry.outputs || {})) {
if (nodeOut?.images?.length) return { filename: nodeOut.images[0].filename, subfolder: nodeOut.images[0].subfolder || '' };
}
}
await new Promise((r) => setTimeout(r, 2000));
}
throw new Error('Timed out waiting for ComfyUI generation');
}
async function comfyGenerateImage(
promptGraph: Record<string, any>,
dst: string,
width: number,
height: number,
timeoutMs: number,
): Promise<{ output: string; width: number; height: number } | { error: string }> {
try {
const promptId = await comfySubmit(promptGraph);
const { filename, subfolder } = await comfyPollResult(promptId, timeoutMs);
const srcFile = path.join(COMFY_OUTPUT_DIR, subfolder, filename);
fs.mkdirSync(path.dirname(dst), { recursive: true });
fs.copyFileSync(srcFile, dst);
return { output: dst, width, height };
} catch (e: any) {
return { error: e?.message || String(e) };
}
}
function sdxlWorkflow(p: { prompt: string; negative_prompt: string; width: number; height: number; steps: number; guidance_scale: number; seed: number }) {
return {
'3': { class_type: 'KSampler', inputs: { cfg: p.guidance_scale, denoise: 1.0, latent_image: ['5', 0], model: ['4', 0], negative: ['7', 0], positive: ['6', 0], sampler_name: 'euler', scheduler: 'normal', seed: p.seed, steps: p.steps } },
'4': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: 'sd_xl_base_1.0.safetensors' } },
'5': { class_type: 'EmptyLatentImage', inputs: { batch_size: 1, height: p.height, width: p.width } },
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['4', 1], text: p.prompt } },
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['4', 1], text: p.negative_prompt || '' } },
'8': { class_type: 'VAEDecode', inputs: { samples: ['3', 0], vae: ['4', 2] } },
'9': { class_type: 'SaveImage', inputs: { filename_prefix: 'sdxl_api', images: ['8', 0] } },
};
}
function fluxSchnellWorkflow(p: { prompt: string; width: number; height: number; steps: number; seed: number }) {
return {
'1': { class_type: 'UNETLoader', inputs: { unet_name: 'flux1-schnell-fp8.safetensors', weight_dtype: 'default' } },
'2': { class_type: 'DualCLIPLoader', inputs: { clip_name1: 'clip_l.safetensors', clip_name2: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'flux' } },
'3': { class_type: 'VAELoader', inputs: { vae_name: 'ae.safetensors' } },
'4': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.prompt } },
'5': { class_type: 'EmptySD3LatentImage', inputs: { batch_size: 1, height: p.height, width: p.width } },
'6': { class_type: 'KSampler', inputs: { cfg: 1.0, denoise: 1.0, latent_image: ['5', 0], model: ['1', 0], negative: ['4', 0], positive: ['4', 0], sampler_name: 'euler', scheduler: 'simple', seed: p.seed, steps: p.steps } },
'7': { class_type: 'VAEDecode', inputs: { samples: ['6', 0], vae: ['3', 0] } },
'8': { class_type: 'SaveImage', inputs: { filename_prefix: 'flux_api', images: ['7', 0] } },
};
}
// FLUX.1-dev + PuLID (identity-locked img2img). Unlike the txt2img workflows
// above, guidance=3.5/denoise~0.45 (the old SDXL-tuned defaults) produced almost
// no visible style change at all when this was benchmarked — FLUX's flow-matching
// sampler needs guidance~8 and denoise~0.85 before the caricature prompt actually
// overrides the source photo. See IMG2IMG_STYLE_PRESETS below for the tuned values.
function fluxPulidStyleWorkflow(p: { imageFilename: string; prompt: string; negative_prompt: string; denoise: number; guidance: number; steps: number; seed: number; pulid_weight: number }) {
const graph: Record<string, any> = {
'1': { class_type: 'UNETLoader', inputs: { unet_name: 'flux1-dev-fp8-e4m3fn.safetensors', weight_dtype: 'default' } },
'2': { class_type: 'DualCLIPLoader', inputs: { clip_name1: 'clip_l.safetensors', clip_name2: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'flux' } },
'3': { class_type: 'VAELoader', inputs: { vae_name: 'ae.safetensors' } },
'4': { class_type: 'LoadImage', inputs: { image: p.imageFilename } },
'4b': { class_type: 'ImageScaleToTotalPixels', inputs: { image: ['4', 0], upscale_method: 'lanczos', megapixels: 1.0, resolution_steps: 16 } },
'5': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.prompt } },
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.negative_prompt || '' } },
'7': { class_type: 'FluxGuidance', inputs: { conditioning: ['5', 0], guidance: p.guidance } },
'12': { class_type: 'VAEEncode', inputs: { pixels: ['4b', 0], vae: ['3', 0] } },
'13': { class_type: 'KSampler', inputs: { model: ['1', 0], positive: ['7', 0], negative: ['6', 0], latent_image: ['12', 0], seed: p.seed, steps: p.steps, cfg: 1.0, sampler_name: 'euler', scheduler: 'simple', denoise: p.denoise } },
'14': { class_type: 'VAEDecode', inputs: { samples: ['13', 0], vae: ['3', 0] } },
'15': { class_type: 'SaveImage', inputs: { filename_prefix: 'flux_pulid_api', images: ['14', 0] } },
};
// face_lock=false (or no face found upstream): skip PuLID nodes entirely rather
// than including a disconnected node — ComfyUI validates the graph as a DAG, so
// a "no-op" node with missing required inputs would just fail to queue.
if (p.pulid_weight > 0) {
graph['8'] = { class_type: 'PulidFluxModelLoader', inputs: { pulid_file: 'pulid_flux_v0.9.1.safetensors' } };
graph['9'] = { class_type: 'PulidFluxInsightFaceLoader', inputs: { provider: 'CUDA' } };
graph['10'] = { class_type: 'PulidFluxEvaClipLoader', inputs: {} };
graph['11'] = { class_type: 'ApplyPulidFlux', inputs: { model: ['1', 0], pulid_flux: ['8', 0], eva_clip: ['10', 0], face_analysis: ['9', 0], image: ['4b', 0], weight: p.pulid_weight, start_at: 0.0, end_at: 1.0 } };
graph['13'].inputs.model = ['11', 0];
}
return graph;
}
// LTX-Video text-to-video, ported to ComfyUI 2026-08-05 (see project_comfyui_evaluation
// memory) — the old venv-script path left GPU1 fully free between requests, but once
// ComfyUI started staying resident there for images, the two independent processes
// started fighting over the same VRAM (one real request took >200s and had to be
// killed). Moving LTX into ComfyUI too puts all of GPU1 under one memory manager.
// Node graph copied from ComfyUI's own bundled ltxv_text_to_video.json template —
// SamplerCustom+LTXVScheduler+KSamplerSelect, not plain KSampler, is how LTX is meant
// to be driven. The T5-XXL text encoder is the same file already downloaded for FLUX.
function ltxVideoWorkflow(p: { prompt: string; negative_prompt: string; width: number; height: number; length: number; fps: number; steps: number; guidance_scale: number; seed: number; ckpt_name?: string }) {
return {
'38': { class_type: 'CLIPLoader', inputs: { clip_name: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'ltxv', device: 'default' } },
'44': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: p.ckpt_name || 'ltxv-2b-0.9.8-distilled-fp8.safetensors' } },
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.prompt } },
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.negative_prompt } },
'70': { class_type: 'EmptyLTXVLatentVideo', inputs: { width: p.width, height: p.height, length: p.length, batch_size: 1 } },
'69': { class_type: 'LTXVConditioning', inputs: { positive: ['6', 0], negative: ['7', 0], frame_rate: p.fps } },
'71': { class_type: 'LTXVScheduler', inputs: { steps: p.steps, max_shift: 2.05, base_shift: 0.95, stretch: true, terminal: 0.1, latent: ['70', 0] } },
'73': { class_type: 'KSamplerSelect', inputs: { sampler_name: 'euler' } },
'72': { class_type: 'SamplerCustom', inputs: { model: ['44', 0], add_noise: true, noise_seed: p.seed, cfg: p.guidance_scale, positive: ['69', 0], negative: ['69', 1], sampler: ['73', 0], sigmas: ['71', 0], latent_image: ['70', 0] } },
'8': { class_type: 'VAEDecode', inputs: { samples: ['72', 0], vae: ['44', 2] } },
'78': { class_type: 'CreateVideo', inputs: { images: ['8', 0], fps: p.fps } },
'79': { class_type: 'SaveVideo', inputs: { video: ['78', 0], filename_prefix: 'ltx_api', format: 'mp4', codec: 'h264' } },
};
}
// Image-to-video variant — same LTX checkpoint/text-encoder as ltxVideoWorkflow above,
// but the source photo goes through LTXVImgToVideo *before* LTXVConditioning (wiring
// copied from ComfyUI's bundled ltxv_image_to_video.json template, node IDs kept
// matching that template for traceability). LTXVImgToVideo bakes the image into both
// the conditioning and the starting latent, replacing EmptyLTXVLatentVideo entirely —
// there's no plain "add an image on top of the txt2vid graph" path, the whole latent
// source changes.
function ltxImageToVideoWorkflow(p: { imageFilename: string; prompt: string; negative_prompt: string; width: number; height: number; length: number; fps: number; steps: number; guidance_scale: number; strength: number; seed: number; ckpt_name?: string }) {
return {
'38': { class_type: 'CLIPLoader', inputs: { clip_name: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'ltxv', device: 'default' } },
'44': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: p.ckpt_name || 'ltxv-2b-0.9.8-distilled-fp8.safetensors' } },
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.prompt } },
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.negative_prompt } },
'78': { class_type: 'LoadImage', inputs: { image: p.imageFilename } },
'77': { class_type: 'LTXVImgToVideo', inputs: { positive: ['6', 0], negative: ['7', 0], vae: ['44', 2], image: ['78', 0], width: p.width, height: p.height, length: p.length, batch_size: 1, strength: p.strength } },
'69': { class_type: 'LTXVConditioning', inputs: { positive: ['77', 0], negative: ['77', 1], frame_rate: p.fps } },
'71': { class_type: 'LTXVScheduler', inputs: { steps: p.steps, max_shift: 2.05, base_shift: 0.95, stretch: true, terminal: 0.1, latent: ['77', 2] } },
'73': { class_type: 'KSamplerSelect', inputs: { sampler_name: 'euler' } },
'72': { class_type: 'SamplerCustom', inputs: { model: ['44', 0], add_noise: true, noise_seed: p.seed, cfg: p.guidance_scale, positive: ['69', 0], negative: ['69', 1], sampler: ['73', 0], sigmas: ['71', 0], latent_image: ['77', 2] } },
'8': { class_type: 'VAEDecode', inputs: { samples: ['72', 0], vae: ['44', 2] } },
'80': { class_type: 'CreateVideo', inputs: { images: ['8', 0], fps: p.fps } },
'81': { class_type: 'SaveVideo', inputs: { video: ['80', 0], filename_prefix: 'ltx_i2v_api', format: 'mp4', codec: 'h264' } },
};
}
// Same submit/poll plumbing as comfyGenerateImage but the SaveVideo node's output file
// isn't a PNG — copy verbatim rather than assuming an image extension.
async function comfyGenerateVideo(promptGraph: Record<string, any>, dst: string, timeoutMs: number): Promise<{ output: string } | { error: string }> {
try {
const promptId = await comfySubmit(promptGraph);
const { filename, subfolder } = await comfyPollResult(promptId, timeoutMs);
const srcFile = path.join(COMFY_OUTPUT_DIR, subfolder, filename);
fs.mkdirSync(path.dirname(dst), { recursive: true });
fs.copyFileSync(srcFile, dst);
return { output: dst };
} catch (e: any) {
return { error: e?.message || String(e) };
}
}
const IMG2IMG_STYLE_PRESETS: Record<string, { prompt: string; negative: string; strength: number; guidance_scale: number }> = {
cartoon: {
// Tuned on FLUX.1-dev + PuLID (2026-08-05) — see fluxPulidStyleWorkflow comment.
// The old SDXL-era numbers (strength 0.45 / guidance 7.0) are no longer used.
prompt: 'caricature portrait illustration, exaggerated facial features, bold clean outlines, vibrant flat colors, humorous comic art style, digital illustration',
negative: 'photorealistic, blurry, deformed hands, extra limbs, low quality, watermark, text',
strength: 0.85,
guidance_scale: 8.0,
},
// watercolor/sketch: still dropped for the same identity-drift reasons as before
// (see git history) — image_edit's OpenCV filters remain the fallback for those.
};
export const imageStyleTransformTool = {
name: 'image_style_transform',
description: [
'Transform an existing photo (e.g. a portrait) into a new artistic style using local FLUX.1-dev img2img with PuLID identity locking, via ComfyUI (runs on-machine GPU, no external API).',
'Supports style="cartoon" (caricature/comic illustration) — more artistically interpretive than image_edit\'s classic OpenCV cartoon filter, at the cost of speed (~60-80s vs ~2-3s). A face embedding (insightface + EVA-CLIP) conditions the generation via PuLID so the result stays recognizably the same person even under a strong style prompt.',
'For anime/sketch/watercolor/black-and-white, use image_edit\'s stylize/filter operations instead — those are faster (both "watercolor" and "sketch" presets were tried and dropped: the style jump needed enough strength that results sometimes drifted the face into a different-looking person).',
'Returns the transformed image inline in the chat.',
].join('\n'),
schema: {
image: 'Path to the source image to transform (required)',
style: '"cartoon" (caricature/comic illustration) — currently the only supported style',
face_lock: 'Whether to lock facial identity via PuLID (optional, default true). Turn off to compare plain img2img — face-lock conditioning sometimes reads as slightly uncanny/over-smoothed; plain img2img gives looser but more natural-looking results.',
strength: 'How strongly to restyle, 0.05-1.0 (optional; default 0.85 — FLUX\'s flow-matching sampler needs much higher values than the old SDXL default to produce a visible style change at all). Lower preserves the original photo\'s pose/expression more, higher restyles more aggressively but can drift away from them.',
steps: 'Denoising steps (default 20, range 10-40)',
guidance_scale: 'How closely to follow the style prompt (default 8.0)',
seed: 'Random seed for reproducibility (optional)',
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
},
jsonSchema: {
type: 'object',
properties: {
image: { type: 'string' },
style: { type: 'string', enum: ['cartoon'] },
face_lock: { type: 'boolean' },
strength: { type: 'number' },
steps: { type: 'number' },
guidance_scale: { type: 'number' },
seed: { type: 'number' },
output: { type: 'string' },
},
required: ['image', 'style'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const imageArg = String(args?.image || '').trim();
if (!imageArg) return { success: false, error: 'image is required' };
const style = String(args?.style || '').trim();
const preset = IMG2IMG_STYLE_PRESETS[style];
if (!preset) return { success: false, error: `Unknown style: ${style} (expected "cartoon")` };
const workspacePath = getWorkspacePath(args);
const srcPath = path.isAbsolute(imageArg) ? imageArg : path.resolve(workspacePath, imageArg);
if (!isPathInsideDir(workspacePath, srcPath)) return { success: false, error: 'Access denied: path escapes workspace' };
if (!fs.existsSync(srcPath)) return { success: false, error: `Source image not found: ${imageArg}` };
let outPath = String(args?.output || '').trim();
if (!outPath) {
outPath = path.join(workspacePath, `${style}_${Date.now()}.png`);
} else if (!path.isAbsolute(outPath)) {
outPath = path.resolve(workspacePath, outPath);
}
const faceLock = args?.face_lock ?? true;
const denoise = Math.min(1, Math.max(0.05, args?.strength ?? preset.strength));
const guidance = toFiniteNumber(args?.guidance_scale, preset.guidance_scale);
const steps = Math.min(40, Math.max(10, args?.steps ?? 20));
const seed = args?.seed != null ? toFiniteNumber(args.seed, 0) : randomSeed();
// ComfyUI reads source images from its own input/ dir — copy in under a unique
// name (same host, so a filesystem copy, no HTTP upload round-trip needed).
fs.mkdirSync(COMFY_INPUT_DIR, { recursive: true });
const inputFilename = `style_src_${Date.now()}${path.extname(srcPath) || '.png'}`;
fs.copyFileSync(srcPath, path.join(COMFY_INPUT_DIR, inputFilename));
const graph = fluxPulidStyleWorkflow({
imageFilename: inputFilename,
prompt: preset.prompt,
negative_prompt: preset.negative,
denoise, guidance, steps, seed,
pulid_weight: faceLock ? 1.0 : 0,
});
// 600s (not the 180s the txt2img workflows use): first request after a
// service restart also has to load PuLID/EVA-CLIP/insightface on top of FLUX.1-dev.
const result = await comfyGenerateImage(graph, outPath, 0, 0, 600_000);
if ('error' in result || !result.output) {
return { success: false, error: 'error' in result ? result.error : 'Generator returned no output file' };
}
return {
success: true,
stdout: [
`Style: ${style}${faceLock ? ' (face-locked via PuLID)' : ''}`,
'',
buildImageMarkdown(result.output, workspacePath),
].join('\n'),
data: { output: result.output, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
};
},
};
export const imageGenerateTool = {
name: 'image_generate',
description: [
'Generate an image from a text prompt using a local diffusion model via ComfyUI (runs on-machine GPU, no external API).',
'quality="fast" (default): SDXL, ~15-20 seconds, default resolution 1024×1024, good for casual/quick illustrations.',
'quality="high": FLUX.1-schnell (fp8), ~15-25 seconds, noticeably more photorealistic detail and prompt accuracy — and no longer much slower than fast mode (moved off the old 4-bit bitsandbytes path, which needed ~40-60s), so lean toward "high" more readily than before. Still default to fast for quick/casual requests.',
'Returns the generated image inline in the chat.',
].join('\n'),
schema: {
prompt: 'Text description of the image to generate (English works best)',
quality: '"fast" (SDXL, default) or "high" (FLUX.1-schnell, much slower but noticeably better detail/realism)',
negative_prompt: 'Things to avoid in the image (optional, e.g. "blurry, low quality, deformed") — ignored in quality="high" mode (FLUX.1-schnell does not support it)',
width: 'Image width in pixels, multiple of 8 (default 1024)',
height: 'Image height in pixels, multiple of 8 (default 1024)',
steps: 'Denoising steps — more = higher quality but slower (fast mode: default 30, range 15–50; high mode: default 4, range 1–8)',
guidance_scale: 'How closely to follow the prompt (default 7.0, range 1–20) — fast mode only, ignored in quality="high"',
seed: 'Random seed for reproducibility (optional)',
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
},
jsonSchema: {
type: 'object',
properties: {
prompt: { type: 'string' },
quality: { type: 'string', enum: ['fast', 'high'] },
negative_prompt: { type: 'string' },
width: { type: 'number' },
height: { type: 'number' },
steps: { type: 'number' },
guidance_scale: { type: 'number' },
seed: { type: 'number' },
output: { type: 'string' },
},
required: ['prompt'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const prompt = String(args?.prompt || '').trim();
if (!prompt) return { success: false, error: 'prompt is required' };
const isHighQuality = args?.quality === 'high';
const workspacePath = getWorkspacePath(args);
let outPath = String(args?.output || '').trim();
if (!outPath) {
outPath = path.join(workspacePath, `${isHighQuality ? 'flux' : 'sdxl'}_${Date.now()}.png`);
} else if (!path.isAbsolute(outPath)) {
outPath = path.resolve(workspacePath, outPath);
}
const negativePromptRaw = args?.negative_prompt || '';
const [translatedPrompt, translatedNegative] = await Promise.all([
translatePromptToEnglish(prompt),
isHighQuality ? Promise.resolve(null) : translatePromptToEnglish(negativePromptRaw),
]);
const width = Math.round((args?.width ?? 1024) / 16) * 16;
const height = Math.round((args?.height ?? 1024) / 16) * 16;
const seed = args?.seed != null ? toFiniteNumber(args.seed, 0) : randomSeed();
const graph = isHighQuality
? fluxSchnellWorkflow({ prompt: translatedPrompt || prompt, width, height, steps: Math.min(8, Math.max(1, args?.steps ?? 4)), seed })
: sdxlWorkflow({
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height,
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
guidance_scale: toFiniteNumber(args?.guidance_scale, 7.0),
seed,
});
const result = await comfyGenerateImage(graph, outPath, width, height, 180_000);
if ('error' in result || !result.output) {
return { success: false, error: 'error' in result ? result.error : 'Generator returned no output file' };
}
return {
success: true,
stdout: [
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
`Generated: ${result.width} × ${result.height} px`,
'',
buildImageMarkdown(result.output, workspacePath),
].filter((line): line is string => line !== null).join('\n'),
data: { ...result, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
};
},
};
// ---------------------------------------------------------------------------
// video_generate — LTX-Video moved to ComfyUI (see ltxVideoWorkflow above);
// CogVideoX-2B still runs the old venv-script path below.
// ---------------------------------------------------------------------------
// CogVideoX-2B — added 2026-07-17 as a second option alongside LTX-Video for quality
// comparison. Barely fits: peak VRAM measured at ~11GB of the 12GB card even with
// enable_model_cpu_offload(), so there's little headroom for pushing resolution/frames
// higher than the defaults below (unlike LTX-Video, which has real slack). Noticeably
// slower than LTX-Video too. fp16 per the model card's own recommendation for the 2B
// checkpoint (the 5B checkpoint recommends bf16, but 5B doesn't fit here at all — OOMs
// even offloaded, its active-component compute footprint alone exceeds 12GB).
const COGVIDEOX_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch
from diffusers import CogVideoXPipeline
from diffusers.utils import export_to_video
pipe = CogVideoXPipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16)
pipe.enable_model_cpu_offload()
pipe.vae.enable_tiling()
video = pipe(
prompt=${JSON.stringify(p.prompt)},
negative_prompt=${JSON.stringify(p.negative_prompt)},
width=int(${p.width}), height=int(${p.height}),
num_frames=int(${p.num_frames}),
num_inference_steps=int(${p.steps}),
guidance_scale=float(${p.guidance_scale}),
).frames[0]
dst = ${JSON.stringify(p.dst)}
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
export_to_video(video, dst, fps=int(${p.fps}))
stat = os.stat(dst)
print("###RESULT###" + json.dumps({
"output": dst, "width": int(${p.width}), "height": int(${p.height}),
"num_frames": int(${p.num_frames}), "fps": int(${p.fps}),
"size_bytes": stat.st_size,
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
}))
except Exception as e:
import traceback
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
export const videoGenerateTool = {
name: 'video_generate',
description: [
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
'Three models available via the "model" param: "ltx" (default — LTX-Video 2B distilled via ComfyUI, ~25-40s), "ltx_hq" (LTX-Video 2B non-distilled/"dev" checkpoint via ComfyUI — noticeably better motion coherence and detail, ~40-90s, VRAM headroom is tight at ~10GB/12GB so avoid stacking a large image_generate call at the same time), and "cogvideox" (THUDM CogVideoX-2B — still the old venv-script path, slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try a couple and compare if unsure which fits the request.',
'Optional "image" param turns this into image-to-video: pass a path to an existing photo and the video will start from it instead of pure noise (only supported for "ltx"/"ltx_hq", not "cogvideox" — no image-conditioned CogVideoX checkpoint is installed).',
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
].join('\n'),
schema: {
prompt: 'Text description of the video/scene to generate (English works best)',
model: '"ltx" (default, fast, distilled), "ltx_hq" (LTX-Video non-distilled "dev" checkpoint, better quality, slower), or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
image: 'Path to a source photo to animate (image-to-video). Optional — omit for plain text-to-video. Only works with model "ltx"/"ltx_hq"; ignored (with an error) for "cogvideox".',
strength: 'How much the video is allowed to drift from the source image, 0-1 (only used with "image", default 0.5 — LTX\'s own example workflow uses 0.15 for near-static motion, higher values allow more change)',
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
height: 'Video height in pixels, multiple of 32 (default 480)',
num_frames: 'Number of frames — duration = num_frames / fps (default 65 for ltx, 49 for cogvideox)',
fps: 'Output frame rate (default 24 for ltx, 8 for cogvideox)',
steps: 'Denoising steps — more = higher quality but slower (default 40 for ltx / 25 for cogvideox — cogvideox is much slower per step, higher values risk exceeding reverse-proxy timeouts, range 15–50)',
guidance_scale: 'How closely to follow the prompt (default 3.0 for ltx / 6.0 for cogvideox, range 1–10)',
output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)',
},
jsonSchema: {
type: 'object',
properties: {
prompt: { type: 'string' },
model: { type: 'string', enum: ['ltx', 'ltx_hq', 'cogvideox'] },
image: { type: 'string' },
strength: { type: 'number' },
negative_prompt: { type: 'string' },
width: { type: 'number' },
height: { type: 'number' },
num_frames: { type: 'number' },
fps: { type: 'number' },
steps: { type: 'number' },
guidance_scale: { type: 'number' },
output: { type: 'string' },
},
required: ['prompt'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const prompt = String(args?.prompt || '').trim();
if (!prompt) return { success: false, error: 'prompt is required' };
const model = args?.model === 'cogvideox' ? 'cogvideox' : args?.model === 'ltx_hq' ? 'ltx_hq' : 'ltx';
const isCogVideoX = model === 'cogvideox';
const isHQ = model === 'ltx_hq';
const workspacePath = getWorkspacePath(args);
let outPath = String(args?.output || '').trim();
if (!outPath) {
outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : isHQ ? 'ltx_hq' : 'ltx'}_${Date.now()}.mp4`);
} else if (!path.isAbsolute(outPath)) {
outPath = path.resolve(workspacePath, outPath);
}
const imageArg = String(args?.image || '').trim();
if (imageArg && isCogVideoX) {
return { success: false, error: 'image-to-video is not supported for model "cogvideox" — use "ltx" or "ltx_hq"' };
}
let inputImageFilename: string | undefined;
if (imageArg) {
const srcPath = path.isAbsolute(imageArg) ? imageArg : path.resolve(workspacePath, imageArg);
if (!isPathInsideDir(workspacePath, srcPath)) return { success: false, error: 'Access denied: path escapes workspace' };
if (!fs.existsSync(srcPath)) return { success: false, error: `Source image not found: ${imageArg}` };
fs.mkdirSync(COMFY_INPUT_DIR, { recursive: true });
inputImageFilename = `vid_src_${Date.now()}${path.extname(srcPath) || '.png'}`;
fs.copyFileSync(srcPath, path.join(COMFY_INPUT_DIR, inputImageFilename));
}
const strength = Math.min(1, Math.max(0, toFiniteNumber(args?.strength, 0.5)));
const negativePromptRaw = args?.negative_prompt || 'worst quality, blurry, distorted, deformed';
const [translatedPrompt, translatedNegative] = await Promise.all([
translatePromptToEnglish(prompt),
translatePromptToEnglish(negativePromptRaw),
]);
const width = Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32;
const height = Math.round((args?.height ?? 480) / 32) * 32;
const fps = toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24);
const guidance_scale = Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0)));
// CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50
// (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any
// reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it
// lower to keep typical runs under ~4 minutes; still overridable via the steps param.
const steps = Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40)));
let num_frames: number;
let outputMeta: { output: string };
if (isCogVideoX) {
num_frames = toFiniteNumber(args?.num_frames, 49);
const params = {
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, num_frames, fps, steps, guidance_scale,
dst: outPath,
};
const result = await runVenvPython(COGVIDEOX_SCRIPT(params), 600_000);
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
}
outputMeta = { output: result.output };
} else {
// LTX-Video wants length as 8k+1 (its temporal VAE downsamples by 8) — snap
// whatever was requested to the nearest valid value instead of erroring.
const requestedFrames = toFiniteNumber(args?.num_frames, 65);
num_frames = Math.max(9, Math.round((requestedFrames - 1) / 8) * 8 + 1);
const ckpt_name = isHQ ? 'ltxv-2b-0.9.6-dev-04-25.safetensors' : undefined;
const graph = inputImageFilename
? ltxImageToVideoWorkflow({
imageFilename: inputImageFilename,
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, length: num_frames, fps, steps, guidance_scale, strength,
seed: randomSeed(),
ckpt_name,
})
: ltxVideoWorkflow({
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height, length: num_frames, fps, steps, guidance_scale,
seed: randomSeed(),
ckpt_name,
});
// The non-distilled "dev" checkpoint is ~1.4x the weight size of the distilled one and
// runs full CFG (two forward passes/step instead of one), so it's meaningfully slower —
// give it more headroom than the regular ltx path's 300s.
const result = await comfyGenerateVideo(graph, outPath, isHQ ? 420_000 : 300_000);
if ('error' in result) {
return { success: false, error: result.error };
}
outputMeta = result;
}
const relOut = path.relative(workspacePath, outputMeta.output).replace(/\\/g, '/');
const sizeMB = (fs.statSync(outputMeta.output).size / 1024 / 1024).toFixed(2);
const durationSec = (num_frames / fps).toFixed(1);
return {
success: true,
stdout: [
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
`Generated: ${width} × ${height} px | ${durationSec}s (${num_frames}f @ ${fps}fps) | ${sizeMB} MB`,
'',
`[${path.basename(outputMeta.output)}](/api/files/${relOut})`,
].filter((line): line is string => line !== null).join('\n'),
data: { output: outputMeta.output, width, height, num_frames, fps, rel_path: relOut },
};
},
};