studio-app.html에 시작 이미지 업로드/제거 UI + strength 슬라이더 추가, imagegen.ts에 ltxImageToVideoWorkflow (ComfyUI 공식 ltxv_image_to_video.json 템플릿 기반 LoadImage→LTXVImgToVideo→LTXVConditioning 그래프) 추가. 실제 생성 테스트로 프레임0(원본 이미지 일치)·프레임15(모션 발생) 확인 완료된 작업을 커밋만 안 하고 남겨뒀던 것. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
670 lines
38 KiB
TypeScript
670 lines
38 KiB
TypeScript
import { spawn } from 'child_process';
|
||
import path from 'path';
|
||
import fs from 'fs';
|
||
import { ToolResult } from '../types.js';
|
||
import { getWorkspacePath } from '../config/paths.js';
|
||
import { buildImageMarkdown, toFiniteNumber, isPathInsideDir } from './image.js';
|
||
import { getOllamaConfig } from './web.js';
|
||
|
||
// SDXL/LTX-Video's text encoders (CLIP/T5) are trained overwhelmingly on English
|
||
// captions, so non-English prompts (e.g. Korean) produce poor prompt adherence.
|
||
// Route non-ASCII prompts through the configured Ollama chat model for a quick
|
||
// English translation before handing them to the diffusion pipeline.
|
||
async function translatePromptToEnglish(text: string): Promise<string | null> {
|
||
if (!text || !/[^\x00-\x7F]/.test(text)) return null;
|
||
try {
|
||
const { endpoint, model } = getOllamaConfig();
|
||
const res = await fetch(`${endpoint}/api/chat`, {
|
||
method: 'POST',
|
||
headers: { 'Content-Type': 'application/json' },
|
||
body: JSON.stringify({
|
||
model,
|
||
messages: [{
|
||
role: 'user',
|
||
content: `Translate the following image/video generation prompt into natural, descriptive English. Output ONLY the translated English text — no quotes, no explanation:\n\n${text}`,
|
||
}],
|
||
stream: false,
|
||
}),
|
||
signal: AbortSignal.timeout(30_000),
|
||
});
|
||
if (!res.ok) return null;
|
||
const data: any = await res.json();
|
||
const translated = String(data.message?.content || '').trim();
|
||
return translated || null;
|
||
} catch {
|
||
return null;
|
||
}
|
||
}
|
||
|
||
// Local diffusion models (SDXL, LTX-Video) run in a dedicated venv with their
|
||
// own torch/diffusers stack, pinned to the second GPU (04:00.0 — kept free of
|
||
// the voice engine that permanently resides on GPU0). See
|
||
// /srv/homeclaw/.smallclaw/imagegen-venv.
|
||
const VENV_PYTHON = '/srv/homeclaw/.smallclaw/imagegen-venv/bin/python3';
|
||
const HF_HOME = '/srv/homeclaw/.smallclaw/imagegen-venv/hf-cache';
|
||
const GEN_GPU = '1';
|
||
|
||
function runVenvPython(script: string, timeoutMs: number): Promise<any> {
|
||
return new Promise((resolve) => {
|
||
const child = spawn(VENV_PYTHON, ['-c', script], {
|
||
timeout: timeoutMs,
|
||
// BNB_CUDA_VERSION: bitsandbytes (used by the FLUX high-quality path) ships no
|
||
// prebuilt binary for our CUDA 13.2 torch build yet — pin it to the newest
|
||
// available (13.0) binary, which is ABI-compatible. No-op for SDXL/LTX, which
|
||
// don't use bitsandbytes.
|
||
env: { ...process.env, HF_HOME, CUDA_VISIBLE_DEVICES: GEN_GPU, BNB_CUDA_VERSION: '130' },
|
||
});
|
||
let out = '';
|
||
let err = '';
|
||
child.stdout.on('data', (d: Buffer) => { out += d.toString('utf8'); });
|
||
child.stderr.on('data', (d: Buffer) => { err += d.toString('utf8'); });
|
||
child.on('close', (code: number | null, signal: string | null) => {
|
||
const marker = out.lastIndexOf('###RESULT###');
|
||
if (marker === -1) {
|
||
// Process exited (crashed, OOM-killed, or timed out) without ever printing a
|
||
// result marker — never silently fall through to a success-shaped {}, since
|
||
// that leaves result.output undefined and crashes the caller downstream.
|
||
resolve({
|
||
error: `Generator process exited without output (code=${code}, signal=${signal})`,
|
||
trace: (err || out).slice(-1500),
|
||
});
|
||
return;
|
||
}
|
||
const jsonPart = out.slice(marker + '###RESULT###'.length);
|
||
try {
|
||
resolve(JSON.parse(jsonPart.trim()));
|
||
} catch {
|
||
resolve({ error: 'Failed to parse generator output', raw: (jsonPart || err).slice(-1500) });
|
||
}
|
||
});
|
||
child.on('error', (e: Error) => resolve({ error: e.message }));
|
||
});
|
||
}
|
||
|
||
// ---------------------------------------------------------------------------
|
||
// ComfyUI backend — replaces the old per-request venv-script spawning for
|
||
// image_generate and image_style_transform (2026-08-05). ComfyUI runs as a
|
||
// persistent systemd --user service (~/.config/systemd/user/comfyui.service,
|
||
// GPU1, port 8188) with models pre-loaded in models/checkpoints|diffusion_models|
|
||
// clip|vae|pulid|insightface|facexlib under /home/kim/comfyui. See
|
||
// project_comfyui_evaluation / project_windy_jetstream_feature memory for how
|
||
// these were chosen and benchmarked — FLUX.1-schnell (fp8) beat the old 4-bit
|
||
// bitsandbytes path 2-3x on the same GPU; SDXL was a wash so it moved over too
|
||
// for one consistent backend. video_generate (LTX/CogVideoX) is untouched —
|
||
// still runs through the venv-script path below, no ComfyUI workflow for those.
|
||
const COMFY_URL = 'http://127.0.0.1:8188';
|
||
const COMFY_DIR = '/home/kim/comfyui';
|
||
const COMFY_OUTPUT_DIR = path.join(COMFY_DIR, 'output');
|
||
const COMFY_INPUT_DIR = path.join(COMFY_DIR, 'input');
|
||
|
||
function randomSeed(): number {
|
||
return Math.floor(Math.random() * 0xFFFFFFFF);
|
||
}
|
||
|
||
async function comfySubmit(promptGraph: Record<string, any>): Promise<string> {
|
||
const res = await fetch(`${COMFY_URL}/prompt`, {
|
||
method: 'POST',
|
||
headers: { 'Content-Type': 'application/json' },
|
||
body: JSON.stringify({ prompt: promptGraph }),
|
||
});
|
||
const data: any = await res.json().catch(() => ({}));
|
||
if (data?.node_errors && Object.keys(data.node_errors).length) {
|
||
throw new Error('ComfyUI rejected the workflow: ' + JSON.stringify(data.node_errors));
|
||
}
|
||
if (!data?.prompt_id) throw new Error(data?.error ? JSON.stringify(data.error) : 'ComfyUI did not return a prompt_id');
|
||
return data.prompt_id;
|
||
}
|
||
|
||
// No websocket/progress push used here (keeps this dependency-free) — just poll
|
||
// /history, same as the manual testing that validated every workflow below.
|
||
async function comfyPollResult(promptId: string, timeoutMs: number): Promise<{ filename: string; subfolder: string }> {
|
||
const start = Date.now();
|
||
while (Date.now() - start < timeoutMs) {
|
||
const res = await fetch(`${COMFY_URL}/history/${promptId}`);
|
||
const data: any = await res.json().catch(() => ({}));
|
||
const entry = data?.[promptId];
|
||
if (entry) {
|
||
if (entry.status?.status_str === 'error') {
|
||
const errMsg = entry.status?.messages?.find((m: any) => m[0] === 'execution_error')?.[1]?.exception_message;
|
||
throw new Error(errMsg || 'ComfyUI execution failed');
|
||
}
|
||
for (const nodeOut of Object.values<any>(entry.outputs || {})) {
|
||
if (nodeOut?.images?.length) return { filename: nodeOut.images[0].filename, subfolder: nodeOut.images[0].subfolder || '' };
|
||
}
|
||
}
|
||
await new Promise((r) => setTimeout(r, 2000));
|
||
}
|
||
throw new Error('Timed out waiting for ComfyUI generation');
|
||
}
|
||
|
||
async function comfyGenerateImage(
|
||
promptGraph: Record<string, any>,
|
||
dst: string,
|
||
width: number,
|
||
height: number,
|
||
timeoutMs: number,
|
||
): Promise<{ output: string; width: number; height: number } | { error: string }> {
|
||
try {
|
||
const promptId = await comfySubmit(promptGraph);
|
||
const { filename, subfolder } = await comfyPollResult(promptId, timeoutMs);
|
||
const srcFile = path.join(COMFY_OUTPUT_DIR, subfolder, filename);
|
||
fs.mkdirSync(path.dirname(dst), { recursive: true });
|
||
fs.copyFileSync(srcFile, dst);
|
||
return { output: dst, width, height };
|
||
} catch (e: any) {
|
||
return { error: e?.message || String(e) };
|
||
}
|
||
}
|
||
|
||
function sdxlWorkflow(p: { prompt: string; negative_prompt: string; width: number; height: number; steps: number; guidance_scale: number; seed: number }) {
|
||
return {
|
||
'3': { class_type: 'KSampler', inputs: { cfg: p.guidance_scale, denoise: 1.0, latent_image: ['5', 0], model: ['4', 0], negative: ['7', 0], positive: ['6', 0], sampler_name: 'euler', scheduler: 'normal', seed: p.seed, steps: p.steps } },
|
||
'4': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: 'sd_xl_base_1.0.safetensors' } },
|
||
'5': { class_type: 'EmptyLatentImage', inputs: { batch_size: 1, height: p.height, width: p.width } },
|
||
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['4', 1], text: p.prompt } },
|
||
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['4', 1], text: p.negative_prompt || '' } },
|
||
'8': { class_type: 'VAEDecode', inputs: { samples: ['3', 0], vae: ['4', 2] } },
|
||
'9': { class_type: 'SaveImage', inputs: { filename_prefix: 'sdxl_api', images: ['8', 0] } },
|
||
};
|
||
}
|
||
|
||
function fluxSchnellWorkflow(p: { prompt: string; width: number; height: number; steps: number; seed: number }) {
|
||
return {
|
||
'1': { class_type: 'UNETLoader', inputs: { unet_name: 'flux1-schnell-fp8.safetensors', weight_dtype: 'default' } },
|
||
'2': { class_type: 'DualCLIPLoader', inputs: { clip_name1: 'clip_l.safetensors', clip_name2: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'flux' } },
|
||
'3': { class_type: 'VAELoader', inputs: { vae_name: 'ae.safetensors' } },
|
||
'4': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.prompt } },
|
||
'5': { class_type: 'EmptySD3LatentImage', inputs: { batch_size: 1, height: p.height, width: p.width } },
|
||
'6': { class_type: 'KSampler', inputs: { cfg: 1.0, denoise: 1.0, latent_image: ['5', 0], model: ['1', 0], negative: ['4', 0], positive: ['4', 0], sampler_name: 'euler', scheduler: 'simple', seed: p.seed, steps: p.steps } },
|
||
'7': { class_type: 'VAEDecode', inputs: { samples: ['6', 0], vae: ['3', 0] } },
|
||
'8': { class_type: 'SaveImage', inputs: { filename_prefix: 'flux_api', images: ['7', 0] } },
|
||
};
|
||
}
|
||
|
||
// FLUX.1-dev + PuLID (identity-locked img2img). Unlike the txt2img workflows
|
||
// above, guidance=3.5/denoise~0.45 (the old SDXL-tuned defaults) produced almost
|
||
// no visible style change at all when this was benchmarked — FLUX's flow-matching
|
||
// sampler needs guidance~8 and denoise~0.85 before the caricature prompt actually
|
||
// overrides the source photo. See IMG2IMG_STYLE_PRESETS below for the tuned values.
|
||
function fluxPulidStyleWorkflow(p: { imageFilename: string; prompt: string; negative_prompt: string; denoise: number; guidance: number; steps: number; seed: number; pulid_weight: number }) {
|
||
const graph: Record<string, any> = {
|
||
'1': { class_type: 'UNETLoader', inputs: { unet_name: 'flux1-dev-fp8-e4m3fn.safetensors', weight_dtype: 'default' } },
|
||
'2': { class_type: 'DualCLIPLoader', inputs: { clip_name1: 'clip_l.safetensors', clip_name2: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'flux' } },
|
||
'3': { class_type: 'VAELoader', inputs: { vae_name: 'ae.safetensors' } },
|
||
'4': { class_type: 'LoadImage', inputs: { image: p.imageFilename } },
|
||
'4b': { class_type: 'ImageScaleToTotalPixels', inputs: { image: ['4', 0], upscale_method: 'lanczos', megapixels: 1.0, resolution_steps: 16 } },
|
||
'5': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.prompt } },
|
||
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['2', 0], text: p.negative_prompt || '' } },
|
||
'7': { class_type: 'FluxGuidance', inputs: { conditioning: ['5', 0], guidance: p.guidance } },
|
||
'12': { class_type: 'VAEEncode', inputs: { pixels: ['4b', 0], vae: ['3', 0] } },
|
||
'13': { class_type: 'KSampler', inputs: { model: ['1', 0], positive: ['7', 0], negative: ['6', 0], latent_image: ['12', 0], seed: p.seed, steps: p.steps, cfg: 1.0, sampler_name: 'euler', scheduler: 'simple', denoise: p.denoise } },
|
||
'14': { class_type: 'VAEDecode', inputs: { samples: ['13', 0], vae: ['3', 0] } },
|
||
'15': { class_type: 'SaveImage', inputs: { filename_prefix: 'flux_pulid_api', images: ['14', 0] } },
|
||
};
|
||
// face_lock=false (or no face found upstream): skip PuLID nodes entirely rather
|
||
// than including a disconnected node — ComfyUI validates the graph as a DAG, so
|
||
// a "no-op" node with missing required inputs would just fail to queue.
|
||
if (p.pulid_weight > 0) {
|
||
graph['8'] = { class_type: 'PulidFluxModelLoader', inputs: { pulid_file: 'pulid_flux_v0.9.1.safetensors' } };
|
||
graph['9'] = { class_type: 'PulidFluxInsightFaceLoader', inputs: { provider: 'CUDA' } };
|
||
graph['10'] = { class_type: 'PulidFluxEvaClipLoader', inputs: {} };
|
||
graph['11'] = { class_type: 'ApplyPulidFlux', inputs: { model: ['1', 0], pulid_flux: ['8', 0], eva_clip: ['10', 0], face_analysis: ['9', 0], image: ['4b', 0], weight: p.pulid_weight, start_at: 0.0, end_at: 1.0 } };
|
||
graph['13'].inputs.model = ['11', 0];
|
||
}
|
||
return graph;
|
||
}
|
||
|
||
// LTX-Video text-to-video, ported to ComfyUI 2026-08-05 (see project_comfyui_evaluation
|
||
// memory) — the old venv-script path left GPU1 fully free between requests, but once
|
||
// ComfyUI started staying resident there for images, the two independent processes
|
||
// started fighting over the same VRAM (one real request took >200s and had to be
|
||
// killed). Moving LTX into ComfyUI too puts all of GPU1 under one memory manager.
|
||
// Node graph copied from ComfyUI's own bundled ltxv_text_to_video.json template —
|
||
// SamplerCustom+LTXVScheduler+KSamplerSelect, not plain KSampler, is how LTX is meant
|
||
// to be driven. The T5-XXL text encoder is the same file already downloaded for FLUX.
|
||
function ltxVideoWorkflow(p: { prompt: string; negative_prompt: string; width: number; height: number; length: number; fps: number; steps: number; guidance_scale: number; seed: number; ckpt_name?: string }) {
|
||
return {
|
||
'38': { class_type: 'CLIPLoader', inputs: { clip_name: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'ltxv', device: 'default' } },
|
||
'44': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: p.ckpt_name || 'ltxv-2b-0.9.8-distilled-fp8.safetensors' } },
|
||
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.prompt } },
|
||
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.negative_prompt } },
|
||
'70': { class_type: 'EmptyLTXVLatentVideo', inputs: { width: p.width, height: p.height, length: p.length, batch_size: 1 } },
|
||
'69': { class_type: 'LTXVConditioning', inputs: { positive: ['6', 0], negative: ['7', 0], frame_rate: p.fps } },
|
||
'71': { class_type: 'LTXVScheduler', inputs: { steps: p.steps, max_shift: 2.05, base_shift: 0.95, stretch: true, terminal: 0.1, latent: ['70', 0] } },
|
||
'73': { class_type: 'KSamplerSelect', inputs: { sampler_name: 'euler' } },
|
||
'72': { class_type: 'SamplerCustom', inputs: { model: ['44', 0], add_noise: true, noise_seed: p.seed, cfg: p.guidance_scale, positive: ['69', 0], negative: ['69', 1], sampler: ['73', 0], sigmas: ['71', 0], latent_image: ['70', 0] } },
|
||
'8': { class_type: 'VAEDecode', inputs: { samples: ['72', 0], vae: ['44', 2] } },
|
||
'78': { class_type: 'CreateVideo', inputs: { images: ['8', 0], fps: p.fps } },
|
||
'79': { class_type: 'SaveVideo', inputs: { video: ['78', 0], filename_prefix: 'ltx_api', format: 'mp4', codec: 'h264' } },
|
||
};
|
||
}
|
||
|
||
// Image-to-video variant — same LTX checkpoint/text-encoder as ltxVideoWorkflow above,
|
||
// but the source photo goes through LTXVImgToVideo *before* LTXVConditioning (wiring
|
||
// copied from ComfyUI's bundled ltxv_image_to_video.json template, node IDs kept
|
||
// matching that template for traceability). LTXVImgToVideo bakes the image into both
|
||
// the conditioning and the starting latent, replacing EmptyLTXVLatentVideo entirely —
|
||
// there's no plain "add an image on top of the txt2vid graph" path, the whole latent
|
||
// source changes.
|
||
function ltxImageToVideoWorkflow(p: { imageFilename: string; prompt: string; negative_prompt: string; width: number; height: number; length: number; fps: number; steps: number; guidance_scale: number; strength: number; seed: number; ckpt_name?: string }) {
|
||
return {
|
||
'38': { class_type: 'CLIPLoader', inputs: { clip_name: 't5xxl_fp8_e4m3fn_scaled.safetensors', type: 'ltxv', device: 'default' } },
|
||
'44': { class_type: 'CheckpointLoaderSimple', inputs: { ckpt_name: p.ckpt_name || 'ltxv-2b-0.9.8-distilled-fp8.safetensors' } },
|
||
'6': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.prompt } },
|
||
'7': { class_type: 'CLIPTextEncode', inputs: { clip: ['38', 0], text: p.negative_prompt } },
|
||
'78': { class_type: 'LoadImage', inputs: { image: p.imageFilename } },
|
||
'77': { class_type: 'LTXVImgToVideo', inputs: { positive: ['6', 0], negative: ['7', 0], vae: ['44', 2], image: ['78', 0], width: p.width, height: p.height, length: p.length, batch_size: 1, strength: p.strength } },
|
||
'69': { class_type: 'LTXVConditioning', inputs: { positive: ['77', 0], negative: ['77', 1], frame_rate: p.fps } },
|
||
'71': { class_type: 'LTXVScheduler', inputs: { steps: p.steps, max_shift: 2.05, base_shift: 0.95, stretch: true, terminal: 0.1, latent: ['77', 2] } },
|
||
'73': { class_type: 'KSamplerSelect', inputs: { sampler_name: 'euler' } },
|
||
'72': { class_type: 'SamplerCustom', inputs: { model: ['44', 0], add_noise: true, noise_seed: p.seed, cfg: p.guidance_scale, positive: ['69', 0], negative: ['69', 1], sampler: ['73', 0], sigmas: ['71', 0], latent_image: ['77', 2] } },
|
||
'8': { class_type: 'VAEDecode', inputs: { samples: ['72', 0], vae: ['44', 2] } },
|
||
'80': { class_type: 'CreateVideo', inputs: { images: ['8', 0], fps: p.fps } },
|
||
'81': { class_type: 'SaveVideo', inputs: { video: ['80', 0], filename_prefix: 'ltx_i2v_api', format: 'mp4', codec: 'h264' } },
|
||
};
|
||
}
|
||
|
||
// Same submit/poll plumbing as comfyGenerateImage but the SaveVideo node's output file
|
||
// isn't a PNG — copy verbatim rather than assuming an image extension.
|
||
async function comfyGenerateVideo(promptGraph: Record<string, any>, dst: string, timeoutMs: number): Promise<{ output: string } | { error: string }> {
|
||
try {
|
||
const promptId = await comfySubmit(promptGraph);
|
||
const { filename, subfolder } = await comfyPollResult(promptId, timeoutMs);
|
||
const srcFile = path.join(COMFY_OUTPUT_DIR, subfolder, filename);
|
||
fs.mkdirSync(path.dirname(dst), { recursive: true });
|
||
fs.copyFileSync(srcFile, dst);
|
||
return { output: dst };
|
||
} catch (e: any) {
|
||
return { error: e?.message || String(e) };
|
||
}
|
||
}
|
||
|
||
const IMG2IMG_STYLE_PRESETS: Record<string, { prompt: string; negative: string; strength: number; guidance_scale: number }> = {
|
||
cartoon: {
|
||
// Tuned on FLUX.1-dev + PuLID (2026-08-05) — see fluxPulidStyleWorkflow comment.
|
||
// The old SDXL-era numbers (strength 0.45 / guidance 7.0) are no longer used.
|
||
prompt: 'caricature portrait illustration, exaggerated facial features, bold clean outlines, vibrant flat colors, humorous comic art style, digital illustration',
|
||
negative: 'photorealistic, blurry, deformed hands, extra limbs, low quality, watermark, text',
|
||
strength: 0.85,
|
||
guidance_scale: 8.0,
|
||
},
|
||
// watercolor/sketch: still dropped for the same identity-drift reasons as before
|
||
// (see git history) — image_edit's OpenCV filters remain the fallback for those.
|
||
};
|
||
|
||
export const imageStyleTransformTool = {
|
||
name: 'image_style_transform',
|
||
description: [
|
||
'Transform an existing photo (e.g. a portrait) into a new artistic style using local FLUX.1-dev img2img with PuLID identity locking, via ComfyUI (runs on-machine GPU, no external API).',
|
||
'Supports style="cartoon" (caricature/comic illustration) — more artistically interpretive than image_edit\'s classic OpenCV cartoon filter, at the cost of speed (~60-80s vs ~2-3s). A face embedding (insightface + EVA-CLIP) conditions the generation via PuLID so the result stays recognizably the same person even under a strong style prompt.',
|
||
'For anime/sketch/watercolor/black-and-white, use image_edit\'s stylize/filter operations instead — those are faster (both "watercolor" and "sketch" presets were tried and dropped: the style jump needed enough strength that results sometimes drifted the face into a different-looking person).',
|
||
'Returns the transformed image inline in the chat.',
|
||
].join('\n'),
|
||
schema: {
|
||
image: 'Path to the source image to transform (required)',
|
||
style: '"cartoon" (caricature/comic illustration) — currently the only supported style',
|
||
face_lock: 'Whether to lock facial identity via PuLID (optional, default true). Turn off to compare plain img2img — face-lock conditioning sometimes reads as slightly uncanny/over-smoothed; plain img2img gives looser but more natural-looking results.',
|
||
strength: 'How strongly to restyle, 0.05-1.0 (optional; default 0.85 — FLUX\'s flow-matching sampler needs much higher values than the old SDXL default to produce a visible style change at all). Lower preserves the original photo\'s pose/expression more, higher restyles more aggressively but can drift away from them.',
|
||
steps: 'Denoising steps (default 20, range 10-40)',
|
||
guidance_scale: 'How closely to follow the style prompt (default 8.0)',
|
||
seed: 'Random seed for reproducibility (optional)',
|
||
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
|
||
},
|
||
jsonSchema: {
|
||
type: 'object',
|
||
properties: {
|
||
image: { type: 'string' },
|
||
style: { type: 'string', enum: ['cartoon'] },
|
||
face_lock: { type: 'boolean' },
|
||
strength: { type: 'number' },
|
||
steps: { type: 'number' },
|
||
guidance_scale: { type: 'number' },
|
||
seed: { type: 'number' },
|
||
output: { type: 'string' },
|
||
},
|
||
required: ['image', 'style'],
|
||
additionalProperties: false,
|
||
},
|
||
execute: async (args: any): Promise<ToolResult> => {
|
||
const imageArg = String(args?.image || '').trim();
|
||
if (!imageArg) return { success: false, error: 'image is required' };
|
||
const style = String(args?.style || '').trim();
|
||
const preset = IMG2IMG_STYLE_PRESETS[style];
|
||
if (!preset) return { success: false, error: `Unknown style: ${style} (expected "cartoon")` };
|
||
|
||
const workspacePath = getWorkspacePath(args);
|
||
const srcPath = path.isAbsolute(imageArg) ? imageArg : path.resolve(workspacePath, imageArg);
|
||
if (!isPathInsideDir(workspacePath, srcPath)) return { success: false, error: 'Access denied: path escapes workspace' };
|
||
if (!fs.existsSync(srcPath)) return { success: false, error: `Source image not found: ${imageArg}` };
|
||
|
||
let outPath = String(args?.output || '').trim();
|
||
if (!outPath) {
|
||
outPath = path.join(workspacePath, `${style}_${Date.now()}.png`);
|
||
} else if (!path.isAbsolute(outPath)) {
|
||
outPath = path.resolve(workspacePath, outPath);
|
||
}
|
||
|
||
const faceLock = args?.face_lock ?? true;
|
||
const denoise = Math.min(1, Math.max(0.05, args?.strength ?? preset.strength));
|
||
const guidance = toFiniteNumber(args?.guidance_scale, preset.guidance_scale);
|
||
const steps = Math.min(40, Math.max(10, args?.steps ?? 20));
|
||
const seed = args?.seed != null ? toFiniteNumber(args.seed, 0) : randomSeed();
|
||
|
||
// ComfyUI reads source images from its own input/ dir — copy in under a unique
|
||
// name (same host, so a filesystem copy, no HTTP upload round-trip needed).
|
||
fs.mkdirSync(COMFY_INPUT_DIR, { recursive: true });
|
||
const inputFilename = `style_src_${Date.now()}${path.extname(srcPath) || '.png'}`;
|
||
fs.copyFileSync(srcPath, path.join(COMFY_INPUT_DIR, inputFilename));
|
||
|
||
const graph = fluxPulidStyleWorkflow({
|
||
imageFilename: inputFilename,
|
||
prompt: preset.prompt,
|
||
negative_prompt: preset.negative,
|
||
denoise, guidance, steps, seed,
|
||
pulid_weight: faceLock ? 1.0 : 0,
|
||
});
|
||
// 600s (not the 180s the txt2img workflows use): first request after a
|
||
// service restart also has to load PuLID/EVA-CLIP/insightface on top of FLUX.1-dev.
|
||
const result = await comfyGenerateImage(graph, outPath, 0, 0, 600_000);
|
||
if ('error' in result || !result.output) {
|
||
return { success: false, error: 'error' in result ? result.error : 'Generator returned no output file' };
|
||
}
|
||
|
||
return {
|
||
success: true,
|
||
stdout: [
|
||
`Style: ${style}${faceLock ? ' (face-locked via PuLID)' : ''}`,
|
||
'',
|
||
buildImageMarkdown(result.output, workspacePath),
|
||
].join('\n'),
|
||
data: { output: result.output, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
|
||
};
|
||
},
|
||
};
|
||
|
||
export const imageGenerateTool = {
|
||
name: 'image_generate',
|
||
description: [
|
||
'Generate an image from a text prompt using a local diffusion model via ComfyUI (runs on-machine GPU, no external API).',
|
||
'quality="fast" (default): SDXL, ~15-20 seconds, default resolution 1024×1024, good for casual/quick illustrations.',
|
||
'quality="high": FLUX.1-schnell (fp8), ~15-25 seconds, noticeably more photorealistic detail and prompt accuracy — and no longer much slower than fast mode (moved off the old 4-bit bitsandbytes path, which needed ~40-60s), so lean toward "high" more readily than before. Still default to fast for quick/casual requests.',
|
||
'Returns the generated image inline in the chat.',
|
||
].join('\n'),
|
||
schema: {
|
||
prompt: 'Text description of the image to generate (English works best)',
|
||
quality: '"fast" (SDXL, default) or "high" (FLUX.1-schnell, much slower but noticeably better detail/realism)',
|
||
negative_prompt: 'Things to avoid in the image (optional, e.g. "blurry, low quality, deformed") — ignored in quality="high" mode (FLUX.1-schnell does not support it)',
|
||
width: 'Image width in pixels, multiple of 8 (default 1024)',
|
||
height: 'Image height in pixels, multiple of 8 (default 1024)',
|
||
steps: 'Denoising steps — more = higher quality but slower (fast mode: default 30, range 15–50; high mode: default 4, range 1–8)',
|
||
guidance_scale: 'How closely to follow the prompt (default 7.0, range 1–20) — fast mode only, ignored in quality="high"',
|
||
seed: 'Random seed for reproducibility (optional)',
|
||
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
|
||
},
|
||
jsonSchema: {
|
||
type: 'object',
|
||
properties: {
|
||
prompt: { type: 'string' },
|
||
quality: { type: 'string', enum: ['fast', 'high'] },
|
||
negative_prompt: { type: 'string' },
|
||
width: { type: 'number' },
|
||
height: { type: 'number' },
|
||
steps: { type: 'number' },
|
||
guidance_scale: { type: 'number' },
|
||
seed: { type: 'number' },
|
||
output: { type: 'string' },
|
||
},
|
||
required: ['prompt'],
|
||
additionalProperties: false,
|
||
},
|
||
execute: async (args: any): Promise<ToolResult> => {
|
||
const prompt = String(args?.prompt || '').trim();
|
||
if (!prompt) return { success: false, error: 'prompt is required' };
|
||
const isHighQuality = args?.quality === 'high';
|
||
|
||
const workspacePath = getWorkspacePath(args);
|
||
let outPath = String(args?.output || '').trim();
|
||
if (!outPath) {
|
||
outPath = path.join(workspacePath, `${isHighQuality ? 'flux' : 'sdxl'}_${Date.now()}.png`);
|
||
} else if (!path.isAbsolute(outPath)) {
|
||
outPath = path.resolve(workspacePath, outPath);
|
||
}
|
||
|
||
const negativePromptRaw = args?.negative_prompt || '';
|
||
const [translatedPrompt, translatedNegative] = await Promise.all([
|
||
translatePromptToEnglish(prompt),
|
||
isHighQuality ? Promise.resolve(null) : translatePromptToEnglish(negativePromptRaw),
|
||
]);
|
||
|
||
const width = Math.round((args?.width ?? 1024) / 16) * 16;
|
||
const height = Math.round((args?.height ?? 1024) / 16) * 16;
|
||
|
||
const seed = args?.seed != null ? toFiniteNumber(args.seed, 0) : randomSeed();
|
||
const graph = isHighQuality
|
||
? fluxSchnellWorkflow({ prompt: translatedPrompt || prompt, width, height, steps: Math.min(8, Math.max(1, args?.steps ?? 4)), seed })
|
||
: sdxlWorkflow({
|
||
prompt: translatedPrompt || prompt,
|
||
negative_prompt: translatedNegative || negativePromptRaw,
|
||
width, height,
|
||
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
|
||
guidance_scale: toFiniteNumber(args?.guidance_scale, 7.0),
|
||
seed,
|
||
});
|
||
const result = await comfyGenerateImage(graph, outPath, width, height, 180_000);
|
||
if ('error' in result || !result.output) {
|
||
return { success: false, error: 'error' in result ? result.error : 'Generator returned no output file' };
|
||
}
|
||
|
||
return {
|
||
success: true,
|
||
stdout: [
|
||
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
|
||
`Generated: ${result.width} × ${result.height} px`,
|
||
'',
|
||
buildImageMarkdown(result.output, workspacePath),
|
||
].filter((line): line is string => line !== null).join('\n'),
|
||
data: { ...result, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
|
||
};
|
||
},
|
||
};
|
||
|
||
// ---------------------------------------------------------------------------
|
||
// video_generate — LTX-Video moved to ComfyUI (see ltxVideoWorkflow above);
|
||
// CogVideoX-2B still runs the old venv-script path below.
|
||
// ---------------------------------------------------------------------------
|
||
|
||
// CogVideoX-2B — added 2026-07-17 as a second option alongside LTX-Video for quality
|
||
// comparison. Barely fits: peak VRAM measured at ~11GB of the 12GB card even with
|
||
// enable_model_cpu_offload(), so there's little headroom for pushing resolution/frames
|
||
// higher than the defaults below (unlike LTX-Video, which has real slack). Noticeably
|
||
// slower than LTX-Video too. fp16 per the model card's own recommendation for the 2B
|
||
// checkpoint (the 5B checkpoint recommends bf16, but 5B doesn't fit here at all — OOMs
|
||
// even offloaded, its active-component compute footprint alone exceeds 12GB).
|
||
const COGVIDEOX_SCRIPT = (p: Record<string, any>) => `
|
||
import os, json, sys
|
||
try:
|
||
import torch
|
||
from diffusers import CogVideoXPipeline
|
||
from diffusers.utils import export_to_video
|
||
|
||
pipe = CogVideoXPipeline.from_pretrained("THUDM/CogVideoX-2b", torch_dtype=torch.float16)
|
||
pipe.enable_model_cpu_offload()
|
||
pipe.vae.enable_tiling()
|
||
|
||
video = pipe(
|
||
prompt=${JSON.stringify(p.prompt)},
|
||
negative_prompt=${JSON.stringify(p.negative_prompt)},
|
||
width=int(${p.width}), height=int(${p.height}),
|
||
num_frames=int(${p.num_frames}),
|
||
num_inference_steps=int(${p.steps}),
|
||
guidance_scale=float(${p.guidance_scale}),
|
||
).frames[0]
|
||
|
||
dst = ${JSON.stringify(p.dst)}
|
||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||
export_to_video(video, dst, fps=int(${p.fps}))
|
||
|
||
stat = os.stat(dst)
|
||
print("###RESULT###" + json.dumps({
|
||
"output": dst, "width": int(${p.width}), "height": int(${p.height}),
|
||
"num_frames": int(${p.num_frames}), "fps": int(${p.fps}),
|
||
"size_bytes": stat.st_size,
|
||
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
|
||
}))
|
||
except Exception as e:
|
||
import traceback
|
||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||
`;
|
||
|
||
export const videoGenerateTool = {
|
||
name: 'video_generate',
|
||
description: [
|
||
'Generate a short video clip from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
|
||
'Three models available via the "model" param: "ltx" (default — LTX-Video 2B distilled via ComfyUI, ~25-40s), "ltx_hq" (LTX-Video 2B non-distilled/"dev" checkpoint via ComfyUI — noticeably better motion coherence and detail, ~40-90s, VRAM headroom is tight at ~10GB/12GB so avoid stacking a large image_generate call at the same time), and "cogvideox" (THUDM CogVideoX-2B — still the old venv-script path, slower, ~1-3min, VRAM is tight so avoid pushing resolution/frames above the defaults). Try a couple and compare if unsure which fits the request.',
|
||
'Optional "image" param turns this into image-to-video: pass a path to an existing photo and the video will start from it instead of pure noise (only supported for "ltx"/"ltx_hq", not "cogvideox" — no image-conditioned CogVideoX checkpoint is installed).',
|
||
'Output is an h264 mp4. Returns a download link — there is no inline video preview in chat yet.',
|
||
].join('\n'),
|
||
schema: {
|
||
prompt: 'Text description of the video/scene to generate (English works best)',
|
||
model: '"ltx" (default, fast, distilled), "ltx_hq" (LTX-Video non-distilled "dev" checkpoint, better quality, slower), or "cogvideox" (THUDM CogVideoX-2B, slower, tighter VRAM headroom)',
|
||
image: 'Path to a source photo to animate (image-to-video). Optional — omit for plain text-to-video. Only works with model "ltx"/"ltx_hq"; ignored (with an error) for "cogvideox".',
|
||
strength: 'How much the video is allowed to drift from the source image, 0-1 (only used with "image", default 0.5 — LTX\'s own example workflow uses 0.15 for near-static motion, higher values allow more change)',
|
||
negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")',
|
||
width: 'Video width in pixels, multiple of 32 (default 704 for ltx, 720 for cogvideox)',
|
||
height: 'Video height in pixels, multiple of 32 (default 480)',
|
||
num_frames: 'Number of frames — duration = num_frames / fps (default 65 for ltx, 49 for cogvideox)',
|
||
fps: 'Output frame rate (default 24 for ltx, 8 for cogvideox)',
|
||
steps: 'Denoising steps — more = higher quality but slower (default 40 for ltx / 25 for cogvideox — cogvideox is much slower per step, higher values risk exceeding reverse-proxy timeouts, range 15–50)',
|
||
guidance_scale: 'How closely to follow the prompt (default 3.0 for ltx / 6.0 for cogvideox, range 1–10)',
|
||
output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)',
|
||
},
|
||
jsonSchema: {
|
||
type: 'object',
|
||
properties: {
|
||
prompt: { type: 'string' },
|
||
model: { type: 'string', enum: ['ltx', 'ltx_hq', 'cogvideox'] },
|
||
image: { type: 'string' },
|
||
strength: { type: 'number' },
|
||
negative_prompt: { type: 'string' },
|
||
width: { type: 'number' },
|
||
height: { type: 'number' },
|
||
num_frames: { type: 'number' },
|
||
fps: { type: 'number' },
|
||
steps: { type: 'number' },
|
||
guidance_scale: { type: 'number' },
|
||
output: { type: 'string' },
|
||
},
|
||
required: ['prompt'],
|
||
additionalProperties: false,
|
||
},
|
||
execute: async (args: any): Promise<ToolResult> => {
|
||
const prompt = String(args?.prompt || '').trim();
|
||
if (!prompt) return { success: false, error: 'prompt is required' };
|
||
|
||
const model = args?.model === 'cogvideox' ? 'cogvideox' : args?.model === 'ltx_hq' ? 'ltx_hq' : 'ltx';
|
||
const isCogVideoX = model === 'cogvideox';
|
||
const isHQ = model === 'ltx_hq';
|
||
|
||
const workspacePath = getWorkspacePath(args);
|
||
let outPath = String(args?.output || '').trim();
|
||
if (!outPath) {
|
||
outPath = path.join(workspacePath, `${isCogVideoX ? 'cogvideox' : isHQ ? 'ltx_hq' : 'ltx'}_${Date.now()}.mp4`);
|
||
} else if (!path.isAbsolute(outPath)) {
|
||
outPath = path.resolve(workspacePath, outPath);
|
||
}
|
||
|
||
const imageArg = String(args?.image || '').trim();
|
||
if (imageArg && isCogVideoX) {
|
||
return { success: false, error: 'image-to-video is not supported for model "cogvideox" — use "ltx" or "ltx_hq"' };
|
||
}
|
||
let inputImageFilename: string | undefined;
|
||
if (imageArg) {
|
||
const srcPath = path.isAbsolute(imageArg) ? imageArg : path.resolve(workspacePath, imageArg);
|
||
if (!isPathInsideDir(workspacePath, srcPath)) return { success: false, error: 'Access denied: path escapes workspace' };
|
||
if (!fs.existsSync(srcPath)) return { success: false, error: `Source image not found: ${imageArg}` };
|
||
fs.mkdirSync(COMFY_INPUT_DIR, { recursive: true });
|
||
inputImageFilename = `vid_src_${Date.now()}${path.extname(srcPath) || '.png'}`;
|
||
fs.copyFileSync(srcPath, path.join(COMFY_INPUT_DIR, inputImageFilename));
|
||
}
|
||
const strength = Math.min(1, Math.max(0, toFiniteNumber(args?.strength, 0.5)));
|
||
|
||
const negativePromptRaw = args?.negative_prompt || 'worst quality, blurry, distorted, deformed';
|
||
const [translatedPrompt, translatedNegative] = await Promise.all([
|
||
translatePromptToEnglish(prompt),
|
||
translatePromptToEnglish(negativePromptRaw),
|
||
]);
|
||
|
||
const width = Math.round((args?.width ?? (isCogVideoX ? 720 : 704)) / 32) * 32;
|
||
const height = Math.round((args?.height ?? 480) / 32) * 32;
|
||
const fps = toFiniteNumber(args?.fps, isCogVideoX ? 8 : 24);
|
||
const guidance_scale = Math.min(10, Math.max(1, args?.guidance_scale ?? (isCogVideoX ? 6.0 : 3.0)));
|
||
// CogVideoX-2B is much slower per step than LTX-Video — the model card's default of 50
|
||
// (and even our earlier 40) routinely pushed generation past 8-9 minutes, well beyond any
|
||
// reasonable reverse-proxy read timeout (see NPM proxy_read_timeout incident). Default it
|
||
// lower to keep typical runs under ~4 minutes; still overridable via the steps param.
|
||
const steps = Math.min(50, Math.max(15, args?.steps ?? (isCogVideoX ? 25 : 40)));
|
||
|
||
let num_frames: number;
|
||
let outputMeta: { output: string };
|
||
if (isCogVideoX) {
|
||
num_frames = toFiniteNumber(args?.num_frames, 49);
|
||
const params = {
|
||
prompt: translatedPrompt || prompt,
|
||
negative_prompt: translatedNegative || negativePromptRaw,
|
||
width, height, num_frames, fps, steps, guidance_scale,
|
||
dst: outPath,
|
||
};
|
||
const result = await runVenvPython(COGVIDEOX_SCRIPT(params), 600_000);
|
||
if (result.error || !result.output) {
|
||
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
|
||
}
|
||
outputMeta = { output: result.output };
|
||
} else {
|
||
// LTX-Video wants length as 8k+1 (its temporal VAE downsamples by 8) — snap
|
||
// whatever was requested to the nearest valid value instead of erroring.
|
||
const requestedFrames = toFiniteNumber(args?.num_frames, 65);
|
||
num_frames = Math.max(9, Math.round((requestedFrames - 1) / 8) * 8 + 1);
|
||
const ckpt_name = isHQ ? 'ltxv-2b-0.9.6-dev-04-25.safetensors' : undefined;
|
||
const graph = inputImageFilename
|
||
? ltxImageToVideoWorkflow({
|
||
imageFilename: inputImageFilename,
|
||
prompt: translatedPrompt || prompt,
|
||
negative_prompt: translatedNegative || negativePromptRaw,
|
||
width, height, length: num_frames, fps, steps, guidance_scale, strength,
|
||
seed: randomSeed(),
|
||
ckpt_name,
|
||
})
|
||
: ltxVideoWorkflow({
|
||
prompt: translatedPrompt || prompt,
|
||
negative_prompt: translatedNegative || negativePromptRaw,
|
||
width, height, length: num_frames, fps, steps, guidance_scale,
|
||
seed: randomSeed(),
|
||
ckpt_name,
|
||
});
|
||
// The non-distilled "dev" checkpoint is ~1.4x the weight size of the distilled one and
|
||
// runs full CFG (two forward passes/step instead of one), so it's meaningfully slower —
|
||
// give it more headroom than the regular ltx path's 300s.
|
||
const result = await comfyGenerateVideo(graph, outPath, isHQ ? 420_000 : 300_000);
|
||
if ('error' in result) {
|
||
return { success: false, error: result.error };
|
||
}
|
||
outputMeta = result;
|
||
}
|
||
|
||
const relOut = path.relative(workspacePath, outputMeta.output).replace(/\\/g, '/');
|
||
const sizeMB = (fs.statSync(outputMeta.output).size / 1024 / 1024).toFixed(2);
|
||
const durationSec = (num_frames / fps).toFixed(1);
|
||
|
||
return {
|
||
success: true,
|
||
stdout: [
|
||
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
|
||
`Generated: ${width} × ${height} px | ${durationSec}s (${num_frames}f @ ${fps}fps) | ${sizeMB} MB`,
|
||
'',
|
||
`[${path.basename(outputMeta.output)}](/api/files/${relOut})`,
|
||
].filter((line): line is string => line !== null).join('\n'),
|
||
data: { output: outputMeta.output, width, height, num_frames, fps, rel_path: relOut },
|
||
};
|
||
},
|
||
};
|