v4.1.8: 이미지/동영상 생성 스튜디오 앱 + SDXL/FLUX 품질 개선
- 이미지·동영상 생성 전용 앱(studio-app.html) + API 라우트 추가, 갤러리 기능 포함 - SD1.5 → SDXL 교체 (프롬프트 반영력/해상도 개선), 한글 프롬프트 자동 영어 번역 추가 - image_generate에 quality="high" 옵션 추가 (FLUX.1-schnell 4비트 양자화, 게이트 없는 커뮤니티 미러 사용) - video_generate에 guidance_scale 노출, 기본 steps 상향 - 생성기 서브프로세스가 결과 없이 죽는 경우의 크래시 방지 + 에러 로깅 강화 - 멀티 GPU 환경에서 시스템 통계 GPU 미터가 깨지던 nvidia-smi 파싱 버그 수정 Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
+189
-39
@@ -4,8 +4,39 @@ import fs from 'fs';
|
||||
import { ToolResult } from '../types.js';
|
||||
import { getWorkspacePath } from '../config/paths.js';
|
||||
import { buildImageMarkdown } from './image.js';
|
||||
import { getOllamaConfig } from './web.js';
|
||||
|
||||
// Local diffusion models (SD1.5, LTX-Video) run in a dedicated venv with their
|
||||
// SDXL/LTX-Video's text encoders (CLIP/T5) are trained overwhelmingly on English
|
||||
// captions, so non-English prompts (e.g. Korean) produce poor prompt adherence.
|
||||
// Route non-ASCII prompts through the configured Ollama chat model for a quick
|
||||
// English translation before handing them to the diffusion pipeline.
|
||||
async function translatePromptToEnglish(text: string): Promise<string | null> {
|
||||
if (!text || !/[^\x00-\x7F]/.test(text)) return null;
|
||||
try {
|
||||
const { endpoint, model } = getOllamaConfig();
|
||||
const res = await fetch(`${endpoint}/api/chat`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({
|
||||
model,
|
||||
messages: [{
|
||||
role: 'user',
|
||||
content: `Translate the following image/video generation prompt into natural, descriptive English. Output ONLY the translated English text — no quotes, no explanation:\n\n${text}`,
|
||||
}],
|
||||
stream: false,
|
||||
}),
|
||||
signal: AbortSignal.timeout(30_000),
|
||||
});
|
||||
if (!res.ok) return null;
|
||||
const data: any = await res.json();
|
||||
const translated = String(data.message?.content || '').trim();
|
||||
return translated || null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
// Local diffusion models (SDXL, LTX-Video) run in a dedicated venv with their
|
||||
// own torch/diffusers stack, pinned to the second GPU (04:00.0 — kept free of
|
||||
// the voice engine that permanently resides on GPU0). See
|
||||
// /srv/homeclaw/.smallclaw/imagegen-venv.
|
||||
@@ -17,17 +48,31 @@ function runVenvPython(script: string, timeoutMs: number): Promise<any> {
|
||||
return new Promise((resolve) => {
|
||||
const child = spawn(VENV_PYTHON, ['-c', script], {
|
||||
timeout: timeoutMs,
|
||||
env: { ...process.env, HF_HOME, CUDA_VISIBLE_DEVICES: GEN_GPU },
|
||||
// BNB_CUDA_VERSION: bitsandbytes (used by the FLUX high-quality path) ships no
|
||||
// prebuilt binary for our CUDA 13.2 torch build yet — pin it to the newest
|
||||
// available (13.0) binary, which is ABI-compatible. No-op for SDXL/LTX, which
|
||||
// don't use bitsandbytes.
|
||||
env: { ...process.env, HF_HOME, CUDA_VISIBLE_DEVICES: GEN_GPU, BNB_CUDA_VERSION: '130' },
|
||||
});
|
||||
let out = '';
|
||||
let err = '';
|
||||
child.stdout.on('data', (d: Buffer) => { out += d.toString('utf8'); });
|
||||
child.stderr.on('data', (d: Buffer) => { err += d.toString('utf8'); });
|
||||
child.on('close', () => {
|
||||
child.on('close', (code: number | null, signal: string | null) => {
|
||||
const marker = out.lastIndexOf('###RESULT###');
|
||||
const jsonPart = marker !== -1 ? out.slice(marker + '###RESULT###'.length) : out;
|
||||
if (marker === -1) {
|
||||
// Process exited (crashed, OOM-killed, or timed out) without ever printing a
|
||||
// result marker — never silently fall through to a success-shaped {}, since
|
||||
// that leaves result.output undefined and crashes the caller downstream.
|
||||
resolve({
|
||||
error: `Generator process exited without output (code=${code}, signal=${signal})`,
|
||||
trace: (err || out).slice(-1500),
|
||||
});
|
||||
return;
|
||||
}
|
||||
const jsonPart = out.slice(marker + '###RESULT###'.length);
|
||||
try {
|
||||
resolve(JSON.parse(jsonPart.trim() || '{}'));
|
||||
resolve(JSON.parse(jsonPart.trim()));
|
||||
} catch {
|
||||
resolve({ error: 'Failed to parse generator output', raw: (jsonPart || err).slice(-1500) });
|
||||
}
|
||||
@@ -37,19 +82,20 @@ function runVenvPython(script: string, timeoutMs: number): Promise<any> {
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_generate — local Stable Diffusion 1.5 text-to-image
|
||||
// image_generate — local SDXL text-to-image
|
||||
// ---------------------------------------------------------------------------
|
||||
const SD15_SCRIPT = (p: Record<string, any>) => `
|
||||
const SDXL_SCRIPT = (p: Record<string, any>) => `
|
||||
import os, json, sys
|
||||
try:
|
||||
import torch
|
||||
from diffusers import StableDiffusionPipeline
|
||||
from diffusers import StableDiffusionXLPipeline
|
||||
|
||||
pipe = StableDiffusionPipeline.from_pretrained(
|
||||
"stable-diffusion-v1-5/stable-diffusion-v1-5",
|
||||
torch_dtype=torch.float16, safety_checker=None,
|
||||
pipe = StableDiffusionXLPipeline.from_pretrained(
|
||||
"stabilityai/stable-diffusion-xl-base-1.0",
|
||||
torch_dtype=torch.float16, variant="fp16", use_safetensors=True,
|
||||
)
|
||||
pipe = pipe.to("cuda")
|
||||
pipe.enable_vae_slicing()
|
||||
|
||||
kwargs = dict(
|
||||
prompt=${JSON.stringify(p.prompt)},
|
||||
@@ -73,20 +119,87 @@ except Exception as e:
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_generate (quality: 'high') — local FLUX.1-schnell, 4-bit quantized
|
||||
// ---------------------------------------------------------------------------
|
||||
// Community mirror of the (gated) official repo — same Apache-2.0 weights,
|
||||
// just rehosted without the HF license-gate. Needed because neither FLUX.1-dev
|
||||
// nor -schnell can be pulled from black-forest-labs/* without an HF token tied
|
||||
// to an account that has clicked through the license on huggingface.co.
|
||||
const FLUX_MODEL_ID = 'Niansuh/FLUX.1-schnell';
|
||||
const FLUX_SCRIPT = (p: Record<string, any>) => `
|
||||
import os, json, sys
|
||||
try:
|
||||
import torch, shutil
|
||||
from diffusers import FluxPipeline, FluxTransformer2DModel, BitsAndBytesConfig as DBnBConfig
|
||||
from transformers import T5EncoderModel, BitsAndBytesConfig as TBnBConfig
|
||||
from huggingface_hub import snapshot_download
|
||||
|
||||
MODEL_ID = "${FLUX_MODEL_ID}"
|
||||
|
||||
# This mirror ships scheduler/config.json instead of the scheduler_config.json
|
||||
# filename diffusers expects — patch it once per cache (idempotent).
|
||||
snap_dir = snapshot_download(MODEL_ID, allow_patterns=["scheduler/config.json"])
|
||||
sched_cfg = os.path.join(snap_dir, "scheduler", "scheduler_config.json")
|
||||
if not os.path.exists(sched_cfg):
|
||||
shutil.copy(os.path.join(snap_dir, "scheduler", "config.json"), sched_cfg)
|
||||
|
||||
transformer_4bit = FluxTransformer2DModel.from_pretrained(
|
||||
MODEL_ID, subfolder="transformer",
|
||||
quantization_config=DBnBConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16),
|
||||
torch_dtype=torch.bfloat16,
|
||||
)
|
||||
text_encoder_2_4bit = T5EncoderModel.from_pretrained(
|
||||
MODEL_ID, subfolder="text_encoder_2",
|
||||
quantization_config=TBnBConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16),
|
||||
torch_dtype=torch.bfloat16,
|
||||
)
|
||||
pipe = FluxPipeline.from_pretrained(
|
||||
MODEL_ID,
|
||||
transformer=transformer_4bit,
|
||||
text_encoder_2=text_encoder_2_4bit,
|
||||
torch_dtype=torch.bfloat16,
|
||||
)
|
||||
pipe.enable_model_cpu_offload()
|
||||
|
||||
kwargs = dict(
|
||||
prompt=${JSON.stringify(p.prompt)},
|
||||
guidance_scale=0.0,
|
||||
num_inference_steps=int(${p.steps}),
|
||||
max_sequence_length=256,
|
||||
width=int(${p.width}), height=int(${p.height}),
|
||||
)
|
||||
${p.seed != null ? `kwargs["generator"] = torch.Generator("cpu").manual_seed(int(${p.seed}))` : ''}
|
||||
|
||||
image = pipe(**kwargs).images[0]
|
||||
dst = ${JSON.stringify(p.dst)}
|
||||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||||
image.save(dst)
|
||||
print("###RESULT###" + json.dumps({
|
||||
"output": dst, "width": image.width, "height": image.height,
|
||||
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
|
||||
}))
|
||||
except Exception as e:
|
||||
import traceback
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
|
||||
export const imageGenerateTool = {
|
||||
name: 'image_generate',
|
||||
description: [
|
||||
'Generate an image from a text prompt using a local Stable Diffusion 1.5 model (runs on-machine GPU, no external API).',
|
||||
'Best for quick, casual illustrations at up to ~768px. Takes a few seconds.',
|
||||
'Generate an image from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
|
||||
'quality="fast" (default): SDXL, ~10-20 seconds, default resolution 1024×1024, good for casual/quick illustrations.',
|
||||
'quality="high": FLUX.1-schnell (4-bit quantized), ~40-60 seconds total (model load + generation), noticeably more photorealistic detail and prompt accuracy. Use only when the user explicitly asks for higher quality/detail/photorealism, or for a "고품질" request — otherwise default to fast.',
|
||||
'Returns the generated image inline in the chat.',
|
||||
].join('\n'),
|
||||
schema: {
|
||||
prompt: 'Text description of the image to generate (English works best for SD1.5)',
|
||||
negative_prompt: 'Things to avoid in the image (optional, e.g. "blurry, low quality, deformed")',
|
||||
width: 'Image width in pixels, multiple of 8 (default 512)',
|
||||
height: 'Image height in pixels, multiple of 8 (default 512)',
|
||||
steps: 'Denoising steps — more = higher quality but slower (default 25, range 10–50)',
|
||||
guidance_scale: 'How closely to follow the prompt (default 7.5, range 1–20)',
|
||||
prompt: 'Text description of the image to generate (English works best)',
|
||||
quality: '"fast" (SDXL, default) or "high" (FLUX.1-schnell, much slower but noticeably better detail/realism)',
|
||||
negative_prompt: 'Things to avoid in the image (optional, e.g. "blurry, low quality, deformed") — ignored in quality="high" mode (FLUX.1-schnell does not support it)',
|
||||
width: 'Image width in pixels, multiple of 8 (default 1024)',
|
||||
height: 'Image height in pixels, multiple of 8 (default 1024)',
|
||||
steps: 'Denoising steps — more = higher quality but slower (fast mode: default 30, range 15–50; high mode: default 4, range 1–8)',
|
||||
guidance_scale: 'How closely to follow the prompt (default 7.0, range 1–20) — fast mode only, ignored in quality="high"',
|
||||
seed: 'Random seed for reproducibility (optional)',
|
||||
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
|
||||
},
|
||||
@@ -94,6 +207,7 @@ export const imageGenerateTool = {
|
||||
type: 'object',
|
||||
properties: {
|
||||
prompt: { type: 'string' },
|
||||
quality: { type: 'string', enum: ['fast', 'high'] },
|
||||
negative_prompt: { type: 'string' },
|
||||
width: { type: 'number' },
|
||||
height: { type: 'number' },
|
||||
@@ -108,36 +222,59 @@ export const imageGenerateTool = {
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const prompt = String(args?.prompt || '').trim();
|
||||
if (!prompt) return { success: false, error: 'prompt is required' };
|
||||
const isHighQuality = args?.quality === 'high';
|
||||
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
let outPath = String(args?.output || '').trim();
|
||||
if (!outPath) {
|
||||
outPath = path.join(workspacePath, `sd15_${Date.now()}.png`);
|
||||
outPath = path.join(workspacePath, `${isHighQuality ? 'flux' : 'sdxl'}_${Date.now()}.png`);
|
||||
} else if (!path.isAbsolute(outPath)) {
|
||||
outPath = path.resolve(workspacePath, outPath);
|
||||
}
|
||||
|
||||
const params = {
|
||||
prompt,
|
||||
negative_prompt: args?.negative_prompt || '',
|
||||
width: Math.round((args?.width ?? 512) / 8) * 8,
|
||||
height: Math.round((args?.height ?? 512) / 8) * 8,
|
||||
steps: Math.min(50, Math.max(10, args?.steps ?? 25)),
|
||||
guidance_scale: args?.guidance_scale ?? 7.5,
|
||||
seed: args?.seed,
|
||||
dst: outPath,
|
||||
};
|
||||
const negativePromptRaw = args?.negative_prompt || '';
|
||||
const [translatedPrompt, translatedNegative] = await Promise.all([
|
||||
translatePromptToEnglish(prompt),
|
||||
isHighQuality ? Promise.resolve(null) : translatePromptToEnglish(negativePromptRaw),
|
||||
]);
|
||||
|
||||
const result = await runVenvPython(SD15_SCRIPT(params), 180_000);
|
||||
if (result.error) return { success: false, error: result.error, stderr: result.trace || result.raw };
|
||||
const width = Math.round((args?.width ?? 1024) / 16) * 16;
|
||||
const height = Math.round((args?.height ?? 1024) / 16) * 16;
|
||||
|
||||
let result: any;
|
||||
if (isHighQuality) {
|
||||
const params = {
|
||||
prompt: translatedPrompt || prompt,
|
||||
width, height,
|
||||
steps: Math.min(8, Math.max(1, args?.steps ?? 4)),
|
||||
seed: args?.seed,
|
||||
dst: outPath,
|
||||
};
|
||||
result = await runVenvPython(FLUX_SCRIPT(params), 180_000);
|
||||
} else {
|
||||
const params = {
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width, height,
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
|
||||
guidance_scale: args?.guidance_scale ?? 7.0,
|
||||
seed: args?.seed,
|
||||
dst: outPath,
|
||||
};
|
||||
result = await runVenvPython(SDXL_SCRIPT(params), 180_000);
|
||||
}
|
||||
if (result.error || !result.output) {
|
||||
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stdout: [
|
||||
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
|
||||
`Generated: ${result.width} × ${result.height} px`,
|
||||
'',
|
||||
buildImageMarkdown(result.output, workspacePath),
|
||||
].join('\n'),
|
||||
].filter((line): line is string => line !== null).join('\n'),
|
||||
data: { ...result, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
|
||||
};
|
||||
},
|
||||
@@ -162,6 +299,7 @@ try:
|
||||
width=int(${p.width}), height=int(${p.height}),
|
||||
num_frames=int(${p.num_frames}),
|
||||
num_inference_steps=int(${p.steps}),
|
||||
guidance_scale=float(${p.guidance_scale}),
|
||||
).frames[0]
|
||||
|
||||
dst = ${JSON.stringify(p.dst)}
|
||||
@@ -194,7 +332,8 @@ export const videoGenerateTool = {
|
||||
height: 'Video height in pixels, multiple of 32 (default 480)',
|
||||
num_frames: 'Number of frames — duration = num_frames / fps (default 65, ~2.7s at 24fps)',
|
||||
fps: 'Output frame rate (default 24)',
|
||||
steps: 'Denoising steps — more = higher quality but slower (default 30, range 15–50)',
|
||||
steps: 'Denoising steps — more = higher quality but slower (default 40, range 15–50)',
|
||||
guidance_scale: 'How closely to follow the prompt (default 3.0, range 1–10 — LTX-Video responds best to a narrower range than SDXL)',
|
||||
output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)',
|
||||
},
|
||||
jsonSchema: {
|
||||
@@ -207,6 +346,7 @@ export const videoGenerateTool = {
|
||||
num_frames: { type: 'number' },
|
||||
fps: { type: 'number' },
|
||||
steps: { type: 'number' },
|
||||
guidance_scale: { type: 'number' },
|
||||
output: { type: 'string' },
|
||||
},
|
||||
required: ['prompt'],
|
||||
@@ -224,19 +364,28 @@ export const videoGenerateTool = {
|
||||
outPath = path.resolve(workspacePath, outPath);
|
||||
}
|
||||
|
||||
const negativePromptRaw = args?.negative_prompt || 'worst quality, blurry, distorted, deformed';
|
||||
const [translatedPrompt, translatedNegative] = await Promise.all([
|
||||
translatePromptToEnglish(prompt),
|
||||
translatePromptToEnglish(negativePromptRaw),
|
||||
]);
|
||||
|
||||
const params = {
|
||||
prompt,
|
||||
negative_prompt: args?.negative_prompt || 'worst quality, blurry, distorted, deformed',
|
||||
prompt: translatedPrompt || prompt,
|
||||
negative_prompt: translatedNegative || negativePromptRaw,
|
||||
width: Math.round((args?.width ?? 704) / 32) * 32,
|
||||
height: Math.round((args?.height ?? 480) / 32) * 32,
|
||||
num_frames: args?.num_frames ?? 65,
|
||||
fps: args?.fps ?? 24,
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
|
||||
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? 3.0)),
|
||||
dst: outPath,
|
||||
};
|
||||
|
||||
const result = await runVenvPython(LTX_SCRIPT(params), 600_000);
|
||||
if (result.error) return { success: false, error: result.error, stderr: result.trace || result.raw };
|
||||
if (result.error || !result.output) {
|
||||
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
|
||||
}
|
||||
|
||||
const relOut = path.relative(workspacePath, result.output).replace(/\\/g, '/');
|
||||
const sizeMB = (result.size_bytes / 1024 / 1024).toFixed(2);
|
||||
@@ -245,10 +394,11 @@ export const videoGenerateTool = {
|
||||
return {
|
||||
success: true,
|
||||
stdout: [
|
||||
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
|
||||
`Generated: ${result.width} × ${result.height} px | ${durationSec}s (${result.num_frames}f @ ${result.fps}fps) | ${sizeMB} MB`,
|
||||
'',
|
||||
`[${path.basename(result.output)}](/api/files/${relOut})`,
|
||||
].join('\n'),
|
||||
].filter((line): line is string => line !== null).join('\n'),
|
||||
data: { ...result, rel_path: relOut },
|
||||
};
|
||||
},
|
||||
|
||||
+1
-1
@@ -992,7 +992,7 @@ export const webFetchTool = {
|
||||
},
|
||||
};
|
||||
|
||||
function getOllamaConfig(): { endpoint: string; model: string } {
|
||||
export function getOllamaConfig(): { endpoint: string; model: string } {
|
||||
try {
|
||||
const cm = getConfig();
|
||||
const data = cm.getConfig();
|
||||
|
||||
Reference in New Issue
Block a user