v4.1.8: 이미지/동영상 생성 스튜디오 앱 + SDXL/FLUX 품질 개선

- 이미지·동영상 생성 전용 앱(studio-app.html) + API 라우트 추가, 갤러리 기능 포함
- SD1.5 → SDXL 교체 (프롬프트 반영력/해상도 개선), 한글 프롬프트 자동 영어 번역 추가
- image_generate에 quality="high" 옵션 추가 (FLUX.1-schnell 4비트 양자화, 게이트 없는 커뮤니티 미러 사용)
- video_generate에 guidance_scale 노출, 기본 steps 상향
- 생성기 서브프로세스가 결과 없이 죽는 경우의 크래시 방지 + 에러 로깅 강화
- 멀티 GPU 환경에서 시스템 통계 GPU 미터가 깨지던 nvidia-smi 파싱 버그 수정

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-15 02:38:58 +09:00
co-authored by Claude Sonnet 5
parent 2427aeeaba
commit 88bad1e7d9
7 changed files with 624 additions and 51 deletions
+189 -39
View File
@@ -4,8 +4,39 @@ import fs from 'fs';
import { ToolResult } from '../types.js';
import { getWorkspacePath } from '../config/paths.js';
import { buildImageMarkdown } from './image.js';
import { getOllamaConfig } from './web.js';
// Local diffusion models (SD1.5, LTX-Video) run in a dedicated venv with their
// SDXL/LTX-Video's text encoders (CLIP/T5) are trained overwhelmingly on English
// captions, so non-English prompts (e.g. Korean) produce poor prompt adherence.
// Route non-ASCII prompts through the configured Ollama chat model for a quick
// English translation before handing them to the diffusion pipeline.
async function translatePromptToEnglish(text: string): Promise<string | null> {
if (!text || !/[^\x00-\x7F]/.test(text)) return null;
try {
const { endpoint, model } = getOllamaConfig();
const res = await fetch(`${endpoint}/api/chat`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
model,
messages: [{
role: 'user',
content: `Translate the following image/video generation prompt into natural, descriptive English. Output ONLY the translated English text — no quotes, no explanation:\n\n${text}`,
}],
stream: false,
}),
signal: AbortSignal.timeout(30_000),
});
if (!res.ok) return null;
const data: any = await res.json();
const translated = String(data.message?.content || '').trim();
return translated || null;
} catch {
return null;
}
}
// Local diffusion models (SDXL, LTX-Video) run in a dedicated venv with their
// own torch/diffusers stack, pinned to the second GPU (04:00.0 — kept free of
// the voice engine that permanently resides on GPU0). See
// /srv/homeclaw/.smallclaw/imagegen-venv.
@@ -17,17 +48,31 @@ function runVenvPython(script: string, timeoutMs: number): Promise<any> {
return new Promise((resolve) => {
const child = spawn(VENV_PYTHON, ['-c', script], {
timeout: timeoutMs,
env: { ...process.env, HF_HOME, CUDA_VISIBLE_DEVICES: GEN_GPU },
// BNB_CUDA_VERSION: bitsandbytes (used by the FLUX high-quality path) ships no
// prebuilt binary for our CUDA 13.2 torch build yet — pin it to the newest
// available (13.0) binary, which is ABI-compatible. No-op for SDXL/LTX, which
// don't use bitsandbytes.
env: { ...process.env, HF_HOME, CUDA_VISIBLE_DEVICES: GEN_GPU, BNB_CUDA_VERSION: '130' },
});
let out = '';
let err = '';
child.stdout.on('data', (d: Buffer) => { out += d.toString('utf8'); });
child.stderr.on('data', (d: Buffer) => { err += d.toString('utf8'); });
child.on('close', () => {
child.on('close', (code: number | null, signal: string | null) => {
const marker = out.lastIndexOf('###RESULT###');
const jsonPart = marker !== -1 ? out.slice(marker + '###RESULT###'.length) : out;
if (marker === -1) {
// Process exited (crashed, OOM-killed, or timed out) without ever printing a
// result marker — never silently fall through to a success-shaped {}, since
// that leaves result.output undefined and crashes the caller downstream.
resolve({
error: `Generator process exited without output (code=${code}, signal=${signal})`,
trace: (err || out).slice(-1500),
});
return;
}
const jsonPart = out.slice(marker + '###RESULT###'.length);
try {
resolve(JSON.parse(jsonPart.trim() || '{}'));
resolve(JSON.parse(jsonPart.trim()));
} catch {
resolve({ error: 'Failed to parse generator output', raw: (jsonPart || err).slice(-1500) });
}
@@ -37,19 +82,20 @@ function runVenvPython(script: string, timeoutMs: number): Promise<any> {
}
// ---------------------------------------------------------------------------
// image_generate — local Stable Diffusion 1.5 text-to-image
// image_generate — local SDXL text-to-image
// ---------------------------------------------------------------------------
const SD15_SCRIPT = (p: Record<string, any>) => `
const SDXL_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch
from diffusers import StableDiffusionPipeline
from diffusers import StableDiffusionXLPipeline
pipe = StableDiffusionPipeline.from_pretrained(
"stable-diffusion-v1-5/stable-diffusion-v1-5",
torch_dtype=torch.float16, safety_checker=None,
pipe = StableDiffusionXLPipeline.from_pretrained(
"stabilityai/stable-diffusion-xl-base-1.0",
torch_dtype=torch.float16, variant="fp16", use_safetensors=True,
)
pipe = pipe.to("cuda")
pipe.enable_vae_slicing()
kwargs = dict(
prompt=${JSON.stringify(p.prompt)},
@@ -73,20 +119,87 @@ except Exception as e:
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
// ---------------------------------------------------------------------------
// image_generate (quality: 'high') — local FLUX.1-schnell, 4-bit quantized
// ---------------------------------------------------------------------------
// Community mirror of the (gated) official repo — same Apache-2.0 weights,
// just rehosted without the HF license-gate. Needed because neither FLUX.1-dev
// nor -schnell can be pulled from black-forest-labs/* without an HF token tied
// to an account that has clicked through the license on huggingface.co.
const FLUX_MODEL_ID = 'Niansuh/FLUX.1-schnell';
const FLUX_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch, shutil
from diffusers import FluxPipeline, FluxTransformer2DModel, BitsAndBytesConfig as DBnBConfig
from transformers import T5EncoderModel, BitsAndBytesConfig as TBnBConfig
from huggingface_hub import snapshot_download
MODEL_ID = "${FLUX_MODEL_ID}"
# This mirror ships scheduler/config.json instead of the scheduler_config.json
# filename diffusers expects — patch it once per cache (idempotent).
snap_dir = snapshot_download(MODEL_ID, allow_patterns=["scheduler/config.json"])
sched_cfg = os.path.join(snap_dir, "scheduler", "scheduler_config.json")
if not os.path.exists(sched_cfg):
shutil.copy(os.path.join(snap_dir, "scheduler", "config.json"), sched_cfg)
transformer_4bit = FluxTransformer2DModel.from_pretrained(
MODEL_ID, subfolder="transformer",
quantization_config=DBnBConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16),
torch_dtype=torch.bfloat16,
)
text_encoder_2_4bit = T5EncoderModel.from_pretrained(
MODEL_ID, subfolder="text_encoder_2",
quantization_config=TBnBConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16),
torch_dtype=torch.bfloat16,
)
pipe = FluxPipeline.from_pretrained(
MODEL_ID,
transformer=transformer_4bit,
text_encoder_2=text_encoder_2_4bit,
torch_dtype=torch.bfloat16,
)
pipe.enable_model_cpu_offload()
kwargs = dict(
prompt=${JSON.stringify(p.prompt)},
guidance_scale=0.0,
num_inference_steps=int(${p.steps}),
max_sequence_length=256,
width=int(${p.width}), height=int(${p.height}),
)
${p.seed != null ? `kwargs["generator"] = torch.Generator("cpu").manual_seed(int(${p.seed}))` : ''}
image = pipe(**kwargs).images[0]
dst = ${JSON.stringify(p.dst)}
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
image.save(dst)
print("###RESULT###" + json.dumps({
"output": dst, "width": image.width, "height": image.height,
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
}))
except Exception as e:
import traceback
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
export const imageGenerateTool = {
name: 'image_generate',
description: [
'Generate an image from a text prompt using a local Stable Diffusion 1.5 model (runs on-machine GPU, no external API).',
'Best for quick, casual illustrations at up to ~768px. Takes a few seconds.',
'Generate an image from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
'quality="fast" (default): SDXL, ~10-20 seconds, default resolution 1024×1024, good for casual/quick illustrations.',
'quality="high": FLUX.1-schnell (4-bit quantized), ~40-60 seconds total (model load + generation), noticeably more photorealistic detail and prompt accuracy. Use only when the user explicitly asks for higher quality/detail/photorealism, or for a "고품질" request — otherwise default to fast.',
'Returns the generated image inline in the chat.',
].join('\n'),
schema: {
prompt: 'Text description of the image to generate (English works best for SD1.5)',
negative_prompt: 'Things to avoid in the image (optional, e.g. "blurry, low quality, deformed")',
width: 'Image width in pixels, multiple of 8 (default 512)',
height: 'Image height in pixels, multiple of 8 (default 512)',
steps: 'Denoising steps — more = higher quality but slower (default 25, range 10–50)',
guidance_scale: 'How closely to follow the prompt (default 7.5, range 1–20)',
prompt: 'Text description of the image to generate (English works best)',
quality: '"fast" (SDXL, default) or "high" (FLUX.1-schnell, much slower but noticeably better detail/realism)',
negative_prompt: 'Things to avoid in the image (optional, e.g. "blurry, low quality, deformed") — ignored in quality="high" mode (FLUX.1-schnell does not support it)',
width: 'Image width in pixels, multiple of 8 (default 1024)',
height: 'Image height in pixels, multiple of 8 (default 1024)',
steps: 'Denoising steps — more = higher quality but slower (fast mode: default 30, range 15–50; high mode: default 4, range 1–8)',
guidance_scale: 'How closely to follow the prompt (default 7.0, range 1–20) — fast mode only, ignored in quality="high"',
seed: 'Random seed for reproducibility (optional)',
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
},
@@ -94,6 +207,7 @@ export const imageGenerateTool = {
type: 'object',
properties: {
prompt: { type: 'string' },
quality: { type: 'string', enum: ['fast', 'high'] },
negative_prompt: { type: 'string' },
width: { type: 'number' },
height: { type: 'number' },
@@ -108,36 +222,59 @@ export const imageGenerateTool = {
execute: async (args: any): Promise<ToolResult> => {
const prompt = String(args?.prompt || '').trim();
if (!prompt) return { success: false, error: 'prompt is required' };
const isHighQuality = args?.quality === 'high';
const workspacePath = getWorkspacePath(args);
let outPath = String(args?.output || '').trim();
if (!outPath) {
outPath = path.join(workspacePath, `sd15_${Date.now()}.png`);
outPath = path.join(workspacePath, `${isHighQuality ? 'flux' : 'sdxl'}_${Date.now()}.png`);
} else if (!path.isAbsolute(outPath)) {
outPath = path.resolve(workspacePath, outPath);
}
const params = {
prompt,
negative_prompt: args?.negative_prompt || '',
width: Math.round((args?.width ?? 512) / 8) * 8,
height: Math.round((args?.height ?? 512) / 8) * 8,
steps: Math.min(50, Math.max(10, args?.steps ?? 25)),
guidance_scale: args?.guidance_scale ?? 7.5,
seed: args?.seed,
dst: outPath,
};
const negativePromptRaw = args?.negative_prompt || '';
const [translatedPrompt, translatedNegative] = await Promise.all([
translatePromptToEnglish(prompt),
isHighQuality ? Promise.resolve(null) : translatePromptToEnglish(negativePromptRaw),
]);
const result = await runVenvPython(SD15_SCRIPT(params), 180_000);
if (result.error) return { success: false, error: result.error, stderr: result.trace || result.raw };
const width = Math.round((args?.width ?? 1024) / 16) * 16;
const height = Math.round((args?.height ?? 1024) / 16) * 16;
let result: any;
if (isHighQuality) {
const params = {
prompt: translatedPrompt || prompt,
width, height,
steps: Math.min(8, Math.max(1, args?.steps ?? 4)),
seed: args?.seed,
dst: outPath,
};
result = await runVenvPython(FLUX_SCRIPT(params), 180_000);
} else {
const params = {
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height,
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
guidance_scale: args?.guidance_scale ?? 7.0,
seed: args?.seed,
dst: outPath,
};
result = await runVenvPython(SDXL_SCRIPT(params), 180_000);
}
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
}
return {
success: true,
stdout: [
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
`Generated: ${result.width} × ${result.height} px`,
'',
buildImageMarkdown(result.output, workspacePath),
].join('\n'),
].filter((line): line is string => line !== null).join('\n'),
data: { ...result, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
};
},
@@ -162,6 +299,7 @@ try:
width=int(${p.width}), height=int(${p.height}),
num_frames=int(${p.num_frames}),
num_inference_steps=int(${p.steps}),
guidance_scale=float(${p.guidance_scale}),
).frames[0]
dst = ${JSON.stringify(p.dst)}
@@ -194,7 +332,8 @@ export const videoGenerateTool = {
height: 'Video height in pixels, multiple of 32 (default 480)',
num_frames: 'Number of frames — duration = num_frames / fps (default 65, ~2.7s at 24fps)',
fps: 'Output frame rate (default 24)',
steps: 'Denoising steps — more = higher quality but slower (default 30, range 15–50)',
steps: 'Denoising steps — more = higher quality but slower (default 40, range 15–50)',
guidance_scale: 'How closely to follow the prompt (default 3.0, range 1–10 — LTX-Video responds best to a narrower range than SDXL)',
output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)',
},
jsonSchema: {
@@ -207,6 +346,7 @@ export const videoGenerateTool = {
num_frames: { type: 'number' },
fps: { type: 'number' },
steps: { type: 'number' },
guidance_scale: { type: 'number' },
output: { type: 'string' },
},
required: ['prompt'],
@@ -224,19 +364,28 @@ export const videoGenerateTool = {
outPath = path.resolve(workspacePath, outPath);
}
const negativePromptRaw = args?.negative_prompt || 'worst quality, blurry, distorted, deformed';
const [translatedPrompt, translatedNegative] = await Promise.all([
translatePromptToEnglish(prompt),
translatePromptToEnglish(negativePromptRaw),
]);
const params = {
prompt,
negative_prompt: args?.negative_prompt || 'worst quality, blurry, distorted, deformed',
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width: Math.round((args?.width ?? 704) / 32) * 32,
height: Math.round((args?.height ?? 480) / 32) * 32,
num_frames: args?.num_frames ?? 65,
fps: args?.fps ?? 24,
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? 3.0)),
dst: outPath,
};
const result = await runVenvPython(LTX_SCRIPT(params), 600_000);
if (result.error) return { success: false, error: result.error, stderr: result.trace || result.raw };
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
}
const relOut = path.relative(workspacePath, result.output).replace(/\\/g, '/');
const sizeMB = (result.size_bytes / 1024 / 1024).toFixed(2);
@@ -245,10 +394,11 @@ export const videoGenerateTool = {
return {
success: true,
stdout: [
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
`Generated: ${result.width} × ${result.height} px | ${durationSec}s (${result.num_frames}f @ ${result.fps}fps) | ${sizeMB} MB`,
'',
`[${path.basename(result.output)}](/api/files/${relOut})`,
].join('\n'),
].filter((line): line is string => line !== null).join('\n'),
data: { ...result, rel_path: relOut },
};
},
+1 -1
View File
@@ -992,7 +992,7 @@ export const webFetchTool = {
},
};
function getOllamaConfig(): { endpoint: string; model: string } {
export function getOllamaConfig(): { endpoint: string; model: string } {
try {
const cm = getConfig();
const data = cm.getConfig();