v4.1.8: 이미지/동영상 생성 스튜디오 앱 + SDXL/FLUX 품질 개선

- 이미지·동영상 생성 전용 앱(studio-app.html) + API 라우트 추가, 갤러리 기능 포함
- SD1.5 → SDXL 교체 (프롬프트 반영력/해상도 개선), 한글 프롬프트 자동 영어 번역 추가
- image_generate에 quality="high" 옵션 추가 (FLUX.1-schnell 4비트 양자화, 게이트 없는 커뮤니티 미러 사용)
- video_generate에 guidance_scale 노출, 기본 steps 상향
- 생성기 서브프로세스가 결과 없이 죽는 경우의 크래시 방지 + 에러 로깅 강화
- 멀티 GPU 환경에서 시스템 통계 GPU 미터가 깨지던 nvidia-smi 파싱 버그 수정

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-15 02:38:58 +09:00
co-authored by Claude Sonnet 5
parent 2427aeeaba
commit 88bad1e7d9
7 changed files with 624 additions and 51 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "smallclaw",
"version": "4.1.7",
"version": "4.1.8",
"description": "Local AI agent framework powered by Ollama - OpenClaw alternative",
"main": "dist/index.js",
"bin": {
+119
View File
@@ -0,0 +1,119 @@
import { Express, Request, Response } from 'express';
import fs from 'fs';
import path from 'path';
import { imageGenerateTool, videoGenerateTool } from '../../tools/imagegen.js';
const IMAGE_EXTS = new Set(['.png', '.jpg', '.jpeg', '.webp']);
const VIDEO_EXTS = new Set(['.mp4']);
function galleryDir(workspace: string): string {
return path.join(workspace, 'generated-media');
}
export function registerImagegenRoutes(app: Express): void {
app.post('/api/imagegen/generate-image', async (req: Request, res: Response) => {
const user = (req as any).user;
if (!user) return res.status(401).json({ error: 'Unauthorized' });
console.log(`[imagegen] generate-image start user=${user.username} prompt="${String(req.body?.prompt || '').slice(0, 120)}"`);
try {
const outDir = galleryDir(user.workspace);
fs.mkdirSync(outDir, { recursive: true });
const filename = `img_${Date.now()}.png`;
const result = await imageGenerateTool.execute({
...req.body,
output: path.join('generated-media', filename),
_workspacePath: user.workspace,
});
if (!result.success) {
console.error(`[imagegen] generate-image failed for user=${user.username}: ${result.error}`, result.stderr ? `\nstderr: ${result.stderr}` : '');
return res.status(500).json({ error: result.error, stderr: result.stderr });
}
res.json({
success: true,
url: `/api/files/${result.data.rel_path}`,
width: result.data.width,
height: result.data.height,
});
} catch (err: any) {
// Belt-and-suspenders: never let an unexpected throw fall through to Express's
// default HTML error page — the studio app's fetch always expects JSON back.
console.error(`[imagegen] generate-image threw for user=${user.username}:`, err);
res.status(500).json({ error: err?.message || 'image_generate failed unexpectedly' });
}
});
app.post('/api/imagegen/generate-video', async (req: Request, res: Response) => {
const user = (req as any).user;
if (!user) return res.status(401).json({ error: 'Unauthorized' });
console.log(`[imagegen] generate-video start user=${user.username} prompt="${String(req.body?.prompt || '').slice(0, 120)}"`);
try {
const outDir = galleryDir(user.workspace);
fs.mkdirSync(outDir, { recursive: true });
const filename = `vid_${Date.now()}.mp4`;
const result = await videoGenerateTool.execute({
...req.body,
output: path.join('generated-media', filename),
_workspacePath: user.workspace,
});
if (!result.success) {
console.error(`[imagegen] generate-video failed for user=${user.username}: ${result.error}`, result.stderr ? `\nstderr: ${result.stderr}` : '');
return res.status(500).json({ error: result.error, stderr: result.stderr });
}
res.json({
success: true,
url: `/api/files/${result.data.rel_path}`,
width: result.data.width,
height: result.data.height,
num_frames: result.data.num_frames,
fps: result.data.fps,
});
} catch (err: any) {
console.error(`[imagegen] generate-video threw for user=${user.username}:`, err);
res.status(500).json({ error: err?.message || 'video_generate failed unexpectedly' });
}
});
app.get('/api/imagegen/gallery', (req: Request, res: Response) => {
const user = (req as any).user;
if (!user) return res.status(401).json({ error: 'Unauthorized' });
const outDir = galleryDir(user.workspace);
if (!fs.existsSync(outDir)) return res.json({ items: [] });
const items = fs.readdirSync(outDir)
.map((name) => {
const ext = path.extname(name).toLowerCase();
const type = IMAGE_EXTS.has(ext) ? 'image' : VIDEO_EXTS.has(ext) ? 'video' : null;
if (!type) return null;
const stat = fs.statSync(path.join(outDir, name));
return {
name,
type,
url: `/api/files/generated-media/${encodeURIComponent(name)}`,
size: stat.size,
mtime: stat.mtimeMs,
};
})
.filter((x): x is NonNullable<typeof x> => x !== null)
.sort((a, b) => b.mtime - a.mtime);
res.json({ items });
});
app.delete('/api/imagegen/gallery/:name', (req: Request, res: Response) => {
const user = (req as any).user;
if (!user) return res.status(401).json({ error: 'Unauthorized' });
const name = path.basename(String(req.params.name || ''));
const target = path.join(galleryDir(user.workspace), name);
if (!fs.existsSync(target)) return res.status(404).json({ error: 'Not found' });
fs.unlinkSync(target);
res.json({ success: true });
});
}
+27 -10
View File
@@ -59,6 +59,7 @@ import { writeSkillPackFromContent } from '../skills/processor.js';
import { summarizeSkillForApi } from '../tools/skills.js';
import { getToolRegistry } from '../tools/registry.js';
import { registerSettingsRoutes } from './routes/settings.js';
import { registerImagegenRoutes } from './routes/imagegen.js';
import {
browserOpen,
browserSnapshot,
@@ -5785,7 +5786,7 @@ async function handleChat(
content: isTranslateSession ? `You are a medical translator. Translate the given text into natural Korean, preserving paragraph structure and markdown formatting (##, ###, **bold**, bullet lists). Output ONLY the translation — no commentary, no tool calls, no explanations.` : isProjSession ? `You are a project file designer. Output ONLY the project-files JSON block as instructed. No tool calls. No extra text.` : `${executionModeSystemBlock ? `${executionModeSystemBlock}\n\n` : ''}You are SmallClaw 🦞, a local AI assistant.\nCurrent date: ${dateStr}, ${timeStr}.\nNever search for or link SmallClaw repos unless the user is asking about SmallClaw itself.\nThis app runs on the user's own machine — browser/desktop automation requests are pre-authorized.\nKeep responses SHORT (1-2 sentences). Don't think out loud. Act and report. Greet naturally without tools.
ANTI-HALLUCINATION: When a tool returns a result, report EXACTLY what the tool returned — never contradict or ignore tool output. If a tool says "(no rows)", say so. Never invent data, file contents, table names, or command output. If you don't know something, call a tool to find out or say you don't know.
IMAGE EDITING RULE: NEVER call image_edit (or any editing tool) when a user uploads a photo without explicitly requesting edits. Uploading a photo is NOT a request to edit it. Only call image_edit when the user's message explicitly asks for an edit (e.g. "수채화로 바꿔줘", "회전해줘"). Violating this rule is a critical error.
IMAGE/VIDEO GENERATION: When a user asks to create/draw/generate a NEW image from a description (no existing photo involved), call image_generate (local SD1.5, a few seconds). When they ask for a short video/clip/animation from a description, call video_generate (local LTX-Video, 30–90 seconds — tell the user it'll take a bit before calling it). Both run entirely on local GPU hardware, no external API or cost. Do not confuse these with image_edit, which only modifies an existing uploaded/generated image.${browserRuleBlock}
IMAGE/VIDEO GENERATION: When a user asks to create/draw/generate a NEW image from a description (no existing photo involved), call image_generate (local SDXL, 10-20 seconds by default). If the user explicitly asks for higher quality/detail/photorealism (e.g. "고품질로", "디테일 살려서"), pass quality="high" instead — this uses FLUX.1-schnell, noticeably better detail but 40-60 seconds total, so tell the user it'll take a bit before calling it. When they ask for a short video/clip/animation from a description, call video_generate (local LTX-Video, 30–90 seconds — tell the user it'll take a bit before calling it). All run entirely on local GPU hardware, no external API or cost. Do not confuse these with image_edit, which only modifies an existing uploaded/generated image.${browserRuleBlock}
CHEMISTRY NOTATION: NEVER draw molecular structures as ASCII art (H/C/#/=/\\/| characters arranged to look like a diagram) — it always renders as garbled, misaligned text. Instead: for a formula or reaction, use LaTeX inside $...$ (e.g. $\\ce{C4H10}$, $\\ce{2H2 + O2 -> 2H2O}$ — mhchem extension is loaded). For an actual 2D structure with real bond lines (rings, branches), output a \`\`\`smiles\`\`\` code block containing the SMILES string (e.g. \`\`\`smiles\\nc1ccccc1\\n\`\`\` for benzene, \`\`\`smiles\\nCC(=O)Oc1ccccc1C(=O)O\\n\`\`\` for aspirin) — the client automatically renders it as a proper 2D diagram with bond lines.
CODE OUTPUT: When writing code in a fenced code block, start with a filename comment on line 1: \`# filename: snake_game.py\` (Python), \`// filename: app.js\` (JS/C), \`<!-- filename: index.html -->\` (HTML). Never repeat code already written in this conversation. For modifications to existing files, use coder_overwrite_lines or coder_insert_lines (not coder_write_file — it only works for NEW files). All code changes are presented as diffs for the user to review before being applied. Write code directly — do not ask for permission.
PACKAGE INSTALL: NEVER run pip install, npm install, apt-get, or any package installation command. If a package is missing, just write the code and mention the package name in a comment — let the user decide whether to install it. Do NOT attempt to install packages yourself.
@@ -12671,6 +12672,7 @@ app.post('/api/schedules/parse', (req: any, res: any) => {
// ── Settings routes (extracted to routes/settings.ts) ──────────────────────────
registerSettingsRoutes(app);
registerImagegenRoutes(app);
// GET /api/credentials/status — list which vault keys are currently stored (names only, no values)
app.get('/api/credentials/status', (_req, res) => {
@@ -12904,17 +12906,32 @@ app.get('/api/system-stats', async (req, res) => {
'nvidia-smi --query-gpu=name,utilization.gpu,memory.used,memory.total --format=csv,noheader,nounits',
{ timeout: 3000, encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] },
);
const parts = smiOut.trim().split(',').map((s: string) => s.trim());
if (parts.length >= 4) {
const vramUsedMb = Number(parts[2]);
const vramTotalMb = Number(parts[3]);
// One line per physical GPU — a machine with 2+ cards must split by line
// before splitting by comma, or fields from adjacent lines bleed together
// (e.g. memory.total gets "12288\nNVIDIA GeForce RTX 3060" glued on, NaN's
// the VRAM percent). Aggregate across all detected GPUs: sum VRAM (so the
// meter reflects total usage), max utilization (so activity on any card shows).
const gpuLines = smiOut.trim().split('\n')
.map((line: string) => line.split(',').map((s: string) => s.trim()))
.filter((parts: string[]) => parts.length >= 4);
if (gpuLines.length > 0) {
let vramUsedMbSum = 0;
let vramTotalMbSum = 0;
let utilMax = 0;
const names: string[] = [];
for (const parts of gpuLines) {
vramUsedMbSum += Number(parts[2]) || 0;
vramTotalMbSum += Number(parts[3]) || 0;
utilMax = Math.max(utilMax, Number(parts[1]) || 0);
names.push(parts[0]);
}
gpuStats = {
available: true,
name: parts[0],
gpu_util_percent: Number(parts[1]),
vram_used_percent: vramTotalMb > 0 ? (vramUsedMb / vramTotalMb) * 100 : 0,
vram_used_gb: vramUsedMb / 1024,
vram_total_gb: vramTotalMb / 1024,
name: gpuLines.length > 1 ? `${names[0]} ×${gpuLines.length}` : names[0],
gpu_util_percent: utilMax,
vram_used_percent: vramTotalMbSum > 0 ? (vramUsedMbSum / vramTotalMbSum) * 100 : 0,
vram_used_gb: vramUsedMbSum / 1024,
vram_total_gb: vramTotalMbSum / 1024,
};
}
} catch { /* nvidia-smi already confirmed working at startup; ignore transient errors */ }
+189 -39
View File
@@ -4,8 +4,39 @@ import fs from 'fs';
import { ToolResult } from '../types.js';
import { getWorkspacePath } from '../config/paths.js';
import { buildImageMarkdown } from './image.js';
import { getOllamaConfig } from './web.js';
// Local diffusion models (SD1.5, LTX-Video) run in a dedicated venv with their
// SDXL/LTX-Video's text encoders (CLIP/T5) are trained overwhelmingly on English
// captions, so non-English prompts (e.g. Korean) produce poor prompt adherence.
// Route non-ASCII prompts through the configured Ollama chat model for a quick
// English translation before handing them to the diffusion pipeline.
async function translatePromptToEnglish(text: string): Promise<string | null> {
if (!text || !/[^\x00-\x7F]/.test(text)) return null;
try {
const { endpoint, model } = getOllamaConfig();
const res = await fetch(`${endpoint}/api/chat`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
model,
messages: [{
role: 'user',
content: `Translate the following image/video generation prompt into natural, descriptive English. Output ONLY the translated English text — no quotes, no explanation:\n\n${text}`,
}],
stream: false,
}),
signal: AbortSignal.timeout(30_000),
});
if (!res.ok) return null;
const data: any = await res.json();
const translated = String(data.message?.content || '').trim();
return translated || null;
} catch {
return null;
}
}
// Local diffusion models (SDXL, LTX-Video) run in a dedicated venv with their
// own torch/diffusers stack, pinned to the second GPU (04:00.0 — kept free of
// the voice engine that permanently resides on GPU0). See
// /srv/homeclaw/.smallclaw/imagegen-venv.
@@ -17,17 +48,31 @@ function runVenvPython(script: string, timeoutMs: number): Promise<any> {
return new Promise((resolve) => {
const child = spawn(VENV_PYTHON, ['-c', script], {
timeout: timeoutMs,
env: { ...process.env, HF_HOME, CUDA_VISIBLE_DEVICES: GEN_GPU },
// BNB_CUDA_VERSION: bitsandbytes (used by the FLUX high-quality path) ships no
// prebuilt binary for our CUDA 13.2 torch build yet — pin it to the newest
// available (13.0) binary, which is ABI-compatible. No-op for SDXL/LTX, which
// don't use bitsandbytes.
env: { ...process.env, HF_HOME, CUDA_VISIBLE_DEVICES: GEN_GPU, BNB_CUDA_VERSION: '130' },
});
let out = '';
let err = '';
child.stdout.on('data', (d: Buffer) => { out += d.toString('utf8'); });
child.stderr.on('data', (d: Buffer) => { err += d.toString('utf8'); });
child.on('close', () => {
child.on('close', (code: number | null, signal: string | null) => {
const marker = out.lastIndexOf('###RESULT###');
const jsonPart = marker !== -1 ? out.slice(marker + '###RESULT###'.length) : out;
if (marker === -1) {
// Process exited (crashed, OOM-killed, or timed out) without ever printing a
// result marker — never silently fall through to a success-shaped {}, since
// that leaves result.output undefined and crashes the caller downstream.
resolve({
error: `Generator process exited without output (code=${code}, signal=${signal})`,
trace: (err || out).slice(-1500),
});
return;
}
const jsonPart = out.slice(marker + '###RESULT###'.length);
try {
resolve(JSON.parse(jsonPart.trim() || '{}'));
resolve(JSON.parse(jsonPart.trim()));
} catch {
resolve({ error: 'Failed to parse generator output', raw: (jsonPart || err).slice(-1500) });
}
@@ -37,19 +82,20 @@ function runVenvPython(script: string, timeoutMs: number): Promise<any> {
}
// ---------------------------------------------------------------------------
// image_generate — local Stable Diffusion 1.5 text-to-image
// image_generate — local SDXL text-to-image
// ---------------------------------------------------------------------------
const SD15_SCRIPT = (p: Record<string, any>) => `
const SDXL_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch
from diffusers import StableDiffusionPipeline
from diffusers import StableDiffusionXLPipeline
pipe = StableDiffusionPipeline.from_pretrained(
"stable-diffusion-v1-5/stable-diffusion-v1-5",
torch_dtype=torch.float16, safety_checker=None,
pipe = StableDiffusionXLPipeline.from_pretrained(
"stabilityai/stable-diffusion-xl-base-1.0",
torch_dtype=torch.float16, variant="fp16", use_safetensors=True,
)
pipe = pipe.to("cuda")
pipe.enable_vae_slicing()
kwargs = dict(
prompt=${JSON.stringify(p.prompt)},
@@ -73,20 +119,87 @@ except Exception as e:
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
// ---------------------------------------------------------------------------
// image_generate (quality: 'high') — local FLUX.1-schnell, 4-bit quantized
// ---------------------------------------------------------------------------
// Community mirror of the (gated) official repo — same Apache-2.0 weights,
// just rehosted without the HF license-gate. Needed because neither FLUX.1-dev
// nor -schnell can be pulled from black-forest-labs/* without an HF token tied
// to an account that has clicked through the license on huggingface.co.
const FLUX_MODEL_ID = 'Niansuh/FLUX.1-schnell';
const FLUX_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch, shutil
from diffusers import FluxPipeline, FluxTransformer2DModel, BitsAndBytesConfig as DBnBConfig
from transformers import T5EncoderModel, BitsAndBytesConfig as TBnBConfig
from huggingface_hub import snapshot_download
MODEL_ID = "${FLUX_MODEL_ID}"
# This mirror ships scheduler/config.json instead of the scheduler_config.json
# filename diffusers expects — patch it once per cache (idempotent).
snap_dir = snapshot_download(MODEL_ID, allow_patterns=["scheduler/config.json"])
sched_cfg = os.path.join(snap_dir, "scheduler", "scheduler_config.json")
if not os.path.exists(sched_cfg):
shutil.copy(os.path.join(snap_dir, "scheduler", "config.json"), sched_cfg)
transformer_4bit = FluxTransformer2DModel.from_pretrained(
MODEL_ID, subfolder="transformer",
quantization_config=DBnBConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16),
torch_dtype=torch.bfloat16,
)
text_encoder_2_4bit = T5EncoderModel.from_pretrained(
MODEL_ID, subfolder="text_encoder_2",
quantization_config=TBnBConfig(load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16),
torch_dtype=torch.bfloat16,
)
pipe = FluxPipeline.from_pretrained(
MODEL_ID,
transformer=transformer_4bit,
text_encoder_2=text_encoder_2_4bit,
torch_dtype=torch.bfloat16,
)
pipe.enable_model_cpu_offload()
kwargs = dict(
prompt=${JSON.stringify(p.prompt)},
guidance_scale=0.0,
num_inference_steps=int(${p.steps}),
max_sequence_length=256,
width=int(${p.width}), height=int(${p.height}),
)
${p.seed != null ? `kwargs["generator"] = torch.Generator("cpu").manual_seed(int(${p.seed}))` : ''}
image = pipe(**kwargs).images[0]
dst = ${JSON.stringify(p.dst)}
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
image.save(dst)
print("###RESULT###" + json.dumps({
"output": dst, "width": image.width, "height": image.height,
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
}))
except Exception as e:
import traceback
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
export const imageGenerateTool = {
name: 'image_generate',
description: [
'Generate an image from a text prompt using a local Stable Diffusion 1.5 model (runs on-machine GPU, no external API).',
'Best for quick, casual illustrations at up to ~768px. Takes a few seconds.',
'Generate an image from a text prompt using a local diffusion model (runs on-machine GPU, no external API).',
'quality="fast" (default): SDXL, ~10-20 seconds, default resolution 1024×1024, good for casual/quick illustrations.',
'quality="high": FLUX.1-schnell (4-bit quantized), ~40-60 seconds total (model load + generation), noticeably more photorealistic detail and prompt accuracy. Use only when the user explicitly asks for higher quality/detail/photorealism, or for a "고품질" request — otherwise default to fast.',
'Returns the generated image inline in the chat.',
].join('\n'),
schema: {
prompt: 'Text description of the image to generate (English works best for SD1.5)',
negative_prompt: 'Things to avoid in the image (optional, e.g. "blurry, low quality, deformed")',
width: 'Image width in pixels, multiple of 8 (default 512)',
height: 'Image height in pixels, multiple of 8 (default 512)',
steps: 'Denoising steps — more = higher quality but slower (default 25, range 10–50)',
guidance_scale: 'How closely to follow the prompt (default 7.5, range 1–20)',
prompt: 'Text description of the image to generate (English works best)',
quality: '"fast" (SDXL, default) or "high" (FLUX.1-schnell, much slower but noticeably better detail/realism)',
negative_prompt: 'Things to avoid in the image (optional, e.g. "blurry, low quality, deformed") — ignored in quality="high" mode (FLUX.1-schnell does not support it)',
width: 'Image width in pixels, multiple of 8 (default 1024)',
height: 'Image height in pixels, multiple of 8 (default 1024)',
steps: 'Denoising steps — more = higher quality but slower (fast mode: default 30, range 15–50; high mode: default 4, range 1–8)',
guidance_scale: 'How closely to follow the prompt (default 7.0, range 1–20) — fast mode only, ignored in quality="high"',
seed: 'Random seed for reproducibility (optional)',
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
},
@@ -94,6 +207,7 @@ export const imageGenerateTool = {
type: 'object',
properties: {
prompt: { type: 'string' },
quality: { type: 'string', enum: ['fast', 'high'] },
negative_prompt: { type: 'string' },
width: { type: 'number' },
height: { type: 'number' },
@@ -108,36 +222,59 @@ export const imageGenerateTool = {
execute: async (args: any): Promise<ToolResult> => {
const prompt = String(args?.prompt || '').trim();
if (!prompt) return { success: false, error: 'prompt is required' };
const isHighQuality = args?.quality === 'high';
const workspacePath = getWorkspacePath(args);
let outPath = String(args?.output || '').trim();
if (!outPath) {
outPath = path.join(workspacePath, `sd15_${Date.now()}.png`);
outPath = path.join(workspacePath, `${isHighQuality ? 'flux' : 'sdxl'}_${Date.now()}.png`);
} else if (!path.isAbsolute(outPath)) {
outPath = path.resolve(workspacePath, outPath);
}
const params = {
prompt,
negative_prompt: args?.negative_prompt || '',
width: Math.round((args?.width ?? 512) / 8) * 8,
height: Math.round((args?.height ?? 512) / 8) * 8,
steps: Math.min(50, Math.max(10, args?.steps ?? 25)),
guidance_scale: args?.guidance_scale ?? 7.5,
seed: args?.seed,
dst: outPath,
};
const negativePromptRaw = args?.negative_prompt || '';
const [translatedPrompt, translatedNegative] = await Promise.all([
translatePromptToEnglish(prompt),
isHighQuality ? Promise.resolve(null) : translatePromptToEnglish(negativePromptRaw),
]);
const result = await runVenvPython(SD15_SCRIPT(params), 180_000);
if (result.error) return { success: false, error: result.error, stderr: result.trace || result.raw };
const width = Math.round((args?.width ?? 1024) / 16) * 16;
const height = Math.round((args?.height ?? 1024) / 16) * 16;
let result: any;
if (isHighQuality) {
const params = {
prompt: translatedPrompt || prompt,
width, height,
steps: Math.min(8, Math.max(1, args?.steps ?? 4)),
seed: args?.seed,
dst: outPath,
};
result = await runVenvPython(FLUX_SCRIPT(params), 180_000);
} else {
const params = {
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width, height,
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
guidance_scale: args?.guidance_scale ?? 7.0,
seed: args?.seed,
dst: outPath,
};
result = await runVenvPython(SDXL_SCRIPT(params), 180_000);
}
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
}
return {
success: true,
stdout: [
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
`Generated: ${result.width} × ${result.height} px`,
'',
buildImageMarkdown(result.output, workspacePath),
].join('\n'),
].filter((line): line is string => line !== null).join('\n'),
data: { ...result, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
};
},
@@ -162,6 +299,7 @@ try:
width=int(${p.width}), height=int(${p.height}),
num_frames=int(${p.num_frames}),
num_inference_steps=int(${p.steps}),
guidance_scale=float(${p.guidance_scale}),
).frames[0]
dst = ${JSON.stringify(p.dst)}
@@ -194,7 +332,8 @@ export const videoGenerateTool = {
height: 'Video height in pixels, multiple of 32 (default 480)',
num_frames: 'Number of frames — duration = num_frames / fps (default 65, ~2.7s at 24fps)',
fps: 'Output frame rate (default 24)',
steps: 'Denoising steps — more = higher quality but slower (default 30, range 15–50)',
steps: 'Denoising steps — more = higher quality but slower (default 40, range 15–50)',
guidance_scale: 'How closely to follow the prompt (default 3.0, range 1–10 — LTX-Video responds best to a narrower range than SDXL)',
output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)',
},
jsonSchema: {
@@ -207,6 +346,7 @@ export const videoGenerateTool = {
num_frames: { type: 'number' },
fps: { type: 'number' },
steps: { type: 'number' },
guidance_scale: { type: 'number' },
output: { type: 'string' },
},
required: ['prompt'],
@@ -224,19 +364,28 @@ export const videoGenerateTool = {
outPath = path.resolve(workspacePath, outPath);
}
const negativePromptRaw = args?.negative_prompt || 'worst quality, blurry, distorted, deformed';
const [translatedPrompt, translatedNegative] = await Promise.all([
translatePromptToEnglish(prompt),
translatePromptToEnglish(negativePromptRaw),
]);
const params = {
prompt,
negative_prompt: args?.negative_prompt || 'worst quality, blurry, distorted, deformed',
prompt: translatedPrompt || prompt,
negative_prompt: translatedNegative || negativePromptRaw,
width: Math.round((args?.width ?? 704) / 32) * 32,
height: Math.round((args?.height ?? 480) / 32) * 32,
num_frames: args?.num_frames ?? 65,
fps: args?.fps ?? 24,
steps: Math.min(50, Math.max(15, args?.steps ?? 30)),
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
guidance_scale: Math.min(10, Math.max(1, args?.guidance_scale ?? 3.0)),
dst: outPath,
};
const result = await runVenvPython(LTX_SCRIPT(params), 600_000);
if (result.error) return { success: false, error: result.error, stderr: result.trace || result.raw };
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
}
const relOut = path.relative(workspacePath, result.output).replace(/\\/g, '/');
const sizeMB = (result.size_bytes / 1024 / 1024).toFixed(2);
@@ -245,10 +394,11 @@ export const videoGenerateTool = {
return {
success: true,
stdout: [
translatedPrompt ? `(translated prompt: ${translatedPrompt})` : null,
`Generated: ${result.width} × ${result.height} px | ${durationSec}s (${result.num_frames}f @ ${result.fps}fps) | ${sizeMB} MB`,
'',
`[${path.basename(result.output)}](/api/files/${relOut})`,
].join('\n'),
].filter((line): line is string => line !== null).join('\n'),
data: { ...result, rel_path: relOut },
};
},
+1 -1
View File
@@ -992,7 +992,7 @@ export const webFetchTool = {
},
};
function getOllamaConfig(): { endpoint: string; model: string } {
export function getOllamaConfig(): { endpoint: string; model: string } {
try {
const cm = getConfig();
const data = cm.getConfig();
+1
View File
@@ -65,6 +65,7 @@
<button class="mode-btn" onclick="window.open('/pptx-wizard.html','_blank')" title="슬라이드 제작">📊 슬라이드</button>
<button class="mode-btn" onclick="window.open('/language-app.html','_blank')" title="우즈베크어 학습">🇺🇿 언어</button>
<button class="mode-btn" onclick="window.open('/music-app.html','_blank')" title="음악 스튜디오">🎵 음악</button>
<button class="mode-btn" onclick="window.open('/studio-app.html','_blank')" title="이미지·동영상 생성">🎨 스튜디오</button>
<button class="mode-btn" onclick="window.open('/accountant-app.html','_blank')" title="회계 어시스턴트">📊 회계</button>
<button class="mode-btn" onclick="window.open('/mind-app.html','_blank')" title="마음 상담실">🧠 상담</button>
<button class="mode-btn" onclick="window.open('/lawyer-app.html','_blank')" title="법률 어시스턴트">⚖️ 법률</button>
+286
View File
@@ -0,0 +1,286 @@
<!DOCTYPE html>
<html lang="ko">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>🎨 이미지·동영상 스튜디오</title>
<link rel="icon" type="image/png" sizes="64x64" href="cherry_logo.png">
<link rel="stylesheet" href="styles.css">
<script>try{var _t=localStorage.getItem('cherryclaw_theme')||'dark';document.documentElement.setAttribute('data-theme',_t);}catch(e){}</script>
<style>
:root{--brand:#a78bfa;}
html,body{height:100%;margin:0;padding:0;}
body{background:var(--bg);color:var(--text);font-family:system-ui,sans-serif;font-weight:500;display:flex;flex-direction:column;overflow:hidden;height:100dvh;}
.st-hdr{display:flex;align-items:center;gap:8px;padding:6px 12px;background:var(--panel);border-bottom:1px solid var(--line);flex-shrink:0;}
.st-hdr h1{font-size:13px;font-weight:700;margin:0;}
.st-hdr-sp{flex:1;}
.hdr-btn{background:none;border:1.5px solid var(--line);border-radius:6px;padding:3px 9px;font-size:11px;color:var(--muted);cursor:pointer;font-family:inherit;transition:.15s;}
.hdr-btn:hover{border-color:var(--brand);color:var(--brand);}
.st-tabs{display:flex;gap:4px;padding:8px 12px 0;flex-shrink:0;}
.st-tab{background:none;border:none;border-bottom:2px solid transparent;padding:8px 14px;font-size:13px;font-weight:700;color:var(--muted);cursor:pointer;font-family:inherit;}
.st-tab.active{color:var(--brand);border-bottom-color:var(--brand);}
.st-layout{flex:1;display:flex;min-height:0;overflow:hidden;}
.st-form{width:320px;flex-shrink:0;display:flex;flex-direction:column;gap:10px;padding:14px;border-right:1px solid var(--line);background:var(--panel);overflow-y:auto;}
.st-field{display:flex;flex-direction:column;gap:4px;}
.st-field label{font-size:11px;font-weight:700;color:var(--muted);}
.st-field textarea,.st-field input,.st-field select{background:var(--panel-2);border:1px solid var(--line);border-radius:6px;padding:7px 8px;font-size:12px;color:var(--text);font-family:inherit;outline:none;resize:vertical;}
.st-field textarea:focus,.st-field input:focus,.st-field select:focus{border-color:var(--brand);}
.st-row{display:flex;gap:8px;}
.st-row .st-field{flex:1;}
.st-gen-btn{background:var(--brand);border:none;border-radius:8px;padding:10px;font-size:13px;font-weight:700;color:#1e1b2e;cursor:pointer;font-family:inherit;margin-top:4px;transition:.15s;}
.st-gen-btn:hover{opacity:.9;}
.st-gen-btn:disabled{opacity:.5;cursor:not-allowed;}
.st-hint{font-size:10px;color:var(--muted);line-height:1.4;}
.st-main{flex:1;display:flex;flex-direction:column;min-width:0;overflow:hidden;}
.st-result{flex:1;display:flex;align-items:center;justify-content:center;padding:20px;overflow:auto;min-height:0;}
.st-result img,.st-result video{max-width:100%;max-height:100%;border-radius:10px;box-shadow:0 4px 24px rgba(0,0,0,.25);}
.st-placeholder{color:var(--muted);font-size:13px;text-align:center;}
.st-spinner{width:32px;height:32px;border:3px solid var(--line);border-top-color:var(--brand);border-radius:50%;animation:st-spin 0.8s linear infinite;margin:0 auto 10px;}
@keyframes st-spin{to{transform:rotate(360deg);}}
.st-status{font-size:12px;color:var(--muted);margin-top:6px;}
.st-error{color:#ef4444;font-size:12px;}
.st-gallery{flex-shrink:0;border-top:1px solid var(--line);background:var(--panel);padding:8px 12px;max-height:130px;overflow-y:auto;}
.st-gallery-hdr{font-size:10px;font-weight:700;color:var(--muted);text-transform:uppercase;letter-spacing:.05em;margin-bottom:6px;}
.st-gallery-grid{display:flex;gap:8px;flex-wrap:wrap;}
.st-gitem{position:relative;width:76px;height:76px;border-radius:6px;overflow:hidden;border:1px solid var(--line);cursor:pointer;flex-shrink:0;background:var(--panel-2);}
.st-gitem img,.st-gitem video{width:100%;height:100%;object-fit:cover;}
.st-gitem .st-gtype{position:absolute;top:2px;left:2px;font-size:9px;background:rgba(0,0,0,.6);color:#fff;padding:1px 4px;border-radius:3px;}
.st-gitem .st-gdel{position:absolute;top:2px;right:2px;background:rgba(0,0,0,.6);color:#fff;border:none;border-radius:3px;font-size:10px;width:16px;height:16px;line-height:1;cursor:pointer;display:none;}
.st-gitem:hover .st-gdel{display:block;}
</style>
</head>
<body>
<header class="st-hdr">
<div style="font-weight:700;font-size:13px">🎨 이미지·동영상 스튜디오</div>
<div class="st-hdr-sp"></div>
<button class="hdr-btn" onclick="toggleTheme()">🌓</button>
<button class="hdr-btn" onclick="window.close()">닫기</button>
</header>
<div class="st-tabs">
<button class="st-tab active" id="tab-image" onclick="switchTab('image')">🖼️ 이미지</button>
<button class="st-tab" id="tab-video" onclick="switchTab('video')">🎬 동영상</button>
</div>
<div class="st-layout">
<div class="st-form">
<div class="st-field">
<label>프롬프트</label>
<textarea id="f-prompt" rows="4" placeholder="예: a red panda reading a book, cozy lighting"></textarea>
</div>
<div class="st-field">
<label>네거티브 프롬프트 (선택)</label>
<textarea id="f-neg" rows="2" placeholder="blurry, low quality, deformed"></textarea>
</div>
<div class="st-row">
<div class="st-field">
<label>가로</label>
<input type="number" id="f-width" value="1024" step="8">
</div>
<div class="st-field">
<label>세로</label>
<input type="number" id="f-height" value="1024" step="8">
</div>
</div>
<div class="st-row">
<div class="st-field">
<label>스텝수 (품질 ↔ 속도)</label>
<input type="number" id="f-steps" value="30" min="10" max="50">
</div>
<div class="st-field">
<label>가이던스 스케일</label>
<input type="number" id="f-guidance" value="7.0" step="0.5">
</div>
</div>
<div class="st-row" id="row-image-only">
<div class="st-field">
<label>품질</label>
<select id="f-quality" onchange="onQualityChange()">
<option value="fast">빠름 (SDXL · 10~20초)</option>
<option value="high">고품질 (FLUX.1 · 40~60초)</option>
</select>
</div>
<div class="st-field">
<label>시드 (선택)</label>
<input type="number" id="f-seed" placeholder="랜덤">
</div>
</div>
<div class="st-row" id="row-video-only" style="display:none">
<div class="st-field">
<label>프레임수</label>
<input type="number" id="f-frames" value="65" step="8">
</div>
<div class="st-field">
<label>FPS</label>
<input type="number" id="f-fps" value="24">
</div>
</div>
<button class="st-gen-btn" id="gen-btn" onclick="generate()">생성</button>
<div class="st-hint" id="hint-text">SDXL 로컬 생성 · 보통 10~20초</div>
</div>
<div class="st-main">
<div class="st-result" id="result-area">
<div class="st-placeholder">프롬프트를 입력하고 생성 버튼을 눌러주세요</div>
</div>
<div class="st-gallery">
<div class="st-gallery-hdr">최근 생성 이력</div>
<div class="st-gallery-grid" id="gallery-grid"></div>
</div>
</div>
</div>
<script>
const TOKEN_KEY='smallclaw_token';
function getToken(){try{return sessionStorage.getItem(TOKEN_KEY)||localStorage.getItem(TOKEN_KEY)||'';}catch{return '';}}
function authH(extra){const t=getToken();const h={...(extra||{})};if(t)h['Authorization']='Bearer '+t;return h;}
async function checkAuth(){
if(!getToken()){location.href='/login.html?redirect='+encodeURIComponent(location.pathname);return false;}
try{const r=await fetch('/api/auth/status',{headers:authH()});if(!r.ok){location.href='/login.html?redirect='+encodeURIComponent(location.pathname);return false;}const d=await r.json();if(!d.authenticated){location.href='/login.html?redirect='+encodeURIComponent(location.pathname);return false;}return true;}
catch{location.href='/login.html?redirect='+encodeURIComponent(location.pathname);return false;}
}
function toggleTheme(){const d=document.documentElement;const n=d.getAttribute('data-theme')==='dark'?'light':'dark';d.setAttribute('data-theme',n);try{localStorage.setItem('cherryclaw_theme',n);}catch{}}
let currentTab='image';
function switchTab(tab){
currentTab=tab;
document.getElementById('tab-image').classList.toggle('active', tab==='image');
document.getElementById('tab-video').classList.toggle('active', tab==='video');
document.getElementById('row-image-only').style.display = tab==='image' ? 'flex' : 'none';
document.getElementById('row-video-only').style.display = tab==='video' ? 'flex' : 'none';
if(tab==='image'){
document.getElementById('f-width').value=1024; document.getElementById('f-height').value=1024;
document.getElementById('f-quality').value='fast';
onQualityChange();
} else {
document.getElementById('f-width').value=704; document.getElementById('f-height').value=480;
document.getElementById('f-steps').value=40; document.getElementById('f-guidance').value=3.0;
document.getElementById('hint-text').textContent='LTX-Video 로컬 생성 · 보통 30~90초, 시간이 걸립니다';
}
}
function onQualityChange(){
const q=document.getElementById('f-quality').value;
const stepsEl=document.getElementById('f-steps');
if(q==='high'){
stepsEl.min=1; stepsEl.max=8; stepsEl.value=4;
document.getElementById('f-guidance').disabled=true;
document.getElementById('hint-text').textContent='FLUX.1-schnell 로컬 생성(4비트 양자화) · 보통 40~60초, 시간이 걸립니다';
} else {
stepsEl.min=10; stepsEl.max=50; stepsEl.value=30;
document.getElementById('f-guidance').disabled=false;
document.getElementById('f-guidance').value=7.0;
document.getElementById('hint-text').textContent='SDXL 로컬 생성 · 보통 10~20초';
}
}
let genStartTime=0, genTimer=null;
function startTimer(){
genStartTime=Date.now();
const btn=document.getElementById('gen-btn');
btn.disabled=true;
genTimer=setInterval(()=>{
const s=((Date.now()-genStartTime)/1000).toFixed(0);
btn.textContent='생성 중... ('+s+'초)';
},500);
}
function stopTimer(){
clearInterval(genTimer);
document.getElementById('gen-btn').disabled=false;
document.getElementById('gen-btn').textContent='생성';
}
async function generate(){
const prompt=document.getElementById('f-prompt').value.trim();
if(!prompt){alert('프롬프트를 입력해주세요');return;}
const resultArea=document.getElementById('result-area');
const isHighQuality = currentTab==='image' && document.getElementById('f-quality').value==='high';
const statusMsg = currentTab==='video' ? '동영상 생성 중... 최대 1~2분 정도 걸릴 수 있어요'
: isHighQuality ? '고품질(FLUX.1) 이미지 생성 중... 최대 1분 정도 걸릴 수 있어요'
: '이미지 생성 중...';
resultArea.innerHTML='<div style="text-align:center"><div class="st-spinner"></div><div class="st-status">'+statusMsg+'</div></div>';
startTimer();
const body={
prompt,
negative_prompt: document.getElementById('f-neg').value.trim(),
width: parseInt(document.getElementById('f-width').value)||undefined,
height: parseInt(document.getElementById('f-height').value)||undefined,
steps: parseInt(document.getElementById('f-steps').value)||undefined,
guidance_scale: parseFloat(document.getElementById('f-guidance').value)||undefined,
};
if(currentTab==='image'){
body.quality=document.getElementById('f-quality').value;
const seed=document.getElementById('f-seed').value;
if(seed) body.seed=parseInt(seed);
} else {
body.num_frames=parseInt(document.getElementById('f-frames').value)||undefined;
body.fps=parseInt(document.getElementById('f-fps').value)||undefined;
}
try{
const endpoint = currentTab==='image' ? '/api/imagegen/generate-image' : '/api/imagegen/generate-video';
const r=await fetch(endpoint,{method:'POST',headers:authH({'Content-Type':'application/json'}),body:JSON.stringify(body)});
const d=await r.json();
stopTimer();
if(!r.ok || !d.success){
resultArea.innerHTML='<div class="st-error">생성 실패: '+(d.error||'알 수 없는 오류')+'</div>';
return;
}
if(currentTab==='image'){
resultArea.innerHTML='<img src="'+d.url+'" alt="generated">';
} else {
resultArea.innerHTML='<video src="'+d.url+'" controls autoplay loop></video>';
}
loadGallery();
}catch(e){
stopTimer();
resultArea.innerHTML='<div class="st-error">요청 실패: '+e.message+'</div>';
}
}
async function loadGallery(){
try{
const r=await fetch('/api/imagegen/gallery',{headers:authH()});
const d=await r.json();
const grid=document.getElementById('gallery-grid');
grid.innerHTML='';
for(const item of (d.items||[])){
const el=document.createElement('div');
el.className='st-gitem';
const media = item.type==='image'
? '<img src="'+item.url+'">'
: '<video src="'+item.url+'" muted></video>';
el.innerHTML=media+'<span class="st-gtype">'+(item.type==='image'?'🖼️':'🎬')+'</span><button class="st-gdel" title="삭제">✕</button>';
el.querySelector('.st-gitem > img, .st-gitem > video')?.addEventListener('click',()=>{
document.getElementById('result-area').innerHTML = item.type==='image'
? '<img src="'+item.url+'">' : '<video src="'+item.url+'" controls autoplay loop></video>';
});
el.addEventListener('click',(ev)=>{ if(ev.target.tagName!=='BUTTON'){
document.getElementById('result-area').innerHTML = item.type==='image'
? '<img src="'+item.url+'">' : '<video src="'+item.url+'" controls autoplay loop></video>';
}});
el.querySelector('.st-gdel').addEventListener('click', async (ev)=>{
ev.stopPropagation();
if(!confirm(item.name+' 삭제할까요?'))return;
await fetch('/api/imagegen/gallery/'+encodeURIComponent(item.name),{method:'DELETE',headers:authH()});
loadGallery();
});
grid.appendChild(el);
}
}catch(e){}
}
(async function init(){
const ok=await checkAuth();if(!ok)return;
loadGallery();
})();
</script>
</body>
</html>