v4.1.9: 스튜디오 이미지 편집·저장 플로우 + 음성엔진 정리 + GPU 게이지 분리

- 스튜디오 앱: 사진 편집(스타일 변환/이모티콘 세트) + 결과 뷰어 내 크롭/회전/필터/보정/텍스트 오버레이 편집 툴바 + 저장 플로우(중간 편집은 갤러리에 안 남기고 최종본만 저장+업로드 목록에 재사용 등록)
- image_style_transform: SDXL img2img + IP-Adapter-FaceID로 cartoon 스타일 추가 (watercolor/sketch는 identity drift로 재검토 후 제외)
- 음성엔진 provider 값 xtts_gpu → omnivoice_gpu로 리네이밍 (실제 로딩 모델과 이름 일치, XTTS는 이미 OmniVoice로 교체된 지 오래)
- 메인창 사이드바 GPU 게이지를 GPU0/GPU1로 분리 표시
- 단종된 gemini-3-flash-preview를 기본 모델값에서 전부 제거, kimi-k2.6:cloud로 교체
- vault 백업을 cron→systemd timer(Persistent=true)로 전환, /DATA 전체 동기화 스크립트 추가

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-15 18:16:20 +09:00
co-authored by Claude Sonnet 5
parent 88bad1e7d9
commit 2228aedbf7
19 changed files with 1366 additions and 96 deletions
+8 -3
View File
@@ -490,11 +490,15 @@ try:
elif op == "remove_bg":
try:
from rembg import remove as _rembg_remove
from rembg import remove as _rembg_remove, new_session as _rembg_new_session
except ImportError:
print(json.dumps({"error": "rembg not installed. Run: pip install rembg[cpu]"})); sys.exit(0)
_inp = img.convert("RGBA")
img = _rembg_remove(_inp)
# birefnet-portrait: much finer hair-strand matting than the old u2net default,
# at the cost of speed (~4s -> ~25s on CPU, no GPU onnxruntime provider installed).
# Worth it for the sticker-set pipeline where cutout edge quality matters.
_session = _rembg_new_session("birefnet-portrait")
img = _rembg_remove(_inp, session=_session)
# force PNG output (preserves transparency)
if not dst.lower().endswith(".png"):
dst = os.path.splitext(dst)[0] + ".png"
@@ -624,7 +628,7 @@ export const imageEditTool = {
' watermark — overlay text (text, position, opacity)',
' speech_bubble — add a speech bubble with tail (text, position, bg_color, text_color, border_color, font_size)',
' stylize — convert to artistic style. Neural: anime(default)|painting|celeba|anime_v1 (AnimeGANv2). Classic: sketch|sketch_color|cartoon|watercolor',
' remove_bg — remove background using AI (rembg); output is PNG with transparency',
' remove_bg — remove background using AI (rembg, birefnet-portrait model — fine hair-strand matting); output is PNG with transparency. Slower than most operations (~20-30s on CPU).',
' convert — change file format (output path determines format)',
'Returns the output file path and new dimensions.',
].join('\n'),
@@ -726,6 +730,7 @@ export const imageEditTool = {
opacity: args?.opacity ?? 0.5,
quality: args?.quality ?? 92,
style: args?.style ?? 'painting',
preset: args?.preset ?? 'warm',
};
// Neural stylize (anime/painting/celeba) downloads PyTorch models on first run (~5 min).
+191
View File
@@ -184,6 +184,197 @@ except Exception as e:
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
// ---------------------------------------------------------------------------
// image_style_transform — local SDXL img2img (cartoon/watercolor only)
// ---------------------------------------------------------------------------
// image_edit's stylize op (AnimeGANv2 + OpenCV) already covers anime/sketch/bw
// well — those are identity-preserving and near-instant. But its "cartoon" and
// "watercolor" presets are mechanical edge/color filters that read as a mild
// photo filter rather than real artistic reinterpretation. SDXL img2img trades
// speed and some facial-identity fidelity for a genuinely more painterly result
// on those two styles specifically.
const IMG2IMG_SCRIPT = (p: Record<string, any>) => `
import os, json, sys
try:
import torch
from diffusers import StableDiffusionXLImg2ImgPipeline
from PIL import Image, ImageOps
pipe = StableDiffusionXLImg2ImgPipeline.from_pretrained(
"stabilityai/stable-diffusion-xl-base-1.0",
torch_dtype=torch.float16, variant="fp16", use_safetensors=True,
)
pipe = pipe.to("cuda")
pipe.enable_vae_slicing()
init_image = ImageOps.exif_transpose(Image.open(${JSON.stringify(p.src)})).convert("RGB")
max_side = int(${p.max_side})
w, h = init_image.size
scale = min(1.0, max_side / max(w, h))
w, h = max(8, int(w * scale) // 8 * 8), max(8, int(h * scale) // 8 * 8)
init_image = init_image.resize((w, h))
# Face-ID lock (IP-Adapter-FaceID/SDXL): without this, img2img alone lets the
# style prompt pull the face toward SDXL's own learned priors — this is what
# caused the watercolor identity-drift problem. Extracting a face embedding and
# conditioning generation on it keeps the result recognizably the same person.
# Falls back to plain img2img if no face is detected (e.g. non-portrait photo).
face_embeds = None
if ${p.face_lock ? 'True' : 'False'}:
try:
import cv2
from insightface.app import FaceAnalysis
_face_app = FaceAnalysis(name="buffalo_l", providers=["CPUExecutionProvider"])
_face_app.prepare(ctx_id=0, det_size=(640, 640))
_cv_img = cv2.imread(${JSON.stringify(p.src)})
_faces = _face_app.get(_cv_img)
if _faces:
_emb = torch.from_numpy(_faces[0].normed_embedding).unsqueeze(0).unsqueeze(0)
face_embeds = torch.cat([torch.zeros_like(_emb), _emb], dim=0).to(dtype=torch.float16, device="cuda")
except Exception:
face_embeds = None
kwargs = dict(
prompt=${JSON.stringify(p.prompt)},
negative_prompt=${JSON.stringify(p.negative_prompt)},
image=init_image,
strength=float(${p.strength}),
num_inference_steps=int(${p.steps}),
guidance_scale=float(${p.guidance_scale}),
)
${p.seed != null ? `kwargs["generator"] = torch.Generator("cuda").manual_seed(int(${p.seed}))` : ''}
if face_embeds is not None:
pipe.load_ip_adapter("h94/IP-Adapter-FaceID", subfolder=None, weight_name="ip-adapter-faceid_sdxl.bin", image_encoder_folder=None)
pipe.set_ip_adapter_scale(0.8)
kwargs["ip_adapter_image_embeds"] = [face_embeds]
image = pipe(**kwargs).images[0]
dst = ${JSON.stringify(p.dst)}
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
image.save(dst)
print("###RESULT###" + json.dumps({
"output": dst, "width": image.width, "height": image.height,
"face_locked": face_embeds is not None,
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
}))
except Exception as e:
import traceback
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
`;
const IMG2IMG_STYLE_PRESETS: Record<string, { prompt: string; negative: string; strength: number; guidance_scale: number }> = {
cartoon: {
// strength was 0.6 — looked like a genuine caricature but the pose/expression
// ("인상") drifted too far from the source photo even with face-ID identity
// locked, since face-ID only conditions identity, not expression/composition.
// 0.45 keeps the photo's actual expression and framing intact while still
// applying visible cel-shaded/cartoon coloring and clean outlines.
prompt: 'caricature portrait illustration, exaggerated facial features, bold clean outlines, vibrant flat colors, humorous comic art style, digital illustration',
negative: 'photorealistic, blurry, deformed hands, extra limbs, low quality, watermark, text',
strength: 0.45,
guidance_scale: 7.0,
},
// "watercolor" was tried here too but dropped: the "watercolor painting portrait"
// prompt pulled SDXL toward its own learned face priors much harder than cartoon
// does — even down to strength 0.25-0.3 it reliably drifted the subject toward a
// different-looking (often different-gender-presenting) face, with the watercolor
// effect barely visible at the strengths low enough to keep identity intact.
// image_edit's classic OpenCV watercolor filter (mild but identity-safe) is used
// instead — see EDIT_STYLE_MAP in gateway/routes/imagegen.ts.
//
// "sketch" was tried too and also dropped: going from a color photo to monochrome
// graphite linework needed strength/guidance high enough (0.8/9.0) that, with no
// seed pinned, run-to-run variance sometimes drifted the face into a different-
// looking person (same failure shape as watercolor). Backing off to 0.6/8.0 for
// more consistent identity made the result look worse than image_edit's classic
// sketch filter — flat/muddy rather than either a clean sketch or a good likeness.
// Reverted to image_edit's OpenCV sketch filter (see EDIT_STYLE_MAP).
};
export const imageStyleTransformTool = {
name: 'image_style_transform',
description: [
'Transform an existing photo (e.g. a portrait) into a new artistic style using local SDXL img2img with IP-Adapter-FaceID identity locking (runs on-machine GPU, no external API).',
'Supports style="cartoon" (caricature/comic illustration) — more artistically interpretive than image_edit\'s classic OpenCV cartoon filter, at the cost of speed (~15-40s vs ~2-3s). When a face is detected in the source photo, a face embedding (insightface/buffalo_l) conditions the generation via IP-Adapter-FaceID so the result stays recognizably the same person even under a strong style prompt — falls back to plain img2img if no face is found.',
'For anime/sketch/watercolor/black-and-white, use image_edit\'s stylize/filter operations instead — those are faster (both "watercolor" and "sketch" SDXL presets were tried and dropped: the style jump needed enough strength that, with no seed pinned, results sometimes drifted the face into a different-looking person).',
'Returns the transformed image inline in the chat.',
].join('\n'),
schema: {
image: 'Path to the source image to transform (required)',
style: '"cartoon" (caricature/comic illustration) — currently the only supported style',
face_lock: 'Whether to lock facial identity via IP-Adapter-FaceID when a face is detected (optional, default true). Turn off to compare plain img2img — face-ID conditioning sometimes reads as slightly uncanny/over-smoothed; plain img2img gives looser but more natural-looking results.',
strength: 'How strongly to restyle, 0.05-1.0 (optional; default 0.45). Lower preserves the original photo\'s pose/expression more, higher restyles more aggressively but can drift away from them.',
steps: 'Denoising steps (default 40, range 15-50)',
guidance_scale: 'How closely to follow the style prompt (default 7.0)',
seed: 'Random seed for reproducibility (optional)',
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
},
jsonSchema: {
type: 'object',
properties: {
image: { type: 'string' },
style: { type: 'string', enum: ['cartoon'] },
face_lock: { type: 'boolean' },
strength: { type: 'number' },
steps: { type: 'number' },
guidance_scale: { type: 'number' },
seed: { type: 'number' },
output: { type: 'string' },
},
required: ['image', 'style'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const imageArg = String(args?.image || '').trim();
if (!imageArg) return { success: false, error: 'image is required' };
const style = String(args?.style || '').trim();
const preset = IMG2IMG_STYLE_PRESETS[style];
if (!preset) return { success: false, error: `Unknown style: ${style} (expected "cartoon")` };
const workspacePath = getWorkspacePath(args);
const srcPath = path.isAbsolute(imageArg) ? imageArg : path.resolve(workspacePath, imageArg);
if (!fs.existsSync(srcPath)) return { success: false, error: `Source image not found: ${imageArg}` };
let outPath = String(args?.output || '').trim();
if (!outPath) {
outPath = path.join(workspacePath, `${style}_${Date.now()}.png`);
} else if (!path.isAbsolute(outPath)) {
outPath = path.resolve(workspacePath, outPath);
}
const params = {
src: srcPath,
prompt: preset.prompt,
negative_prompt: preset.negative,
strength: Math.min(1, Math.max(0.05, args?.strength ?? preset.strength)),
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
guidance_scale: args?.guidance_scale ?? preset.guidance_scale,
seed: args?.seed,
face_lock: args?.face_lock ?? true,
max_side: 1024,
dst: outPath,
};
// 600s (not 180s like the other scripts): the face-ID adapter weights (~1.7GB)
// download from HF Hub on first run and can take several minutes.
const result = await runVenvPython(IMG2IMG_SCRIPT(params), 600_000);
if (result.error || !result.output) {
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
}
return {
success: true,
stdout: [
`Style: ${style}`,
`Transformed: ${result.width} × ${result.height} px`,
'',
buildImageMarkdown(result.output, workspacePath),
].join('\n'),
data: { ...result, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
};
},
};
export const imageGenerateTool = {
name: 'image_generate',
description: [
+2 -1
View File
@@ -17,7 +17,7 @@ import { openalexSearchTool, semanticSearchTool } from './scholar.js';
import { pdfReadTool } from './pdf.js';
import { pdfExtractImagesTool, pdfExtractTablesTool } from './pdf-extract.js';
import { imageReadTool, imagePreviewTool, imageInfoTool, imageEditTool } from './image.js';
import { imageGenerateTool, videoGenerateTool } from './imagegen.js';
import { imageGenerateTool, videoGenerateTool, imageStyleTransformTool } from './imagegen.js';
import { audioTranscribeTool } from './audio-transcribe.js';
import { pythonEvalTool } from './python.js';
import { sqliteTool } from './sqlite.js';
@@ -206,6 +206,7 @@ class ToolRegistry {
this.registerSafe(imageInfoTool);
this.registerSafe(imageEditTool);
this.registerSafe(imageGenerateTool);
this.registerSafe(imageStyleTransformTool);
this.registerSafe(videoGenerateTool);
this.registerSafe(audioTranscribeTool);
this.registerSafe(pythonEvalTool);
+1 -1
View File
@@ -35,7 +35,7 @@ export async function synthesizeSpeech(text: string): Promise<Buffer> {
const provider = cfg?.voice?.tts?.provider;
const truncatedText = text.slice(0, 4000);
if (provider === 'xtts_gpu') {
if (provider === 'omnivoice_gpu') {
const tempDir = cfg?.voice?.tempDir || os.tmpdir();
const ffmpegPath = cfg?.voice?.ffmpegPath || 'ffmpeg';
const { synthesizeFullGPU } = await import('./voice-engine-client.js');