v4.1.9: 스튜디오 이미지 편집·저장 플로우 + 음성엔진 정리 + GPU 게이지 분리
- 스튜디오 앱: 사진 편집(스타일 변환/이모티콘 세트) + 결과 뷰어 내 크롭/회전/필터/보정/텍스트 오버레이 편집 툴바 + 저장 플로우(중간 편집은 갤러리에 안 남기고 최종본만 저장+업로드 목록에 재사용 등록) - image_style_transform: SDXL img2img + IP-Adapter-FaceID로 cartoon 스타일 추가 (watercolor/sketch는 identity drift로 재검토 후 제외) - 음성엔진 provider 값 xtts_gpu → omnivoice_gpu로 리네이밍 (실제 로딩 모델과 이름 일치, XTTS는 이미 OmniVoice로 교체된 지 오래) - 메인창 사이드바 GPU 게이지를 GPU0/GPU1로 분리 표시 - 단종된 gemini-3-flash-preview를 기본 모델값에서 전부 제거, kimi-k2.6:cloud로 교체 - vault 백업을 cron→systemd timer(Persistent=true)로 전환, /DATA 전체 동기화 스크립트 추가 Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
+8
-3
@@ -490,11 +490,15 @@ try:
|
||||
|
||||
elif op == "remove_bg":
|
||||
try:
|
||||
from rembg import remove as _rembg_remove
|
||||
from rembg import remove as _rembg_remove, new_session as _rembg_new_session
|
||||
except ImportError:
|
||||
print(json.dumps({"error": "rembg not installed. Run: pip install rembg[cpu]"})); sys.exit(0)
|
||||
_inp = img.convert("RGBA")
|
||||
img = _rembg_remove(_inp)
|
||||
# birefnet-portrait: much finer hair-strand matting than the old u2net default,
|
||||
# at the cost of speed (~4s -> ~25s on CPU, no GPU onnxruntime provider installed).
|
||||
# Worth it for the sticker-set pipeline where cutout edge quality matters.
|
||||
_session = _rembg_new_session("birefnet-portrait")
|
||||
img = _rembg_remove(_inp, session=_session)
|
||||
# force PNG output (preserves transparency)
|
||||
if not dst.lower().endswith(".png"):
|
||||
dst = os.path.splitext(dst)[0] + ".png"
|
||||
@@ -624,7 +628,7 @@ export const imageEditTool = {
|
||||
' watermark — overlay text (text, position, opacity)',
|
||||
' speech_bubble — add a speech bubble with tail (text, position, bg_color, text_color, border_color, font_size)',
|
||||
' stylize — convert to artistic style. Neural: anime(default)|painting|celeba|anime_v1 (AnimeGANv2). Classic: sketch|sketch_color|cartoon|watercolor',
|
||||
' remove_bg — remove background using AI (rembg); output is PNG with transparency',
|
||||
' remove_bg — remove background using AI (rembg, birefnet-portrait model — fine hair-strand matting); output is PNG with transparency. Slower than most operations (~20-30s on CPU).',
|
||||
' convert — change file format (output path determines format)',
|
||||
'Returns the output file path and new dimensions.',
|
||||
].join('\n'),
|
||||
@@ -726,6 +730,7 @@ export const imageEditTool = {
|
||||
opacity: args?.opacity ?? 0.5,
|
||||
quality: args?.quality ?? 92,
|
||||
style: args?.style ?? 'painting',
|
||||
preset: args?.preset ?? 'warm',
|
||||
};
|
||||
|
||||
// Neural stylize (anime/painting/celeba) downloads PyTorch models on first run (~5 min).
|
||||
|
||||
@@ -184,6 +184,197 @@ except Exception as e:
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_style_transform — local SDXL img2img (cartoon/watercolor only)
|
||||
// ---------------------------------------------------------------------------
|
||||
// image_edit's stylize op (AnimeGANv2 + OpenCV) already covers anime/sketch/bw
|
||||
// well — those are identity-preserving and near-instant. But its "cartoon" and
|
||||
// "watercolor" presets are mechanical edge/color filters that read as a mild
|
||||
// photo filter rather than real artistic reinterpretation. SDXL img2img trades
|
||||
// speed and some facial-identity fidelity for a genuinely more painterly result
|
||||
// on those two styles specifically.
|
||||
const IMG2IMG_SCRIPT = (p: Record<string, any>) => `
|
||||
import os, json, sys
|
||||
try:
|
||||
import torch
|
||||
from diffusers import StableDiffusionXLImg2ImgPipeline
|
||||
from PIL import Image, ImageOps
|
||||
|
||||
pipe = StableDiffusionXLImg2ImgPipeline.from_pretrained(
|
||||
"stabilityai/stable-diffusion-xl-base-1.0",
|
||||
torch_dtype=torch.float16, variant="fp16", use_safetensors=True,
|
||||
)
|
||||
pipe = pipe.to("cuda")
|
||||
pipe.enable_vae_slicing()
|
||||
|
||||
init_image = ImageOps.exif_transpose(Image.open(${JSON.stringify(p.src)})).convert("RGB")
|
||||
max_side = int(${p.max_side})
|
||||
w, h = init_image.size
|
||||
scale = min(1.0, max_side / max(w, h))
|
||||
w, h = max(8, int(w * scale) // 8 * 8), max(8, int(h * scale) // 8 * 8)
|
||||
init_image = init_image.resize((w, h))
|
||||
|
||||
# Face-ID lock (IP-Adapter-FaceID/SDXL): without this, img2img alone lets the
|
||||
# style prompt pull the face toward SDXL's own learned priors — this is what
|
||||
# caused the watercolor identity-drift problem. Extracting a face embedding and
|
||||
# conditioning generation on it keeps the result recognizably the same person.
|
||||
# Falls back to plain img2img if no face is detected (e.g. non-portrait photo).
|
||||
face_embeds = None
|
||||
if ${p.face_lock ? 'True' : 'False'}:
|
||||
try:
|
||||
import cv2
|
||||
from insightface.app import FaceAnalysis
|
||||
_face_app = FaceAnalysis(name="buffalo_l", providers=["CPUExecutionProvider"])
|
||||
_face_app.prepare(ctx_id=0, det_size=(640, 640))
|
||||
_cv_img = cv2.imread(${JSON.stringify(p.src)})
|
||||
_faces = _face_app.get(_cv_img)
|
||||
if _faces:
|
||||
_emb = torch.from_numpy(_faces[0].normed_embedding).unsqueeze(0).unsqueeze(0)
|
||||
face_embeds = torch.cat([torch.zeros_like(_emb), _emb], dim=0).to(dtype=torch.float16, device="cuda")
|
||||
except Exception:
|
||||
face_embeds = None
|
||||
|
||||
kwargs = dict(
|
||||
prompt=${JSON.stringify(p.prompt)},
|
||||
negative_prompt=${JSON.stringify(p.negative_prompt)},
|
||||
image=init_image,
|
||||
strength=float(${p.strength}),
|
||||
num_inference_steps=int(${p.steps}),
|
||||
guidance_scale=float(${p.guidance_scale}),
|
||||
)
|
||||
${p.seed != null ? `kwargs["generator"] = torch.Generator("cuda").manual_seed(int(${p.seed}))` : ''}
|
||||
|
||||
if face_embeds is not None:
|
||||
pipe.load_ip_adapter("h94/IP-Adapter-FaceID", subfolder=None, weight_name="ip-adapter-faceid_sdxl.bin", image_encoder_folder=None)
|
||||
pipe.set_ip_adapter_scale(0.8)
|
||||
kwargs["ip_adapter_image_embeds"] = [face_embeds]
|
||||
|
||||
image = pipe(**kwargs).images[0]
|
||||
dst = ${JSON.stringify(p.dst)}
|
||||
os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True)
|
||||
image.save(dst)
|
||||
print("###RESULT###" + json.dumps({
|
||||
"output": dst, "width": image.width, "height": image.height,
|
||||
"face_locked": face_embeds is not None,
|
||||
"vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2,
|
||||
}))
|
||||
except Exception as e:
|
||||
import traceback
|
||||
print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]}))
|
||||
`;
|
||||
|
||||
const IMG2IMG_STYLE_PRESETS: Record<string, { prompt: string; negative: string; strength: number; guidance_scale: number }> = {
|
||||
cartoon: {
|
||||
// strength was 0.6 — looked like a genuine caricature but the pose/expression
|
||||
// ("인상") drifted too far from the source photo even with face-ID identity
|
||||
// locked, since face-ID only conditions identity, not expression/composition.
|
||||
// 0.45 keeps the photo's actual expression and framing intact while still
|
||||
// applying visible cel-shaded/cartoon coloring and clean outlines.
|
||||
prompt: 'caricature portrait illustration, exaggerated facial features, bold clean outlines, vibrant flat colors, humorous comic art style, digital illustration',
|
||||
negative: 'photorealistic, blurry, deformed hands, extra limbs, low quality, watermark, text',
|
||||
strength: 0.45,
|
||||
guidance_scale: 7.0,
|
||||
},
|
||||
// "watercolor" was tried here too but dropped: the "watercolor painting portrait"
|
||||
// prompt pulled SDXL toward its own learned face priors much harder than cartoon
|
||||
// does — even down to strength 0.25-0.3 it reliably drifted the subject toward a
|
||||
// different-looking (often different-gender-presenting) face, with the watercolor
|
||||
// effect barely visible at the strengths low enough to keep identity intact.
|
||||
// image_edit's classic OpenCV watercolor filter (mild but identity-safe) is used
|
||||
// instead — see EDIT_STYLE_MAP in gateway/routes/imagegen.ts.
|
||||
//
|
||||
// "sketch" was tried too and also dropped: going from a color photo to monochrome
|
||||
// graphite linework needed strength/guidance high enough (0.8/9.0) that, with no
|
||||
// seed pinned, run-to-run variance sometimes drifted the face into a different-
|
||||
// looking person (same failure shape as watercolor). Backing off to 0.6/8.0 for
|
||||
// more consistent identity made the result look worse than image_edit's classic
|
||||
// sketch filter — flat/muddy rather than either a clean sketch or a good likeness.
|
||||
// Reverted to image_edit's OpenCV sketch filter (see EDIT_STYLE_MAP).
|
||||
};
|
||||
|
||||
export const imageStyleTransformTool = {
|
||||
name: 'image_style_transform',
|
||||
description: [
|
||||
'Transform an existing photo (e.g. a portrait) into a new artistic style using local SDXL img2img with IP-Adapter-FaceID identity locking (runs on-machine GPU, no external API).',
|
||||
'Supports style="cartoon" (caricature/comic illustration) — more artistically interpretive than image_edit\'s classic OpenCV cartoon filter, at the cost of speed (~15-40s vs ~2-3s). When a face is detected in the source photo, a face embedding (insightface/buffalo_l) conditions the generation via IP-Adapter-FaceID so the result stays recognizably the same person even under a strong style prompt — falls back to plain img2img if no face is found.',
|
||||
'For anime/sketch/watercolor/black-and-white, use image_edit\'s stylize/filter operations instead — those are faster (both "watercolor" and "sketch" SDXL presets were tried and dropped: the style jump needed enough strength that, with no seed pinned, results sometimes drifted the face into a different-looking person).',
|
||||
'Returns the transformed image inline in the chat.',
|
||||
].join('\n'),
|
||||
schema: {
|
||||
image: 'Path to the source image to transform (required)',
|
||||
style: '"cartoon" (caricature/comic illustration) — currently the only supported style',
|
||||
face_lock: 'Whether to lock facial identity via IP-Adapter-FaceID when a face is detected (optional, default true). Turn off to compare plain img2img — face-ID conditioning sometimes reads as slightly uncanny/over-smoothed; plain img2img gives looser but more natural-looking results.',
|
||||
strength: 'How strongly to restyle, 0.05-1.0 (optional; default 0.45). Lower preserves the original photo\'s pose/expression more, higher restyles more aggressively but can drift away from them.',
|
||||
steps: 'Denoising steps (default 40, range 15-50)',
|
||||
guidance_scale: 'How closely to follow the style prompt (default 7.0)',
|
||||
seed: 'Random seed for reproducibility (optional)',
|
||||
output: 'Output file path (optional; defaults to a timestamped file in the workspace)',
|
||||
},
|
||||
jsonSchema: {
|
||||
type: 'object',
|
||||
properties: {
|
||||
image: { type: 'string' },
|
||||
style: { type: 'string', enum: ['cartoon'] },
|
||||
face_lock: { type: 'boolean' },
|
||||
strength: { type: 'number' },
|
||||
steps: { type: 'number' },
|
||||
guidance_scale: { type: 'number' },
|
||||
seed: { type: 'number' },
|
||||
output: { type: 'string' },
|
||||
},
|
||||
required: ['image', 'style'],
|
||||
additionalProperties: false,
|
||||
},
|
||||
execute: async (args: any): Promise<ToolResult> => {
|
||||
const imageArg = String(args?.image || '').trim();
|
||||
if (!imageArg) return { success: false, error: 'image is required' };
|
||||
const style = String(args?.style || '').trim();
|
||||
const preset = IMG2IMG_STYLE_PRESETS[style];
|
||||
if (!preset) return { success: false, error: `Unknown style: ${style} (expected "cartoon")` };
|
||||
|
||||
const workspacePath = getWorkspacePath(args);
|
||||
const srcPath = path.isAbsolute(imageArg) ? imageArg : path.resolve(workspacePath, imageArg);
|
||||
if (!fs.existsSync(srcPath)) return { success: false, error: `Source image not found: ${imageArg}` };
|
||||
|
||||
let outPath = String(args?.output || '').trim();
|
||||
if (!outPath) {
|
||||
outPath = path.join(workspacePath, `${style}_${Date.now()}.png`);
|
||||
} else if (!path.isAbsolute(outPath)) {
|
||||
outPath = path.resolve(workspacePath, outPath);
|
||||
}
|
||||
|
||||
const params = {
|
||||
src: srcPath,
|
||||
prompt: preset.prompt,
|
||||
negative_prompt: preset.negative,
|
||||
strength: Math.min(1, Math.max(0.05, args?.strength ?? preset.strength)),
|
||||
steps: Math.min(50, Math.max(15, args?.steps ?? 40)),
|
||||
guidance_scale: args?.guidance_scale ?? preset.guidance_scale,
|
||||
seed: args?.seed,
|
||||
face_lock: args?.face_lock ?? true,
|
||||
max_side: 1024,
|
||||
dst: outPath,
|
||||
};
|
||||
// 600s (not 180s like the other scripts): the face-ID adapter weights (~1.7GB)
|
||||
// download from HF Hub on first run and can take several minutes.
|
||||
const result = await runVenvPython(IMG2IMG_SCRIPT(params), 600_000);
|
||||
if (result.error || !result.output) {
|
||||
return { success: false, error: result.error || 'Generator returned no output file', stderr: result.trace || result.raw };
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
stdout: [
|
||||
`Style: ${style}`,
|
||||
`Transformed: ${result.width} × ${result.height} px`,
|
||||
'',
|
||||
buildImageMarkdown(result.output, workspacePath),
|
||||
].join('\n'),
|
||||
data: { ...result, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') },
|
||||
};
|
||||
},
|
||||
};
|
||||
|
||||
export const imageGenerateTool = {
|
||||
name: 'image_generate',
|
||||
description: [
|
||||
|
||||
@@ -17,7 +17,7 @@ import { openalexSearchTool, semanticSearchTool } from './scholar.js';
|
||||
import { pdfReadTool } from './pdf.js';
|
||||
import { pdfExtractImagesTool, pdfExtractTablesTool } from './pdf-extract.js';
|
||||
import { imageReadTool, imagePreviewTool, imageInfoTool, imageEditTool } from './image.js';
|
||||
import { imageGenerateTool, videoGenerateTool } from './imagegen.js';
|
||||
import { imageGenerateTool, videoGenerateTool, imageStyleTransformTool } from './imagegen.js';
|
||||
import { audioTranscribeTool } from './audio-transcribe.js';
|
||||
import { pythonEvalTool } from './python.js';
|
||||
import { sqliteTool } from './sqlite.js';
|
||||
@@ -206,6 +206,7 @@ class ToolRegistry {
|
||||
this.registerSafe(imageInfoTool);
|
||||
this.registerSafe(imageEditTool);
|
||||
this.registerSafe(imageGenerateTool);
|
||||
this.registerSafe(imageStyleTransformTool);
|
||||
this.registerSafe(videoGenerateTool);
|
||||
this.registerSafe(audioTranscribeTool);
|
||||
this.registerSafe(pythonEvalTool);
|
||||
|
||||
+1
-1
@@ -35,7 +35,7 @@ export async function synthesizeSpeech(text: string): Promise<Buffer> {
|
||||
const provider = cfg?.voice?.tts?.provider;
|
||||
const truncatedText = text.slice(0, 4000);
|
||||
|
||||
if (provider === 'xtts_gpu') {
|
||||
if (provider === 'omnivoice_gpu') {
|
||||
const tempDir = cfg?.voice?.tempDir || os.tmpdir();
|
||||
const ffmpegPath = cfg?.voice?.ffmpegPath || 'ffmpeg';
|
||||
const { synthesizeFullGPU } = await import('./voice-engine-client.js');
|
||||
|
||||
Reference in New Issue
Block a user