From 2427aeeabaf654d991fa2db59ad5063b58edacde Mon Sep 17 00:00:00 2001 From: kim Date: Tue, 14 Jul 2026 20:42:42 +0900 Subject: [PATCH] =?UTF-8?q?v4.1.7:=20SD1.5/LTX-Video=20=EB=A1=9C=EC=BB=AC?= =?UTF-8?q?=20=EC=9D=B4=EB=AF=B8=EC=A7=80=C2=B7=EB=8F=99=EC=98=81=EC=83=81?= =?UTF-8?q?=20=EC=83=9D=EC=84=B1=20=ED=88=B4=20=EC=B6=94=EA=B0=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit image_generate(SD1.5), video_generate(LTX-Video)를 전용 venv+GPU1에서 구동하는 채팅 툴로 추가. image_edit과 구분되도록 시스템 프롬프트에 사용 규칙 명시. --- package.json | 2 +- src/gateway/server-v2.ts | 3 +- src/tools/imagegen.ts | 255 +++++++++++++++++++++++++++++++++++++++ src/tools/registry.ts | 3 + 4 files changed, 261 insertions(+), 2 deletions(-) create mode 100644 src/tools/imagegen.ts diff --git a/package.json b/package.json index 5c982d9..e1623cd 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "smallclaw", - "version": "4.1.6", + "version": "4.1.7", "description": "Local AI agent framework powered by Ollama - OpenClaw alternative", "main": "dist/index.js", "bin": { diff --git a/src/gateway/server-v2.ts b/src/gateway/server-v2.ts index 072f165..dd51752 100644 --- a/src/gateway/server-v2.ts +++ b/src/gateway/server-v2.ts @@ -5784,7 +5784,8 @@ async function handleChat( role: 'system', content: isTranslateSession ? `You are a medical translator. Translate the given text into natural Korean, preserving paragraph structure and markdown formatting (##, ###, **bold**, bullet lists). Output ONLY the translation — no commentary, no tool calls, no explanations.` : isProjSession ? `You are a project file designer. Output ONLY the project-files JSON block as instructed. No tool calls. No extra text.` : `${executionModeSystemBlock ? `${executionModeSystemBlock}\n\n` : ''}You are SmallClaw 🦞, a local AI assistant.\nCurrent date: ${dateStr}, ${timeStr}.\nNever search for or link SmallClaw repos unless the user is asking about SmallClaw itself.\nThis app runs on the user's own machine — browser/desktop automation requests are pre-authorized.\nKeep responses SHORT (1-2 sentences). Don't think out loud. Act and report. Greet naturally without tools. ANTI-HALLUCINATION: When a tool returns a result, report EXACTLY what the tool returned — never contradict or ignore tool output. If a tool says "(no rows)", say so. Never invent data, file contents, table names, or command output. If you don't know something, call a tool to find out or say you don't know. -IMAGE EDITING RULE: NEVER call image_edit (or any editing tool) when a user uploads a photo without explicitly requesting edits. Uploading a photo is NOT a request to edit it. Only call image_edit when the user's message explicitly asks for an edit (e.g. "수채화로 바꿔줘", "회전해줘"). Violating this rule is a critical error.${browserRuleBlock} +IMAGE EDITING RULE: NEVER call image_edit (or any editing tool) when a user uploads a photo without explicitly requesting edits. Uploading a photo is NOT a request to edit it. Only call image_edit when the user's message explicitly asks for an edit (e.g. "수채화로 바꿔줘", "회전해줘"). Violating this rule is a critical error. +IMAGE/VIDEO GENERATION: When a user asks to create/draw/generate a NEW image from a description (no existing photo involved), call image_generate (local SD1.5, a few seconds). When they ask for a short video/clip/animation from a description, call video_generate (local LTX-Video, 30–90 seconds — tell the user it'll take a bit before calling it). Both run entirely on local GPU hardware, no external API or cost. Do not confuse these with image_edit, which only modifies an existing uploaded/generated image.${browserRuleBlock} CHEMISTRY NOTATION: NEVER draw molecular structures as ASCII art (H/C/#/=/\\/| characters arranged to look like a diagram) — it always renders as garbled, misaligned text. Instead: for a formula or reaction, use LaTeX inside $...$ (e.g. $\\ce{C4H10}$, $\\ce{2H2 + O2 -> 2H2O}$ — mhchem extension is loaded). For an actual 2D structure with real bond lines (rings, branches), output a \`\`\`smiles\`\`\` code block containing the SMILES string (e.g. \`\`\`smiles\\nc1ccccc1\\n\`\`\` for benzene, \`\`\`smiles\\nCC(=O)Oc1ccccc1C(=O)O\\n\`\`\` for aspirin) — the client automatically renders it as a proper 2D diagram with bond lines. CODE OUTPUT: When writing code in a fenced code block, start with a filename comment on line 1: \`# filename: snake_game.py\` (Python), \`// filename: app.js\` (JS/C), \`\` (HTML). Never repeat code already written in this conversation. For modifications to existing files, use coder_overwrite_lines or coder_insert_lines (not coder_write_file — it only works for NEW files). All code changes are presented as diffs for the user to review before being applied. Write code directly — do not ask for permission. PACKAGE INSTALL: NEVER run pip install, npm install, apt-get, or any package installation command. If a package is missing, just write the code and mention the package name in a comment — let the user decide whether to install it. Do NOT attempt to install packages yourself. diff --git a/src/tools/imagegen.ts b/src/tools/imagegen.ts new file mode 100644 index 0000000..e3f91b7 --- /dev/null +++ b/src/tools/imagegen.ts @@ -0,0 +1,255 @@ +import { spawn } from 'child_process'; +import path from 'path'; +import fs from 'fs'; +import { ToolResult } from '../types.js'; +import { getWorkspacePath } from '../config/paths.js'; +import { buildImageMarkdown } from './image.js'; + +// Local diffusion models (SD1.5, LTX-Video) run in a dedicated venv with their +// own torch/diffusers stack, pinned to the second GPU (04:00.0 — kept free of +// the voice engine that permanently resides on GPU0). See +// /srv/homeclaw/.smallclaw/imagegen-venv. +const VENV_PYTHON = '/srv/homeclaw/.smallclaw/imagegen-venv/bin/python3'; +const HF_HOME = '/srv/homeclaw/.smallclaw/imagegen-venv/hf-cache'; +const GEN_GPU = '1'; + +function runVenvPython(script: string, timeoutMs: number): Promise { + return new Promise((resolve) => { + const child = spawn(VENV_PYTHON, ['-c', script], { + timeout: timeoutMs, + env: { ...process.env, HF_HOME, CUDA_VISIBLE_DEVICES: GEN_GPU }, + }); + let out = ''; + let err = ''; + child.stdout.on('data', (d: Buffer) => { out += d.toString('utf8'); }); + child.stderr.on('data', (d: Buffer) => { err += d.toString('utf8'); }); + child.on('close', () => { + const marker = out.lastIndexOf('###RESULT###'); + const jsonPart = marker !== -1 ? out.slice(marker + '###RESULT###'.length) : out; + try { + resolve(JSON.parse(jsonPart.trim() || '{}')); + } catch { + resolve({ error: 'Failed to parse generator output', raw: (jsonPart || err).slice(-1500) }); + } + }); + child.on('error', (e: Error) => resolve({ error: e.message })); + }); +} + +// --------------------------------------------------------------------------- +// image_generate — local Stable Diffusion 1.5 text-to-image +// --------------------------------------------------------------------------- +const SD15_SCRIPT = (p: Record) => ` +import os, json, sys +try: + import torch + from diffusers import StableDiffusionPipeline + + pipe = StableDiffusionPipeline.from_pretrained( + "stable-diffusion-v1-5/stable-diffusion-v1-5", + torch_dtype=torch.float16, safety_checker=None, + ) + pipe = pipe.to("cuda") + + kwargs = dict( + prompt=${JSON.stringify(p.prompt)}, + negative_prompt=${JSON.stringify(p.negative_prompt || '')} or None, + width=int(${p.width}), height=int(${p.height}), + num_inference_steps=int(${p.steps}), + guidance_scale=float(${p.guidance_scale}), + ) + ${p.seed != null ? `kwargs["generator"] = torch.Generator("cuda").manual_seed(int(${p.seed}))` : ''} + + image = pipe(**kwargs).images[0] + dst = ${JSON.stringify(p.dst)} + os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True) + image.save(dst) + print("###RESULT###" + json.dumps({ + "output": dst, "width": image.width, "height": image.height, + "vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2, + })) +except Exception as e: + import traceback + print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]})) +`; + +export const imageGenerateTool = { + name: 'image_generate', + description: [ + 'Generate an image from a text prompt using a local Stable Diffusion 1.5 model (runs on-machine GPU, no external API).', + 'Best for quick, casual illustrations at up to ~768px. Takes a few seconds.', + 'Returns the generated image inline in the chat.', + ].join('\n'), + schema: { + prompt: 'Text description of the image to generate (English works best for SD1.5)', + negative_prompt: 'Things to avoid in the image (optional, e.g. "blurry, low quality, deformed")', + width: 'Image width in pixels, multiple of 8 (default 512)', + height: 'Image height in pixels, multiple of 8 (default 512)', + steps: 'Denoising steps — more = higher quality but slower (default 25, range 10–50)', + guidance_scale: 'How closely to follow the prompt (default 7.5, range 1–20)', + seed: 'Random seed for reproducibility (optional)', + output: 'Output file path (optional; defaults to a timestamped file in the workspace)', + }, + jsonSchema: { + type: 'object', + properties: { + prompt: { type: 'string' }, + negative_prompt: { type: 'string' }, + width: { type: 'number' }, + height: { type: 'number' }, + steps: { type: 'number' }, + guidance_scale: { type: 'number' }, + seed: { type: 'number' }, + output: { type: 'string' }, + }, + required: ['prompt'], + additionalProperties: false, + }, + execute: async (args: any): Promise => { + const prompt = String(args?.prompt || '').trim(); + if (!prompt) return { success: false, error: 'prompt is required' }; + + const workspacePath = getWorkspacePath(args); + let outPath = String(args?.output || '').trim(); + if (!outPath) { + outPath = path.join(workspacePath, `sd15_${Date.now()}.png`); + } else if (!path.isAbsolute(outPath)) { + outPath = path.resolve(workspacePath, outPath); + } + + const params = { + prompt, + negative_prompt: args?.negative_prompt || '', + width: Math.round((args?.width ?? 512) / 8) * 8, + height: Math.round((args?.height ?? 512) / 8) * 8, + steps: Math.min(50, Math.max(10, args?.steps ?? 25)), + guidance_scale: args?.guidance_scale ?? 7.5, + seed: args?.seed, + dst: outPath, + }; + + const result = await runVenvPython(SD15_SCRIPT(params), 180_000); + if (result.error) return { success: false, error: result.error, stderr: result.trace || result.raw }; + + return { + success: true, + stdout: [ + `Generated: ${result.width} × ${result.height} px`, + '', + buildImageMarkdown(result.output, workspacePath), + ].join('\n'), + data: { ...result, rel_path: path.relative(workspacePath, result.output).replace(/\\/g, '/') }, + }; + }, +}; + +// --------------------------------------------------------------------------- +// video_generate — local LTX-Video text-to-video +// --------------------------------------------------------------------------- +const LTX_SCRIPT = (p: Record) => ` +import os, json, sys +try: + import torch + from diffusers import LTXPipeline + from diffusers.utils import export_to_video + + pipe = LTXPipeline.from_pretrained("Lightricks/LTX-Video", torch_dtype=torch.bfloat16) + pipe.enable_model_cpu_offload() + + video = pipe( + prompt=${JSON.stringify(p.prompt)}, + negative_prompt=${JSON.stringify(p.negative_prompt)}, + width=int(${p.width}), height=int(${p.height}), + num_frames=int(${p.num_frames}), + num_inference_steps=int(${p.steps}), + ).frames[0] + + dst = ${JSON.stringify(p.dst)} + os.makedirs(os.path.dirname(os.path.abspath(dst)), exist_ok=True) + export_to_video(video, dst, fps=int(${p.fps})) + + stat = os.stat(dst) + print("###RESULT###" + json.dumps({ + "output": dst, "width": int(${p.width}), "height": int(${p.height}), + "num_frames": int(${p.num_frames}), "fps": int(${p.fps}), + "size_bytes": stat.st_size, + "vram_peak_mb": torch.cuda.max_memory_allocated() / 1024**2, + })) +except Exception as e: + import traceback + print("###RESULT###" + json.dumps({"error": str(e), "trace": traceback.format_exc()[-800:]})) +`; + +export const videoGenerateTool = { + name: 'video_generate', + description: [ + 'Generate a short video clip from a text prompt using a local LTX-Video model (runs on-machine GPU, no external API).', + 'Takes roughly 30–90 seconds depending on resolution/steps/frame count. Output is an h264 mp4.', + 'Returns a download link — there is no inline video preview in chat yet.', + ].join('\n'), + schema: { + prompt: 'Text description of the video/scene to generate (English works best)', + negative_prompt: 'Things to avoid (optional, default "worst quality, blurry, distorted")', + width: 'Video width in pixels, multiple of 32 (default 704)', + height: 'Video height in pixels, multiple of 32 (default 480)', + num_frames: 'Number of frames — duration = num_frames / fps (default 65, ~2.7s at 24fps)', + fps: 'Output frame rate (default 24)', + steps: 'Denoising steps — more = higher quality but slower (default 30, range 15–50)', + output: 'Output file path (optional; defaults to a timestamped .mp4 in the workspace)', + }, + jsonSchema: { + type: 'object', + properties: { + prompt: { type: 'string' }, + negative_prompt: { type: 'string' }, + width: { type: 'number' }, + height: { type: 'number' }, + num_frames: { type: 'number' }, + fps: { type: 'number' }, + steps: { type: 'number' }, + output: { type: 'string' }, + }, + required: ['prompt'], + additionalProperties: false, + }, + execute: async (args: any): Promise => { + const prompt = String(args?.prompt || '').trim(); + if (!prompt) return { success: false, error: 'prompt is required' }; + + const workspacePath = getWorkspacePath(args); + let outPath = String(args?.output || '').trim(); + if (!outPath) { + outPath = path.join(workspacePath, `ltx_${Date.now()}.mp4`); + } else if (!path.isAbsolute(outPath)) { + outPath = path.resolve(workspacePath, outPath); + } + + const params = { + prompt, + negative_prompt: args?.negative_prompt || 'worst quality, blurry, distorted, deformed', + width: Math.round((args?.width ?? 704) / 32) * 32, + height: Math.round((args?.height ?? 480) / 32) * 32, + num_frames: args?.num_frames ?? 65, + fps: args?.fps ?? 24, + steps: Math.min(50, Math.max(15, args?.steps ?? 30)), + dst: outPath, + }; + + const result = await runVenvPython(LTX_SCRIPT(params), 600_000); + if (result.error) return { success: false, error: result.error, stderr: result.trace || result.raw }; + + const relOut = path.relative(workspacePath, result.output).replace(/\\/g, '/'); + const sizeMB = (result.size_bytes / 1024 / 1024).toFixed(2); + const durationSec = (result.num_frames / result.fps).toFixed(1); + + return { + success: true, + stdout: [ + `Generated: ${result.width} × ${result.height} px | ${durationSec}s (${result.num_frames}f @ ${result.fps}fps) | ${sizeMB} MB`, + '', + `[${path.basename(result.output)}](/api/files/${relOut})`, + ].join('\n'), + data: { ...result, rel_path: relOut }, + }; + }, +}; diff --git a/src/tools/registry.ts b/src/tools/registry.ts index d28d3f1..ab8c2ca 100644 --- a/src/tools/registry.ts +++ b/src/tools/registry.ts @@ -17,6 +17,7 @@ import { openalexSearchTool, semanticSearchTool } from './scholar.js'; import { pdfReadTool } from './pdf.js'; import { pdfExtractImagesTool, pdfExtractTablesTool } from './pdf-extract.js'; import { imageReadTool, imagePreviewTool, imageInfoTool, imageEditTool } from './image.js'; +import { imageGenerateTool, videoGenerateTool } from './imagegen.js'; import { audioTranscribeTool } from './audio-transcribe.js'; import { pythonEvalTool } from './python.js'; import { sqliteTool } from './sqlite.js'; @@ -204,6 +205,8 @@ class ToolRegistry { this.registerSafe(imagePreviewTool); this.registerSafe(imageInfoTool); this.registerSafe(imageEditTool); + this.registerSafe(imageGenerateTool); + this.registerSafe(videoGenerateTool); this.registerSafe(audioTranscribeTool); this.registerSafe(pythonEvalTool); this.registerSafe(sqliteTool);