From 7bc1b1bb65921d2b35e0dc75fdfbbaa740e06410 Mon Sep 17 00:00:00 2001 From: kim Date: Fri, 24 Jul 2026 16:06:29 +0900 Subject: [PATCH] =?UTF-8?q?v4.3.0:=20Chroma=20=EB=B2=A1=ED=84=B0=20?= =?UTF-8?q?=EB=A9=94=EB=AA=A8=EB=A6=AC=20=EC=8B=9C=EC=8A=A4=ED=85=9C=20?= =?UTF-8?q?=EA=B5=AC=EC=B6=95=20+=20=ED=95=98=EB=93=9C=EC=BD=94=EB=94=A9?= =?UTF-8?q?=20=EA=B2=8C=EC=9D=B4=ED=8A=B8=20=EC=A0=95=EB=A6=AC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - USER.md/SOUL.md 문자캡 잘림 보완용 벡터 검색(Chroma, EmbeddingGemma) 신설 - 일별 대화로그 LLM 추출 파이프라인(user_facts/daily_extracts 컬렉션 분리) - 검증-강제 게이트(isFactualInfoRequest 등)를 prompt-gates.ts로 분리, 개인 인프라 별명(지서버/클로서버) 관련 불필요한 web_search 강제 호출 버그 수정 - Qwen3-Embedding-8B는 실측 결과 이미지생성 GPU와 자원 충돌 확인되어 폐기, EmbeddingGemma(경량+다국어)로 최종 선택 Co-Authored-By: Claude Sonnet 5 --- .gitignore | 3 + .../users/papa/workspace/prompts/USER.md | 2 +- package-lock.json | 101 ++++++++- package.json | 3 +- scripts/extract-daily-memory.ts | 119 ++++++++++ scripts/migrate-memory-to-chroma.ts | 147 +++++++++++++ scripts/reembed-collections.ts | 49 +++++ src/gateway/memory-vector.ts | 165 ++++++++++++++ src/gateway/prompt-gates.ts | 155 +++++++++++++ src/gateway/server-v2.ts | 203 ++++++++---------- 10 files changed, 827 insertions(+), 120 deletions(-) create mode 100644 scripts/extract-daily-memory.ts create mode 100644 scripts/migrate-memory-to-chroma.ts create mode 100644 scripts/reembed-collections.ts create mode 100644 src/gateway/memory-vector.ts create mode 100644 src/gateway/prompt-gates.ts diff --git a/.gitignore b/.gitignore index 893254f..9eb42b8 100644 --- a/.gitignore +++ b/.gitignore @@ -121,6 +121,9 @@ mnt/ .tmp_openclaw_*/ .tmp_codex_* +# --- VECTOR MEMORY (Chroma venv + data, large binary) --- +.chroma/ + # --- NODE --- node_modules/ dist/ diff --git a/.smallclaw/users/papa/workspace/prompts/USER.md b/.smallclaw/users/papa/workspace/prompts/USER.md index c1a2adb..9cbee74 100644 --- a/.smallclaw/users/papa/workspace/prompts/USER.md +++ b/.smallclaw/users/papa/workspace/prompts/USER.md @@ -22,7 +22,7 @@ - Local LLM Server name: 지서버 (Z-Server). [2026-07-03] [2026-07-03] - Web server name: 클로서버 (Claw-Server). [2026-07-03] [2026-07-03] - Web server (클로서버) 사양: HP ProDesk 600 G1 DM, 16GB RAM. [2026-07-06] [2026-07-06] -- Web server (클로서버) 사양 업그레이드: Dell T5810, Xeon E5-2683 v4, 40GB RAM, RTX 3060 12GB. [2026-07-09] [2026-07-09] +- Web server (클로서버) 사양 업그레이드: Dell T5810, Xeon E5-2683 v4, 40GB RAM, RTX 3060 12GB ×2개(총 24GB VRAM). [2026-07-09] [2026-07-09] - 클로서버(Claw-Server) AI 미디어 파이프라인 구성: [음성 STT/TTS] diff --git a/package-lock.json b/package-lock.json index e65c6fb..979c4bf 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "smallclaw", - "version": "4.2.1", + "version": "4.3.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "smallclaw", - "version": "4.2.1", + "version": "4.3.0", "license": "MIT", "dependencies": { "@xterm/addon-fit": "^0.10.0", @@ -14,6 +14,7 @@ "@xterm/xterm": "^5.5.0", "avr8js": "^0.21.0", "better-sqlite3": "^12.9.0", + "chromadb": "^3.5.0", "commander": "^14.0.3", "cors": "^2.8.6", "croner": "^10.0.1", @@ -879,6 +880,102 @@ "integrity": "sha512-jJ0bqzaylmJtVnNgzTeSOs8DPavpbYgEr/b0YL8/2GO3xJEhInFmhKMUnEJQjZumK7KXGFhUy89PrsJWlakBVg==", "license": "ISC" }, + "node_modules/chromadb": { + "version": "3.5.0", + "resolved": "https://registry.npmjs.org/chromadb/-/chromadb-3.5.0.tgz", + "integrity": "sha512-A68f2RQjY3R95Pzy0j3ykOu6fi+WWCN1tjbvzrZXZoYQ1d7ZjgWr1FyzitZCeaz/0qmc8y2LEK7QmlxloOAs9Q==", + "dependencies": { + "semver": "^7.7.1" + }, + "bin": { + "chroma": "dist/cli.mjs" + }, + "engines": { + "node": ">=20" + }, + "optionalDependencies": { + "chromadb-js-bindings-darwin-arm64": "^1.3.4", + "chromadb-js-bindings-darwin-x64": "^1.3.4", + "chromadb-js-bindings-linux-arm64-gnu": "^1.3.4", + "chromadb-js-bindings-linux-x64-gnu": "^1.3.4", + "chromadb-js-bindings-win32-x64-msvc": "^1.3.4" + } + }, + "node_modules/chromadb-js-bindings-darwin-arm64": { + "version": "1.3.4", + "resolved": "https://registry.npmjs.org/chromadb-js-bindings-darwin-arm64/-/chromadb-js-bindings-darwin-arm64-1.3.4.tgz", + "integrity": "sha512-pDdWCcR0hCz3pSDiIijBito7YG6FumewfzYE2mIjSwjrO1CKSfQKLwDdTVK4ZNpVGQa16ePoMX+pg7ShSUARJg==", + "cpu": [ + "arm64" + ], + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/chromadb-js-bindings-darwin-x64": { + "version": "1.3.4", + "resolved": "https://registry.npmjs.org/chromadb-js-bindings-darwin-x64/-/chromadb-js-bindings-darwin-x64-1.3.4.tgz", + "integrity": "sha512-5vKVFXSFo+S9qC9by/7oofjcXqQK4IGHiUnPrvTfc7B52gNlPSKeyDO1wha556COc3KtXSZjICEH0nYBJ8UXng==", + "cpu": [ + "x64" + ], + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/chromadb-js-bindings-linux-arm64-gnu": { + "version": "1.3.4", + "resolved": "https://registry.npmjs.org/chromadb-js-bindings-linux-arm64-gnu/-/chromadb-js-bindings-linux-arm64-gnu-1.3.4.tgz", + "integrity": "sha512-ELiqDZzU5mZd1BvZugGrhoMkVeNyQaa+PGizd06WIpIJVoHY+S+9ZBjdeM6ZkRnL3vTxD79XBrosxNWhzqy/kg==", + "cpu": [ + "arm64" + ], + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/chromadb-js-bindings-linux-x64-gnu": { + "version": "1.3.4", + "resolved": "https://registry.npmjs.org/chromadb-js-bindings-linux-x64-gnu/-/chromadb-js-bindings-linux-x64-gnu-1.3.4.tgz", + "integrity": "sha512-HmTUe7PEIaBngCQ70Yyle81z2wYyg9OMgJYyRUNBYCBp4ED6zTmdaOro41Vs/c/h2UyQeBXFIKNmHDeD2Nr2fw==", + "cpu": [ + "x64" + ], + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/chromadb-js-bindings-win32-x64-msvc": { + "version": "1.3.4", + "resolved": "https://registry.npmjs.org/chromadb-js-bindings-win32-x64-msvc/-/chromadb-js-bindings-win32-x64-msvc-1.3.4.tgz", + "integrity": "sha512-punrofQRzKspNMxHmv24qoPeRycV2T3KfedekGV4lOuPPG14i6qggS4x/OWLSAK7ozPfBbrzCejvCa5wfvujhQ==", + "cpu": [ + "x64" + ], + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 10" + } + }, "node_modules/commander": { "version": "14.0.3", "resolved": "https://registry.npmjs.org/commander/-/commander-14.0.3.tgz", diff --git a/package.json b/package.json index a322652..3d49e05 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "smallclaw", - "version": "4.2.2", + "version": "4.3.0", "description": "Local AI agent framework powered by Ollama - OpenClaw alternative", "main": "dist/index.js", "bin": { @@ -42,6 +42,7 @@ "@xterm/xterm": "^5.5.0", "avr8js": "^0.21.0", "better-sqlite3": "^12.9.0", + "chromadb": "^3.5.0", "commander": "^14.0.3", "cors": "^2.8.6", "croner": "^10.0.1", diff --git a/scripts/extract-daily-memory.ts b/scripts/extract-daily-memory.ts new file mode 100644 index 0000000..3480d28 --- /dev/null +++ b/scripts/extract-daily-memory.ts @@ -0,0 +1,119 @@ +// Backfill: LLM-extract atomic, reusable facts from raw daily conversation logs and store them +// in the daily_extracts Chroma collection (separate from memory_write's clean user_facts +// collection — see project_vector_memory_chroma memory for why). Naive paragraph-chunking of +// raw transcripts was tried first and abandoned (too noisy, see migrate-memory-to-chroma.ts) — +// this replaces that approach with real extraction. +import fs from 'fs'; +import path from 'path'; +import { addVector, hasAnyVector, DAILY_EXTRACTS_COLLECTION } from '../src/gateway/memory-vector'; + +const WORKSPACES: { workspace: string; label: string }[] = [ + { workspace: '/srv/homeclaw/.smallclaw/users/papa/workspace', label: 'papa' }, + { workspace: '/srv/homeclaw/.smallclaw/users/jasmine/workspace', label: 'jasmine' }, + { workspace: '/srv/homeclaw/.smallclaw/users/cherry/workspace', label: 'cherry' }, + { workspace: '/srv/homeclaw/.smallclaw/workspace', label: 'global' }, +]; + +const OLLAMA_ENDPOINT = 'http://localhost:11434'; +const EXTRACT_MODEL = 'gemma4:31b-cloud'; // matches homeclaw's configured primary model + +async function extractFacts(dayContent: string): Promise { + const prompt = `다음은 하루치 대화 로그입니다. 이 중에서 나중에 다시 참고할 가치가 있는 사실만 뽑아줘. + +규칙: +- 재사용 가능한 구체적 정보만 (하드웨어 스펙, 설정값, 결정사항, 프로젝트 진행상황, 사용자의 취향/제약사항 등) +- 일회성 질문-답변, 잡담, 이미 끝난 계산/설명은 제외 +- 각 사실은 한 줄에 하나씩, 문맥 없이도 이해되는 완결된 문장으로 작성 (예: "클로서버는 RTX 3060 12GB 2개를 사용한다" 처럼 — 어느 대상 얘기인지 반드시 명시) +- 뽑을 게 없으면 "없음"이라고만 답해 +- 사실 목록 외의 설명이나 서두는 쓰지 마 + +--- 로그 시작 --- +${dayContent} +--- 로그 끝 ---`; + + const res = await fetch(`${OLLAMA_ENDPOINT}/api/chat`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + model: EXTRACT_MODEL, + messages: [{ role: 'user', content: prompt }], + stream: false, + }), + signal: AbortSignal.timeout(120_000), + }); + if (!res.ok) throw new Error(`extract HTTP ${res.status}`); + const data: any = await res.json(); + const raw = String(data?.message?.content || '').trim(); + if (!raw || raw === '없음') return []; + + return raw + .split('\n') + .map(l => l.trim().replace(/^[-*]\s*/, '')) + .filter(l => l && l !== '없음'); +} + +async function processWorkspace(workspace: string, label: string) { + const memDir = path.join(workspace, 'memory'); + if (!fs.existsSync(memDir)) { + console.log(`[${label}] no memory dir, skip`); + return; + } + // Never process today's file — it's still being written to during the day, so extracting + // it early would capture a partial day and then skip it forever (hasAnyVector marks it + // "done"). A nightly cron run naturally picks up "yesterday" once it's actually complete. + const today = new Date().toISOString().slice(0, 10); + const dailyFiles = fs.readdirSync(memDir) + .filter(f => /^\d{4}-\d{2}-\d{2}\.md$/.test(f)) + .filter(f => f.replace('.md', '') < today) + .sort(); + let totalFacts = 0, daysProcessed = 0, daysSkipped = 0, daysFailed = 0; + + for (const f of dailyFiles) { + const date = f.replace('.md', ''); + const content = fs.readFileSync(path.join(memDir, f), 'utf-8').trim(); + if (!content) { daysSkipped++; continue; } + + if (await hasAnyVector(DAILY_EXTRACTS_COLLECTION, { workspace, date })) { + daysSkipped++; + continue; + } + + try { + const facts = await extractFacts(content); + if (facts.length === 0) { + // Record a sentinel so hasAnyVector() sees this date as processed on future reruns — + // otherwise a legitimately fact-free day gets re-sent to the LLM every single rerun. + await addVector(DAILY_EXTRACTS_COLLECTION, { + id: `${workspace}:daily-extract:${date}:none`, + text: '(해당 날짜에 재사용 가치 있는 사실 없음)', + metadata: { workspace, date, source: 'daily-extract', empty: true }, + }); + } else { + for (let i = 0; i < facts.length; i++) { + const id = `${workspace}:daily-extract:${date}:${i}`; + await addVector(DAILY_EXTRACTS_COLLECTION, { + id, + text: facts[i], + metadata: { workspace, date, source: 'daily-extract' }, + }); + } + } + totalFacts += facts.length; + daysProcessed++; + console.log(`[${label}] ${date}: ${facts.length}개 추출`); + } catch (err: any) { + daysFailed++; + console.warn(`[${label}] ${date} 실패: ${err.message}`); + } + } + + console.log(`[${label}] 완료 — ${daysProcessed}일 처리, ${totalFacts}개 사실, ${daysSkipped}일 스킵(빈파일), ${daysFailed}일 실패`); +} + +async function main() { + for (const { workspace, label } of WORKSPACES) { + await processWorkspace(workspace, label); + } +} + +main().then(() => process.exit(0)).catch(err => { console.error(err); process.exit(1); }); diff --git a/scripts/migrate-memory-to-chroma.ts b/scripts/migrate-memory-to-chroma.ts new file mode 100644 index 0000000..e0d1164 --- /dev/null +++ b/scripts/migrate-memory-to-chroma.ts @@ -0,0 +1,147 @@ +// One-off backfill: index existing USER.md/SOUL.md bullet facts and daily memory files into +// Chroma. Run once after wiring up memory-vector.ts. Safe to re-run (uses deterministic IDs, +// Chroma's add() upserts on matching ID... actually add() errors on duplicate ID in this +// version, so this script skips ids already present). +import fs from 'fs'; +import path from 'path'; +import { addMemoryVector } from '../src/gateway/memory-vector'; + +const WORKSPACES: { workspace: string; label: string }[] = [ + { workspace: '/srv/homeclaw/.smallclaw/users/papa/workspace', label: 'papa' }, + { workspace: '/srv/homeclaw/.smallclaw/users/jasmine/workspace', label: 'jasmine' }, + { workspace: '/srv/homeclaw/.smallclaw/users/cherry/workspace', label: 'cherry' }, + { workspace: '/srv/homeclaw/.smallclaw/workspace', label: 'global' }, +]; + +// nomic-embed-text's context window is 2048 tokens — long daily-memory files (some 30KB+) +// blow past that and the embed call 500s. Chunk on paragraph boundaries, ~3000 chars/chunk +// (conservative for mixed Korean/English token density) rather than embedding the whole file. +function chunkText(text: string, maxChars = 1500): string[] { + const paragraphs = text.split(/\n\n+/); + const chunks: string[] = []; + let current = ''; + for (const para of paragraphs) { + if (current && (current.length + para.length + 2) > maxChars) { + chunks.push(current); + current = para; + } else { + current = current ? `${current}\n\n${para}` : para; + } + // A single paragraph longer than maxChars on its own — hard-split it. + while (current.length > maxChars) { + chunks.push(current.slice(0, maxChars)); + current = current.slice(maxChars); + } + } + if (current) chunks.push(current); + return chunks; +} + +// memory_write always appends a trailing [YYYY-MM-DD] to what it writes (see the memory_write +// handler in server-v2.ts) — so a bullet WITHOUT one was never written atomically via that path; +// it's hand-authored nested content (a "parent:" line introducing a block, often with [bracket] +// sub-headers underneath, e.g. "클로서버 AI 미디어 파이프라인 구성:" > "[음성 STT/TTS]" > "- TTS 메인: ..."). +// Splitting those sub-bullets into standalone vectors loses which parent they belong to — verified +// 2026-07-24: a recalled "TTS 메인: OmniVoice..." fact (really about 클로서버) got misattributed to +// 지서버 by the model once it lost that context. Fix: group undated bullets with their nearest +// colon-ending parent (+ bracket sub-header, if any) into one combined fact instead of splitting. +function parseBullets(fileContent: string, fileLabel: 'user' | 'soul'): { category: string; text: string; date: string }[] { + const facts: { category: string; text: string; date: string }[] = []; + let currentCategory = 'general'; + let parentLine: string | null = null; + let bracketHeader: string | null = null; + let buffer: string[] = []; + + const flush = () => { + if (buffer.length === 0) return; + const prefixParts = [parentLine, bracketHeader].filter(Boolean); + const text = (prefixParts.length ? `${prefixParts.join(' — ')}: ` : '') + buffer.join(' / '); + facts.push({ category: currentCategory, text, date: '' }); + buffer = []; + }; + + for (const line of fileContent.split('\n')) { + const headerMatch = line.match(/^##\s+(.+)$/); + if (headerMatch) { + flush(); + parentLine = null; + bracketHeader = null; + currentCategory = headerMatch[1].trim(); + continue; + } + const bracketMatch = line.match(/^\[(.+)\]$/); + if (bracketMatch) { + flush(); + bracketHeader = bracketMatch[1].trim(); + continue; + } + const bulletMatch = line.match(/^-\s+(.+)$/); + if (!bulletMatch) continue; + const text = bulletMatch[1].trim(); + const dateMatch = text.match(/\[(\d{4}-\d{2}-\d{2})\]/); + if (dateMatch) { + // Standalone dated fact — flush any pending nested block first, then emit this as-is. + flush(); + parentLine = null; + bracketHeader = null; + facts.push({ category: currentCategory, text, date: dateMatch[1] }); + } else if (/:$/.test(text)) { + // Introduces a new nested block. + flush(); + parentLine = text.replace(/:$/, ''); + bracketHeader = null; + } else if (parentLine) { + // Sub-item of the current nested block. + buffer.push(text); + } else { + // Undated, no parent context (e.g. a plain continuation bullet) — keep as its own fact + // rather than silently dropping it. + facts.push({ category: currentCategory, text, date: '' }); + } + } + flush(); + return facts; +} + +async function migrateWorkspace(workspace: string, label: string) { + let added = 0, skipped = 0, failed = 0; + + for (const file of ['USER.md', 'SOUL.md'] as const) { + const p = path.join(workspace, 'prompts', file); + if (!fs.existsSync(p)) continue; + const content = fs.readFileSync(p, 'utf-8'); + const facts = parseBullets(content, file === 'USER.md' ? 'user' : 'soul'); + for (let i = 0; i < facts.length; i++) { + const f = facts[i]; + const id = `${workspace}:${file}:${f.category}:migrated:${i}`; + try { + await addMemoryVector({ + id, + text: f.text, + metadata: { workspace, file: file === 'USER.md' ? 'user' : 'soul', category: f.category, date: f.date || 'unknown', source: 'migration' }, + }); + added++; + } catch (err: any) { + if (String(err.message || '').includes('already exists') || String(err.message || '').includes('Insert of existing embedding ID')) skipped++; + else { failed++; console.warn(` [${label}] failed ${id}: ${err.message}`); } + } + } + } + + // Daily memory files (raw conversation transcripts) are deliberately NOT migrated. + // 2026-07-24 test: naive paragraph-chunked transcripts embed noisily and drowned out the + // clean USER.md/SOUL.md bullet facts in top-K results (e.g. a "지서버 hardware spec" query + // surfaced unrelated weather/incident-report chunks instead of the actual hardware fact). + // Chunking raw dialogue well enough to be useful (extraction/summarization, not naive + // paragraph splitting) is a separate, harder problem — out of scope for this pass. + + console.log(`[${label}] added=${added} skipped(existing)=${skipped} failed=${failed}`); +} + +async function main() { + for (const { workspace, label } of WORKSPACES) { + await migrateWorkspace(workspace, label); + } +} + +main().then(() => process.exit(0)).catch(err => { console.error(err); process.exit(1); }); diff --git a/scripts/reembed-collections.ts b/scripts/reembed-collections.ts new file mode 100644 index 0000000..26a9194 --- /dev/null +++ b/scripts/reembed-collections.ts @@ -0,0 +1,49 @@ +// One-off: re-embed existing Chroma collections with a new embedding model. Reuses already- +// stored document text + metadata + ids — does NOT re-run migrate-memory-to-chroma.ts's USER.md +// parsing or extract-daily-memory.ts's LLM extraction, since only the embedding vectors need to +// change, not the underlying facts. Run whenever EMBED_MODEL changes in memory-vector.ts. +import { ChromaClient } from 'chromadb'; +import { addVector, USER_FACTS_COLLECTION, DAILY_EXTRACTS_COLLECTION } from '../src/gateway/memory-vector'; + +const CHROMA_HOST = 'localhost'; +const CHROMA_PORT = 8100; + +async function reembedCollection(name: string) { + const client = new ChromaClient({ host: CHROMA_HOST, port: CHROMA_PORT }); + let existing: { ids: string[]; documents: (string | null)[]; metadatas: any[] } | null = null; + try { + const col = await client.getCollection({ name }); + const res = await col.get({ limit: 10000 }); + existing = { ids: res.ids || [], documents: res.documents || [], metadatas: res.metadatas || [] }; + } catch (err: any) { + console.log(`[${name}] no existing collection or empty (${err.message}) — nothing to reembed`); + return; + } + + console.log(`[${name}] ${existing.ids.length}개 기존 항목 발견, 재임베딩 시작`); + await client.deleteCollection({ name }); + + let done = 0, failed = 0; + for (let i = 0; i < existing.ids.length; i++) { + const id = existing.ids[i]; + const text = existing.documents[i]; + const metadata = existing.metadatas[i] || undefined; + if (!text) continue; + try { + await addVector(name, { id, text, metadata }); + done++; + if (done % 50 === 0) console.log(`[${name}] ${done}/${existing.ids.length}...`); + } catch (err: any) { + failed++; + console.warn(`[${name}] failed ${id}: ${err.message}`); + } + } + console.log(`[${name}] 완료 — ${done}개 재임베딩, ${failed}개 실패`); +} + +async function main() { + await reembedCollection(USER_FACTS_COLLECTION); + await reembedCollection(DAILY_EXTRACTS_COLLECTION); +} + +main().then(() => process.exit(0)).catch(err => { console.error(err); process.exit(1); }); diff --git a/src/gateway/memory-vector.ts b/src/gateway/memory-vector.ts new file mode 100644 index 0000000..6ca8a56 --- /dev/null +++ b/src/gateway/memory-vector.ts @@ -0,0 +1,165 @@ +// Vector-based memory: embeddings via local Ollama (nomic-embed-text), storage/search via +// local Chroma (systemd --user service, port 8100). Sits alongside the flat USER.md/SOUL.md +// files — this is for the long-tail of facts that would otherwise get silently truncated by +// loadFile's fixed char caps (see server-v2.ts loadFile). Core identity stays in the small files; +// everything else that accumulates over time goes here and gets pulled in by relevance, not recency. +// +// Two collections (2026-07-24 design decision, see project_vector_memory_chroma memory): +// - USER_FACTS_COLLECTION: USER.md/SOUL.md-sourced facts via memory_write. Clean, high-trust. +// - DAILY_EXTRACTS_COLLECTION: LLM-extracted facts from raw daily conversation logs. Noisier — +// callers should apply a stricter relevance threshold against this one. +import { ChromaClient, type Collection } from 'chromadb'; +import { getConfig } from '../config/config.js'; + +const CHROMA_HOST = 'localhost'; +const CHROMA_PORT = 8100; +// EmbeddingGemma (300M, Gemma3-based) — switched from nomic-embed-text 2026-07-24: higher MTEB +// multilingual score (61.15 vs 53.01), genuinely multilingual-trained (better Korean semantics +// than nomic's English-centric training), and a smaller real footprint (~680MB loaded vs +// nomic's 274MB — still tiny, no GPU contention risk unlike the Qwen3-Embedding-8B path we +// tried and rejected — that one OOM'd SDXL image generation on GPU1 just from residing there). +// Kept resident (keep_alive below) since the footprint is negligible either way. +const EMBED_MODEL = 'embeddinggemma'; +const EMBED_KEEP_ALIVE = '24h'; + +export const USER_FACTS_COLLECTION = 'homeclaw_memory'; +export const DAILY_EXTRACTS_COLLECTION = 'daily_extracts'; + +function getOllamaEndpoint(): string { + return (getConfig().getConfig() as any).ollama?.endpoint || 'http://localhost:11434'; +} + +// EmbeddingGemma requires its own task-prefix convention (different from nomic's +// search_document:/search_query: — verified via Google's model card, 2026-07-24) or retrieval +// quality degrades the same way nomic's did without its prefix. Documents get "title: none | +// text: ", queries get "task: search result | query: ". +async function embedText(text: string, taskType: 'document' | 'query'): Promise { + const prompt = taskType === 'document' + ? `title: none | text: ${text}` + : `task: search result | query: ${text}`; + const res = await fetch(`${getOllamaEndpoint()}/api/embeddings`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ model: EMBED_MODEL, prompt, keep_alive: EMBED_KEEP_ALIVE }), + signal: AbortSignal.timeout(15_000), + }); + if (!res.ok) throw new Error(`embedText failed: HTTP ${res.status}`); + const data: any = await res.json(); + if (!Array.isArray(data.embedding)) throw new Error('embedText: no embedding in response'); + return data.embedding; +} + +// Exposed so callers querying multiple collections for the same turn (e.g. user_facts + +// daily_extracts) embed the query text once and reuse it, instead of one embed call per +// collection. +export async function embedQuery(text: string): Promise { + return embedText(text, 'query'); +} + +let clientSingleton: ChromaClient | null = null; +const collectionCache = new Map(); + +async function getCollection(name: string): Promise { + const cached = collectionCache.get(name); + if (cached) return cached; + if (!clientSingleton) clientSingleton = new ChromaClient({ host: CHROMA_HOST, port: CHROMA_PORT }); + // cosine space required — default (L2) ranked an unrelated fact above the relevant one + // in the same 2026-07-24 test that surfaced the prefix requirement. + const col = await clientSingleton.getOrCreateCollection({ + name, + metadata: { 'hnsw:space': 'cosine' }, + // We always supply embeddings explicitly (via Ollama) — without this, Chroma tries to + // instantiate its own DefaultEmbeddingFunction on every call and logs a noisy warning. + embeddingFunction: null, + }); + collectionCache.set(name, col); + return col; +} + +// Chroma requires a `where` clause to have exactly one top-level operator — a flat +// multi-key object like {workspace: X, date: Y} errors with "Expected 'where' to have +// exactly one operator, but got 2" (verified 2026-07-24). Callers write plain flat objects; +// this wraps them in $and automatically so they don't all need to know Chroma's where-clause +// quirks. +function toChromaWhere(where?: Record): any { + if (!where) return undefined; + const keys = Object.keys(where); + if (keys.length <= 1) return where; + return { $and: keys.map(k => ({ [k]: where[k] })) }; +} + +export interface MemoryVectorRecord { + id: string; + text: string; + metadata?: Record; +} + +export async function addVector(collectionName: string, rec: MemoryVectorRecord): Promise { + const col = await getCollection(collectionName); + const embedding = await embedText(rec.text, 'document'); + await col.add({ + ids: [rec.id], + embeddings: [embedding], + documents: [rec.text], + metadatas: rec.metadata ? [rec.metadata] : undefined, + }); +} + +export interface MemoryVectorHit { + text: string; + distance: number; + metadata?: Record; +} + +export async function queryVectorsWithEmbedding( + collectionName: string, + embedding: number[], + topK = 6, + where?: Record, +): Promise { + const col = await getCollection(collectionName); + const res = await col.query({ queryEmbeddings: [embedding], nResults: topK, where: toChromaWhere(where) }); + const docs = res.documents?.[0] || []; + const dists = res.distances?.[0] || []; + const metas = res.metadatas?.[0] || []; + return docs.map((text, i) => ({ + text: text || '', + distance: dists[i] ?? Infinity, + metadata: metas[i] as Record | undefined, + })); +} + +// Cheap existence check (no embedding call) — lets batch jobs like extract-daily-memory.ts +// skip already-processed items without needing to know their exact IDs up front. +export async function hasAnyVector( + collectionName: string, + where: Record, +): Promise { + const col = await getCollection(collectionName); + const res = await col.get({ where: toChromaWhere(where), limit: 1 }); + return (res.ids?.length || 0) > 0; +} + +export async function queryVectors( + collectionName: string, + query: string, + topK = 6, + where?: Record, +): Promise { + const embedding = await embedText(query, 'query'); + return queryVectorsWithEmbedding(collectionName, embedding, topK, where); +} + +// --- Backward-compatible convenience wrappers for the user_facts collection (memory_write) --- + +export async function addMemoryVector(rec: MemoryVectorRecord): Promise { + return addVector(USER_FACTS_COLLECTION, rec); +} + +export async function queryMemoryVectors( + query: string, + topK = 6, + where?: Record, +): Promise { + return queryVectors(USER_FACTS_COLLECTION, query, topK, where); +} diff --git a/src/gateway/prompt-gates.ts b/src/gateway/prompt-gates.ts new file mode 100644 index 0000000..68b3e44 --- /dev/null +++ b/src/gateway/prompt-gates.ts @@ -0,0 +1,155 @@ +// Hardcoded (regex-based) gates that force or block model behavior deterministically, as a +// backstop for cases where prompt-level instructions alone aren't reliably followed (e.g. a +// model skipping a required tool call and confidently fabricating an answer instead). +// +// EXEMPTIONS: verification-forcing gates (isFactualInfoRequest, looksLikeUnverifiedSpecClaim) +// must all agree on when NOT to force a search — otherwise fixing one gate's exemption still +// leaves the others forcing the same pointless retry. Verified 2026-07-24: fixing only +// isFactualInfoRequest (input-side) for personal-infra questions still left +// looksLikeUnverifiedSpecClaim (output-side) forcing a retry on the same case, since the +// personal-nickname exemption lived only in the first function. Route every such gate through +// isExemptFromVerification() below instead of duplicating the exemption regex per-gate. + +// The user's own private infra nicknames — these describe personal hardware/servers, not a +// public product, so web_search can never resolve them (there's nothing to find out there). +// Add new private-only terms here — every verification-forcing gate picks it up automatically. +const EXEMPT_FROM_VERIFICATION = [ + /지서버|클로서버|z-server|claw-server/i, +]; + +export function isExemptFromVerification(message: string): boolean { + const m = String(message || ''); + return EXEMPT_FROM_VERIFICATION.some(re => re.test(m)); +} + +export function isGreetingLikeMessage(text: string): boolean { + const raw = String(text || '').trim(); + if (!raw || raw.length > 120) return false; + if (/\b(search|open|read|write|file|code|task|build|fix|debug|run|install|http|www\.|\.com|please|could you|can you)\b/i.test(raw)) { + return false; + } + return /^(hi|hello|hey|yo|sup|howdy|good (morning|afternoon|evening)|hey claw|hello claw|hi claw|hey smallclaw|hello smallclaw|hi smallclaw|how are you)[!.?\s]*$/i.test(raw); +} + +export function isExecutionLikeRequest(message: string): boolean { + const m = String(message || ''); + return /\b(create|build|implement|develop|scaffold|generate|fix|debug|edit|update|refactor|rewrite|patch|setup|configure|calendar|app|component|project|file|folder|directory|workspace|code|desktop|window|screen|mouse|keyboard|clipboard|vs code|vscode)\b/i.test(m) + || /(만들어|생성해|구현해|개발해|고쳐|수정해|편집해|디버그|리팩터|리팩토링|패치|설정해|프로젝트|파일|폴더|디렉터리|워크스페이스|코드|바탕화면|화면|마우스|키보드|클립보드)/.test(m); +} + +export function isLiveDataRequest(message: string): boolean { + const m = String(message || ''); + return /(뉴스|속보|날씨|기온|예보|미세먼지|황사|환율|주가|금리|시세|장마)/.test(m) + || /\b(news|weather|forecast|stock price|exchange rate)\b/i.test(m); +} + +// Broader than isLiveDataRequest: catches factual/product-info questions (specs, prices, +// versions, comparisons, "what models exist") where the model tends to answer fluently from +// memorized training data instead of checking — those numbers go stale or are just wrong +// (e.g. confidently inventing GPU TFLOPS/VRAM figures). Deliberately excludes anything that +// looks like a coding/build request or a greeting, since those aren't "look this up" asks. +export function isFactualInfoRequest(message: string): boolean { + const m = String(message || '').trim(); + if (!m) return false; + if (isExecutionLikeRequest(m) || isGreetingLikeMessage(m)) return false; + if (isExemptFromVerification(m)) return false; + return /\b(vs\.?|versus)\b/i.test(m) + || /(비교|차이점?|스펙|사양|성능|가격|버전|몇\s*(세대|개|년|만원)|추천해|어떤\s*(게|것|모델|제품|카드)|뭐가\s*있|무엇이\s*있|종류가)/.test(m); +} + +// Output-side counterpart to isFactualInfoRequest()/isLiveDataRequest(): those only catch +// predictable INPUT phrasings ("스펙", "비교", "몇 세대"), and casual paraphrases slip through +// unmatched ("초당 몇 토큰 나올까?", "이거 어떨까?") — a model answering those confidently from +// memory ("약 70~80GB", "31 tok/s가 나옵니다") never gets flagged even though it's the exact +// fabrication ANTI-HALLUCINATION warns about. This instead scans what the model actually WROTE +// for a concrete spec/price/benchmark number (digit + a spec-like unit) or a Korean hedge +// phrase ("~것으로 예상", "추정") — a much smaller, harder-to-dodge surface than every possible +// question wording. Skips code fences so numeric literals in code don't false-positive. +// Caller must separately check isExemptFromVerification(message) — a personal-infra answer +// legitimately contains GB/tok-s-shaped numbers and shouldn't force a retry just for that. +export function looksLikeUnverifiedSpecClaim(content: string): boolean { + const text = String(content || ''); + if (!text) return false; + const withoutCode = text.replace(/```[\s\S]*?```/g, ''); + return /\d[\d,.]*\s*(GB|TB|MB|tok\/s|tps|TFLOPS?|GFLOPS?|GHz|MHz|watts?|원|달러|USD|\$|토큰\s*\/\s*초)/i.test(withoutCode) + || /초당\s*(약\s*)?\d/.test(withoutCode) + || /(것으로\s*예상|추정됩니다|추정치|짐작)/.test(withoutCode); +} + +// --- Step 2 (2026-07-24): browser/desktop/messaging-completion gates, moved alongside the +// verification gates above for the same reason — one place to scan for the full set of +// hardcoded behaviors instead of hunting through server-v2.ts. + +export function isBrowserAutomationRequest(message: string): boolean { + const m = String(message || ''); + const hasBrowserVerb = /\b(open|go to|navigate|visit|browse|click|type|fill|press|submit|log ?in|login|use my computer)\b/i.test(m) + || /(열어|들어가|접속해|방문해|클릭해|눌러|입력해|채워|제출해|로그인해)/.test(m); + const hasTarget = /(?:https?:\/\/)?(?:www\.)?[a-z0-9][a-z0-9.-]+\.[a-z]{2,}(?:\/\S*)?/i.test(m) + || /\b(chatgpt|google|reddit|x\.com|twitter|github|youtube|chrome)\b/i.test(m) + || /(네이버|다음|사이트)/.test(m); + return hasBrowserVerb && hasTarget; +} + +export function isDesktopAutomationRequest(message: string): boolean { + const m = String(message || ''); + const hasDesktopVerb = /\b(check|look|see|open|focus|click|type|press|read|copy|paste|use my computer|screenshot)\b/i.test(m) + || /(확인해|봐줘|열어|클릭해|눌러|입력해|읽어|복사해|붙여넣|스크린샷|캡처)/.test(m); + const hasDesktopTarget = /\b(desktop|screen|window|app|application|vs code|vscode|terminal|notepad|clipboard|codex)\b/i.test(m) + || /(바탕화면|화면|창|어플|터미널|메모장|클립보드)/.test(m); + const statusAsk = /\b(is|did|has).*\b(done|finished|complete|completed)\b/i.test(m) + || /(됐|끝났|완료|다\s*됐|다\s*끝났)/.test(m); + return (hasDesktopVerb && hasDesktopTarget) || (statusAsk && /\b(vs code|vscode|codex)\b/i.test(m)); +} + +// A grounding tool call that "succeeded" at the API level but came back with zero real +// content (empty stdout, or one of the known "no results" sentinel strings) is functionally +// the same as no grounding call at all — but the AUTO-RECOVER logic only checks whether a +// grounding tool WAS called, not whether it returned anything usable. Left unchecked, a model +// facing an all-empty turn (e.g. a region NewsData.io has no coverage for, or a Korean query +// string that can't match non-Korean-language articles) still answers fluently from training +// data and cites sources as if they came from the search it just ran. +export function isUsableGroundingResult(r: { error?: boolean; result?: string }): boolean { + if (r.error) return false; + const text = String(r.result || '').trim(); + if (!text) return false; + if (/^\(no articles found\)$/i.test(text)) return false; + if (/에\s*대한\s*검색\s*결과가\s*없습니다\.?$/i.test(text)) return false; + return true; +} + +export function looksLikeSafetyRefusal(text: string): boolean { + const s = String(text || '').trim().toLowerCase(); + if (!s) return false; + return ( + /disallowed|can't (help|assist|do that|use your computer)|cannot (help|assist|do that|use your computer)|unable to (help|assist|do that)/i.test(s) + || /i (can't|cannot) (control|operate|use) (your|the) computer/i.test(s) + || /against (policy|safety)/i.test(s) + ); +} + +export function hasConcreteCompletion(text: string): boolean { + const s = String(text || '').trim(); + if (!s) return false; + return /\b(done|completed|created|updated|fixed|implemented|finished|saved|wrote|executed|here(?:'s| is) (?:the|your)|success(?:fully)?)\b/i.test(s); +} + +export function isMessagingRequest(message: string): boolean { + const m = String(message || ''); + return (/(이메일|메일)/.test(m) && /(보내|전송|발송)/.test(m)) + || /(카카오톡?|카톡)/.test(m) && /(보내|전송|발송)/.test(m) + || /\b(send)\b.*\b(email|mail|kakao)\b/i.test(m); +} + +export function claimsMessageSent(text: string): boolean { + const s = String(text || ''); + return /(보냈습니다|보냈어요|전송했습니다|전송했어요|발송했습니다|발송했어요|보내드렸습니다|전송\s*완료|발송\s*완료)/.test(s) + || /\b(sent the (email|message)|email (has been|was) sent|message (has been|was) sent)\b/i.test(s); +} + +export function isBrowserToolName(name: string): boolean { + return /^browser_(open|snapshot|click|fill|press_key|wait|scroll|close)$/i.test(String(name || '')); +} + +export function isDesktopToolName(name: string): boolean { + return /^desktop_(screenshot|find_window|focus_window|click|drag|wait|type|press_key|get_clipboard|set_clipboard)$/i.test(String(name || '')); +} diff --git a/src/gateway/server-v2.ts b/src/gateway/server-v2.ts index f347b9a..23f2dec 100644 --- a/src/gateway/server-v2.ts +++ b/src/gateway/server-v2.ts @@ -16,6 +16,31 @@ import fs from 'fs'; import fsp from 'fs/promises'; import crypto from 'crypto'; import { WebSocketServer, WebSocket } from 'ws'; +import { + addMemoryVector, + queryMemoryVectors, + embedQuery, + queryVectorsWithEmbedding, + USER_FACTS_COLLECTION, + DAILY_EXTRACTS_COLLECTION, +} from './memory-vector'; +import { + isGreetingLikeMessage, + isExecutionLikeRequest, + isLiveDataRequest, + isFactualInfoRequest, + looksLikeUnverifiedSpecClaim, + isExemptFromVerification, + isBrowserAutomationRequest, + isDesktopAutomationRequest, + isUsableGroundingResult, + looksLikeSafetyRefusal, + hasConcreteCompletion, + isMessagingRequest, + claimsMessageSent, + isBrowserToolName, + isDesktopToolName, +} from './prompt-gates'; import { registerArduinoRoutes } from './routes-arduino'; import { registerAndroidRoutes, attachAndroidWsProxy } from './routes-android'; import { registerPptxRoutes } from './routes-pptx'; @@ -1591,6 +1616,39 @@ async function buildPersonalityContext( // SOUL only when tool categories are detected — pure conversational turns don't need it const soulContent = cats.size > 0 ? loadFile('SOUL.md', 4000) : ''; + // Vector-store recall: relevance-based, not the fixed-recency/fixed-char-cap USER.md/SOUL.md + // path — this is what lets facts survive past loadFile's truncation cutoff (see loadFile + // above). Scoped to this workspace via the `where` filter so users never see each other's + // facts. Best-effort — a Chroma/embedding outage must not break prompt building. + // + // Two collections, one query embedding (2026-07-24 design — see project_vector_memory_chroma + // memory): user_facts (memory_write-sourced, clean) gets a loose threshold; daily_extracts + // (LLM-extracted from raw conversation logs, noisier) gets a stricter one. Both merged and + // ranked together by distance so the model just sees one flat, relevance-sorted list. + let vectorMemory = ''; + if (messageText.trim().length >= 4) { + try { + const queryEmbedding = await embedQuery(messageText); + const [userFactHits, dailyExtractHits] = await Promise.all([ + queryVectorsWithEmbedding(USER_FACTS_COLLECTION, queryEmbedding, 6, { workspace: workspacePath }), + queryVectorsWithEmbedding(DAILY_EXTRACTS_COLLECTION, queryEmbedding, 6, { workspace: workspacePath }), + ]); + // Thresholds recalibrated for EmbeddingGemma's wider distance spread (2026-07-24 switch + // from nomic-embed-text) — a 3-fact spot check gave ~0.51 for a genuinely relevant match, + // ~0.66 for same-topic-wrong-entity, ~0.97 for unrelated. Rough starting points, not a + // rigorous calibration — revisit if recall feels off/noisy in real use. + const relevant = [ + ...userFactHits.filter(h => h.distance < 0.65), + ...dailyExtractHits.filter(h => h.distance < 0.55 && !h.metadata?.empty), + ].sort((a, b) => a.distance - b.distance).slice(0, 8); + if (relevant.length > 0) { + vectorMemory = relevant.map(h => `- ${h.text}`).join('\n'); + } + } catch (err: any) { + console.warn('[buildPersonalityContext] vector memory query failed (non-fatal):', err.message); + } + } + const parts = [ loadFile('IDENTITY.md', 1500) ? `[IDENTITY]\n${loadFile('IDENTITY.md', 1500)}` : '', loadFile('USER.md', 3000) ? `[USER]\n${loadFile('USER.md', 3000)}` : '', @@ -1599,6 +1657,7 @@ async function buildPersonalityContext( intradayNotes ? `[TODAY_NOTES]\n${intradayNotes}` : '', toolBlockParts.length > 0 ? `[TOOLS]\n${toolBlockParts.join('\n\n')}${toolsHint}` : (toolsHint ? `[TOOLS]${toolsHint}` : ''), memorySnippets ? `[RELEVANT_MEMORY]\n${memorySnippets}` : '', + vectorMemory ? `[RECALLED_FACTS]\n${vectorMemory}` : '', self ? `[SELF]\n${self}` : '', ].filter(Boolean); @@ -4423,6 +4482,14 @@ print(json.dumps({'slides': slides, 'total': len(prs.slides)}, ensure_ascii=Fals mwFileContent = mwFileContent.slice(0, insertAt) + '\n\n' + mwSectionHeader + '\n' + mwEntry + mwFileContent.slice(insertAt); } fs.writeFileSync(mwPath, mwFileContent, 'utf-8'); + // Best-effort: also index in the vector store so it survives USER.md/SOUL.md's fixed + // char-cap truncation and can be retrieved by relevance later, not just recency. Never + // let a Chroma/embedding hiccup fail the primary (file) write. + addMemoryVector({ + id: `${workspacePath}:${mwFile}:${mwCategory}:${Date.now()}`, + text: mwContent, + metadata: { workspace: workspacePath, file: mwFile, category: mwCategory, date: mwDate }, + }).catch(err => console.warn('[memory_write] vector index failed (non-fatal):', err.message)); return { name, args, result: `Written to ${mwFilename} [${mwCategory}]: ${mwContent}`, error: false }; } @@ -4657,15 +4724,6 @@ function normalizeForDedup(text: string): string { return raw.replace(/[^a-z0-9]+/g, ''); } -function isGreetingLikeMessage(text: string): boolean { - const raw = String(text || '').trim(); - if (!raw || raw.length > 120) return false; - if (/\b(search|open|read|write|file|code|task|build|fix|debug|run|install|http|www\.|\.com|please|could you|can you)\b/i.test(raw)) { - return false; - } - return /^(hi|hello|hey|yo|sup|howdy|good (morning|afternoon|evening)|hey claw|hello claw|hi claw|hey smallclaw|hello smallclaw|hi smallclaw|how are you)[!.?\s]*$/i.test(raw); -} - function sanitizeFinalReply( text: string, opts: { preflightReason?: string } = {}, @@ -4751,33 +4809,6 @@ function stripExplicitThinkTags(text: string): { cleaned: string; thinking: stri } -function isExecutionLikeRequest(message: string): boolean { - const m = String(message || ''); - return /\b(create|build|implement|develop|scaffold|generate|fix|debug|edit|update|refactor|rewrite|patch|setup|configure|calendar|app|component|project|file|folder|directory|workspace|code|desktop|window|screen|mouse|keyboard|clipboard|vs code|vscode)\b/i.test(m) - || /(만들어|생성해|구현해|개발해|고쳐|수정해|편집해|디버그|리팩터|리팩토링|패치|설정해|프로젝트|파일|폴더|디렉터리|워크스페이스|코드|바탕화면|화면|마우스|키보드|클립보드)/.test(m); -} - -function isBrowserAutomationRequest(message: string): boolean { - const m = String(message || ''); - const hasBrowserVerb = /\b(open|go to|navigate|visit|browse|click|type|fill|press|submit|log ?in|login|use my computer)\b/i.test(m) - || /(열어|들어가|접속해|방문해|클릭해|눌러|입력해|채워|제출해|로그인해)/.test(m); - const hasTarget = /(?:https?:\/\/)?(?:www\.)?[a-z0-9][a-z0-9.-]+\.[a-z]{2,}(?:\/\S*)?/i.test(m) - || /\b(chatgpt|google|reddit|x\.com|twitter|github|youtube|chrome)\b/i.test(m) - || /(네이버|다음|사이트)/.test(m); - return hasBrowserVerb && hasTarget; -} - -function isDesktopAutomationRequest(message: string): boolean { - const m = String(message || ''); - const hasDesktopVerb = /\b(check|look|see|open|focus|click|type|press|read|copy|paste|use my computer|screenshot)\b/i.test(m) - || /(확인해|봐줘|열어|클릭해|눌러|입력해|읽어|복사해|붙여넣|스크린샷|캡처)/.test(m); - const hasDesktopTarget = /\b(desktop|screen|window|app|application|vs code|vscode|terminal|notepad|clipboard|codex)\b/i.test(m) - || /(바탕화면|화면|창|어플|터미널|메모장|클립보드)/.test(m); - const statusAsk = /\b(is|did|has).*\b(done|finished|complete|completed)\b/i.test(m) - || /(됐|끝났|완료|다\s*됐|다\s*끝났)/.test(m); - return (hasDesktopVerb && hasDesktopTarget) || (statusAsk && /\b(vs code|vscode|codex)\b/i.test(m)); -} - function extractLikelyUrl(message: string): string | null { const raw = String(message || ''); const directUrlMatch = raw.match(/\bhttps?:\/\/[^\s)]+/i); @@ -4788,48 +4819,12 @@ function extractLikelyUrl(message: string): string | null { return normalized.replace(/["'<>]/g, ''); } -function isLiveDataRequest(message: string): boolean { - const m = String(message || ''); - return /(뉴스|속보|날씨|기온|예보|미세먼지|황사|환율|주가|금리|시세|장마)/.test(m) - || /\b(news|weather|forecast|stock price|exchange rate)\b/i.test(m); -} - -// Broader than isLiveDataRequest: catches factual/product-info questions (specs, prices, -// versions, comparisons, "what models exist") where the model tends to answer fluently from -// memorized training data instead of checking — those numbers go stale or are just wrong -// (e.g. confidently inventing GPU TFLOPS/VRAM figures). Deliberately excludes anything that -// looks like a coding/build request or a greeting, since those aren't "look this up" asks. -function isFactualInfoRequest(message: string): boolean { - const m = String(message || '').trim(); - if (!m) return false; - if (isExecutionLikeRequest(m) || isGreetingLikeMessage(m)) return false; - return /\b(vs\.?|versus)\b/i.test(m) - || /(비교|차이점?|스펙|사양|성능|가격|버전|몇\s*(세대|개|년|만원)|추천해|어떤\s*(게|것|모델|제품|카드)|뭐가\s*있|무엇이\s*있|종류가)/.test(m); -} - // Per-model behavioral overrides. Some models (e.g. mistral-large-3) have observed quirks — // skipping tool calls and confidently fabricating instead — that the generic system prompt // doesn't fully correct. Rather than hardcoding model names in prompt text, config.json's // models.profiles[modelName].extraSystemPrompt lets us attach model-specific reminders that // only apply when that exact model is the one actually serving the turn, discovered/edited // without a code deploy. -// A grounding tool call that "succeeded" at the API level but came back with zero real -// content (empty stdout, or one of the known "no results" sentinel strings) is functionally -// the same as no grounding call at all — but the existing AUTO-RECOVER logic only checks -// whether a grounding tool WAS called, not whether it returned anything usable. Left -// unchecked, a model facing an all-empty turn (e.g. a region NewsData.io has no coverage -// for, or a Korean query string that can't match non-Korean-language articles) still answers -// fluently from training data and cites sources as if they came from the search it just ran — -// reads as authoritative, which is worse than an honest "no results found". -function isUsableGroundingResult(r: { error?: boolean; result?: string }): boolean { - if (r.error) return false; - const text = String(r.result || '').trim(); - if (!text) return false; - if (/^\(no articles found\)$/i.test(text)) return false; - if (/에\s*대한\s*검색\s*결과가\s*없습니다\.?$/i.test(text)) return false; - return true; -} - function getModelProfileExtraPrompt(modelName: string): string { const m = String(modelName || '').trim(); if (!m) return ''; @@ -4842,16 +4837,6 @@ function getModelProfileExtraPrompt(modelName: string): string { } } -function looksLikeSafetyRefusal(text: string): boolean { - const s = String(text || '').trim().toLowerCase(); - if (!s) return false; - return ( - /disallowed|can't (help|assist|do that|use your computer)|cannot (help|assist|do that|use your computer)|unable to (help|assist|do that)/i.test(s) - || /i (can't|cannot) (control|operate|use) (your|the) computer/i.test(s) - || /against (policy|safety)/i.test(s) - ); -} - function looksLikeIntentOnlyReply(text: string): boolean { const s = String(text || '').trim(); if (!s) return true; @@ -4864,33 +4849,6 @@ function looksLikeIntentOnlyReply(text: string): boolean { return intentPattern.test(s); } -function hasConcreteCompletion(text: string): boolean { - const s = String(text || '').trim(); - if (!s) return false; - return /\b(done|completed|created|updated|fixed|implemented|finished|saved|wrote|executed|here(?:'s| is) (?:the|your)|success(?:fully)?)\b/i.test(s); -} - -function isMessagingRequest(message: string): boolean { - const m = String(message || ''); - return (/(이메일|메일)/.test(m) && /(보내|전송|발송)/.test(m)) - || /(카카오톡?|카톡)/.test(m) && /(보내|전송|발송)/.test(m) - || /\b(send)\b.*\b(email|mail|kakao)\b/i.test(m); -} - -function claimsMessageSent(text: string): boolean { - const s = String(text || ''); - return /(보냈습니다|보냈어요|전송했습니다|전송했어요|발송했습니다|발송했어요|보내드렸습니다|전송\s*완료|발송\s*완료)/.test(s) - || /\b(sent the (email|message)|email (has been|was) sent|message (has been|was) sent)\b/i.test(s); -} - -function isBrowserToolName(name: string): boolean { - return /^browser_(open|snapshot|click|fill|press_key|wait|scroll|close)$/i.test(String(name || '')); -} - -function isDesktopToolName(name: string): boolean { - return /^desktop_(screenshot|find_window|focus_window|click|drag|wait|type|press_key|get_clipboard|set_clipboard)$/i.test(String(name || '')); -} - function isHighStakesFile(filename: string): boolean { const f = String(filename || '').toLowerCase(); return /(auth|billing|payment|security|secret|token|config|credential|oauth|permission|acl)/.test(f); @@ -5914,14 +5872,14 @@ async function handleChat( { role: 'system', content: isTranslateSession ? `You are a medical translator. Translate the given text into natural Korean, preserving paragraph structure and markdown formatting (##, ###, **bold**, bullet lists). Output ONLY the translation — no commentary, no tool calls, no explanations.` : isProjSession ? `You are a project file designer. Output ONLY the project-files JSON block as instructed. No tool calls. No extra text.` : `${executionModeSystemBlock ? `${executionModeSystemBlock}\n\n` : ''}You are SmallClaw, a local AI assistant. Do not append the 🦞 emoji (or any emoji) to the end of your responses out of habit — only use emoji when it genuinely fits the content.\nCurrent date: ${dateStr}, ${timeStr}.\nNever search for or link SmallClaw repos unless the user is asking about SmallClaw itself.\nThis app runs on the user's own machine — browser/desktop automation requests are pre-authorized.\nKeep responses SHORT (1-2 sentences). Don't think out loud. Act and report. Greet naturally without tools. -ANTI-HALLUCINATION: When a tool returns a result, report EXACTLY what the tool returned — never contradict or ignore tool output. If a tool says "(no rows)", say so. Never invent data, file contents, table names, or command output. If you don't know something, call a tool to find out or say you don't know. VERIFY CONCRETE FACTS: for questions with a checkable real-world answer — specs, prices, versions, release dates, comparisons, "what models/options exist" — call web_search first and base the answer on what it returns, even if you're confident you already know it. Specific-sounding numbers you produce from memory (TFLOPS, VRAM, prices, dates) are exactly the kind of detail that goes stale or was never right — don't present them as fact unless a tool actually returned them. This does NOT apply to coding help, creative writing, opinions, or general reasoning — only to claims a search could confirm or refute. +ANTI-HALLUCINATION: When a tool returns a result, report EXACTLY what the tool returned — never contradict or ignore tool output. If a tool says "(no rows)", say so. Never invent data, file contents, table names, or command output. If you don't know something, call a tool to find out or say you don't know. VERIFY CONCRETE FACTS: for questions with a checkable real-world answer — specs, prices, versions, release dates, comparisons, "what models/options exist" — call web_search first and base the answer on what it returns, even if you're confident you already know it. Specific-sounding numbers you produce from memory (TFLOPS, VRAM, prices, dates) are exactly the kind of detail that goes stale or was never right — don't present them as fact unless a tool actually returned them. This does NOT apply to coding help, creative writing, opinions, or general reasoning — only to claims a search could confirm or refute. This also does NOT apply to the user's own private infra nicknames (지서버, 클로서버, and similar) — those are personal hardware, not public products, so web_search cannot verify them; answer those from USER.md/SOUL.md/[RECALLED_FACTS] context instead. TEMPORAL CONSISTENCY: The current date and day-of-week is given above ("Current date: ${dateStr}") — this is ground truth, more reliable than any search snippet's phrasing. Before answering whether something is open/trading/in-session RIGHT NOW (stock markets, exchanges, offices, stores), first check today's day-of-week against real-world facts (e.g. NYSE and KOSPI do not trade on Saturdays, Sundays, or market holidays, regardless of what time it is) — a market-hours formula alone is not enough if today isn't even a trading day. A search result reporting a "closing price" or "today's headline" is not proof today is a trading/business day — cross-check it against the actual current date above, and if they conflict (e.g. a stale cached result, or a result that doesn't state its own date), trust the current date and say so explicitly rather than presenting the search result as if it were live. IMAGE EDITING RULE: NEVER call image_edit (or any editing tool) when a user uploads a photo without explicitly requesting edits. Uploading a photo is NOT a request to edit it. Only call image_edit when the user's message explicitly asks for an edit (e.g. "수채화로 바꿔줘", "회전해줘"). Violating this rule is a critical error. IMAGE/VIDEO GENERATION: When a user asks to create/draw/generate a NEW image from a description (no existing photo involved), call image_generate (local SDXL, 10-20 seconds by default). If the user explicitly asks for higher quality/detail/photorealism (e.g. "고품질로", "디테일 살려서"), pass quality="high" instead — this uses FLUX.1-schnell, noticeably better detail but 40-60 seconds total, so tell the user it'll take a bit before calling it. When they ask for a short video/clip/animation from a description, call video_generate (local LTX-Video, 30–90 seconds — tell the user it'll take a bit before calling it). All run entirely on local GPU hardware, no external API or cost. Do not confuse these with image_edit, which only modifies an existing uploaded/generated image.${browserRuleBlock} CHEMISTRY NOTATION: NEVER draw molecular structures as ASCII art (H/C/#/=/\\/| characters arranged to look like a diagram) — it always renders as garbled, misaligned text. Instead: for a formula or reaction, use LaTeX inside $...$ (e.g. $\\ce{C4H10}$, $\\ce{2H2 + O2 -> 2H2O}$ — mhchem extension is loaded). For an actual 2D structure with real bond lines (rings, branches), output a \`\`\`smiles\`\`\` code block containing the SMILES string (e.g. \`\`\`smiles\\nc1ccccc1\\n\`\`\` for benzene, \`\`\`smiles\\nCC(=O)Oc1ccccc1C(=O)O\\n\`\`\` for aspirin) — the client automatically renders it as a proper 2D diagram with bond lines. CODE OUTPUT: When writing code in a fenced code block, start with a filename comment on line 1: \`# filename: snake_game.py\` (Python), \`// filename: app.js\` (JS/C), \`\` (HTML). Never repeat code already written in this conversation. For modifications to existing files, use coder_overwrite_lines or coder_insert_lines (not coder_write_file — it only works for NEW files). All code changes are presented as diffs for the user to review before being applied. Write code directly — do not ask for permission. PACKAGE INSTALL: NEVER run pip install, npm install, apt-get, or any package installation command. If a package is missing, just write the code and mention the package name in a comment — let the user decide whether to install it. Do NOT attempt to install packages yourself. -OUTPUT FORMAT: When presenting 3+ items (news articles, emails, search results, lists), always use a markdown table or structured bullet list with clear headers. Never dump them as a long paragraph. Example: news → table with columns 제목|요약|출처.${modelProfileBlock}${callerContext ? '\n\n' + callerContext : ''}${browserStateCtx}${personalityCtx}${skillsManager.buildPromptContextForUser(username ? getUserWorkspace(username) : null, 16000, message)}`, +OUTPUT FORMAT: When presenting 3+ items (emails, search results, lists), always use a markdown table or structured bullet list with clear headers. Never dump them as a long paragraph. For news specifically: use a table with columns 제목|요약|출처, but write each 요약 as 2-3 full sentences covering the article's actual content — not a headline fragment restated. For weather: do NOT force a table — explain current conditions and forecast in natural, fuller sentences (trend, precipitation chance, notable changes vs yesterday/normal), not just bare numbers.${modelProfileBlock}${callerContext ? '\n\n' + callerContext : ''}${browserStateCtx}${personalityCtx}${skillsManager.buildPromptContextForUser(username ? getUserWorkspace(username) : null, 16000, message)}`, }, ]; // Emit the fixed-overhead size (system prompt + tool schemas) as soon as both are known — @@ -5970,13 +5928,13 @@ OUTPUT FORMAT: When presenting 3+ items (news articles, emails, search results, messages.push({ role: 'assistant', content: 'Got it — checking now.' }); messages.push({ role: 'user', - content: `Reminder: this needs live/current data. Call a tool first (${isNewsRequest ? 'news_search for news' : 'web_search for news'}, weather_kma/weather_search for weather/forecast, etc.) before writing anything. Do NOT answer from memory or training data.${isNewsRequest ? ' If this message names a specific topic, pass it as news_search\'s query param; otherwise (a bare "뉴스"/"news" request) call news_search with no query for general headlines — do NOT default to unrelated topics from earlier in this conversation unless the user explicitly asks for updates on them.' : ''}`, + content: `Reminder: this needs live/current data. Call a tool first (${isNewsRequest ? 'news_search for news' : 'web_search for news'}, weather_kma/weather_search for weather/forecast, etc.) before writing anything. Do NOT answer from memory or training data.${isNewsRequest ? ' If this message names a specific topic, pass it as news_search\'s query param; otherwise (a bare "뉴스"/"news" request) call news_search several times across different categories (e.g. "top", "business", "technology", "world", "sports") rather than a single bare no-category call — a lone bare call tends to return only 4-5 stories clustered around 1-2 dominant topics; spanning categories like this builds a well-rounded ~8-10 item digest. Do NOT default to unrelated topics from earlier in this conversation unless the user explicitly asks for updates on them.' : ''}`, }); } else if (isFactualInfoRequest(message)) { messages.push({ role: 'assistant', content: 'Got it — let me verify that first.' }); messages.push({ role: 'user', - content: 'Reminder: this asks about concrete facts (specs, prices, versions, comparisons, or "what models/options exist"). Call web_search first to verify before answering — do not rely on memorized details, which are frequently stale or simply wrong. If search turns up nothing useful, say so explicitly rather than filling the gap from memory.', + content: 'Reminder: this asks about concrete facts (specs, prices, versions, comparisons, or "what models/options exist"). For a PUBLIC product/fact, call web_search first to verify before answering — do not rely on memorized details, which are frequently stale or simply wrong. But if this is about the user\'s OWN personal setup (their hardware nicknames like 지서버/클로서버, their own servers/accounts/configs — anything that only exists in USER.md/SOUL.md/[RECALLED_FACTS], not a public product), that is NOT web-searchable — check memory (USER.md, [RECALLED_FACTS] context, memory_read) first instead of wasting a search call on it. Only fall back to web_search for the personal-setup case if memory genuinely has nothing on it. If search turns up nothing useful for a public fact, say so explicitly rather than filling the gap from memory.', }); } @@ -7483,7 +7441,15 @@ RULES: // even when the model's reply looks like a confident final answer rather than stalled // reasoning (a model that skips tools and answers fluently is more dangerous than one // that visibly hesitates, since the fabrication reads as authoritative). - const liveDataRequest = isLiveDataRequest(message) || isFactualInfoRequest(message); + // Output-side catch-all: the input didn't match any known "factual question" phrasing, + // but the model went ahead and stated a spec/price/benchmark number anyway. Excludes + // execution-like requests (coding, file ops) so numbers in those contexts don't misfire. + // Same personal-infra exemption as isFactualInfoRequest (see PERSONAL_INFRA_NICKNAMES) — + // this is the OUTPUT-side counterpart and was missing it: a 지서버/클로서버 hardware answer + // legitimately contains GB/tok/s-shaped numbers, which would otherwise still force a + // pointless web_search retry even after the input-side gate was fixed for the same case. + const unverifiedSpecClaim = !isExecutionLikeRequest(message) && !isExemptFromVerification(message) && looksLikeUnverifiedSpecClaim(content); + const liveDataRequest = isLiveDataRequest(message) || isFactualInfoRequest(message) || unverifiedSpecClaim; // Whether a tool that could actually ground THIS kind of request has already run — // any other tool call (file listing, email check, etc.) doesn't count. const groundingToolPattern = browserAutomationRequest @@ -7494,7 +7460,12 @@ RULES: const hasGroundingToolCall = allToolResults.some((r) => groundingToolPattern.test(String(r?.name || ''))); if (((queryNeedsTools && (looksLikeReasoning || looksLikeRefusal)) || liveDataRequest) && !hasGroundingToolCall) { toolSkipForcedRetries++; - console.log(`[v2] AUTO-RECOVER (${toolSkipForcedRetries}/${MAX_TOOL_SKIP_FORCED_RETRIES}): Model dumped ${content.length} chars${liveDataRequest ? ' (live-data query answered without a grounding tool call)' : ' of reasoning instead of calling tools'}${allToolResults.length > 0 ? ` [${allToolResults.length} unrelated tool call(s) already made this turn]` : ''}. Re-prompting...`); + const liveDataReason = unverifiedSpecClaim + ? ' (unverified spec/price/benchmark number in the answer, no matching input phrasing)' + : liveDataRequest + ? ' (live-data query answered without a grounding tool call)' + : ''; + console.log(`[v2] AUTO-RECOVER (${toolSkipForcedRetries}/${MAX_TOOL_SKIP_FORCED_RETRIES}): Model dumped ${content.length} chars${liveDataReason || ' of reasoning instead of calling tools'}${allToolResults.length > 0 ? ` [${allToolResults.length} unrelated tool call(s) already made this turn]` : ''}. Re-prompting...`); allThinking += (allThinking ? '\n\n' : '') + content; sendSSE('thinking', { thinking: content.slice(0, 500) + '...' }); // Inject a forceful nudge and retry this round @@ -7535,7 +7506,7 @@ RULES: ? 'the news_search tool (NOT web_search)' : 'the web_search tool'; const scopeHint = isNewsRequest - ? ' If this message names a specific topic, pass it as news_search\'s query param; otherwise (a bare "뉴스"/"news" request) call news_search with no query for general headlines — do NOT default to unrelated topics from earlier in this conversation unless the user explicitly asks for updates on them.' + ? ' If this message names a specific topic, pass it as news_search\'s query param; otherwise (a bare "뉴스"/"news" request) call news_search several times across different categories (e.g. "top", "business", "technology", "world", "sports") rather than a single bare no-category call — a lone bare call tends to return only 4-5 stories clustered around 1-2 dominant topics; spanning categories like this builds a well-rounded ~8-10 item digest. Do NOT default to unrelated topics from earlier in this conversation unless the user explicitly asks for updates on them.' : ''; messages.push({ role: 'assistant', content: 'Let me check that now.' }); messages.push({ role: 'user', content: `Yes, use ${toolHint} right now. Do NOT think or plan — just call it.${scopeHint}` });