v4.3.0: Chroma 벡터 메모리 시스템 구축 + 하드코딩 게이트 정리

- USER.md/SOUL.md 문자캡 잘림 보완용 벡터 검색(Chroma, EmbeddingGemma) 신설
- 일별 대화로그 LLM 추출 파이프라인(user_facts/daily_extracts 컬렉션 분리)
- 검증-강제 게이트(isFactualInfoRequest 등)를 prompt-gates.ts로 분리, 개인 인프라
  별명(지서버/클로서버) 관련 불필요한 web_search 강제 호출 버그 수정
- Qwen3-Embedding-8B는 실측 결과 이미지생성 GPU와 자원 충돌 확인되어 폐기,
  EmbeddingGemma(경량+다국어)로 최종 선택

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-24 16:06:29 +09:00
co-authored by Claude Sonnet 5
parent df856877e8
commit 7bc1b1bb65
10 changed files with 827 additions and 120 deletions
+147
View File
@@ -0,0 +1,147 @@
// One-off backfill: index existing USER.md/SOUL.md bullet facts and daily memory files into
// Chroma. Run once after wiring up memory-vector.ts. Safe to re-run (uses deterministic IDs,
// Chroma's add() upserts on matching ID... actually add() errors on duplicate ID in this
// version, so this script skips ids already present).
import fs from 'fs';
import path from 'path';
import { addMemoryVector } from '../src/gateway/memory-vector';
const WORKSPACES: { workspace: string; label: string }[] = [
{ workspace: '/srv/homeclaw/.smallclaw/users/papa/workspace', label: 'papa' },
{ workspace: '/srv/homeclaw/.smallclaw/users/jasmine/workspace', label: 'jasmine' },
{ workspace: '/srv/homeclaw/.smallclaw/users/cherry/workspace', label: 'cherry' },
{ workspace: '/srv/homeclaw/.smallclaw/workspace', label: 'global' },
];
// nomic-embed-text's context window is 2048 tokens — long daily-memory files (some 30KB+)
// blow past that and the embed call 500s. Chunk on paragraph boundaries, ~3000 chars/chunk
// (conservative for mixed Korean/English token density) rather than embedding the whole file.
function chunkText(text: string, maxChars = 1500): string[] {
const paragraphs = text.split(/\n\n+/);
const chunks: string[] = [];
let current = '';
for (const para of paragraphs) {
if (current && (current.length + para.length + 2) > maxChars) {
chunks.push(current);
current = para;
} else {
current = current ? `${current}\n\n${para}` : para;
}
// A single paragraph longer than maxChars on its own — hard-split it.
while (current.length > maxChars) {
chunks.push(current.slice(0, maxChars));
current = current.slice(maxChars);
}
}
if (current) chunks.push(current);
return chunks;
}
// memory_write always appends a trailing [YYYY-MM-DD] to what it writes (see the memory_write
// handler in server-v2.ts) — so a bullet WITHOUT one was never written atomically via that path;
// it's hand-authored nested content (a "parent:" line introducing a block, often with [bracket]
// sub-headers underneath, e.g. "클로서버 AI 미디어 파이프라인 구성:" > "[음성 STT/TTS]" > "- TTS 메인: ...").
// Splitting those sub-bullets into standalone vectors loses which parent they belong to — verified
// 2026-07-24: a recalled "TTS 메인: OmniVoice..." fact (really about 클로서버) got misattributed to
// 지서버 by the model once it lost that context. Fix: group undated bullets with their nearest
// colon-ending parent (+ bracket sub-header, if any) into one combined fact instead of splitting.
function parseBullets(fileContent: string, fileLabel: 'user' | 'soul'): { category: string; text: string; date: string }[] {
const facts: { category: string; text: string; date: string }[] = [];
let currentCategory = 'general';
let parentLine: string | null = null;
let bracketHeader: string | null = null;
let buffer: string[] = [];
const flush = () => {
if (buffer.length === 0) return;
const prefixParts = [parentLine, bracketHeader].filter(Boolean);
const text = (prefixParts.length ? `${prefixParts.join(' — ')}: ` : '') + buffer.join(' / ');
facts.push({ category: currentCategory, text, date: '' });
buffer = [];
};
for (const line of fileContent.split('\n')) {
const headerMatch = line.match(/^##\s+(.+)$/);
if (headerMatch) {
flush();
parentLine = null;
bracketHeader = null;
currentCategory = headerMatch[1].trim();
continue;
}
const bracketMatch = line.match(/^\[(.+)\]$/);
if (bracketMatch) {
flush();
bracketHeader = bracketMatch[1].trim();
continue;
}
const bulletMatch = line.match(/^-\s+(.+)$/);
if (!bulletMatch) continue;
const text = bulletMatch[1].trim();
const dateMatch = text.match(/\[(\d{4}-\d{2}-\d{2})\]/);
if (dateMatch) {
// Standalone dated fact — flush any pending nested block first, then emit this as-is.
flush();
parentLine = null;
bracketHeader = null;
facts.push({ category: currentCategory, text, date: dateMatch[1] });
} else if (/:$/.test(text)) {
// Introduces a new nested block.
flush();
parentLine = text.replace(/:$/, '');
bracketHeader = null;
} else if (parentLine) {
// Sub-item of the current nested block.
buffer.push(text);
} else {
// Undated, no parent context (e.g. a plain continuation bullet) — keep as its own fact
// rather than silently dropping it.
facts.push({ category: currentCategory, text, date: '' });
}
}
flush();
return facts;
}
async function migrateWorkspace(workspace: string, label: string) {
let added = 0, skipped = 0, failed = 0;
for (const file of ['USER.md', 'SOUL.md'] as const) {
const p = path.join(workspace, 'prompts', file);
if (!fs.existsSync(p)) continue;
const content = fs.readFileSync(p, 'utf-8');
const facts = parseBullets(content, file === 'USER.md' ? 'user' : 'soul');
for (let i = 0; i < facts.length; i++) {
const f = facts[i];
const id = `${workspace}:${file}:${f.category}:migrated:${i}`;
try {
await addMemoryVector({
id,
text: f.text,
metadata: { workspace, file: file === 'USER.md' ? 'user' : 'soul', category: f.category, date: f.date || 'unknown', source: 'migration' },
});
added++;
} catch (err: any) {
if (String(err.message || '').includes('already exists') || String(err.message || '').includes('Insert of existing embedding ID')) skipped++;
else { failed++; console.warn(` [${label}] failed ${id}: ${err.message}`); }
}
}
}
// Daily memory files (raw conversation transcripts) are deliberately NOT migrated.
// 2026-07-24 test: naive paragraph-chunked transcripts embed noisily and drowned out the
// clean USER.md/SOUL.md bullet facts in top-K results (e.g. a "지서버 hardware spec" query
// surfaced unrelated weather/incident-report chunks instead of the actual hardware fact).
// Chunking raw dialogue well enough to be useful (extraction/summarization, not naive
// paragraph splitting) is a separate, harder problem — out of scope for this pass.
console.log(`[${label}] added=${added} skipped(existing)=${skipped} failed=${failed}`);
}
async function main() {
for (const { workspace, label } of WORKSPACES) {
await migrateWorkspace(workspace, label);
}
}
main().then(() => process.exit(0)).catch(err => { console.error(err); process.exit(1); });