// One-off backfill: index existing USER.md/SOUL.md bullet facts and daily memory files into // Chroma. Run once after wiring up memory-vector.ts. Safe to re-run (uses deterministic IDs, // Chroma's add() upserts on matching ID... actually add() errors on duplicate ID in this // version, so this script skips ids already present). import fs from 'fs'; import path from 'path'; import { addMemoryVector } from '../src/gateway/memory-vector'; const WORKSPACES: { workspace: string; label: string }[] = [ { workspace: '/srv/homeclaw/.smallclaw/users/papa/workspace', label: 'papa' }, { workspace: '/srv/homeclaw/.smallclaw/users/jasmine/workspace', label: 'jasmine' }, { workspace: '/srv/homeclaw/.smallclaw/users/cherry/workspace', label: 'cherry' }, { workspace: '/srv/homeclaw/.smallclaw/workspace', label: 'global' }, ]; // nomic-embed-text's context window is 2048 tokens — long daily-memory files (some 30KB+) // blow past that and the embed call 500s. Chunk on paragraph boundaries, ~3000 chars/chunk // (conservative for mixed Korean/English token density) rather than embedding the whole file. function chunkText(text: string, maxChars = 1500): string[] { const paragraphs = text.split(/\n\n+/); const chunks: string[] = []; let current = ''; for (const para of paragraphs) { if (current && (current.length + para.length + 2) > maxChars) { chunks.push(current); current = para; } else { current = current ? `${current}\n\n${para}` : para; } // A single paragraph longer than maxChars on its own — hard-split it. while (current.length > maxChars) { chunks.push(current.slice(0, maxChars)); current = current.slice(maxChars); } } if (current) chunks.push(current); return chunks; } // memory_write always appends a trailing [YYYY-MM-DD] to what it writes (see the memory_write // handler in server-v2.ts) — so a bullet WITHOUT one was never written atomically via that path; // it's hand-authored nested content (a "parent:" line introducing a block, often with [bracket] // sub-headers underneath, e.g. "클로서버 AI 미디어 파이프라인 구성:" > "[음성 STT/TTS]" > "- TTS 메인: ..."). // Splitting those sub-bullets into standalone vectors loses which parent they belong to — verified // 2026-07-24: a recalled "TTS 메인: OmniVoice..." fact (really about 클로서버) got misattributed to // 지서버 by the model once it lost that context. Fix: group undated bullets with their nearest // colon-ending parent (+ bracket sub-header, if any) into one combined fact instead of splitting. function parseBullets(fileContent: string, fileLabel: 'user' | 'soul'): { category: string; text: string; date: string }[] { const facts: { category: string; text: string; date: string }[] = []; let currentCategory = 'general'; let parentLine: string | null = null; let bracketHeader: string | null = null; let buffer: string[] = []; const flush = () => { if (buffer.length === 0) return; const prefixParts = [parentLine, bracketHeader].filter(Boolean); const text = (prefixParts.length ? `${prefixParts.join(' — ')}: ` : '') + buffer.join(' / '); facts.push({ category: currentCategory, text, date: '' }); buffer = []; }; for (const line of fileContent.split('\n')) { const headerMatch = line.match(/^##\s+(.+)$/); if (headerMatch) { flush(); parentLine = null; bracketHeader = null; currentCategory = headerMatch[1].trim(); continue; } const bracketMatch = line.match(/^\[(.+)\]$/); if (bracketMatch) { flush(); bracketHeader = bracketMatch[1].trim(); continue; } const bulletMatch = line.match(/^-\s+(.+)$/); if (!bulletMatch) continue; const text = bulletMatch[1].trim(); const dateMatch = text.match(/\[(\d{4}-\d{2}-\d{2})\]/); if (dateMatch) { // Standalone dated fact — flush any pending nested block first, then emit this as-is. flush(); parentLine = null; bracketHeader = null; facts.push({ category: currentCategory, text, date: dateMatch[1] }); } else if (/:$/.test(text)) { // Introduces a new nested block. flush(); parentLine = text.replace(/:$/, ''); bracketHeader = null; } else if (parentLine) { // Sub-item of the current nested block. buffer.push(text); } else { // Undated, no parent context (e.g. a plain continuation bullet) — keep as its own fact // rather than silently dropping it. facts.push({ category: currentCategory, text, date: '' }); } } flush(); return facts; } async function migrateWorkspace(workspace: string, label: string) { let added = 0, skipped = 0, failed = 0; for (const file of ['USER.md', 'SOUL.md'] as const) { const p = path.join(workspace, 'prompts', file); if (!fs.existsSync(p)) continue; const content = fs.readFileSync(p, 'utf-8'); const facts = parseBullets(content, file === 'USER.md' ? 'user' : 'soul'); for (let i = 0; i < facts.length; i++) { const f = facts[i]; const id = `${workspace}:${file}:${f.category}:migrated:${i}`; try { await addMemoryVector({ id, text: f.text, metadata: { workspace, file: file === 'USER.md' ? 'user' : 'soul', category: f.category, date: f.date || 'unknown', source: 'migration' }, }); added++; } catch (err: any) { if (String(err.message || '').includes('already exists') || String(err.message || '').includes('Insert of existing embedding ID')) skipped++; else { failed++; console.warn(` [${label}] failed ${id}: ${err.message}`); } } } } // Daily memory files (raw conversation transcripts) are deliberately NOT migrated. // 2026-07-24 test: naive paragraph-chunked transcripts embed noisily and drowned out the // clean USER.md/SOUL.md bullet facts in top-K results (e.g. a "지서버 hardware spec" query // surfaced unrelated weather/incident-report chunks instead of the actual hardware fact). // Chunking raw dialogue well enough to be useful (extraction/summarization, not naive // paragraph splitting) is a separate, harder problem — out of scope for this pass. console.log(`[${label}] added=${added} skipped(existing)=${skipped} failed=${failed}`); } async function main() { for (const { workspace, label } of WORKSPACES) { await migrateWorkspace(workspace, label); } } main().then(() => process.exit(0)).catch(err => { console.error(err); process.exit(1); });