v4.3.0: Chroma 벡터 메모리 시스템 구축 + 하드코딩 게이트 정리
- USER.md/SOUL.md 문자캡 잘림 보완용 벡터 검색(Chroma, EmbeddingGemma) 신설 - 일별 대화로그 LLM 추출 파이프라인(user_facts/daily_extracts 컬렉션 분리) - 검증-강제 게이트(isFactualInfoRequest 등)를 prompt-gates.ts로 분리, 개인 인프라 별명(지서버/클로서버) 관련 불필요한 web_search 강제 호출 버그 수정 - Qwen3-Embedding-8B는 실측 결과 이미지생성 GPU와 자원 충돌 확인되어 폐기, EmbeddingGemma(경량+다국어)로 최종 선택 Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,147 @@
|
||||
// One-off backfill: index existing USER.md/SOUL.md bullet facts and daily memory files into
|
||||
// Chroma. Run once after wiring up memory-vector.ts. Safe to re-run (uses deterministic IDs,
|
||||
// Chroma's add() upserts on matching ID... actually add() errors on duplicate ID in this
|
||||
// version, so this script skips ids already present).
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import { addMemoryVector } from '../src/gateway/memory-vector';
|
||||
|
||||
const WORKSPACES: { workspace: string; label: string }[] = [
|
||||
{ workspace: '/srv/homeclaw/.smallclaw/users/papa/workspace', label: 'papa' },
|
||||
{ workspace: '/srv/homeclaw/.smallclaw/users/jasmine/workspace', label: 'jasmine' },
|
||||
{ workspace: '/srv/homeclaw/.smallclaw/users/cherry/workspace', label: 'cherry' },
|
||||
{ workspace: '/srv/homeclaw/.smallclaw/workspace', label: 'global' },
|
||||
];
|
||||
|
||||
// nomic-embed-text's context window is 2048 tokens — long daily-memory files (some 30KB+)
|
||||
// blow past that and the embed call 500s. Chunk on paragraph boundaries, ~3000 chars/chunk
|
||||
// (conservative for mixed Korean/English token density) rather than embedding the whole file.
|
||||
function chunkText(text: string, maxChars = 1500): string[] {
|
||||
const paragraphs = text.split(/\n\n+/);
|
||||
const chunks: string[] = [];
|
||||
let current = '';
|
||||
for (const para of paragraphs) {
|
||||
if (current && (current.length + para.length + 2) > maxChars) {
|
||||
chunks.push(current);
|
||||
current = para;
|
||||
} else {
|
||||
current = current ? `${current}\n\n${para}` : para;
|
||||
}
|
||||
// A single paragraph longer than maxChars on its own — hard-split it.
|
||||
while (current.length > maxChars) {
|
||||
chunks.push(current.slice(0, maxChars));
|
||||
current = current.slice(maxChars);
|
||||
}
|
||||
}
|
||||
if (current) chunks.push(current);
|
||||
return chunks;
|
||||
}
|
||||
|
||||
// memory_write always appends a trailing [YYYY-MM-DD] to what it writes (see the memory_write
|
||||
// handler in server-v2.ts) — so a bullet WITHOUT one was never written atomically via that path;
|
||||
// it's hand-authored nested content (a "parent:" line introducing a block, often with [bracket]
|
||||
// sub-headers underneath, e.g. "클로서버 AI 미디어 파이프라인 구성:" > "[음성 STT/TTS]" > "- TTS 메인: ...").
|
||||
// Splitting those sub-bullets into standalone vectors loses which parent they belong to — verified
|
||||
// 2026-07-24: a recalled "TTS 메인: OmniVoice..." fact (really about 클로서버) got misattributed to
|
||||
// 지서버 by the model once it lost that context. Fix: group undated bullets with their nearest
|
||||
// colon-ending parent (+ bracket sub-header, if any) into one combined fact instead of splitting.
|
||||
function parseBullets(fileContent: string, fileLabel: 'user' | 'soul'): { category: string; text: string; date: string }[] {
|
||||
const facts: { category: string; text: string; date: string }[] = [];
|
||||
let currentCategory = 'general';
|
||||
let parentLine: string | null = null;
|
||||
let bracketHeader: string | null = null;
|
||||
let buffer: string[] = [];
|
||||
|
||||
const flush = () => {
|
||||
if (buffer.length === 0) return;
|
||||
const prefixParts = [parentLine, bracketHeader].filter(Boolean);
|
||||
const text = (prefixParts.length ? `${prefixParts.join(' — ')}: ` : '') + buffer.join(' / ');
|
||||
facts.push({ category: currentCategory, text, date: '' });
|
||||
buffer = [];
|
||||
};
|
||||
|
||||
for (const line of fileContent.split('\n')) {
|
||||
const headerMatch = line.match(/^##\s+(.+)$/);
|
||||
if (headerMatch) {
|
||||
flush();
|
||||
parentLine = null;
|
||||
bracketHeader = null;
|
||||
currentCategory = headerMatch[1].trim();
|
||||
continue;
|
||||
}
|
||||
const bracketMatch = line.match(/^\[(.+)\]$/);
|
||||
if (bracketMatch) {
|
||||
flush();
|
||||
bracketHeader = bracketMatch[1].trim();
|
||||
continue;
|
||||
}
|
||||
const bulletMatch = line.match(/^-\s+(.+)$/);
|
||||
if (!bulletMatch) continue;
|
||||
const text = bulletMatch[1].trim();
|
||||
const dateMatch = text.match(/\[(\d{4}-\d{2}-\d{2})\]/);
|
||||
if (dateMatch) {
|
||||
// Standalone dated fact — flush any pending nested block first, then emit this as-is.
|
||||
flush();
|
||||
parentLine = null;
|
||||
bracketHeader = null;
|
||||
facts.push({ category: currentCategory, text, date: dateMatch[1] });
|
||||
} else if (/:$/.test(text)) {
|
||||
// Introduces a new nested block.
|
||||
flush();
|
||||
parentLine = text.replace(/:$/, '');
|
||||
bracketHeader = null;
|
||||
} else if (parentLine) {
|
||||
// Sub-item of the current nested block.
|
||||
buffer.push(text);
|
||||
} else {
|
||||
// Undated, no parent context (e.g. a plain continuation bullet) — keep as its own fact
|
||||
// rather than silently dropping it.
|
||||
facts.push({ category: currentCategory, text, date: '' });
|
||||
}
|
||||
}
|
||||
flush();
|
||||
return facts;
|
||||
}
|
||||
|
||||
async function migrateWorkspace(workspace: string, label: string) {
|
||||
let added = 0, skipped = 0, failed = 0;
|
||||
|
||||
for (const file of ['USER.md', 'SOUL.md'] as const) {
|
||||
const p = path.join(workspace, 'prompts', file);
|
||||
if (!fs.existsSync(p)) continue;
|
||||
const content = fs.readFileSync(p, 'utf-8');
|
||||
const facts = parseBullets(content, file === 'USER.md' ? 'user' : 'soul');
|
||||
for (let i = 0; i < facts.length; i++) {
|
||||
const f = facts[i];
|
||||
const id = `${workspace}:${file}:${f.category}:migrated:${i}`;
|
||||
try {
|
||||
await addMemoryVector({
|
||||
id,
|
||||
text: f.text,
|
||||
metadata: { workspace, file: file === 'USER.md' ? 'user' : 'soul', category: f.category, date: f.date || 'unknown', source: 'migration' },
|
||||
});
|
||||
added++;
|
||||
} catch (err: any) {
|
||||
if (String(err.message || '').includes('already exists') || String(err.message || '').includes('Insert of existing embedding ID')) skipped++;
|
||||
else { failed++; console.warn(` [${label}] failed ${id}: ${err.message}`); }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Daily memory files (raw conversation transcripts) are deliberately NOT migrated.
|
||||
// 2026-07-24 test: naive paragraph-chunked transcripts embed noisily and drowned out the
|
||||
// clean USER.md/SOUL.md bullet facts in top-K results (e.g. a "지서버 hardware spec" query
|
||||
// surfaced unrelated weather/incident-report chunks instead of the actual hardware fact).
|
||||
// Chunking raw dialogue well enough to be useful (extraction/summarization, not naive
|
||||
// paragraph splitting) is a separate, harder problem — out of scope for this pass.
|
||||
|
||||
console.log(`[${label}] added=${added} skipped(existing)=${skipped} failed=${failed}`);
|
||||
}
|
||||
|
||||
async function main() {
|
||||
for (const { workspace, label } of WORKSPACES) {
|
||||
await migrateWorkspace(workspace, label);
|
||||
}
|
||||
}
|
||||
|
||||
main().then(() => process.exit(0)).catch(err => { console.error(err); process.exit(1); });
|
||||
Reference in New Issue
Block a user