- USER.md/SOUL.md 문자캡 잘림 보완용 벡터 검색(Chroma, EmbeddingGemma) 신설 - 일별 대화로그 LLM 추출 파이프라인(user_facts/daily_extracts 컬렉션 분리) - 검증-강제 게이트(isFactualInfoRequest 등)를 prompt-gates.ts로 분리, 개인 인프라 별명(지서버/클로서버) 관련 불필요한 web_search 강제 호출 버그 수정 - Qwen3-Embedding-8B는 실측 결과 이미지생성 GPU와 자원 충돌 확인되어 폐기, EmbeddingGemma(경량+다국어)로 최종 선택 Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
148 lines
6.4 KiB
TypeScript
148 lines
6.4 KiB
TypeScript
// One-off backfill: index existing USER.md/SOUL.md bullet facts and daily memory files into
|
|
// Chroma. Run once after wiring up memory-vector.ts. Safe to re-run (uses deterministic IDs,
|
|
// Chroma's add() upserts on matching ID... actually add() errors on duplicate ID in this
|
|
// version, so this script skips ids already present).
|
|
import fs from 'fs';
|
|
import path from 'path';
|
|
import { addMemoryVector } from '../src/gateway/memory-vector';
|
|
|
|
const WORKSPACES: { workspace: string; label: string }[] = [
|
|
{ workspace: '/srv/homeclaw/.smallclaw/users/papa/workspace', label: 'papa' },
|
|
{ workspace: '/srv/homeclaw/.smallclaw/users/jasmine/workspace', label: 'jasmine' },
|
|
{ workspace: '/srv/homeclaw/.smallclaw/users/cherry/workspace', label: 'cherry' },
|
|
{ workspace: '/srv/homeclaw/.smallclaw/workspace', label: 'global' },
|
|
];
|
|
|
|
// nomic-embed-text's context window is 2048 tokens — long daily-memory files (some 30KB+)
|
|
// blow past that and the embed call 500s. Chunk on paragraph boundaries, ~3000 chars/chunk
|
|
// (conservative for mixed Korean/English token density) rather than embedding the whole file.
|
|
function chunkText(text: string, maxChars = 1500): string[] {
|
|
const paragraphs = text.split(/\n\n+/);
|
|
const chunks: string[] = [];
|
|
let current = '';
|
|
for (const para of paragraphs) {
|
|
if (current && (current.length + para.length + 2) > maxChars) {
|
|
chunks.push(current);
|
|
current = para;
|
|
} else {
|
|
current = current ? `${current}\n\n${para}` : para;
|
|
}
|
|
// A single paragraph longer than maxChars on its own — hard-split it.
|
|
while (current.length > maxChars) {
|
|
chunks.push(current.slice(0, maxChars));
|
|
current = current.slice(maxChars);
|
|
}
|
|
}
|
|
if (current) chunks.push(current);
|
|
return chunks;
|
|
}
|
|
|
|
// memory_write always appends a trailing [YYYY-MM-DD] to what it writes (see the memory_write
|
|
// handler in server-v2.ts) — so a bullet WITHOUT one was never written atomically via that path;
|
|
// it's hand-authored nested content (a "parent:" line introducing a block, often with [bracket]
|
|
// sub-headers underneath, e.g. "클로서버 AI 미디어 파이프라인 구성:" > "[음성 STT/TTS]" > "- TTS 메인: ...").
|
|
// Splitting those sub-bullets into standalone vectors loses which parent they belong to — verified
|
|
// 2026-07-24: a recalled "TTS 메인: OmniVoice..." fact (really about 클로서버) got misattributed to
|
|
// 지서버 by the model once it lost that context. Fix: group undated bullets with their nearest
|
|
// colon-ending parent (+ bracket sub-header, if any) into one combined fact instead of splitting.
|
|
function parseBullets(fileContent: string, fileLabel: 'user' | 'soul'): { category: string; text: string; date: string }[] {
|
|
const facts: { category: string; text: string; date: string }[] = [];
|
|
let currentCategory = 'general';
|
|
let parentLine: string | null = null;
|
|
let bracketHeader: string | null = null;
|
|
let buffer: string[] = [];
|
|
|
|
const flush = () => {
|
|
if (buffer.length === 0) return;
|
|
const prefixParts = [parentLine, bracketHeader].filter(Boolean);
|
|
const text = (prefixParts.length ? `${prefixParts.join(' — ')}: ` : '') + buffer.join(' / ');
|
|
facts.push({ category: currentCategory, text, date: '' });
|
|
buffer = [];
|
|
};
|
|
|
|
for (const line of fileContent.split('\n')) {
|
|
const headerMatch = line.match(/^##\s+(.+)$/);
|
|
if (headerMatch) {
|
|
flush();
|
|
parentLine = null;
|
|
bracketHeader = null;
|
|
currentCategory = headerMatch[1].trim();
|
|
continue;
|
|
}
|
|
const bracketMatch = line.match(/^\[(.+)\]$/);
|
|
if (bracketMatch) {
|
|
flush();
|
|
bracketHeader = bracketMatch[1].trim();
|
|
continue;
|
|
}
|
|
const bulletMatch = line.match(/^-\s+(.+)$/);
|
|
if (!bulletMatch) continue;
|
|
const text = bulletMatch[1].trim();
|
|
const dateMatch = text.match(/\[(\d{4}-\d{2}-\d{2})\]/);
|
|
if (dateMatch) {
|
|
// Standalone dated fact — flush any pending nested block first, then emit this as-is.
|
|
flush();
|
|
parentLine = null;
|
|
bracketHeader = null;
|
|
facts.push({ category: currentCategory, text, date: dateMatch[1] });
|
|
} else if (/:$/.test(text)) {
|
|
// Introduces a new nested block.
|
|
flush();
|
|
parentLine = text.replace(/:$/, '');
|
|
bracketHeader = null;
|
|
} else if (parentLine) {
|
|
// Sub-item of the current nested block.
|
|
buffer.push(text);
|
|
} else {
|
|
// Undated, no parent context (e.g. a plain continuation bullet) — keep as its own fact
|
|
// rather than silently dropping it.
|
|
facts.push({ category: currentCategory, text, date: '' });
|
|
}
|
|
}
|
|
flush();
|
|
return facts;
|
|
}
|
|
|
|
async function migrateWorkspace(workspace: string, label: string) {
|
|
let added = 0, skipped = 0, failed = 0;
|
|
|
|
for (const file of ['USER.md', 'SOUL.md'] as const) {
|
|
const p = path.join(workspace, 'prompts', file);
|
|
if (!fs.existsSync(p)) continue;
|
|
const content = fs.readFileSync(p, 'utf-8');
|
|
const facts = parseBullets(content, file === 'USER.md' ? 'user' : 'soul');
|
|
for (let i = 0; i < facts.length; i++) {
|
|
const f = facts[i];
|
|
const id = `${workspace}:${file}:${f.category}:migrated:${i}`;
|
|
try {
|
|
await addMemoryVector({
|
|
id,
|
|
text: f.text,
|
|
metadata: { workspace, file: file === 'USER.md' ? 'user' : 'soul', category: f.category, date: f.date || 'unknown', source: 'migration' },
|
|
});
|
|
added++;
|
|
} catch (err: any) {
|
|
if (String(err.message || '').includes('already exists') || String(err.message || '').includes('Insert of existing embedding ID')) skipped++;
|
|
else { failed++; console.warn(` [${label}] failed ${id}: ${err.message}`); }
|
|
}
|
|
}
|
|
}
|
|
|
|
// Daily memory files (raw conversation transcripts) are deliberately NOT migrated.
|
|
// 2026-07-24 test: naive paragraph-chunked transcripts embed noisily and drowned out the
|
|
// clean USER.md/SOUL.md bullet facts in top-K results (e.g. a "지서버 hardware spec" query
|
|
// surfaced unrelated weather/incident-report chunks instead of the actual hardware fact).
|
|
// Chunking raw dialogue well enough to be useful (extraction/summarization, not naive
|
|
// paragraph splitting) is a separate, harder problem — out of scope for this pass.
|
|
|
|
console.log(`[${label}] added=${added} skipped(existing)=${skipped} failed=${failed}`);
|
|
}
|
|
|
|
async function main() {
|
|
for (const { workspace, label } of WORKSPACES) {
|
|
await migrateWorkspace(workspace, label);
|
|
}
|
|
}
|
|
|
|
main().then(() => process.exit(0)).catch(err => { console.error(err); process.exit(1); });
|