// One-time backfill: mirror existing Chroma vector-memory records into the new FTS5 // keyword-search sidecar (memory-fts.ts, added 2026-08-06 hybrid-search upgrade). Only // needed once — addVector() keeps the two in sync for everything written after this ran. import { ChromaClient } from 'chromadb'; import { USER_FACTS_COLLECTION, DAILY_EXTRACTS_COLLECTION } from '../src/gateway/memory/memory-vector'; import { upsertFtsRecord } from '../src/gateway/memory/memory-fts'; async function backfillCollection(client: ChromaClient, name: string) { const col = await client.getOrCreateCollection({ name, embeddingFunction: null }); const count = await col.count(); console.log(`[${name}] ${count} records`); const batchSize = 200; let offset = 0; let written = 0; while (offset < count) { const res = await col.get({ limit: batchSize, offset }); const ids = res.ids || []; const docs = res.documents || []; const metas = res.metadatas || []; for (let i = 0; i < ids.length; i++) { const text = docs[i] || ''; if (!text) continue; const workspace = String((metas[i] as any)?.workspace || ''); upsertFtsRecord(name, ids[i], text, workspace); written++; } offset += batchSize; } console.log(`[${name}] backfilled ${written} records into FTS`); } async function main() { const client = new ChromaClient({ host: 'localhost', port: 8100 }); await backfillCollection(client, USER_FACTS_COLLECTION); await backfillCollection(client, DAILY_EXTRACTS_COLLECTION); console.log('done'); } main().catch(e => { console.error(e); process.exit(1); });