From 70c1e1c23fd884802528e6b784b4f9902ec4fc99 Mon Sep 17 00:00:00 2001 From: kim Date: Thu, 6 Aug 2026 15:12:50 +0900 Subject: [PATCH] =?UTF-8?q?RAG=20=EA=B8=B0=EC=96=B5=20=EA=B2=80=EC=83=89?= =?UTF-8?q?=EC=97=90=20=ED=95=98=EC=9D=B4=EB=B8=8C=EB=A6=AC=EB=93=9C=20?= =?UTF-8?q?=EA=B2=80=EC=83=89(=EB=B2=A1=ED=84=B0+=ED=82=A4=EC=9B=8C?= =?UTF-8?q?=EB=93=9C)=20+=20=EC=9E=AC=EB=9E=AD=ED=82=B9=20=EC=B6=94?= =?UTF-8?q?=EA=B0=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Chroma 벡터검색만으로는 IP·모델명 같은 정확한 용어를 놓치는 경우가 있어 SQLite FTS5(trigram) 키워드검색을 병합하고, 애매한 경우(예: 클로서버 vs 지서버 GPU 스펙 혼동)만 LLM 재랭킹으로 오답을 걸러내도록 함. 재랭킹은 후보가 없거나 확실한 단일매치일 때는 건너뛰어 대부분의 대화에서는 지연시간 증가가 거의 없음. 기존 570개 기록은 scripts/backfill-fts.ts로 백필. Co-Authored-By: Claude Sonnet 5 --- .gitignore | 4 ++ scripts/backfill-fts.ts | 39 +++++++++++ src/gateway/chat/personality-context.ts | 86 ++++++++++++++++++++++--- src/gateway/memory/memory-fts.ts | 78 ++++++++++++++++++++++ src/gateway/memory/memory-rerank.ts | 65 +++++++++++++++++++ src/gateway/memory/memory-vector.ts | 11 ++++ 6 files changed, 273 insertions(+), 10 deletions(-) create mode 100644 scripts/backfill-fts.ts create mode 100644 src/gateway/memory/memory-fts.ts create mode 100644 src/gateway/memory/memory-rerank.ts diff --git a/.gitignore b/.gitignore index 3ef8639..2a99b57 100644 --- a/.gitignore +++ b/.gitignore @@ -21,6 +21,10 @@ .smallclaw/databases/*.db.bak.* .smallclaw/databases/nohup.out +# Derived FTS mirror of Chroma vector-memory data (rag-hybrid-search upgrade) — fully +# regenerable via scripts/backfill-fts.ts, not source-of-truth data. +.smallclaw/databases/rag_fts.db + # --- PYTHON CACHE --- scripts/__pycache__/ **/__pycache__/ diff --git a/scripts/backfill-fts.ts b/scripts/backfill-fts.ts new file mode 100644 index 0000000..ae967c9 --- /dev/null +++ b/scripts/backfill-fts.ts @@ -0,0 +1,39 @@ +// One-time backfill: mirror existing Chroma vector-memory records into the new FTS5 +// keyword-search sidecar (memory-fts.ts, added 2026-08-06 hybrid-search upgrade). Only +// needed once — addVector() keeps the two in sync for everything written after this ran. +import { ChromaClient } from 'chromadb'; +import { USER_FACTS_COLLECTION, DAILY_EXTRACTS_COLLECTION } from '../src/gateway/memory/memory-vector'; +import { upsertFtsRecord } from '../src/gateway/memory/memory-fts'; + +async function backfillCollection(client: ChromaClient, name: string) { + const col = await client.getOrCreateCollection({ name, embeddingFunction: null }); + const count = await col.count(); + console.log(`[${name}] ${count} records`); + const batchSize = 200; + let offset = 0; + let written = 0; + while (offset < count) { + const res = await col.get({ limit: batchSize, offset }); + const ids = res.ids || []; + const docs = res.documents || []; + const metas = res.metadatas || []; + for (let i = 0; i < ids.length; i++) { + const text = docs[i] || ''; + if (!text) continue; + const workspace = String((metas[i] as any)?.workspace || ''); + upsertFtsRecord(name, ids[i], text, workspace); + written++; + } + offset += batchSize; + } + console.log(`[${name}] backfilled ${written} records into FTS`); +} + +async function main() { + const client = new ChromaClient({ host: 'localhost', port: 8100 }); + await backfillCollection(client, USER_FACTS_COLLECTION); + await backfillCollection(client, DAILY_EXTRACTS_COLLECTION); + console.log('done'); +} + +main().catch(e => { console.error(e); process.exit(1); }); diff --git a/src/gateway/chat/personality-context.ts b/src/gateway/chat/personality-context.ts index ddc5d65..186ddb5 100644 --- a/src/gateway/chat/personality-context.ts +++ b/src/gateway/chat/personality-context.ts @@ -8,6 +8,8 @@ import { USER_FACTS_COLLECTION, DAILY_EXTRACTS_COLLECTION, } from '../memory/memory-vector'; +import { keywordSearch } from '../memory/memory-fts'; +import { rerankCandidates } from '../memory/memory-rerank'; // resolvePromptPath stays in server.ts (used well beyond this cluster — e.g. // routes-agents.ts also takes it as a dep), so it's threaded in. @@ -401,30 +403,94 @@ return async function buildPersonalityContext( // Vector-store recall: relevance-based, not the fixed-recency/fixed-char-cap USER.md/SOUL.md // path — this is what lets facts survive past loadFile's truncation cutoff (see loadFile // above). Scoped to this workspace via the `where` filter so users never see each other's - // facts. Best-effort — a Chroma/embedding outage must not break prompt building. + // facts. Best-effort — a Chroma/embedding/rerank outage must not break prompt building. // // Two collections, one query embedding (2026-07-24 design — see project_vector_memory_chroma - // memory): user_facts (memory_write-sourced, clean) gets a loose threshold; daily_extracts - // (LLM-extracted from raw conversation logs, noisier) gets a stricter one. Both merged and - // ranked together by distance so the model just sees one flat, relevance-sorted list. + // memory): user_facts (memory_write-sourced, clean), daily_extracts (LLM-extracted from raw + // conversation logs, noisier). + // + // Hybrid search + rerank (2026-08-06 upgrade — see this session's rag-upgrade memory): + // dense vector search alone blurs exact literal terms (IPs, model names, hostnames), and a + // pure distance cutoff can't tell "superficially similar" apart from "actually relevant" — + // so this widens the vector candidate pool and merges in FTS5 keyword hits (memory-fts.ts) + // for literal-term recall, then reranks the merged pool with a cheap LLM call (memory- + // rerank.ts) instead of a bare threshold. + // + // Reranking is gated, not unconditional: a first pass measured it adding ~5-7s to *every* + // turn (proportional to the ~25-candidate pool needed for reranking to actually have enough + // to disambiguate — shrinking the pool for speed silently defeated the point, see session + // notes), which is too much tax for turns where there's nothing ambiguous to resolve. Most + // turns fall into one of two cheap-to-detect cases: (a) no candidates at all — ordinary + // chit-chat with nothing memory-relevant — or (b) one clearly dominant vector match with no + // competing signal, where a distance gap alone is enough confidence. Only genuinely + // contested cases (multiple plausible candidates, e.g. 클로서버 vs 지서버 GPU facts) pay for + // the rerank call. Both hybrid search and reranking remain best-effort layers on top of a + // retrieval path that must keep working without them. let vectorMemory = ''; if (messageText.trim().length >= 4) { try { const queryEmbedding = await embedQuery(messageText); const [userFactHits, dailyExtractHits] = await Promise.all([ - queryVectorsWithEmbedding(USER_FACTS_COLLECTION, queryEmbedding, 6, { workspace: workspacePath }), - queryVectorsWithEmbedding(DAILY_EXTRACTS_COLLECTION, queryEmbedding, 6, { workspace: workspacePath }), + queryVectorsWithEmbedding(USER_FACTS_COLLECTION, queryEmbedding, 15, { workspace: workspacePath }), + queryVectorsWithEmbedding(DAILY_EXTRACTS_COLLECTION, queryEmbedding, 15, { workspace: workspacePath }), ]); + const keywordHits = [ + ...keywordSearch(USER_FACTS_COLLECTION, messageText, 15, workspacePath), + ...keywordSearch(DAILY_EXTRACTS_COLLECTION, messageText, 15, workspacePath), + ]; + // Thresholds recalibrated for EmbeddingGemma's wider distance spread (2026-07-24 switch // from nomic-embed-text) — a 3-fact spot check gave ~0.51 for a genuinely relevant match, // ~0.66 for same-topic-wrong-entity, ~0.97 for unrelated. Rough starting points, not a - // rigorous calibration — revisit if recall feels off/noisy in real use. - const relevant = [ + // rigorous calibration — revisit if recall feels off/noisy in real use. Used both as the + // confidence signal for the skip-rerank fast path and as the fallback if reranking fails. + const allVectorHits = [...userFactHits, ...dailyExtractHits].sort((a, b) => a.distance - b.distance); + const byDistanceThreshold = [ ...userFactHits.filter(h => h.distance < 0.65), ...dailyExtractHits.filter(h => h.distance < 0.55 && !h.metadata?.empty), ].sort((a, b) => a.distance - b.distance).slice(0, 8); - if (relevant.length > 0) { - vectorMemory = relevant.map(h => `- ${h.text}`).join('\n'); + + // Merge vector + keyword candidates, deduped by id (vector hits win the text on overlap — + // same source of truth either way). Capped so the rerank prompt (when it does run) stays + // small — this is meant to be a fast classification call, not another full turn. + const merged = new Map(); + for (const h of allVectorHits) if (h.id) merged.set(h.id, h.text); + let hasKeywordOnlyHit = false; + for (const h of keywordHits) { + if (!h.id) continue; + if (!merged.has(h.id)) { hasKeywordOnlyHit = true; merged.set(h.id, h.text); } + } + const candidates = Array.from(merged, ([id, text]) => ({ id, text })).slice(0, 25); + + // Chroma's query() always returns up to topK neighbors regardless of how far they + // are — "candidates.length" is nearly always the full 25 cap and useless as a gate on + // its own. Distance quality is the real signal: nothing crossing the existing + // thresholds (and no keyword hits either) means nothing is actually relevant, matching + // what the pre-rerank baseline would have shown anyway. + const top = allVectorHits[0]; + const runnerUp = allVectorHits[1]; + const nothingRelevant = byDistanceThreshold.length === 0 && keywordHits.length === 0; + const confidentSingleMatch = !hasKeywordOnlyHit && !!top && top.distance < 0.35 + && (!runnerUp || runnerUp.distance - top.distance > 0.2); + + if (nothingRelevant) { + // Nothing at all — ordinary conversational turn, nothing to recall. + } else if (confidentSingleMatch) { + // Too few candidates to be worth disambiguating, or one clear winner with no + // competing signal — skip the rerank round-trip and trust retrieval directly. + const fallback = byDistanceThreshold.length > 0 + ? byDistanceThreshold + : candidates.map(c => ({ text: c.text } as { text: string })); + vectorMemory = fallback.slice(0, 8).map(h => `- ${h.text}`).join('\n'); + } else { + const rerankedIds = await rerankCandidates(messageText, candidates, 8); + if (rerankedIds === null) { + // Rerank call failed outright — fall back to plain distance-threshold behavior. + if (byDistanceThreshold.length > 0) vectorMemory = byDistanceThreshold.map(h => `- ${h.text}`).join('\n'); + } else if (rerankedIds.length > 0) { + const byId = new Map(candidates.map(c => [c.id, c.text])); + vectorMemory = rerankedIds.map(id => `- ${byId.get(id)}`).join('\n'); + } } } catch (err: any) { console.warn('[buildPersonalityContext] vector memory query failed (non-fatal):', err.message); diff --git a/src/gateway/memory/memory-fts.ts b/src/gateway/memory/memory-fts.ts new file mode 100644 index 0000000..1ce03d3 --- /dev/null +++ b/src/gateway/memory/memory-fts.ts @@ -0,0 +1,78 @@ +// Keyword-search sidecar for the RAG memory system (2026-08-06 hybrid-search upgrade). +// Chroma only does dense vector search, which blurs exact literal terms — IPs, hostnames, +// model names, port numbers — that show up constantly in this memory's content (see +// project_wol_gate / project_jiserver_model_benchmarks style facts). This FTS5 mirror +// catches those exact-substring matches and gets merged with vector results before +// reranking (see memory-rerank.ts) in personality-context.ts. +// +// Uses FTS5's trigram tokenizer (character n-grams) instead of the default unicode61 +// (whitespace word-segmentation) tokenizer — Korean has no whitespace word boundaries the +// way English does, so a word-tokenizer badly under-matches Korean text. Trigram works +// identically regardless of script/language, at the cost of being pure substring matching +// (no stemming) — acceptable here since the whole point is literal-term recall. +import Database from 'better-sqlite3'; +import path from 'path'; +import fs from 'fs'; +import { getConfig } from '../../config/config.js'; + +let dbSingleton: Database.Database | null = null; + +function getDbPath(): string { + const dir = path.join(getConfig().getConfigDir(), 'databases'); + fs.mkdirSync(dir, { recursive: true }); + return path.join(dir, 'rag_fts.db'); +} + +function getDb(): Database.Database { + if (dbSingleton) return dbSingleton; + const db = new Database(getDbPath()); + db.pragma('journal_mode = WAL'); + db.exec(` + CREATE VIRTUAL TABLE IF NOT EXISTS memory_fts USING fts5( + collection, doc_id, text, workspace, + tokenize='trigram' + ); + `); + dbSingleton = db; + return db; +} + +export function upsertFtsRecord(collection: string, docId: string, text: string, workspace: string): void { + const db = getDb(); + db.prepare(`DELETE FROM memory_fts WHERE collection = ? AND doc_id = ?`).run(collection, docId); + db.prepare(`INSERT INTO memory_fts (collection, doc_id, text, workspace) VALUES (?, ?, ?, ?)`).run(collection, docId, text, workspace || ''); +} + +export interface FtsHit { id: string; text: string; } + +// FTS5's query grammar treats quotes/parens/colons/leading-dash specially — quoting each +// extracted term as its own phrase neutralizes that (still trigram-matched inside the +// quotes) so a query containing e.g. "192.168.0.1" or "3060" doesn't throw a syntax error. +// OR-joined (not AND/phrase) because the goal is recall of *any* distinctive term in the +// query, not requiring the whole query to match verbatim. +function buildMatchQuery(query: string): string | null { + const terms = query + .split(/\s+/) + .map(t => t.trim()) + .filter(t => t.length >= 2) + .slice(0, 12) + .map(t => `"${t.replace(/"/g, '""')}"`); + if (terms.length === 0) return null; + return terms.join(' OR '); +} + +export function keywordSearch(collection: string, query: string, topK: number, workspace?: string): FtsHit[] { + const match = buildMatchQuery(query); + if (!match) return []; + try { + const db = getDb(); + const rows = workspace + ? db.prepare(`SELECT doc_id, text FROM memory_fts WHERE memory_fts MATCH ? AND collection = ? AND workspace = ? ORDER BY rank LIMIT ?`).all(match, collection, workspace, topK) + : db.prepare(`SELECT doc_id, text FROM memory_fts WHERE memory_fts MATCH ? AND collection = ? ORDER BY rank LIMIT ?`).all(match, collection, topK); + return (rows as any[]).map(r => ({ id: r.doc_id, text: r.text })); + } catch { + // Malformed MATCH query or FTS quirk — degrade to no keyword hits rather than + // breaking the whole retrieval path (same best-effort philosophy as the vector side). + return []; + } +} diff --git a/src/gateway/memory/memory-rerank.ts b/src/gateway/memory/memory-rerank.ts new file mode 100644 index 0000000..573d744 --- /dev/null +++ b/src/gateway/memory/memory-rerank.ts @@ -0,0 +1,65 @@ +// LLM-based reranking for the RAG memory system (2026-08-06 upgrade). Vector cosine-distance +// and FTS5 rank alone often surface candidates that are superficially similar/keyword-adjacent +// but not actually relevant to what was asked — a single extra classification-style LLM call +// over the merged candidate pool measurably improves precision, which is the standard "why +// bother reranking" argument in RAG literature and the reason this was the top recommendation +// over other options considered (see this session's rag-upgrade memory). +// +// Kept deliberately cheap: one call per turn (not one per candidate), a fast/cheap model, +// short timeout, and a null return on any failure so the caller can fall back to the plain +// distance-threshold behavior — this must never be a hard dependency for prompt building, +// same "best-effort" rule as the vector/FTS retrieval it sits on top of. +import { getConfig } from '../../config/config.js'; + +// "flash"-tier model chosen specifically for latency, not quality — this call is a small +// classification task (pick relevant indices from a short list), not a task worth spending +// primary-model-grade reasoning time on. Revisit if this model is ever unpulled/renamed. +const RERANK_MODEL = 'deepseek-v4-flash:cloud'; +// Only fires on genuinely ambiguous retrieval (see the gating in personality-context.ts) — +// most turns never reach this call at all, so a generous timeout here doesn't cost anything +// on the common path. 8s was measured to occasionally clip real (successful, just slow) +// cloud-model responses under normal latency variance. +const RERANK_TIMEOUT_MS = 15_000; + +export interface RerankCandidate { id: string; text: string; } + +function getOllamaEndpoint(): string { + return (getConfig().getConfig() as any).ollama?.endpoint || 'http://localhost:11434'; +} + +// Returns the subset of candidate ids that are actually relevant, ranked best-first, +// capped at topK. Returns null (not []) on failure so callers can distinguish "reranked, +// turned out nothing was relevant" from "reranking itself didn't happen." +export async function rerankCandidates(query: string, candidates: RerankCandidate[], topK: number): Promise { + if (candidates.length === 0) return []; + const listText = candidates.map((c, i) => `[${i}] ${c.text}`).join('\n'); + const prompt = `사용자 질문: "${query}" + +아래는 장기 기억에서 검색된 후보 목록입니다. 이 중 사용자 질문에 실제로 관련 있는 것만, 관련성 높은 순서로 번호만 골라줘 (최대 ${topK}개). +관련 있는 게 하나도 없으면 "없음"이라고만 답해. 설명 없이 번호만 쉼표로 구분해서 답해 (예: "2,0,5"). + +${listText}`; + + try { + const res = await fetch(`${getOllamaEndpoint()}/api/chat`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ model: RERANK_MODEL, messages: [{ role: 'user', content: prompt }], stream: false }), + signal: AbortSignal.timeout(RERANK_TIMEOUT_MS), + }); + if (!res.ok) throw new Error(`rerank HTTP ${res.status}`); + const data: any = await res.json(); + const raw = String(data?.message?.content || '').trim(); + if (!raw || raw === '없음') return []; + + const seen = new Set(); + const indices = raw + .split(',') + .map(s => parseInt(s.trim(), 10)) + .filter(n => Number.isInteger(n) && n >= 0 && n < candidates.length && !seen.has(n) && seen.add(n)); + return indices.slice(0, topK).map(i => candidates[i].id); + } catch (err: any) { + console.warn('[memory-rerank] rerank failed (falling back to distance-threshold):', err.message); + return null; + } +} diff --git a/src/gateway/memory/memory-vector.ts b/src/gateway/memory/memory-vector.ts index ae9fa9c..486c122 100644 --- a/src/gateway/memory/memory-vector.ts +++ b/src/gateway/memory/memory-vector.ts @@ -10,6 +10,7 @@ // callers should apply a stricter relevance threshold against this one. import { ChromaClient, type Collection } from 'chromadb'; import { getConfig } from '../../config/config.js'; +import { upsertFtsRecord } from './memory-fts.js'; const CHROMA_HOST = 'localhost'; const CHROMA_PORT = 8100; @@ -103,9 +104,17 @@ export async function addVector(collectionName: string, rec: MemoryVectorRecord) documents: [rec.text], metadatas: rec.metadata ? [rec.metadata] : undefined, }); + // Keyword-search mirror for hybrid retrieval (see memory-fts.ts) — kept best-effort: + // a write failing here must not fail the vector write it's piggybacking on. + try { + upsertFtsRecord(collectionName, rec.id, rec.text, String(rec.metadata?.workspace || '')); + } catch (err: any) { + console.warn('[memory-vector] FTS mirror write failed (non-fatal):', err.message); + } } export interface MemoryVectorHit { + id: string; text: string; distance: number; metadata?: Record; @@ -119,10 +128,12 @@ export async function queryVectorsWithEmbedding( ): Promise { const col = await getCollection(collectionName); const res = await col.query({ queryEmbeddings: [embedding], nResults: topK, where: toChromaWhere(where) }); + const ids = res.ids?.[0] || []; const docs = res.documents?.[0] || []; const dists = res.distances?.[0] || []; const metas = res.metadatas?.[0] || []; return docs.map((text, i) => ({ + id: ids[i] || '', text: text || '', distance: dists[i] ?? Infinity, metadata: metas[i] as Record | undefined,