RAG 기억 검색에 하이브리드 검색(벡터+키워드) + 재랭킹 추가

Chroma 벡터검색만으로는 IP·모델명 같은 정확한 용어를 놓치는 경우가 있어
SQLite FTS5(trigram) 키워드검색을 병합하고, 애매한 경우(예: 클로서버 vs
지서버 GPU 스펙 혼동)만 LLM 재랭킹으로 오답을 걸러내도록 함. 재랭킹은
후보가 없거나 확실한 단일매치일 때는 건너뛰어 대부분의 대화에서는 지연시간
증가가 거의 없음. 기존 570개 기록은 scripts/backfill-fts.ts로 백필.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-08-06 15:12:50 +09:00
co-authored by Claude Sonnet 5
parent 6a61c3f4af
commit 70c1e1c23f
6 changed files with 273 additions and 10 deletions
+4
View File
@@ -21,6 +21,10 @@
.smallclaw/databases/*.db.bak.*
.smallclaw/databases/nohup.out
# Derived FTS mirror of Chroma vector-memory data (rag-hybrid-search upgrade) — fully
# regenerable via scripts/backfill-fts.ts, not source-of-truth data.
.smallclaw/databases/rag_fts.db
# --- PYTHON CACHE ---
scripts/__pycache__/
**/__pycache__/
+39
View File
@@ -0,0 +1,39 @@
// One-time backfill: mirror existing Chroma vector-memory records into the new FTS5
// keyword-search sidecar (memory-fts.ts, added 2026-08-06 hybrid-search upgrade). Only
// needed once — addVector() keeps the two in sync for everything written after this ran.
import { ChromaClient } from 'chromadb';
import { USER_FACTS_COLLECTION, DAILY_EXTRACTS_COLLECTION } from '../src/gateway/memory/memory-vector';
import { upsertFtsRecord } from '../src/gateway/memory/memory-fts';
async function backfillCollection(client: ChromaClient, name: string) {
const col = await client.getOrCreateCollection({ name, embeddingFunction: null });
const count = await col.count();
console.log(`[${name}] ${count} records`);
const batchSize = 200;
let offset = 0;
let written = 0;
while (offset < count) {
const res = await col.get({ limit: batchSize, offset });
const ids = res.ids || [];
const docs = res.documents || [];
const metas = res.metadatas || [];
for (let i = 0; i < ids.length; i++) {
const text = docs[i] || '';
if (!text) continue;
const workspace = String((metas[i] as any)?.workspace || '');
upsertFtsRecord(name, ids[i], text, workspace);
written++;
}
offset += batchSize;
}
console.log(`[${name}] backfilled ${written} records into FTS`);
}
async function main() {
const client = new ChromaClient({ host: 'localhost', port: 8100 });
await backfillCollection(client, USER_FACTS_COLLECTION);
await backfillCollection(client, DAILY_EXTRACTS_COLLECTION);
console.log('done');
}
main().catch(e => { console.error(e); process.exit(1); });
+76 -10
View File
@@ -8,6 +8,8 @@ import {
USER_FACTS_COLLECTION,
DAILY_EXTRACTS_COLLECTION,
} from '../memory/memory-vector';
import { keywordSearch } from '../memory/memory-fts';
import { rerankCandidates } from '../memory/memory-rerank';
// resolvePromptPath stays in server.ts (used well beyond this cluster — e.g.
// routes-agents.ts also takes it as a dep), so it's threaded in.
@@ -401,30 +403,94 @@ return async function buildPersonalityContext(
// Vector-store recall: relevance-based, not the fixed-recency/fixed-char-cap USER.md/SOUL.md
// path — this is what lets facts survive past loadFile's truncation cutoff (see loadFile
// above). Scoped to this workspace via the `where` filter so users never see each other's
// facts. Best-effort — a Chroma/embedding outage must not break prompt building.
// facts. Best-effort — a Chroma/embedding/rerank outage must not break prompt building.
//
// Two collections, one query embedding (2026-07-24 design — see project_vector_memory_chroma
// memory): user_facts (memory_write-sourced, clean) gets a loose threshold; daily_extracts
// (LLM-extracted from raw conversation logs, noisier) gets a stricter one. Both merged and
// ranked together by distance so the model just sees one flat, relevance-sorted list.
// memory): user_facts (memory_write-sourced, clean), daily_extracts (LLM-extracted from raw
// conversation logs, noisier).
//
// Hybrid search + rerank (2026-08-06 upgrade — see this session's rag-upgrade memory):
// dense vector search alone blurs exact literal terms (IPs, model names, hostnames), and a
// pure distance cutoff can't tell "superficially similar" apart from "actually relevant" —
// so this widens the vector candidate pool and merges in FTS5 keyword hits (memory-fts.ts)
// for literal-term recall, then reranks the merged pool with a cheap LLM call (memory-
// rerank.ts) instead of a bare threshold.
//
// Reranking is gated, not unconditional: a first pass measured it adding ~5-7s to *every*
// turn (proportional to the ~25-candidate pool needed for reranking to actually have enough
// to disambiguate — shrinking the pool for speed silently defeated the point, see session
// notes), which is too much tax for turns where there's nothing ambiguous to resolve. Most
// turns fall into one of two cheap-to-detect cases: (a) no candidates at all — ordinary
// chit-chat with nothing memory-relevant — or (b) one clearly dominant vector match with no
// competing signal, where a distance gap alone is enough confidence. Only genuinely
// contested cases (multiple plausible candidates, e.g. 클로서버 vs 지서버 GPU facts) pay for
// the rerank call. Both hybrid search and reranking remain best-effort layers on top of a
// retrieval path that must keep working without them.
let vectorMemory = '';
if (messageText.trim().length >= 4) {
try {
const queryEmbedding = await embedQuery(messageText);
const [userFactHits, dailyExtractHits] = await Promise.all([
queryVectorsWithEmbedding(USER_FACTS_COLLECTION, queryEmbedding, 6, { workspace: workspacePath }),
queryVectorsWithEmbedding(DAILY_EXTRACTS_COLLECTION, queryEmbedding, 6, { workspace: workspacePath }),
queryVectorsWithEmbedding(USER_FACTS_COLLECTION, queryEmbedding, 15, { workspace: workspacePath }),
queryVectorsWithEmbedding(DAILY_EXTRACTS_COLLECTION, queryEmbedding, 15, { workspace: workspacePath }),
]);
const keywordHits = [
...keywordSearch(USER_FACTS_COLLECTION, messageText, 15, workspacePath),
...keywordSearch(DAILY_EXTRACTS_COLLECTION, messageText, 15, workspacePath),
];
// Thresholds recalibrated for EmbeddingGemma's wider distance spread (2026-07-24 switch
// from nomic-embed-text) — a 3-fact spot check gave ~0.51 for a genuinely relevant match,
// ~0.66 for same-topic-wrong-entity, ~0.97 for unrelated. Rough starting points, not a
// rigorous calibration — revisit if recall feels off/noisy in real use.
const relevant = [
// rigorous calibration — revisit if recall feels off/noisy in real use. Used both as the
// confidence signal for the skip-rerank fast path and as the fallback if reranking fails.
const allVectorHits = [...userFactHits, ...dailyExtractHits].sort((a, b) => a.distance - b.distance);
const byDistanceThreshold = [
...userFactHits.filter(h => h.distance < 0.65),
...dailyExtractHits.filter(h => h.distance < 0.55 && !h.metadata?.empty),
].sort((a, b) => a.distance - b.distance).slice(0, 8);
if (relevant.length > 0) {
vectorMemory = relevant.map(h => `- ${h.text}`).join('\n');
// Merge vector + keyword candidates, deduped by id (vector hits win the text on overlap —
// same source of truth either way). Capped so the rerank prompt (when it does run) stays
// small — this is meant to be a fast classification call, not another full turn.
const merged = new Map<string, string>();
for (const h of allVectorHits) if (h.id) merged.set(h.id, h.text);
let hasKeywordOnlyHit = false;
for (const h of keywordHits) {
if (!h.id) continue;
if (!merged.has(h.id)) { hasKeywordOnlyHit = true; merged.set(h.id, h.text); }
}
const candidates = Array.from(merged, ([id, text]) => ({ id, text })).slice(0, 25);
// Chroma's query() always returns up to topK neighbors regardless of how far they
// are — "candidates.length" is nearly always the full 25 cap and useless as a gate on
// its own. Distance quality is the real signal: nothing crossing the existing
// thresholds (and no keyword hits either) means nothing is actually relevant, matching
// what the pre-rerank baseline would have shown anyway.
const top = allVectorHits[0];
const runnerUp = allVectorHits[1];
const nothingRelevant = byDistanceThreshold.length === 0 && keywordHits.length === 0;
const confidentSingleMatch = !hasKeywordOnlyHit && !!top && top.distance < 0.35
&& (!runnerUp || runnerUp.distance - top.distance > 0.2);
if (nothingRelevant) {
// Nothing at all — ordinary conversational turn, nothing to recall.
} else if (confidentSingleMatch) {
// Too few candidates to be worth disambiguating, or one clear winner with no
// competing signal — skip the rerank round-trip and trust retrieval directly.
const fallback = byDistanceThreshold.length > 0
? byDistanceThreshold
: candidates.map(c => ({ text: c.text } as { text: string }));
vectorMemory = fallback.slice(0, 8).map(h => `- ${h.text}`).join('\n');
} else {
const rerankedIds = await rerankCandidates(messageText, candidates, 8);
if (rerankedIds === null) {
// Rerank call failed outright — fall back to plain distance-threshold behavior.
if (byDistanceThreshold.length > 0) vectorMemory = byDistanceThreshold.map(h => `- ${h.text}`).join('\n');
} else if (rerankedIds.length > 0) {
const byId = new Map(candidates.map(c => [c.id, c.text]));
vectorMemory = rerankedIds.map(id => `- ${byId.get(id)}`).join('\n');
}
}
} catch (err: any) {
console.warn('[buildPersonalityContext] vector memory query failed (non-fatal):', err.message);
+78
View File
@@ -0,0 +1,78 @@
// Keyword-search sidecar for the RAG memory system (2026-08-06 hybrid-search upgrade).
// Chroma only does dense vector search, which blurs exact literal terms — IPs, hostnames,
// model names, port numbers — that show up constantly in this memory's content (see
// project_wol_gate / project_jiserver_model_benchmarks style facts). This FTS5 mirror
// catches those exact-substring matches and gets merged with vector results before
// reranking (see memory-rerank.ts) in personality-context.ts.
//
// Uses FTS5's trigram tokenizer (character n-grams) instead of the default unicode61
// (whitespace word-segmentation) tokenizer — Korean has no whitespace word boundaries the
// way English does, so a word-tokenizer badly under-matches Korean text. Trigram works
// identically regardless of script/language, at the cost of being pure substring matching
// (no stemming) — acceptable here since the whole point is literal-term recall.
import Database from 'better-sqlite3';
import path from 'path';
import fs from 'fs';
import { getConfig } from '../../config/config.js';
let dbSingleton: Database.Database | null = null;
function getDbPath(): string {
const dir = path.join(getConfig().getConfigDir(), 'databases');
fs.mkdirSync(dir, { recursive: true });
return path.join(dir, 'rag_fts.db');
}
function getDb(): Database.Database {
if (dbSingleton) return dbSingleton;
const db = new Database(getDbPath());
db.pragma('journal_mode = WAL');
db.exec(`
CREATE VIRTUAL TABLE IF NOT EXISTS memory_fts USING fts5(
collection, doc_id, text, workspace,
tokenize='trigram'
);
`);
dbSingleton = db;
return db;
}
export function upsertFtsRecord(collection: string, docId: string, text: string, workspace: string): void {
const db = getDb();
db.prepare(`DELETE FROM memory_fts WHERE collection = ? AND doc_id = ?`).run(collection, docId);
db.prepare(`INSERT INTO memory_fts (collection, doc_id, text, workspace) VALUES (?, ?, ?, ?)`).run(collection, docId, text, workspace || '');
}
export interface FtsHit { id: string; text: string; }
// FTS5's query grammar treats quotes/parens/colons/leading-dash specially — quoting each
// extracted term as its own phrase neutralizes that (still trigram-matched inside the
// quotes) so a query containing e.g. "192.168.0.1" or "3060" doesn't throw a syntax error.
// OR-joined (not AND/phrase) because the goal is recall of *any* distinctive term in the
// query, not requiring the whole query to match verbatim.
function buildMatchQuery(query: string): string | null {
const terms = query
.split(/\s+/)
.map(t => t.trim())
.filter(t => t.length >= 2)
.slice(0, 12)
.map(t => `"${t.replace(/"/g, '""')}"`);
if (terms.length === 0) return null;
return terms.join(' OR ');
}
export function keywordSearch(collection: string, query: string, topK: number, workspace?: string): FtsHit[] {
const match = buildMatchQuery(query);
if (!match) return [];
try {
const db = getDb();
const rows = workspace
? db.prepare(`SELECT doc_id, text FROM memory_fts WHERE memory_fts MATCH ? AND collection = ? AND workspace = ? ORDER BY rank LIMIT ?`).all(match, collection, workspace, topK)
: db.prepare(`SELECT doc_id, text FROM memory_fts WHERE memory_fts MATCH ? AND collection = ? ORDER BY rank LIMIT ?`).all(match, collection, topK);
return (rows as any[]).map(r => ({ id: r.doc_id, text: r.text }));
} catch {
// Malformed MATCH query or FTS quirk — degrade to no keyword hits rather than
// breaking the whole retrieval path (same best-effort philosophy as the vector side).
return [];
}
}
+65
View File
@@ -0,0 +1,65 @@
// LLM-based reranking for the RAG memory system (2026-08-06 upgrade). Vector cosine-distance
// and FTS5 rank alone often surface candidates that are superficially similar/keyword-adjacent
// but not actually relevant to what was asked — a single extra classification-style LLM call
// over the merged candidate pool measurably improves precision, which is the standard "why
// bother reranking" argument in RAG literature and the reason this was the top recommendation
// over other options considered (see this session's rag-upgrade memory).
//
// Kept deliberately cheap: one call per turn (not one per candidate), a fast/cheap model,
// short timeout, and a null return on any failure so the caller can fall back to the plain
// distance-threshold behavior — this must never be a hard dependency for prompt building,
// same "best-effort" rule as the vector/FTS retrieval it sits on top of.
import { getConfig } from '../../config/config.js';
// "flash"-tier model chosen specifically for latency, not quality — this call is a small
// classification task (pick relevant indices from a short list), not a task worth spending
// primary-model-grade reasoning time on. Revisit if this model is ever unpulled/renamed.
const RERANK_MODEL = 'deepseek-v4-flash:cloud';
// Only fires on genuinely ambiguous retrieval (see the gating in personality-context.ts) —
// most turns never reach this call at all, so a generous timeout here doesn't cost anything
// on the common path. 8s was measured to occasionally clip real (successful, just slow)
// cloud-model responses under normal latency variance.
const RERANK_TIMEOUT_MS = 15_000;
export interface RerankCandidate { id: string; text: string; }
function getOllamaEndpoint(): string {
return (getConfig().getConfig() as any).ollama?.endpoint || 'http://localhost:11434';
}
// Returns the subset of candidate ids that are actually relevant, ranked best-first,
// capped at topK. Returns null (not []) on failure so callers can distinguish "reranked,
// turned out nothing was relevant" from "reranking itself didn't happen."
export async function rerankCandidates(query: string, candidates: RerankCandidate[], topK: number): Promise<string[] | null> {
if (candidates.length === 0) return [];
const listText = candidates.map((c, i) => `[${i}] ${c.text}`).join('\n');
const prompt = `사용자 질문: "${query}"
아래는 장기 기억에서 검색된 후보 목록입니다. 이 중 사용자 질문에 실제로 관련 있는 것만, 관련성 높은 순서로 번호만 골라줘 (최대 ${topK}개).
관련 있는 게 하나도 없으면 "없음"이라고만 답해. 설명 없이 번호만 쉼표로 구분해서 답해 (예: "2,0,5").
${listText}`;
try {
const res = await fetch(`${getOllamaEndpoint()}/api/chat`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ model: RERANK_MODEL, messages: [{ role: 'user', content: prompt }], stream: false }),
signal: AbortSignal.timeout(RERANK_TIMEOUT_MS),
});
if (!res.ok) throw new Error(`rerank HTTP ${res.status}`);
const data: any = await res.json();
const raw = String(data?.message?.content || '').trim();
if (!raw || raw === '없음') return [];
const seen = new Set<number>();
const indices = raw
.split(',')
.map(s => parseInt(s.trim(), 10))
.filter(n => Number.isInteger(n) && n >= 0 && n < candidates.length && !seen.has(n) && seen.add(n));
return indices.slice(0, topK).map(i => candidates[i].id);
} catch (err: any) {
console.warn('[memory-rerank] rerank failed (falling back to distance-threshold):', err.message);
return null;
}
}
+11
View File
@@ -10,6 +10,7 @@
// callers should apply a stricter relevance threshold against this one.
import { ChromaClient, type Collection } from 'chromadb';
import { getConfig } from '../../config/config.js';
import { upsertFtsRecord } from './memory-fts.js';
const CHROMA_HOST = 'localhost';
const CHROMA_PORT = 8100;
@@ -103,9 +104,17 @@ export async function addVector(collectionName: string, rec: MemoryVectorRecord)
documents: [rec.text],
metadatas: rec.metadata ? [rec.metadata] : undefined,
});
// Keyword-search mirror for hybrid retrieval (see memory-fts.ts) — kept best-effort:
// a write failing here must not fail the vector write it's piggybacking on.
try {
upsertFtsRecord(collectionName, rec.id, rec.text, String(rec.metadata?.workspace || ''));
} catch (err: any) {
console.warn('[memory-vector] FTS mirror write failed (non-fatal):', err.message);
}
}
export interface MemoryVectorHit {
id: string;
text: string;
distance: number;
metadata?: Record<string, unknown>;
@@ -119,10 +128,12 @@ export async function queryVectorsWithEmbedding(
): Promise<MemoryVectorHit[]> {
const col = await getCollection(collectionName);
const res = await col.query({ queryEmbeddings: [embedding], nResults: topK, where: toChromaWhere(where) });
const ids = res.ids?.[0] || [];
const docs = res.documents?.[0] || [];
const dists = res.distances?.[0] || [];
const metas = res.metadatas?.[0] || [];
return docs.map((text, i) => ({
id: ids[i] || '',
text: text || '',
distance: dists[i] ?? Infinity,
metadata: metas[i] as Record<string, unknown> | undefined,