RAG 기억 검색에 하이브리드 검색(벡터+키워드) + 재랭킹 추가
Chroma 벡터검색만으로는 IP·모델명 같은 정확한 용어를 놓치는 경우가 있어 SQLite FTS5(trigram) 키워드검색을 병합하고, 애매한 경우(예: 클로서버 vs 지서버 GPU 스펙 혼동)만 LLM 재랭킹으로 오답을 걸러내도록 함. 재랭킹은 후보가 없거나 확실한 단일매치일 때는 건너뛰어 대부분의 대화에서는 지연시간 증가가 거의 없음. 기존 570개 기록은 scripts/backfill-fts.ts로 백필. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -21,6 +21,10 @@
|
||||
.smallclaw/databases/*.db.bak.*
|
||||
.smallclaw/databases/nohup.out
|
||||
|
||||
# Derived FTS mirror of Chroma vector-memory data (rag-hybrid-search upgrade) — fully
|
||||
# regenerable via scripts/backfill-fts.ts, not source-of-truth data.
|
||||
.smallclaw/databases/rag_fts.db
|
||||
|
||||
# --- PYTHON CACHE ---
|
||||
scripts/__pycache__/
|
||||
**/__pycache__/
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
// One-time backfill: mirror existing Chroma vector-memory records into the new FTS5
|
||||
// keyword-search sidecar (memory-fts.ts, added 2026-08-06 hybrid-search upgrade). Only
|
||||
// needed once — addVector() keeps the two in sync for everything written after this ran.
|
||||
import { ChromaClient } from 'chromadb';
|
||||
import { USER_FACTS_COLLECTION, DAILY_EXTRACTS_COLLECTION } from '../src/gateway/memory/memory-vector';
|
||||
import { upsertFtsRecord } from '../src/gateway/memory/memory-fts';
|
||||
|
||||
async function backfillCollection(client: ChromaClient, name: string) {
|
||||
const col = await client.getOrCreateCollection({ name, embeddingFunction: null });
|
||||
const count = await col.count();
|
||||
console.log(`[${name}] ${count} records`);
|
||||
const batchSize = 200;
|
||||
let offset = 0;
|
||||
let written = 0;
|
||||
while (offset < count) {
|
||||
const res = await col.get({ limit: batchSize, offset });
|
||||
const ids = res.ids || [];
|
||||
const docs = res.documents || [];
|
||||
const metas = res.metadatas || [];
|
||||
for (let i = 0; i < ids.length; i++) {
|
||||
const text = docs[i] || '';
|
||||
if (!text) continue;
|
||||
const workspace = String((metas[i] as any)?.workspace || '');
|
||||
upsertFtsRecord(name, ids[i], text, workspace);
|
||||
written++;
|
||||
}
|
||||
offset += batchSize;
|
||||
}
|
||||
console.log(`[${name}] backfilled ${written} records into FTS`);
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const client = new ChromaClient({ host: 'localhost', port: 8100 });
|
||||
await backfillCollection(client, USER_FACTS_COLLECTION);
|
||||
await backfillCollection(client, DAILY_EXTRACTS_COLLECTION);
|
||||
console.log('done');
|
||||
}
|
||||
|
||||
main().catch(e => { console.error(e); process.exit(1); });
|
||||
@@ -8,6 +8,8 @@ import {
|
||||
USER_FACTS_COLLECTION,
|
||||
DAILY_EXTRACTS_COLLECTION,
|
||||
} from '../memory/memory-vector';
|
||||
import { keywordSearch } from '../memory/memory-fts';
|
||||
import { rerankCandidates } from '../memory/memory-rerank';
|
||||
|
||||
// resolvePromptPath stays in server.ts (used well beyond this cluster — e.g.
|
||||
// routes-agents.ts also takes it as a dep), so it's threaded in.
|
||||
@@ -401,30 +403,94 @@ return async function buildPersonalityContext(
|
||||
// Vector-store recall: relevance-based, not the fixed-recency/fixed-char-cap USER.md/SOUL.md
|
||||
// path — this is what lets facts survive past loadFile's truncation cutoff (see loadFile
|
||||
// above). Scoped to this workspace via the `where` filter so users never see each other's
|
||||
// facts. Best-effort — a Chroma/embedding outage must not break prompt building.
|
||||
// facts. Best-effort — a Chroma/embedding/rerank outage must not break prompt building.
|
||||
//
|
||||
// Two collections, one query embedding (2026-07-24 design — see project_vector_memory_chroma
|
||||
// memory): user_facts (memory_write-sourced, clean) gets a loose threshold; daily_extracts
|
||||
// (LLM-extracted from raw conversation logs, noisier) gets a stricter one. Both merged and
|
||||
// ranked together by distance so the model just sees one flat, relevance-sorted list.
|
||||
// memory): user_facts (memory_write-sourced, clean), daily_extracts (LLM-extracted from raw
|
||||
// conversation logs, noisier).
|
||||
//
|
||||
// Hybrid search + rerank (2026-08-06 upgrade — see this session's rag-upgrade memory):
|
||||
// dense vector search alone blurs exact literal terms (IPs, model names, hostnames), and a
|
||||
// pure distance cutoff can't tell "superficially similar" apart from "actually relevant" —
|
||||
// so this widens the vector candidate pool and merges in FTS5 keyword hits (memory-fts.ts)
|
||||
// for literal-term recall, then reranks the merged pool with a cheap LLM call (memory-
|
||||
// rerank.ts) instead of a bare threshold.
|
||||
//
|
||||
// Reranking is gated, not unconditional: a first pass measured it adding ~5-7s to *every*
|
||||
// turn (proportional to the ~25-candidate pool needed for reranking to actually have enough
|
||||
// to disambiguate — shrinking the pool for speed silently defeated the point, see session
|
||||
// notes), which is too much tax for turns where there's nothing ambiguous to resolve. Most
|
||||
// turns fall into one of two cheap-to-detect cases: (a) no candidates at all — ordinary
|
||||
// chit-chat with nothing memory-relevant — or (b) one clearly dominant vector match with no
|
||||
// competing signal, where a distance gap alone is enough confidence. Only genuinely
|
||||
// contested cases (multiple plausible candidates, e.g. 클로서버 vs 지서버 GPU facts) pay for
|
||||
// the rerank call. Both hybrid search and reranking remain best-effort layers on top of a
|
||||
// retrieval path that must keep working without them.
|
||||
let vectorMemory = '';
|
||||
if (messageText.trim().length >= 4) {
|
||||
try {
|
||||
const queryEmbedding = await embedQuery(messageText);
|
||||
const [userFactHits, dailyExtractHits] = await Promise.all([
|
||||
queryVectorsWithEmbedding(USER_FACTS_COLLECTION, queryEmbedding, 6, { workspace: workspacePath }),
|
||||
queryVectorsWithEmbedding(DAILY_EXTRACTS_COLLECTION, queryEmbedding, 6, { workspace: workspacePath }),
|
||||
queryVectorsWithEmbedding(USER_FACTS_COLLECTION, queryEmbedding, 15, { workspace: workspacePath }),
|
||||
queryVectorsWithEmbedding(DAILY_EXTRACTS_COLLECTION, queryEmbedding, 15, { workspace: workspacePath }),
|
||||
]);
|
||||
const keywordHits = [
|
||||
...keywordSearch(USER_FACTS_COLLECTION, messageText, 15, workspacePath),
|
||||
...keywordSearch(DAILY_EXTRACTS_COLLECTION, messageText, 15, workspacePath),
|
||||
];
|
||||
|
||||
// Thresholds recalibrated for EmbeddingGemma's wider distance spread (2026-07-24 switch
|
||||
// from nomic-embed-text) — a 3-fact spot check gave ~0.51 for a genuinely relevant match,
|
||||
// ~0.66 for same-topic-wrong-entity, ~0.97 for unrelated. Rough starting points, not a
|
||||
// rigorous calibration — revisit if recall feels off/noisy in real use.
|
||||
const relevant = [
|
||||
// rigorous calibration — revisit if recall feels off/noisy in real use. Used both as the
|
||||
// confidence signal for the skip-rerank fast path and as the fallback if reranking fails.
|
||||
const allVectorHits = [...userFactHits, ...dailyExtractHits].sort((a, b) => a.distance - b.distance);
|
||||
const byDistanceThreshold = [
|
||||
...userFactHits.filter(h => h.distance < 0.65),
|
||||
...dailyExtractHits.filter(h => h.distance < 0.55 && !h.metadata?.empty),
|
||||
].sort((a, b) => a.distance - b.distance).slice(0, 8);
|
||||
if (relevant.length > 0) {
|
||||
vectorMemory = relevant.map(h => `- ${h.text}`).join('\n');
|
||||
|
||||
// Merge vector + keyword candidates, deduped by id (vector hits win the text on overlap —
|
||||
// same source of truth either way). Capped so the rerank prompt (when it does run) stays
|
||||
// small — this is meant to be a fast classification call, not another full turn.
|
||||
const merged = new Map<string, string>();
|
||||
for (const h of allVectorHits) if (h.id) merged.set(h.id, h.text);
|
||||
let hasKeywordOnlyHit = false;
|
||||
for (const h of keywordHits) {
|
||||
if (!h.id) continue;
|
||||
if (!merged.has(h.id)) { hasKeywordOnlyHit = true; merged.set(h.id, h.text); }
|
||||
}
|
||||
const candidates = Array.from(merged, ([id, text]) => ({ id, text })).slice(0, 25);
|
||||
|
||||
// Chroma's query() always returns up to topK neighbors regardless of how far they
|
||||
// are — "candidates.length" is nearly always the full 25 cap and useless as a gate on
|
||||
// its own. Distance quality is the real signal: nothing crossing the existing
|
||||
// thresholds (and no keyword hits either) means nothing is actually relevant, matching
|
||||
// what the pre-rerank baseline would have shown anyway.
|
||||
const top = allVectorHits[0];
|
||||
const runnerUp = allVectorHits[1];
|
||||
const nothingRelevant = byDistanceThreshold.length === 0 && keywordHits.length === 0;
|
||||
const confidentSingleMatch = !hasKeywordOnlyHit && !!top && top.distance < 0.35
|
||||
&& (!runnerUp || runnerUp.distance - top.distance > 0.2);
|
||||
|
||||
if (nothingRelevant) {
|
||||
// Nothing at all — ordinary conversational turn, nothing to recall.
|
||||
} else if (confidentSingleMatch) {
|
||||
// Too few candidates to be worth disambiguating, or one clear winner with no
|
||||
// competing signal — skip the rerank round-trip and trust retrieval directly.
|
||||
const fallback = byDistanceThreshold.length > 0
|
||||
? byDistanceThreshold
|
||||
: candidates.map(c => ({ text: c.text } as { text: string }));
|
||||
vectorMemory = fallback.slice(0, 8).map(h => `- ${h.text}`).join('\n');
|
||||
} else {
|
||||
const rerankedIds = await rerankCandidates(messageText, candidates, 8);
|
||||
if (rerankedIds === null) {
|
||||
// Rerank call failed outright — fall back to plain distance-threshold behavior.
|
||||
if (byDistanceThreshold.length > 0) vectorMemory = byDistanceThreshold.map(h => `- ${h.text}`).join('\n');
|
||||
} else if (rerankedIds.length > 0) {
|
||||
const byId = new Map(candidates.map(c => [c.id, c.text]));
|
||||
vectorMemory = rerankedIds.map(id => `- ${byId.get(id)}`).join('\n');
|
||||
}
|
||||
}
|
||||
} catch (err: any) {
|
||||
console.warn('[buildPersonalityContext] vector memory query failed (non-fatal):', err.message);
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
// Keyword-search sidecar for the RAG memory system (2026-08-06 hybrid-search upgrade).
|
||||
// Chroma only does dense vector search, which blurs exact literal terms — IPs, hostnames,
|
||||
// model names, port numbers — that show up constantly in this memory's content (see
|
||||
// project_wol_gate / project_jiserver_model_benchmarks style facts). This FTS5 mirror
|
||||
// catches those exact-substring matches and gets merged with vector results before
|
||||
// reranking (see memory-rerank.ts) in personality-context.ts.
|
||||
//
|
||||
// Uses FTS5's trigram tokenizer (character n-grams) instead of the default unicode61
|
||||
// (whitespace word-segmentation) tokenizer — Korean has no whitespace word boundaries the
|
||||
// way English does, so a word-tokenizer badly under-matches Korean text. Trigram works
|
||||
// identically regardless of script/language, at the cost of being pure substring matching
|
||||
// (no stemming) — acceptable here since the whole point is literal-term recall.
|
||||
import Database from 'better-sqlite3';
|
||||
import path from 'path';
|
||||
import fs from 'fs';
|
||||
import { getConfig } from '../../config/config.js';
|
||||
|
||||
let dbSingleton: Database.Database | null = null;
|
||||
|
||||
function getDbPath(): string {
|
||||
const dir = path.join(getConfig().getConfigDir(), 'databases');
|
||||
fs.mkdirSync(dir, { recursive: true });
|
||||
return path.join(dir, 'rag_fts.db');
|
||||
}
|
||||
|
||||
function getDb(): Database.Database {
|
||||
if (dbSingleton) return dbSingleton;
|
||||
const db = new Database(getDbPath());
|
||||
db.pragma('journal_mode = WAL');
|
||||
db.exec(`
|
||||
CREATE VIRTUAL TABLE IF NOT EXISTS memory_fts USING fts5(
|
||||
collection, doc_id, text, workspace,
|
||||
tokenize='trigram'
|
||||
);
|
||||
`);
|
||||
dbSingleton = db;
|
||||
return db;
|
||||
}
|
||||
|
||||
export function upsertFtsRecord(collection: string, docId: string, text: string, workspace: string): void {
|
||||
const db = getDb();
|
||||
db.prepare(`DELETE FROM memory_fts WHERE collection = ? AND doc_id = ?`).run(collection, docId);
|
||||
db.prepare(`INSERT INTO memory_fts (collection, doc_id, text, workspace) VALUES (?, ?, ?, ?)`).run(collection, docId, text, workspace || '');
|
||||
}
|
||||
|
||||
export interface FtsHit { id: string; text: string; }
|
||||
|
||||
// FTS5's query grammar treats quotes/parens/colons/leading-dash specially — quoting each
|
||||
// extracted term as its own phrase neutralizes that (still trigram-matched inside the
|
||||
// quotes) so a query containing e.g. "192.168.0.1" or "3060" doesn't throw a syntax error.
|
||||
// OR-joined (not AND/phrase) because the goal is recall of *any* distinctive term in the
|
||||
// query, not requiring the whole query to match verbatim.
|
||||
function buildMatchQuery(query: string): string | null {
|
||||
const terms = query
|
||||
.split(/\s+/)
|
||||
.map(t => t.trim())
|
||||
.filter(t => t.length >= 2)
|
||||
.slice(0, 12)
|
||||
.map(t => `"${t.replace(/"/g, '""')}"`);
|
||||
if (terms.length === 0) return null;
|
||||
return terms.join(' OR ');
|
||||
}
|
||||
|
||||
export function keywordSearch(collection: string, query: string, topK: number, workspace?: string): FtsHit[] {
|
||||
const match = buildMatchQuery(query);
|
||||
if (!match) return [];
|
||||
try {
|
||||
const db = getDb();
|
||||
const rows = workspace
|
||||
? db.prepare(`SELECT doc_id, text FROM memory_fts WHERE memory_fts MATCH ? AND collection = ? AND workspace = ? ORDER BY rank LIMIT ?`).all(match, collection, workspace, topK)
|
||||
: db.prepare(`SELECT doc_id, text FROM memory_fts WHERE memory_fts MATCH ? AND collection = ? ORDER BY rank LIMIT ?`).all(match, collection, topK);
|
||||
return (rows as any[]).map(r => ({ id: r.doc_id, text: r.text }));
|
||||
} catch {
|
||||
// Malformed MATCH query or FTS quirk — degrade to no keyword hits rather than
|
||||
// breaking the whole retrieval path (same best-effort philosophy as the vector side).
|
||||
return [];
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
// LLM-based reranking for the RAG memory system (2026-08-06 upgrade). Vector cosine-distance
|
||||
// and FTS5 rank alone often surface candidates that are superficially similar/keyword-adjacent
|
||||
// but not actually relevant to what was asked — a single extra classification-style LLM call
|
||||
// over the merged candidate pool measurably improves precision, which is the standard "why
|
||||
// bother reranking" argument in RAG literature and the reason this was the top recommendation
|
||||
// over other options considered (see this session's rag-upgrade memory).
|
||||
//
|
||||
// Kept deliberately cheap: one call per turn (not one per candidate), a fast/cheap model,
|
||||
// short timeout, and a null return on any failure so the caller can fall back to the plain
|
||||
// distance-threshold behavior — this must never be a hard dependency for prompt building,
|
||||
// same "best-effort" rule as the vector/FTS retrieval it sits on top of.
|
||||
import { getConfig } from '../../config/config.js';
|
||||
|
||||
// "flash"-tier model chosen specifically for latency, not quality — this call is a small
|
||||
// classification task (pick relevant indices from a short list), not a task worth spending
|
||||
// primary-model-grade reasoning time on. Revisit if this model is ever unpulled/renamed.
|
||||
const RERANK_MODEL = 'deepseek-v4-flash:cloud';
|
||||
// Only fires on genuinely ambiguous retrieval (see the gating in personality-context.ts) —
|
||||
// most turns never reach this call at all, so a generous timeout here doesn't cost anything
|
||||
// on the common path. 8s was measured to occasionally clip real (successful, just slow)
|
||||
// cloud-model responses under normal latency variance.
|
||||
const RERANK_TIMEOUT_MS = 15_000;
|
||||
|
||||
export interface RerankCandidate { id: string; text: string; }
|
||||
|
||||
function getOllamaEndpoint(): string {
|
||||
return (getConfig().getConfig() as any).ollama?.endpoint || 'http://localhost:11434';
|
||||
}
|
||||
|
||||
// Returns the subset of candidate ids that are actually relevant, ranked best-first,
|
||||
// capped at topK. Returns null (not []) on failure so callers can distinguish "reranked,
|
||||
// turned out nothing was relevant" from "reranking itself didn't happen."
|
||||
export async function rerankCandidates(query: string, candidates: RerankCandidate[], topK: number): Promise<string[] | null> {
|
||||
if (candidates.length === 0) return [];
|
||||
const listText = candidates.map((c, i) => `[${i}] ${c.text}`).join('\n');
|
||||
const prompt = `사용자 질문: "${query}"
|
||||
|
||||
아래는 장기 기억에서 검색된 후보 목록입니다. 이 중 사용자 질문에 실제로 관련 있는 것만, 관련성 높은 순서로 번호만 골라줘 (최대 ${topK}개).
|
||||
관련 있는 게 하나도 없으면 "없음"이라고만 답해. 설명 없이 번호만 쉼표로 구분해서 답해 (예: "2,0,5").
|
||||
|
||||
${listText}`;
|
||||
|
||||
try {
|
||||
const res = await fetch(`${getOllamaEndpoint()}/api/chat`, {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ model: RERANK_MODEL, messages: [{ role: 'user', content: prompt }], stream: false }),
|
||||
signal: AbortSignal.timeout(RERANK_TIMEOUT_MS),
|
||||
});
|
||||
if (!res.ok) throw new Error(`rerank HTTP ${res.status}`);
|
||||
const data: any = await res.json();
|
||||
const raw = String(data?.message?.content || '').trim();
|
||||
if (!raw || raw === '없음') return [];
|
||||
|
||||
const seen = new Set<number>();
|
||||
const indices = raw
|
||||
.split(',')
|
||||
.map(s => parseInt(s.trim(), 10))
|
||||
.filter(n => Number.isInteger(n) && n >= 0 && n < candidates.length && !seen.has(n) && seen.add(n));
|
||||
return indices.slice(0, topK).map(i => candidates[i].id);
|
||||
} catch (err: any) {
|
||||
console.warn('[memory-rerank] rerank failed (falling back to distance-threshold):', err.message);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
@@ -10,6 +10,7 @@
|
||||
// callers should apply a stricter relevance threshold against this one.
|
||||
import { ChromaClient, type Collection } from 'chromadb';
|
||||
import { getConfig } from '../../config/config.js';
|
||||
import { upsertFtsRecord } from './memory-fts.js';
|
||||
|
||||
const CHROMA_HOST = 'localhost';
|
||||
const CHROMA_PORT = 8100;
|
||||
@@ -103,9 +104,17 @@ export async function addVector(collectionName: string, rec: MemoryVectorRecord)
|
||||
documents: [rec.text],
|
||||
metadatas: rec.metadata ? [rec.metadata] : undefined,
|
||||
});
|
||||
// Keyword-search mirror for hybrid retrieval (see memory-fts.ts) — kept best-effort:
|
||||
// a write failing here must not fail the vector write it's piggybacking on.
|
||||
try {
|
||||
upsertFtsRecord(collectionName, rec.id, rec.text, String(rec.metadata?.workspace || ''));
|
||||
} catch (err: any) {
|
||||
console.warn('[memory-vector] FTS mirror write failed (non-fatal):', err.message);
|
||||
}
|
||||
}
|
||||
|
||||
export interface MemoryVectorHit {
|
||||
id: string;
|
||||
text: string;
|
||||
distance: number;
|
||||
metadata?: Record<string, unknown>;
|
||||
@@ -119,10 +128,12 @@ export async function queryVectorsWithEmbedding(
|
||||
): Promise<MemoryVectorHit[]> {
|
||||
const col = await getCollection(collectionName);
|
||||
const res = await col.query({ queryEmbeddings: [embedding], nResults: topK, where: toChromaWhere(where) });
|
||||
const ids = res.ids?.[0] || [];
|
||||
const docs = res.documents?.[0] || [];
|
||||
const dists = res.distances?.[0] || [];
|
||||
const metas = res.metadatas?.[0] || [];
|
||||
return docs.map((text, i) => ({
|
||||
id: ids[i] || '',
|
||||
text: text || '',
|
||||
distance: dists[i] ?? Infinity,
|
||||
metadata: metas[i] as Record<string, unknown> | undefined,
|
||||
|
||||
Reference in New Issue
Block a user