// Backfill: LLM-extract atomic, reusable facts from raw daily conversation logs and store them // in the daily_extracts Chroma collection (separate from memory_write's clean user_facts // collection — see project_vector_memory_chroma memory for why). Naive paragraph-chunking of // raw transcripts was tried first and abandoned (too noisy, see migrate-memory-to-chroma.ts) — // this replaces that approach with real extraction. import fs from 'fs'; import path from 'path'; import { addVector, hasAnyVector, DAILY_EXTRACTS_COLLECTION } from '../src/gateway/memory/memory-vector'; const WORKSPACES: { workspace: string; label: string }[] = [ { workspace: '/srv/homeclaw/.smallclaw/users/papa/workspace', label: 'papa' }, { workspace: '/srv/homeclaw/.smallclaw/users/jasmine/workspace', label: 'jasmine' }, { workspace: '/srv/homeclaw/.smallclaw/users/cherry/workspace', label: 'cherry' }, { workspace: '/srv/homeclaw/.smallclaw/workspace', label: 'global' }, ]; const OLLAMA_ENDPOINT = 'http://localhost:11434'; const EXTRACT_MODEL = 'gemma4:31b-cloud'; // matches homeclaw's configured primary model async function extractFacts(dayContent: string): Promise { const prompt = `다음은 하루치 대화 로그입니다. 이 중에서 나중에 다시 참고할 가치가 있는 사실만 뽑아줘. 규칙: - 재사용 가능한 구체적 정보만 (하드웨어 스펙, 설정값, 결정사항, 프로젝트 진행상황, 사용자의 취향/제약사항 등) - 일회성 질문-답변, 잡담, 이미 끝난 계산/설명은 제외 - 각 사실은 한 줄에 하나씩, 문맥 없이도 이해되는 완결된 문장으로 작성 (예: "클로서버는 RTX 3060 12GB 2개를 사용한다" 처럼 — 어느 대상 얘기인지 반드시 명시) - 뽑을 게 없으면 "없음"이라고만 답해 - 사실 목록 외의 설명이나 서두는 쓰지 마 --- 로그 시작 --- ${dayContent} --- 로그 끝 ---`; const res = await fetch(`${OLLAMA_ENDPOINT}/api/chat`, { method: 'POST', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ model: EXTRACT_MODEL, messages: [{ role: 'user', content: prompt }], stream: false, }), signal: AbortSignal.timeout(120_000), }); if (!res.ok) throw new Error(`extract HTTP ${res.status}`); const data: any = await res.json(); const raw = String(data?.message?.content || '').trim(); if (!raw || raw === '없음') return []; return raw .split('\n') .map(l => l.trim().replace(/^[-*]\s*/, '')) .filter(l => l && l !== '없음'); } async function processWorkspace(workspace: string, label: string) { const memDir = path.join(workspace, 'memory'); if (!fs.existsSync(memDir)) { console.log(`[${label}] no memory dir, skip`); return; } // Never process today's file — it's still being written to during the day, so extracting // it early would capture a partial day and then skip it forever (hasAnyVector marks it // "done"). A nightly cron run naturally picks up "yesterday" once it's actually complete. const today = new Date().toISOString().slice(0, 10); const dailyFiles = fs.readdirSync(memDir) .filter(f => /^\d{4}-\d{2}-\d{2}\.md$/.test(f)) .filter(f => f.replace('.md', '') < today) .sort(); let totalFacts = 0, daysProcessed = 0, daysSkipped = 0, daysFailed = 0; for (const f of dailyFiles) { const date = f.replace('.md', ''); const content = fs.readFileSync(path.join(memDir, f), 'utf-8').trim(); if (!content) { daysSkipped++; continue; } if (await hasAnyVector(DAILY_EXTRACTS_COLLECTION, { workspace, date })) { daysSkipped++; continue; } try { const facts = await extractFacts(content); if (facts.length === 0) { // Record a sentinel so hasAnyVector() sees this date as processed on future reruns — // otherwise a legitimately fact-free day gets re-sent to the LLM every single rerun. await addVector(DAILY_EXTRACTS_COLLECTION, { id: `${workspace}:daily-extract:${date}:none`, text: '(해당 날짜에 재사용 가치 있는 사실 없음)', metadata: { workspace, date, source: 'daily-extract', empty: true }, }); } else { for (let i = 0; i < facts.length; i++) { const id = `${workspace}:daily-extract:${date}:${i}`; await addVector(DAILY_EXTRACTS_COLLECTION, { id, text: facts[i], metadata: { workspace, date, source: 'daily-extract' }, }); } } totalFacts += facts.length; daysProcessed++; console.log(`[${label}] ${date}: ${facts.length}개 추출`); } catch (err: any) { daysFailed++; console.warn(`[${label}] ${date} 실패: ${err.message}`); } } console.log(`[${label}] 완료 — ${daysProcessed}일 처리, ${totalFacts}개 사실, ${daysSkipped}일 스킵(빈파일), ${daysFailed}일 실패`); } async function main() { for (const { workspace, label } of WORKSPACES) { await processWorkspace(workspace, label); } } main().then(() => process.exit(0)).catch(err => { console.error(err); process.exit(1); });