memory-vector.ts가 src/gateway/memory/ 하위로 옮겨진 뒤 이 스크립트만 경로가 안 고쳐져서 MODULE_NOT_FOUND로 매일 새벽 1시 systemd 타이머가 3일 연속 즉시 실패하고 있었음(daily_extracts Chroma 컬렉션이 07-24 이후로 안 쌓이던 원인). 경로 수정 후 밀린 07-25~27 백필 확인. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
120 lines
5.1 KiB
TypeScript
120 lines
5.1 KiB
TypeScript
// Backfill: LLM-extract atomic, reusable facts from raw daily conversation logs and store them
|
|
// in the daily_extracts Chroma collection (separate from memory_write's clean user_facts
|
|
// collection — see project_vector_memory_chroma memory for why). Naive paragraph-chunking of
|
|
// raw transcripts was tried first and abandoned (too noisy, see migrate-memory-to-chroma.ts) —
|
|
// this replaces that approach with real extraction.
|
|
import fs from 'fs';
|
|
import path from 'path';
|
|
import { addVector, hasAnyVector, DAILY_EXTRACTS_COLLECTION } from '../src/gateway/memory/memory-vector';
|
|
|
|
const WORKSPACES: { workspace: string; label: string }[] = [
|
|
{ workspace: '/srv/homeclaw/.smallclaw/users/papa/workspace', label: 'papa' },
|
|
{ workspace: '/srv/homeclaw/.smallclaw/users/jasmine/workspace', label: 'jasmine' },
|
|
{ workspace: '/srv/homeclaw/.smallclaw/users/cherry/workspace', label: 'cherry' },
|
|
{ workspace: '/srv/homeclaw/.smallclaw/workspace', label: 'global' },
|
|
];
|
|
|
|
const OLLAMA_ENDPOINT = 'http://localhost:11434';
|
|
const EXTRACT_MODEL = 'gemma4:31b-cloud'; // matches homeclaw's configured primary model
|
|
|
|
async function extractFacts(dayContent: string): Promise<string[]> {
|
|
const prompt = `다음은 하루치 대화 로그입니다. 이 중에서 나중에 다시 참고할 가치가 있는 사실만 뽑아줘.
|
|
|
|
규칙:
|
|
- 재사용 가능한 구체적 정보만 (하드웨어 스펙, 설정값, 결정사항, 프로젝트 진행상황, 사용자의 취향/제약사항 등)
|
|
- 일회성 질문-답변, 잡담, 이미 끝난 계산/설명은 제외
|
|
- 각 사실은 한 줄에 하나씩, 문맥 없이도 이해되는 완결된 문장으로 작성 (예: "클로서버는 RTX 3060 12GB 2개를 사용한다" 처럼 — 어느 대상 얘기인지 반드시 명시)
|
|
- 뽑을 게 없으면 "없음"이라고만 답해
|
|
- 사실 목록 외의 설명이나 서두는 쓰지 마
|
|
|
|
--- 로그 시작 ---
|
|
${dayContent}
|
|
--- 로그 끝 ---`;
|
|
|
|
const res = await fetch(`${OLLAMA_ENDPOINT}/api/chat`, {
|
|
method: 'POST',
|
|
headers: { 'Content-Type': 'application/json' },
|
|
body: JSON.stringify({
|
|
model: EXTRACT_MODEL,
|
|
messages: [{ role: 'user', content: prompt }],
|
|
stream: false,
|
|
}),
|
|
signal: AbortSignal.timeout(120_000),
|
|
});
|
|
if (!res.ok) throw new Error(`extract HTTP ${res.status}`);
|
|
const data: any = await res.json();
|
|
const raw = String(data?.message?.content || '').trim();
|
|
if (!raw || raw === '없음') return [];
|
|
|
|
return raw
|
|
.split('\n')
|
|
.map(l => l.trim().replace(/^[-*]\s*/, ''))
|
|
.filter(l => l && l !== '없음');
|
|
}
|
|
|
|
async function processWorkspace(workspace: string, label: string) {
|
|
const memDir = path.join(workspace, 'memory');
|
|
if (!fs.existsSync(memDir)) {
|
|
console.log(`[${label}] no memory dir, skip`);
|
|
return;
|
|
}
|
|
// Never process today's file — it's still being written to during the day, so extracting
|
|
// it early would capture a partial day and then skip it forever (hasAnyVector marks it
|
|
// "done"). A nightly cron run naturally picks up "yesterday" once it's actually complete.
|
|
const today = new Date().toISOString().slice(0, 10);
|
|
const dailyFiles = fs.readdirSync(memDir)
|
|
.filter(f => /^\d{4}-\d{2}-\d{2}\.md$/.test(f))
|
|
.filter(f => f.replace('.md', '') < today)
|
|
.sort();
|
|
let totalFacts = 0, daysProcessed = 0, daysSkipped = 0, daysFailed = 0;
|
|
|
|
for (const f of dailyFiles) {
|
|
const date = f.replace('.md', '');
|
|
const content = fs.readFileSync(path.join(memDir, f), 'utf-8').trim();
|
|
if (!content) { daysSkipped++; continue; }
|
|
|
|
if (await hasAnyVector(DAILY_EXTRACTS_COLLECTION, { workspace, date })) {
|
|
daysSkipped++;
|
|
continue;
|
|
}
|
|
|
|
try {
|
|
const facts = await extractFacts(content);
|
|
if (facts.length === 0) {
|
|
// Record a sentinel so hasAnyVector() sees this date as processed on future reruns —
|
|
// otherwise a legitimately fact-free day gets re-sent to the LLM every single rerun.
|
|
await addVector(DAILY_EXTRACTS_COLLECTION, {
|
|
id: `${workspace}:daily-extract:${date}:none`,
|
|
text: '(해당 날짜에 재사용 가치 있는 사실 없음)',
|
|
metadata: { workspace, date, source: 'daily-extract', empty: true },
|
|
});
|
|
} else {
|
|
for (let i = 0; i < facts.length; i++) {
|
|
const id = `${workspace}:daily-extract:${date}:${i}`;
|
|
await addVector(DAILY_EXTRACTS_COLLECTION, {
|
|
id,
|
|
text: facts[i],
|
|
metadata: { workspace, date, source: 'daily-extract' },
|
|
});
|
|
}
|
|
}
|
|
totalFacts += facts.length;
|
|
daysProcessed++;
|
|
console.log(`[${label}] ${date}: ${facts.length}개 추출`);
|
|
} catch (err: any) {
|
|
daysFailed++;
|
|
console.warn(`[${label}] ${date} 실패: ${err.message}`);
|
|
}
|
|
}
|
|
|
|
console.log(`[${label}] 완료 — ${daysProcessed}일 처리, ${totalFacts}개 사실, ${daysSkipped}일 스킵(빈파일), ${daysFailed}일 실패`);
|
|
}
|
|
|
|
async function main() {
|
|
for (const { workspace, label } of WORKSPACES) {
|
|
await processWorkspace(workspace, label);
|
|
}
|
|
}
|
|
|
|
main().then(() => process.exit(0)).catch(err => { console.error(err); process.exit(1); });
|