Files
homeclaw/scripts/extract-daily-memory.ts
T
kimandClaude Sonnet 5 e180c3978e v4.3.19-24: 일별 메모리 추출 스크립트 깨진 import 경로 수정
memory-vector.ts가 src/gateway/memory/ 하위로 옮겨진 뒤 이 스크립트만
경로가 안 고쳐져서 MODULE_NOT_FOUND로 매일 새벽 1시 systemd 타이머가
3일 연속 즉시 실패하고 있었음(daily_extracts Chroma 컬렉션이 07-24
이후로 안 쌓이던 원인). 경로 수정 후 밀린 07-25~27 백필 확인.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-07-28 13:22:18 +09:00

120 lines
5.1 KiB
TypeScript

// Backfill: LLM-extract atomic, reusable facts from raw daily conversation logs and store them
// in the daily_extracts Chroma collection (separate from memory_write's clean user_facts
// collection — see project_vector_memory_chroma memory for why). Naive paragraph-chunking of
// raw transcripts was tried first and abandoned (too noisy, see migrate-memory-to-chroma.ts) —
// this replaces that approach with real extraction.
import fs from 'fs';
import path from 'path';
import { addVector, hasAnyVector, DAILY_EXTRACTS_COLLECTION } from '../src/gateway/memory/memory-vector';
const WORKSPACES: { workspace: string; label: string }[] = [
{ workspace: '/srv/homeclaw/.smallclaw/users/papa/workspace', label: 'papa' },
{ workspace: '/srv/homeclaw/.smallclaw/users/jasmine/workspace', label: 'jasmine' },
{ workspace: '/srv/homeclaw/.smallclaw/users/cherry/workspace', label: 'cherry' },
{ workspace: '/srv/homeclaw/.smallclaw/workspace', label: 'global' },
];
const OLLAMA_ENDPOINT = 'http://localhost:11434';
const EXTRACT_MODEL = 'gemma4:31b-cloud'; // matches homeclaw's configured primary model
async function extractFacts(dayContent: string): Promise<string[]> {
const prompt = `다음은 하루치 대화 로그입니다. 이 중에서 나중에 다시 참고할 가치가 있는 사실만 뽑아줘.
규칙:
- 재사용 가능한 구체적 정보만 (하드웨어 스펙, 설정값, 결정사항, 프로젝트 진행상황, 사용자의 취향/제약사항 등)
- 일회성 질문-답변, 잡담, 이미 끝난 계산/설명은 제외
- 각 사실은 한 줄에 하나씩, 문맥 없이도 이해되는 완결된 문장으로 작성 (예: "클로서버는 RTX 3060 12GB 2개를 사용한다" 처럼 — 어느 대상 얘기인지 반드시 명시)
- 뽑을 게 없으면 "없음"이라고만 답해
- 사실 목록 외의 설명이나 서두는 쓰지 마
--- 로그 시작 ---
${dayContent}
--- 로그 끝 ---`;
const res = await fetch(`${OLLAMA_ENDPOINT}/api/chat`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
model: EXTRACT_MODEL,
messages: [{ role: 'user', content: prompt }],
stream: false,
}),
signal: AbortSignal.timeout(120_000),
});
if (!res.ok) throw new Error(`extract HTTP ${res.status}`);
const data: any = await res.json();
const raw = String(data?.message?.content || '').trim();
if (!raw || raw === '없음') return [];
return raw
.split('\n')
.map(l => l.trim().replace(/^[-*]\s*/, ''))
.filter(l => l && l !== '없음');
}
async function processWorkspace(workspace: string, label: string) {
const memDir = path.join(workspace, 'memory');
if (!fs.existsSync(memDir)) {
console.log(`[${label}] no memory dir, skip`);
return;
}
// Never process today's file — it's still being written to during the day, so extracting
// it early would capture a partial day and then skip it forever (hasAnyVector marks it
// "done"). A nightly cron run naturally picks up "yesterday" once it's actually complete.
const today = new Date().toISOString().slice(0, 10);
const dailyFiles = fs.readdirSync(memDir)
.filter(f => /^\d{4}-\d{2}-\d{2}\.md$/.test(f))
.filter(f => f.replace('.md', '') < today)
.sort();
let totalFacts = 0, daysProcessed = 0, daysSkipped = 0, daysFailed = 0;
for (const f of dailyFiles) {
const date = f.replace('.md', '');
const content = fs.readFileSync(path.join(memDir, f), 'utf-8').trim();
if (!content) { daysSkipped++; continue; }
if (await hasAnyVector(DAILY_EXTRACTS_COLLECTION, { workspace, date })) {
daysSkipped++;
continue;
}
try {
const facts = await extractFacts(content);
if (facts.length === 0) {
// Record a sentinel so hasAnyVector() sees this date as processed on future reruns —
// otherwise a legitimately fact-free day gets re-sent to the LLM every single rerun.
await addVector(DAILY_EXTRACTS_COLLECTION, {
id: `${workspace}:daily-extract:${date}:none`,
text: '(해당 날짜에 재사용 가치 있는 사실 없음)',
metadata: { workspace, date, source: 'daily-extract', empty: true },
});
} else {
for (let i = 0; i < facts.length; i++) {
const id = `${workspace}:daily-extract:${date}:${i}`;
await addVector(DAILY_EXTRACTS_COLLECTION, {
id,
text: facts[i],
metadata: { workspace, date, source: 'daily-extract' },
});
}
}
totalFacts += facts.length;
daysProcessed++;
console.log(`[${label}] ${date}: ${facts.length}개 추출`);
} catch (err: any) {
daysFailed++;
console.warn(`[${label}] ${date} 실패: ${err.message}`);
}
}
console.log(`[${label}] 완료 — ${daysProcessed}일 처리, ${totalFacts}개 사실, ${daysSkipped}일 스킵(빈파일), ${daysFailed}일 실패`);
}
async function main() {
for (const { workspace, label } of WORKSPACES) {
await processWorkspace(workspace, label);
}
}
main().then(() => process.exit(0)).catch(err => { console.error(err); process.exit(1); });