Files
homeclaw/tests/top-headlines.test.ts
T
kimandClaude Opus 5 faee483cf1 fix: 뉴스 검색이 2분 넘게 걸리던 원인 두 가지
사용자 신고: "지금 뉴스검색을 2분 넘게 하고 있는데?"
"오늘 미국 주요 뉴스" 한 턴에서 도구를 11번 불렀다. 라운드마다 muse-glimmer가
통째로 한 번씩 생성해야 하는데 22 tok/s dense라 라운드 수가 그대로 시간이 된다.

1) 1면 RSS가 제목과 링크만 넘겨서, 모델이 요약을 쓰려고 기사를 하나씩
   web_fetch 했다(6번, 그중 NYT는 403). 정작 피드엔 description이 이미 실려
   있었다 — NYT 138자, NPR 293자 — 파서가 지나치고 버리고 있었을 뿐이다.
   Headline에 summary를 넣고 formatHeadlines가 함께 내보낸다. 실측: us 피드
   12건 전부 요약 확보.

   NPR은 <em> 식으로 이중 인코딩해 보내므로 디코드→태그제거→디코드를
   한 번 더 돈다. 한 번만 돌면 <em>이 그대로 남는다.

2) 가드 두 개가 서로 물려 헛바퀴 3회를 돌았다. 모델이 category를 붙여 다시
   부르면 → breadth 가드가 "요청 안 한 category"라며 떼어냄 → 인자가 앞 호출과
   같아짐 → 중복으로 스킵 → 모델은 데이터를 못 받았으니 다른 category로 또 시도.

   중복 스킵에는 원래 replay 경로가 있는데 대상이 coder_list_files와
   coder_read_file 둘뿐이라, 뉴스·검색은 데이터 대신 "Already ran this exact
   call"이라는 문구만 받았다. 결과가 없으니 변형해서 또 부르는 게 당연하다.
   news_search/web_search/web_fetch를 replay 대상에 넣어 이전 결과를 그대로
   돌려준다 — 모델이 원하던 걸 받으면 반복이 끝난다.

테스트 8개 추가(요약 추출, Atom summary, 이중 인코딩, CDATA, 요약 없는 항목,
길이 제한, 출력 포함, 빈 줄 미발생) — 354개 통과.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-15 08:39:45 +09:00

146 lines
6.1 KiB
TypeScript

/**
* top-headlines — 1면 RSS 파싱
*
* 순수 함수만 테스트한다. 피드 수신은 네트워크에 의존하므로 다루지 않는다.
*
* 이 파서가 조용히 깨지면 "뉴스가 없는 날"처럼 보인다 — 남의 문서 구조에 기대는 코드라
* 언젠가 바뀐다. 저장된 조각으로 회귀를 잡는 게 이 테스트의 목적이다.
*/
import { test, describe } from 'node:test';
import assert from 'node:assert/strict';
import { parseFeedItems, hasTopHeadlineFeeds, formatHeadlines } from '../src/tools/top-headlines';
const RSS = `<?xml version="1.0"?><rss><channel>
<title>NYT &gt; Top Stories</title>
<item><title>Lisa Demuth Wins Republican Primary</title><link>https://ex.invalid/a</link></item>
<item><title><![CDATA[Rescuers scramble: 180 dead & counting]]></title><link>https://ex.invalid/b</link></item>
</channel></rss>`;
const ATOM = `<?xml version="1.0"?><feed>
<entry><title>Front page story</title><link rel="alternate" href="https://ex.invalid/c"/></entry>
</feed>`;
describe('parseFeedItems', () => {
test('RSS item에서 제목과 링크를 뽑는다', () => {
const items = parseFeedItems(RSS, 'nytimes.com');
assert.equal(items.length, 2);
assert.equal(items[0].title, 'Lisa Demuth Wins Republican Primary');
assert.equal(items[0].link, 'https://ex.invalid/a');
assert.equal(items[0].source, 'nytimes.com');
});
test('채널 제목은 기사로 세지 않는다', () => {
assert.ok(!parseFeedItems(RSS, 'x').some(i => i.title.includes('Top Stories')));
});
test('CDATA와 엔티티를 푼다', () => {
assert.equal(parseFeedItems(RSS, 'x')[1].title, 'Rescuers scramble: 180 dead & counting');
});
test('Atom entry도 읽는다 — link는 href 속성에 있다', () => {
const items = parseFeedItems(ATOM, 'x');
assert.equal(items[0].title, 'Front page story');
assert.equal(items[0].link, 'https://ex.invalid/c');
});
test('limit을 지킨다', () => {
assert.equal(parseFeedItems(RSS, 'x', 1).length, 1);
});
test('구조가 바뀌면 0건 — 호출부가 NewsData로 폴백할 신호', () => {
assert.deepEqual(parseFeedItems('<html>완전히 다른 문서</html>', 'x'), []);
assert.deepEqual(parseFeedItems('', 'x'), []);
});
});
describe('hasTopHeadlineFeeds', () => {
test('피드가 있는 나라', () => {
for (const c of ['us', 'gb', 'de', 'fr', 'it', 'es']) assert.equal(hasTopHeadlineFeeds(c), true, c);
});
test('다국가 목록은 하나라도 있으면 참', () => {
assert.equal(hasTopHeadlineFeeds('gb,de,fr,it,es'), true);
assert.equal(hasTopHeadlineFeeds('jp,de'), true);
});
test('피드가 없는 나라 — NewsData 경로로 간다', () => {
assert.equal(hasTopHeadlineFeeds('kr'), false);
assert.equal(hasTopHeadlineFeeds('jp'), false);
assert.equal(hasTopHeadlineFeeds(''), false);
});
});
describe('formatHeadlines', () => {
test('번호·링크·출처를 담는다', () => {
const out = formatHeadlines(parseFeedItems(RSS, 'nytimes.com'));
assert.match(out, /^\[1\] Lisa Demuth/);
assert.match(out, /Source: nytimes\.com/);
});
});
/**
* 요약 추출 (2026-08-15 추가)
*
* 제목만 넘기면 모델이 기사를 하나씩 web_fetch 하러 간다 — 실측: "오늘 미국 주요 뉴스" 한 턴에
* web_fetch 6번, 매번 22 tok/s 모델의 생성 라운드 하나씩. 정작 피드엔 description이 이미
* 실려 있었는데(NYT 138자, NPR 293자) 파서가 지나치고 버리고 있었다.
*/
describe('요약 추출', () => {
test('description을 뽑는다', () => {
const xml = `<rss><channel><item>
<title>제목</title><link>https://e.com/1</link>
<description>디에고 가르시아가 물류 허브가 되었다.</description>
</item></channel></rss>`;
assert.equal(parseFeedItems(xml, 'e.com')[0].summary, '디에고 가르시아가 물류 허브가 되었다.');
});
test('Atom의 summary도 뽑는다', () => {
const xml = `<feed><entry>
<title>T</title><link href="https://e.com/2"/><summary>요약본이다.</summary>
</entry></feed>`;
assert.equal(parseFeedItems(xml, 'e.com')[0].summary, '요약본이다.');
});
test('이중 인코딩된 마크업을 걷어낸다 — NPR이 &lt;em&gt;로 보내온다', () => {
const xml = `<rss><channel><item><title>T</title><link>u</link>
<description>Clint Smith&apos;s &lt;em&gt;How the Word Is Passed&lt;/em&gt; explores it.</description>
</item></channel></rss>`;
const s = parseFeedItems(xml, 'x')[0].summary;
assert.ok(!s.includes('<em>') && !s.includes('&lt;'), `태그가 남았다: ${s}`);
assert.ok(s.includes("Clint Smith's") && s.includes('How the Word Is Passed'));
});
test('CDATA 안의 요약도 뽑는다', () => {
const xml = `<rss><channel><item><title>T</title><link>u</link>
<description><![CDATA[<p>본문 요약</p>]]></description></item></channel></rss>`;
assert.equal(parseFeedItems(xml, 'x')[0].summary, '본문 요약');
});
test('요약이 없어도 기사 자체는 살린다', () => {
const xml = `<rss><channel><item><title>제목만</title><link>u</link></item></channel></rss>`;
const items = parseFeedItems(xml, 'x');
assert.equal(items.length, 1);
assert.equal(items[0].summary, '');
});
test('지나치게 긴 요약은 자른다', () => {
const xml = `<rss><channel><item><title>T</title><link>u</link>
<description>${'가'.repeat(2000)}</description></item></channel></rss>`;
assert.ok(parseFeedItems(xml, 'x')[0].summary.length <= 400);
});
test('formatHeadlines가 요약을 실어 보낸다', () => {
const xml = `<rss><channel><item><title>제목</title><link>https://e.com/1</link>
<description>핵심 요약</description></item></channel></rss>`;
const out = formatHeadlines(parseFeedItems(xml, 'e.com'));
assert.ok(out.includes('핵심 요약'), out);
assert.ok(out.includes('https://e.com/1'));
});
test('요약이 없으면 빈 줄을 넣지 않는다', () => {
const xml = `<rss><channel><item><title>제목만</title><link>u</link></item></channel></rss>`;
assert.ok(!/\n\s*\n/.test(formatHeadlines(parseFeedItems(xml, 'x'))));
});
});