feat: thinking 분리를 thinking-separation.ts로 추출 + GLM 영어 셀프토크 구조 기반 분리

- server.ts의 separateThinkingFromContent를 chat/thinking-separation.ts로 추출 (단위 테스트 가능하게)
- stripEnglishScratchPrefix 신설: 영어 런 40자+ + 문장부호 뒤 무공백 한글 봉합점(seam)을
  구조로 탐지 — 키워드가 아니라 구조라 한글 조각이 섞인 영어 계획 단락도 처리
- 스크래치 증거 게이트(메타 라벨 또는 한글 없는 40자+ 단락 2개)로 오탐 방지,
  정상 혼합문("Open-Meteo 모델은")은 보존
- think 태그 내용을 버리지 않고 thinking으로 보존하던 손실 수정
- 실측 표본 2건(09-22 추석 날씨 세션)을 단위 테스트로 고정, 6/6 통과

Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
kim
2026-09-23 00:41:49 +09:00
co-authored by Claude Code
parent fe934cbd36
commit f8d1c9f9df
3 changed files with 229 additions and 67 deletions
+133
View File
@@ -0,0 +1,133 @@
// separateThinkingFromContent — 모델 응답에서 thinking을 분리한다.
// 2026-09-22 server.ts에서 추출(단위 테스트를 위해). 실측 데이터: tests/thinking-separation.test.ts
function separateThinkingFromContent(text: string): { reply: string; thinking: string } {
if (!text) return { reply: '', thinking: '' };
// think 태그 안의 내용은 버리지 않고 thinking으로 보존한다(기존엔 그냥 삭제돼 손실).
const thinkParts: string[] = [];
const cleaned = text
.replace(/<think[\s\S]*?<\/think>/gi, (mm) => { thinkParts.push(mm.replace(/<\/?think>/gi, '')); return ''; })
.replace(/<think[\s\S]*/gi, (mm) => { thinkParts.push(mm.replace(/<think/gi, '')); return ''; })
.replace(/<\/think>/gi, '')
.trim();
const captured = thinkParts.join('\n\n').trim();
const base = classify(cleaned);
return {
reply: base.reply,
thinking: [captured, base.thinking].filter(Boolean).join('\n\n'),
};
}
function classify(cleaned: string): { reply: string; thinking: string } {
if (!cleaned) return { reply: '', thinking: '' };
// glm-5.3-flash 영어 셀프토크 접두사(구조 기반) 먼저 시도 — 키워드 휴리스틱에 안 잡히는 패턴
const scratch = stripEnglishScratchPrefix(cleaned);
if (scratch) {
// 접두사 제거 후 남은 답변에 대해 기존 키워드 휴리스틱을 한 번 더 돌려 중첩 셀프토크도 처리
const inner = separateThinkingFromContent(scratch.reply);
return {
reply: inner.reply,
thinking: [scratch.thinking, inner.thinking].filter(Boolean).join('\n\n'),
};
}
// Fast-path: if the entire output looks like pure reasoning (starts with common
// reasoning starters and is very long), treat the whole thing as thinking
if (cleaned.length > 500 && /^(Okay|Ok,|Let me|First|Hmm|Wait|The user|I need|I should|So,)/i.test(cleaned)) {
// Try to find the last sentence that looks like a real reply
const sentences = cleaned.split(/(?<=[.!?])\s+/);
let lastUseful: string | undefined;
for (let i = sentences.length - 1; i >= 0; i--) {
const s = sentences[i];
if (s.length > 10 && s.length < 200 && !/\b(the user|I need to|I should|let me|wait,|hmm|the rules|the tools|the instructions)\b/i.test(s)) {
lastUseful = s;
break;
}
}
if (lastUseful) {
return { reply: lastUseful.trim(), thinking: cleaned };
}
return { reply: '', thinking: cleaned };
}
const paragraphs = cleaned.split(/\n{2,}/).map(p => p.trim()).filter(Boolean);
const reasoningRE = /\b(the user|the tools|the instructions|I need to|I should|let me|the problem|the question|the answer|looking at|first,|second,|wait,|hmm|the response|the correct|the assistant|check the rules|according to|the file|the current|the plan)\b/i;
const starterRE = /^(Okay|Ok|Alright|Let me|First|Hmm|So,? |Wait|The user|Looking|I need|I should|Now,? |Since|Given|Based on|Check)/i;
let lastIdx = -1;
for (let i = 0; i < paragraphs.length; i++) {
if (reasoningRE.test(paragraphs[i]) || starterRE.test(paragraphs[i])) lastIdx = i;
}
if (lastIdx === -1) return { reply: cleaned, thinking: '' };
if (lastIdx >= paragraphs.length - 1) {
const last = paragraphs[paragraphs.length - 1];
const sentences = last.split(/(?<=[.!?])\s+/);
for (let i = sentences.length - 1; i >= 0; i--) {
if (!reasoningRE.test(sentences[i]) && sentences[i].length < 200) {
return {
reply: sentences.slice(i).join(' ').trim(),
thinking: [...paragraphs.slice(0, -1), sentences.slice(0, i).join(' ')].join('\n\n').trim(),
};
}
}
return { reply: cleaned, thinking: '' };
}
const reply = paragraphs.slice(lastIdx + 1).join('\n\n');
const replyChars = reply.replace(/\s/g, '').length;
if (replyChars < 10 && cleaned.length > reply.length) {
return { reply: cleaned, thinking: '' };
}
return {
thinking: paragraphs.slice(0, lastIdx + 1).join('\n\n'),
reply,
};
}
// glm-5.3-flash(2026-09-22부터 primary)가 답변 앞에 영어로 셀프토크(내부 규칙 인용,
// "Key message:", "Answer format:" 등)를 붙였다가 한글 답변과 한 단락에서 무공백 융합하는
// 실측 패턴(2026-09-22 "추석 비소식" 2건).
//
// 경계 판정은 키워드가 아니라 **구조**로 한다: 영어 런(ASCII만, 40자 이상)이 문장부호로 끝나고
// 그 뒤에 공백 없이 한글이 바로 붙는 "봉합점(seam)"을 찾는다. 정상 텍스트는 영어 문장과 한글이
// 무공백으로 붙는 일이 거의 없으므로("Open-Meteo 모델은" 같은 짧은 섞임은 런이 40자 미만이라 안 걸림),
// 언어 혼합에 관계없이 한글 조각이 박힌 영어 계획 단락도 처리할 수 있다(실측 표본 2건 모두 여기 해당).
//
// 봉합점을 찾아도 바로 자르지 않고, 앞부분에 스크래치 증거가 있어야 발동한다 —
// ① 이전 단락 중 메타 라벨("Answer format:" 등)로 시작하는 것, 또는
// ② 한글 없는 40자 이상 단락이 2개 이상.
// 둘 중 하나도 없으면 정상 응답으로 본다.
const SEAM_RE = /([A-Za-z0-9 .,;:'"()\-]{40,}[.!?:][")]?)([가-힣])/g;
const SCRATCH_LABEL_RE = /^(Also note|Also mention|Answer format|Key message|Key points|Keep it|Actually the rule|Note that|Remember that|Final note|One more thing)\b/i;
function stripEnglishScratchPrefix(text: string): { reply: string; thinking: string } | null {
if (!text) return null;
const HANGUL = /[가-힣]/;
// 답변 언어가 한국어인 경우에만 적용한다(마지막 500자에 한글이 있어야 함).
if (!HANGUL.test(text.slice(-500))) return null;
// 첫 번째 봉합점이 어금 뒤쪽에 진짜 경계가 있을 수 있다(예: 인용 속 영어 뒤 한글 조각).
// 증거가 있는 첫 봉합점을 채택한다.
for (const m of text.matchAll(SEAM_RE)) {
const before = text.slice(0, m.index);
const prior = before.split(/\n{2,}/).map(p => p.trim()).filter(Boolean);
const hasEvidence =
prior.some(p => SCRATCH_LABEL_RE.test(p)) ||
prior.filter(p => p.length >= 40 && !HANGUL.test(p)).length >= 2;
if (!hasEvidence) continue;
const hangulPos = m.index + m[1].length;
const reply = text.slice(hangulPos);
if (reply.replace(/\s/g, '').length < 40) continue; // 너무 잘려나가면 다음 후보로
return { reply, thinking: before.trim() };
}
return null;
}
export { separateThinkingFromContent };
+1 -67
View File
@@ -72,6 +72,7 @@ import { createBuildTools } from './chat/build-tools';
import { createExecuteTool, type ToolResult } from './chat/execute-tool';
import { createHandleChat } from './chat/handle-chat';
import { createPersonalityContext } from './chat/personality-context';
import { separateThinkingFromContent } from './chat/thinking-separation';
import {
createHandleTaskControl,
inferTaskChannelFromSession,
@@ -1628,73 +1629,6 @@ function logToolCall(workspacePath: string, toolName: string, args: any, result:
} catch {}
}
function separateThinkingFromContent(text: string): { reply: string; thinking: string } {
if (!text) return { reply: '', thinking: '' };
let cleaned = text
.replace(/<think>[\s\S]*?<\/think>/gi, '')
.replace(/<think>[\s\S]*/gi, '')
.replace(/<\/think>/gi, '')
.trim();
if (!cleaned) return { reply: '', thinking: text };
// Fast-path: if the entire output looks like pure reasoning (starts with common
// reasoning starters and is very long), treat the whole thing as thinking
if (cleaned.length > 500 && /^(Okay|Ok,|Let me|First|Hmm|Wait|The user|I need|I should|So,)/i.test(cleaned)) {
// Try to find the last sentence that looks like a real reply
const sentences = cleaned.split(/(?<=[.!?])\s+/);
let lastUseful: string | undefined;
for (let i = sentences.length - 1; i >= 0; i--) {
const s = sentences[i];
if (s.length > 10 && s.length < 200 && !/\b(the user|I need to|I should|let me|wait,|hmm|the rules|the tools|the instructions)\b/i.test(s)) {
lastUseful = s;
break;
}
}
if (lastUseful) {
return { reply: lastUseful.trim(), thinking: cleaned };
}
return { reply: '', thinking: cleaned };
}
const paragraphs = cleaned.split(/\n{2,}/).map(p => p.trim()).filter(Boolean);
const reasoningRE = /\b(the user|the tools|the instructions|I need to|I should|let me|the problem|the question|the answer|looking at|first,|second,|wait,|hmm|the response|the correct|the assistant|check the rules|according to|the file|the current|the plan)\b/i;
const starterRE = /^(Okay|Ok|Alright|Let me|First|Hmm|So,? |Wait|The user|Looking|I need|I should|Now,? |Since|Given|Based on|Check)/i;
let lastIdx = -1;
for (let i = 0; i < paragraphs.length; i++) {
if (reasoningRE.test(paragraphs[i]) || starterRE.test(paragraphs[i])) lastIdx = i;
}
if (lastIdx === -1) return { reply: cleaned, thinking: '' };
if (lastIdx >= paragraphs.length - 1) {
const last = paragraphs[paragraphs.length - 1];
const sentences = last.split(/(?<=[.!?])\s+/);
for (let i = sentences.length - 1; i >= 0; i--) {
if (!reasoningRE.test(sentences[i]) && sentences[i].length < 200) {
return {
reply: sentences.slice(i).join(' ').trim(),
thinking: [...paragraphs.slice(0, -1), sentences.slice(0, i).join(' ')].join('\n\n').trim(),
};
}
}
return { reply: cleaned, thinking: '' };
}
const reply = paragraphs.slice(lastIdx + 1).join('\n\n');
const replyChars = reply.replace(/\s/g, '').length;
if (replyChars < 10 && cleaned.length > reply.length) {
return { reply: cleaned, thinking: '' };
}
return {
thinking: paragraphs.slice(0, lastIdx + 1).join('\n\n'),
reply,
};
}
function normalizeForDedup(text: string): string {
const raw = String(text || '').toLowerCase().trim();
if (!raw) return '';
+95
View File
@@ -0,0 +1,95 @@
/**
* separateThinkingFromContent — glm-5.3-flash 영어 셀프토크 접두사 분리
*
* 2026-09-22 실측: glm-5.3-flash:cloud(당시 primary)가 날씨 답변 앞에 영어로 내부 계획을 붙였다가
* 마지막 단락에서 한글 답변과 무공백 융합("No emoji habit.아쉽게도…")해 통째로 노출됨.
* 키워드 휴리스틱이 이 패턴을 못 잡아 구조 기반 stripEnglishScratchPrefix를 추가.
* 표본 2건은 그날의 실제 세션 텍스트(09175258 세션)다.
*/
import { test, describe } from 'node:test';
import assert from 'node:assert/strict';
import { separateThinkingFromContent } from '../src/gateway/chat/thinking-separation';
// ── 실측 표본 1: "추석에 비소식 없나?" 응답 ──
const SAMPLE_1 = `Also note discrepancy: Open-Meteo showed Sep 24 as mostly clear/dry (0% precip prob through 22:00), while KMA says 30% for Sep 24. KMA is observation-backed and preferred for Korea. I'll lead with KMA numbers. The openmeteo hourly for Sep 24 showed 0% precip probability — slight disagreement, but KMA is the authoritative source for Korea. I'll present KMA as main, and can mention the trend.
Answer format: weather answers should be natural sentences, not forced tables. But a small table for the holiday period could help... The rules say for weather do NOT force a table — explain in natural fuller sentences. I'll write in Korean, conversational but informative.
Key message: 아쉽게도 비 소식이 있습니다. 추석 당일(9/25 금) 강수확률 60%, 비 예보. 전날(9/24 목)은 흐림 30%. 토요일(9/26)도 60% 비. 일요일(9/27)부터는 개어서 4%. 서울 기준이며, 지역 알려주면 다시 확인해주겠다고.
Also mention: 발표 시각 2026-09-22 23:00 KST 기준 단기예보라, 3일 뒤 예보라 변동 가능성 있음 — 추석 당일까지는 예보가 바뀔 수 있으니 며칠 뒤 다시 확인 권장.
Keep it reasonably concise but full sentences. No emoji habit.아쉽게도 비 소식이 있습니다. 기상청 단기예보(오늘 밤 11시 발표, 서울 기준)에 따르면 추석 연휴 후반부에 비가 예보돼요.
- **9/24 (목, 추석 전날)**: 흐림, 19~28°C, 강수확률 30% — 아직 비 예보는 아니고 흐려지는 날
- **9/25 (금, 추석 당일)**: **비, 강수확률 60%**, 19~28°C — 당일은 비 소식 확실하게 잡혀 있음
즉 귀성길인 목요일까지는 큰 문제가 없지만, 추석 당일과 토요일 귀경길에 비를 만날 가능성이 높습니다.
서울 기준으로 봤는데, 다른 지역(고향)이 궁금하면 알려주세요.`;
// ── 실측 표본 2: "남부 지방은 어때?" 응답 ──
const SAMPLE_2 = `Actually the rule: "When comparing several cities (e.g. '오늘 유럽 대도시 최고 기온'), a table is fine for the numbers themselves, but always follow it with a few sentences of actual commentary." So table + commentary is fine here.
Key points:
- 남부 3대 도시(부산·대구·광주) 모두 비슷한 패턴
- 추석 당일(9/25 금): 세 도시 모두 흐림, 강수확률 30% — 서울(비 60%)보다 하루 늦게 비가 옴
Keep it focused and natural. Korean language.남부 지방은 서울보다 비가 하루 늦게 옵니다. 부산·대구·광주 세 도시를 기상청 단기예보(오늘 밤 11시 발표)로 확인했는데, 패턴이 거의 같아요.
| 날짜 | 부산 | 대구 | 광주 |
|------|------|------|------|
| 9/26 (토) | **비 60%** | **비 60%** | **비 60%** |
역시 3일 뒤까지의 예보라 변동 가능성은 있으니, 토요일 출발 전에 다시 확인하시는 게 좋습니다.`;
describe('glm 영어 셀프토크 접두사 분리', () => {
test('실측 표본 1 — 영어 계획 노출분을 thinking으로 떼어낸다', () => {
const { reply, thinking } = separateThinkingFromContent(SAMPLE_1);
assert.ok(!/Also note|Answer format|Also mention|Keep it reasonably/.test(reply),
`reply에 영어 계획 잔여물이 남음: ${reply.slice(0, 120)}`);
assert.ok(reply.startsWith('아쉽게도 비 소식이 있습니다'), `reply 시작이 한글 답변이 아님: ${reply.slice(0, 60)}`);
assert.ok(/귀성길인 목요일/.test(reply), '답변 본문이 잘리면 안 됨');
assert.ok(/Open-Meteo/.test(thinking), '분리된 thinking에 계획이 있어야 함');
});
test('실측 표본 2 — "Korean language."뒤 무공백 융합을 끊는다', () => {
const { reply, thinking } = separateThinkingFromContent(SAMPLE_2);
assert.ok(!/Actually the rule|Keep it focused/.test(reply),
`reply에 영어 계획 잔여물이 남음: ${reply.slice(0, 120)}`);
assert.ok(reply.startsWith('남부 지방은 서울보다'), `reply 시작이 한글 답변이 아님: ${reply.slice(0, 60)}`);
assert.ok(/부산·대구·광주 세 도시를 기상청/.test(reply), '융합 단락의 한글 본문이 보존돼야 함');
assert.ok(/9\/26 \(토\)/.test(reply), '마크다운 표가 잘리면 안 됨');
assert.ok(/Actually the rule/.test(thinking), '분리된 thinking에 계획이 있어야 함');
});
test('정상 한글 답변은 건드리지 않는다', () => {
const normal = '오늘 서울 날씨는 맑습니다. 최고 기온은 24도예요.\n\n- 아침 16도\n- 낮 24도\n\n즐거운 하루 보내세요.';
const { reply, thinking } = separateThinkingFromContent(normal);
assert.equal(thinking, '');
assert.equal(reply, normal);
});
test('영어 단어 뒤 바로 한글 오는 정상 혼합문은 보존된다', () => {
const mixed = 'Open-Meteo 모델은 목요일을 맑은 날로 보고 있습니다.\n\n기상청 기준으로 정리하면 다음과 같습니다.\n\n- 9/24: 흐림 30%\n- 9/25: 비 60%\n\n출발 전에 다시 확인하세요.';
const { reply, thinking } = separateThinkingFromContent(mixed);
assert.equal(thinking, '');
assert.ok(reply.startsWith('Open-Meteo'), '영어 단어로 시작하는 정상 답변이 잘리면 안 됨');
});
test('짧은 영어 접두는 기존 키워드 휴리스틱이 걸러낸다', () => {
const short = 'Let me summarize the forecast for you.\n\n답변입니다. 추석 당일 비 예보 60%입니다.\n\n추가로 토요일도 비 60%예요.\n\n일요일부터는 개어서 맑습니다.';
const { reply, thinking } = separateThinkingFromContent(short);
// "Let me" 단락은 starterRE/reasoningRE에 걸려 thinking으로 간다 — reply는 한글 답변만 남는다
assert.ok(/Let me/.test(thinking), '영어 접두 단락이 thinking으로 가야 함');
assert.ok(reply.startsWith('답변입니다'), `reply가 한글 답변이어야 함: ${reply.slice(0, 60)}`);
});
test('기존 <think> 태그 방식은 그대로 작동한다', () => {
const tagged = '<think>사용자가 추석 비를 물었다. KMA를 보자.</think>아쉽게도 추석 당일 비 예보가 있습니다. 강수확률 60%예요.';
const { reply, thinking } = separateThinkingFromContent(tagged);
assert.ok(reply.startsWith('아쉽게도'), `<think> 태그가 안 떼어짐: ${reply.slice(0, 60)}`);
assert.ok(/추석 비를 물었다/.test(thinking), 'think 내용이 thinking으로 가야 함');
});
});