feat: thinking 분리를 thinking-separation.ts로 추출 + GLM 영어 셀프토크 구조 기반 분리
- server.ts의 separateThinkingFromContent를 chat/thinking-separation.ts로 추출 (단위 테스트 가능하게)
- stripEnglishScratchPrefix 신설: 영어 런 40자+ + 문장부호 뒤 무공백 한글 봉합점(seam)을
구조로 탐지 — 키워드가 아니라 구조라 한글 조각이 섞인 영어 계획 단락도 처리
- 스크래치 증거 게이트(메타 라벨 또는 한글 없는 40자+ 단락 2개)로 오탐 방지,
정상 혼합문("Open-Meteo 모델은")은 보존
- think 태그 내용을 버리지 않고 thinking으로 보존하던 손실 수정
- 실측 표본 2건(09-22 추석 날씨 세션)을 단위 테스트로 고정, 6/6 통과
Co-Authored-By: Claude Code <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,133 @@
|
||||
// separateThinkingFromContent — 모델 응답에서 thinking을 분리한다.
|
||||
// 2026-09-22 server.ts에서 추출(단위 테스트를 위해). 실측 데이터: tests/thinking-separation.test.ts
|
||||
|
||||
|
||||
function separateThinkingFromContent(text: string): { reply: string; thinking: string } {
|
||||
if (!text) return { reply: '', thinking: '' };
|
||||
|
||||
// think 태그 안의 내용은 버리지 않고 thinking으로 보존한다(기존엔 그냥 삭제돼 손실).
|
||||
const thinkParts: string[] = [];
|
||||
const cleaned = text
|
||||
.replace(/<think[\s\S]*?<\/think>/gi, (mm) => { thinkParts.push(mm.replace(/<\/?think>/gi, '')); return ''; })
|
||||
.replace(/<think[\s\S]*/gi, (mm) => { thinkParts.push(mm.replace(/<think/gi, '')); return ''; })
|
||||
.replace(/<\/think>/gi, '')
|
||||
.trim();
|
||||
const captured = thinkParts.join('\n\n').trim();
|
||||
const base = classify(cleaned);
|
||||
return {
|
||||
reply: base.reply,
|
||||
thinking: [captured, base.thinking].filter(Boolean).join('\n\n'),
|
||||
};
|
||||
}
|
||||
|
||||
function classify(cleaned: string): { reply: string; thinking: string } {
|
||||
if (!cleaned) return { reply: '', thinking: '' };
|
||||
|
||||
// glm-5.3-flash 영어 셀프토크 접두사(구조 기반) 먼저 시도 — 키워드 휴리스틱에 안 잡히는 패턴
|
||||
const scratch = stripEnglishScratchPrefix(cleaned);
|
||||
if (scratch) {
|
||||
// 접두사 제거 후 남은 답변에 대해 기존 키워드 휴리스틱을 한 번 더 돌려 중첩 셀프토크도 처리
|
||||
const inner = separateThinkingFromContent(scratch.reply);
|
||||
return {
|
||||
reply: inner.reply,
|
||||
thinking: [scratch.thinking, inner.thinking].filter(Boolean).join('\n\n'),
|
||||
};
|
||||
}
|
||||
|
||||
// Fast-path: if the entire output looks like pure reasoning (starts with common
|
||||
// reasoning starters and is very long), treat the whole thing as thinking
|
||||
if (cleaned.length > 500 && /^(Okay|Ok,|Let me|First|Hmm|Wait|The user|I need|I should|So,)/i.test(cleaned)) {
|
||||
// Try to find the last sentence that looks like a real reply
|
||||
const sentences = cleaned.split(/(?<=[.!?])\s+/);
|
||||
let lastUseful: string | undefined;
|
||||
for (let i = sentences.length - 1; i >= 0; i--) {
|
||||
const s = sentences[i];
|
||||
if (s.length > 10 && s.length < 200 && !/\b(the user|I need to|I should|let me|wait,|hmm|the rules|the tools|the instructions)\b/i.test(s)) {
|
||||
lastUseful = s;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (lastUseful) {
|
||||
return { reply: lastUseful.trim(), thinking: cleaned };
|
||||
}
|
||||
return { reply: '', thinking: cleaned };
|
||||
}
|
||||
|
||||
const paragraphs = cleaned.split(/\n{2,}/).map(p => p.trim()).filter(Boolean);
|
||||
const reasoningRE = /\b(the user|the tools|the instructions|I need to|I should|let me|the problem|the question|the answer|looking at|first,|second,|wait,|hmm|the response|the correct|the assistant|check the rules|according to|the file|the current|the plan)\b/i;
|
||||
const starterRE = /^(Okay|Ok|Alright|Let me|First|Hmm|So,? |Wait|The user|Looking|I need|I should|Now,? |Since|Given|Based on|Check)/i;
|
||||
|
||||
let lastIdx = -1;
|
||||
for (let i = 0; i < paragraphs.length; i++) {
|
||||
if (reasoningRE.test(paragraphs[i]) || starterRE.test(paragraphs[i])) lastIdx = i;
|
||||
}
|
||||
|
||||
if (lastIdx === -1) return { reply: cleaned, thinking: '' };
|
||||
if (lastIdx >= paragraphs.length - 1) {
|
||||
const last = paragraphs[paragraphs.length - 1];
|
||||
const sentences = last.split(/(?<=[.!?])\s+/);
|
||||
for (let i = sentences.length - 1; i >= 0; i--) {
|
||||
if (!reasoningRE.test(sentences[i]) && sentences[i].length < 200) {
|
||||
return {
|
||||
reply: sentences.slice(i).join(' ').trim(),
|
||||
thinking: [...paragraphs.slice(0, -1), sentences.slice(0, i).join(' ')].join('\n\n').trim(),
|
||||
};
|
||||
}
|
||||
}
|
||||
return { reply: cleaned, thinking: '' };
|
||||
}
|
||||
|
||||
const reply = paragraphs.slice(lastIdx + 1).join('\n\n');
|
||||
const replyChars = reply.replace(/\s/g, '').length;
|
||||
if (replyChars < 10 && cleaned.length > reply.length) {
|
||||
return { reply: cleaned, thinking: '' };
|
||||
}
|
||||
|
||||
return {
|
||||
thinking: paragraphs.slice(0, lastIdx + 1).join('\n\n'),
|
||||
reply,
|
||||
};
|
||||
}
|
||||
|
||||
// glm-5.3-flash(2026-09-22부터 primary)가 답변 앞에 영어로 셀프토크(내부 규칙 인용,
|
||||
// "Key message:", "Answer format:" 등)를 붙였다가 한글 답변과 한 단락에서 무공백 융합하는
|
||||
// 실측 패턴(2026-09-22 "추석 비소식" 2건).
|
||||
//
|
||||
// 경계 판정은 키워드가 아니라 **구조**로 한다: 영어 런(ASCII만, 40자 이상)이 문장부호로 끝나고
|
||||
// 그 뒤에 공백 없이 한글이 바로 붙는 "봉합점(seam)"을 찾는다. 정상 텍스트는 영어 문장과 한글이
|
||||
// 무공백으로 붙는 일이 거의 없으므로("Open-Meteo 모델은" 같은 짧은 섞임은 런이 40자 미만이라 안 걸림),
|
||||
// 언어 혼합에 관계없이 한글 조각이 박힌 영어 계획 단락도 처리할 수 있다(실측 표본 2건 모두 여기 해당).
|
||||
//
|
||||
// 봉합점을 찾아도 바로 자르지 않고, 앞부분에 스크래치 증거가 있어야 발동한다 —
|
||||
// ① 이전 단락 중 메타 라벨("Answer format:" 등)로 시작하는 것, 또는
|
||||
// ② 한글 없는 40자 이상 단락이 2개 이상.
|
||||
// 둘 중 하나도 없으면 정상 응답으로 본다.
|
||||
const SEAM_RE = /([A-Za-z0-9 .,;:'"()\-]{40,}[.!?:][")]?)([가-힣])/g;
|
||||
const SCRATCH_LABEL_RE = /^(Also note|Also mention|Answer format|Key message|Key points|Keep it|Actually the rule|Note that|Remember that|Final note|One more thing)\b/i;
|
||||
|
||||
function stripEnglishScratchPrefix(text: string): { reply: string; thinking: string } | null {
|
||||
if (!text) return null;
|
||||
const HANGUL = /[가-힣]/;
|
||||
// 답변 언어가 한국어인 경우에만 적용한다(마지막 500자에 한글이 있어야 함).
|
||||
if (!HANGUL.test(text.slice(-500))) return null;
|
||||
|
||||
// 첫 번째 봉합점이 어금 뒤쪽에 진짜 경계가 있을 수 있다(예: 인용 속 영어 뒤 한글 조각).
|
||||
// 증거가 있는 첫 봉합점을 채택한다.
|
||||
for (const m of text.matchAll(SEAM_RE)) {
|
||||
const before = text.slice(0, m.index);
|
||||
const prior = before.split(/\n{2,}/).map(p => p.trim()).filter(Boolean);
|
||||
const hasEvidence =
|
||||
prior.some(p => SCRATCH_LABEL_RE.test(p)) ||
|
||||
prior.filter(p => p.length >= 40 && !HANGUL.test(p)).length >= 2;
|
||||
if (!hasEvidence) continue;
|
||||
|
||||
const hangulPos = m.index + m[1].length;
|
||||
const reply = text.slice(hangulPos);
|
||||
if (reply.replace(/\s/g, '').length < 40) continue; // 너무 잘려나가면 다음 후보로
|
||||
return { reply, thinking: before.trim() };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
|
||||
export { separateThinkingFromContent };
|
||||
+1
-67
@@ -72,6 +72,7 @@ import { createBuildTools } from './chat/build-tools';
|
||||
import { createExecuteTool, type ToolResult } from './chat/execute-tool';
|
||||
import { createHandleChat } from './chat/handle-chat';
|
||||
import { createPersonalityContext } from './chat/personality-context';
|
||||
import { separateThinkingFromContent } from './chat/thinking-separation';
|
||||
import {
|
||||
createHandleTaskControl,
|
||||
inferTaskChannelFromSession,
|
||||
@@ -1628,73 +1629,6 @@ function logToolCall(workspacePath: string, toolName: string, args: any, result:
|
||||
} catch {}
|
||||
}
|
||||
|
||||
|
||||
function separateThinkingFromContent(text: string): { reply: string; thinking: string } {
|
||||
if (!text) return { reply: '', thinking: '' };
|
||||
|
||||
let cleaned = text
|
||||
.replace(/<think>[\s\S]*?<\/think>/gi, '')
|
||||
.replace(/<think>[\s\S]*/gi, '')
|
||||
.replace(/<\/think>/gi, '')
|
||||
.trim();
|
||||
|
||||
if (!cleaned) return { reply: '', thinking: text };
|
||||
|
||||
// Fast-path: if the entire output looks like pure reasoning (starts with common
|
||||
// reasoning starters and is very long), treat the whole thing as thinking
|
||||
if (cleaned.length > 500 && /^(Okay|Ok,|Let me|First|Hmm|Wait|The user|I need|I should|So,)/i.test(cleaned)) {
|
||||
// Try to find the last sentence that looks like a real reply
|
||||
const sentences = cleaned.split(/(?<=[.!?])\s+/);
|
||||
let lastUseful: string | undefined;
|
||||
for (let i = sentences.length - 1; i >= 0; i--) {
|
||||
const s = sentences[i];
|
||||
if (s.length > 10 && s.length < 200 && !/\b(the user|I need to|I should|let me|wait,|hmm|the rules|the tools|the instructions)\b/i.test(s)) {
|
||||
lastUseful = s;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (lastUseful) {
|
||||
return { reply: lastUseful.trim(), thinking: cleaned };
|
||||
}
|
||||
return { reply: '', thinking: cleaned };
|
||||
}
|
||||
|
||||
const paragraphs = cleaned.split(/\n{2,}/).map(p => p.trim()).filter(Boolean);
|
||||
const reasoningRE = /\b(the user|the tools|the instructions|I need to|I should|let me|the problem|the question|the answer|looking at|first,|second,|wait,|hmm|the response|the correct|the assistant|check the rules|according to|the file|the current|the plan)\b/i;
|
||||
const starterRE = /^(Okay|Ok|Alright|Let me|First|Hmm|So,? |Wait|The user|Looking|I need|I should|Now,? |Since|Given|Based on|Check)/i;
|
||||
|
||||
let lastIdx = -1;
|
||||
for (let i = 0; i < paragraphs.length; i++) {
|
||||
if (reasoningRE.test(paragraphs[i]) || starterRE.test(paragraphs[i])) lastIdx = i;
|
||||
}
|
||||
|
||||
if (lastIdx === -1) return { reply: cleaned, thinking: '' };
|
||||
if (lastIdx >= paragraphs.length - 1) {
|
||||
const last = paragraphs[paragraphs.length - 1];
|
||||
const sentences = last.split(/(?<=[.!?])\s+/);
|
||||
for (let i = sentences.length - 1; i >= 0; i--) {
|
||||
if (!reasoningRE.test(sentences[i]) && sentences[i].length < 200) {
|
||||
return {
|
||||
reply: sentences.slice(i).join(' ').trim(),
|
||||
thinking: [...paragraphs.slice(0, -1), sentences.slice(0, i).join(' ')].join('\n\n').trim(),
|
||||
};
|
||||
}
|
||||
}
|
||||
return { reply: cleaned, thinking: '' };
|
||||
}
|
||||
|
||||
const reply = paragraphs.slice(lastIdx + 1).join('\n\n');
|
||||
const replyChars = reply.replace(/\s/g, '').length;
|
||||
if (replyChars < 10 && cleaned.length > reply.length) {
|
||||
return { reply: cleaned, thinking: '' };
|
||||
}
|
||||
|
||||
return {
|
||||
thinking: paragraphs.slice(0, lastIdx + 1).join('\n\n'),
|
||||
reply,
|
||||
};
|
||||
}
|
||||
|
||||
function normalizeForDedup(text: string): string {
|
||||
const raw = String(text || '').toLowerCase().trim();
|
||||
if (!raw) return '';
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
/**
|
||||
* separateThinkingFromContent — glm-5.3-flash 영어 셀프토크 접두사 분리
|
||||
*
|
||||
* 2026-09-22 실측: glm-5.3-flash:cloud(당시 primary)가 날씨 답변 앞에 영어로 내부 계획을 붙였다가
|
||||
* 마지막 단락에서 한글 답변과 무공백 융합("No emoji habit.아쉽게도…")해 통째로 노출됨.
|
||||
* 키워드 휴리스틱이 이 패턴을 못 잡아 구조 기반 stripEnglishScratchPrefix를 추가.
|
||||
* 표본 2건은 그날의 실제 세션 텍스트(09175258 세션)다.
|
||||
*/
|
||||
|
||||
import { test, describe } from 'node:test';
|
||||
import assert from 'node:assert/strict';
|
||||
import { separateThinkingFromContent } from '../src/gateway/chat/thinking-separation';
|
||||
|
||||
// ── 실측 표본 1: "추석에 비소식 없나?" 응답 ──
|
||||
const SAMPLE_1 = `Also note discrepancy: Open-Meteo showed Sep 24 as mostly clear/dry (0% precip prob through 22:00), while KMA says 30% for Sep 24. KMA is observation-backed and preferred for Korea. I'll lead with KMA numbers. The openmeteo hourly for Sep 24 showed 0% precip probability — slight disagreement, but KMA is the authoritative source for Korea. I'll present KMA as main, and can mention the trend.
|
||||
|
||||
Answer format: weather answers should be natural sentences, not forced tables. But a small table for the holiday period could help... The rules say for weather do NOT force a table — explain in natural fuller sentences. I'll write in Korean, conversational but informative.
|
||||
|
||||
Key message: 아쉽게도 비 소식이 있습니다. 추석 당일(9/25 금) 강수확률 60%, 비 예보. 전날(9/24 목)은 흐림 30%. 토요일(9/26)도 60% 비. 일요일(9/27)부터는 개어서 4%. 서울 기준이며, 지역 알려주면 다시 확인해주겠다고.
|
||||
|
||||
Also mention: 발표 시각 2026-09-22 23:00 KST 기준 단기예보라, 3일 뒤 예보라 변동 가능성 있음 — 추석 당일까지는 예보가 바뀔 수 있으니 며칠 뒤 다시 확인 권장.
|
||||
|
||||
Keep it reasonably concise but full sentences. No emoji habit.아쉽게도 비 소식이 있습니다. 기상청 단기예보(오늘 밤 11시 발표, 서울 기준)에 따르면 추석 연휴 후반부에 비가 예보돼요.
|
||||
|
||||
- **9/24 (목, 추석 전날)**: 흐림, 19~28°C, 강수확률 30% — 아직 비 예보는 아니고 흐려지는 날
|
||||
- **9/25 (금, 추석 당일)**: **비, 강수확률 60%**, 19~28°C — 당일은 비 소식 확실하게 잡혀 있음
|
||||
|
||||
즉 귀성길인 목요일까지는 큰 문제가 없지만, 추석 당일과 토요일 귀경길에 비를 만날 가능성이 높습니다.
|
||||
|
||||
서울 기준으로 봤는데, 다른 지역(고향)이 궁금하면 알려주세요.`;
|
||||
|
||||
// ── 실측 표본 2: "남부 지방은 어때?" 응답 ──
|
||||
const SAMPLE_2 = `Actually the rule: "When comparing several cities (e.g. '오늘 유럽 대도시 최고 기온'), a table is fine for the numbers themselves, but always follow it with a few sentences of actual commentary." So table + commentary is fine here.
|
||||
|
||||
Key points:
|
||||
- 남부 3대 도시(부산·대구·광주) 모두 비슷한 패턴
|
||||
- 추석 당일(9/25 금): 세 도시 모두 흐림, 강수확률 30% — 서울(비 60%)보다 하루 늦게 비가 옴
|
||||
|
||||
Keep it focused and natural. Korean language.남부 지방은 서울보다 비가 하루 늦게 옵니다. 부산·대구·광주 세 도시를 기상청 단기예보(오늘 밤 11시 발표)로 확인했는데, 패턴이 거의 같아요.
|
||||
|
||||
| 날짜 | 부산 | 대구 | 광주 |
|
||||
|------|------|------|------|
|
||||
| 9/26 (토) | **비 60%** | **비 60%** | **비 60%** |
|
||||
|
||||
역시 3일 뒤까지의 예보라 변동 가능성은 있으니, 토요일 출발 전에 다시 확인하시는 게 좋습니다.`;
|
||||
|
||||
describe('glm 영어 셀프토크 접두사 분리', () => {
|
||||
test('실측 표본 1 — 영어 계획 노출분을 thinking으로 떼어낸다', () => {
|
||||
const { reply, thinking } = separateThinkingFromContent(SAMPLE_1);
|
||||
assert.ok(!/Also note|Answer format|Also mention|Keep it reasonably/.test(reply),
|
||||
`reply에 영어 계획 잔여물이 남음: ${reply.slice(0, 120)}`);
|
||||
assert.ok(reply.startsWith('아쉽게도 비 소식이 있습니다'), `reply 시작이 한글 답변이 아님: ${reply.slice(0, 60)}`);
|
||||
assert.ok(/귀성길인 목요일/.test(reply), '답변 본문이 잘리면 안 됨');
|
||||
assert.ok(/Open-Meteo/.test(thinking), '분리된 thinking에 계획이 있어야 함');
|
||||
});
|
||||
|
||||
test('실측 표본 2 — "Korean language."뒤 무공백 융합을 끊는다', () => {
|
||||
const { reply, thinking } = separateThinkingFromContent(SAMPLE_2);
|
||||
assert.ok(!/Actually the rule|Keep it focused/.test(reply),
|
||||
`reply에 영어 계획 잔여물이 남음: ${reply.slice(0, 120)}`);
|
||||
assert.ok(reply.startsWith('남부 지방은 서울보다'), `reply 시작이 한글 답변이 아님: ${reply.slice(0, 60)}`);
|
||||
assert.ok(/부산·대구·광주 세 도시를 기상청/.test(reply), '융합 단락의 한글 본문이 보존돼야 함');
|
||||
assert.ok(/9\/26 \(토\)/.test(reply), '마크다운 표가 잘리면 안 됨');
|
||||
assert.ok(/Actually the rule/.test(thinking), '분리된 thinking에 계획이 있어야 함');
|
||||
});
|
||||
|
||||
test('정상 한글 답변은 건드리지 않는다', () => {
|
||||
const normal = '오늘 서울 날씨는 맑습니다. 최고 기온은 24도예요.\n\n- 아침 16도\n- 낮 24도\n\n즐거운 하루 보내세요.';
|
||||
const { reply, thinking } = separateThinkingFromContent(normal);
|
||||
assert.equal(thinking, '');
|
||||
assert.equal(reply, normal);
|
||||
});
|
||||
|
||||
test('영어 단어 뒤 바로 한글 오는 정상 혼합문은 보존된다', () => {
|
||||
const mixed = 'Open-Meteo 모델은 목요일을 맑은 날로 보고 있습니다.\n\n기상청 기준으로 정리하면 다음과 같습니다.\n\n- 9/24: 흐림 30%\n- 9/25: 비 60%\n\n출발 전에 다시 확인하세요.';
|
||||
const { reply, thinking } = separateThinkingFromContent(mixed);
|
||||
assert.equal(thinking, '');
|
||||
assert.ok(reply.startsWith('Open-Meteo'), '영어 단어로 시작하는 정상 답변이 잘리면 안 됨');
|
||||
});
|
||||
|
||||
test('짧은 영어 접두는 기존 키워드 휴리스틱이 걸러낸다', () => {
|
||||
const short = 'Let me summarize the forecast for you.\n\n답변입니다. 추석 당일 비 예보 60%입니다.\n\n추가로 토요일도 비 60%예요.\n\n일요일부터는 개어서 맑습니다.';
|
||||
const { reply, thinking } = separateThinkingFromContent(short);
|
||||
// "Let me" 단락은 starterRE/reasoningRE에 걸려 thinking으로 간다 — reply는 한글 답변만 남는다
|
||||
assert.ok(/Let me/.test(thinking), '영어 접두 단락이 thinking으로 가야 함');
|
||||
assert.ok(reply.startsWith('답변입니다'), `reply가 한글 답변이어야 함: ${reply.slice(0, 60)}`);
|
||||
});
|
||||
|
||||
test('기존 <think> 태그 방식은 그대로 작동한다', () => {
|
||||
const tagged = '<think>사용자가 추석 비를 물었다. KMA를 보자.</think>아쉽게도 추석 당일 비 예보가 있습니다. 강수확률 60%예요.';
|
||||
const { reply, thinking } = separateThinkingFromContent(tagged);
|
||||
assert.ok(reply.startsWith('아쉽게도'), `<think> 태그가 안 떼어짐: ${reply.slice(0, 60)}`);
|
||||
assert.ok(/추석 비를 물었다/.test(thinking), 'think 내용이 thinking으로 가야 함');
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user