diff --git a/src/gateway/chat/thinking-separation.ts b/src/gateway/chat/thinking-separation.ts new file mode 100644 index 0000000..2fd9e92 --- /dev/null +++ b/src/gateway/chat/thinking-separation.ts @@ -0,0 +1,133 @@ +// separateThinkingFromContent — 모델 응답에서 thinking을 분리한다. +// 2026-09-22 server.ts에서 추출(단위 테스트를 위해). 실측 데이터: tests/thinking-separation.test.ts + + +function separateThinkingFromContent(text: string): { reply: string; thinking: string } { + if (!text) return { reply: '', thinking: '' }; + + // think 태그 안의 내용은 버리지 않고 thinking으로 보존한다(기존엔 그냥 삭제돼 손실). + const thinkParts: string[] = []; + const cleaned = text + .replace(//gi, (mm) => { thinkParts.push(mm.replace(/<\/?think>/gi, '')); return ''; }) + .replace(/ { thinkParts.push(mm.replace(//gi, '') + .trim(); + const captured = thinkParts.join('\n\n').trim(); + const base = classify(cleaned); + return { + reply: base.reply, + thinking: [captured, base.thinking].filter(Boolean).join('\n\n'), + }; +} + +function classify(cleaned: string): { reply: string; thinking: string } { + if (!cleaned) return { reply: '', thinking: '' }; + + // glm-5.3-flash 영어 셀프토크 접두사(구조 기반) 먼저 시도 — 키워드 휴리스틱에 안 잡히는 패턴 + const scratch = stripEnglishScratchPrefix(cleaned); + if (scratch) { + // 접두사 제거 후 남은 답변에 대해 기존 키워드 휴리스틱을 한 번 더 돌려 중첩 셀프토크도 처리 + const inner = separateThinkingFromContent(scratch.reply); + return { + reply: inner.reply, + thinking: [scratch.thinking, inner.thinking].filter(Boolean).join('\n\n'), + }; + } + + // Fast-path: if the entire output looks like pure reasoning (starts with common + // reasoning starters and is very long), treat the whole thing as thinking + if (cleaned.length > 500 && /^(Okay|Ok,|Let me|First|Hmm|Wait|The user|I need|I should|So,)/i.test(cleaned)) { + // Try to find the last sentence that looks like a real reply + const sentences = cleaned.split(/(?<=[.!?])\s+/); + let lastUseful: string | undefined; + for (let i = sentences.length - 1; i >= 0; i--) { + const s = sentences[i]; + if (s.length > 10 && s.length < 200 && !/\b(the user|I need to|I should|let me|wait,|hmm|the rules|the tools|the instructions)\b/i.test(s)) { + lastUseful = s; + break; + } + } + if (lastUseful) { + return { reply: lastUseful.trim(), thinking: cleaned }; + } + return { reply: '', thinking: cleaned }; + } + + const paragraphs = cleaned.split(/\n{2,}/).map(p => p.trim()).filter(Boolean); + const reasoningRE = /\b(the user|the tools|the instructions|I need to|I should|let me|the problem|the question|the answer|looking at|first,|second,|wait,|hmm|the response|the correct|the assistant|check the rules|according to|the file|the current|the plan)\b/i; + const starterRE = /^(Okay|Ok|Alright|Let me|First|Hmm|So,? |Wait|The user|Looking|I need|I should|Now,? |Since|Given|Based on|Check)/i; + + let lastIdx = -1; + for (let i = 0; i < paragraphs.length; i++) { + if (reasoningRE.test(paragraphs[i]) || starterRE.test(paragraphs[i])) lastIdx = i; + } + + if (lastIdx === -1) return { reply: cleaned, thinking: '' }; + if (lastIdx >= paragraphs.length - 1) { + const last = paragraphs[paragraphs.length - 1]; + const sentences = last.split(/(?<=[.!?])\s+/); + for (let i = sentences.length - 1; i >= 0; i--) { + if (!reasoningRE.test(sentences[i]) && sentences[i].length < 200) { + return { + reply: sentences.slice(i).join(' ').trim(), + thinking: [...paragraphs.slice(0, -1), sentences.slice(0, i).join(' ')].join('\n\n').trim(), + }; + } + } + return { reply: cleaned, thinking: '' }; + } + + const reply = paragraphs.slice(lastIdx + 1).join('\n\n'); + const replyChars = reply.replace(/\s/g, '').length; + if (replyChars < 10 && cleaned.length > reply.length) { + return { reply: cleaned, thinking: '' }; + } + + return { + thinking: paragraphs.slice(0, lastIdx + 1).join('\n\n'), + reply, + }; +} + +// glm-5.3-flash(2026-09-22부터 primary)가 답변 앞에 영어로 셀프토크(내부 규칙 인용, +// "Key message:", "Answer format:" 등)를 붙였다가 한글 답변과 한 단락에서 무공백 융합하는 +// 실측 패턴(2026-09-22 "추석 비소식" 2건). +// +// 경계 판정은 키워드가 아니라 **구조**로 한다: 영어 런(ASCII만, 40자 이상)이 문장부호로 끝나고 +// 그 뒤에 공백 없이 한글이 바로 붙는 "봉합점(seam)"을 찾는다. 정상 텍스트는 영어 문장과 한글이 +// 무공백으로 붙는 일이 거의 없으므로("Open-Meteo 모델은" 같은 짧은 섞임은 런이 40자 미만이라 안 걸림), +// 언어 혼합에 관계없이 한글 조각이 박힌 영어 계획 단락도 처리할 수 있다(실측 표본 2건 모두 여기 해당). +// +// 봉합점을 찾아도 바로 자르지 않고, 앞부분에 스크래치 증거가 있어야 발동한다 — +// ① 이전 단락 중 메타 라벨("Answer format:" 등)로 시작하는 것, 또는 +// ② 한글 없는 40자 이상 단락이 2개 이상. +// 둘 중 하나도 없으면 정상 응답으로 본다. +const SEAM_RE = /([A-Za-z0-9 .,;:'"()\-]{40,}[.!?:][")]?)([가-힣])/g; +const SCRATCH_LABEL_RE = /^(Also note|Also mention|Answer format|Key message|Key points|Keep it|Actually the rule|Note that|Remember that|Final note|One more thing)\b/i; + +function stripEnglishScratchPrefix(text: string): { reply: string; thinking: string } | null { + if (!text) return null; + const HANGUL = /[가-힣]/; + // 답변 언어가 한국어인 경우에만 적용한다(마지막 500자에 한글이 있어야 함). + if (!HANGUL.test(text.slice(-500))) return null; + + // 첫 번째 봉합점이 어금 뒤쪽에 진짜 경계가 있을 수 있다(예: 인용 속 영어 뒤 한글 조각). + // 증거가 있는 첫 봉합점을 채택한다. + for (const m of text.matchAll(SEAM_RE)) { + const before = text.slice(0, m.index); + const prior = before.split(/\n{2,}/).map(p => p.trim()).filter(Boolean); + const hasEvidence = + prior.some(p => SCRATCH_LABEL_RE.test(p)) || + prior.filter(p => p.length >= 40 && !HANGUL.test(p)).length >= 2; + if (!hasEvidence) continue; + + const hangulPos = m.index + m[1].length; + const reply = text.slice(hangulPos); + if (reply.replace(/\s/g, '').length < 40) continue; // 너무 잘려나가면 다음 후보로 + return { reply, thinking: before.trim() }; + } + return null; +} + + +export { separateThinkingFromContent }; \ No newline at end of file diff --git a/src/gateway/server.ts b/src/gateway/server.ts index 5c61fa4..3dd2feb 100644 --- a/src/gateway/server.ts +++ b/src/gateway/server.ts @@ -72,6 +72,7 @@ import { createBuildTools } from './chat/build-tools'; import { createExecuteTool, type ToolResult } from './chat/execute-tool'; import { createHandleChat } from './chat/handle-chat'; import { createPersonalityContext } from './chat/personality-context'; +import { separateThinkingFromContent } from './chat/thinking-separation'; import { createHandleTaskControl, inferTaskChannelFromSession, @@ -1628,73 +1629,6 @@ function logToolCall(workspacePath: string, toolName: string, args: any, result: } catch {} } - -function separateThinkingFromContent(text: string): { reply: string; thinking: string } { - if (!text) return { reply: '', thinking: '' }; - - let cleaned = text - .replace(/[\s\S]*?<\/think>/gi, '') - .replace(/[\s\S]*/gi, '') - .replace(/<\/think>/gi, '') - .trim(); - - if (!cleaned) return { reply: '', thinking: text }; - - // Fast-path: if the entire output looks like pure reasoning (starts with common - // reasoning starters and is very long), treat the whole thing as thinking - if (cleaned.length > 500 && /^(Okay|Ok,|Let me|First|Hmm|Wait|The user|I need|I should|So,)/i.test(cleaned)) { - // Try to find the last sentence that looks like a real reply - const sentences = cleaned.split(/(?<=[.!?])\s+/); - let lastUseful: string | undefined; - for (let i = sentences.length - 1; i >= 0; i--) { - const s = sentences[i]; - if (s.length > 10 && s.length < 200 && !/\b(the user|I need to|I should|let me|wait,|hmm|the rules|the tools|the instructions)\b/i.test(s)) { - lastUseful = s; - break; - } - } - if (lastUseful) { - return { reply: lastUseful.trim(), thinking: cleaned }; - } - return { reply: '', thinking: cleaned }; - } - - const paragraphs = cleaned.split(/\n{2,}/).map(p => p.trim()).filter(Boolean); - const reasoningRE = /\b(the user|the tools|the instructions|I need to|I should|let me|the problem|the question|the answer|looking at|first,|second,|wait,|hmm|the response|the correct|the assistant|check the rules|according to|the file|the current|the plan)\b/i; - const starterRE = /^(Okay|Ok|Alright|Let me|First|Hmm|So,? |Wait|The user|Looking|I need|I should|Now,? |Since|Given|Based on|Check)/i; - - let lastIdx = -1; - for (let i = 0; i < paragraphs.length; i++) { - if (reasoningRE.test(paragraphs[i]) || starterRE.test(paragraphs[i])) lastIdx = i; - } - - if (lastIdx === -1) return { reply: cleaned, thinking: '' }; - if (lastIdx >= paragraphs.length - 1) { - const last = paragraphs[paragraphs.length - 1]; - const sentences = last.split(/(?<=[.!?])\s+/); - for (let i = sentences.length - 1; i >= 0; i--) { - if (!reasoningRE.test(sentences[i]) && sentences[i].length < 200) { - return { - reply: sentences.slice(i).join(' ').trim(), - thinking: [...paragraphs.slice(0, -1), sentences.slice(0, i).join(' ')].join('\n\n').trim(), - }; - } - } - return { reply: cleaned, thinking: '' }; - } - - const reply = paragraphs.slice(lastIdx + 1).join('\n\n'); - const replyChars = reply.replace(/\s/g, '').length; - if (replyChars < 10 && cleaned.length > reply.length) { - return { reply: cleaned, thinking: '' }; - } - - return { - thinking: paragraphs.slice(0, lastIdx + 1).join('\n\n'), - reply, - }; -} - function normalizeForDedup(text: string): string { const raw = String(text || '').toLowerCase().trim(); if (!raw) return ''; diff --git a/tests/thinking-separation.test.ts b/tests/thinking-separation.test.ts new file mode 100644 index 0000000..aafd7f6 --- /dev/null +++ b/tests/thinking-separation.test.ts @@ -0,0 +1,95 @@ +/** + * separateThinkingFromContent — glm-5.3-flash 영어 셀프토크 접두사 분리 + * + * 2026-09-22 실측: glm-5.3-flash:cloud(당시 primary)가 날씨 답변 앞에 영어로 내부 계획을 붙였다가 + * 마지막 단락에서 한글 답변과 무공백 융합("No emoji habit.아쉽게도…")해 통째로 노출됨. + * 키워드 휴리스틱이 이 패턴을 못 잡아 구조 기반 stripEnglishScratchPrefix를 추가. + * 표본 2건은 그날의 실제 세션 텍스트(09175258 세션)다. + */ + +import { test, describe } from 'node:test'; +import assert from 'node:assert/strict'; +import { separateThinkingFromContent } from '../src/gateway/chat/thinking-separation'; + +// ── 실측 표본 1: "추석에 비소식 없나?" 응답 ── +const SAMPLE_1 = `Also note discrepancy: Open-Meteo showed Sep 24 as mostly clear/dry (0% precip prob through 22:00), while KMA says 30% for Sep 24. KMA is observation-backed and preferred for Korea. I'll lead with KMA numbers. The openmeteo hourly for Sep 24 showed 0% precip probability — slight disagreement, but KMA is the authoritative source for Korea. I'll present KMA as main, and can mention the trend. + +Answer format: weather answers should be natural sentences, not forced tables. But a small table for the holiday period could help... The rules say for weather do NOT force a table — explain in natural fuller sentences. I'll write in Korean, conversational but informative. + +Key message: 아쉽게도 비 소식이 있습니다. 추석 당일(9/25 금) 강수확률 60%, 비 예보. 전날(9/24 목)은 흐림 30%. 토요일(9/26)도 60% 비. 일요일(9/27)부터는 개어서 4%. 서울 기준이며, 지역 알려주면 다시 확인해주겠다고. + +Also mention: 발표 시각 2026-09-22 23:00 KST 기준 단기예보라, 3일 뒤 예보라 변동 가능성 있음 — 추석 당일까지는 예보가 바뀔 수 있으니 며칠 뒤 다시 확인 권장. + +Keep it reasonably concise but full sentences. No emoji habit.아쉽게도 비 소식이 있습니다. 기상청 단기예보(오늘 밤 11시 발표, 서울 기준)에 따르면 추석 연휴 후반부에 비가 예보돼요. + +- **9/24 (목, 추석 전날)**: 흐림, 19~28°C, 강수확률 30% — 아직 비 예보는 아니고 흐려지는 날 +- **9/25 (금, 추석 당일)**: **비, 강수확률 60%**, 19~28°C — 당일은 비 소식 확실하게 잡혀 있음 + +즉 귀성길인 목요일까지는 큰 문제가 없지만, 추석 당일과 토요일 귀경길에 비를 만날 가능성이 높습니다. + +서울 기준으로 봤는데, 다른 지역(고향)이 궁금하면 알려주세요.`; + +// ── 실측 표본 2: "남부 지방은 어때?" 응답 ── +const SAMPLE_2 = `Actually the rule: "When comparing several cities (e.g. '오늘 유럽 대도시 최고 기온'), a table is fine for the numbers themselves, but always follow it with a few sentences of actual commentary." So table + commentary is fine here. + +Key points: +- 남부 3대 도시(부산·대구·광주) 모두 비슷한 패턴 +- 추석 당일(9/25 금): 세 도시 모두 흐림, 강수확률 30% — 서울(비 60%)보다 하루 늦게 비가 옴 + +Keep it focused and natural. Korean language.남부 지방은 서울보다 비가 하루 늦게 옵니다. 부산·대구·광주 세 도시를 기상청 단기예보(오늘 밤 11시 발표)로 확인했는데, 패턴이 거의 같아요. + +| 날짜 | 부산 | 대구 | 광주 | +|------|------|------|------| +| 9/26 (토) | **비 60%** | **비 60%** | **비 60%** | + +역시 3일 뒤까지의 예보라 변동 가능성은 있으니, 토요일 출발 전에 다시 확인하시는 게 좋습니다.`; + +describe('glm 영어 셀프토크 접두사 분리', () => { + test('실측 표본 1 — 영어 계획 노출분을 thinking으로 떼어낸다', () => { + const { reply, thinking } = separateThinkingFromContent(SAMPLE_1); + assert.ok(!/Also note|Answer format|Also mention|Keep it reasonably/.test(reply), + `reply에 영어 계획 잔여물이 남음: ${reply.slice(0, 120)}`); + assert.ok(reply.startsWith('아쉽게도 비 소식이 있습니다'), `reply 시작이 한글 답변이 아님: ${reply.slice(0, 60)}`); + assert.ok(/귀성길인 목요일/.test(reply), '답변 본문이 잘리면 안 됨'); + assert.ok(/Open-Meteo/.test(thinking), '분리된 thinking에 계획이 있어야 함'); + }); + + test('실측 표본 2 — "Korean language."뒤 무공백 융합을 끊는다', () => { + const { reply, thinking } = separateThinkingFromContent(SAMPLE_2); + assert.ok(!/Actually the rule|Keep it focused/.test(reply), + `reply에 영어 계획 잔여물이 남음: ${reply.slice(0, 120)}`); + assert.ok(reply.startsWith('남부 지방은 서울보다'), `reply 시작이 한글 답변이 아님: ${reply.slice(0, 60)}`); + assert.ok(/부산·대구·광주 세 도시를 기상청/.test(reply), '융합 단락의 한글 본문이 보존돼야 함'); + assert.ok(/9\/26 \(토\)/.test(reply), '마크다운 표가 잘리면 안 됨'); + assert.ok(/Actually the rule/.test(thinking), '분리된 thinking에 계획이 있어야 함'); + }); + + test('정상 한글 답변은 건드리지 않는다', () => { + const normal = '오늘 서울 날씨는 맑습니다. 최고 기온은 24도예요.\n\n- 아침 16도\n- 낮 24도\n\n즐거운 하루 보내세요.'; + const { reply, thinking } = separateThinkingFromContent(normal); + assert.equal(thinking, ''); + assert.equal(reply, normal); + }); + + test('영어 단어 뒤 바로 한글 오는 정상 혼합문은 보존된다', () => { + const mixed = 'Open-Meteo 모델은 목요일을 맑은 날로 보고 있습니다.\n\n기상청 기준으로 정리하면 다음과 같습니다.\n\n- 9/24: 흐림 30%\n- 9/25: 비 60%\n\n출발 전에 다시 확인하세요.'; + const { reply, thinking } = separateThinkingFromContent(mixed); + assert.equal(thinking, ''); + assert.ok(reply.startsWith('Open-Meteo'), '영어 단어로 시작하는 정상 답변이 잘리면 안 됨'); + }); + + test('짧은 영어 접두는 기존 키워드 휴리스틱이 걸러낸다', () => { + const short = 'Let me summarize the forecast for you.\n\n답변입니다. 추석 당일 비 예보 60%입니다.\n\n추가로 토요일도 비 60%예요.\n\n일요일부터는 개어서 맑습니다.'; + const { reply, thinking } = separateThinkingFromContent(short); + // "Let me" 단락은 starterRE/reasoningRE에 걸려 thinking으로 간다 — reply는 한글 답변만 남는다 + assert.ok(/Let me/.test(thinking), '영어 접두 단락이 thinking으로 가야 함'); + assert.ok(reply.startsWith('답변입니다'), `reply가 한글 답변이어야 함: ${reply.slice(0, 60)}`); + }); + + test('기존 태그 방식은 그대로 작동한다', () => { + const tagged = '사용자가 추석 비를 물었다. KMA를 보자.아쉽게도 추석 당일 비 예보가 있습니다. 강수확률 60%예요.'; + const { reply, thinking } = separateThinkingFromContent(tagged); + assert.ok(reply.startsWith('아쉽게도'), ` 태그가 안 떼어짐: ${reply.slice(0, 60)}`); + assert.ok(/추석 비를 물었다/.test(thinking), 'think 내용이 thinking으로 가야 함'); + }); +}); \ No newline at end of file