Files
homeclaw/src/gateway/chat/tool-scope.ts
T
kimandClaude Code bb6a64bff4 feat: write_scad 도구 추가 + 전체조립/부품 저장위치 분리 + PART 오타 흡수
write_scad — .scad 신규작성/수정 전용 도구(사용자 요청: "그래도 툴 하나
만들어주자"). python_eval의 open(path,'w') 우회는 os.makedirs가 막혀있어
폴더 없으면 실패하는 등 신뢰도가 낮았음. 저장 직후 실제 OpenSCAD 컴파일까지
해서 문법오류/빈 형상을 그 자리에서 알려주고 모델이 고쳐서 재시도하게 함.
registry.ts+build-tools.ts(실제 스키마) 둘 다 등록, tool-scope.ts 게이트 추가.
직접 테스트(정상/문법오류)+실채팅 테스트 통과.

PART 값 오타 흡수 — stl_cad의 compile 액션과 scad_to_stl 둘 다: scad 파일에
없는 PART 문자열을 넘기면 OpenSCAD가 에러 없이 조용히 빈 STL을 만드는 함정이
있었음(실측: 워크숍 채팅이 "tilt_arm"을 "Tilt Arm"/"tiltarm"/"tilt-arm"으로
반복해서 틀려 빈 STL만 생성). scad 소스에서 PART 분기 문자열을 미리 뽑아
대소문자/공백/하이픈 차이를 정규화해서 매칭하고, 그래도 안 맞으면 openscad를
돌리지도 않고 사용 가능한 PART 값 목록을 바로 에러로 돌려줌.

전체조립 vs 부품 저장위치 분리 — handle-chat.ts의 workshopProjectCtx 힌트:
part 지정해서 부품 하나만 만들 때는 CAD/print/에, part 없이 전체조립본
(시각화용, 인쇄 대상 아님) 만들 때는 CAD/ 바로 밑에 저장하도록 구분.

Co-Authored-By: Claude Code <noreply@anthropic.com>
2026-09-22 17:17:58 +09:00

421 lines
38 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* tool-scope.ts
*
* Decides WHICH tools the model is offered on a given turn.
*
* Extracted from handleChat() on 2026-07-29. It had lived as a closure inside a 2,959-line
* function, which made it untestable — the only way to check a gate was to boot the gateway,
* send a real chat message, and read the token count back out of an SSE event. Every condition
* here exists because some tool was riding along on unrelated turns and costing schema tokens,
* so a silent regression is invisible: nothing breaks, the prompt just quietly gets fatter or a
* needed tool quietly disappears.
*
* Pure by construction: no I/O, no config reads, no clock. Everything it needs arrives in
* ToolScopeInput, which is what makes it testable.
*
* Gating philosophy — two kinds, do not mix them up:
* - APP SESSION gates (wt_/lw_/ac_ prefixes) are unconditional inside that app's tab, because
* a follow-up like "내일은?" carries no keyword but is still a weather turn.
* - KEYWORD gates are a cost optimisation and are allowed to be wrong: worst case the model
* lacks a tool it could have used and says so. Never gate a SAFETY rule this way — see the
* system-prompt blocks in handle-chat.ts, which gate on tool availability instead precisely
* because a missed keyword there would drop a suppression rule.
*/
import { DISASTER_PATTERN } from '../guards/prompt-gates';
export interface ToolScopeInput {
/** The user's message for this turn — every keyword gate tests against it. */
message: string;
/**
* Current message plus the previous couple of user turns, for the gates where a follow-up
* legitimately inherits the topic ("그럼 모레는?" after a weather question). Optional: falls
* back to `message`, so callers that don't have history behave exactly as before.
*
* Applied ONLY to the weather/news/disaster gates (2026-08-12, user's call). Those decide
* whether a tool is OFFERED, so a stale match costs a few hundred schema tokens and nothing
* else. The other gates are left on the current message on purpose — widening them would let a
* coding word from two turns ago drag coder tools into an unrelated turn.
*/
recentUserText?: string;
/**
* The assistant's own previous reply, truncated. Used ONLY by hasStlCadKeyword (2026-09-22):
* a short retry like "다시 시도해봐" carries none of the stl/cad/scad keywords itself, and if
* the original request has already scrolled out of recentUserText's 2-turn window, the tool
* silently vanishes mid-task (실측 — 워크숍 채팅에서 "scad_to_stl 도구를 쓰라고 안내는 받았는데
* 실제 목록엔 없다"고 보고). The assistant's last turn is a strong same-topic signal here since
* our own tool descriptions inject the tool name into it.
*/
lastAssistantMsg?: string;
/** Session id; the wt_/lw_/ac_ prefixes select app-specific tool sets. */
sessionId: string;
/** Workspace path used for per-user skill toggles; null when unauthenticated. */
userWorkspace: string | null;
/** Injected so this module stays free of config/disk access. */
isSkillEnabledForUser: (slug: string, workspace: string | null) => boolean;
}
/** Boot (BOOT.md) turns get a read-only pair — they summarise state, they do not act. */
export const bootAllowedTools = new Set(['coder_list_files', 'coder_read_file']);
/**
* Code-editor sessions (code_ai_*) allow file/shell/search tools but block the ones that make
* no sense in an editor: browser automation, messaging, timers, presentations, image tools.
*/
export const codeAiBlockedTools = new Set(['start_task', 'task_control', 'write_note', 'browser_open', 'browser_snapshot', 'browser_click', 'browser_fill', 'browser_press_key', 'browser_wait', 'browser_close', 'image_search', 'image_url', 'image_edit', 'image_info', 'image_read', 'weather_search', 'weather_map_screenshot', 'start_timer', 'send_message', 'send_telegram', 'kakao_send_message', 'create_presentation', 'edit_presentation', 'spawn_subagent', 'schedule_job',
// desktop_* is Windows-only (desktop-tools.ts throws off process.platform); every call on
// this Linux server fails, same reasoning as dormantDesktopTools below for the main filter.
...(process.platform !== 'win32' ? ['desktop_screenshot', 'desktop_find_window', 'desktop_focus_window', 'desktop_click', 'desktop_drag', 'desktop_wait', 'desktop_press_key', 'desktop_type', 'desktop_set_clipboard', 'desktop_get_clipboard'] : [])]);
/**
* Builds the per-turn predicate. Returns a function of the tool NAME (not the whole schema
* object) so tests can drive it with plain strings.
*/
export function buildSkillToolFilter(input: ToolScopeInput): (toolName: string) => boolean {
const message = String(input.message || '');
// 후속 요청이 앞 턴의 주제를 물려받는 게이트에서만 쓴다. 아래 주석 참고.
const topicText = String(input.recentUserText || input.message || '');
const sessionId = String(input.sessionId || '');
const userWorkspace = input.userWorkspace;
// Weather comes in two tiers (2026-07-29). "내일 날씨" is an everyday question, not a
// specialist one, but all nine weather tools sat behind the meteorologist skill — so with the
// skill off a plain weather question had no weather tool at all and fell back to web_search.
// These two cover the ordinary case between them (weather_kma = Korean observations,
// weather_openmeteo = everywhere else, both with precipitation chance) for 446 tokens, and
// only when a weather keyword is present. They need no skill toggle.
const basicWeatherToolNames = new Set(['weather_kma', 'weather_openmeteo']);
// Air quality is a separate concern from the forecast, and weather_search is a redundant
// fallback whose schema costs as much as openmeteo while carrying less data — so they load
// on the weather keyword but are not part of the everyday pair above.
const weatherToolNames = new Set(['weather_search', 'weather_airpollution', 'weather_airkorea']);
// 2026-09-02 (efficiency audit): these four are climate-reanalysis/projection archives, not
// "what's the weather" tools — ERA5 and CDS are Copernicus reanalysis, CMIP6 is scenario
// projections, NASA POWER is agro-climate history. They were riding the plain weather keyword,
// so a casual "오늘 서울 날씨 어때?" carried 2,878 chars ≈ 820 tokens of archive API surface it
// could never use. Split onto their own keyword: a question about 기후/평년/과거/추세 is a
// different question from today's forecast, and phrasing it names the difference reliably.
const climateToolNames = new Set(['weather_nasa_power', 'weather_era5', 'weather_cds', 'weather_cmip6']);
// 2026-09-04 도달률 감사(실사용 1,011건): "메모리 가격 추세"가 이 게이트를 통과해 기후
// 재분석 아카이브 4종(~820토큰)을 통째로 실었다. 위 주석은 "기후/평년/과거/추세라는 표현이
// 차이를 확실히 짚어 준다"고 적었지만, 실제로는 "추세"와 "장기"가 기후와 아무 상관 없는 문장에
// 훨씬 자주 쓰인다(가격 추세, 장기 계획, 장기적으로…). 두 낱말만 기후·기상 명사와 같이 나올 때
// 걸리도록 좁힌다 — 나머지 키워드(기후/평년/재분석/era5 등)는 그 자체로 충분히 특정적이라 그대로 둔다.
const CLIMATE_SUBJECT = /기후|기온|날씨|강수|강우|우량|적설|기상|해수면|온난화|climate|weather|temperature|precipitation|rainfall/i;
const hasClimateKeyword =
/기후|기후변화|온난화|평년|과거\s*(기온|날씨|자료|데이터)|재분석|시나리오|연평균|기상\s*통계|climate|reanalysis|historical\s*(weather|climate)|era5|cmip|projection/i.test(topicText)
|| (/추세|장기/.test(topicText) && CLIMATE_SUBJECT.test(topicText));
const legalToolNames = new Set(['korean_law_search', 'korean_law_fetch', 'us_case_search']);
const pptxToolNames = new Set(['create_presentation', 'edit_presentation']);
// In practice these are only ever used while building an academic presentation (gather
// papers → create_presentation), not in standalone chat — gate them the same way as the
// pptx tools themselves rather than always loading 5 more schemas.
const academicToolNames = new Set(['pubmed_search', 'pubmed_fetch', 'pubmed_fulltext', 'openalex_search', 'semantic_search']);
// cms_hospital_compare — CMS Medicare hospital data (ownership / star rating / mortality /
// readmission). ~450-token schema, only relevant to US-hospital-performance questions, so
// gate it rather than carry it on every turn.
// drug_dur_check — 식약처 DUR(병용금기·임부금기 등). 닥터앱 탭에서는 무조건, 일반 챗에서는
// 약 이름·상호작용을 실제로 물을 때만. "약"은 너무 흔해서 단독으로 넣지 않고 문맥형으로만.
const hasDrugKeyword =
/병용금기|임부금기|수유부|복약|약물\s*상호작용|약\s*상호작용|같이\s*(먹어|복용|드셔)|함께\s*(먹어|복용)|DUR|처방전|약봉투|성분\s*겹|중복\s*처방|이\s*약(을|은|이|,|\s|의|도|과|와|\?)|먹는\s*약|드시는\s*약|복용\s*중(인|during)?\s*약|drug\s*interaction|contraindicat/i
.test(message);
// 2026-09-04: "미국의 영리의료법인 현황"·"영리의료법인의 법인세율?"이 이 게이트를 통과하지
// 못했다. 사용자가 실제로 쓴 말은 "병원"이 아니라 "의료법인"이었는데 목록에 그 낱말이 없었다
// ("영리 병원의 비율"만 우연히 걸렸다). 영리/비영리 소유형태를 묻는 순간이 바로 이 도구가
// 있는 이유라, 그 수식어에 붙는 기관 명사를 넓힌다.
const hasHospitalDataKeyword =
/\bCMS\b|메디케어|\bmedicare\b|care\s*compare|hospital\s*compare|hospital\s*(rating|ratings|star|quality|ownership|performance)|(병원|의료법인|의료기관|의료재단|요양기관).{0,6}(비교|평가|등급|순위|별점|소유|영리|비영리|성과|퀄리티|데이터|현황|비율)|(영리|비영리|투자자\s*소유).{0,4}(병원|의료법인|의료기관|의료재단|병상)|(for-?profit|non-?profit|nonprofit|investor-owned).{0,20}(hospital|health\s*system|medical\s*(center|corporation))/i
.test(message);
// coder_* (10 tools) is the file-authoring toolkit for actual code/script work — it has its
// own dedicated branch for isCodeAiSession (code.html) below, so this only affects the
// general chat path (main app + app-specific tabs like ac_/wt_/lw_/dental_/mn_, none of which
// are coding contexts). Ungated, it rode along on every single turn regardless of topic.
const coderToolNames = new Set(['coder_read_file', 'coder_write_file', 'coder_overwrite_lines', 'coder_insert_lines', 'coder_delete_lines', 'coder_str_replace', 'coder_patch_file', 'coder_delete_file', 'coder_move_file', 'coder_list_files']);
const hasCodeKeyword = /코드|스크립트|소스\s*코드|알고리즘|파이썬|python|javascript|typescript|자바스크립트|타입스크립트|프로그램|함수|클래스|디버그|버그|리팩터|refactor|\.py\b|\.js\b|\.ts\b|\.html\b|\.css\b|\.json\b|\.sh\b|\bcode\b|script|algorithm/i.test(message);
// Email/kakao/db/image-edit — same story as excel: no dedicated skill toggle or app tab,
// so gate purely on keyword. shell's own description tells the model to pair it with
// coder_write_file, so it rides the same code-intent gate rather than its own.
const emailToolNames = new Set(['email_list', 'email_read', 'email_send', 'email_search', 'email_delete']);
// 2026-07-29 bugfix (found by tests/tool-scope.test.ts). A Hangul syllable followed by a
// word-boundary anchor can NEVER match: \b is an ASCII boundary and Hangul is not an ASCII
// word character, so there is no boundary between 일 and a following space. "메일 확인해줘"
// therefore failed this gate outright and the email tools stayed hidden; only the separate
// "이메일" alternative ever fired. The same dead pattern was removed from hasCodeKeyword
// (함수/클래스/버그), hasNewsKeyword (기사) and hasPptxKeyword (덱) in the same pass — all of
// them were silently no-op alternatives. Dropping the anchor is safe for Korean: these words
// are unambiguous as substrings (카카오메일, 기사 등). Keep \b only on ASCII tokens like ppt\b.
const hasEmailKeyword = /이메일|메일|email|inbox|받은편지함/i.test(message);
// 2026-09-04 감사: `![KakaoMap_20260903_162614.png](/api/files/uploads/…)` 한 줄이 이 게이트를
// 통과해 kakao_send_message를 실었다. 사용자가 카카오맵 캡처를 올렸을 뿐인데 메신저 발송 도구가
// 붙는다 — 그리고 executeTool은 스키마 소속을 확인하지 않으므로([[project_tool_schema_leak]])
// 프롬프트가 이름만 언급해도 실제로 호출될 수 있는 종류의 도구다. 첨부 파일명은 사용자의 '요청'이
// 아니므로 판정에서 뺀다. 카카오'맵'도 메신저가 아니라 지도라 함께 제외한다.
const messageWithoutAttachments = message.replace(/!?\[[^\]]*\]\([^)]*\)/g, ' ');
const hasKakaoKeyword = /카카오(?!\s*맵)|kakao(?!map)/i.test(messageWithoutAttachments);
// USER.md/SOUL.md에 사실을 직접 적어 넣는 3종. 아래 게이트 주석 참고.
// 사용자가 기억을 **지시**하거나 **조회**할 때만 싣는다. topicText를 쓰는 이유는 "그것도
// 기억해둬" 같은 후속 발화가 앞 턴의 주제를 물려받기 때문이다.
const manualMemoryToolNames = new Set(['memory_browse', 'memory_write', 'memory_read']);
const hasMemoryIntentKeyword =
/기억(해|해둬|해\s*둬|해\s*줘|하고|해두|시켜|나|해야|할|한|났|안\s*나|못\s*해)|잊지\s*마|잊어버리지|외워|저장해|기록해|메모해|적어\s*둬|프로필|persona|USER\.md|SOUL\.md|\bremember\b|\bmemorize\b|don'?t forget|keep in mind|note that/i
.test(topicText);
const hasSqliteKeyword = /데이터베이스|sqlite|쿼리|query\b|\.db\b/i.test(message);
const hasImageEditKeyword = /편집|보정|자르기|리사이즈|크롭|워터마크|배경\s*제거|필터|흑백|그레이스케일|스타일화|crop|resize|watermark|remove.?bg|grayscale|stylize/i.test(message);
// news_search is the single biggest ungated schema (2,035 chars, mostly usage guidance for
// its country/category/language params) yet was included on every turn regardless of topic.
const hasNewsKeyword = /뉴스|시사|헤드라인|속보|기사|news|headline/i.test(topicText);
// image_read (OCR) only makes sense once an actual image is in play — the upload flow
// inserts attachments as markdown image links (![name](/api/files/...png)), so detect that
// directly instead of guessing at phrasing like the other keyword gates.
const hasImageAttachment = /\.(png|jpe?g|webp|bmp|tiff?|gif)\b/i.test(message);
const hasRunCommandKeyword = /실행해|열어줘|띄워줘|런칭|실행시켜|메모장|notepad|vscode|vs\s*code|launch\s+(app|the|notepad|vscode)/i.test(message);
// Same shape as hasImageAttachment — pdf_read/pdf_extract_images (1,792 chars combined, the
// biggest remaining pair) only make sense once a PDF is actually referenced.
const hasPdfAttachment = /\.pdf\b|pdf\s*(파일|읽어|추출)/i.test(message);
const hasScheduleKeyword = /예약|스케줄|반복|매일|매주|매월|정기적|주기적|cron|schedule|recurring/i.test(message);
// eonet_events/satellite_snapshot — casual "what's happening on Earth" feature,
// not relevant to unrelated turns.
const hasSatelliteKeyword = /위성|자연재해|산불|태풍|화산|지진|산사태|가뭄|황사|연무|해빙|satellite|wildfire|volcano|eonet|natural\s*disaster/i.test(message);
// nvr_snapshot — 집 CCTV 한 프레임 캡처해 모델이 직접 확인. "마당에 누구 있어?", "택배 왔나
// 봐줘", "현관/주차장 확인", "고양이 있나" 같은 요청에만. 카메라/감시 어휘 + "확인해줘·봐줘"류
// 동사, 또는 알려진 카메라 위치명이 있을 때 연다.
//
// topicText(직전 2턴 포함)로 검사한다 — 실사고(2026-09-07): "cctv확인해봐"로 도구가 한 번
// 돌았는데, 바로 다음 "아니다 cam2" / "사진 보여줘" 후속턴엔 키워드가 없어 도구가 사라지자
// 모델이 "저는 그런 툴이 없습니다"라고 3턴 연속 답했다. 스테일 매치 비용은 작은 스키마 토큰뿐.
const hasCctvKeyword = /cctv|감시\s*카메라|카메라(로|에|를|\s|$)|웹캠|현관(?:문|앞|쪽)?|마당|주차장|차고|대문|택배|초인종|스냅샷|캡처해|누가?\s*(?:왔|있|지나)|누구\s*(?:있|왔)|밖에?\s*(?:누구|뭐|무슨)|집\s*(?:앞|밖|상황)|\bcam\s*\d|채널\s*\d.*(?:봐|보여|확인|찍)|door\s*cam|security\s*cam|who'?s?\s*(?:at\s*the\s*door|outside)/i.test(topicText);
// print3d_model — STL 슬라이싱+K2 프린터 업로드. topicText로 검사(prepare→start 확인
// 왕복이 여러 턴 걸치므로 nvr_snapshot과 같은 이유로 최근 대화 전체를 봄).
const hasPrint3dKeyword = /3d\s*프린트|3d\s*print|stl|슬라이싱|슬라이서|g-?code|출력해|프린터로|k2\s*(?:콤보|프린터)?/i.test(topicText);
// stl_cad/scad_to_stl — STL 확인/렌더링/구멍뚫기/scad 컴파일(OpenSCAD). print3d와 겹치는
// "stl" 외에 CAD 편집 관련 어휘도 추가로 매칭. 직전 어시스턴트 답변도 같이 봄(위
// lastAssistantMsg 주석 참고) — 짧은 "다시 시도해봐" 재시도에 도구가 사라지는 걸 막는다.
const stlCadPattern = /stl|cad|캐드|구멍\s*뚫|렌더링|바운딩박스|openscad|치수\s*확인/i;
const hasStlCadKeyword = stlCadPattern.test(topicText) || stlCadPattern.test(String(input.lastAssistantMsg || '').slice(-800));
// workshop_project — "작업실"(여러 메이커 프로젝트) 대시보드 부품/작업 편집. "로봇" 단독은
// 로봇청소기(Valetudo) 대화와 겹치므로 프로젝트 고유 어휘로만 매칭한다. 새 프로젝트 종류가
// 늘어날 때마다 이 목록도 넓혀줘야 함(앱 세션 안에서는 아래 isWorkshopAppSession이 대신
// 커버하므로 메인 채팅에서 키워드 없이 말할 때만 이 게이트가 문제가 됨).
const hasWorkshopKeyword = /작업실\s*프로젝트|로봇\s*프로젝트|메카넘|mecanum|로봇팔|robot\s*arm|so-?101|so-?arm|rplidar|라이다.*로봇|로봇.*(?:부품|예산|작업|단계|대시보드)|esp32|하우징/i.test(topicText);
// 2026-08-10: news_search was gated on the literal word "뉴스" (hasNewsKeyword below), so a
// disaster question with no news-flavored wording ("돌핀 지금 어디 있지?") never got access to
// it — even after handle-chat.ts's reminder started explicitly recommending news_search for
// exactly this case (real incident: search for the storm's position turned up nothing useful,
// while the actual story — landfall in China, evacuations, damage — was straightforward news
// coverage). Reuses prompt-gates.ts's DISASTER_PATTERN rather than keeping a second copy of
// the same keyword list here, per this file's own header note on gates needing to agree.
const hasDisasterKeyword = DISASTER_PATTERN.test(topicText);
// 2026-09-02: the per-user skills_state.json toggles are legacy — the UI moved to per-app
// gating and nobody maintains these flags any more. They were still the only load-bearing
// consumer of that state, and stale values were silently WITHHOLDING tools: papa has
// meteorologist:false, so "오늘 서울 미세먼지 어때?" got weather_kma/openmeteo only —
// and weather_kma carries no air-quality data at all, so the answer could only be a
// failure or an invention. weather_airkorea (에어코리아, key approved and working) was
// sitting right there, gated off by a flag from a UI the user no longer uses.
//
// The keyword gates below are the real cost control and they are already narrow. A toggle
// that is never updated is not a safety mechanism, it is a trap — so the skill condition is
// dropped and the keyword alone decides. `isSkillEnabledForUser` stays on ToolScopeInput for
// the app-session gates and for whenever skills get real prompt files again.
const hasPptxKeyword = /슬라이드|발표|pptx|ppt\b|프레젠테이션|피피티|덱|presentation/i.test(message);
// 2026-08-25: 논문 도구는 pptx 키워드에만 묶여 있어서, "논문 찾아줘"로는 아예 안 켜지고
// 모델이 web_search로 때웠다(실사용 로그에서 확인). 학술 검색은 web_search보다 확실히
// 낫다 — openalex_search는 키 없이 2억 건에서 제목·저자·연도·저널·DOI·인용수·오픈액세스
// 여부를 구조화해 준다. 그래서 학술 의도 자체를 게이트로 추가한다.
// 한글 뒤에는 \b를 쓰지 않는다(ASCII 경계라 절대 매칭 안 됨 — 위 hasEmailKeyword 주석 참고).
const hasAcademicKeyword = /논문|페이퍼|학술|저널|문헌|선행연구|인용수|피인용|초록|아카이브|학회지|\bpapers?\b|\bpubmed\b|\bdoi\b|arxiv|scholar|citations?|journals?|literature/i.test(message);
// Weather/legal skills being ON means the user does that kind of work sometimes, not that
// every single turn is about it — without a keyword gate the full 9 weather / 3 legal tools
// rode along on unrelated turns too. Mirrors the pptx pattern (skill + keyword) below.
// 기상 현상 목록. "비"·"눈"처럼 다른 뜻과 겹치는 짧은 단어가 섞여 있으므로, 단독으로 쓰지 말고
// 반드시 아래처럼 물음 형태("~소식/예보/전망/상황")와 짝지어서만 쓸 것.
const WEATHER_PHENOMENON = '비|눈|태풍|장마|한파|폭염|황사|미세먼지|우박|안개|서리|더위|추위|바람|구름|강수|기온|날씨';
// 2026-08-12 실사고: "향후 비소식은 없나?"에 이 목록이 하나도 안 걸려 weather_kma·
// weather_openmeteo가 스키마에서 통째로 빠졌고, 모델은 날씨 도구가 없으니 web_search로
// 갔다가 2019년 기사를 물어와 "예보를 확인할 수 없다"고 답했다. 이어서 사용자가
// "기상청에 알아봐"라고 명시했는데 그 말조차 목록에 없어서 두 번째 턴도 똑같이 실패했다.
// 대화 전체가 날씨 도구 없이 돈 것이다.
//
// 이 게이트의 비용 비대칭은 프롬프트 게이트와 반대다 — 오탐은 도구 스키마 몇백 토큰이
// 늘 뿐이지만, 미탐은 답변 경로 자체가 틀린다. 그러니 여기서는 넉넉하게 잡는다.
//
// 실사용 메시지 103건으로 교정([[feedback_calibrate_on_real_data]]): "기상청" 신규 2건
// (전부 진짜 날씨), "비" 문맥형 신규 2건(전부 진짜 날씨), 오탐 0건. "비"는 단독으로 넣으면
// 비용·비교·준비에 걸리므로 반드시 문맥형으로만 잡는다("비소식"처럼 붙여 쓰는 것도 포함).
const hasWeatherKeyword = /날씨|기온|강수|미세먼지|대기질|태풍|기후|일기예보|장마|폭염|한파|자외선|황사|weather|forecast/i.test(topicText)
|| /기상청|예보|기상\s*상황/.test(topicText)
// "OO 소식 있나?"는 날씨를 묻는 가장 흔한 말투인데, 현상별로 흩어 놓으면 다음 현상에서
// 또 빠진다(우박·안개·서리가 실제로 빠져 있었다). 현상 목록 × 물음 형태로 한 번에 잡는다.
// "소식" 단독은 절대 넣지 않는다 — 회사 소식·업계 소식·새로운 소식에 다 걸린다.
|| new RegExp(`(${WEATHER_PHENOMENON})\\s*(소식|예보|전망|상황)`).test(topicText)
|| /비가\s*(오|와|내리|올)|비\s*(온다|와요|올까|내려)|소나기|호우|폭우|우산/.test(topicText)
|| /눈이\s*(오|내리)|눈\s*올/.test(topicText)
|| /무더위|더위|추위|쌀쌀|습도|풍속|체감\s*온도/.test(topicText);
const hasLegalKeyword = /법률|판례|소송|변호사|법원|법조문|조항|계약서|형법|민법|법령|legal|lawsuit|court|statute/i.test(message);
// excel_read/excel_write have no dedicated skill toggle (accountant-app.html's persona is
// page-local, not a registered skill) — gate on the accountant app tab (ac_ sessions) or an
// explicit excel/spreadsheet keyword, same shape as the weather/legal gates above. Previously
// ungated, so both tool schemas rode along on every single chat turn regardless of app.
const hasExcelKeyword = /엑셀|xlsx|xls\b|스프레드시트|spreadsheet|excel/i.test(message);
// ollama_web_search/ollama_web_fetch are explicitly self-deprecating in their own descriptions
// ("web_search가 없는 환경에서만 fallback으로 사용") — and the provider fallback they exist for
// is already handled server-side inside executeWebSearch, so the model never needs to reach for
// them on its own. Advertising them cost 214 tokens per turn to tell the model not to use a
// tool. Gate on the same explicit phrasing personality-context.ts already uses for its
// ollama_web hint block, so the pair stays available when the user actually asks for it by name
// (otherwise that hint would describe tools no longer in the schema).
const ollamaWebToolNames = new Set(['ollama_web_search', 'ollama_web_fetch']);
const hasOllamaWebKeyword = /ollama[\s_]?(웹|web|검색|search)|올라마\s*(웹|검색|로\s*검색)/i.test(message);
// Sub-agent/task-orchestration tools that tool_audit.log shows near-zero real usage for
// (2026-05-04~2026-07-12: start_task/request_secondary_assist/subagent_spawn/
// delegate_to_specialist/spawn_subagent = 0 calls; task_control/spawn_agent = a handful, all
// stale). Kept registered for future agent-composition work,
// just not sent to the model in normal chat. schedule_job is excluded from this set — it's
// self-contained (list/create/update/pause/resume/delete/run_now) and still actively used.
const dormantAgentTools = new Set(['start_task', 'task_control', 'request_secondary_assist', 'subagent_spawn', 'delegate_to_specialist', 'spawn_subagent', 'spawn_agent']);
// Browser automation — parked. tool_audit.log shows every browser_open call from 07-09
// onward failing with "Chrome launched but did not respond on port 9222". Repeated debug
// runs showed the live `tools` array alternates between two mutually-exclusive sets for the
// SAME "ping" request against the SAME running process — open/snapshot/click/fill/press_key/
// wait/close one time, scroll/get_images the next — apparently some live Chrome-reachability
// probe swapping which schema set gets exposed. Excluding all 9 names covers both variants.
// Re-enable by removing this set once browser automation is fixed / needed (e.g. home shopping).
const dormantBrowserTools = new Set(['browser_open', 'browser_snapshot', 'browser_click', 'browser_fill', 'browser_press_key', 'browser_wait', 'browser_close', 'browser_scroll', 'browser_get_images']);
// desktop_* (10 tools, ~2.7k of the tool-schema overhead) is Windows-only remote-desktop
// control (mouse/keyboard/screenshot) — desktop-tools.ts itself throws "Desktop tools are
// currently supported on Windows only." off process.platform, and tool_audit.log shows every
// desktop_screenshot call ever made on this server failing with exactly that. Mirror the same
// platform check here so the schemas aren't sent on a platform where every call is guaranteed
// to fail (auto-recovers if this ever runs on Windows, unlike a hardcoded exclude).
const dormantDesktopTools = process.platform !== 'win32'
? new Set(['desktop_screenshot', 'desktop_find_window', 'desktop_focus_window', 'desktop_click', 'desktop_drag', 'desktop_wait', 'desktop_press_key', 'desktop_type', 'desktop_set_clipboard', 'desktop_get_clipboard'])
: new Set<string>();
// MCP database servers built for a specific dedicated app tab (dental-agent.html →
// 'dental_' sessions, mind-app.html → 'mn_' sessions) — 62% of the tool-schema overhead
// (12,124 of 19,531 tokens measured) came from these 3 servers × 8 sub-tools each, riding
// along on every main-chat turn for every user even though they're single-purpose lookup
// tables for one app. Scope them to the app session that actually uses them.
const isDentalAppSession = /^dental_/.test(String(sessionId || ''));
const isMindAppSession = /^mn_/.test(String(sessionId || ''));
// Weather/lawyer have their own dedicated app tabs (weather-app.html → 'wt_' sessions,
// lawyer-app.html → 'lw_' sessions). Inside those apps the skill+keyword gate below is
// wrong — a "wt_" session IS the weather context even if a given message has no weather
// keyword (e.g. a follow-up "내일은?"), so always include the tools there.
const isWeatherAppSession = /^wt_/.test(String(sessionId || ''));
const isLawyerAppSession = /^lw_/.test(String(sessionId || ''));
const isAccountantAppSession = /^ac_/.test(String(sessionId || ''));
// 2026-09-02: two more app tabs ship a session prefix but never got a gate, so inside them a
// natural follow-up ("그럼 작년은?", "이 사건 자료 더 찾아줘") fell back to main-chat keyword
// matching and came up with no specialist tool at all.
// detective-app.html → 'dt_': 판례 12회 / 법률 11회 언급. Same legal tool family as the
// lawyer tab — it is a case-research app, so it needs 국가법령정보센터/CourtListener.
// investor-app.html → 'iv_': 환율 6회, 주가·증시·엑셀·뉴스. Market questions are live-data
// questions; without news_search a "작년 대비" follow-up has nothing to read.
// doctor-app.html → 'dr_' (2026-09-02, 사용자 요청): 개인 의료기록 정리 앱이지만 "이 약
// 장기복용 괜찮나?", "이 백신 효과 몇 년?", "이 수치 정상 범위인가?" 류 질문에 지금은
// web_search로 블로그를 물어온다. 학술 도구(PubMed/OpenAlex/Semantic Scholar)를 이 탭
// 안에서만 열어 근거 문헌으로 답하게 한다. 앱 상단 면책("진단·처방 대체 아님")은 유지되고,
// 세션 id는 dr_main / dr_case_<id> / dr_<rand> 모두 dr_ 접두사라 한 번에 잡힌다.
// language-app ('lg_') is deliberately NOT here — a tutor served fine by the base tool set.
const isDetectiveAppSession = /^dt_/.test(String(sessionId || ''));
const isInvestorAppSession = /^iv_/.test(String(sessionId || ''));
const isDoctorAppSession = /^dr_/.test(String(sessionId || ''));
// workshop-app.html의 전용 채팅 패널('ws_' 세션) 안에서는 그 대화 전체가 이미 어떤
// 프로젝트 얘기라는 게 확실하므로, 키워드 없이 짧게 말해도("이 부품 지워줘") workshop_project가
// 열리게 함(원래 robot-app.html/'rb_' 세션이었던 것을 여러 프로젝트로 일반화, 2026-09-20).
const isWorkshopAppSession = /^ws_/.test(String(sessionId || ''));
const filter = (name: string): boolean => {
if (dormantAgentTools.has(name)) return false;
if (dormantBrowserTools.has(name)) return false;
if (dormantDesktopTools.has(name)) return false;
if (name.startsWith('mcp__dental-dict-sqlite__') && !isDentalAppSession) return false;
if ((name.startsWith('mcp__psychotherapy-cases-sqlite__') || name.startsWith('mcp__psychiatry-cases-sqlite__')) && !isMindAppSession) return false;
if (basicWeatherToolNames.has(name) && !isWeatherAppSession && !hasWeatherKeyword) return false;
if (weatherToolNames.has(name) && !isWeatherAppSession && !hasWeatherKeyword) return false;
if (climateToolNames.has(name) && !isWeatherAppSession && !hasClimateKeyword) return false;
// Windy map screenshots only make sense inside the dedicated weather app tab (the
// model there has already generated the embed.windy.com URL to capture) — unlike the
// other weather tools, not offered via the meteorologist skill in general chat.
//
// Exception (2026-08-10): the KMA/JMA typhoon-track capture path (KMA_TYPHOON_TRACK_URL /
// JMA_TYPHOON_MAP_URL in weather.ts) doesn't depend on a prior Windy embed URL the way the
// radar/satellite path does — it's a fixed, always-valid URL. Real incident: a main-chat
// "태풍 돌핀 지금 어디 있지?" had no access to this tool at all, so the model fell back to
// web_search with a weak self-generated query and missed the actual news. The KMA/JMA
// track page is the authoritative source for exactly this question and was sitting behind
// a gate that had nothing to do with typhoons — same hasSatelliteKeyword gate that already
// covers eonet_events/satellite_snapshot below, since "태풍" is in that keyword list.
if (name === 'weather_map_screenshot' && !isWeatherAppSession && !hasSatelliteKeyword) return false;
if (name === 'nhc_active_storms' && !isWeatherAppSession) return false;
if (ollamaWebToolNames.has(name) && !hasOllamaWebKeyword) return false;
if (legalToolNames.has(name) && !isLawyerAppSession && !isDetectiveAppSession && !hasLegalKeyword) return false;
if ((name === 'excel_read' || name === 'excel_write') && !isAccountantAppSession && !isInvestorAppSession && !hasExcelKeyword) return false;
if (coderToolNames.has(name) && !hasCodeKeyword) return false;
if (name === 'shell' && !hasCodeKeyword) return false;
// 2026-07-29 bugfix (found by tests/tool-scope.test.ts on extraction): the condition used to
// be `presenterEnabled && !hasPptxKeyword`, i.e. the exclusion only ran when the presenter
// skill was ENABLED. With the skill OFF the gate never fired at all and both schemas rode
// along on every turn — inverted, since a disabled skill should restrict more, not less.
// Scope of the actual impact (checked against skills_state.json the same day): papa and
// cherry have presenter=true so they were unaffected; jasmine has it false and was carrying
// ~1,600 tokens of presentation API surface (create_presentation alone is 833, the largest
// tool schema in the codebase) on every turn including plain greetings.
// Now matches the weather/legal shape directly above: exclude unless skill AND keyword agree.
if (pptxToolNames.has(name) && !hasPptxKeyword) return false;
if (academicToolNames.has(name) && !isDoctorAppSession && !hasPptxKeyword && !hasAcademicKeyword) return false;
if (name === 'cms_hospital_compare' && !hasHospitalDataKeyword) return false;
if (name === 'drug_dur_check' && !isDoctorAppSession && !hasDrugKeyword) return false;
if (emailToolNames.has(name) && !hasEmailKeyword) return false;
if (name === 'kakao_send_message' && !hasKakaoKeyword) return false;
// USER.md/SOUL.md 수동 메모리 3종. 2026-09-04 효율 감사로 게이팅.
//
// 근거는 순수하게 사용량이다: 실사용 1,011턴에서 browse 0회 / read 2회 / write 1회인데
// **매 턴 무조건** 실려 357토큰을 냈다. 게이트가 걸리면 0.3%의 턴에만 실린다.
//
// 도구 자체는 멀쩡하다. USER.md·SOUL.md는 `workspace/prompts/` 아래에 실제로 있고
// resolvePromptPath가 거기서 찾는다(papa의 USER.md는 3.2KB, 카테고리 5개). 감사 중에
// 워크스페이스 루트만 보고 "파일이 없다"고 잘못 판단했다가 정정한 자리다 — 셋 다
// 파일이 없으면 에러를 내므로, 없다는 판단이 맞았다면 통째로 죽은 도구였을 뻔했다.
//
// 제거가 아니라 게이팅인 이유: 사용자가 "기억해둬"라고 명시하는 턴에는 반드시 있어야 한다.
// 평상시 장기기억은 일별 로그 + Chroma 자동 추출이 담당하지만([[project_vector_memory_chroma]]),
// 그건 하루 뒤에 반영되는 배치라 "지금 이걸 기억해"의 대체재가 아니다.
// memory_stats는 벡터 스토어를 보므로 게이트 대상이 아니다(실제로 13회 호출됨).
if (manualMemoryToolNames.has(name) && !hasMemoryIntentKeyword) return false;
if (name === 'sqlite_query' && !hasSqliteKeyword) return false;
if (name === 'image_edit' && !hasImageEditKeyword) return false;
if (name === 'image_read' && !hasImageAttachment) return false;
if (name === 'news_search' && !isInvestorAppSession && !isDetectiveAppSession && !hasNewsKeyword && !hasDisasterKeyword) return false;
if ((name === 'pdf_read' || name === 'pdf_extract_images') && !hasPdfAttachment) return false;
if (name === 'schedule_job' && !hasScheduleKeyword) return false;
if ((name === 'eonet_events' || name === 'satellite_snapshot') && !hasSatelliteKeyword) return false;
if (name === 'nvr_snapshot' && !hasCctvKeyword) return false;
if (name === 'print3d_model' && !hasPrint3dKeyword) return false;
if ((name === 'stl_cad' || name === 'scad_to_stl' || name === 'write_scad') && !hasStlCadKeyword) return false;
if (name === 'workshop_project' && !isWorkshopAppSession && !hasWorkshopKeyword) return false;
if (name === 'run_command' && !hasRunCommandKeyword) return false;
return true;
};
return filter;
}
/**
* Applies the right tool set for the turn. The four branches are mutually exclusive and
* ordered by specificity: boot < inert sessions < code editor < normal chat.
*/
export function selectToolsForTurn(
allTools: any[],
input: ToolScopeInput,
flags: { isBootStartupTurn: boolean; isCodeAiSession: boolean; isTranslateSession: boolean; isProjSession: boolean },
): any[] {
const nameOf = (t: any) => String(t?.function?.name || '');
if (flags.isBootStartupTurn) return allTools.filter(t => bootAllowedTools.has(nameOf(t)));
// Translation and project-file sessions answer from the prompt alone; handing them tools only
// invites a stray call that corrupts a strict output format.
if (flags.isTranslateSession || flags.isProjSession) return [];
if (flags.isCodeAiSession) return allTools.filter(t => !codeAiBlockedTools.has(nameOf(t)));
const filter = buildSkillToolFilter(input);
return allTools.filter(t => filter(nameOf(t)));
}