write_scad — .scad 신규작성/수정 전용 도구(사용자 요청: "그래도 툴 하나 만들어주자"). python_eval의 open(path,'w') 우회는 os.makedirs가 막혀있어 폴더 없으면 실패하는 등 신뢰도가 낮았음. 저장 직후 실제 OpenSCAD 컴파일까지 해서 문법오류/빈 형상을 그 자리에서 알려주고 모델이 고쳐서 재시도하게 함. registry.ts+build-tools.ts(실제 스키마) 둘 다 등록, tool-scope.ts 게이트 추가. 직접 테스트(정상/문법오류)+실채팅 테스트 통과. PART 값 오타 흡수 — stl_cad의 compile 액션과 scad_to_stl 둘 다: scad 파일에 없는 PART 문자열을 넘기면 OpenSCAD가 에러 없이 조용히 빈 STL을 만드는 함정이 있었음(실측: 워크숍 채팅이 "tilt_arm"을 "Tilt Arm"/"tiltarm"/"tilt-arm"으로 반복해서 틀려 빈 STL만 생성). scad 소스에서 PART 분기 문자열을 미리 뽑아 대소문자/공백/하이픈 차이를 정규화해서 매칭하고, 그래도 안 맞으면 openscad를 돌리지도 않고 사용 가능한 PART 값 목록을 바로 에러로 돌려줌. 전체조립 vs 부품 저장위치 분리 — handle-chat.ts의 workshopProjectCtx 힌트: part 지정해서 부품 하나만 만들 때는 CAD/print/에, part 없이 전체조립본 (시각화용, 인쇄 대상 아님) 만들 때는 CAD/ 바로 밑에 저장하도록 구분. Co-Authored-By: Claude Code <noreply@anthropic.com>
421 lines
38 KiB
TypeScript
421 lines
38 KiB
TypeScript
/**
|
||
* tool-scope.ts
|
||
*
|
||
* Decides WHICH tools the model is offered on a given turn.
|
||
*
|
||
* Extracted from handleChat() on 2026-07-29. It had lived as a closure inside a 2,959-line
|
||
* function, which made it untestable — the only way to check a gate was to boot the gateway,
|
||
* send a real chat message, and read the token count back out of an SSE event. Every condition
|
||
* here exists because some tool was riding along on unrelated turns and costing schema tokens,
|
||
* so a silent regression is invisible: nothing breaks, the prompt just quietly gets fatter or a
|
||
* needed tool quietly disappears.
|
||
*
|
||
* Pure by construction: no I/O, no config reads, no clock. Everything it needs arrives in
|
||
* ToolScopeInput, which is what makes it testable.
|
||
*
|
||
* Gating philosophy — two kinds, do not mix them up:
|
||
* - APP SESSION gates (wt_/lw_/ac_ prefixes) are unconditional inside that app's tab, because
|
||
* a follow-up like "내일은?" carries no keyword but is still a weather turn.
|
||
* - KEYWORD gates are a cost optimisation and are allowed to be wrong: worst case the model
|
||
* lacks a tool it could have used and says so. Never gate a SAFETY rule this way — see the
|
||
* system-prompt blocks in handle-chat.ts, which gate on tool availability instead precisely
|
||
* because a missed keyword there would drop a suppression rule.
|
||
*/
|
||
|
||
import { DISASTER_PATTERN } from '../guards/prompt-gates';
|
||
|
||
export interface ToolScopeInput {
|
||
/** The user's message for this turn — every keyword gate tests against it. */
|
||
message: string;
|
||
/**
|
||
* Current message plus the previous couple of user turns, for the gates where a follow-up
|
||
* legitimately inherits the topic ("그럼 모레는?" after a weather question). Optional: falls
|
||
* back to `message`, so callers that don't have history behave exactly as before.
|
||
*
|
||
* Applied ONLY to the weather/news/disaster gates (2026-08-12, user's call). Those decide
|
||
* whether a tool is OFFERED, so a stale match costs a few hundred schema tokens and nothing
|
||
* else. The other gates are left on the current message on purpose — widening them would let a
|
||
* coding word from two turns ago drag coder tools into an unrelated turn.
|
||
*/
|
||
recentUserText?: string;
|
||
/**
|
||
* The assistant's own previous reply, truncated. Used ONLY by hasStlCadKeyword (2026-09-22):
|
||
* a short retry like "다시 시도해봐" carries none of the stl/cad/scad keywords itself, and if
|
||
* the original request has already scrolled out of recentUserText's 2-turn window, the tool
|
||
* silently vanishes mid-task (실측 — 워크숍 채팅에서 "scad_to_stl 도구를 쓰라고 안내는 받았는데
|
||
* 실제 목록엔 없다"고 보고). The assistant's last turn is a strong same-topic signal here since
|
||
* our own tool descriptions inject the tool name into it.
|
||
*/
|
||
lastAssistantMsg?: string;
|
||
/** Session id; the wt_/lw_/ac_ prefixes select app-specific tool sets. */
|
||
sessionId: string;
|
||
/** Workspace path used for per-user skill toggles; null when unauthenticated. */
|
||
userWorkspace: string | null;
|
||
/** Injected so this module stays free of config/disk access. */
|
||
isSkillEnabledForUser: (slug: string, workspace: string | null) => boolean;
|
||
}
|
||
|
||
/** Boot (BOOT.md) turns get a read-only pair — they summarise state, they do not act. */
|
||
export const bootAllowedTools = new Set(['coder_list_files', 'coder_read_file']);
|
||
|
||
/**
|
||
* Code-editor sessions (code_ai_*) allow file/shell/search tools but block the ones that make
|
||
* no sense in an editor: browser automation, messaging, timers, presentations, image tools.
|
||
*/
|
||
export const codeAiBlockedTools = new Set(['start_task', 'task_control', 'write_note', 'browser_open', 'browser_snapshot', 'browser_click', 'browser_fill', 'browser_press_key', 'browser_wait', 'browser_close', 'image_search', 'image_url', 'image_edit', 'image_info', 'image_read', 'weather_search', 'weather_map_screenshot', 'start_timer', 'send_message', 'send_telegram', 'kakao_send_message', 'create_presentation', 'edit_presentation', 'spawn_subagent', 'schedule_job',
|
||
// desktop_* is Windows-only (desktop-tools.ts throws off process.platform); every call on
|
||
// this Linux server fails, same reasoning as dormantDesktopTools below for the main filter.
|
||
...(process.platform !== 'win32' ? ['desktop_screenshot', 'desktop_find_window', 'desktop_focus_window', 'desktop_click', 'desktop_drag', 'desktop_wait', 'desktop_press_key', 'desktop_type', 'desktop_set_clipboard', 'desktop_get_clipboard'] : [])]);
|
||
|
||
/**
|
||
* Builds the per-turn predicate. Returns a function of the tool NAME (not the whole schema
|
||
* object) so tests can drive it with plain strings.
|
||
*/
|
||
export function buildSkillToolFilter(input: ToolScopeInput): (toolName: string) => boolean {
|
||
const message = String(input.message || '');
|
||
// 후속 요청이 앞 턴의 주제를 물려받는 게이트에서만 쓴다. 아래 주석 참고.
|
||
const topicText = String(input.recentUserText || input.message || '');
|
||
const sessionId = String(input.sessionId || '');
|
||
const userWorkspace = input.userWorkspace;
|
||
|
||
// Weather comes in two tiers (2026-07-29). "내일 날씨" is an everyday question, not a
|
||
// specialist one, but all nine weather tools sat behind the meteorologist skill — so with the
|
||
// skill off a plain weather question had no weather tool at all and fell back to web_search.
|
||
// These two cover the ordinary case between them (weather_kma = Korean observations,
|
||
// weather_openmeteo = everywhere else, both with precipitation chance) for 446 tokens, and
|
||
// only when a weather keyword is present. They need no skill toggle.
|
||
const basicWeatherToolNames = new Set(['weather_kma', 'weather_openmeteo']);
|
||
// Air quality is a separate concern from the forecast, and weather_search is a redundant
|
||
// fallback whose schema costs as much as openmeteo while carrying less data — so they load
|
||
// on the weather keyword but are not part of the everyday pair above.
|
||
const weatherToolNames = new Set(['weather_search', 'weather_airpollution', 'weather_airkorea']);
|
||
// 2026-09-02 (efficiency audit): these four are climate-reanalysis/projection archives, not
|
||
// "what's the weather" tools — ERA5 and CDS are Copernicus reanalysis, CMIP6 is scenario
|
||
// projections, NASA POWER is agro-climate history. They were riding the plain weather keyword,
|
||
// so a casual "오늘 서울 날씨 어때?" carried 2,878 chars ≈ 820 tokens of archive API surface it
|
||
// could never use. Split onto their own keyword: a question about 기후/평년/과거/추세 is a
|
||
// different question from today's forecast, and phrasing it names the difference reliably.
|
||
const climateToolNames = new Set(['weather_nasa_power', 'weather_era5', 'weather_cds', 'weather_cmip6']);
|
||
// 2026-09-04 도달률 감사(실사용 1,011건): "메모리 가격 추세"가 이 게이트를 통과해 기후
|
||
// 재분석 아카이브 4종(~820토큰)을 통째로 실었다. 위 주석은 "기후/평년/과거/추세라는 표현이
|
||
// 차이를 확실히 짚어 준다"고 적었지만, 실제로는 "추세"와 "장기"가 기후와 아무 상관 없는 문장에
|
||
// 훨씬 자주 쓰인다(가격 추세, 장기 계획, 장기적으로…). 두 낱말만 기후·기상 명사와 같이 나올 때
|
||
// 걸리도록 좁힌다 — 나머지 키워드(기후/평년/재분석/era5 등)는 그 자체로 충분히 특정적이라 그대로 둔다.
|
||
const CLIMATE_SUBJECT = /기후|기온|날씨|강수|강우|우량|적설|기상|해수면|온난화|climate|weather|temperature|precipitation|rainfall/i;
|
||
const hasClimateKeyword =
|
||
/기후|기후변화|온난화|평년|과거\s*(기온|날씨|자료|데이터)|재분석|시나리오|연평균|기상\s*통계|climate|reanalysis|historical\s*(weather|climate)|era5|cmip|projection/i.test(topicText)
|
||
|| (/추세|장기/.test(topicText) && CLIMATE_SUBJECT.test(topicText));
|
||
const legalToolNames = new Set(['korean_law_search', 'korean_law_fetch', 'us_case_search']);
|
||
const pptxToolNames = new Set(['create_presentation', 'edit_presentation']);
|
||
// In practice these are only ever used while building an academic presentation (gather
|
||
// papers → create_presentation), not in standalone chat — gate them the same way as the
|
||
// pptx tools themselves rather than always loading 5 more schemas.
|
||
const academicToolNames = new Set(['pubmed_search', 'pubmed_fetch', 'pubmed_fulltext', 'openalex_search', 'semantic_search']);
|
||
// cms_hospital_compare — CMS Medicare hospital data (ownership / star rating / mortality /
|
||
// readmission). ~450-token schema, only relevant to US-hospital-performance questions, so
|
||
// gate it rather than carry it on every turn.
|
||
// drug_dur_check — 식약처 DUR(병용금기·임부금기 등). 닥터앱 탭에서는 무조건, 일반 챗에서는
|
||
// 약 이름·상호작용을 실제로 물을 때만. "약"은 너무 흔해서 단독으로 넣지 않고 문맥형으로만.
|
||
const hasDrugKeyword =
|
||
/병용금기|임부금기|수유부|복약|약물\s*상호작용|약\s*상호작용|같이\s*(먹어|복용|드셔)|함께\s*(먹어|복용)|DUR|처방전|약봉투|성분\s*겹|중복\s*처방|이\s*약(을|은|이|,|\s|의|도|과|와|\?)|먹는\s*약|드시는\s*약|복용\s*중(인|during)?\s*약|drug\s*interaction|contraindicat/i
|
||
.test(message);
|
||
// 2026-09-04: "미국의 영리의료법인 현황"·"영리의료법인의 법인세율?"이 이 게이트를 통과하지
|
||
// 못했다. 사용자가 실제로 쓴 말은 "병원"이 아니라 "의료법인"이었는데 목록에 그 낱말이 없었다
|
||
// ("영리 병원의 비율"만 우연히 걸렸다). 영리/비영리 소유형태를 묻는 순간이 바로 이 도구가
|
||
// 있는 이유라, 그 수식어에 붙는 기관 명사를 넓힌다.
|
||
const hasHospitalDataKeyword =
|
||
/\bCMS\b|메디케어|\bmedicare\b|care\s*compare|hospital\s*compare|hospital\s*(rating|ratings|star|quality|ownership|performance)|(병원|의료법인|의료기관|의료재단|요양기관).{0,6}(비교|평가|등급|순위|별점|소유|영리|비영리|성과|퀄리티|데이터|현황|비율)|(영리|비영리|투자자\s*소유).{0,4}(병원|의료법인|의료기관|의료재단|병상)|(for-?profit|non-?profit|nonprofit|investor-owned).{0,20}(hospital|health\s*system|medical\s*(center|corporation))/i
|
||
.test(message);
|
||
// coder_* (10 tools) is the file-authoring toolkit for actual code/script work — it has its
|
||
// own dedicated branch for isCodeAiSession (code.html) below, so this only affects the
|
||
// general chat path (main app + app-specific tabs like ac_/wt_/lw_/dental_/mn_, none of which
|
||
// are coding contexts). Ungated, it rode along on every single turn regardless of topic.
|
||
const coderToolNames = new Set(['coder_read_file', 'coder_write_file', 'coder_overwrite_lines', 'coder_insert_lines', 'coder_delete_lines', 'coder_str_replace', 'coder_patch_file', 'coder_delete_file', 'coder_move_file', 'coder_list_files']);
|
||
const hasCodeKeyword = /코드|스크립트|소스\s*코드|알고리즘|파이썬|python|javascript|typescript|자바스크립트|타입스크립트|프로그램|함수|클래스|디버그|버그|리팩터|refactor|\.py\b|\.js\b|\.ts\b|\.html\b|\.css\b|\.json\b|\.sh\b|\bcode\b|script|algorithm/i.test(message);
|
||
// Email/kakao/db/image-edit — same story as excel: no dedicated skill toggle or app tab,
|
||
// so gate purely on keyword. shell's own description tells the model to pair it with
|
||
// coder_write_file, so it rides the same code-intent gate rather than its own.
|
||
const emailToolNames = new Set(['email_list', 'email_read', 'email_send', 'email_search', 'email_delete']);
|
||
// 2026-07-29 bugfix (found by tests/tool-scope.test.ts). A Hangul syllable followed by a
|
||
// word-boundary anchor can NEVER match: \b is an ASCII boundary and Hangul is not an ASCII
|
||
// word character, so there is no boundary between 일 and a following space. "메일 확인해줘"
|
||
// therefore failed this gate outright and the email tools stayed hidden; only the separate
|
||
// "이메일" alternative ever fired. The same dead pattern was removed from hasCodeKeyword
|
||
// (함수/클래스/버그), hasNewsKeyword (기사) and hasPptxKeyword (덱) in the same pass — all of
|
||
// them were silently no-op alternatives. Dropping the anchor is safe for Korean: these words
|
||
// are unambiguous as substrings (카카오메일, 기사 등). Keep \b only on ASCII tokens like ppt\b.
|
||
const hasEmailKeyword = /이메일|메일|email|inbox|받은편지함/i.test(message);
|
||
// 2026-09-04 감사: `` 한 줄이 이 게이트를
|
||
// 통과해 kakao_send_message를 실었다. 사용자가 카카오맵 캡처를 올렸을 뿐인데 메신저 발송 도구가
|
||
// 붙는다 — 그리고 executeTool은 스키마 소속을 확인하지 않으므로([[project_tool_schema_leak]])
|
||
// 프롬프트가 이름만 언급해도 실제로 호출될 수 있는 종류의 도구다. 첨부 파일명은 사용자의 '요청'이
|
||
// 아니므로 판정에서 뺀다. 카카오'맵'도 메신저가 아니라 지도라 함께 제외한다.
|
||
const messageWithoutAttachments = message.replace(/!?\[[^\]]*\]\([^)]*\)/g, ' ');
|
||
const hasKakaoKeyword = /카카오(?!\s*맵)|kakao(?!map)/i.test(messageWithoutAttachments);
|
||
// USER.md/SOUL.md에 사실을 직접 적어 넣는 3종. 아래 게이트 주석 참고.
|
||
// 사용자가 기억을 **지시**하거나 **조회**할 때만 싣는다. topicText를 쓰는 이유는 "그것도
|
||
// 기억해둬" 같은 후속 발화가 앞 턴의 주제를 물려받기 때문이다.
|
||
const manualMemoryToolNames = new Set(['memory_browse', 'memory_write', 'memory_read']);
|
||
const hasMemoryIntentKeyword =
|
||
/기억(해|해둬|해\s*둬|해\s*줘|하고|해두|시켜|나|해야|할|한|났|안\s*나|못\s*해)|잊지\s*마|잊어버리지|외워|저장해|기록해|메모해|적어\s*둬|프로필|persona|USER\.md|SOUL\.md|\bremember\b|\bmemorize\b|don'?t forget|keep in mind|note that/i
|
||
.test(topicText);
|
||
const hasSqliteKeyword = /데이터베이스|sqlite|쿼리|query\b|\.db\b/i.test(message);
|
||
const hasImageEditKeyword = /편집|보정|자르기|리사이즈|크롭|워터마크|배경\s*제거|필터|흑백|그레이스케일|스타일화|crop|resize|watermark|remove.?bg|grayscale|stylize/i.test(message);
|
||
// news_search is the single biggest ungated schema (2,035 chars, mostly usage guidance for
|
||
// its country/category/language params) yet was included on every turn regardless of topic.
|
||
const hasNewsKeyword = /뉴스|시사|헤드라인|속보|기사|news|headline/i.test(topicText);
|
||
// image_read (OCR) only makes sense once an actual image is in play — the upload flow
|
||
// inserts attachments as markdown image links (), so detect that
|
||
// directly instead of guessing at phrasing like the other keyword gates.
|
||
const hasImageAttachment = /\.(png|jpe?g|webp|bmp|tiff?|gif)\b/i.test(message);
|
||
const hasRunCommandKeyword = /실행해|열어줘|띄워줘|런칭|실행시켜|메모장|notepad|vscode|vs\s*code|launch\s+(app|the|notepad|vscode)/i.test(message);
|
||
// Same shape as hasImageAttachment — pdf_read/pdf_extract_images (1,792 chars combined, the
|
||
// biggest remaining pair) only make sense once a PDF is actually referenced.
|
||
const hasPdfAttachment = /\.pdf\b|pdf\s*(파일|읽어|추출)/i.test(message);
|
||
const hasScheduleKeyword = /예약|스케줄|반복|매일|매주|매월|정기적|주기적|cron|schedule|recurring/i.test(message);
|
||
// eonet_events/satellite_snapshot — casual "what's happening on Earth" feature,
|
||
// not relevant to unrelated turns.
|
||
const hasSatelliteKeyword = /위성|자연재해|산불|태풍|화산|지진|산사태|가뭄|황사|연무|해빙|satellite|wildfire|volcano|eonet|natural\s*disaster/i.test(message);
|
||
// nvr_snapshot — 집 CCTV 한 프레임 캡처해 모델이 직접 확인. "마당에 누구 있어?", "택배 왔나
|
||
// 봐줘", "현관/주차장 확인", "고양이 있나" 같은 요청에만. 카메라/감시 어휘 + "확인해줘·봐줘"류
|
||
// 동사, 또는 알려진 카메라 위치명이 있을 때 연다.
|
||
//
|
||
// topicText(직전 2턴 포함)로 검사한다 — 실사고(2026-09-07): "cctv확인해봐"로 도구가 한 번
|
||
// 돌았는데, 바로 다음 "아니다 cam2" / "사진 보여줘" 후속턴엔 키워드가 없어 도구가 사라지자
|
||
// 모델이 "저는 그런 툴이 없습니다"라고 3턴 연속 답했다. 스테일 매치 비용은 작은 스키마 토큰뿐.
|
||
const hasCctvKeyword = /cctv|감시\s*카메라|카메라(로|에|를|\s|$)|웹캠|현관(?:문|앞|쪽)?|마당|주차장|차고|대문|택배|초인종|스냅샷|캡처해|누가?\s*(?:왔|있|지나)|누구\s*(?:있|왔)|밖에?\s*(?:누구|뭐|무슨)|집\s*(?:앞|밖|상황)|\bcam\s*\d|채널\s*\d.*(?:봐|보여|확인|찍)|door\s*cam|security\s*cam|who'?s?\s*(?:at\s*the\s*door|outside)/i.test(topicText);
|
||
// print3d_model — STL 슬라이싱+K2 프린터 업로드. topicText로 검사(prepare→start 확인
|
||
// 왕복이 여러 턴 걸치므로 nvr_snapshot과 같은 이유로 최근 대화 전체를 봄).
|
||
const hasPrint3dKeyword = /3d\s*프린트|3d\s*print|stl|슬라이싱|슬라이서|g-?code|출력해|프린터로|k2\s*(?:콤보|프린터)?/i.test(topicText);
|
||
// stl_cad/scad_to_stl — STL 확인/렌더링/구멍뚫기/scad 컴파일(OpenSCAD). print3d와 겹치는
|
||
// "stl" 외에 CAD 편집 관련 어휘도 추가로 매칭. 직전 어시스턴트 답변도 같이 봄(위
|
||
// lastAssistantMsg 주석 참고) — 짧은 "다시 시도해봐" 재시도에 도구가 사라지는 걸 막는다.
|
||
const stlCadPattern = /stl|cad|캐드|구멍\s*뚫|렌더링|바운딩박스|openscad|치수\s*확인/i;
|
||
const hasStlCadKeyword = stlCadPattern.test(topicText) || stlCadPattern.test(String(input.lastAssistantMsg || '').slice(-800));
|
||
// workshop_project — "작업실"(여러 메이커 프로젝트) 대시보드 부품/작업 편집. "로봇" 단독은
|
||
// 로봇청소기(Valetudo) 대화와 겹치므로 프로젝트 고유 어휘로만 매칭한다. 새 프로젝트 종류가
|
||
// 늘어날 때마다 이 목록도 넓혀줘야 함(앱 세션 안에서는 아래 isWorkshopAppSession이 대신
|
||
// 커버하므로 메인 채팅에서 키워드 없이 말할 때만 이 게이트가 문제가 됨).
|
||
const hasWorkshopKeyword = /작업실\s*프로젝트|로봇\s*프로젝트|메카넘|mecanum|로봇팔|robot\s*arm|so-?101|so-?arm|rplidar|라이다.*로봇|로봇.*(?:부품|예산|작업|단계|대시보드)|esp32|하우징/i.test(topicText);
|
||
// 2026-08-10: news_search was gated on the literal word "뉴스" (hasNewsKeyword below), so a
|
||
// disaster question with no news-flavored wording ("돌핀 지금 어디 있지?") never got access to
|
||
// it — even after handle-chat.ts's reminder started explicitly recommending news_search for
|
||
// exactly this case (real incident: search for the storm's position turned up nothing useful,
|
||
// while the actual story — landfall in China, evacuations, damage — was straightforward news
|
||
// coverage). Reuses prompt-gates.ts's DISASTER_PATTERN rather than keeping a second copy of
|
||
// the same keyword list here, per this file's own header note on gates needing to agree.
|
||
const hasDisasterKeyword = DISASTER_PATTERN.test(topicText);
|
||
// 2026-09-02: the per-user skills_state.json toggles are legacy — the UI moved to per-app
|
||
// gating and nobody maintains these flags any more. They were still the only load-bearing
|
||
// consumer of that state, and stale values were silently WITHHOLDING tools: papa has
|
||
// meteorologist:false, so "오늘 서울 미세먼지 어때?" got weather_kma/openmeteo only —
|
||
// and weather_kma carries no air-quality data at all, so the answer could only be a
|
||
// failure or an invention. weather_airkorea (에어코리아, key approved and working) was
|
||
// sitting right there, gated off by a flag from a UI the user no longer uses.
|
||
//
|
||
// The keyword gates below are the real cost control and they are already narrow. A toggle
|
||
// that is never updated is not a safety mechanism, it is a trap — so the skill condition is
|
||
// dropped and the keyword alone decides. `isSkillEnabledForUser` stays on ToolScopeInput for
|
||
// the app-session gates and for whenever skills get real prompt files again.
|
||
const hasPptxKeyword = /슬라이드|발표|pptx|ppt\b|프레젠테이션|피피티|덱|presentation/i.test(message);
|
||
// 2026-08-25: 논문 도구는 pptx 키워드에만 묶여 있어서, "논문 찾아줘"로는 아예 안 켜지고
|
||
// 모델이 web_search로 때웠다(실사용 로그에서 확인). 학술 검색은 web_search보다 확실히
|
||
// 낫다 — openalex_search는 키 없이 2억 건에서 제목·저자·연도·저널·DOI·인용수·오픈액세스
|
||
// 여부를 구조화해 준다. 그래서 학술 의도 자체를 게이트로 추가한다.
|
||
// 한글 뒤에는 \b를 쓰지 않는다(ASCII 경계라 절대 매칭 안 됨 — 위 hasEmailKeyword 주석 참고).
|
||
const hasAcademicKeyword = /논문|페이퍼|학술|저널|문헌|선행연구|인용수|피인용|초록|아카이브|학회지|\bpapers?\b|\bpubmed\b|\bdoi\b|arxiv|scholar|citations?|journals?|literature/i.test(message);
|
||
// Weather/legal skills being ON means the user does that kind of work sometimes, not that
|
||
// every single turn is about it — without a keyword gate the full 9 weather / 3 legal tools
|
||
// rode along on unrelated turns too. Mirrors the pptx pattern (skill + keyword) below.
|
||
// 기상 현상 목록. "비"·"눈"처럼 다른 뜻과 겹치는 짧은 단어가 섞여 있으므로, 단독으로 쓰지 말고
|
||
// 반드시 아래처럼 물음 형태("~소식/예보/전망/상황")와 짝지어서만 쓸 것.
|
||
const WEATHER_PHENOMENON = '비|눈|태풍|장마|한파|폭염|황사|미세먼지|우박|안개|서리|더위|추위|바람|구름|강수|기온|날씨';
|
||
|
||
// 2026-08-12 실사고: "향후 비소식은 없나?"에 이 목록이 하나도 안 걸려 weather_kma·
|
||
// weather_openmeteo가 스키마에서 통째로 빠졌고, 모델은 날씨 도구가 없으니 web_search로
|
||
// 갔다가 2019년 기사를 물어와 "예보를 확인할 수 없다"고 답했다. 이어서 사용자가
|
||
// "기상청에 알아봐"라고 명시했는데 그 말조차 목록에 없어서 두 번째 턴도 똑같이 실패했다.
|
||
// 대화 전체가 날씨 도구 없이 돈 것이다.
|
||
//
|
||
// 이 게이트의 비용 비대칭은 프롬프트 게이트와 반대다 — 오탐은 도구 스키마 몇백 토큰이
|
||
// 늘 뿐이지만, 미탐은 답변 경로 자체가 틀린다. 그러니 여기서는 넉넉하게 잡는다.
|
||
//
|
||
// 실사용 메시지 103건으로 교정([[feedback_calibrate_on_real_data]]): "기상청" 신규 2건
|
||
// (전부 진짜 날씨), "비" 문맥형 신규 2건(전부 진짜 날씨), 오탐 0건. "비"는 단독으로 넣으면
|
||
// 비용·비교·준비에 걸리므로 반드시 문맥형으로만 잡는다("비소식"처럼 붙여 쓰는 것도 포함).
|
||
const hasWeatherKeyword = /날씨|기온|강수|미세먼지|대기질|태풍|기후|일기예보|장마|폭염|한파|자외선|황사|weather|forecast/i.test(topicText)
|
||
|| /기상청|예보|기상\s*상황/.test(topicText)
|
||
// "OO 소식 있나?"는 날씨를 묻는 가장 흔한 말투인데, 현상별로 흩어 놓으면 다음 현상에서
|
||
// 또 빠진다(우박·안개·서리가 실제로 빠져 있었다). 현상 목록 × 물음 형태로 한 번에 잡는다.
|
||
// "소식" 단독은 절대 넣지 않는다 — 회사 소식·업계 소식·새로운 소식에 다 걸린다.
|
||
|| new RegExp(`(${WEATHER_PHENOMENON})\\s*(소식|예보|전망|상황)`).test(topicText)
|
||
|| /비가\s*(오|와|내리|올)|비\s*(온다|와요|올까|내려)|소나기|호우|폭우|우산/.test(topicText)
|
||
|| /눈이\s*(오|내리)|눈\s*올/.test(topicText)
|
||
|| /무더위|더위|추위|쌀쌀|습도|풍속|체감\s*온도/.test(topicText);
|
||
const hasLegalKeyword = /법률|판례|소송|변호사|법원|법조문|조항|계약서|형법|민법|법령|legal|lawsuit|court|statute/i.test(message);
|
||
// excel_read/excel_write have no dedicated skill toggle (accountant-app.html's persona is
|
||
// page-local, not a registered skill) — gate on the accountant app tab (ac_ sessions) or an
|
||
// explicit excel/spreadsheet keyword, same shape as the weather/legal gates above. Previously
|
||
// ungated, so both tool schemas rode along on every single chat turn regardless of app.
|
||
const hasExcelKeyword = /엑셀|xlsx|xls\b|스프레드시트|spreadsheet|excel/i.test(message);
|
||
// ollama_web_search/ollama_web_fetch are explicitly self-deprecating in their own descriptions
|
||
// ("web_search가 없는 환경에서만 fallback으로 사용") — and the provider fallback they exist for
|
||
// is already handled server-side inside executeWebSearch, so the model never needs to reach for
|
||
// them on its own. Advertising them cost 214 tokens per turn to tell the model not to use a
|
||
// tool. Gate on the same explicit phrasing personality-context.ts already uses for its
|
||
// ollama_web hint block, so the pair stays available when the user actually asks for it by name
|
||
// (otherwise that hint would describe tools no longer in the schema).
|
||
const ollamaWebToolNames = new Set(['ollama_web_search', 'ollama_web_fetch']);
|
||
const hasOllamaWebKeyword = /ollama[\s_]?(웹|web|검색|search)|올라마\s*(웹|검색|로\s*검색)/i.test(message);
|
||
// Sub-agent/task-orchestration tools that tool_audit.log shows near-zero real usage for
|
||
// (2026-05-04~2026-07-12: start_task/request_secondary_assist/subagent_spawn/
|
||
// delegate_to_specialist/spawn_subagent = 0 calls; task_control/spawn_agent = a handful, all
|
||
// stale). Kept registered for future agent-composition work,
|
||
// just not sent to the model in normal chat. schedule_job is excluded from this set — it's
|
||
// self-contained (list/create/update/pause/resume/delete/run_now) and still actively used.
|
||
const dormantAgentTools = new Set(['start_task', 'task_control', 'request_secondary_assist', 'subagent_spawn', 'delegate_to_specialist', 'spawn_subagent', 'spawn_agent']);
|
||
// Browser automation — parked. tool_audit.log shows every browser_open call from 07-09
|
||
// onward failing with "Chrome launched but did not respond on port 9222". Repeated debug
|
||
// runs showed the live `tools` array alternates between two mutually-exclusive sets for the
|
||
// SAME "ping" request against the SAME running process — open/snapshot/click/fill/press_key/
|
||
// wait/close one time, scroll/get_images the next — apparently some live Chrome-reachability
|
||
// probe swapping which schema set gets exposed. Excluding all 9 names covers both variants.
|
||
// Re-enable by removing this set once browser automation is fixed / needed (e.g. home shopping).
|
||
const dormantBrowserTools = new Set(['browser_open', 'browser_snapshot', 'browser_click', 'browser_fill', 'browser_press_key', 'browser_wait', 'browser_close', 'browser_scroll', 'browser_get_images']);
|
||
// desktop_* (10 tools, ~2.7k of the tool-schema overhead) is Windows-only remote-desktop
|
||
// control (mouse/keyboard/screenshot) — desktop-tools.ts itself throws "Desktop tools are
|
||
// currently supported on Windows only." off process.platform, and tool_audit.log shows every
|
||
// desktop_screenshot call ever made on this server failing with exactly that. Mirror the same
|
||
// platform check here so the schemas aren't sent on a platform where every call is guaranteed
|
||
// to fail (auto-recovers if this ever runs on Windows, unlike a hardcoded exclude).
|
||
const dormantDesktopTools = process.platform !== 'win32'
|
||
? new Set(['desktop_screenshot', 'desktop_find_window', 'desktop_focus_window', 'desktop_click', 'desktop_drag', 'desktop_wait', 'desktop_press_key', 'desktop_type', 'desktop_set_clipboard', 'desktop_get_clipboard'])
|
||
: new Set<string>();
|
||
// MCP database servers built for a specific dedicated app tab (dental-agent.html →
|
||
// 'dental_' sessions, mind-app.html → 'mn_' sessions) — 62% of the tool-schema overhead
|
||
// (12,124 of 19,531 tokens measured) came from these 3 servers × 8 sub-tools each, riding
|
||
// along on every main-chat turn for every user even though they're single-purpose lookup
|
||
// tables for one app. Scope them to the app session that actually uses them.
|
||
const isDentalAppSession = /^dental_/.test(String(sessionId || ''));
|
||
const isMindAppSession = /^mn_/.test(String(sessionId || ''));
|
||
// Weather/lawyer have their own dedicated app tabs (weather-app.html → 'wt_' sessions,
|
||
// lawyer-app.html → 'lw_' sessions). Inside those apps the skill+keyword gate below is
|
||
// wrong — a "wt_" session IS the weather context even if a given message has no weather
|
||
// keyword (e.g. a follow-up "내일은?"), so always include the tools there.
|
||
const isWeatherAppSession = /^wt_/.test(String(sessionId || ''));
|
||
const isLawyerAppSession = /^lw_/.test(String(sessionId || ''));
|
||
const isAccountantAppSession = /^ac_/.test(String(sessionId || ''));
|
||
// 2026-09-02: two more app tabs ship a session prefix but never got a gate, so inside them a
|
||
// natural follow-up ("그럼 작년은?", "이 사건 자료 더 찾아줘") fell back to main-chat keyword
|
||
// matching and came up with no specialist tool at all.
|
||
// detective-app.html → 'dt_': 판례 12회 / 법률 11회 언급. Same legal tool family as the
|
||
// lawyer tab — it is a case-research app, so it needs 국가법령정보센터/CourtListener.
|
||
// investor-app.html → 'iv_': 환율 6회, 주가·증시·엑셀·뉴스. Market questions are live-data
|
||
// questions; without news_search a "작년 대비" follow-up has nothing to read.
|
||
// doctor-app.html → 'dr_' (2026-09-02, 사용자 요청): 개인 의료기록 정리 앱이지만 "이 약
|
||
// 장기복용 괜찮나?", "이 백신 효과 몇 년?", "이 수치 정상 범위인가?" 류 질문에 지금은
|
||
// web_search로 블로그를 물어온다. 학술 도구(PubMed/OpenAlex/Semantic Scholar)를 이 탭
|
||
// 안에서만 열어 근거 문헌으로 답하게 한다. 앱 상단 면책("진단·처방 대체 아님")은 유지되고,
|
||
// 세션 id는 dr_main / dr_case_<id> / dr_<rand> 모두 dr_ 접두사라 한 번에 잡힌다.
|
||
// language-app ('lg_') is deliberately NOT here — a tutor served fine by the base tool set.
|
||
const isDetectiveAppSession = /^dt_/.test(String(sessionId || ''));
|
||
const isInvestorAppSession = /^iv_/.test(String(sessionId || ''));
|
||
const isDoctorAppSession = /^dr_/.test(String(sessionId || ''));
|
||
// workshop-app.html의 전용 채팅 패널('ws_' 세션) 안에서는 그 대화 전체가 이미 어떤
|
||
// 프로젝트 얘기라는 게 확실하므로, 키워드 없이 짧게 말해도("이 부품 지워줘") workshop_project가
|
||
// 열리게 함(원래 robot-app.html/'rb_' 세션이었던 것을 여러 프로젝트로 일반화, 2026-09-20).
|
||
const isWorkshopAppSession = /^ws_/.test(String(sessionId || ''));
|
||
const filter = (name: string): boolean => {
|
||
if (dormantAgentTools.has(name)) return false;
|
||
if (dormantBrowserTools.has(name)) return false;
|
||
if (dormantDesktopTools.has(name)) return false;
|
||
if (name.startsWith('mcp__dental-dict-sqlite__') && !isDentalAppSession) return false;
|
||
if ((name.startsWith('mcp__psychotherapy-cases-sqlite__') || name.startsWith('mcp__psychiatry-cases-sqlite__')) && !isMindAppSession) return false;
|
||
if (basicWeatherToolNames.has(name) && !isWeatherAppSession && !hasWeatherKeyword) return false;
|
||
if (weatherToolNames.has(name) && !isWeatherAppSession && !hasWeatherKeyword) return false;
|
||
if (climateToolNames.has(name) && !isWeatherAppSession && !hasClimateKeyword) return false;
|
||
// Windy map screenshots only make sense inside the dedicated weather app tab (the
|
||
// model there has already generated the embed.windy.com URL to capture) — unlike the
|
||
// other weather tools, not offered via the meteorologist skill in general chat.
|
||
//
|
||
// Exception (2026-08-10): the KMA/JMA typhoon-track capture path (KMA_TYPHOON_TRACK_URL /
|
||
// JMA_TYPHOON_MAP_URL in weather.ts) doesn't depend on a prior Windy embed URL the way the
|
||
// radar/satellite path does — it's a fixed, always-valid URL. Real incident: a main-chat
|
||
// "태풍 돌핀 지금 어디 있지?" had no access to this tool at all, so the model fell back to
|
||
// web_search with a weak self-generated query and missed the actual news. The KMA/JMA
|
||
// track page is the authoritative source for exactly this question and was sitting behind
|
||
// a gate that had nothing to do with typhoons — same hasSatelliteKeyword gate that already
|
||
// covers eonet_events/satellite_snapshot below, since "태풍" is in that keyword list.
|
||
if (name === 'weather_map_screenshot' && !isWeatherAppSession && !hasSatelliteKeyword) return false;
|
||
if (name === 'nhc_active_storms' && !isWeatherAppSession) return false;
|
||
if (ollamaWebToolNames.has(name) && !hasOllamaWebKeyword) return false;
|
||
if (legalToolNames.has(name) && !isLawyerAppSession && !isDetectiveAppSession && !hasLegalKeyword) return false;
|
||
if ((name === 'excel_read' || name === 'excel_write') && !isAccountantAppSession && !isInvestorAppSession && !hasExcelKeyword) return false;
|
||
if (coderToolNames.has(name) && !hasCodeKeyword) return false;
|
||
if (name === 'shell' && !hasCodeKeyword) return false;
|
||
// 2026-07-29 bugfix (found by tests/tool-scope.test.ts on extraction): the condition used to
|
||
// be `presenterEnabled && !hasPptxKeyword`, i.e. the exclusion only ran when the presenter
|
||
// skill was ENABLED. With the skill OFF the gate never fired at all and both schemas rode
|
||
// along on every turn — inverted, since a disabled skill should restrict more, not less.
|
||
// Scope of the actual impact (checked against skills_state.json the same day): papa and
|
||
// cherry have presenter=true so they were unaffected; jasmine has it false and was carrying
|
||
// ~1,600 tokens of presentation API surface (create_presentation alone is 833, the largest
|
||
// tool schema in the codebase) on every turn including plain greetings.
|
||
// Now matches the weather/legal shape directly above: exclude unless skill AND keyword agree.
|
||
if (pptxToolNames.has(name) && !hasPptxKeyword) return false;
|
||
if (academicToolNames.has(name) && !isDoctorAppSession && !hasPptxKeyword && !hasAcademicKeyword) return false;
|
||
if (name === 'cms_hospital_compare' && !hasHospitalDataKeyword) return false;
|
||
if (name === 'drug_dur_check' && !isDoctorAppSession && !hasDrugKeyword) return false;
|
||
if (emailToolNames.has(name) && !hasEmailKeyword) return false;
|
||
if (name === 'kakao_send_message' && !hasKakaoKeyword) return false;
|
||
// USER.md/SOUL.md 수동 메모리 3종. 2026-09-04 효율 감사로 게이팅.
|
||
//
|
||
// 근거는 순수하게 사용량이다: 실사용 1,011턴에서 browse 0회 / read 2회 / write 1회인데
|
||
// **매 턴 무조건** 실려 357토큰을 냈다. 게이트가 걸리면 0.3%의 턴에만 실린다.
|
||
//
|
||
// 도구 자체는 멀쩡하다. USER.md·SOUL.md는 `workspace/prompts/` 아래에 실제로 있고
|
||
// resolvePromptPath가 거기서 찾는다(papa의 USER.md는 3.2KB, 카테고리 5개). 감사 중에
|
||
// 워크스페이스 루트만 보고 "파일이 없다"고 잘못 판단했다가 정정한 자리다 — 셋 다
|
||
// 파일이 없으면 에러를 내므로, 없다는 판단이 맞았다면 통째로 죽은 도구였을 뻔했다.
|
||
//
|
||
// 제거가 아니라 게이팅인 이유: 사용자가 "기억해둬"라고 명시하는 턴에는 반드시 있어야 한다.
|
||
// 평상시 장기기억은 일별 로그 + Chroma 자동 추출이 담당하지만([[project_vector_memory_chroma]]),
|
||
// 그건 하루 뒤에 반영되는 배치라 "지금 이걸 기억해"의 대체재가 아니다.
|
||
// memory_stats는 벡터 스토어를 보므로 게이트 대상이 아니다(실제로 13회 호출됨).
|
||
if (manualMemoryToolNames.has(name) && !hasMemoryIntentKeyword) return false;
|
||
if (name === 'sqlite_query' && !hasSqliteKeyword) return false;
|
||
if (name === 'image_edit' && !hasImageEditKeyword) return false;
|
||
if (name === 'image_read' && !hasImageAttachment) return false;
|
||
if (name === 'news_search' && !isInvestorAppSession && !isDetectiveAppSession && !hasNewsKeyword && !hasDisasterKeyword) return false;
|
||
if ((name === 'pdf_read' || name === 'pdf_extract_images') && !hasPdfAttachment) return false;
|
||
if (name === 'schedule_job' && !hasScheduleKeyword) return false;
|
||
if ((name === 'eonet_events' || name === 'satellite_snapshot') && !hasSatelliteKeyword) return false;
|
||
if (name === 'nvr_snapshot' && !hasCctvKeyword) return false;
|
||
if (name === 'print3d_model' && !hasPrint3dKeyword) return false;
|
||
if ((name === 'stl_cad' || name === 'scad_to_stl' || name === 'write_scad') && !hasStlCadKeyword) return false;
|
||
if (name === 'workshop_project' && !isWorkshopAppSession && !hasWorkshopKeyword) return false;
|
||
if (name === 'run_command' && !hasRunCommandKeyword) return false;
|
||
return true;
|
||
};
|
||
return filter;
|
||
}
|
||
|
||
/**
|
||
* Applies the right tool set for the turn. The four branches are mutually exclusive and
|
||
* ordered by specificity: boot < inert sessions < code editor < normal chat.
|
||
*/
|
||
export function selectToolsForTurn(
|
||
allTools: any[],
|
||
input: ToolScopeInput,
|
||
flags: { isBootStartupTurn: boolean; isCodeAiSession: boolean; isTranslateSession: boolean; isProjSession: boolean },
|
||
): any[] {
|
||
const nameOf = (t: any) => String(t?.function?.name || '');
|
||
if (flags.isBootStartupTurn) return allTools.filter(t => bootAllowedTools.has(nameOf(t)));
|
||
// Translation and project-file sessions answer from the prompt alone; handing them tools only
|
||
// invites a stray call that corrupts a strict output format.
|
||
if (flags.isTranslateSession || flags.isProjSession) return [];
|
||
if (flags.isCodeAiSession) return allTools.filter(t => !codeAiBlockedTools.has(nameOf(t)));
|
||
const filter = buildSkillToolFilter(input);
|
||
return allTools.filter(t => filter(nameOf(t)));
|
||
}
|