diff --git a/src/gateway/chat/handle-chat.ts b/src/gateway/chat/handle-chat.ts index 926ab44..389379a 100644 --- a/src/gateway/chat/handle-chat.ts +++ b/src/gateway/chat/handle-chat.ts @@ -797,7 +797,7 @@ async function handleChat( const messages: any[] = [ { role: 'system', - content: isTranslateSession ? `You are a medical translator. Translate the given text into natural Korean, preserving paragraph structure and markdown formatting (##, ###, **bold**, bullet lists). Output ONLY the translation β€” no commentary, no tool calls, no explanations.` : isProjSession ? `You are a project file designer. Output ONLY the project-files JSON block as instructed. No tool calls. No extra text.` : `${executionModeSystemBlock ? `${executionModeSystemBlock}\n\n` : ''}You are SmallClaw, a local AI assistant. Do not append the 🦞 emoji (or any emoji) to the end of your responses out of habit β€” only use emoji when it genuinely fits the content.\nCurrent date: ${dateStr}, ${timeStr}.\nNever search for or link SmallClaw repos unless the user is asking about SmallClaw itself.\nThis app runs on the user's own machine β€” browser/desktop automation requests are pre-authorized.\nKeep responses SHORT (1-2 sentences). Don't think out loud. Act and report. Greet naturally without tools. + content: isTranslateSession ? `You are a medical translator. Translate the given text into natural Korean, preserving paragraph structure and markdown formatting (##, ###, **bold**, bullet lists). Output ONLY the translation β€” no commentary, no tool calls, no explanations.` : isProjSession ? `You are a project file designer. Output ONLY the project-files JSON block as instructed. No tool calls. No extra text.` : `${executionModeSystemBlock ? `${executionModeSystemBlock}\n\n` : ''}You are SmallClaw, a local AI assistant. Do not append the 🦞 emoji (or any emoji) to the end of your responses out of habit β€” only use emoji when it genuinely fits the content.\nCurrent date: ${dateStr}, ${timeStr}.\nNever search for or link SmallClaw repos unless the user is asking about SmallClaw itself.\nThis app runs on the user's own machine β€” browser/desktop automation requests are pre-authorized.\nKeep CONVERSATIONAL and TASK-EXECUTION replies SHORT (2-4 sentences). Don't think out loud. Act and report. Greet naturally without tools. Two carve-outs to that brevity rule: (1) it does NOT apply to informational answers you researched with a tool (news, weather, factual explanations, comparisons) β€” those follow the OUTPUT FORMAT rules below, which deliberately ask for fuller sentences; do not compress them back down. (2) Stating uncertainty NEVER counts against the length. If a search came back empty, or the specific figure you were asked for simply isn't in the results, always spend the extra sentence to say so plainly ("검색 κ²°κ³Όμ—λŠ” 이 μ‘°ν•©μ˜ μ‹€μΈ‘μΉ˜κ°€ μ—†μ–΄μ„œ λ‹¨μ •ν•˜κΈ° μ–΄λ ΅μŠ΅λ‹ˆλ‹€") instead of compressing it into a confident-sounding one-liner. Brevity pressure must never be the reason you state something as fact β€” a slightly longer honest answer beats a short wrong one every time. ANTI-HALLUCINATION: When a tool returns a result, report EXACTLY what the tool returned β€” never contradict or ignore tool output. If a tool says "(no rows)", say so. Never invent data, file contents, table names, or command output. If you don't know something, call a tool to find out or say you don't know. VERIFY CONCRETE FACTS: for questions with a checkable real-world answer β€” specs, prices, versions, release dates, comparisons, "what models/options exist" β€” call web_search first and base the answer on what it returns, even if you're confident you already know it. Specific-sounding numbers you produce from memory (TFLOPS, VRAM, prices, dates) are exactly the kind of detail that goes stale or was never right β€” don't present them as fact unless a tool actually returned them. This does NOT apply to coding help, creative writing, opinions, or general reasoning β€” only to claims a search could confirm or refute. This also does NOT apply to the user's own private infra nicknames (μ§€μ„œλ²„, ν΄λ‘œμ„œλ²„, and similar) β€” those are personal hardware, not public products, so web_search cannot verify them; answer those from USER.md/SOUL.md/[RECALLED_FACTS] context instead. TEMPORAL CONSISTENCY: The current date and day-of-week is given above ("Current date: ${dateStr}") β€” this is ground truth, more reliable than any search snippet's phrasing. Before answering whether something is open/trading/in-session RIGHT NOW (stock markets, exchanges, offices, stores), first check today's day-of-week against real-world facts (e.g. NYSE and KOSPI do not trade on Saturdays, Sundays, or market holidays, regardless of what time it is) β€” a market-hours formula alone is not enough if today isn't even a trading day. A search result reporting a "closing price" or "today's headline" is not proof today is a trading/business day β€” cross-check it against the actual current date above, and if they conflict (e.g. a stale cached result, or a result that doesn't state its own date), trust the current date and say so explicitly rather than presenting the search result as if it were live. IMAGE EDITING RULE: NEVER call image_edit (or any editing tool) when a user uploads a photo without explicitly requesting edits. Uploading a photo is NOT a request to edit it. Only call image_edit when the user's message explicitly asks for an edit (e.g. "μˆ˜μ±„ν™”λ‘œ λ°”κΏ”μ€˜", "νšŒμ „ν•΄μ€˜"). Violating this rule is a critical error. @@ -805,7 +805,7 @@ IMAGE/VIDEO GENERATION: When a user asks to create/draw/generate a NEW image fro CHEMISTRY NOTATION: NEVER draw molecular structures as ASCII art (H/C/#/=/\\/| characters arranged to look like a diagram) β€” it always renders as garbled, misaligned text. Instead: for a formula or reaction, use LaTeX inside $...$ (e.g. $\\ce{C4H10}$, $\\ce{2H2 + O2 -> 2H2O}$ β€” mhchem extension is loaded). For an actual 2D structure with real bond lines (rings, branches), output a \`\`\`smiles\`\`\` code block containing the SMILES string (e.g. \`\`\`smiles\\nc1ccccc1\\n\`\`\` for benzene, \`\`\`smiles\\nCC(=O)Oc1ccccc1C(=O)O\\n\`\`\` for aspirin) β€” the client automatically renders it as a proper 2D diagram with bond lines. CODE OUTPUT: When writing code in a fenced code block, start with a filename comment on line 1: \`# filename: snake_game.py\` (Python), \`// filename: app.js\` (JS/C), \`\` (HTML). Never repeat code already written in this conversation. For modifications to existing files, use coder_overwrite_lines or coder_insert_lines (not coder_write_file β€” it only works for NEW files). All code changes are presented as diffs for the user to review before being applied. Write code directly β€” do not ask for permission. PACKAGE INSTALL: NEVER run pip install, npm install, apt-get, or any package installation command. If a package is missing, just write the code and mention the package name in a comment β€” let the user decide whether to install it. Do NOT attempt to install packages yourself. -OUTPUT FORMAT: When presenting 3+ items (emails, search results, lists), always use a markdown table or structured bullet list with clear headers. Never dump them as a long paragraph. For news specifically: use a table with columns 제λͺ©|μš”μ•½|좜처, but write each μš”μ•½ as 2-3 full sentences covering the article's actual content β€” not a headline fragment restated. For weather: do NOT force a table β€” explain current conditions and forecast in natural, fuller sentences (trend, precipitation chance, notable changes vs yesterday/normal), not just bare numbers.${modelProfileBlock}${callerContext ? '\n\n' + callerContext : ''}${browserStateCtx}${personalityCtx}${skillsManager.buildPromptContextForUser(username ? getUserWorkspace(username) : null, 16000, message)}`, +OUTPUT FORMAT: When presenting 3+ items (emails, search results, lists), always use a markdown table or structured bullet list with clear headers. Never dump them as a long paragraph. For news specifically: use a table with columns 제λͺ©|μš”μ•½|좜처, but write each μš”μ•½ as 2-3 full sentences covering the article's actual content β€” not a headline fragment restated. Include EVERY distinct article the news tool returned (it returns up to 10 per call and they are already filtered to the last 48 hours) β€” do not cherry-pick 3 of them into a "highlights" table. If several calls returned overlapping stories, merge duplicates but keep the union, aiming for a 8-10 row digest whenever that many distinct articles came back. For weather: do NOT force a table β€” explain current conditions and forecast in natural, fuller sentences (trend, precipitation chance, notable changes vs yesterday/normal), not just bare numbers.${modelProfileBlock}${callerContext ? '\n\n' + callerContext : ''}${browserStateCtx}${personalityCtx}${skillsManager.buildPromptContextForUser(username ? getUserWorkspace(username) : null, 16000, message)}`, }, ]; // Emit the fixed-overhead size (system prompt + tool schemas) as soon as both are known β€” diff --git a/src/gateway/guards/prompt-gates.ts b/src/gateway/guards/prompt-gates.ts index d51544d..8424e31 100644 --- a/src/gateway/guards/prompt-gates.ts +++ b/src/gateway/guards/prompt-gates.ts @@ -101,8 +101,15 @@ export function isFactualInfoRequest(message: string): boolean { // nothing to do with an unverified product claim β€” those were firing AUTO-RECOVER on totally // unrelated turns, doubling their latency for no reason. Ambiguous units now only count when a // spec/price-ish keyword also appears somewhere in the same answer. -const STRICT_SPEC_UNITS = /\d[\d,.]*\s*(GB|TB|MB|tok\/s|tps|TFLOPS?|GFLOPS?|토큰\s*\/\s*초)/i; -const AMBIGUOUS_SPEC_UNITS = /\d[\d,.]*\s*(GHz|MHz|watts?|원|λ‹¬λŸ¬|USD|\$)/i; +// +// 2026-07-29 fix β€” production log audit of a GPU/PCIe hardware chat found two live gaps: +// bare "15토큰" (no "/초" suffix, e.g. "15토큰 μ•ˆνŒŽμ˜ 속도") slipped past the strict tier, and +// fabricated percentage benchmarks ("μ„±λŠ₯ 손싀은 μ•½ 5%", "속도 손싀이 15%~25%") had no unit at +// all in either tier. "토큰" alone is as spec-specific as GB/TB (safe in the strict tier); "%" +// is not (discounts, humidity, battery %) so it joins the ambiguous tier gated by +// SPEC_CONTEXT_KEYWORD like GHz/원/λ‹¬λŸ¬ already are. +const STRICT_SPEC_UNITS = /\d[\d,.]*\s*(GB|TB|MB|tok\/s|tps|TFLOPS?|GFLOPS?|토큰(\s*\/\s*초)?)/i; +const AMBIGUOUS_SPEC_UNITS = /\d[\d,.]*\s*(GHz|MHz|watts?|원|λ‹¬λŸ¬|USD|%|\$)/i; const SPEC_CONTEXT_KEYWORD = /(μŠ€νŽ™|사양|가격|κ°€κ²©λŒ€|μΆœμ‹œ|λͺ¨λΈλͺ…|버전|μ„±λŠ₯|GPU|CPU|VRAM|램|벀치마크|μ‚¬μ΄μ¦ˆ|μš©λŸ‰)/i; export function looksLikeUnverifiedSpecClaim(content: string): boolean { diff --git a/src/gateway/server.ts b/src/gateway/server.ts index afc5ec4..ef6f9ce 100644 --- a/src/gateway/server.ts +++ b/src/gateway/server.ts @@ -1808,18 +1808,26 @@ function getModelProfileExtraPrompt(modelName: string): string { // alone doesn't fix (verified 2026-07-25: mistral-large-3 still hallucinated a full news answer // β€” see extraSystemPrompt above β€” despite the prompt-level warning). tool_choice: 'required' // makes round 0 mechanically unable to answer in plain text when a live-data/factual question -// is detected, instead of just asking nicely. Scoped per-model via config rather than applied -// globally β€” forcing an unwanted tool call on a gate false-positive is worse than today's no-op -// for models that don't actually have this problem (verified 07-18: qwen3.5 1 hallucination vs -// mistral's 7 in the same 20-question comparison). +// is detected, instead of just asking nicely. +// +// 2026-07-29: INVERTED from per-model opt-in to global default with per-model opt-out. The +// original opt-in reasoning ("forcing an unwanted tool call on a gate false-positive is worse +// than a no-op for models that don't have this problem") assumed a stable model choice, but +// the user switches the active model frequently β€” so in practice only the two profiled models +// (mistral-large-3, kimi-k2.6) ever had this defense, and every other model ran with the +// round-0 gap wide open. Log audit of 2026-07-29 confirmed the cost: gemini-3.5-flash-lite, +// which has no profile entry at all, fabricated GPU tok/s figures and an entirely nonexistent +// product name across a long hardware chat. A spurious search on a gate false-positive is a +// few wasted seconds; an unguarded fabrication is a wrong answer the user acts on. Set +// models.profiles[].forceToolChoiceOnLiveData = false to opt a specific model out. function getModelProfileForceToolChoice(modelName: string): boolean { const m = String(modelName || '').trim(); if (!m) return false; try { const raw = getConfig().getConfig() as any; - return raw?.models?.profiles?.[m]?.forceToolChoiceOnLiveData === true; + return raw?.models?.profiles?.[m]?.forceToolChoiceOnLiveData !== false; } catch { - return false; + return true; } } diff --git a/src/providers/ollama-adapter.ts b/src/providers/ollama-adapter.ts index d61d604..a5ad898 100644 --- a/src/providers/ollama-adapter.ts +++ b/src/providers/ollama-adapter.ts @@ -13,22 +13,10 @@ import path from 'path'; import type { LLMProvider, ChatMessage, ContentPart, ChatOptions, ChatResult, GenerateOptions, GenerateResult, ModelInfo } from './LLMProvider'; import { getConfig } from '../config/config'; - -// Lightweight JSONL usage log for cost-comparison analysis (Ollama Cloud flat-fee -// vs metered APIs like Gemini). Append-only, one line per completed call. -const USAGE_LOG_PATH = (() => { - try { return path.join(getConfig().getConfigDir(), 'logs', 'token-usage.jsonl'); } - catch { return null; } -})(); -function logUsage(model: string, promptTokens: number, completionTokens: number) { - if (!USAGE_LOG_PATH || (!promptTokens && !completionTokens)) return; - try { - fs.mkdirSync(path.dirname(USAGE_LOG_PATH), { recursive: true }); - fs.appendFileSync(USAGE_LOG_PATH, JSON.stringify({ - ts: new Date().toISOString(), model, promptTokens, completionTokens, - }) + '\n'); - } catch { /* non-fatal */ } -} +// Moved to providers/usage-log.ts (2026-07-29) so non-Ollama adapters log too β€” it used +// to be defined privately here, which meant Gemini-served traffic never appeared in the +// cost-comparison log at all. See that file's header for the full story. +import { logUsage } from './usage-log'; /** * Resolve a message's content for Ollama. diff --git a/src/providers/openai-codex-adapter.ts b/src/providers/openai-codex-adapter.ts index b35f9b6..248749a 100644 --- a/src/providers/openai-codex-adapter.ts +++ b/src/providers/openai-codex-adapter.ts @@ -16,6 +16,7 @@ import type { LLMProvider, ChatMessage, ContentPart, ChatOptions, ChatResult, GenerateOptions, GenerateResult, ModelInfo } from './LLMProvider'; import { loadTokens, getValidToken } from '../auth/openai-oauth'; import { contentToString } from './content-utils'; +import { logUsage } from './usage-log'; const CODEX_ENDPOINT = 'https://chatgpt.com/backend-api/codex/responses'; @@ -200,11 +201,11 @@ export class OpenAICodexAdapter implements LLMProvider { } // Parse SSE stream to extract the completed response - const result = await this.parseSSEStream(response); + const result = await this.parseSSEStream(response, model); return result; } - private async parseSSEStream(response: Response): Promise { + private async parseSSEStream(response: Response, model: string): Promise { const reader = response.body?.getReader(); if (!reader) throw new Error('No response body from Codex endpoint'); @@ -212,6 +213,7 @@ export class OpenAICodexAdapter implements LLMProvider { let buffer = ''; let finalContent = ''; let toolCalls: any[] = []; + let streamUsage: { promptTokens: number; completionTokens: number } | null = null; try { while (true) { @@ -259,6 +261,16 @@ export class OpenAICodexAdapter implements LLMProvider { // response.completed contains the full final snapshot if (type === 'response.completed') { + // Responses-API usage lives on the final snapshot, not the deltas β€” the only + // point in this stream where token counts exist (added 2026-07-29 alongside + // the openai-compat fix; this adapter was the other provider missing from the + // cost-comparison log). + const u = event.response?.usage; + if (u) { + const pt = Number(u.input_tokens ?? u.prompt_tokens ?? 0) || 0; + const ct = Number(u.output_tokens ?? u.completion_tokens ?? 0) || 0; + if (pt || ct) { streamUsage = { promptTokens: pt, completionTokens: ct }; } + } const outputs = event.response?.output || []; for (const item of outputs) { if (item.type === 'message') { @@ -301,7 +313,11 @@ export class OpenAICodexAdapter implements LLMProvider { tool_calls: toolCalls.length > 0 ? toolCalls : undefined, }; - return { message }; + if (streamUsage) logUsage(model, streamUsage.promptTokens, streamUsage.completionTokens); + return { + message, + ...(streamUsage ? { usage: { ...streamUsage, ctxWindow: 0 } } : {}), + }; } async generate(prompt: string, model: string, options?: GenerateOptions): Promise { diff --git a/src/providers/openai-compat-adapter.ts b/src/providers/openai-compat-adapter.ts index 73c89ac..1427202 100644 --- a/src/providers/openai-compat-adapter.ts +++ b/src/providers/openai-compat-adapter.ts @@ -11,6 +11,7 @@ import type { LLMProvider, ChatMessage, ChatOptions, ChatResult, GenerateOptions, GenerateResult, ModelInfo, ProviderID } from './LLMProvider'; import { contentToString } from './content-utils'; import { recordGoogleRequest } from './google-usage'; +import { logUsage, parseOpenAiUsage } from './usage-log'; export interface OpenAICompatConfig { endpoint: string; @@ -117,7 +118,17 @@ export class OpenAICompatAdapter implements LLMProvider { content: choice?.message?.content ?? '', tool_calls: choice?.message?.tool_calls, }; - return { message }; + // This adapter deliberately has no chatStream(), so every streaming caller + // (chatWithThinkingStream) falls back through here β€” making this the single + // point where Gemini/LM Studio/OpenAI token counts can be captured at all. + // Until 2026-07-29 the usage field was dropped on the floor here, so those + // providers reported no usage to the UI gauge and wrote nothing to the cost log. + const usage = parseOpenAiUsage(data.usage); + if (usage) logUsage(model, usage.promptTokens, usage.completionTokens); + return { + message, + ...(usage ? { usage: { ...usage, ctxWindow: Number(options?.num_ctx) || 0 } } : {}), + }; } async generate(prompt: string, model: string, options?: GenerateOptions): Promise { @@ -144,6 +155,8 @@ export class OpenAICompatAdapter implements LLMProvider { const data = await this.post('/v1/chat/completions', body); const content = data.choices?.[0]?.message?.content ?? ''; + const usage = parseOpenAiUsage(data.usage); + if (usage) logUsage(model, usage.promptTokens, usage.completionTokens); return { response: contentToString(content) }; } diff --git a/src/providers/usage-log.ts b/src/providers/usage-log.ts new file mode 100644 index 0000000..73c260b --- /dev/null +++ b/src/providers/usage-log.ts @@ -0,0 +1,43 @@ +/** + * usage-log.ts + * Append-only JSONL token-usage log, shared by every provider adapter. + * + * Lives here rather than inside one adapter because the log exists to compare cost + * across providers (Ollama Cloud's flat fee vs metered APIs like Gemini) β€” a log that + * only one adapter writes to silently answers that question wrong. Verified 2026-07-29: + * this logger was defined privately inside ollama-adapter.ts, so a full morning of + * traffic served by the Google/Gemini provider produced ZERO rows, making Gemini look + * free in the very comparison the log was created for. + */ + +import fs from 'fs'; +import path from 'path'; +import { getConfig } from '../config/config'; + +const USAGE_LOG_PATH = (() => { + try { return path.join(getConfig().getConfigDir(), 'logs', 'token-usage.jsonl'); } + catch { return null; } +})(); + +export function logUsage(model: string, promptTokens: number, completionTokens: number): void { + if (!USAGE_LOG_PATH || (!promptTokens && !completionTokens)) return; + try { + fs.mkdirSync(path.dirname(USAGE_LOG_PATH), { recursive: true }); + fs.appendFileSync(USAGE_LOG_PATH, JSON.stringify({ + ts: new Date().toISOString(), model, promptTokens, completionTokens, + }) + '\n'); + } catch { /* non-fatal */ } +} + +/** + * Pull token counts out of an OpenAI-compatible `usage` object. + * Returns null when the field is absent or carries no real numbers, so callers can + * skip both logging and populating a bogus all-zero TokenUsage. + */ +export function parseOpenAiUsage(usage: any): { promptTokens: number; completionTokens: number } | null { + if (!usage || typeof usage !== 'object') return null; + const promptTokens = Number(usage.prompt_tokens ?? usage.promptTokens ?? 0) || 0; + const completionTokens = Number(usage.completion_tokens ?? usage.completionTokens ?? 0) || 0; + if (!promptTokens && !completionTokens) return null; + return { promptTokens, completionTokens }; +}