Files
homeclaw/.smallclaw/databases/vision_benchmark.py
T
kimandClaude Sonnet 4.6 508d11615a Add user workspaces, dental DB scripts, server memory + code editor improvements
- cherry: orofacial/TMD PPTX presentations + 음악가 구강건강 가이드
- papa: sinus/infraoccluded PPTX, garden-monitor Arduino, MIDI files
- databases: PCSP scraper, translation scripts, vision benchmark, batch JSONs
- web-ui: code.js/companion.py/styles.css major updates, autostart installer
- src: ollama-client, session, LLMProvider, ollama-adapter patches
- .gitignore: add claude-key, SQLite WAL/SHM, __pycache__, nohup.out

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-07 23:13:03 +09:00

132 lines
5.0 KiB
Python

#!/usr/bin/env python3
"""
Vision speed benchmark: kimi-k2.6:cloud vs mistral-large-3:675b-cloud
10 requests each, with image + text input.
"""
import base64, json, time, statistics, urllib.request, urllib.error, sys
OLLAMA_URL = "http://localhost:11434/api/chat"
OLLAMA_TAGS_URL = "http://localhost:11434/api/tags"
# thinking 모델은 토큰을 많이 써야 content가 나옴
MODEL_OPTIONS = {
"kimi-k2.6:cloud": {"num_predict": 2000},
"glm-5.1:cloud": {"num_predict": 2000},
"default": {"num_predict": 1000},
}
MODELS = [
"gpt-oss:120b-cloud",
"gemini-3-flash-preview:cloud",
]
# Test questions (varied to avoid caching)
QUESTIONS = [
"양자컴퓨터가 현재 암호화 기술에 미치는 위협을 간단히 설명해줘.",
"우울증과 번아웃의 차이점은 뭐야?",
"파이썬에서 GIL이 뭔지, 멀티스레딩에 어떤 영향을 주는지 설명해줘.",
"기후변화 대응에서 탄소세와 탄소 배출권 거래제의 장단점 비교해줘.",
"한국 부동산 시장에서 전세 제도의 장단점을 설명해줘.",
"RAG(Retrieval-Augmented Generation)가 뭔지 쉽게 설명해줘.",
"소크라테스의 무지의 지(知)가 현대 사회에서 갖는 의미는?",
"비트코인의 작업증명(PoW)과 이더리움의 지분증명(PoS) 차이를 설명해줘.",
"외상 후 스트레스 장애(PTSD) 치료에서 EMDR 요법이 효과적인 이유는?",
"마이크로서비스 아키텍처의 장단점과 언제 쓰는 게 좋은지 알려줘.",
]
def load_image_b64(path: str) -> str:
with open(path, "rb") as f:
return base64.b64encode(f.read()).decode()
def chat(model: str, question: str, img_b64: str = "") -> tuple[float, str, str]:
msg: dict = {"role": "user", "content": question}
if img_b64:
msg["images"] = [img_b64]
body = {
"model": model,
"stream": False,
"messages": [msg],
"options": MODEL_OPTIONS.get(model, MODEL_OPTIONS["default"]),
}
data = json.dumps(body).encode()
req = urllib.request.Request(OLLAMA_URL, data=data,
headers={"Content-Type": "application/json"})
t0 = time.perf_counter()
try:
with urllib.request.urlopen(req, timeout=120) as r:
resp = json.loads(r.read())
elapsed = time.perf_counter() - t0
msg = resp.get("message", {})
content = msg.get("content", "").strip()
thinking = msg.get("thinking", "").strip()
return elapsed, content, thinking
except Exception as e:
elapsed = time.perf_counter() - t0
return elapsed, f"ERROR: {e}", ""
def benchmark(model: str, img_b64: str = "") -> tuple[list[float], list[str], list[str]]:
times, responses, thinkings = [], [], []
print(f"\n{'='*50}")
print(f" {model}")
print(f"{'='*50}")
for i, q in enumerate(QUESTIONS):
t, text, thinking = chat(model, q, img_b64)
status = "✓" if not text.startswith("ERROR") else "✗"
has_thinking = "💭" if thinking else " "
print(f" [{i+1:02d}] {status}{has_thinking} {t:6.2f}s")
times.append(t)
responses.append(text)
thinkings.append(thinking)
sys.stdout.flush()
return times, responses, thinkings
def main():
print(f"테스트 모델 ({len(MODELS)}개): {', '.join(MODELS)}")
print("모드: 일반 채팅 (이미지 없음)")
results = {}
for model in MODELS:
times, responses, thinkings = benchmark(model)
results[model] = {"times": times, "responses": responses, "thinkings": thinkings}
# 응답 내용 비교
print(f"\n{'='*60}")
print(" 응답 내용 비교")
print(f"{'='*60}")
for i, q in enumerate(QUESTIONS):
print(f"\n[Q{i+1}] {q}")
for model in MODELS:
name = model.split(":")[0].split("-")[0].upper()
resp = results[model]["responses"][i]
thinking = results[model]["thinkings"][i]
t = results[model]["times"][i]
print(f"\n [{name} {t:.1f}s]")
if thinking:
print(f" 💭 {thinking[:200]}")
print(f" ✏️ {resp if resp else '(없음)'}")
print()
# 속도 요약
print(f"\n{'='*60}")
print(" 속도 요약")
print(f"{'='*60}")
for model, data in results.items():
times = data["times"]
valid = [t for t in times if t < 115]
if valid:
print(f"\n {model}")
print(f" 평균: {statistics.mean(valid):.2f}s 중앙값: {statistics.median(valid):.2f}s "
f"최소: {min(valid):.2f}s 최대: {max(valid):.2f}s")
avgs = {m: statistics.mean([t for t in d["times"] if t < 115] or [999])
for m, d in results.items()}
winner = min(avgs, key=avgs.get)
loser = max(avgs, key=avgs.get)
diff = avgs[loser] - avgs[winner]
print(f"\n 속도 승자: {winner} ({diff:.2f}s, {diff/avgs[loser]*100:.0f}% 빠름)")
print(f"{'='*60}")
if __name__ == "__main__":
main()