- 스킬앱 HTML 내장화: accountant/lawyer/investor/weather/music/mind/dental-agent 각 앱이 SKILL_CONTENT를 HTML에 직접 내장 → .smallclaw/skills/ SKILL.md 파일 제거 - 새 게이트웨이 라우트: routes-dental, routes-music, routes-mcp, routes-voice - skillContext 서버 지원: /api/chat에서 skill 컨텍스트를 callerCtx로 주입 - 세션 삭제 API: DELETE /api/admin/sessions/:username/:id - Telegram defaultUsername: 어드민 계정 자동 추론 - Ollama /api/ollama/models에 vision 지원 여부 플래그 추가 - DB 워크플로 개선: workflow_images.py 대규모 리팩터링 - 아두이노 에뮬레이터 추가 개선 (arduino-emulator.html +400줄) - 우즈베크어 학습 콘텐츠 및 MIDI 파일 추가 Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
237 lines
9.0 KiB
Python
237 lines
9.0 KiB
Python
"""Image-verification workflow for the dental dictionary.
|
|
|
|
Uses a vision model via Ollama to check whether an image actually depicts
|
|
the dental term it is mapped to. Two entry points:
|
|
|
|
validate_image(src, korean, english, ...) -> dict
|
|
Validate a single image (local path or http URL) against a term.
|
|
|
|
verify_db(category=None, limit=None, fix=False)
|
|
Re-check terms that already have an image_url and report mismatches.
|
|
With fix=True, clears bad mappings (image_url -> NULL) so a later
|
|
`add-images` pass can re-fill them.
|
|
|
|
The one-off per-round assignment scripts that originally lived in /tmp are
|
|
archived under docs/archive_image_rounds/ for reference.
|
|
Called by manage.py's `verify` subcommand.
|
|
"""
|
|
|
|
import base64
|
|
import json
|
|
import os
|
|
import sqlite3
|
|
import time
|
|
import urllib.request
|
|
|
|
from config import DB_PATH, IMG_BASE, IMG_URL_PREFIX, OLLAMA_URL, VERIFY_MODEL, connect_db
|
|
|
|
_RULE_PROCEDURE = (
|
|
"- 용어가 특정 술식·기법이면, 일반적인 장치·기구·해부 사진만으로는 '아니오'. "
|
|
"해당 술식의 과정이나 결과가 실제로 드러나야 '예'.\n"
|
|
)
|
|
_RULE_MATERIAL = (
|
|
"- 용어가 재료·약물·장비·기구·소프트웨어 자체라면, 그 제품·물질·장치를 "
|
|
"명확히 보여주는 사진(앰풀·바이알·재료 덩어리·장비 외관·소프트웨어 UI·"
|
|
"스캔 결과 화면 등)은 '예'. 단, 용어와 무관한 다른 제품 사진은 '아니오'.\n"
|
|
)
|
|
_RULE_ANATOMY = (
|
|
"- 용어가 해부 구조·조직·세포·발생 단계라면, 해당 구조를 보여주는 "
|
|
"해부 사진·도식·현미경 사진·조직학 슬라이드는 '예'. 일반 임상 사진은 "
|
|
"해당 구조가 명확히 드러날 때만 '예'.\n"
|
|
)
|
|
|
|
_RULE_RADIOLOGY = (
|
|
"- 방사선 촬영 기법 도식, 포지셔닝 다이어그램, 방사선 해부 도해, 방사선 방호 도표, "
|
|
"촬영 조건 비교표(kVp·mA·노출시간 등)는 용어와 관련될 때 '예'. "
|
|
"순수 통계 차트·환자 집계 표·흐름도는 '아니오'.\n"
|
|
)
|
|
|
|
_MATERIAL_CATEGORIES = {"치과마취학", "치과생체재료학", "디지털치의학"}
|
|
_ANATOMY_CATEGORIES = {"기초치의학"}
|
|
_RADIOLOGY_CATEGORIES = {"구강악안면영상의학"}
|
|
|
|
|
|
def _category_rule(category):
|
|
if category in _MATERIAL_CATEGORIES:
|
|
return _RULE_MATERIAL
|
|
if category in _ANATOMY_CATEGORIES:
|
|
return _RULE_ANATOMY
|
|
if category in _RADIOLOGY_CATEGORIES:
|
|
return _RULE_RADIOLOGY
|
|
return _RULE_PROCEDURE
|
|
|
|
|
|
_PROMPT = (
|
|
"치과 용어: {korean} ({english}){cat}\n"
|
|
"{hint}"
|
|
"이 이미지가 위 치과 용어를 설명하거나 예시하는 데 적합한가요?\n"
|
|
"다음 기준을 엄격히 적용하세요:\n"
|
|
"- 흐름도(flowchart), 통계 차트, 순수 수치 데이터 표는 반드시 '아니오'. "
|
|
"기술 도식·해부 도해·포지셔닝 다이어그램은 용어와 관련될 때 '예'.\n"
|
|
"- 이미지 안에 적힌 글자·라벨·제목·캡션은 판정 근거로 삼지 마세요. "
|
|
"오직 시각적 내용만으로 판단합니다.\n"
|
|
"{rule3}"
|
|
"반드시 아래 형식으로만 답하세요:\n적합: 예/아니오\n이유: (한 줄)"
|
|
)
|
|
|
|
|
|
def _load_image_b64(src):
|
|
"""Return base64 of an image given a local path or an http(s) URL."""
|
|
if src.startswith("http://") or src.startswith("https://"):
|
|
for attempt in range(4):
|
|
try:
|
|
req = urllib.request.Request(src, headers={"User-Agent": "Mozilla/5.0"})
|
|
with urllib.request.urlopen(req, timeout=15) as r:
|
|
data = r.read()
|
|
return base64.b64encode(data).decode()
|
|
except Exception as e:
|
|
if "429" in str(e) and attempt < 3:
|
|
wait = 15 * (attempt + 1)
|
|
time.sleep(wait)
|
|
else:
|
|
raise
|
|
with open(src, "rb") as f:
|
|
return base64.b64encode(f.read()).decode()
|
|
|
|
|
|
def _resolve(image_url):
|
|
"""Map a stored image_url to something _load_image_b64 can open.
|
|
|
|
Handles all URL forms in the DB:
|
|
- http(s)://... -> used as-is (remote URL)
|
|
- /api/files/uploads/dental_images/<rel> -> IMG_BASE/<rel>
|
|
- /api/files/dental_images/<rel> -> IMG_BASE/<rel>
|
|
- dental_images/<rel> -> IMG_BASE/<rel>
|
|
"""
|
|
if image_url.startswith("http"):
|
|
return image_url
|
|
s = image_url
|
|
for prefix in ("/api/files/uploads/dental_images/", "/api/files/dental_images/"):
|
|
if s.startswith(prefix):
|
|
return os.path.join(IMG_BASE, s[len(prefix):])
|
|
if s.startswith("dental_images/"):
|
|
return os.path.join(IMG_BASE, s[len("dental_images/"):])
|
|
return os.path.join(IMG_BASE, s.lstrip("/"))
|
|
|
|
|
|
def verify_updates(conn, updates, source="", model=VERIFY_MODEL, delay=0.0):
|
|
"""Filter a list of (image_url, term_id) candidate mappings.
|
|
|
|
Keeps only those whose image the vision model judges a fit for the term.
|
|
Used by workflow_images source functions to gate their batch writes so
|
|
that every source verifies before saving. Returns the filtered list.
|
|
"""
|
|
if not updates:
|
|
return updates
|
|
tag = f"[{source}] " if source else ""
|
|
print(f" {tag}verifying {len(updates)} candidate mappings "
|
|
f"(model={model})...", flush=True)
|
|
kept, dropped = [], 0
|
|
for image_url, term_id in updates:
|
|
row = conn.execute(
|
|
"SELECT korean, english, category FROM terms WHERE id=?", (term_id,)
|
|
).fetchone()
|
|
if not row:
|
|
continue
|
|
korean, english, category = row
|
|
res = validate_image(_resolve(image_url), korean, english, category,
|
|
model=model)
|
|
if res["ok"]:
|
|
kept.append((image_url, term_id))
|
|
else:
|
|
dropped += 1
|
|
print(f" ✗ {korean} — {res['reason'][:60]}", flush=True)
|
|
if delay:
|
|
time.sleep(delay)
|
|
print(f" {tag}verified: {len(kept)} kept, {dropped} dropped", flush=True)
|
|
return kept
|
|
|
|
|
|
def validate_image(src, korean, english, category=None, hint=None,
|
|
model=VERIFY_MODEL):
|
|
"""Ask the vision model whether `src` fits the term.
|
|
|
|
Returns {"ok": bool, "reason": str, "raw": str}. On any error,
|
|
ok is False and reason explains why.
|
|
"""
|
|
try:
|
|
b64 = _load_image_b64(src)
|
|
except Exception as e:
|
|
return {"ok": False, "reason": f"load failed: {e}", "raw": ""}
|
|
|
|
prompt = _PROMPT.format(
|
|
korean=korean,
|
|
english=english or "",
|
|
cat=f", 카테고리: {category}" if category else "",
|
|
hint=f"{hint}\n" if hint else "",
|
|
rule3=_category_rule(category),
|
|
)
|
|
payload = {
|
|
"model": model,
|
|
"messages": [{"role": "user", "content": prompt, "images": [b64]}],
|
|
"stream": False,
|
|
}
|
|
try:
|
|
data = json.dumps(payload).encode()
|
|
req = urllib.request.Request(
|
|
OLLAMA_URL, data=data, headers={"Content-Type": "application/json"}
|
|
)
|
|
with urllib.request.urlopen(req, timeout=60) as r:
|
|
text = json.loads(r.read()).get("message", {}).get("content", "")
|
|
except Exception as e:
|
|
return {"ok": False, "reason": f"model error: {e}", "raw": ""}
|
|
|
|
ok = "적합: 예" in text
|
|
reason = next(
|
|
(l.split(":", 1)[1].strip() for l in text.splitlines() if l.startswith("이유:")),
|
|
"",
|
|
)
|
|
return {"ok": ok, "reason": reason, "raw": text}
|
|
|
|
|
|
def verify_db(category=None, limit=None, fix=False, delay=0.3, model=VERIFY_MODEL):
|
|
"""Re-check terms that already have an image_url.
|
|
|
|
Reports each mismatch; with fix=True also clears it (image_url -> NULL).
|
|
Returns the number of mismatches found.
|
|
"""
|
|
conn = connect_db()
|
|
sql = (
|
|
"SELECT id, korean, english, category, image_url FROM terms "
|
|
"WHERE image_url IS NOT NULL AND image_url != ''"
|
|
)
|
|
params = []
|
|
if category:
|
|
sql += " AND category = ?"
|
|
params.append(category)
|
|
sql += " ORDER BY category, id"
|
|
if limit:
|
|
sql += " LIMIT ?"
|
|
params.append(limit)
|
|
|
|
rows = conn.execute(sql, params).fetchall()
|
|
print(f"Verifying {len(rows)} mapped terms"
|
|
f"{f' in {category}' if category else ''} (model={model})", flush=True)
|
|
|
|
bad = []
|
|
for i, (tid, korean, english, cat, image_url) in enumerate(rows):
|
|
res = validate_image(_resolve(image_url), korean, english, cat, model=model)
|
|
if not res["ok"]:
|
|
bad.append(tid)
|
|
print(f" ✗ [{cat}] {korean} — {res['reason'][:60]}", flush=True)
|
|
if (i + 1) % 25 == 0:
|
|
print(f" ... {i + 1}/{len(rows)} checked, {len(bad)} bad", flush=True)
|
|
time.sleep(delay)
|
|
|
|
if fix and bad:
|
|
conn.executemany(
|
|
"UPDATE terms SET image_url = NULL WHERE id = ?", [(t,) for t in bad]
|
|
)
|
|
conn.commit()
|
|
print(f"\nCleared {len(bad)} bad mappings (image_url -> NULL)")
|
|
else:
|
|
print(f"\n{len(bad)} mismatches found"
|
|
f"{' (run with --fix to clear them)' if bad else ''}")
|
|
conn.close()
|
|
return len(bad)
|