"""Image-verification workflow for the dental dictionary. Uses a vision model via Ollama to check whether an image actually depicts the dental term it is mapped to. Two entry points: validate_image(src, korean, english, ...) -> dict Validate a single image (local path or http URL) against a term. verify_db(category=None, limit=None, fix=False) Re-check terms that already have an image_url and report mismatches. With fix=True, clears bad mappings (image_url -> NULL) so a later `add-images` pass can re-fill them. The one-off per-round assignment scripts that originally lived in /tmp are archived under docs/archive_image_rounds/ for reference. Called by manage.py's `verify` subcommand. """ import base64 import json import os import sqlite3 import time import urllib.request from config import DB_PATH, IMG_BASE, IMG_URL_PREFIX, OLLAMA_URL, VERIFY_MODEL, connect_db _RULE_PROCEDURE = ( "- 용어가 특정 술식·기법이면, 일반적인 장치·기구·해부 사진만으로는 '아니오'. " "해당 술식의 과정이나 결과가 실제로 드러나야 '예'.\n" ) _RULE_MATERIAL = ( "- 용어가 재료·약물·장비·기구·소프트웨어 자체라면, 그 제품·물질·장치를 " "명확히 보여주는 사진(앰풀·바이알·재료 덩어리·장비 외관·소프트웨어 UI·" "스캔 결과 화면 등)은 '예'. 단, 용어와 무관한 다른 제품 사진은 '아니오'.\n" ) _RULE_ANATOMY = ( "- 용어가 해부 구조·조직·세포·발생 단계라면, 해당 구조를 보여주는 " "해부 사진·도식·현미경 사진·조직학 슬라이드는 '예'. 일반 임상 사진은 " "해당 구조가 명확히 드러날 때만 '예'.\n" ) _RULE_RADIOLOGY = ( "- 방사선 촬영 기법 도식, 포지셔닝 다이어그램, 방사선 해부 도해, 방사선 방호 도표, " "촬영 조건 비교표(kVp·mA·노출시간 등)는 용어와 관련될 때 '예'. " "순수 통계 차트·환자 집계 표·흐름도는 '아니오'.\n" ) _MATERIAL_CATEGORIES = {"치과마취학", "치과생체재료학", "디지털치의학"} _ANATOMY_CATEGORIES = {"기초치의학"} _RADIOLOGY_CATEGORIES = {"구강악안면영상의학"} def _category_rule(category): if category in _MATERIAL_CATEGORIES: return _RULE_MATERIAL if category in _ANATOMY_CATEGORIES: return _RULE_ANATOMY if category in _RADIOLOGY_CATEGORIES: return _RULE_RADIOLOGY return _RULE_PROCEDURE _PROMPT = ( "치과 용어: {korean} ({english}){cat}\n" "{hint}" "이 이미지가 위 치과 용어를 설명하거나 예시하는 데 적합한가요?\n" "다음 기준을 엄격히 적용하세요:\n" "- 흐름도(flowchart), 통계 차트, 순수 수치 데이터 표는 반드시 '아니오'. " "기술 도식·해부 도해·포지셔닝 다이어그램은 용어와 관련될 때 '예'.\n" "- 이미지 안에 적힌 글자·라벨·제목·캡션은 판정 근거로 삼지 마세요. " "오직 시각적 내용만으로 판단합니다.\n" "{rule3}" "반드시 아래 형식으로만 답하세요:\n적합: 예/아니오\n이유: (한 줄)" ) def _load_image_b64(src): """Return base64 of an image given a local path or an http(s) URL.""" if src.startswith("http://") or src.startswith("https://"): for attempt in range(4): try: req = urllib.request.Request(src, headers={"User-Agent": "Mozilla/5.0"}) with urllib.request.urlopen(req, timeout=15) as r: data = r.read() return base64.b64encode(data).decode() except Exception as e: if "429" in str(e) and attempt < 3: wait = 15 * (attempt + 1) time.sleep(wait) else: raise with open(src, "rb") as f: return base64.b64encode(f.read()).decode() def _resolve(image_url): """Map a stored image_url to something _load_image_b64 can open. Handles all URL forms in the DB: - http(s)://... -> used as-is (remote URL) - /api/files/uploads/dental_images/ -> IMG_BASE/ - /api/files/dental_images/ -> IMG_BASE/ - dental_images/ -> IMG_BASE/ """ if image_url.startswith("http"): return image_url s = image_url for prefix in ("/api/files/uploads/dental_images/", "/api/files/dental_images/"): if s.startswith(prefix): return os.path.join(IMG_BASE, s[len(prefix):]) if s.startswith("dental_images/"): return os.path.join(IMG_BASE, s[len("dental_images/"):]) return os.path.join(IMG_BASE, s.lstrip("/")) def verify_updates(conn, updates, source="", model=VERIFY_MODEL, delay=0.0): """Filter a list of (image_url, term_id) candidate mappings. Keeps only those whose image the vision model judges a fit for the term. Used by workflow_images source functions to gate their batch writes so that every source verifies before saving. Returns the filtered list. """ if not updates: return updates tag = f"[{source}] " if source else "" print(f" {tag}verifying {len(updates)} candidate mappings " f"(model={model})...", flush=True) kept, dropped = [], 0 for image_url, term_id in updates: row = conn.execute( "SELECT korean, english, category FROM terms WHERE id=?", (term_id,) ).fetchone() if not row: continue korean, english, category = row res = validate_image(_resolve(image_url), korean, english, category, model=model) if res["ok"]: kept.append((image_url, term_id)) else: dropped += 1 print(f" ✗ {korean} — {res['reason'][:60]}", flush=True) if delay: time.sleep(delay) print(f" {tag}verified: {len(kept)} kept, {dropped} dropped", flush=True) return kept def validate_image(src, korean, english, category=None, hint=None, model=VERIFY_MODEL): """Ask the vision model whether `src` fits the term. Returns {"ok": bool, "reason": str, "raw": str}. On any error, ok is False and reason explains why. """ try: b64 = _load_image_b64(src) except Exception as e: return {"ok": False, "reason": f"load failed: {e}", "raw": ""} prompt = _PROMPT.format( korean=korean, english=english or "", cat=f", 카테고리: {category}" if category else "", hint=f"{hint}\n" if hint else "", rule3=_category_rule(category), ) payload = { "model": model, "messages": [{"role": "user", "content": prompt, "images": [b64]}], "stream": False, } try: data = json.dumps(payload).encode() req = urllib.request.Request( OLLAMA_URL, data=data, headers={"Content-Type": "application/json"} ) with urllib.request.urlopen(req, timeout=60) as r: text = json.loads(r.read()).get("message", {}).get("content", "") except Exception as e: return {"ok": False, "reason": f"model error: {e}", "raw": ""} ok = "적합: 예" in text reason = next( (l.split(":", 1)[1].strip() for l in text.splitlines() if l.startswith("이유:")), "", ) return {"ok": ok, "reason": reason, "raw": text} def verify_db(category=None, limit=None, fix=False, delay=0.3, model=VERIFY_MODEL): """Re-check terms that already have an image_url. Reports each mismatch; with fix=True also clears it (image_url -> NULL). Returns the number of mismatches found. """ conn = connect_db() sql = ( "SELECT id, korean, english, category, image_url FROM terms " "WHERE image_url IS NOT NULL AND image_url != ''" ) params = [] if category: sql += " AND category = ?" params.append(category) sql += " ORDER BY category, id" if limit: sql += " LIMIT ?" params.append(limit) rows = conn.execute(sql, params).fetchall() print(f"Verifying {len(rows)} mapped terms" f"{f' in {category}' if category else ''} (model={model})", flush=True) bad = [] for i, (tid, korean, english, cat, image_url) in enumerate(rows): res = validate_image(_resolve(image_url), korean, english, cat, model=model) if not res["ok"]: bad.append(tid) print(f" ✗ [{cat}] {korean} — {res['reason'][:60]}", flush=True) if (i + 1) % 25 == 0: print(f" ... {i + 1}/{len(rows)} checked, {len(bad)} bad", flush=True) time.sleep(delay) if fix and bad: conn.executemany( "UPDATE terms SET image_url = NULL WHERE id = ?", [(t,) for t in bad] ) conn.commit() print(f"\nCleared {len(bad)} bad mappings (image_url -> NULL)") else: print(f"\n{len(bad)} mismatches found" f"{' (run with --fix to clear them)' if bad else ''}") conn.close() return len(bad)