Files
homeclaw/.smallclaw/databases/workflow_images.py
T
kimandClaude Sonnet 4.6 9ad63ec8b1 v3.0.8: 스킬앱 독립화 + 치과에이전트 + 음악/MCP/보이스 라우트 + DB 워크플로 개선
- 스킬앱 HTML 내장화: accountant/lawyer/investor/weather/music/mind/dental-agent
  각 앱이 SKILL_CONTENT를 HTML에 직접 내장 → .smallclaw/skills/ SKILL.md 파일 제거
- 새 게이트웨이 라우트: routes-dental, routes-music, routes-mcp, routes-voice
- skillContext 서버 지원: /api/chat에서 skill 컨텍스트를 callerCtx로 주입
- 세션 삭제 API: DELETE /api/admin/sessions/:username/:id
- Telegram defaultUsername: 어드민 계정 자동 추론
- Ollama /api/ollama/models에 vision 지원 여부 플래그 추가
- DB 워크플로 개선: workflow_images.py 대규모 리팩터링
- 아두이노 에뮬레이터 추가 개선 (arduino-emulator.html +400줄)
- 우즈베크어 학습 콘텐츠 및 MIDI 파일 추가

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-21 17:07:58 +09:00

2265 lines
97 KiB
Python

#!/usr/bin/env python3
"""
Dental Dictionary Image Mapping Workflow
=========================================
이 스크립트는 치과 용어 사전(dental_dict.db)의 용어에 이미지 URL을 매핑합니다.
소스:
1. Kaggle 데이터셋 (로컬 파일)
- javedrashid/mouth-and-oral-diseases-mod
7클래스(CaS/CoS/Gum/MC/OC/OLP/OT), 516장, Training+Testing+Validation
로컬 경로: .smallclaw/databases/dental_images/kaggle/
Gemini 검증 스크립트: /tmp/kaggle_image_test{1,2,3}.py
결과: /tmp/kaggle_test_results{,2,3}.json
주의: CoS(구순포진) 이미지는 모두 입술 외부 병변(Herpes labialis) →
DB에 구순포진(id=5202), 재발성구순포진(id=5203), 구순단순포진(id=5204) 추가로 매핑 가능
2. Roboflow Universe (CC BY 4.0)
3. Hugging Face 데이터셋 (DENTEX, Caries, Oral Cancer, Implant, Gingivitis, X-ray)
4. Wikimedia Commons (CC BY-SA)
5. Radiopaedia (CC BY-NC-SA)
6. ITU Dental Datasets 카탈로그 참조
8. Zenodo 공개 레코드
9. Mendeley Data (소아치과)
Kaggle 추가 다운로드 후보:
- salmansajid05/oral-diseases: 246MB, Calculus/Gingivitis/Caries/Hypodontia/MouthUlcer/Discolor
→ 다운로드 완료: .smallclaw/databases/dental_images/salmansajid/
→ 검증 스크립트: /tmp/kaggle_image_test4.py, 결과: /tmp/kaggle_test_results4.json
- jiahongqian/cephalometric-landmarks: 117MB, 측면 두개계측 X선 400장 (랜드마크 CSV)
→ 다운로드 완료: .smallclaw/databases/dental_images/cephalometric_landmarks/cepha400/cepha400/
→ 교정 카테고리 upgrade용 (두개계측분석, 악교정수술, 성장평가 등)
- kambingbersayaphitam/cephalometric-profile-dataset: 117MB, 측면 안모 사진 (Concave/Convex/Plane)
→ 다운로드 완료: .smallclaw/databases/dental_images/cephalometric_profile/Cephalometric Profile Dataset/
→ 교정 골격 분류 upgrade용 (앵글류1/2/3급, 상하악전돌증 등)
→ 검증 스크립트: /tmp/kaggle_image_test5.py, 결과: /tmp/kaggle_test_results5.json
교정 카테고리 이미지 현황 (2026-05-13):
- 100% 배정 완료 (394/394), 단 312개는 generic HuggingFace 파노라마 placeholder
- Round 5로 33개 두개계측/안모 이미지로 교체 완료
- 나머지 312개 (브라켓/와이어/장치 전용 term)는 임상 사진 필요 → PMC 추출 또는 전용 데이터셋
사용법:
python3 dental_image_workflow.py [--source all|kaggle|roboflow|huggingface|wikimedia|radiopaedia|itu|zenodo|mendeley|openi]
주의:
- Wikimedia Commons 검색은 API 속도제한(429)이 있으므로 요청 간 5초 이상 대기
- DB에 먼저 모두 모은 후 한 번에 쓰기 (DB 락 방지)
- API 키: .smallclaw/roboflow_api_key.txt, .smallclaw/huggingface_api_key.txt
"""
import sqlite3
import json
import os
import re
import sys
import time
import urllib.request
import urllib.parse
from config import DB_PATH, IMG_BASE, SMALLCLAW, OLLAMA_URL, VERIFY_MODEL, connect_db
from workflow_verify import verify_updates, validate_image
# ============================================================
# 공통 상수 & 헬퍼 함수
# ============================================================
REMOTE_DIR = os.path.join(IMG_BASE, "remote")
URL_PREFIX = "/api/files/dental_images"
URL_PREFIX_REMOTE = "/api/files/dental_images/remote"
_IMAGE_MAGIC = (
b"\x89PNG\r\n\x1a\n", # PNG
b"\xff\xd8\xff", # JPEG
b"GIF87a", b"GIF89a", # GIF
b"RIFF", # WEBP
b"II\x2a\x00", # TIFF little-endian
b"MM\x00\x2a", # TIFF big-endian
)
def find_unmapped_term(conn, korean):
"""Return term id if korean has no image yet, else None."""
r = conn.execute(
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
(korean,),
).fetchone()
return r[0] if r else None
def localize_image(term_id, url, korean=None, log_prefix=" "):
"""Download one remote image URL to REMOTE_DIR/<term_id>.<ext>.
Returns the new local URL on success, None on failure.
Does NOT write to DB — caller is responsible for UPDATE + commit.
"""
if not url or not url.startswith("http"):
return url
path = urllib.parse.urlparse(url).path
ext = os.path.splitext(path)[1].lower()
if ext not in (".jpg", ".jpeg", ".png", ".gif", ".webp"):
ext = ".jpg"
os.makedirs(REMOTE_DIR, exist_ok=True)
local_name = f"{term_id}{ext}"
local_path = os.path.join(REMOTE_DIR, local_name)
new_url = f"{URL_PREFIX_REMOTE}/{local_name}"
label = korean or f"term {term_id}"
if os.path.exists(local_path):
print(f"{log_prefix}↓ {label} — already local", flush=True)
return new_url
for attempt in range(5):
try:
req = urllib.request.Request(
url, headers={"User-Agent": "DentalDictBot/2.0 (educational)"}
)
with urllib.request.urlopen(req, timeout=30) as r:
ctype = (r.headers.get("Content-Type") or "").lower()
data = r.read()
if not ctype.startswith("image/"):
print(f"{log_prefix}↓ {label} — not an image ({ctype or 'no type'})", flush=True)
return None
if not any(data.startswith(m) for m in _IMAGE_MAGIC):
print(f"{log_prefix}↓ {label} — bad magic bytes", flush=True)
return None
if len(data) < 1000:
print(f"{log_prefix}↓ {label} — too small ({len(data)}B)", flush=True)
return None
with open(local_path, "wb") as f:
f.write(data)
print(f"{log_prefix}↓ {label} — {len(data)//1024}KB ✓", flush=True)
return new_url
except Exception as e:
is_429 = "429" in str(e)
if attempt < 4:
wait = min(30 * (attempt + 1), 120) if is_429 else 5
print(f"{log_prefix}↓ {label} — retry {attempt+1}/5 after {wait}s: {e}", flush=True)
time.sleep(wait)
else:
print(f"{log_prefix}↓ {label} — FAILED: {e} (URL kept)", flush=True)
return None
def apply_updates(conn, updates, source_name, verify=True):
"""공통 패턴: 검증(선택) → bulk UPDATE → http URL 로컬화 → 1회 commit.
updates = [(image_url, term_id), ...]
"""
if verify:
updates = verify_updates(conn, updates, source=source_name)
for url, term_id in updates:
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, term_id))
# 외부 http URL 자동 로컬화 (zenodo, mendeley, roboflow_api 등)
localized = 0
for url, term_id in list(updates):
if url.startswith("http"):
new_url = localize_image(term_id, url)
if new_url:
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (new_url, term_id))
localized += 1
conn.commit()
if localized:
print(f" {source_name}: {localized} URLs localized to remote/", flush=True)
return len(updates)
def bulk_localize(limit=None, category=None, dry_run=False):
"""DB의 http:// image_url을 모두 로컬 파일로 다운로드.
manage.py localize 및 구 download_images.py CLI에서 호출.
"""
os.makedirs(REMOTE_DIR, exist_ok=True)
conn = connect_db()
query = """
SELECT id, korean, image_url FROM terms
WHERE image_url IS NOT NULL AND image_url != '' AND image_url LIKE 'http%'
"""
params = []
if category:
query += " AND category=?"
params.append(category)
rows = conn.execute(query, params).fetchall()
if limit:
rows = rows[:limit]
print(f"Remote images to download: {len(rows)}", flush=True)
downloaded = skipped = failed = 0
for i, (term_id, korean, url) in enumerate(rows, 1):
if dry_run:
print(f" [{i}/{len(rows)}] {korean} — would download {url}", flush=True)
downloaded += 1
continue
prefix = f" [{i}/{len(rows)}] "
already = os.path.exists(os.path.join(
REMOTE_DIR,
f"{term_id}{os.path.splitext(urllib.parse.urlparse(url).path)[1].lower() or '.jpg'}"
))
new_url = localize_image(term_id, url, korean=korean, log_prefix=prefix)
if new_url is None:
failed += 1
elif already:
skipped += 1
else:
downloaded += 1
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (new_url, term_id))
time.sleep(2 if "wikimedia" in url else 0.5)
if not dry_run:
conn.commit()
conn.close()
print(f"\nDone: {downloaded} downloaded, {skipped} already local, {failed} failed", flush=True)
# ============================================================
# 1. Kaggle 데이터셋 매핑
# ============================================================
def map_kaggle_datasets(conn, verify=True):
"""Map images from locally downloaded Kaggle datasets under IMG_BASE.
Present datasets (verified 2026-05-14):
- kaggle/ MOD (Mouth & Oral Diseases): {Training,Testing,Validation}/{CaS,CoS,Gum,MC,OC,OLP,OT}
- salmansajid/ oral-diseases: Calculus / Gingivitis / Data caries / Mouth Ulcer /
Tooth Discoloration / hypodontia
- cephalometric_landmarks/cepha400/cepha400/ 400 lateral cephalometric X-rays
- cephalometric_profile/Cephalometric Profile Dataset/{Concave,Convex,Plane}
image_url is stored as the gateway-served path: /api/files/dental_images/...
Consecutive images are handed to consecutive still-unmapped terms.
"""
updates = []
def list_images(rel_dir):
d = os.path.join(IMG_BASE, rel_dir)
if not os.path.isdir(d):
return []
return [
f"{rel_dir}/{f}"
for f in sorted(os.listdir(d))
if f.lower().endswith((".jpg", ".jpeg", ".png"))
]
def assign(rel_paths, term_list):
i = 0
for term_kr in term_list:
if i >= len(rel_paths):
break
result = conn.execute(
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
(term_kr,),
).fetchone()
if result:
updates.append((f"{URL_PREFIX}/{rel_paths[i]}", result[0]))
i += 1
# --- 1. MOD dataset (kaggle/) — 7 disease classes across 3 splits ---
mod_terms = {
"CaS": ["아프타성 구내염", "구강아프타", "구강아프타성궤양", "아프타성 궤양"],
"CoS": ["구순포진", "재발성구순포진", "구순단순포진", "구강포진"],
"Gum": ["치은염", "치주염"],
"MC": ["구강암"],
"OC": ["편평세포암종", "편평상피암", "구강편평세포암"],
"OLP": ["편평태선"],
"OT": ["구강 칸디다증", "구강캔디다증", "위막형칸디다증"],
}
for cls, term_list in mod_terms.items():
imgs = []
for split in ("Training", "Testing", "Validation"):
imgs.extend(list_images(f"kaggle/{split}/{cls}"))
assign(imgs, term_list)
# --- 2. salmansajid/ oral-diseases (original, non-augmented leaves) ---
salmansajid_map = [
("salmansajid/Calculus/Calculus", ["치석"]),
("salmansajid/Gingivitis/Gingivitis", ["치은염", "치주염"]),
("salmansajid/Data caries/Data caries/caries orignal data set/done",
["치아우식증", "법랑질우식", "상아질우식"]),
("salmansajid/Mouth Ulcer/Mouth Ulcer/ulcer original dataset/ulcer original dataset",
["구강궤양"]),
("salmansajid/Tooth Discoloration/Tooth Discoloration /tooth discoloration original dataset/tooth discoloration original dataset",
["치아 변색", "치아염색", "외인성 변색", "내인성 변색"]),
("salmansajid/hypodontia/hypodontia", ["선천성무치증"]),
]
for rel_dir, term_list in salmansajid_map:
assign(list_images(rel_dir), term_list)
# --- 3. cephalometric_landmarks — 400 lateral cephalometric X-rays ---
assign(
list_images("cephalometric_landmarks/cepha400/cepha400"),
["두부 계측 방사선 사진", "측두두부방사선사진", "두부 방사선 규격 사진",
"측두두부계측방사선사진", "후전두부계측방사선사진", "측방두개규격사진",
"두부 계측 추적", "두부 계측 분석", "두부계측분석"],
)
# --- 4. cephalometric_profile — facial profile photos (Concave/Convex/Plane) ---
profile_imgs = []
for cls in ("Concave", "Convex", "Plane"):
profile_imgs.extend(
list_images(f"cephalometric_profile/Cephalometric Profile Dataset/{cls}")
)
assign(profile_imgs, ["측면 분석", "안모 분석", "안모 분석 체계", "안모비율"])
return apply_updates(conn, updates, "kaggle", verify=verify)
# ============================================================
# 2. Roboflow 데이터셋 매핑
# ============================================================
def map_roboflow_datasets(conn, verify=True):
"""Map images from Roboflow datasets"""
updates = []
COCO_LABEL_MAP = {
"Cavity": ["치아우식증", "법랑질우식", "상아질우식"],
"Fillings": ["치과충전", "복합레진", "아말감"],
"Impacted Tooth": ["매복치", "사랑니"],
"Implant": ["임플란트"],
"infected-teeth": ["근첩농양", "근첩육아종", "근첩병변"],
}
DENTAL_LABEL_MAP = {
"Cavity": ["치아우식증"],
"Fillings": ["치과충전", "복합레진"],
"Impacted Tooth": ["매복치"],
"Implant": ["임플란트", "임플란트보철물"],
}
for dataset_name, label_map, prefix in [
("roboflow_coco", COCO_LABEL_MAP, "COCO"),
("roboflow_dental_xray", DENTAL_LABEL_MAP, "Dental"),
]:
for split in ["train", "valid", "test"]:
json_path = os.path.join(IMG_BASE, dataset_name, split, "_annotations.coco.json")
if not os.path.exists(json_path):
continue
with open(json_path) as f:
data = json.load(f)
cats = {c['id']: c['name'] for c in data['categories']}
for cat_name, korean_terms in label_map.items():
cat_id = None
for cid, cname in cats.items():
if cname == cat_name:
cat_id = cid
break
if not cat_id:
continue
img_ids = set()
for ann in data['annotations']:
if ann['category_id'] == cat_id:
img_ids.add(ann['image_id'])
img_map = {img['id']: img['file_name'] for img in data['images']}
img_list = [img_map[iid] for iid in img_ids if iid in img_map]
for term_kr in korean_terms:
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
if result and img_list:
rel_path = f"/api/files/dental_images/{dataset_name}/{split}/{img_list[0]}"
full_path = os.path.join(IMG_BASE, dataset_name, split, img_list[0])
if os.path.exists(full_path):
updates.append((rel_path, result[0]))
break
return apply_updates(conn, updates, "roboflow", verify=verify)
# ============================================================
# 3. Hugging Face 데이터셋 매핑
# ============================================================
def map_huggingface_datasets(conn, verify=True):
"""Map images from Hugging Face datasets (DENTEX, Caries, Oral Cancer, Implant, Gingivitis, X-ray)"""
updates = []
# --- DENTEX diagnosis categories ---
dentex_dir = os.path.join(IMG_BASE, "huggingface_dentex")
if os.path.exists(dentex_dir):
dentex_id_map = {
0: ["치아우식증", "법랑질우식", "상아질우식"],
1: ["상아질우식", "치수염"],
2: ["근첩농양", "근첩육아종", "근첩낭종", "근첩병변"],
3: ["매복치", "사랑니", "함치낭종"],
}
dentex_name_map = {
"caries": dentex_id_map[0],
"deep caries": dentex_id_map[1],
"periapical lesion": dentex_id_map[2],
"impacted": dentex_id_map[3],
}
seen_dentex = set()
for split in ["train", "valid", "test"]:
json_path = os.path.join(dentex_dir, "DENTEX", f"{split}_triple.json")
if not os.path.exists(json_path):
continue
with open(json_path) as f:
data = json.load(f)
# Build (img_file, [category_keys]) pairs — handles COCO dict or list format
pairs = []
if isinstance(data, dict) and "images" in data:
cats = {c['id']: c['name'] for c in data.get('categories', [])}
ann_by_img = {}
for ann in data.get('annotations', []):
ann_by_img.setdefault(ann['image_id'], []).append(
cats.get(ann['category_id'], ann['category_id'])
)
for img in data['images']:
pairs.append((img['file_name'], ann_by_img.get(img['id'], [])))
elif isinstance(data, list):
for entry in data:
img_file = entry.get("file_name", entry.get("image", ""))
keys = entry.get("diagnosis", entry.get("labels", entry.get("categories", [])))
pairs.append((img_file, keys))
print(f" DENTEX {split}: {len(pairs)} entries")
for img_file, cat_keys in pairs:
if not img_file:
continue
if not os.path.exists(os.path.join(dentex_dir, "DENTEX", split, img_file)):
continue
for key in cat_keys:
terms = dentex_name_map.get(key) if isinstance(key, str) else dentex_id_map.get(key)
if not terms:
continue
for term_kr in terms:
if term_kr in seen_dentex:
continue
result = conn.execute(
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
(term_kr,)
).fetchone()
if result:
rel_path = f"/api/files/dental_images/huggingface_dentex/DENTEX/{split}/{img_file}"
updates.append((rel_path, result[0]))
seen_dentex.add(term_kr)
break
# --- Caries2 dataset (reza362/dental-xray-caries) ---
caries2_dir = os.path.join(IMG_BASE, "huggingface_caries2")
if os.path.exists(caries2_dir):
for split in ["train", "test", "valid", "val"]:
split_dir = os.path.join(caries2_dir, split)
if not os.path.exists(split_dir):
continue
images = sorted([f for f in os.listdir(split_dir) if f.endswith(('.jpg', '.png', '.jpeg'))])
if images:
for term_kr in ["치아우식증", "법랑질우식"]:
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
if result:
rel_path = f"/api/files/dental_images/huggingface_caries2/{split}/{images[0]}"
updates.append((rel_path, result[0]))
break
# --- Implant dataset (Mesh2001/Dental_implant) ---
implant_dir = os.path.join(IMG_BASE, "huggingface_implant", "extracted")
if os.path.exists(implant_dir):
# Classes: 0=Bego, 1=Bicon, 2=ITI
implant_term_map = {
0: ["임플란트"],
1: ["임플란트"],
2: ["임플란트", "임플란트보철물"],
}
for split in ["train", "valid", "test"]:
split_img_dir = os.path.join(implant_dir, split, "images")
if not os.path.exists(split_img_dir):
continue
images = sorted([f for f in os.listdir(split_img_dir) if f.endswith(('.jpg', '.png', '.jpeg'))])
# Pick first image per class prefix
seen_prefixes = set()
for img in images:
prefix = img.split("-")[0].split("_")[0]
if prefix in seen_prefixes:
continue
seen_prefixes.add(prefix)
for term_kr in ["임플란트", "임플란트보철물"]:
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
if result:
rel_path = f"/api/files/dental_images/huggingface_implant/extracted/{split}/images/{img}"
updates.append((rel_path, result[0]))
break
# --- Gingivitis dataset (ekacare/IntraOral_Gingivitis_Image_Captioning) ---
gingivitis_dir = os.path.join(IMG_BASE, "huggingface_gingivitis", "extracted")
if os.path.exists(gingivitis_dir):
gingivitis_terms = {
"mild": ["치은염"],
"moderate": ["치은염", "치주염"],
"severe": ["치주염", "치주낭"],
"gingivitis": ["치은염"],
}
images = sorted([f for f in os.listdir(gingivitis_dir) if f.endswith('.jpg')])
for level, terms in gingivitis_terms.items():
level_images = [img for img in images if f"_{level}" in img]
if not level_images:
continue
for term_kr in terms:
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
if result:
rel_path = f"/api/files/dental_images/huggingface_gingivitis/extracted/{level_images[0]}"
updates.append((rel_path, result[0]))
# --- Oral Cancer dataset (Docty/Oral-Cancer) ---
oral_cancer_dir = os.path.join(IMG_BASE, "huggingface_oral_cancer", "extracted")
if os.path.exists(oral_cancer_dir):
cancer_terms = {
"cancer": ["구강암", "편평세포암", "구강암종"],
"normal": ["정상구강점막"],
}
for label, terms in cancer_terms.items():
label_images = sorted([f for f in os.listdir(oral_cancer_dir) if f"_{label}" in f])
if not label_images:
continue
for term_kr in terms:
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
if result:
rel_path = f"/api/files/dental_images/huggingface_oral_cancer/extracted/{label_images[0]}"
updates.append((rel_path, result[0]))
# --- Hugging Face X-ray segmentation dataset ---
xray_dir = os.path.join(IMG_BASE, "huggingface_xray")
if os.path.exists(xray_dir):
xray_terms = ["파노라마방사선사진", "치아방사선사진", "방사선사진"]
images = []
for root, dirs, files in os.walk(xray_dir):
for f in files:
if f.endswith(('.jpg', '.png', '.jpeg')):
images.append(os.path.relpath(os.path.join(root, f), xray_dir))
images.sort()
if images:
for term_kr in xray_terms:
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
if result:
rel_path = f"/api/files/dental_images/huggingface_xray/{images[0]}"
updates.append((rel_path, result[0]))
# --- Hugging Face caries dataset (original) ---
caries_dir = os.path.join(IMG_BASE, "huggingface_caries")
if os.path.exists(caries_dir):
images = []
for root, dirs, files in os.walk(caries_dir):
for f in files:
if f.endswith(('.jpg', '.png', '.jpeg')):
images.append(os.path.relpath(os.path.join(root, f), caries_dir))
images.sort()
if images:
for term_kr in ["치아우식증"]:
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
if result:
rel_path = f"/api/files/dental_images/huggingface_caries/{images[0]}"
updates.append((rel_path, result[0]))
return apply_updates(conn, updates, "huggingface", verify=verify)
# ============================================================
# 4. Wikimedia Commons 검색 (5초 딜레이 필수)
# ============================================================
def search_wikimedia_commons(conn, max_terms=50, delay=5, category=None,
ollama_url="http://127.0.0.1:11434/api/chat",
model=None):
"""Search Wikimedia Commons for dental term images with vision model validation"""
if model is None:
model = VERIFY_MODEL
def search_commons(query, limit=2):
url = (f"https://commons.wikimedia.org/w/api.php?action=query"
f"&list=search&srnamespace=6"
f"&srsearch={urllib.parse.quote(query)}&format=json&srlimit={limit}")
try:
req = urllib.request.Request(url, headers={"User-Agent": "DentalDictBot/2.0 (educational)"})
with urllib.request.urlopen(req, timeout=15) as resp:
data = json.loads(resp.read())
return data.get("query", {}).get("search", [])
except urllib.error.HTTPError as e:
if e.code == 429:
time.sleep(30)
return []
except Exception:
return []
def get_commons_url(title):
# SVG는 비전 모델이 직접 처리 불가 → iiurlwidth로 PNG 썸네일 URL 함께 요청
is_svg = title.lower().endswith(".svg")
extra = "&iiurlwidth=600" if is_svg else ""
url = (f"https://commons.wikimedia.org/w/api.php?action=query"
f"&titles={urllib.parse.quote(title)}&prop=imageinfo&iiprop=url&format=json{extra}")
try:
req = urllib.request.Request(url, headers={"User-Agent": "DentalDictBot/2.0 (educational)"})
with urllib.request.urlopen(req, timeout=15) as resp:
data = json.loads(resp.read())
pages = data.get("query", {}).get("pages", {})
for pid, page in pages.items():
if "imageinfo" in page:
ii = page["imageinfo"][0]
return ii.get("thumburl") or ii.get("url", "")
except Exception:
pass
return ""
# 수집 후 일괄 쓰기 (DB 락 방지)
updates = []
if category:
terms = conn.execute("""
SELECT id, korean, english, category
FROM terms WHERE (image_url IS NULL OR image_url = '') AND category=?
ORDER BY id LIMIT ?
""", (category, max_terms)).fetchall()
total_unmapped = conn.execute(
"SELECT COUNT(*) FROM terms WHERE (image_url IS NULL OR image_url='') AND category=?",
(category,)
).fetchone()[0]
else:
terms = conn.execute("""
SELECT id, korean, english, category
FROM terms WHERE (image_url IS NULL OR image_url = '')
ORDER BY category, id LIMIT ?
""", (max_terms,)).fetchall()
total_unmapped = conn.execute("SELECT COUNT(*) FROM terms WHERE image_url IS NULL OR image_url=''").fetchone()[0]
if total_unmapped > max_terms:
print(f" {total_unmapped} unmapped terms found — capping at {max_terms} (use --max-wikimedia to raise)", flush=True)
est_min = round(len(terms) * (delay + 3 + 1.5) / 60, 1)
print(f"Searching Wikimedia Commons for {len(terms)} terms (delay={delay}s, +{model} 검증, ~{est_min} min)...", flush=True)
for i, (term_id, korean, english, category) in enumerate(terms):
search_query = english if english else korean
time.sleep(delay)
results = search_commons(search_query)
if not results:
print(f" [{i+1}/{len(terms)}] {korean} no results", flush=True)
continue
best = None
for r in results:
title = r["title"]
if any(title.lower().endswith(ext) for ext in [".jpg", ".jpeg", ".png", ".svg"]):
best = title
break
if not best:
best = results[0]["title"]
time.sleep(3)
desc_url = get_commons_url(best)
if not desc_url:
print(f" [{i+1}/{len(terms)}] {korean} no URL", flush=True)
continue
res = validate_image(desc_url, korean, english or korean, category, model=model)
if res["ok"]:
updates.append((desc_url, term_id, korean))
print(f" [{i+1}/{len(terms)}] {korean} ({category}) ✓", flush=True)
else:
print(f" [{i+1}/{len(terms)}] {korean} ({category}) ❌ {res['reason'][:50]}", flush=True)
# 검증 통과 이미지 → 로컬 다운로드 후 일괄 쓰기
saved = 0
for ext_url, term_id, korean in updates:
local_url = localize_image(term_id, ext_url, korean=korean)
final_url = local_url or ext_url
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (final_url, term_id))
if local_url:
saved += 1
conn.commit()
print(f"Wikimedia: {len(updates)} verified, {saved} localized", flush=True)
return len(updates)
# ============================================================
# 5. Radiopaedia URL 매핑
# ============================================================
def map_radiopaedia_urls(conn):
"""Map Radiopaedia article URLs to terms (stored in source_url, not image_url)"""
existing = {row[1] for row in conn.execute("PRAGMA table_info(terms)")}
if "source_url" not in existing:
conn.execute("ALTER TABLE terms ADD COLUMN source_url TEXT")
mappings = {
# 구강악안면영상의학
"치아우식증": "https://radiopaedia.org/articles/dental-caries",
"치주염": "https://radiopaedia.org/articles/periodontitis",
"근첩농양": "https://radiopaedia.org/articles/dental-abscess",
"근첩낭종": "https://radiopaedia.org/articles/periapical-cyst",
"함치낭종": "https://radiopaedia.org/articles/dentigerous-cyst",
"함치종": "https://radiopaedia.org/articles/odontoma",
"골수염": "https://radiopaedia.org/articles/osteomyelitis",
"임플란트": "https://radiopaedia.org/articles/dental-implant",
"파노라마방사선사진": "https://radiopaedia.org/articles/orthopantomography",
"근첩병변": "https://radiopaedia.org/articles/apical-periodontitis",
"치아": "https://radiopaedia.org/articles/teeth",
"방사선골괴사": "https://radiopaedia.org/articles/mandibular-osteoradionecrosis",
"매복치": "https://radiopaedia.org/articles/impacted-tooth",
"치근흡수": "https://radiopaedia.org/articles/root-resorption",
"치은염": "https://radiopaedia.org/articles/gingivitis",
"하악골": "https://radiopaedia.org/articles/mandible",
"치은퇴축": "https://radiopaedia.org/articles/gingival-recession",
"구강암": "https://radiopaedia.org/articles/oral-cancer",
"악골골절": "https://radiopaedia.org/articles/mandibular-fracture",
"측두하악관절": "https://radiopaedia.org/articles/tmj",
"타액선조영술": "https://radiopaedia.org/articles/sialography",
"타액선염": "https://radiopaedia.org/articles/sialadenitis",
"근첩육아종": "https://radiopaedia.org/articles/periapical-granuloma",
# 구강외과
"발치": "https://radiopaedia.org/articles/tooth-extraction",
"사랑니발치": "https://radiopaedia.org/articles/wisdom-tooth-extraction",
"치조골흡수": "https://radiopaedia.org/articles/alveolar-bone-loss",
# 보존
"근관치료": "https://radiopaedia.org/articles/root-canal-treatment",
"치수염": "https://radiopaedia.org/articles/pulpitis",
# 치주
"치주낭": "https://radiopaedia.org/articles/periodontal-pocket",
"치석": "https://radiopaedia.org/articles/dental-calculus",
# 교정
"교정장치": "https://radiopaedia.org/articles/orthodontic-braces",
# 소아치과
"유전치": "https://radiopaedia.org/articles/deciduous-teeth",
# 구강병리
"편평세포암": "https://radiopaedia.org/articles/squamous-cell-carcinoma-head-and-neck",
"구강점막": "https://radiopaedia.org/articles/oral-mucosa",
# 임플란트
"골유착": "https://radiopaedia.org/articles/osseointegration",
# 기초치의학
"법랑질": "https://radiopaedia.org/articles/tooth-enamel",
"상아질": "https://radiopaedia.org/articles/dentine",
"치수": "https://radiopaedia.org/articles/dental-pulp",
"백아질": "https://radiopaedia.org/articles/cementum",
"치은": "https://radiopaedia.org/articles/gingiva",
"치주인대": "https://radiopaedia.org/articles/periodontal-ligament",
# 영상의학 추가
"CBCT": "https://radiopaedia.org/articles/cone-beam-computed-tomography-dental",
"컴퓨터단층촬영": "https://radiopaedia.org/articles/computed-tomography-head-technique",
"자기공명영상": "https://radiopaedia.org/articles/magnetic-resonance-imaging-head-technique",
"초음파검사": "https://radiopaedia.org/articles/ultrasound-head-and-neck-technique",
"상악동": "https://radiopaedia.org/articles/maxillary-sinus",
"하악관": "https://radiopaedia.org/articles/inferior-alveolar-canal",
}
updates = []
for korean, url in mappings.items():
result = conn.execute(
"SELECT id FROM terms WHERE korean=? AND (source_url IS NULL OR source_url='')",
(korean,)
).fetchone()
if result:
updates.append((url, result[0]))
for url, term_id in updates:
conn.execute("UPDATE terms SET source_url=? WHERE id=?", (url, term_id))
conn.commit()
return len(updates)
# ============================================================
# 6. ITU Dental Datasets 카탈로그 참조
# ============================================================
def map_itu_datasets(conn):
"""Map terms using ITU Dental Datasets catalog references (CC BY 4.0)
ITU catalog datasets (not yet downloaded, referenced for future use):
- Apical Periodontitis Panoramic (Mendeley: kx52tk2ddj/3) - 3,926 images
- Children's Dental Panoramic (figshare: c.6317013) - 193 images
- OCDC H&E OSCC (Mendeley: 9bsc36jyrt/1) - 1,020 images
- NDB-UFES Oral Cancer (Mendeley: bbmmm4wgr8/4) - 237+3,763 patches
- Panoramic Dental Xray Tunisia (Mendeley: 73n3kz2k4k/2) - 221 images
Note: Implant, Gingivitis, Oral Cancer datasets are handled by map_huggingface_datasets().
"""
# ITU-referenced datasets not yet downloaded — placeholder for future mapping
print(" ITU: no additional datasets to map (implant/gingivitis/oral_cancer handled by Hugging Face)")
return 0
# ============================================================
# 7. Roboflow REST API — 추가 프로젝트 (hosted URL, 다운로드 불필요)
# ============================================================
def map_roboflow_api_projects(conn, verify=True):
"""Map images via Roboflow API — additional projects not covered by local files"""
api_key_path = "/home/kim/homeclaw/.smallclaw/roboflow_api_key.txt"
if not os.path.exists(api_key_path):
print(" Roboflow API key not found, skipping")
return 0
with open(api_key_path) as f:
api_key = f.read().strip()
PROJECTS = [
{
"workspace": "yosafat_chandra05-yahoo-com",
"project": "dental-implants-2.0",
"version": 1,
"label_map": {
"implant": ["임플란트", "임플란트보철물"],
"Implant": ["임플란트", "임플란트보철물"],
"implant_bone": ["임플란트", "골유착"],
},
},
{
"workspace": "dental-mate",
"project": "dentalmate",
"version": 1,
"label_map": {
"Cavity": ["치아우식증"],
"Calculus": ["치석"],
"Gingivitis": ["치은염"],
"Mouth_Ulcer": ["구강궤양"],
"Tooth_Discoloration": ["치아변색"],
"Ulcers": ["구강궤양"],
"Hypodontia": ["선천결치증"],
"Caries": ["치아우식증"],
},
},
]
updates = []
for proj in PROJECTS:
ws, project, version = proj["workspace"], proj["project"], proj["version"]
label_map = proj["label_map"]
# Request COCO export — returns {"export": {"link": "...", "size": ...}}
export_url = (
f"https://api.roboflow.com/{ws}/{project}/{version}"
f"/coco?api_key={api_key}"
)
try:
req = urllib.request.Request(export_url, headers={"User-Agent": "DentalDictBot/2.0"})
with urllib.request.urlopen(req, timeout=20) as resp:
export_meta = json.loads(resp.read())
except Exception as e:
print(f" {project}: export API error - {e}")
continue
export_link = export_meta.get("export", {}).get("link", "")
if not export_link:
print(f" {project}: no export link in response")
continue
# Download the COCO annotation JSON from the export link
try:
req = urllib.request.Request(export_link, headers={"User-Agent": "DentalDictBot/2.0"})
with urllib.request.urlopen(req, timeout=60) as resp:
raw = resp.read()
# Export may be a zip; try JSON first
try:
coco_data = json.loads(raw)
except Exception:
import zipfile, io
with zipfile.ZipFile(io.BytesIO(raw)) as zf:
ann_name = next(
(n for n in zf.namelist() if n.endswith(".json")), None
)
if not ann_name:
print(f" {project}: no JSON inside zip")
continue
coco_data = json.loads(zf.read(ann_name))
except Exception as e:
print(f" {project}: COCO download/parse error - {e}")
continue
cats = {c["id"]: c["name"] for c in coco_data.get("categories", [])}
imgs = {img["id"]: img for img in coco_data.get("images", [])}
used_terms = set()
for ann in coco_data.get("annotations", []):
cat_name = cats.get(ann["category_id"], "")
if cat_name not in label_map:
continue
img_info = imgs.get(ann["image_id"], {})
img_url = img_info.get("path", "") or img_info.get("coco_url", "")
if not img_url:
continue
for term_kr in label_map[cat_name]:
if term_kr in used_terms:
continue
result = conn.execute(
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
(term_kr,),
).fetchone()
if result:
updates.append((img_url, result[0]))
used_terms.add(term_kr)
break
print(f" {project}: {len(used_terms)} terms matched")
return apply_updates(conn, updates, "roboflow_api", verify=verify)
# ============================================================
# 8. Zenodo REST API — 공개 데이터셋 (인증 불필요)
# ============================================================
def map_zenodo_records(conn, verify=True):
"""Map images from Zenodo public records via REST API"""
RECORDS = [
{
"id": "15487430",
"description": "14-class panoramic radiograph (CC BY 4.0)",
"label_map": {
"implant": ["임플란트"],
"Implant": ["임플란트"],
"cavity": ["치아우식증"],
"caries": ["치아우식증"],
"Caries": ["치아우식증"],
"periapical": ["근첩병변", "근첩농양"],
"Periapical": ["근첩병변"],
"crown": ["크라운"],
"Crown": ["크라운"],
"RCT": ["근관치료"],
"rct": ["근관치료"],
"filling": ["치과충전"],
"Filling": ["치과충전"],
"impacted": ["매복치"],
"Impacted": ["매복치"],
"bridge": ["고정성보철물"],
"Bridge": ["고정성보철물"],
"calculus": ["치석"],
"Calculus": ["치석"],
"gingivitis": ["치은염"],
"Gingivitis": ["치은염"],
},
},
]
updates = []
for record in RECORDS:
rec_id = record["id"]
label_map = record["label_map"]
try:
req = urllib.request.Request(
f"https://zenodo.org/api/records/{rec_id}",
headers={"User-Agent": "DentalDictBot/2.0"},
)
with urllib.request.urlopen(req, timeout=20) as resp:
meta = json.loads(resp.read())
except Exception as e:
print(f" Zenodo {rec_id}: API error - {e}")
continue
files = meta.get("files", [])
print(f" Zenodo {rec_id}: {len(files)} files ({record['description']})")
img_files = [
f for f in files
if f.get("key", "").lower().endswith((".jpg", ".jpeg", ".png"))
]
ann_files = [
f for f in files
if f.get("key", "").lower().endswith(".json")
]
used_terms = set()
if img_files:
# Direct image files — match by filename keyword
for f in img_files:
key = f.get("key", "").lower()
dl_url = f.get("links", {}).get("self", "")
if not dl_url:
continue
for label, korean_terms in label_map.items():
if label.lower() in key:
for term_kr in korean_terms:
if term_kr in used_terms:
continue
result = conn.execute(
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
(term_kr,),
).fetchone()
if result:
updates.append((dl_url, result[0]))
used_terms.add(term_kr)
break
print(f" Zenodo {rec_id}: {len(used_terms)} terms matched from images")
elif ann_files:
# Try first COCO annotation JSON
ann_url = ann_files[0].get("links", {}).get("self", "")
if ann_url:
try:
req = urllib.request.Request(ann_url, headers={"User-Agent": "DentalDictBot/2.0"})
with urllib.request.urlopen(req, timeout=30) as resp:
ann_data = json.loads(resp.read())
if isinstance(ann_data, dict) and "images" in ann_data:
cats = {c["id"]: c["name"] for c in ann_data.get("categories", [])}
imgs = {img["id"]: img for img in ann_data.get("images", [])}
# Map image filenames to Zenodo download URLs
fname_to_url = {}
for f in files:
fkey = f.get("key", "")
furl = f.get("links", {}).get("self", "")
if furl:
fname_to_url[os.path.basename(fkey)] = furl
for ann in ann_data.get("annotations", []):
cat_name = cats.get(ann["category_id"], "")
if cat_name not in label_map:
continue
img_info = imgs.get(ann["image_id"], {})
fname = os.path.basename(img_info.get("file_name", ""))
img_url = fname_to_url.get(fname, "")
if not img_url:
continue
for term_kr in label_map[cat_name]:
if term_kr in used_terms:
continue
result = conn.execute(
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
(term_kr,),
).fetchone()
if result:
updates.append((img_url, result[0]))
used_terms.add(term_kr)
break
print(f" Zenodo {rec_id}: {len(used_terms)} terms matched from COCO")
except Exception as e:
print(f" Zenodo {rec_id}: annotation error - {e}")
else:
zip_count = sum(1 for f in files if f.get("key", "").lower().endswith(".zip"))
print(f" Zenodo {rec_id}: {zip_count} zip files — download locally first")
return apply_updates(conn, updates, "zenodo", verify=verify)
# ============================================================
# 9. Mendeley Data — 소아치과 이미지 (CC BY 4.0)
# ============================================================
def map_mendeley_dataset(conn, verify=True):
"""Map images from Mendeley Data public datasets (no API key required for public records)
Dataset 6zsnhrds9t: Teeth/Dental images from children 1-14 years
9,562 intraoral photographs, 8 standardized clinical views (CC BY 4.0)
"""
DATASETS = [
{
"id": "6zsnhrds9t",
"version": 1,
"default_terms": ["유치", "유전치", "유구치", "혼합치열", "소아치과"],
},
]
updates = []
for ds in DATASETS:
ds_id, version = ds["id"], ds["version"]
default_terms = ds["default_terms"]
api_url = f"https://data.mendeley.com/api/datasets/{ds_id}/versions/{version}"
try:
req = urllib.request.Request(
api_url,
headers={"User-Agent": "DentalDictBot/2.0", "Accept": "application/json"},
)
with urllib.request.urlopen(req, timeout=20) as resp:
meta = json.loads(resp.read())
except Exception as e:
print(f" Mendeley {ds_id}: API error - {e}")
continue
files = meta.get("files", [])
print(f" Mendeley {ds_id}: {len(files)} files")
img_files = [
f for f in files
if f.get("filename", "").lower().endswith((".jpg", ".jpeg", ".png"))
]
if not img_files:
print(f" Mendeley {ds_id}: no direct image files (likely zip archives)")
continue
used_terms = set()
for f in img_files[: len(default_terms)]:
dl_url = (
f.get("download_url", "")
or (f.get("content_details") or {}).get("download_url", "")
)
if not dl_url:
continue
for term_kr in default_terms:
if term_kr in used_terms:
continue
result = conn.execute(
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
(term_kr,),
).fetchone()
if result:
updates.append((dl_url, result[0]))
used_terms.add(term_kr)
break
print(f" Mendeley {ds_id}: {len(used_terms)} terms matched")
return apply_updates(conn, updates, "mendeley", verify=verify)
# ============================================================
# 10. NLM Open-i — PMC figure-level search + 검증
# ============================================================
def map_openi(conn, max_terms=50, delay=0.5, category=None,
model=VERIFY_MODEL):
"""Map dental terms to images via NLM Open-i.
Open-i indexes figures from PMC at the figure level (richer than
map_pmc_oa's article-HTML scraping). Each result includes a figure
caption which we feed to the validator as a hint, improving match
accuracy especially for procedure / technique terms.
- Endpoint: https://openi.nlm.nih.gov/api/search?query=...&m=...&n=...
- No auth, generous rate (we still throttle with `delay`).
- License: PMC Open Access subset (CC-BY / similar).
- Per term we fetch up to 5 candidates, validate each with the
configured vision model (defaults to VERIFY_MODEL = qwen3.5),
and save the first that passes. Commits per-term (checkpoint).
"""
BASE = "https://openi.nlm.nih.gov"
def openi_search(query, n=5, max_retries=4):
# Open-i is slow and intermittently returns 500 / empty body.
# Back off progressively: 10s, 30s, 60s.
backoff = [10, 30, 60]
url = (f"{BASE}/api/search?"
f"query={urllib.parse.quote(query)}&m=1&n={n}")
last_err = None
for attempt in range(max_retries):
try:
req = urllib.request.Request(
url, headers={"User-Agent": "DentalDictBot/2.0"})
with urllib.request.urlopen(req, timeout=120) as r:
body = r.read()
return json.loads(body).get("list", [])
except Exception as e:
last_err = e
if attempt < max_retries - 1:
wait = backoff[attempt] if attempt < len(backoff) else 60
print(f" ! retry {attempt+1}/{max_retries} after {wait}s: {e}", flush=True)
time.sleep(wait)
print(f" ! Open-i error after {max_retries} tries: {last_err}",
flush=True)
return []
# Pick unmapped terms with an English label; category filter or all.
if category:
terms = conn.execute("""
SELECT id, korean, english, category FROM terms
WHERE (image_url IS NULL OR image_url = '')
AND english IS NOT NULL AND english != ''
AND category = ?
ORDER BY id LIMIT ?
""", (category, max_terms)).fetchall()
else:
terms = conn.execute("""
SELECT id, korean, english, category FROM terms
WHERE (image_url IS NULL OR image_url = '')
AND english IS NOT NULL AND english != ''
ORDER BY category, id LIMIT ?
""", (max_terms,)).fetchall()
print(f"Open-i: {len(terms)} terms"
f"{f' in {category}' if category else ''}"
f" (max={max_terms}, delay={delay}s, model={model})", flush=True)
est = round(len(terms) * (delay + 1 + 10) / 60, 1) # ~1s API + ~10s verify
print(f" Estimated time: ~{est} min", flush=True)
# 카테고리별 검색어 suffix: 기본은 빈 문자열, 방사선은 영상 특화
_QUERY_SUFFIX = {
"구강악안면영상의학": "oral radiograph OR oral X-ray",
}
updates = []
for i, (tid, ko, en, cat) in enumerate(terms):
time.sleep(delay)
suffix = _QUERY_SUFFIX.get(cat, "")
query = f"{en} {suffix}".strip()
results = openi_search(query, n=5)
print(f" [{i+1}/{len(terms)}] {ko} ({cat}) — {len(results)} hits", flush=True)
assigned = False
for r in results:
img_path = r.get("imgLarge") or ""
if not img_path:
continue
img_url = BASE + img_path
caption = (r.get("image") or {}).get("caption", "")
hint = f"논문 그림 캡션 참고: {caption[:200]}" if caption else None
res = validate_image(img_url, ko, en, cat, hint=hint, model=model)
if res["ok"]:
# 검증 즉시 로컬화, 실패 시 외부 URL 유지
local_url = localize_image(tid, img_url, korean=ko)
updates.append((local_url or img_url, tid))
print(f" ✓ 저장 (PMID {r.get('pmid','?')})", flush=True)
assigned = True
break
if not assigned:
print(f" — 적합 이미지 없음", flush=True)
for url, tid in updates:
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
conn.commit()
print(f"Open-i: {len(updates)} mappings committed", flush=True)
return len(updates)
# ============================================================
# 11. Figshare — 공개 학술 그림 (무인증, item_type=1)
# ============================================================
def map_figshare(conn, max_terms=50, delay=1, category=None, model=VERIFY_MODEL):
"""Map dental terms to figures from Figshare (open access, no auth required).
Searches figure-type items (item_type=1) on Figshare by English dental term,
validates each image with the vision model, then localizes.
Rate limit: 5000 req/hr — 1s delay is safe.
"""
def _search(query, page_size=5):
url = "https://api.figshare.com/v2/articles/search"
payload = json.dumps({
"search_for": query,
"item_type": 1, # Figure
"page_size": page_size,
}).encode()
req = urllib.request.Request(url, data=payload, headers={
"Content-Type": "application/json",
"User-Agent": "DentalDictBot/2.0 (educational)",
})
with urllib.request.urlopen(req, timeout=15) as r:
return json.loads(r.read())
def _files(article_id):
url = f"https://api.figshare.com/v2/articles/{article_id}/files"
req = urllib.request.Request(url, headers={"User-Agent": "DentalDictBot/2.0"})
with urllib.request.urlopen(req, timeout=15) as r:
return json.loads(r.read())
if category:
terms = conn.execute("""
SELECT id, korean, english, category FROM terms
WHERE (image_url IS NULL OR image_url='') AND category=?
AND english IS NOT NULL AND english != ''
ORDER BY id LIMIT ?
""", (category, max_terms)).fetchall()
else:
terms = conn.execute("""
SELECT id, korean, english, category FROM terms
WHERE (image_url IS NULL OR image_url='')
AND english IS NOT NULL AND english != ''
ORDER BY category, id LIMIT ?
""", (max_terms,)).fetchall()
print(f"Figshare: {len(terms)} terms"
f"{f' in {category}' if category else ''} (model={model})", flush=True)
# 카테고리별 검색 힌트 — Figshare는 재료·현미경 이미지에 특히 강함
_SUFFIX = {
"치과생체재료학": "SEM microscopy",
"기초치의학": "histology microscopy",
"디지털치의학": "CAD CAM scanning",
"구강악안면영상의학": "radiograph X-ray",
}
updates = []
for i, (tid, ko, en, cat) in enumerate(terms):
time.sleep(delay)
suffix = _SUFFIX.get(cat, "dental")
try:
articles = _search(f"{en} {suffix}", page_size=5)
# 결과 없으면 suffix 없이 재시도
if not articles:
articles = _search(en, page_size=5)
except Exception as e:
print(f" [{i+1}/{len(terms)}] {ko} — search error: {e}", flush=True)
continue
if not articles:
print(f" [{i+1}/{len(terms)}] {ko} — no results", flush=True)
continue
print(f" [{i+1}/{len(terms)}] {ko} — {len(articles)} figures", flush=True)
assigned = False
for art in articles:
try:
files = _files(art["id"])
except Exception:
continue
img_files = [
f for f in files
if f.get("name", "").lower().endswith(
(".jpg", ".jpeg", ".png", ".gif", ".webp", ".tif", ".tiff"))
]
if not img_files:
continue
img_url = img_files[0]["download_url"]
res = validate_image(img_url, ko, en, cat, model=model)
if res["ok"]:
local_url = localize_image(tid, img_url, korean=ko)
updates.append((local_url or img_url, tid))
print(f" ✓ {art.get('title','?')[:50]}", flush=True)
assigned = True
break
if not assigned:
print(f" — 적합 이미지 없음", flush=True)
for url, tid in updates:
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
conn.commit()
print(f"Figshare: {len(updates)} mappings committed", flush=True)
return len(updates)
# ============================================================
# 12. PMC Entrez — 논문 Figure 직접 추출 (JATS XML)
# ============================================================
def map_pmc_entrez(conn, max_terms=50, delay=1, category=None, model=VERIFY_MODEL):
"""Map dental terms to PMC Open Access article figures via NCBI Entrez.
Difference from map_openi() (NLM Open-i index):
- Searches the full PMC OA database (not just Open-i indexed subset)
- Fetches JATS XML → parses ALL <fig><graphic> elements per article
- Can find images for rare/specialized terms openi misses
Rate: 3 req/s without API key → delay=1 is safe.
"""
import xml.etree.ElementTree as ET
ESEARCH = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi"
EFETCH = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/efetch.fcgi"
XLINK = "{http://www.w3.org/1999/xlink}"
PMC_BIN = "https://www.ncbi.nlm.nih.gov/pmc/articles/PMC{pmcid}/bin/{href}{suffix}"
if category:
terms = conn.execute("""
SELECT id, korean, english, category FROM terms
WHERE (image_url IS NULL OR image_url='') AND category=?
AND english IS NOT NULL AND english != ''
ORDER BY id LIMIT ?
""", (category, max_terms)).fetchall()
else:
terms = conn.execute("""
SELECT id, korean, english, category FROM terms
WHERE (image_url IS NULL OR image_url='')
AND english IS NOT NULL AND english != ''
ORDER BY category, id LIMIT ?
""", (max_terms,)).fetchall()
print(f"PMC Entrez: {len(terms)} terms"
f"{f' in {category}' if category else ''} (model={model})", flush=True)
updates = []
for i, (tid, ko, en, cat) in enumerate(terms):
time.sleep(delay)
# 1. esearch: PMC articles for this dental term (OA 필터 없이 더 넓게)
q = urllib.parse.quote(en)
try:
req = urllib.request.Request(
f"{ESEARCH}?db=pmc&term={q}&retmax=5&retmode=json",
headers={"User-Agent": "DentalDictBot/2.0"},
)
with urllib.request.urlopen(req, timeout=15) as r:
pmcids = json.loads(r.read()).get("esearchresult", {}).get("idlist", [])
except Exception as e:
print(f" [{i+1}/{len(terms)}] {ko} — esearch error: {e}", flush=True)
continue
if not pmcids:
print(f" [{i+1}/{len(terms)}] {ko} — no PMC hits", flush=True)
continue
print(f" [{i+1}/{len(terms)}] {ko} — {len(pmcids)} articles", flush=True)
assigned = False
for pmcid in pmcids[:3]:
time.sleep(0.5)
try:
req = urllib.request.Request(
f"{EFETCH}?db=pmc&id={pmcid}&retmode=xml",
headers={"User-Agent": "DentalDictBot/2.0"},
)
with urllib.request.urlopen(req, timeout=30) as r:
root = ET.fromstring(r.read())
except Exception as e:
print(f" PMC{pmcid} fetch error: {e}", flush=True)
continue
# 2. JATS XML에서 <fig><graphic xlink:href> 수집
fig_graphics = []
for fig in root.iter("fig"):
caption = " ".join(fig.itertext())[:200].strip()
for graphic in fig.iter("graphic"):
href = graphic.get(f"{XLINK}href") or graphic.get("href", "")
if href:
fig_graphics.append((href, caption))
if not fig_graphics:
continue
for href, caption in fig_graphics[:6]:
# PMC는 href 그대로, 또는 .jpg/.png 추가 (PMC 변환 관행)
# 예: href="fig1" → bin/fig1.jpg 또는 bin/fig1.png
# 예: href="fig1.jpg" → bin/fig1.jpg.png (PMC TIFF→PNG 변환 시)
img_url = None
for suffix in ["", ".jpg", ".png", ".gif"]:
candidate = PMC_BIN.format(pmcid=pmcid, href=href, suffix=suffix)
try:
urllib.request.urlopen(
urllib.request.Request(
candidate,
headers={"User-Agent": "DentalDictBot/2.0"},
),
timeout=8,
).close()
img_url = candidate
break
except Exception:
continue
if not img_url:
continue
hint = f"논문 그림 캡션: {caption}" if caption else None
res = validate_image(img_url, ko, en, cat, hint=hint, model=model)
if res["ok"]:
local_url = localize_image(tid, img_url, korean=ko)
updates.append((local_url or img_url, tid))
print(f" ✓ PMC{pmcid}/{href} ({res['reason'][:40]})", flush=True)
assigned = True
break
if assigned:
break
if not assigned:
print(f" — 적합 이미지 없음", flush=True)
for url, tid in updates:
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
conn.commit()
print(f"PMC Entrez: {len(updates)} mappings committed", flush=True)
return len(updates)
# ============================================================
# 공통 헬퍼 — 영문명으로 미매핑 term 찾기
# ============================================================
def _find_unmapped_by_en(conn, en, category=None):
"""영문명 en(소문자)과 매칭되는 미매핑 term (id,korean,english,category) 반환, 없으면 None.
정확일치 → 포함(양방향) 순."""
en = (en or "").strip().lower()
if not en:
return None
if category:
rows = conn.execute(
"SELECT id,korean,english,category FROM terms "
"WHERE (image_url IS NULL OR image_url='') AND category=? "
"AND english IS NOT NULL AND english!=''", (category,)).fetchall()
else:
rows = conn.execute(
"SELECT id,korean,english,category FROM terms "
"WHERE (image_url IS NULL OR image_url='') "
"AND english IS NOT NULL AND english!=''").fetchall()
for tid, ko, e, cat in rows:
if e and e.strip().lower() == en:
return (tid, ko, e, cat)
for tid, ko, e, cat in rows:
if e and en in e.strip().lower():
return (tid, ko, e, cat)
for tid, ko, e, cat in rows:
if e and e.strip().lower() and e.strip().lower() in en:
return (tid, ko, e, cat)
return None
# ============================================================
# 13. VCU Oral Pathology Review — 교육용 병리 아틀라스 스크랩
# ============================================================
def map_vcu(conn, max_terms=60, delay=1, category=None, model=VERIFY_MODEL):
"""VCU Oral Pathology Review (scholarscompass.vcu.edu/opr).
~60개 구강 병리 질환, MeSH 인덱스. 인덱스→상세페이지 preview.jpg 추출.
영문 질환명 → DB english 매칭. 비영리 교육용 (radiopaedia와 동일 취급)."""
BASE = "https://scholarscompass.vcu.edu"
INDEX = BASE + "/opr/index.html"
def _fetch(url):
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
return urllib.request.urlopen(req, timeout=20).read().decode("utf-8", "ignore")
# 인덱스 페이지(1~2)에서 질환 링크 수집
conditions = {}
for page in ["", "?page=2"]:
try:
html = _fetch(INDEX + page)
except Exception as e:
print(f" VCU index{page} 오류: {e}", flush=True)
continue
for cid, name in re.findall(r'/opr/(\d+)[^"]*"[^>]*>([^<]+)</a>', html):
name = name.strip()
if name and name.lower() != "view slideshow" and cid not in conditions:
conditions[cid] = name
print(f"VCU: {len(conditions)} conditions (model={model})", flush=True)
if not conditions:
return 0
# 영문명 정제: 괄호 제거
def _clean(name):
return re.sub(r"\s*\([^)]*\)", "", name).strip().lower()
updates = []
used = set()
for i, (cid, name) in enumerate(conditions.items()):
if len(used) >= max_terms:
break
time.sleep(delay)
try:
html = _fetch(f"{BASE}/opr/{cid}")
except Exception as e:
print(f" [{i+1}] {name} — 페이지 오류: {e}", flush=True)
continue
m = re.search(rf"{BASE}/opr/(\d+)/preview\.jpg", html)
if not m:
print(f" [{i+1}] {name} — 이미지 없음", flush=True)
continue
img_url = f"{BASE}/opr/{m.group(1)}/preview.jpg"
clean = _clean(name)
term = _find_unmapped_by_en(conn, clean, category=category)
if not term:
print(f" [{i+1}] {name} — 매칭 term 없음", flush=True)
continue
tid, ko, en, cat = term
if tid in used:
continue
res = validate_image(img_url, ko, en, cat, hint=f"VCU 병리 이미지: {name}", model=model)
if res["ok"]:
local_url = localize_image(tid, img_url, korean=ko)
updates.append((local_url or img_url, tid))
used.add(tid)
print(f" [{i+1}] {name} → {ko} ✓", flush=True)
else:
print(f" [{i+1}] {name} ✗ {res.get('reason','')[:50]}", flush=True)
for url, tid in updates:
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
conn.commit()
print(f"VCU: {len(updates)} mappings committed", flush=True)
return len(updates)
# ============================================================
# 14. OralSDv1 — GitHub 폴더별 일반 구강질환 (6종, ~1163장)
# ============================================================
def map_oralsd(conn, max_terms=50, delay=0.5, category=None, model=VERIFY_MODEL):
"""OralSDv1 (github.com/enderXM249/OralSDv1). 폴더별 6종 일반질환 이미지.
Calculus/Caries/Gingivitis/Tooth_Discoloration/Ulcers/Hypodontia.
GitHub raw URL 직접 사용. 라이선스: 연구용(저자 문의)."""
API = "https://api.github.com/repos/enderXM249/OralSDv1/git/trees/main?recursive=1"
RAW = "https://raw.githubusercontent.com/enderXM249/OralSDv1/main/"
LABEL = {
"Calculus": ["치석"],
"Caries": ["치아우식증", "우식증"],
"Gingivitis": ["치은염"],
"Tooth_Discoloration": ["치아변색"],
"Ulcers": ["구강궤양", "아프투스궤양"],
"Hypodontia": ["무치아증", "선천적결손치"],
}
try:
d = json.loads(urllib.request.urlopen(
urllib.request.Request(API, headers={"User-Agent": "DentalDictBot/2.0"}),
timeout=20).read())
except Exception as e:
print(f" OralSDv1 tree 오류: {e}", flush=True)
return 0
folders = {}
for t in d.get("tree", []):
if t.get("type") == "blob":
p = t["path"]
if "/" in p and p.lower().endswith((".jpg", ".jpeg", ".png")):
folders.setdefault(p.split("/")[0], []).append(p)
print(f"OralSDv1: {len(folders)} condition folders (model={model})", flush=True)
updates = []
used = set()
for folder, paths in folders.items():
if len(used) >= max_terms:
break
korean_terms = LABEL.get(folder)
if not korean_terms:
continue
for term_kr in korean_terms:
row = conn.execute(
"SELECT id,korean,english,category FROM terms "
"WHERE korean=? AND (image_url IS NULL OR image_url='')",
(term_kr,)).fetchone()
if not row or row[0] in used:
continue
tid, ko, en, cat = row
assigned = False
for p in paths[:8]:
img_url = RAW + urllib.parse.quote(p)
time.sleep(delay)
res = validate_image(img_url, ko, en, cat,
hint=f"OralSDv1 {folder}", model=model)
if res["ok"]:
local_url = localize_image(tid, img_url, korean=ko)
updates.append((local_url or img_url, tid))
used.add(tid)
print(f" ✓ {folder} → {ko}", flush=True)
assigned = True
break
if not assigned:
print(f" — {folder}/{ko} 적합 이미지 없음", flush=True)
break
for url, tid in updates:
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
conn.commit()
print(f"OralSDv1: {len(updates)} mappings committed", flush=True)
return len(updates)
# ============================================================
# 15. AKUDENTAL — GitHub 파노라마 방사선 + 인스턴스 매니페스트
# ============================================================
def map_akudental(conn, max_terms=20, delay=0.5, category=None, model=VERIFY_MODEL):
"""AKUDENTAL (github.com/melihoz/akudental). 333 파노라마 방사선, CC-BY-4.0.
매니페스트(akudental_instances.json)에서 이미지 목록 추출 → raw URL.
구강악안면영상의학 카테고리 미매핑 term에 파노라마 이미지 매핑(비전 검증)."""
API = "https://api.github.com/repos/melihoz/akudental/git/trees/main?recursive=1"
RAW = "https://raw.githubusercontent.com/melihoz/akudental/main/"
try:
d = json.loads(urllib.request.urlopen(
urllib.request.Request(API, headers={"User-Agent": "DentalDictBot/2.0"}),
timeout=20).read())
except Exception as e:
print(f" AKUDENTAL tree 오류: {e}", flush=True)
return 0
# 매니페스트에서 이미지 파일명 추출
manifest = None
img_paths = []
for t in d.get("tree", []):
p = t.get("path", "")
if p.endswith("akudental_instances.json"):
manifest = p
elif p.lower().endswith((".jpg", ".jpeg", ".png")) and "image" in p.lower():
img_paths.append(p)
if manifest:
try:
raw = urllib.request.urlopen(urllib.request.Request(
RAW + urllib.parse.quote(manifest),
headers={"User-Agent": "DentalDictBot/2.0"}), timeout=30).read()
data = json.loads(raw)
# 매니페스트 구조 가변 — image 경로 키 탐색
extra = []
if isinstance(data, list):
for item in data[:50]:
if isinstance(item, dict):
for v in item.values():
if isinstance(v, str) and v.lower().endswith((".jpg", ".jpeg", ".png")):
extra.append(v)
elif isinstance(data, dict):
for v in data.values():
if isinstance(v, str) and v.lower().endswith((".jpg", ".jpeg", ".png")):
extra.append(v)
img_paths = extra or img_paths
except Exception as e:
print(f" AKUDENTAL manifest 파싱 오류: {e}", flush=True)
print(f"AKUDENTAL: {len(img_paths)} panoramic images (model={model})", flush=True)
if not img_paths:
return 0
# 파노라마/방사선 관련 미매핑 term (영상의학 카테고리 우선)
if category:
rows = conn.execute(
"SELECT id,korean,english,category FROM terms "
"WHERE (image_url IS NULL OR image_url='') AND category=? "
"AND (english LIKE '%panoramic%' OR english LIKE '%radiograph%' "
"OR korean LIKE '%파노라마%' OR korean LIKE '%방사선%') "
"ORDER BY id LIMIT ?", (category, max_terms)).fetchall()
else:
rows = conn.execute(
"SELECT id,korean,english,category FROM terms "
"WHERE (image_url IS NULL OR image_url='') "
"AND (english LIKE '%panoramic%' OR english LIKE '%radiograph%' "
"OR korean LIKE '%파노라마%' OR korean LIKE '%방사선%') "
"ORDER BY id LIMIT ?", (max_terms,)).fetchall()
print(f"AKUDENTAL: {len(rows)} 후보 term", flush=True)
updates = []
used = set()
for tid, ko, en, cat in rows:
assigned = False
for p in img_paths[:10]:
img_url = RAW + urllib.parse.quote(p)
time.sleep(delay)
res = validate_image(img_url, ko, en, cat, hint="파노라마 방사선", model=model)
if res["ok"]:
local_url = localize_image(tid, img_url, korean=ko)
updates.append((local_url or img_url, tid))
used.add(tid)
print(f" ✓ {ko}", flush=True)
assigned = True
break
if not assigned:
print(f" — {ko} 적합 없음", flush=True)
if len(used) >= max_terms:
break
for url, tid in updates:
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
conn.commit()
print(f"AKUDENTAL: {len(updates)} mappings committed", flush=True)
return len(updates)
# ============================================================
# 16. COde — HuggingFace zirak-ai/COde (964MB zip 로컬 다운로드 필요)
# ============================================================
def _llm_label_match_batch(batch, label_list, model=VERIFY_MODEL):
"""Text-only LLM call: map each (tid, ko, en, cat) in batch to the best
COde label name, or None if no clinical photo could represent it.
Returns {tid: label_str or None}."""
label_block = ", ".join(sorted(label_list))
lines = "\n".join(
f"{i+1}. {en} ({ko}) [{cat}]"
for i, (tid, ko, en, cat) in enumerate(batch)
)
prompt = (
"Pick the best COde label for each dental term, or 'none' if no clinical oral "
"photo could represent it (e.g. bacteria names, histology, classification systems, "
"procedure names with no visual signature).\n"
f"Labels: {label_block}\n\n"
"Format: <number>. <exact label name or none> — one line per term, no explanations.\n\n"
f"Terms:\n{lines}"
)
payload = {"model": model, "messages": [{"role": "user", "content": prompt}], "stream": False}
try:
req = urllib.request.Request(
OLLAMA_URL, data=json.dumps(payload).encode(),
headers={"Content-Type": "application/json"},
)
with urllib.request.urlopen(req, timeout=180) as r:
text = json.loads(r.read()).get("message", {}).get("content", "")
except Exception as e:
print(f" LLM label match error: {e}", flush=True)
return {}
label_lower = {l.lower(): l for l in label_list}
result = {}
for line in text.splitlines():
m = re.match(r"^(\d+)\.\s*(.+)$", line.strip())
if not m:
continue
idx = int(m.group(1)) - 1
raw = m.group(2).strip().lower().rstrip(".")
if idx < 0 or idx >= len(batch):
continue
tid = batch[idx][0]
result[tid] = label_lower.get(raw) # None if "none" or unrecognised
return result
def map_code(conn, verify=True, max_terms=None, category=None, llm_match=False):
"""COde (HuggingFace zirak-ai/COde). ~50k 구강내사진 + 8k 방사선 + 진단.
HF datasets-server 미지원 → COde-Dataset.zip(964MB)을 dental_images/code/에
다운로드 후 풀어야 함.
구조: complete_dataset.csv 매니페스트 (8775행).
- photographs : Images/Photographs/ 내 파일명 (쉼표 구분 다중)
- anomalies_en : 영문 카테고리 라벨 (Gingivitis, Dental Caries, Pulpitis,
Periodontitis, Class II Malocclusion, ... 쉼표 구분)
라벨 → DB term english 매칭 → term별 최대 3장 후보 비전 검증 → 첫 OK 저장.
image_url = /api/files/dental_images/code/Images/Photographs/<fn> (게이트웨이 서빙).
max_terms: 최대 매핑 term 수 (None=전체). category: 해당 카테고리 term만 매핑."""
import csv as _csv
root = os.path.join(IMG_BASE, "code")
csv_path = os.path.join(root, "complete_dataset.csv")
photo_dir = os.path.join(root, "Images", "Photographs")
if not os.path.exists(csv_path):
print("COde: complete_dataset.csv 없음 — COde-Dataset.zip(964MB) 다운로드 후 "
"dental_images/code/ 에 해제하세요.", flush=True)
print(" https://huggingface.co/datasets/zirak-ai/COde/resolve/main/COde-Dataset.zip",
flush=True)
return 0
# 1. CSV → label(소문자) → [절대경로 이미지 후보]
# 단일-anomaly 행의 사진이 그 anomaly를 정확히 담을 확률이 높으므로
# 단일-anomaly 후보를 먼저 배치, 부족 시 다중-anomaly 행 후보로 보충.
single = {} # label -> [imgp] (단일-anomaly 행)
multi = {} # label -> [imgp] (다중-anomaly 행)
n_rows = 0
with open(csv_path, encoding="utf-8-sig") as f:
for row in _csv.DictReader(f):
n_rows += 1
photos = (row.get("photographs") or "").strip()
anom = (row.get("anomalies_en") or "").strip()
if not photos or not anom:
continue
labs = [x.strip().lower() for x in anom.replace('"', "").split(",") if x.strip()]
is_single = len(labs) == 1
for fn in photos.split(","):
fn = fn.strip()
if not fn:
continue
imgp = os.path.join(photo_dir, fn)
if not os.path.exists(imgp):
continue
for lab in labs:
(single if is_single else multi).setdefault(lab, []).append(imgp)
label_to_images = {}
for k in set(single) | set(multi):
label_to_images[k] = list(dict.fromkeys(single.get(k, []) + multi.get(k, [])))[:50]
print(f"COde: {n_rows} rows, {len(label_to_images)} labels, "
f"{sum(len(v) for v in label_to_images.values())} candidate images "
f"(single-anomaly 우선)", flush=True)
if not label_to_images:
return 0
# 2. 미매핑 term → 매칭 라벨 → 후보 검증
where = "(image_url IS NULL OR image_url='') AND english IS NOT NULL AND english!=''"
params = []
if category:
where += " AND category=?"
params.append(category)
terms = conn.execute(
"SELECT id,korean,english,category FROM terms WHERE " + where, params
).fetchall()
print(f"COde: {len(terms)} unmapped terms"
+ (f" (category={category})" if category else ""), flush=True)
def _word_match(enl, lab):
"""enl == lab 또는 한쪽이 단어경계 내에서 다른쪽에 포함.
부분문자열 사고(pit↔pulpitis 등) 방지."""
if enl == lab:
return True
if re.search(r"\b" + re.escape(lab) + r"\b", enl):
return True
if re.search(r"\b" + re.escape(enl) + r"\b", lab):
return True
return False
# COde는 질환 '묘사' 사진만 있으므로 예방/역학/병인 등 비-묘사 개념 term은 스킵
NON_DEPICT = re.compile(
r"\b(prevention|epidemiology|etiology|pathogenesis|classification|"
r"prognosis|prevalence|incidence|management of|treatment of|"
r"diagnosis of|definition|overview|introduction|history of)\b"
)
label_ptrs = {k: 0 for k in label_to_images}
updates = []
used = set()
MAX_TRIES = 3
def _assign(tid, ko, en, cat, best):
"""Try to assign an image from label_to_images[best]. Returns True on success."""
cands = label_to_images[best]
ptr = label_ptrs[best]
tries = 0
while ptr < len(cands) and tries < MAX_TRIES:
imgp = cands[ptr]
ptr += 1
tries += 1
if not os.path.exists(imgp):
continue
if verify:
res = validate_image(imgp, ko, en, cat,
hint=f"COde label: {best}", model=VERIFY_MODEL)
if not res["ok"]:
continue
fn = os.path.basename(imgp)
url = f"{URL_PREFIX}/code/Images/Photographs/{fn}"
updates.append((url, tid))
used.add(tid)
label_ptrs[best] = ptr
print(f" ✓ {ko} ← {best}", flush=True)
return True
label_ptrs[best] = ptr
return False
# Pass 1: 단어경계 문자열 매칭
unmatched = []
for tid, ko, en, cat in terms:
if max_terms and len(updates) >= max_terms:
break
enl = en.strip().lower()
if NON_DEPICT.search(enl):
continue
matches = [lab for lab in label_to_images if _word_match(enl, lab)]
if not matches:
unmatched.append((tid, ko, en, cat))
continue
best = max(matches, key=len)
_assign(tid, ko, en, cat, best)
# Pass 2: LLM 의미론적 라벨 매칭 (문자열 매칭 실패 term)
if llm_match and unmatched:
# 이미 매핑된 term은 제외
pending = [(tid, ko, en, cat) for tid, ko, en, cat in unmatched if tid not in used]
print(f" LLM 라벨 매칭: {len(pending)}개 term → 배치 처리 중...", flush=True)
BATCH = 20
label_list = list(label_to_images.keys())
for i in range(0, len(pending), BATCH):
if max_terms and len(updates) >= max_terms:
break
batch = pending[i:i + BATCH]
print(f" 배치 {i // BATCH + 1}/{(len(pending) + BATCH - 1) // BATCH} "
f"({len(batch)}개)", flush=True)
label_map = _llm_label_match_batch(batch, label_list)
for tid, ko, en, cat in batch:
if max_terms and len(updates) >= max_terms:
break
if tid in used:
continue
best = label_map.get(tid)
if not best:
continue
_assign(tid, ko, en, cat, best)
for url, tid in updates:
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
conn.commit()
print(f"COde: {len(updates)} mappings committed", flush=True)
return len(updates)
# ============================================================
# 17. Stock Image APIs — Pixabay (CC0) + Pexels + Unsplash
# ============================================================
def map_stock(conn, max_terms=50, delay=0.5, category=None,
model=VERIFY_MODEL, verify=True):
"""Pixabay(CC0) → Pexels → Unsplash 순으로 term별 이미지 검색.
키 파일: .smallclaw/pixabay_api_key.txt, pexels_api_key.txt, unsplash_api_key.txt
verify=True: 비전 모델 검증 후 저장. False: 첫 결과 바로 저장(빠름).
"""
def _key(fname):
try:
with open(os.path.join(SMALLCLAW, fname)) as f:
return f.read().strip()
except FileNotFoundError:
return None
PIXABAY = _key("pixabay_api_key.txt")
PEXELS = _key("pexels_api_key.txt")
UNSPLASH = _key("unsplash_api_key.txt")
if not any([PIXABAY, PEXELS, UNSPLASH]):
print("Stock: API 키 없음 — skipping", flush=True)
return 0
UA = "DentalDictBot/2.0 (educational)"
def _get(url, headers={}):
req = urllib.request.Request(url, headers={"User-Agent": UA, **headers})
with urllib.request.urlopen(req, timeout=15) as r:
return json.loads(r.read())
def _pixabay(q, n=5):
if not PIXABAY:
return []
try:
d = _get(f"https://pixabay.com/api/?key={PIXABAY}"
f"&q={urllib.parse.quote(q)}&image_type=photo"
f"&per_page={n}&safesearch=true&category=science")
return [h["webformatURL"] for h in d.get("hits", [])]
except Exception as e:
print(f" Pixabay: {e}", flush=True)
return []
def _pexels(q, n=5):
if not PEXELS:
return []
try:
d = _get(f"https://api.pexels.com/v1/search"
f"?query={urllib.parse.quote(q)}&per_page={n}",
{"Authorization": PEXELS})
return [p["src"]["medium"] for p in d.get("photos", [])]
except Exception as e:
print(f" Pexels: {e}", flush=True)
return []
def _unsplash(q, n=5):
if not UNSPLASH:
return []
try:
d = _get(f"https://api.unsplash.com/search/photos"
f"?query={urllib.parse.quote(q)}&per_page={n}",
{"Authorization": f"Client-ID {UNSPLASH}"})
return [r["urls"]["regular"] for r in d.get("results", [])]
except Exception as e:
print(f" Unsplash: {e}", flush=True)
return []
if category:
terms = conn.execute("""
SELECT id, korean, english, category FROM terms
WHERE (image_url IS NULL OR image_url='') AND category=?
AND english IS NOT NULL AND english != ''
ORDER BY id LIMIT ?
""", (category, max_terms)).fetchall()
else:
terms = conn.execute("""
SELECT id, korean, english, category FROM terms
WHERE (image_url IS NULL OR image_url='')
AND english IS NOT NULL AND english != ''
ORDER BY category, id LIMIT ?
""", (max_terms,)).fetchall()
keys_info = f"Pixabay={'✓' if PIXABAY else '✗'} Pexels={'✓' if PEXELS else '✗'} Unsplash={'✓' if UNSPLASH else '✗'}"
print(f"Stock: {len(terms)} terms"
f"{f' in {category}' if category else ''} ({keys_info}, verify={'on' if verify else 'off'})", flush=True)
updates = []
for i, (tid, ko, en, cat) in enumerate(terms):
time.sleep(delay)
q = f"dental {en}" if "dental" not in en.lower() else en
# Pixabay → Pexels → Unsplash
candidates = (
[("Pixabay", u) for u in _pixabay(q)] +
[("Pexels", u) for u in _pexels(q)] +
[("Unsplash", u) for u in _unsplash(q)]
)
if not candidates:
print(f" [{i+1}/{len(terms)}] {ko} — 결과 없음", flush=True)
continue
print(f" [{i+1}/{len(terms)}] {ko} — {len(candidates)} 후보", flush=True)
assigned = False
for src, img_url in candidates:
if verify:
res = validate_image(img_url, ko, en, cat,
hint=f"Stock photo ({src})", model=model)
if not res["ok"]:
continue
local_url = localize_image(tid, img_url, korean=ko)
final_url = local_url or img_url
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (final_url, tid))
conn.commit()
updates.append((final_url, tid))
print(f" ✓ {src}", flush=True)
assigned = True
break
if not assigned:
print(f" — 적합 이미지 없음", flush=True)
print(f"Stock: {len(updates)} mappings committed", flush=True)
return len(updates)
# ============================================================
# 18. BRAR / AlphaDent — zip-only 데이터셋 (로컬 다운로드 필요)
# ============================================================
def map_brar(conn, verify=True):
"""BRAR-anchored multimodal (Figshare 30155974). 1,104 파노라마 방사선, CC-BY-4.0.
Figshare에 467MB zip만 있어 로컬 다운로드 필요."""
root = os.path.join(IMG_BASE, "brar")
if not os.path.isdir(root) or not any(os.scandir(root)):
print("BRAR: 로컬 다운로드 필요 — Figshare 30155974 (467MB zip)", flush=True)
print(" https://api.figshare.com/v2/articles/30155974/files", flush=True)
print(" dental_images/brar/ 에 풀고 재실행.", flush=True)
return 0
print("BRAR: 로컬 매핑은 미구현 (파노라마 radiograph zip) — 필요시 구현 요청", flush=True)
return 0
def map_alphadent(conn, verify=True):
"""AlphaDent (Zenodo 16582489). 치아 병변 탐지, Apache-2.0.
Zenodo에 4.8GB zip만 있어 로컬 다운로드 필요."""
root = os.path.join(IMG_BASE, "alphadent")
if not os.path.isdir(root) or not any(os.scandir(root)):
print("AlphaDent: 로컬 다운로드 필요 — Zenodo 16582489 (4.8GB zip)", flush=True)
print(" https://zenodo.org/api/records/16582489", flush=True)
print(" dental_images/alphadent/ 에 풀고 재실행.", flush=True)
return 0
print("AlphaDent: 로컬 매핑 미구현 (zip 구조 미확정) — 필요시 구현 요청", flush=True)
return 0
# ============================================================
# 18. OOPID / STS-3D-Tooth — 대용량 niche (수동 다운로드)
# ============================================================
def map_oopid(conn, verify=True):
"""OOPID (oopid.jp). 구강 세포병리 9,593 patch, 17GB, 커스텀 라이선스.
cytology niche — 수동 다운로드 필요."""
print("OOPID: 수동 다운로드 필요 — https://oopid.jp/ (17GB, 커스텀 라이선스)", flush=True)
print(" 구강 세포병리(cytology) niche — 사전 term 매핑 가치 낮음.", flush=True)
return 0
def map_sts3d(conn, verify=True):
"""STS-3D-Tooth (Zenodo 10597292). 3D CBCT 371볼륨, 31GB, CC-BY-4.0.
3D 볼륨 데이터 — 2D term 매핑 부적합."""
print("STS-3D-Tooth: 수동 다운로드 필요 — Zenodo 10597292 (31GB 3D CBCT)", flush=True)
print(" 3D 볼륨 데이터로 2D term 매핑 부적합 — niche.", flush=True)
return 0
# ============================================================
# Main
# ============================================================
# Image sources, in execution order for source="all". See docs/howto.md.
SOURCES = [
"kaggle", "roboflow", "huggingface", "radiopaedia", "itu",
"roboflow_api", "zenodo", "mendeley", "openi", "wikimedia",
"figshare", "pmc",
# 신규 (2026-06): 웹 스크랩/API + 로컬 다운로드 데이터셋
"vcu", "oralsd", "akudental", "code", "alphadent", "brar", "oopid", "sts3d",
# 스톡 이미지 API (Pixabay CC0 / Pexels / Unsplash)
"stock",
]
def run(source="all", verify=True,
max_wikimedia=50, wikimedia_category=None, wikimedia_model=VERIFY_MODEL, delay=5,
max_openi=50, openi_delay=0.5, openi_category=None, openi_model=VERIFY_MODEL,
max_figshare=50, figshare_category=None, figshare_model=VERIFY_MODEL, figshare_delay=1,
max_pmc=50, pmc_category=None, pmc_model=VERIFY_MODEL, pmc_delay=1,
max_vcu=60, vcu_category=None, vcu_model=VERIFY_MODEL, vcu_delay=1,
max_oralsd=50, oralsd_category=None, oralsd_model=VERIFY_MODEL, oralsd_delay=0.5,
max_akudental=20, akudental_category=None, akudental_model=VERIFY_MODEL, akudental_delay=0.5,
max_code=None, code_category=None, code_llm_match=False,
max_stock=50, stock_category=None, stock_model=VERIFY_MODEL, stock_delay=0.5):
"""Map dental dictionary terms to images from one or all sources.
Called by manage.py's `add-images` subcommand. `source` is "all" or one
of SOURCES. When `verify` is True every source validates each candidate
image with the vision model before saving (openi/wikimedia/figshare/pmc always do).
Returns the number of new mappings committed.
"""
conn = connect_db()
if source == "all":
want = SOURCES
elif "," in source:
want = [s.strip() for s in source.split(",") if s.strip() in SOURCES]
else:
want = [source]
total_before = conn.execute(
"SELECT COUNT(*) FROM terms WHERE image_url IS NOT NULL AND image_url != ''"
).fetchone()[0]
print(f"Images before: {total_before} (verify={'on' if verify else 'off'})")
if "kaggle" in want:
print("\n=== Mapping Kaggle datasets ===")
print(f"Kaggle: {map_kaggle_datasets(conn, verify=verify)} mapped")
if "roboflow" in want:
print("\n=== Mapping Roboflow datasets ===")
print(f"Roboflow: {map_roboflow_datasets(conn, verify=verify)} mapped")
if "huggingface" in want:
print("\n=== Mapping Hugging Face datasets ===")
print(f"Hugging Face: {map_huggingface_datasets(conn, verify=verify)} mapped")
if "radiopaedia" in want:
print("\n=== Mapping Radiopaedia URLs ===")
print(f"Radiopaedia: {map_radiopaedia_urls(conn)} mapped")
if "itu" in want:
print("\n=== Mapping ITU/HF datasets ===")
print(f"ITU/HF: {map_itu_datasets(conn)} mapped")
if "roboflow_api" in want:
print("\n=== Mapping Roboflow API (additional projects) ===")
print(f"Roboflow API: {map_roboflow_api_projects(conn, verify=verify)} mapped")
if "zenodo" in want:
print("\n=== Mapping Zenodo records ===")
print(f"Zenodo: {map_zenodo_records(conn, verify=verify)} mapped")
if "mendeley" in want:
print("\n=== Mapping Mendeley Data ===")
print(f"Mendeley: {map_mendeley_dataset(conn, verify=verify)} mapped")
if "openi" in want:
print("\n=== Mapping via NLM Open-i ===")
n = map_openi(conn, max_terms=max_openi, delay=openi_delay,
category=openi_category, model=openi_model)
print(f"Open-i: {n} mapped")
if "wikimedia" in want:
print("\n=== Searching Wikimedia Commons ===")
n = search_wikimedia_commons(conn, max_terms=max_wikimedia,
delay=delay, category=wikimedia_category,
model=wikimedia_model)
print(f"Wikimedia: {n} mapped")
if "figshare" in want:
print("\n=== Figshare 공개 그림 ===")
n = map_figshare(conn, max_terms=max_figshare, delay=figshare_delay,
category=figshare_category, model=figshare_model)
print(f"Figshare: {n} mapped")
if "pmc" in want:
print("\n=== PMC Entrez Figure 추출 ===")
n = map_pmc_entrez(conn, max_terms=max_pmc, delay=pmc_delay,
category=pmc_category, model=pmc_model)
print(f"PMC Entrez: {n} mapped")
if "vcu" in want:
print("\n=== VCU Oral Pathology Review ===")
n = map_vcu(conn, max_terms=max_vcu, delay=vcu_delay,
category=vcu_category, model=vcu_model)
print(f"VCU: {n} mapped")
if "oralsd" in want:
print("\n=== OralSDv1 (GitHub) ===")
n = map_oralsd(conn, max_terms=max_oralsd, delay=oralsd_delay,
category=oralsd_category, model=oralsd_model)
print(f"OralSDv1: {n} mapped")
if "akudental" in want:
print("\n=== AKUDENTAL 파노라마 ===")
n = map_akudental(conn, max_terms=max_akudental, delay=akudental_delay,
category=akudental_category, model=akudental_model)
print(f"AKUDENTAL: {n} mapped")
if "code" in want:
print("\n=== COde (HuggingFace 로컬) ===")
print(f"COde: {map_code(conn, verify=verify, max_terms=max_code, category=code_category, llm_match=code_llm_match)} mapped")
if "alphadent" in want:
print("\n=== AlphaDent (로컬 다운로드) ===")
print(f"AlphaDent: {map_alphadent(conn, verify=verify)} mapped")
if "brar" in want:
print("\n=== BRAR 파노라마 (로컬 다운로드) ===")
print(f"BRAR: {map_brar(conn, verify=verify)} mapped")
if "oopid" in want:
print("\n=== OOPID 세포병리 ===")
print(f"OOPID: {map_oopid(conn, verify=verify)} mapped")
if "sts3d" in want:
print("\n=== STS-3D-Tooth CBCT ===")
print(f"STS-3D: {map_sts3d(conn, verify=verify)} mapped")
if "stock" in want:
print("\n=== Stock Images (Pixabay / Pexels / Unsplash) ===")
n = map_stock(conn, max_terms=max_stock, delay=stock_delay,
category=stock_category, model=stock_model, verify=verify)
print(f"Stock: {n} mapped")
total_after = conn.execute(
"SELECT COUNT(*) FROM terms WHERE image_url IS NOT NULL AND image_url != ''"
).fetchone()[0]
print(f"\n=== Summary ===")
print(f"Images before: {total_before}")
print(f"Images after: {total_after}")
print(f"New mappings: {total_after - total_before}")
conn.close()
return total_after - total_before