- 스킬앱 HTML 내장화: accountant/lawyer/investor/weather/music/mind/dental-agent 각 앱이 SKILL_CONTENT를 HTML에 직접 내장 → .smallclaw/skills/ SKILL.md 파일 제거 - 새 게이트웨이 라우트: routes-dental, routes-music, routes-mcp, routes-voice - skillContext 서버 지원: /api/chat에서 skill 컨텍스트를 callerCtx로 주입 - 세션 삭제 API: DELETE /api/admin/sessions/:username/:id - Telegram defaultUsername: 어드민 계정 자동 추론 - Ollama /api/ollama/models에 vision 지원 여부 플래그 추가 - DB 워크플로 개선: workflow_images.py 대규모 리팩터링 - 아두이노 에뮬레이터 추가 개선 (arduino-emulator.html +400줄) - 우즈베크어 학습 콘텐츠 및 MIDI 파일 추가 Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2265 lines
97 KiB
Python
2265 lines
97 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Dental Dictionary Image Mapping Workflow
|
|
=========================================
|
|
이 스크립트는 치과 용어 사전(dental_dict.db)의 용어에 이미지 URL을 매핑합니다.
|
|
|
|
소스:
|
|
1. Kaggle 데이터셋 (로컬 파일)
|
|
- javedrashid/mouth-and-oral-diseases-mod
|
|
7클래스(CaS/CoS/Gum/MC/OC/OLP/OT), 516장, Training+Testing+Validation
|
|
로컬 경로: .smallclaw/databases/dental_images/kaggle/
|
|
Gemini 검증 스크립트: /tmp/kaggle_image_test{1,2,3}.py
|
|
결과: /tmp/kaggle_test_results{,2,3}.json
|
|
주의: CoS(구순포진) 이미지는 모두 입술 외부 병변(Herpes labialis) →
|
|
DB에 구순포진(id=5202), 재발성구순포진(id=5203), 구순단순포진(id=5204) 추가로 매핑 가능
|
|
2. Roboflow Universe (CC BY 4.0)
|
|
3. Hugging Face 데이터셋 (DENTEX, Caries, Oral Cancer, Implant, Gingivitis, X-ray)
|
|
4. Wikimedia Commons (CC BY-SA)
|
|
5. Radiopaedia (CC BY-NC-SA)
|
|
6. ITU Dental Datasets 카탈로그 참조
|
|
8. Zenodo 공개 레코드
|
|
9. Mendeley Data (소아치과)
|
|
|
|
Kaggle 추가 다운로드 후보:
|
|
- salmansajid05/oral-diseases: 246MB, Calculus/Gingivitis/Caries/Hypodontia/MouthUlcer/Discolor
|
|
→ 다운로드 완료: .smallclaw/databases/dental_images/salmansajid/
|
|
→ 검증 스크립트: /tmp/kaggle_image_test4.py, 결과: /tmp/kaggle_test_results4.json
|
|
- jiahongqian/cephalometric-landmarks: 117MB, 측면 두개계측 X선 400장 (랜드마크 CSV)
|
|
→ 다운로드 완료: .smallclaw/databases/dental_images/cephalometric_landmarks/cepha400/cepha400/
|
|
→ 교정 카테고리 upgrade용 (두개계측분석, 악교정수술, 성장평가 등)
|
|
- kambingbersayaphitam/cephalometric-profile-dataset: 117MB, 측면 안모 사진 (Concave/Convex/Plane)
|
|
→ 다운로드 완료: .smallclaw/databases/dental_images/cephalometric_profile/Cephalometric Profile Dataset/
|
|
→ 교정 골격 분류 upgrade용 (앵글류1/2/3급, 상하악전돌증 등)
|
|
→ 검증 스크립트: /tmp/kaggle_image_test5.py, 결과: /tmp/kaggle_test_results5.json
|
|
|
|
교정 카테고리 이미지 현황 (2026-05-13):
|
|
- 100% 배정 완료 (394/394), 단 312개는 generic HuggingFace 파노라마 placeholder
|
|
- Round 5로 33개 두개계측/안모 이미지로 교체 완료
|
|
- 나머지 312개 (브라켓/와이어/장치 전용 term)는 임상 사진 필요 → PMC 추출 또는 전용 데이터셋
|
|
|
|
사용법:
|
|
python3 dental_image_workflow.py [--source all|kaggle|roboflow|huggingface|wikimedia|radiopaedia|itu|zenodo|mendeley|openi]
|
|
|
|
주의:
|
|
- Wikimedia Commons 검색은 API 속도제한(429)이 있으므로 요청 간 5초 이상 대기
|
|
- DB에 먼저 모두 모은 후 한 번에 쓰기 (DB 락 방지)
|
|
- API 키: .smallclaw/roboflow_api_key.txt, .smallclaw/huggingface_api_key.txt
|
|
"""
|
|
|
|
import sqlite3
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import time
|
|
import urllib.request
|
|
import urllib.parse
|
|
|
|
from config import DB_PATH, IMG_BASE, SMALLCLAW, OLLAMA_URL, VERIFY_MODEL, connect_db
|
|
from workflow_verify import verify_updates, validate_image
|
|
|
|
# ============================================================
|
|
# 공통 상수 & 헬퍼 함수
|
|
# ============================================================
|
|
REMOTE_DIR = os.path.join(IMG_BASE, "remote")
|
|
URL_PREFIX = "/api/files/dental_images"
|
|
URL_PREFIX_REMOTE = "/api/files/dental_images/remote"
|
|
|
|
_IMAGE_MAGIC = (
|
|
b"\x89PNG\r\n\x1a\n", # PNG
|
|
b"\xff\xd8\xff", # JPEG
|
|
b"GIF87a", b"GIF89a", # GIF
|
|
b"RIFF", # WEBP
|
|
b"II\x2a\x00", # TIFF little-endian
|
|
b"MM\x00\x2a", # TIFF big-endian
|
|
)
|
|
|
|
|
|
def find_unmapped_term(conn, korean):
|
|
"""Return term id if korean has no image yet, else None."""
|
|
r = conn.execute(
|
|
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
|
|
(korean,),
|
|
).fetchone()
|
|
return r[0] if r else None
|
|
|
|
|
|
def localize_image(term_id, url, korean=None, log_prefix=" "):
|
|
"""Download one remote image URL to REMOTE_DIR/<term_id>.<ext>.
|
|
|
|
Returns the new local URL on success, None on failure.
|
|
Does NOT write to DB — caller is responsible for UPDATE + commit.
|
|
"""
|
|
if not url or not url.startswith("http"):
|
|
return url
|
|
|
|
path = urllib.parse.urlparse(url).path
|
|
ext = os.path.splitext(path)[1].lower()
|
|
if ext not in (".jpg", ".jpeg", ".png", ".gif", ".webp"):
|
|
ext = ".jpg"
|
|
|
|
os.makedirs(REMOTE_DIR, exist_ok=True)
|
|
local_name = f"{term_id}{ext}"
|
|
local_path = os.path.join(REMOTE_DIR, local_name)
|
|
new_url = f"{URL_PREFIX_REMOTE}/{local_name}"
|
|
label = korean or f"term {term_id}"
|
|
|
|
if os.path.exists(local_path):
|
|
print(f"{log_prefix}↓ {label} — already local", flush=True)
|
|
return new_url
|
|
|
|
for attempt in range(5):
|
|
try:
|
|
req = urllib.request.Request(
|
|
url, headers={"User-Agent": "DentalDictBot/2.0 (educational)"}
|
|
)
|
|
with urllib.request.urlopen(req, timeout=30) as r:
|
|
ctype = (r.headers.get("Content-Type") or "").lower()
|
|
data = r.read()
|
|
if not ctype.startswith("image/"):
|
|
print(f"{log_prefix}↓ {label} — not an image ({ctype or 'no type'})", flush=True)
|
|
return None
|
|
if not any(data.startswith(m) for m in _IMAGE_MAGIC):
|
|
print(f"{log_prefix}↓ {label} — bad magic bytes", flush=True)
|
|
return None
|
|
if len(data) < 1000:
|
|
print(f"{log_prefix}↓ {label} — too small ({len(data)}B)", flush=True)
|
|
return None
|
|
with open(local_path, "wb") as f:
|
|
f.write(data)
|
|
print(f"{log_prefix}↓ {label} — {len(data)//1024}KB ✓", flush=True)
|
|
return new_url
|
|
except Exception as e:
|
|
is_429 = "429" in str(e)
|
|
if attempt < 4:
|
|
wait = min(30 * (attempt + 1), 120) if is_429 else 5
|
|
print(f"{log_prefix}↓ {label} — retry {attempt+1}/5 after {wait}s: {e}", flush=True)
|
|
time.sleep(wait)
|
|
else:
|
|
print(f"{log_prefix}↓ {label} — FAILED: {e} (URL kept)", flush=True)
|
|
return None
|
|
|
|
|
|
def apply_updates(conn, updates, source_name, verify=True):
|
|
"""공통 패턴: 검증(선택) → bulk UPDATE → http URL 로컬화 → 1회 commit.
|
|
|
|
updates = [(image_url, term_id), ...]
|
|
"""
|
|
if verify:
|
|
updates = verify_updates(conn, updates, source=source_name)
|
|
for url, term_id in updates:
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, term_id))
|
|
# 외부 http URL 자동 로컬화 (zenodo, mendeley, roboflow_api 등)
|
|
localized = 0
|
|
for url, term_id in list(updates):
|
|
if url.startswith("http"):
|
|
new_url = localize_image(term_id, url)
|
|
if new_url:
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (new_url, term_id))
|
|
localized += 1
|
|
conn.commit()
|
|
if localized:
|
|
print(f" {source_name}: {localized} URLs localized to remote/", flush=True)
|
|
return len(updates)
|
|
|
|
|
|
def bulk_localize(limit=None, category=None, dry_run=False):
|
|
"""DB의 http:// image_url을 모두 로컬 파일로 다운로드.
|
|
|
|
manage.py localize 및 구 download_images.py CLI에서 호출.
|
|
"""
|
|
os.makedirs(REMOTE_DIR, exist_ok=True)
|
|
conn = connect_db()
|
|
|
|
query = """
|
|
SELECT id, korean, image_url FROM terms
|
|
WHERE image_url IS NOT NULL AND image_url != '' AND image_url LIKE 'http%'
|
|
"""
|
|
params = []
|
|
if category:
|
|
query += " AND category=?"
|
|
params.append(category)
|
|
rows = conn.execute(query, params).fetchall()
|
|
if limit:
|
|
rows = rows[:limit]
|
|
|
|
print(f"Remote images to download: {len(rows)}", flush=True)
|
|
downloaded = skipped = failed = 0
|
|
|
|
for i, (term_id, korean, url) in enumerate(rows, 1):
|
|
if dry_run:
|
|
print(f" [{i}/{len(rows)}] {korean} — would download {url}", flush=True)
|
|
downloaded += 1
|
|
continue
|
|
prefix = f" [{i}/{len(rows)}] "
|
|
already = os.path.exists(os.path.join(
|
|
REMOTE_DIR,
|
|
f"{term_id}{os.path.splitext(urllib.parse.urlparse(url).path)[1].lower() or '.jpg'}"
|
|
))
|
|
new_url = localize_image(term_id, url, korean=korean, log_prefix=prefix)
|
|
if new_url is None:
|
|
failed += 1
|
|
elif already:
|
|
skipped += 1
|
|
else:
|
|
downloaded += 1
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (new_url, term_id))
|
|
time.sleep(2 if "wikimedia" in url else 0.5)
|
|
|
|
if not dry_run:
|
|
conn.commit()
|
|
conn.close()
|
|
print(f"\nDone: {downloaded} downloaded, {skipped} already local, {failed} failed", flush=True)
|
|
|
|
# ============================================================
|
|
# 1. Kaggle 데이터셋 매핑
|
|
# ============================================================
|
|
def map_kaggle_datasets(conn, verify=True):
|
|
"""Map images from locally downloaded Kaggle datasets under IMG_BASE.
|
|
|
|
Present datasets (verified 2026-05-14):
|
|
- kaggle/ MOD (Mouth & Oral Diseases): {Training,Testing,Validation}/{CaS,CoS,Gum,MC,OC,OLP,OT}
|
|
- salmansajid/ oral-diseases: Calculus / Gingivitis / Data caries / Mouth Ulcer /
|
|
Tooth Discoloration / hypodontia
|
|
- cephalometric_landmarks/cepha400/cepha400/ 400 lateral cephalometric X-rays
|
|
- cephalometric_profile/Cephalometric Profile Dataset/{Concave,Convex,Plane}
|
|
|
|
image_url is stored as the gateway-served path: /api/files/dental_images/...
|
|
Consecutive images are handed to consecutive still-unmapped terms.
|
|
"""
|
|
updates = []
|
|
|
|
def list_images(rel_dir):
|
|
d = os.path.join(IMG_BASE, rel_dir)
|
|
if not os.path.isdir(d):
|
|
return []
|
|
return [
|
|
f"{rel_dir}/{f}"
|
|
for f in sorted(os.listdir(d))
|
|
if f.lower().endswith((".jpg", ".jpeg", ".png"))
|
|
]
|
|
|
|
def assign(rel_paths, term_list):
|
|
i = 0
|
|
for term_kr in term_list:
|
|
if i >= len(rel_paths):
|
|
break
|
|
result = conn.execute(
|
|
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
|
|
(term_kr,),
|
|
).fetchone()
|
|
if result:
|
|
updates.append((f"{URL_PREFIX}/{rel_paths[i]}", result[0]))
|
|
i += 1
|
|
|
|
# --- 1. MOD dataset (kaggle/) — 7 disease classes across 3 splits ---
|
|
mod_terms = {
|
|
"CaS": ["아프타성 구내염", "구강아프타", "구강아프타성궤양", "아프타성 궤양"],
|
|
"CoS": ["구순포진", "재발성구순포진", "구순단순포진", "구강포진"],
|
|
"Gum": ["치은염", "치주염"],
|
|
"MC": ["구강암"],
|
|
"OC": ["편평세포암종", "편평상피암", "구강편평세포암"],
|
|
"OLP": ["편평태선"],
|
|
"OT": ["구강 칸디다증", "구강캔디다증", "위막형칸디다증"],
|
|
}
|
|
for cls, term_list in mod_terms.items():
|
|
imgs = []
|
|
for split in ("Training", "Testing", "Validation"):
|
|
imgs.extend(list_images(f"kaggle/{split}/{cls}"))
|
|
assign(imgs, term_list)
|
|
|
|
# --- 2. salmansajid/ oral-diseases (original, non-augmented leaves) ---
|
|
salmansajid_map = [
|
|
("salmansajid/Calculus/Calculus", ["치석"]),
|
|
("salmansajid/Gingivitis/Gingivitis", ["치은염", "치주염"]),
|
|
("salmansajid/Data caries/Data caries/caries orignal data set/done",
|
|
["치아우식증", "법랑질우식", "상아질우식"]),
|
|
("salmansajid/Mouth Ulcer/Mouth Ulcer/ulcer original dataset/ulcer original dataset",
|
|
["구강궤양"]),
|
|
("salmansajid/Tooth Discoloration/Tooth Discoloration /tooth discoloration original dataset/tooth discoloration original dataset",
|
|
["치아 변색", "치아염색", "외인성 변색", "내인성 변색"]),
|
|
("salmansajid/hypodontia/hypodontia", ["선천성무치증"]),
|
|
]
|
|
for rel_dir, term_list in salmansajid_map:
|
|
assign(list_images(rel_dir), term_list)
|
|
|
|
# --- 3. cephalometric_landmarks — 400 lateral cephalometric X-rays ---
|
|
assign(
|
|
list_images("cephalometric_landmarks/cepha400/cepha400"),
|
|
["두부 계측 방사선 사진", "측두두부방사선사진", "두부 방사선 규격 사진",
|
|
"측두두부계측방사선사진", "후전두부계측방사선사진", "측방두개규격사진",
|
|
"두부 계측 추적", "두부 계측 분석", "두부계측분석"],
|
|
)
|
|
|
|
# --- 4. cephalometric_profile — facial profile photos (Concave/Convex/Plane) ---
|
|
profile_imgs = []
|
|
for cls in ("Concave", "Convex", "Plane"):
|
|
profile_imgs.extend(
|
|
list_images(f"cephalometric_profile/Cephalometric Profile Dataset/{cls}")
|
|
)
|
|
assign(profile_imgs, ["측면 분석", "안모 분석", "안모 분석 체계", "안모비율"])
|
|
|
|
return apply_updates(conn, updates, "kaggle", verify=verify)
|
|
|
|
|
|
# ============================================================
|
|
# 2. Roboflow 데이터셋 매핑
|
|
# ============================================================
|
|
def map_roboflow_datasets(conn, verify=True):
|
|
"""Map images from Roboflow datasets"""
|
|
updates = []
|
|
|
|
COCO_LABEL_MAP = {
|
|
"Cavity": ["치아우식증", "법랑질우식", "상아질우식"],
|
|
"Fillings": ["치과충전", "복합레진", "아말감"],
|
|
"Impacted Tooth": ["매복치", "사랑니"],
|
|
"Implant": ["임플란트"],
|
|
"infected-teeth": ["근첩농양", "근첩육아종", "근첩병변"],
|
|
}
|
|
|
|
DENTAL_LABEL_MAP = {
|
|
"Cavity": ["치아우식증"],
|
|
"Fillings": ["치과충전", "복합레진"],
|
|
"Impacted Tooth": ["매복치"],
|
|
"Implant": ["임플란트", "임플란트보철물"],
|
|
}
|
|
|
|
for dataset_name, label_map, prefix in [
|
|
("roboflow_coco", COCO_LABEL_MAP, "COCO"),
|
|
("roboflow_dental_xray", DENTAL_LABEL_MAP, "Dental"),
|
|
]:
|
|
for split in ["train", "valid", "test"]:
|
|
json_path = os.path.join(IMG_BASE, dataset_name, split, "_annotations.coco.json")
|
|
if not os.path.exists(json_path):
|
|
continue
|
|
|
|
with open(json_path) as f:
|
|
data = json.load(f)
|
|
|
|
cats = {c['id']: c['name'] for c in data['categories']}
|
|
|
|
for cat_name, korean_terms in label_map.items():
|
|
cat_id = None
|
|
for cid, cname in cats.items():
|
|
if cname == cat_name:
|
|
cat_id = cid
|
|
break
|
|
if not cat_id:
|
|
continue
|
|
|
|
img_ids = set()
|
|
for ann in data['annotations']:
|
|
if ann['category_id'] == cat_id:
|
|
img_ids.add(ann['image_id'])
|
|
|
|
img_map = {img['id']: img['file_name'] for img in data['images']}
|
|
img_list = [img_map[iid] for iid in img_ids if iid in img_map]
|
|
|
|
for term_kr in korean_terms:
|
|
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
|
|
if result and img_list:
|
|
rel_path = f"/api/files/dental_images/{dataset_name}/{split}/{img_list[0]}"
|
|
full_path = os.path.join(IMG_BASE, dataset_name, split, img_list[0])
|
|
if os.path.exists(full_path):
|
|
updates.append((rel_path, result[0]))
|
|
break
|
|
|
|
return apply_updates(conn, updates, "roboflow", verify=verify)
|
|
|
|
|
|
# ============================================================
|
|
# 3. Hugging Face 데이터셋 매핑
|
|
# ============================================================
|
|
def map_huggingface_datasets(conn, verify=True):
|
|
"""Map images from Hugging Face datasets (DENTEX, Caries, Oral Cancer, Implant, Gingivitis, X-ray)"""
|
|
updates = []
|
|
|
|
# --- DENTEX diagnosis categories ---
|
|
dentex_dir = os.path.join(IMG_BASE, "huggingface_dentex")
|
|
if os.path.exists(dentex_dir):
|
|
dentex_id_map = {
|
|
0: ["치아우식증", "법랑질우식", "상아질우식"],
|
|
1: ["상아질우식", "치수염"],
|
|
2: ["근첩농양", "근첩육아종", "근첩낭종", "근첩병변"],
|
|
3: ["매복치", "사랑니", "함치낭종"],
|
|
}
|
|
dentex_name_map = {
|
|
"caries": dentex_id_map[0],
|
|
"deep caries": dentex_id_map[1],
|
|
"periapical lesion": dentex_id_map[2],
|
|
"impacted": dentex_id_map[3],
|
|
}
|
|
seen_dentex = set()
|
|
for split in ["train", "valid", "test"]:
|
|
json_path = os.path.join(dentex_dir, "DENTEX", f"{split}_triple.json")
|
|
if not os.path.exists(json_path):
|
|
continue
|
|
with open(json_path) as f:
|
|
data = json.load(f)
|
|
|
|
# Build (img_file, [category_keys]) pairs — handles COCO dict or list format
|
|
pairs = []
|
|
if isinstance(data, dict) and "images" in data:
|
|
cats = {c['id']: c['name'] for c in data.get('categories', [])}
|
|
ann_by_img = {}
|
|
for ann in data.get('annotations', []):
|
|
ann_by_img.setdefault(ann['image_id'], []).append(
|
|
cats.get(ann['category_id'], ann['category_id'])
|
|
)
|
|
for img in data['images']:
|
|
pairs.append((img['file_name'], ann_by_img.get(img['id'], [])))
|
|
elif isinstance(data, list):
|
|
for entry in data:
|
|
img_file = entry.get("file_name", entry.get("image", ""))
|
|
keys = entry.get("diagnosis", entry.get("labels", entry.get("categories", [])))
|
|
pairs.append((img_file, keys))
|
|
|
|
print(f" DENTEX {split}: {len(pairs)} entries")
|
|
|
|
for img_file, cat_keys in pairs:
|
|
if not img_file:
|
|
continue
|
|
if not os.path.exists(os.path.join(dentex_dir, "DENTEX", split, img_file)):
|
|
continue
|
|
for key in cat_keys:
|
|
terms = dentex_name_map.get(key) if isinstance(key, str) else dentex_id_map.get(key)
|
|
if not terms:
|
|
continue
|
|
for term_kr in terms:
|
|
if term_kr in seen_dentex:
|
|
continue
|
|
result = conn.execute(
|
|
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
|
|
(term_kr,)
|
|
).fetchone()
|
|
if result:
|
|
rel_path = f"/api/files/dental_images/huggingface_dentex/DENTEX/{split}/{img_file}"
|
|
updates.append((rel_path, result[0]))
|
|
seen_dentex.add(term_kr)
|
|
break
|
|
|
|
# --- Caries2 dataset (reza362/dental-xray-caries) ---
|
|
caries2_dir = os.path.join(IMG_BASE, "huggingface_caries2")
|
|
if os.path.exists(caries2_dir):
|
|
for split in ["train", "test", "valid", "val"]:
|
|
split_dir = os.path.join(caries2_dir, split)
|
|
if not os.path.exists(split_dir):
|
|
continue
|
|
images = sorted([f for f in os.listdir(split_dir) if f.endswith(('.jpg', '.png', '.jpeg'))])
|
|
if images:
|
|
for term_kr in ["치아우식증", "법랑질우식"]:
|
|
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
|
|
if result:
|
|
rel_path = f"/api/files/dental_images/huggingface_caries2/{split}/{images[0]}"
|
|
updates.append((rel_path, result[0]))
|
|
break
|
|
|
|
# --- Implant dataset (Mesh2001/Dental_implant) ---
|
|
implant_dir = os.path.join(IMG_BASE, "huggingface_implant", "extracted")
|
|
if os.path.exists(implant_dir):
|
|
# Classes: 0=Bego, 1=Bicon, 2=ITI
|
|
implant_term_map = {
|
|
0: ["임플란트"],
|
|
1: ["임플란트"],
|
|
2: ["임플란트", "임플란트보철물"],
|
|
}
|
|
for split in ["train", "valid", "test"]:
|
|
split_img_dir = os.path.join(implant_dir, split, "images")
|
|
if not os.path.exists(split_img_dir):
|
|
continue
|
|
images = sorted([f for f in os.listdir(split_img_dir) if f.endswith(('.jpg', '.png', '.jpeg'))])
|
|
# Pick first image per class prefix
|
|
seen_prefixes = set()
|
|
for img in images:
|
|
prefix = img.split("-")[0].split("_")[0]
|
|
if prefix in seen_prefixes:
|
|
continue
|
|
seen_prefixes.add(prefix)
|
|
for term_kr in ["임플란트", "임플란트보철물"]:
|
|
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
|
|
if result:
|
|
rel_path = f"/api/files/dental_images/huggingface_implant/extracted/{split}/images/{img}"
|
|
updates.append((rel_path, result[0]))
|
|
break
|
|
|
|
# --- Gingivitis dataset (ekacare/IntraOral_Gingivitis_Image_Captioning) ---
|
|
gingivitis_dir = os.path.join(IMG_BASE, "huggingface_gingivitis", "extracted")
|
|
if os.path.exists(gingivitis_dir):
|
|
gingivitis_terms = {
|
|
"mild": ["치은염"],
|
|
"moderate": ["치은염", "치주염"],
|
|
"severe": ["치주염", "치주낭"],
|
|
"gingivitis": ["치은염"],
|
|
}
|
|
images = sorted([f for f in os.listdir(gingivitis_dir) if f.endswith('.jpg')])
|
|
for level, terms in gingivitis_terms.items():
|
|
level_images = [img for img in images if f"_{level}" in img]
|
|
if not level_images:
|
|
continue
|
|
for term_kr in terms:
|
|
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
|
|
if result:
|
|
rel_path = f"/api/files/dental_images/huggingface_gingivitis/extracted/{level_images[0]}"
|
|
updates.append((rel_path, result[0]))
|
|
|
|
# --- Oral Cancer dataset (Docty/Oral-Cancer) ---
|
|
oral_cancer_dir = os.path.join(IMG_BASE, "huggingface_oral_cancer", "extracted")
|
|
if os.path.exists(oral_cancer_dir):
|
|
cancer_terms = {
|
|
"cancer": ["구강암", "편평세포암", "구강암종"],
|
|
"normal": ["정상구강점막"],
|
|
}
|
|
for label, terms in cancer_terms.items():
|
|
label_images = sorted([f for f in os.listdir(oral_cancer_dir) if f"_{label}" in f])
|
|
if not label_images:
|
|
continue
|
|
for term_kr in terms:
|
|
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
|
|
if result:
|
|
rel_path = f"/api/files/dental_images/huggingface_oral_cancer/extracted/{label_images[0]}"
|
|
updates.append((rel_path, result[0]))
|
|
|
|
# --- Hugging Face X-ray segmentation dataset ---
|
|
xray_dir = os.path.join(IMG_BASE, "huggingface_xray")
|
|
if os.path.exists(xray_dir):
|
|
xray_terms = ["파노라마방사선사진", "치아방사선사진", "방사선사진"]
|
|
images = []
|
|
for root, dirs, files in os.walk(xray_dir):
|
|
for f in files:
|
|
if f.endswith(('.jpg', '.png', '.jpeg')):
|
|
images.append(os.path.relpath(os.path.join(root, f), xray_dir))
|
|
images.sort()
|
|
if images:
|
|
for term_kr in xray_terms:
|
|
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
|
|
if result:
|
|
rel_path = f"/api/files/dental_images/huggingface_xray/{images[0]}"
|
|
updates.append((rel_path, result[0]))
|
|
|
|
# --- Hugging Face caries dataset (original) ---
|
|
caries_dir = os.path.join(IMG_BASE, "huggingface_caries")
|
|
if os.path.exists(caries_dir):
|
|
images = []
|
|
for root, dirs, files in os.walk(caries_dir):
|
|
for f in files:
|
|
if f.endswith(('.jpg', '.png', '.jpeg')):
|
|
images.append(os.path.relpath(os.path.join(root, f), caries_dir))
|
|
images.sort()
|
|
if images:
|
|
for term_kr in ["치아우식증"]:
|
|
result = conn.execute("SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')", (term_kr,)).fetchone()
|
|
if result:
|
|
rel_path = f"/api/files/dental_images/huggingface_caries/{images[0]}"
|
|
updates.append((rel_path, result[0]))
|
|
|
|
return apply_updates(conn, updates, "huggingface", verify=verify)
|
|
|
|
|
|
# ============================================================
|
|
# 4. Wikimedia Commons 검색 (5초 딜레이 필수)
|
|
# ============================================================
|
|
def search_wikimedia_commons(conn, max_terms=50, delay=5, category=None,
|
|
ollama_url="http://127.0.0.1:11434/api/chat",
|
|
model=None):
|
|
"""Search Wikimedia Commons for dental term images with vision model validation"""
|
|
if model is None:
|
|
model = VERIFY_MODEL
|
|
|
|
def search_commons(query, limit=2):
|
|
url = (f"https://commons.wikimedia.org/w/api.php?action=query"
|
|
f"&list=search&srnamespace=6"
|
|
f"&srsearch={urllib.parse.quote(query)}&format=json&srlimit={limit}")
|
|
try:
|
|
req = urllib.request.Request(url, headers={"User-Agent": "DentalDictBot/2.0 (educational)"})
|
|
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
data = json.loads(resp.read())
|
|
return data.get("query", {}).get("search", [])
|
|
except urllib.error.HTTPError as e:
|
|
if e.code == 429:
|
|
time.sleep(30)
|
|
return []
|
|
except Exception:
|
|
return []
|
|
|
|
def get_commons_url(title):
|
|
# SVG는 비전 모델이 직접 처리 불가 → iiurlwidth로 PNG 썸네일 URL 함께 요청
|
|
is_svg = title.lower().endswith(".svg")
|
|
extra = "&iiurlwidth=600" if is_svg else ""
|
|
url = (f"https://commons.wikimedia.org/w/api.php?action=query"
|
|
f"&titles={urllib.parse.quote(title)}&prop=imageinfo&iiprop=url&format=json{extra}")
|
|
try:
|
|
req = urllib.request.Request(url, headers={"User-Agent": "DentalDictBot/2.0 (educational)"})
|
|
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
data = json.loads(resp.read())
|
|
pages = data.get("query", {}).get("pages", {})
|
|
for pid, page in pages.items():
|
|
if "imageinfo" in page:
|
|
ii = page["imageinfo"][0]
|
|
return ii.get("thumburl") or ii.get("url", "")
|
|
except Exception:
|
|
pass
|
|
return ""
|
|
|
|
# 수집 후 일괄 쓰기 (DB 락 방지)
|
|
updates = []
|
|
if category:
|
|
terms = conn.execute("""
|
|
SELECT id, korean, english, category
|
|
FROM terms WHERE (image_url IS NULL OR image_url = '') AND category=?
|
|
ORDER BY id LIMIT ?
|
|
""", (category, max_terms)).fetchall()
|
|
total_unmapped = conn.execute(
|
|
"SELECT COUNT(*) FROM terms WHERE (image_url IS NULL OR image_url='') AND category=?",
|
|
(category,)
|
|
).fetchone()[0]
|
|
else:
|
|
terms = conn.execute("""
|
|
SELECT id, korean, english, category
|
|
FROM terms WHERE (image_url IS NULL OR image_url = '')
|
|
ORDER BY category, id LIMIT ?
|
|
""", (max_terms,)).fetchall()
|
|
total_unmapped = conn.execute("SELECT COUNT(*) FROM terms WHERE image_url IS NULL OR image_url=''").fetchone()[0]
|
|
if total_unmapped > max_terms:
|
|
print(f" {total_unmapped} unmapped terms found — capping at {max_terms} (use --max-wikimedia to raise)", flush=True)
|
|
est_min = round(len(terms) * (delay + 3 + 1.5) / 60, 1)
|
|
print(f"Searching Wikimedia Commons for {len(terms)} terms (delay={delay}s, +{model} 검증, ~{est_min} min)...", flush=True)
|
|
|
|
for i, (term_id, korean, english, category) in enumerate(terms):
|
|
search_query = english if english else korean
|
|
|
|
time.sleep(delay)
|
|
results = search_commons(search_query)
|
|
|
|
if not results:
|
|
print(f" [{i+1}/{len(terms)}] {korean} no results", flush=True)
|
|
continue
|
|
|
|
best = None
|
|
for r in results:
|
|
title = r["title"]
|
|
if any(title.lower().endswith(ext) for ext in [".jpg", ".jpeg", ".png", ".svg"]):
|
|
best = title
|
|
break
|
|
if not best:
|
|
best = results[0]["title"]
|
|
|
|
time.sleep(3)
|
|
desc_url = get_commons_url(best)
|
|
if not desc_url:
|
|
print(f" [{i+1}/{len(terms)}] {korean} no URL", flush=True)
|
|
continue
|
|
|
|
res = validate_image(desc_url, korean, english or korean, category, model=model)
|
|
if res["ok"]:
|
|
updates.append((desc_url, term_id, korean))
|
|
print(f" [{i+1}/{len(terms)}] {korean} ({category}) ✓", flush=True)
|
|
else:
|
|
print(f" [{i+1}/{len(terms)}] {korean} ({category}) ❌ {res['reason'][:50]}", flush=True)
|
|
|
|
# 검증 통과 이미지 → 로컬 다운로드 후 일괄 쓰기
|
|
saved = 0
|
|
for ext_url, term_id, korean in updates:
|
|
local_url = localize_image(term_id, ext_url, korean=korean)
|
|
final_url = local_url or ext_url
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (final_url, term_id))
|
|
if local_url:
|
|
saved += 1
|
|
conn.commit()
|
|
print(f"Wikimedia: {len(updates)} verified, {saved} localized", flush=True)
|
|
return len(updates)
|
|
|
|
|
|
# ============================================================
|
|
# 5. Radiopaedia URL 매핑
|
|
# ============================================================
|
|
def map_radiopaedia_urls(conn):
|
|
"""Map Radiopaedia article URLs to terms (stored in source_url, not image_url)"""
|
|
existing = {row[1] for row in conn.execute("PRAGMA table_info(terms)")}
|
|
if "source_url" not in existing:
|
|
conn.execute("ALTER TABLE terms ADD COLUMN source_url TEXT")
|
|
mappings = {
|
|
# 구강악안면영상의학
|
|
"치아우식증": "https://radiopaedia.org/articles/dental-caries",
|
|
"치주염": "https://radiopaedia.org/articles/periodontitis",
|
|
"근첩농양": "https://radiopaedia.org/articles/dental-abscess",
|
|
"근첩낭종": "https://radiopaedia.org/articles/periapical-cyst",
|
|
"함치낭종": "https://radiopaedia.org/articles/dentigerous-cyst",
|
|
"함치종": "https://radiopaedia.org/articles/odontoma",
|
|
"골수염": "https://radiopaedia.org/articles/osteomyelitis",
|
|
"임플란트": "https://radiopaedia.org/articles/dental-implant",
|
|
"파노라마방사선사진": "https://radiopaedia.org/articles/orthopantomography",
|
|
"근첩병변": "https://radiopaedia.org/articles/apical-periodontitis",
|
|
"치아": "https://radiopaedia.org/articles/teeth",
|
|
"방사선골괴사": "https://radiopaedia.org/articles/mandibular-osteoradionecrosis",
|
|
"매복치": "https://radiopaedia.org/articles/impacted-tooth",
|
|
"치근흡수": "https://radiopaedia.org/articles/root-resorption",
|
|
"치은염": "https://radiopaedia.org/articles/gingivitis",
|
|
"하악골": "https://radiopaedia.org/articles/mandible",
|
|
"치은퇴축": "https://radiopaedia.org/articles/gingival-recession",
|
|
"구강암": "https://radiopaedia.org/articles/oral-cancer",
|
|
"악골골절": "https://radiopaedia.org/articles/mandibular-fracture",
|
|
"측두하악관절": "https://radiopaedia.org/articles/tmj",
|
|
"타액선조영술": "https://radiopaedia.org/articles/sialography",
|
|
"타액선염": "https://radiopaedia.org/articles/sialadenitis",
|
|
"근첩육아종": "https://radiopaedia.org/articles/periapical-granuloma",
|
|
# 구강외과
|
|
"발치": "https://radiopaedia.org/articles/tooth-extraction",
|
|
"사랑니발치": "https://radiopaedia.org/articles/wisdom-tooth-extraction",
|
|
"치조골흡수": "https://radiopaedia.org/articles/alveolar-bone-loss",
|
|
# 보존
|
|
"근관치료": "https://radiopaedia.org/articles/root-canal-treatment",
|
|
"치수염": "https://radiopaedia.org/articles/pulpitis",
|
|
# 치주
|
|
"치주낭": "https://radiopaedia.org/articles/periodontal-pocket",
|
|
"치석": "https://radiopaedia.org/articles/dental-calculus",
|
|
# 교정
|
|
"교정장치": "https://radiopaedia.org/articles/orthodontic-braces",
|
|
# 소아치과
|
|
"유전치": "https://radiopaedia.org/articles/deciduous-teeth",
|
|
# 구강병리
|
|
"편평세포암": "https://radiopaedia.org/articles/squamous-cell-carcinoma-head-and-neck",
|
|
"구강점막": "https://radiopaedia.org/articles/oral-mucosa",
|
|
# 임플란트
|
|
"골유착": "https://radiopaedia.org/articles/osseointegration",
|
|
# 기초치의학
|
|
"법랑질": "https://radiopaedia.org/articles/tooth-enamel",
|
|
"상아질": "https://radiopaedia.org/articles/dentine",
|
|
"치수": "https://radiopaedia.org/articles/dental-pulp",
|
|
"백아질": "https://radiopaedia.org/articles/cementum",
|
|
"치은": "https://radiopaedia.org/articles/gingiva",
|
|
"치주인대": "https://radiopaedia.org/articles/periodontal-ligament",
|
|
# 영상의학 추가
|
|
"CBCT": "https://radiopaedia.org/articles/cone-beam-computed-tomography-dental",
|
|
"컴퓨터단층촬영": "https://radiopaedia.org/articles/computed-tomography-head-technique",
|
|
"자기공명영상": "https://radiopaedia.org/articles/magnetic-resonance-imaging-head-technique",
|
|
"초음파검사": "https://radiopaedia.org/articles/ultrasound-head-and-neck-technique",
|
|
"상악동": "https://radiopaedia.org/articles/maxillary-sinus",
|
|
"하악관": "https://radiopaedia.org/articles/inferior-alveolar-canal",
|
|
}
|
|
|
|
updates = []
|
|
for korean, url in mappings.items():
|
|
result = conn.execute(
|
|
"SELECT id FROM terms WHERE korean=? AND (source_url IS NULL OR source_url='')",
|
|
(korean,)
|
|
).fetchone()
|
|
if result:
|
|
updates.append((url, result[0]))
|
|
|
|
for url, term_id in updates:
|
|
conn.execute("UPDATE terms SET source_url=? WHERE id=?", (url, term_id))
|
|
conn.commit()
|
|
return len(updates)
|
|
|
|
|
|
# ============================================================
|
|
# 6. ITU Dental Datasets 카탈로그 참조
|
|
# ============================================================
|
|
def map_itu_datasets(conn):
|
|
"""Map terms using ITU Dental Datasets catalog references (CC BY 4.0)
|
|
|
|
ITU catalog datasets (not yet downloaded, referenced for future use):
|
|
- Apical Periodontitis Panoramic (Mendeley: kx52tk2ddj/3) - 3,926 images
|
|
- Children's Dental Panoramic (figshare: c.6317013) - 193 images
|
|
- OCDC H&E OSCC (Mendeley: 9bsc36jyrt/1) - 1,020 images
|
|
- NDB-UFES Oral Cancer (Mendeley: bbmmm4wgr8/4) - 237+3,763 patches
|
|
- Panoramic Dental Xray Tunisia (Mendeley: 73n3kz2k4k/2) - 221 images
|
|
|
|
Note: Implant, Gingivitis, Oral Cancer datasets are handled by map_huggingface_datasets().
|
|
"""
|
|
# ITU-referenced datasets not yet downloaded — placeholder for future mapping
|
|
print(" ITU: no additional datasets to map (implant/gingivitis/oral_cancer handled by Hugging Face)")
|
|
return 0
|
|
|
|
|
|
# ============================================================
|
|
# 7. Roboflow REST API — 추가 프로젝트 (hosted URL, 다운로드 불필요)
|
|
# ============================================================
|
|
def map_roboflow_api_projects(conn, verify=True):
|
|
"""Map images via Roboflow API — additional projects not covered by local files"""
|
|
|
|
api_key_path = "/home/kim/homeclaw/.smallclaw/roboflow_api_key.txt"
|
|
if not os.path.exists(api_key_path):
|
|
print(" Roboflow API key not found, skipping")
|
|
return 0
|
|
with open(api_key_path) as f:
|
|
api_key = f.read().strip()
|
|
|
|
PROJECTS = [
|
|
{
|
|
"workspace": "yosafat_chandra05-yahoo-com",
|
|
"project": "dental-implants-2.0",
|
|
"version": 1,
|
|
"label_map": {
|
|
"implant": ["임플란트", "임플란트보철물"],
|
|
"Implant": ["임플란트", "임플란트보철물"],
|
|
"implant_bone": ["임플란트", "골유착"],
|
|
},
|
|
},
|
|
{
|
|
"workspace": "dental-mate",
|
|
"project": "dentalmate",
|
|
"version": 1,
|
|
"label_map": {
|
|
"Cavity": ["치아우식증"],
|
|
"Calculus": ["치석"],
|
|
"Gingivitis": ["치은염"],
|
|
"Mouth_Ulcer": ["구강궤양"],
|
|
"Tooth_Discoloration": ["치아변색"],
|
|
"Ulcers": ["구강궤양"],
|
|
"Hypodontia": ["선천결치증"],
|
|
"Caries": ["치아우식증"],
|
|
},
|
|
},
|
|
]
|
|
|
|
updates = []
|
|
|
|
for proj in PROJECTS:
|
|
ws, project, version = proj["workspace"], proj["project"], proj["version"]
|
|
label_map = proj["label_map"]
|
|
|
|
# Request COCO export — returns {"export": {"link": "...", "size": ...}}
|
|
export_url = (
|
|
f"https://api.roboflow.com/{ws}/{project}/{version}"
|
|
f"/coco?api_key={api_key}"
|
|
)
|
|
try:
|
|
req = urllib.request.Request(export_url, headers={"User-Agent": "DentalDictBot/2.0"})
|
|
with urllib.request.urlopen(req, timeout=20) as resp:
|
|
export_meta = json.loads(resp.read())
|
|
except Exception as e:
|
|
print(f" {project}: export API error - {e}")
|
|
continue
|
|
|
|
export_link = export_meta.get("export", {}).get("link", "")
|
|
if not export_link:
|
|
print(f" {project}: no export link in response")
|
|
continue
|
|
|
|
# Download the COCO annotation JSON from the export link
|
|
try:
|
|
req = urllib.request.Request(export_link, headers={"User-Agent": "DentalDictBot/2.0"})
|
|
with urllib.request.urlopen(req, timeout=60) as resp:
|
|
raw = resp.read()
|
|
# Export may be a zip; try JSON first
|
|
try:
|
|
coco_data = json.loads(raw)
|
|
except Exception:
|
|
import zipfile, io
|
|
with zipfile.ZipFile(io.BytesIO(raw)) as zf:
|
|
ann_name = next(
|
|
(n for n in zf.namelist() if n.endswith(".json")), None
|
|
)
|
|
if not ann_name:
|
|
print(f" {project}: no JSON inside zip")
|
|
continue
|
|
coco_data = json.loads(zf.read(ann_name))
|
|
except Exception as e:
|
|
print(f" {project}: COCO download/parse error - {e}")
|
|
continue
|
|
|
|
cats = {c["id"]: c["name"] for c in coco_data.get("categories", [])}
|
|
imgs = {img["id"]: img for img in coco_data.get("images", [])}
|
|
|
|
used_terms = set()
|
|
for ann in coco_data.get("annotations", []):
|
|
cat_name = cats.get(ann["category_id"], "")
|
|
if cat_name not in label_map:
|
|
continue
|
|
img_info = imgs.get(ann["image_id"], {})
|
|
img_url = img_info.get("path", "") or img_info.get("coco_url", "")
|
|
if not img_url:
|
|
continue
|
|
for term_kr in label_map[cat_name]:
|
|
if term_kr in used_terms:
|
|
continue
|
|
result = conn.execute(
|
|
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
|
|
(term_kr,),
|
|
).fetchone()
|
|
if result:
|
|
updates.append((img_url, result[0]))
|
|
used_terms.add(term_kr)
|
|
break
|
|
|
|
print(f" {project}: {len(used_terms)} terms matched")
|
|
|
|
return apply_updates(conn, updates, "roboflow_api", verify=verify)
|
|
|
|
|
|
# ============================================================
|
|
# 8. Zenodo REST API — 공개 데이터셋 (인증 불필요)
|
|
# ============================================================
|
|
def map_zenodo_records(conn, verify=True):
|
|
"""Map images from Zenodo public records via REST API"""
|
|
|
|
RECORDS = [
|
|
{
|
|
"id": "15487430",
|
|
"description": "14-class panoramic radiograph (CC BY 4.0)",
|
|
"label_map": {
|
|
"implant": ["임플란트"],
|
|
"Implant": ["임플란트"],
|
|
"cavity": ["치아우식증"],
|
|
"caries": ["치아우식증"],
|
|
"Caries": ["치아우식증"],
|
|
"periapical": ["근첩병변", "근첩농양"],
|
|
"Periapical": ["근첩병변"],
|
|
"crown": ["크라운"],
|
|
"Crown": ["크라운"],
|
|
"RCT": ["근관치료"],
|
|
"rct": ["근관치료"],
|
|
"filling": ["치과충전"],
|
|
"Filling": ["치과충전"],
|
|
"impacted": ["매복치"],
|
|
"Impacted": ["매복치"],
|
|
"bridge": ["고정성보철물"],
|
|
"Bridge": ["고정성보철물"],
|
|
"calculus": ["치석"],
|
|
"Calculus": ["치석"],
|
|
"gingivitis": ["치은염"],
|
|
"Gingivitis": ["치은염"],
|
|
},
|
|
},
|
|
]
|
|
|
|
updates = []
|
|
|
|
for record in RECORDS:
|
|
rec_id = record["id"]
|
|
label_map = record["label_map"]
|
|
|
|
try:
|
|
req = urllib.request.Request(
|
|
f"https://zenodo.org/api/records/{rec_id}",
|
|
headers={"User-Agent": "DentalDictBot/2.0"},
|
|
)
|
|
with urllib.request.urlopen(req, timeout=20) as resp:
|
|
meta = json.loads(resp.read())
|
|
except Exception as e:
|
|
print(f" Zenodo {rec_id}: API error - {e}")
|
|
continue
|
|
|
|
files = meta.get("files", [])
|
|
print(f" Zenodo {rec_id}: {len(files)} files ({record['description']})")
|
|
|
|
img_files = [
|
|
f for f in files
|
|
if f.get("key", "").lower().endswith((".jpg", ".jpeg", ".png"))
|
|
]
|
|
ann_files = [
|
|
f for f in files
|
|
if f.get("key", "").lower().endswith(".json")
|
|
]
|
|
|
|
used_terms = set()
|
|
|
|
if img_files:
|
|
# Direct image files — match by filename keyword
|
|
for f in img_files:
|
|
key = f.get("key", "").lower()
|
|
dl_url = f.get("links", {}).get("self", "")
|
|
if not dl_url:
|
|
continue
|
|
for label, korean_terms in label_map.items():
|
|
if label.lower() in key:
|
|
for term_kr in korean_terms:
|
|
if term_kr in used_terms:
|
|
continue
|
|
result = conn.execute(
|
|
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
|
|
(term_kr,),
|
|
).fetchone()
|
|
if result:
|
|
updates.append((dl_url, result[0]))
|
|
used_terms.add(term_kr)
|
|
break
|
|
print(f" Zenodo {rec_id}: {len(used_terms)} terms matched from images")
|
|
|
|
elif ann_files:
|
|
# Try first COCO annotation JSON
|
|
ann_url = ann_files[0].get("links", {}).get("self", "")
|
|
if ann_url:
|
|
try:
|
|
req = urllib.request.Request(ann_url, headers={"User-Agent": "DentalDictBot/2.0"})
|
|
with urllib.request.urlopen(req, timeout=30) as resp:
|
|
ann_data = json.loads(resp.read())
|
|
|
|
if isinstance(ann_data, dict) and "images" in ann_data:
|
|
cats = {c["id"]: c["name"] for c in ann_data.get("categories", [])}
|
|
imgs = {img["id"]: img for img in ann_data.get("images", [])}
|
|
# Map image filenames to Zenodo download URLs
|
|
fname_to_url = {}
|
|
for f in files:
|
|
fkey = f.get("key", "")
|
|
furl = f.get("links", {}).get("self", "")
|
|
if furl:
|
|
fname_to_url[os.path.basename(fkey)] = furl
|
|
|
|
for ann in ann_data.get("annotations", []):
|
|
cat_name = cats.get(ann["category_id"], "")
|
|
if cat_name not in label_map:
|
|
continue
|
|
img_info = imgs.get(ann["image_id"], {})
|
|
fname = os.path.basename(img_info.get("file_name", ""))
|
|
img_url = fname_to_url.get(fname, "")
|
|
if not img_url:
|
|
continue
|
|
for term_kr in label_map[cat_name]:
|
|
if term_kr in used_terms:
|
|
continue
|
|
result = conn.execute(
|
|
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
|
|
(term_kr,),
|
|
).fetchone()
|
|
if result:
|
|
updates.append((img_url, result[0]))
|
|
used_terms.add(term_kr)
|
|
break
|
|
print(f" Zenodo {rec_id}: {len(used_terms)} terms matched from COCO")
|
|
except Exception as e:
|
|
print(f" Zenodo {rec_id}: annotation error - {e}")
|
|
else:
|
|
zip_count = sum(1 for f in files if f.get("key", "").lower().endswith(".zip"))
|
|
print(f" Zenodo {rec_id}: {zip_count} zip files — download locally first")
|
|
|
|
return apply_updates(conn, updates, "zenodo", verify=verify)
|
|
|
|
|
|
# ============================================================
|
|
# 9. Mendeley Data — 소아치과 이미지 (CC BY 4.0)
|
|
# ============================================================
|
|
def map_mendeley_dataset(conn, verify=True):
|
|
"""Map images from Mendeley Data public datasets (no API key required for public records)
|
|
|
|
Dataset 6zsnhrds9t: Teeth/Dental images from children 1-14 years
|
|
9,562 intraoral photographs, 8 standardized clinical views (CC BY 4.0)
|
|
"""
|
|
DATASETS = [
|
|
{
|
|
"id": "6zsnhrds9t",
|
|
"version": 1,
|
|
"default_terms": ["유치", "유전치", "유구치", "혼합치열", "소아치과"],
|
|
},
|
|
]
|
|
|
|
updates = []
|
|
|
|
for ds in DATASETS:
|
|
ds_id, version = ds["id"], ds["version"]
|
|
default_terms = ds["default_terms"]
|
|
|
|
api_url = f"https://data.mendeley.com/api/datasets/{ds_id}/versions/{version}"
|
|
try:
|
|
req = urllib.request.Request(
|
|
api_url,
|
|
headers={"User-Agent": "DentalDictBot/2.0", "Accept": "application/json"},
|
|
)
|
|
with urllib.request.urlopen(req, timeout=20) as resp:
|
|
meta = json.loads(resp.read())
|
|
except Exception as e:
|
|
print(f" Mendeley {ds_id}: API error - {e}")
|
|
continue
|
|
|
|
files = meta.get("files", [])
|
|
print(f" Mendeley {ds_id}: {len(files)} files")
|
|
|
|
img_files = [
|
|
f for f in files
|
|
if f.get("filename", "").lower().endswith((".jpg", ".jpeg", ".png"))
|
|
]
|
|
|
|
if not img_files:
|
|
print(f" Mendeley {ds_id}: no direct image files (likely zip archives)")
|
|
continue
|
|
|
|
used_terms = set()
|
|
for f in img_files[: len(default_terms)]:
|
|
dl_url = (
|
|
f.get("download_url", "")
|
|
or (f.get("content_details") or {}).get("download_url", "")
|
|
)
|
|
if not dl_url:
|
|
continue
|
|
for term_kr in default_terms:
|
|
if term_kr in used_terms:
|
|
continue
|
|
result = conn.execute(
|
|
"SELECT id FROM terms WHERE korean=? AND (image_url IS NULL OR image_url='')",
|
|
(term_kr,),
|
|
).fetchone()
|
|
if result:
|
|
updates.append((dl_url, result[0]))
|
|
used_terms.add(term_kr)
|
|
break
|
|
|
|
print(f" Mendeley {ds_id}: {len(used_terms)} terms matched")
|
|
|
|
return apply_updates(conn, updates, "mendeley", verify=verify)
|
|
|
|
|
|
# ============================================================
|
|
# 10. NLM Open-i — PMC figure-level search + 검증
|
|
# ============================================================
|
|
def map_openi(conn, max_terms=50, delay=0.5, category=None,
|
|
model=VERIFY_MODEL):
|
|
"""Map dental terms to images via NLM Open-i.
|
|
|
|
Open-i indexes figures from PMC at the figure level (richer than
|
|
map_pmc_oa's article-HTML scraping). Each result includes a figure
|
|
caption which we feed to the validator as a hint, improving match
|
|
accuracy especially for procedure / technique terms.
|
|
|
|
- Endpoint: https://openi.nlm.nih.gov/api/search?query=...&m=...&n=...
|
|
- No auth, generous rate (we still throttle with `delay`).
|
|
- License: PMC Open Access subset (CC-BY / similar).
|
|
- Per term we fetch up to 5 candidates, validate each with the
|
|
configured vision model (defaults to VERIFY_MODEL = qwen3.5),
|
|
and save the first that passes. Commits per-term (checkpoint).
|
|
"""
|
|
BASE = "https://openi.nlm.nih.gov"
|
|
|
|
def openi_search(query, n=5, max_retries=4):
|
|
# Open-i is slow and intermittently returns 500 / empty body.
|
|
# Back off progressively: 10s, 30s, 60s.
|
|
backoff = [10, 30, 60]
|
|
url = (f"{BASE}/api/search?"
|
|
f"query={urllib.parse.quote(query)}&m=1&n={n}")
|
|
last_err = None
|
|
for attempt in range(max_retries):
|
|
try:
|
|
req = urllib.request.Request(
|
|
url, headers={"User-Agent": "DentalDictBot/2.0"})
|
|
with urllib.request.urlopen(req, timeout=120) as r:
|
|
body = r.read()
|
|
return json.loads(body).get("list", [])
|
|
except Exception as e:
|
|
last_err = e
|
|
if attempt < max_retries - 1:
|
|
wait = backoff[attempt] if attempt < len(backoff) else 60
|
|
print(f" ! retry {attempt+1}/{max_retries} after {wait}s: {e}", flush=True)
|
|
time.sleep(wait)
|
|
print(f" ! Open-i error after {max_retries} tries: {last_err}",
|
|
flush=True)
|
|
return []
|
|
|
|
# Pick unmapped terms with an English label; category filter or all.
|
|
if category:
|
|
terms = conn.execute("""
|
|
SELECT id, korean, english, category FROM terms
|
|
WHERE (image_url IS NULL OR image_url = '')
|
|
AND english IS NOT NULL AND english != ''
|
|
AND category = ?
|
|
ORDER BY id LIMIT ?
|
|
""", (category, max_terms)).fetchall()
|
|
else:
|
|
terms = conn.execute("""
|
|
SELECT id, korean, english, category FROM terms
|
|
WHERE (image_url IS NULL OR image_url = '')
|
|
AND english IS NOT NULL AND english != ''
|
|
ORDER BY category, id LIMIT ?
|
|
""", (max_terms,)).fetchall()
|
|
|
|
print(f"Open-i: {len(terms)} terms"
|
|
f"{f' in {category}' if category else ''}"
|
|
f" (max={max_terms}, delay={delay}s, model={model})", flush=True)
|
|
est = round(len(terms) * (delay + 1 + 10) / 60, 1) # ~1s API + ~10s verify
|
|
print(f" Estimated time: ~{est} min", flush=True)
|
|
|
|
# 카테고리별 검색어 suffix: 기본은 빈 문자열, 방사선은 영상 특화
|
|
_QUERY_SUFFIX = {
|
|
"구강악안면영상의학": "oral radiograph OR oral X-ray",
|
|
}
|
|
|
|
updates = []
|
|
for i, (tid, ko, en, cat) in enumerate(terms):
|
|
time.sleep(delay)
|
|
suffix = _QUERY_SUFFIX.get(cat, "")
|
|
query = f"{en} {suffix}".strip()
|
|
results = openi_search(query, n=5)
|
|
print(f" [{i+1}/{len(terms)}] {ko} ({cat}) — {len(results)} hits", flush=True)
|
|
|
|
assigned = False
|
|
for r in results:
|
|
img_path = r.get("imgLarge") or ""
|
|
if not img_path:
|
|
continue
|
|
img_url = BASE + img_path
|
|
caption = (r.get("image") or {}).get("caption", "")
|
|
hint = f"논문 그림 캡션 참고: {caption[:200]}" if caption else None
|
|
res = validate_image(img_url, ko, en, cat, hint=hint, model=model)
|
|
if res["ok"]:
|
|
# 검증 즉시 로컬화, 실패 시 외부 URL 유지
|
|
local_url = localize_image(tid, img_url, korean=ko)
|
|
updates.append((local_url or img_url, tid))
|
|
print(f" ✓ 저장 (PMID {r.get('pmid','?')})", flush=True)
|
|
assigned = True
|
|
break
|
|
|
|
if not assigned:
|
|
print(f" — 적합 이미지 없음", flush=True)
|
|
|
|
for url, tid in updates:
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
|
|
conn.commit()
|
|
print(f"Open-i: {len(updates)} mappings committed", flush=True)
|
|
return len(updates)
|
|
|
|
|
|
|
|
# ============================================================
|
|
# 11. Figshare — 공개 학술 그림 (무인증, item_type=1)
|
|
# ============================================================
|
|
def map_figshare(conn, max_terms=50, delay=1, category=None, model=VERIFY_MODEL):
|
|
"""Map dental terms to figures from Figshare (open access, no auth required).
|
|
|
|
Searches figure-type items (item_type=1) on Figshare by English dental term,
|
|
validates each image with the vision model, then localizes.
|
|
Rate limit: 5000 req/hr — 1s delay is safe.
|
|
"""
|
|
|
|
def _search(query, page_size=5):
|
|
url = "https://api.figshare.com/v2/articles/search"
|
|
payload = json.dumps({
|
|
"search_for": query,
|
|
"item_type": 1, # Figure
|
|
"page_size": page_size,
|
|
}).encode()
|
|
req = urllib.request.Request(url, data=payload, headers={
|
|
"Content-Type": "application/json",
|
|
"User-Agent": "DentalDictBot/2.0 (educational)",
|
|
})
|
|
with urllib.request.urlopen(req, timeout=15) as r:
|
|
return json.loads(r.read())
|
|
|
|
def _files(article_id):
|
|
url = f"https://api.figshare.com/v2/articles/{article_id}/files"
|
|
req = urllib.request.Request(url, headers={"User-Agent": "DentalDictBot/2.0"})
|
|
with urllib.request.urlopen(req, timeout=15) as r:
|
|
return json.loads(r.read())
|
|
|
|
if category:
|
|
terms = conn.execute("""
|
|
SELECT id, korean, english, category FROM terms
|
|
WHERE (image_url IS NULL OR image_url='') AND category=?
|
|
AND english IS NOT NULL AND english != ''
|
|
ORDER BY id LIMIT ?
|
|
""", (category, max_terms)).fetchall()
|
|
else:
|
|
terms = conn.execute("""
|
|
SELECT id, korean, english, category FROM terms
|
|
WHERE (image_url IS NULL OR image_url='')
|
|
AND english IS NOT NULL AND english != ''
|
|
ORDER BY category, id LIMIT ?
|
|
""", (max_terms,)).fetchall()
|
|
|
|
print(f"Figshare: {len(terms)} terms"
|
|
f"{f' in {category}' if category else ''} (model={model})", flush=True)
|
|
|
|
# 카테고리별 검색 힌트 — Figshare는 재료·현미경 이미지에 특히 강함
|
|
_SUFFIX = {
|
|
"치과생체재료학": "SEM microscopy",
|
|
"기초치의학": "histology microscopy",
|
|
"디지털치의학": "CAD CAM scanning",
|
|
"구강악안면영상의학": "radiograph X-ray",
|
|
}
|
|
|
|
updates = []
|
|
for i, (tid, ko, en, cat) in enumerate(terms):
|
|
time.sleep(delay)
|
|
suffix = _SUFFIX.get(cat, "dental")
|
|
try:
|
|
articles = _search(f"{en} {suffix}", page_size=5)
|
|
# 결과 없으면 suffix 없이 재시도
|
|
if not articles:
|
|
articles = _search(en, page_size=5)
|
|
except Exception as e:
|
|
print(f" [{i+1}/{len(terms)}] {ko} — search error: {e}", flush=True)
|
|
continue
|
|
|
|
if not articles:
|
|
print(f" [{i+1}/{len(terms)}] {ko} — no results", flush=True)
|
|
continue
|
|
|
|
print(f" [{i+1}/{len(terms)}] {ko} — {len(articles)} figures", flush=True)
|
|
|
|
assigned = False
|
|
for art in articles:
|
|
try:
|
|
files = _files(art["id"])
|
|
except Exception:
|
|
continue
|
|
|
|
img_files = [
|
|
f for f in files
|
|
if f.get("name", "").lower().endswith(
|
|
(".jpg", ".jpeg", ".png", ".gif", ".webp", ".tif", ".tiff"))
|
|
]
|
|
if not img_files:
|
|
continue
|
|
|
|
img_url = img_files[0]["download_url"]
|
|
res = validate_image(img_url, ko, en, cat, model=model)
|
|
if res["ok"]:
|
|
local_url = localize_image(tid, img_url, korean=ko)
|
|
updates.append((local_url or img_url, tid))
|
|
print(f" ✓ {art.get('title','?')[:50]}", flush=True)
|
|
assigned = True
|
|
break
|
|
|
|
if not assigned:
|
|
print(f" — 적합 이미지 없음", flush=True)
|
|
|
|
for url, tid in updates:
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
|
|
conn.commit()
|
|
print(f"Figshare: {len(updates)} mappings committed", flush=True)
|
|
return len(updates)
|
|
|
|
|
|
# ============================================================
|
|
# 12. PMC Entrez — 논문 Figure 직접 추출 (JATS XML)
|
|
# ============================================================
|
|
def map_pmc_entrez(conn, max_terms=50, delay=1, category=None, model=VERIFY_MODEL):
|
|
"""Map dental terms to PMC Open Access article figures via NCBI Entrez.
|
|
|
|
Difference from map_openi() (NLM Open-i index):
|
|
- Searches the full PMC OA database (not just Open-i indexed subset)
|
|
- Fetches JATS XML → parses ALL <fig><graphic> elements per article
|
|
- Can find images for rare/specialized terms openi misses
|
|
Rate: 3 req/s without API key → delay=1 is safe.
|
|
"""
|
|
import xml.etree.ElementTree as ET
|
|
|
|
ESEARCH = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi"
|
|
EFETCH = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/efetch.fcgi"
|
|
XLINK = "{http://www.w3.org/1999/xlink}"
|
|
PMC_BIN = "https://www.ncbi.nlm.nih.gov/pmc/articles/PMC{pmcid}/bin/{href}{suffix}"
|
|
|
|
if category:
|
|
terms = conn.execute("""
|
|
SELECT id, korean, english, category FROM terms
|
|
WHERE (image_url IS NULL OR image_url='') AND category=?
|
|
AND english IS NOT NULL AND english != ''
|
|
ORDER BY id LIMIT ?
|
|
""", (category, max_terms)).fetchall()
|
|
else:
|
|
terms = conn.execute("""
|
|
SELECT id, korean, english, category FROM terms
|
|
WHERE (image_url IS NULL OR image_url='')
|
|
AND english IS NOT NULL AND english != ''
|
|
ORDER BY category, id LIMIT ?
|
|
""", (max_terms,)).fetchall()
|
|
|
|
print(f"PMC Entrez: {len(terms)} terms"
|
|
f"{f' in {category}' if category else ''} (model={model})", flush=True)
|
|
|
|
updates = []
|
|
for i, (tid, ko, en, cat) in enumerate(terms):
|
|
time.sleep(delay)
|
|
|
|
# 1. esearch: PMC articles for this dental term (OA 필터 없이 더 넓게)
|
|
q = urllib.parse.quote(en)
|
|
try:
|
|
req = urllib.request.Request(
|
|
f"{ESEARCH}?db=pmc&term={q}&retmax=5&retmode=json",
|
|
headers={"User-Agent": "DentalDictBot/2.0"},
|
|
)
|
|
with urllib.request.urlopen(req, timeout=15) as r:
|
|
pmcids = json.loads(r.read()).get("esearchresult", {}).get("idlist", [])
|
|
except Exception as e:
|
|
print(f" [{i+1}/{len(terms)}] {ko} — esearch error: {e}", flush=True)
|
|
continue
|
|
|
|
if not pmcids:
|
|
print(f" [{i+1}/{len(terms)}] {ko} — no PMC hits", flush=True)
|
|
continue
|
|
|
|
print(f" [{i+1}/{len(terms)}] {ko} — {len(pmcids)} articles", flush=True)
|
|
|
|
assigned = False
|
|
for pmcid in pmcids[:3]:
|
|
time.sleep(0.5)
|
|
try:
|
|
req = urllib.request.Request(
|
|
f"{EFETCH}?db=pmc&id={pmcid}&retmode=xml",
|
|
headers={"User-Agent": "DentalDictBot/2.0"},
|
|
)
|
|
with urllib.request.urlopen(req, timeout=30) as r:
|
|
root = ET.fromstring(r.read())
|
|
except Exception as e:
|
|
print(f" PMC{pmcid} fetch error: {e}", flush=True)
|
|
continue
|
|
|
|
# 2. JATS XML에서 <fig><graphic xlink:href> 수집
|
|
fig_graphics = []
|
|
for fig in root.iter("fig"):
|
|
caption = " ".join(fig.itertext())[:200].strip()
|
|
for graphic in fig.iter("graphic"):
|
|
href = graphic.get(f"{XLINK}href") or graphic.get("href", "")
|
|
if href:
|
|
fig_graphics.append((href, caption))
|
|
|
|
if not fig_graphics:
|
|
continue
|
|
|
|
for href, caption in fig_graphics[:6]:
|
|
# PMC는 href 그대로, 또는 .jpg/.png 추가 (PMC 변환 관행)
|
|
# 예: href="fig1" → bin/fig1.jpg 또는 bin/fig1.png
|
|
# 예: href="fig1.jpg" → bin/fig1.jpg.png (PMC TIFF→PNG 변환 시)
|
|
img_url = None
|
|
for suffix in ["", ".jpg", ".png", ".gif"]:
|
|
candidate = PMC_BIN.format(pmcid=pmcid, href=href, suffix=suffix)
|
|
try:
|
|
urllib.request.urlopen(
|
|
urllib.request.Request(
|
|
candidate,
|
|
headers={"User-Agent": "DentalDictBot/2.0"},
|
|
),
|
|
timeout=8,
|
|
).close()
|
|
img_url = candidate
|
|
break
|
|
except Exception:
|
|
continue
|
|
if not img_url:
|
|
continue
|
|
|
|
hint = f"논문 그림 캡션: {caption}" if caption else None
|
|
res = validate_image(img_url, ko, en, cat, hint=hint, model=model)
|
|
if res["ok"]:
|
|
local_url = localize_image(tid, img_url, korean=ko)
|
|
updates.append((local_url or img_url, tid))
|
|
print(f" ✓ PMC{pmcid}/{href} ({res['reason'][:40]})", flush=True)
|
|
assigned = True
|
|
break
|
|
|
|
if assigned:
|
|
break
|
|
|
|
if not assigned:
|
|
print(f" — 적합 이미지 없음", flush=True)
|
|
|
|
for url, tid in updates:
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
|
|
conn.commit()
|
|
print(f"PMC Entrez: {len(updates)} mappings committed", flush=True)
|
|
return len(updates)
|
|
|
|
|
|
# ============================================================
|
|
# 공통 헬퍼 — 영문명으로 미매핑 term 찾기
|
|
# ============================================================
|
|
def _find_unmapped_by_en(conn, en, category=None):
|
|
"""영문명 en(소문자)과 매칭되는 미매핑 term (id,korean,english,category) 반환, 없으면 None.
|
|
정확일치 → 포함(양방향) 순."""
|
|
en = (en or "").strip().lower()
|
|
if not en:
|
|
return None
|
|
if category:
|
|
rows = conn.execute(
|
|
"SELECT id,korean,english,category FROM terms "
|
|
"WHERE (image_url IS NULL OR image_url='') AND category=? "
|
|
"AND english IS NOT NULL AND english!=''", (category,)).fetchall()
|
|
else:
|
|
rows = conn.execute(
|
|
"SELECT id,korean,english,category FROM terms "
|
|
"WHERE (image_url IS NULL OR image_url='') "
|
|
"AND english IS NOT NULL AND english!=''").fetchall()
|
|
for tid, ko, e, cat in rows:
|
|
if e and e.strip().lower() == en:
|
|
return (tid, ko, e, cat)
|
|
for tid, ko, e, cat in rows:
|
|
if e and en in e.strip().lower():
|
|
return (tid, ko, e, cat)
|
|
for tid, ko, e, cat in rows:
|
|
if e and e.strip().lower() and e.strip().lower() in en:
|
|
return (tid, ko, e, cat)
|
|
return None
|
|
|
|
|
|
# ============================================================
|
|
# 13. VCU Oral Pathology Review — 교육용 병리 아틀라스 스크랩
|
|
# ============================================================
|
|
def map_vcu(conn, max_terms=60, delay=1, category=None, model=VERIFY_MODEL):
|
|
"""VCU Oral Pathology Review (scholarscompass.vcu.edu/opr).
|
|
~60개 구강 병리 질환, MeSH 인덱스. 인덱스→상세페이지 preview.jpg 추출.
|
|
영문 질환명 → DB english 매칭. 비영리 교육용 (radiopaedia와 동일 취급)."""
|
|
|
|
BASE = "https://scholarscompass.vcu.edu"
|
|
INDEX = BASE + "/opr/index.html"
|
|
|
|
def _fetch(url):
|
|
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
|
|
return urllib.request.urlopen(req, timeout=20).read().decode("utf-8", "ignore")
|
|
|
|
# 인덱스 페이지(1~2)에서 질환 링크 수집
|
|
conditions = {}
|
|
for page in ["", "?page=2"]:
|
|
try:
|
|
html = _fetch(INDEX + page)
|
|
except Exception as e:
|
|
print(f" VCU index{page} 오류: {e}", flush=True)
|
|
continue
|
|
for cid, name in re.findall(r'/opr/(\d+)[^"]*"[^>]*>([^<]+)</a>', html):
|
|
name = name.strip()
|
|
if name and name.lower() != "view slideshow" and cid not in conditions:
|
|
conditions[cid] = name
|
|
print(f"VCU: {len(conditions)} conditions (model={model})", flush=True)
|
|
if not conditions:
|
|
return 0
|
|
|
|
# 영문명 정제: 괄호 제거
|
|
def _clean(name):
|
|
return re.sub(r"\s*\([^)]*\)", "", name).strip().lower()
|
|
|
|
updates = []
|
|
used = set()
|
|
for i, (cid, name) in enumerate(conditions.items()):
|
|
if len(used) >= max_terms:
|
|
break
|
|
time.sleep(delay)
|
|
try:
|
|
html = _fetch(f"{BASE}/opr/{cid}")
|
|
except Exception as e:
|
|
print(f" [{i+1}] {name} — 페이지 오류: {e}", flush=True)
|
|
continue
|
|
m = re.search(rf"{BASE}/opr/(\d+)/preview\.jpg", html)
|
|
if not m:
|
|
print(f" [{i+1}] {name} — 이미지 없음", flush=True)
|
|
continue
|
|
img_url = f"{BASE}/opr/{m.group(1)}/preview.jpg"
|
|
clean = _clean(name)
|
|
term = _find_unmapped_by_en(conn, clean, category=category)
|
|
if not term:
|
|
print(f" [{i+1}] {name} — 매칭 term 없음", flush=True)
|
|
continue
|
|
tid, ko, en, cat = term
|
|
if tid in used:
|
|
continue
|
|
res = validate_image(img_url, ko, en, cat, hint=f"VCU 병리 이미지: {name}", model=model)
|
|
if res["ok"]:
|
|
local_url = localize_image(tid, img_url, korean=ko)
|
|
updates.append((local_url or img_url, tid))
|
|
used.add(tid)
|
|
print(f" [{i+1}] {name} → {ko} ✓", flush=True)
|
|
else:
|
|
print(f" [{i+1}] {name} ✗ {res.get('reason','')[:50]}", flush=True)
|
|
|
|
for url, tid in updates:
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
|
|
conn.commit()
|
|
print(f"VCU: {len(updates)} mappings committed", flush=True)
|
|
return len(updates)
|
|
|
|
|
|
# ============================================================
|
|
# 14. OralSDv1 — GitHub 폴더별 일반 구강질환 (6종, ~1163장)
|
|
# ============================================================
|
|
def map_oralsd(conn, max_terms=50, delay=0.5, category=None, model=VERIFY_MODEL):
|
|
"""OralSDv1 (github.com/enderXM249/OralSDv1). 폴더별 6종 일반질환 이미지.
|
|
Calculus/Caries/Gingivitis/Tooth_Discoloration/Ulcers/Hypodontia.
|
|
GitHub raw URL 직접 사용. 라이선스: 연구용(저자 문의)."""
|
|
API = "https://api.github.com/repos/enderXM249/OralSDv1/git/trees/main?recursive=1"
|
|
RAW = "https://raw.githubusercontent.com/enderXM249/OralSDv1/main/"
|
|
LABEL = {
|
|
"Calculus": ["치석"],
|
|
"Caries": ["치아우식증", "우식증"],
|
|
"Gingivitis": ["치은염"],
|
|
"Tooth_Discoloration": ["치아변색"],
|
|
"Ulcers": ["구강궤양", "아프투스궤양"],
|
|
"Hypodontia": ["무치아증", "선천적결손치"],
|
|
}
|
|
try:
|
|
d = json.loads(urllib.request.urlopen(
|
|
urllib.request.Request(API, headers={"User-Agent": "DentalDictBot/2.0"}),
|
|
timeout=20).read())
|
|
except Exception as e:
|
|
print(f" OralSDv1 tree 오류: {e}", flush=True)
|
|
return 0
|
|
folders = {}
|
|
for t in d.get("tree", []):
|
|
if t.get("type") == "blob":
|
|
p = t["path"]
|
|
if "/" in p and p.lower().endswith((".jpg", ".jpeg", ".png")):
|
|
folders.setdefault(p.split("/")[0], []).append(p)
|
|
print(f"OralSDv1: {len(folders)} condition folders (model={model})", flush=True)
|
|
|
|
updates = []
|
|
used = set()
|
|
for folder, paths in folders.items():
|
|
if len(used) >= max_terms:
|
|
break
|
|
korean_terms = LABEL.get(folder)
|
|
if not korean_terms:
|
|
continue
|
|
for term_kr in korean_terms:
|
|
row = conn.execute(
|
|
"SELECT id,korean,english,category FROM terms "
|
|
"WHERE korean=? AND (image_url IS NULL OR image_url='')",
|
|
(term_kr,)).fetchone()
|
|
if not row or row[0] in used:
|
|
continue
|
|
tid, ko, en, cat = row
|
|
assigned = False
|
|
for p in paths[:8]:
|
|
img_url = RAW + urllib.parse.quote(p)
|
|
time.sleep(delay)
|
|
res = validate_image(img_url, ko, en, cat,
|
|
hint=f"OralSDv1 {folder}", model=model)
|
|
if res["ok"]:
|
|
local_url = localize_image(tid, img_url, korean=ko)
|
|
updates.append((local_url or img_url, tid))
|
|
used.add(tid)
|
|
print(f" ✓ {folder} → {ko}", flush=True)
|
|
assigned = True
|
|
break
|
|
if not assigned:
|
|
print(f" — {folder}/{ko} 적합 이미지 없음", flush=True)
|
|
break
|
|
|
|
for url, tid in updates:
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
|
|
conn.commit()
|
|
print(f"OralSDv1: {len(updates)} mappings committed", flush=True)
|
|
return len(updates)
|
|
|
|
|
|
# ============================================================
|
|
# 15. AKUDENTAL — GitHub 파노라마 방사선 + 인스턴스 매니페스트
|
|
# ============================================================
|
|
def map_akudental(conn, max_terms=20, delay=0.5, category=None, model=VERIFY_MODEL):
|
|
"""AKUDENTAL (github.com/melihoz/akudental). 333 파노라마 방사선, CC-BY-4.0.
|
|
매니페스트(akudental_instances.json)에서 이미지 목록 추출 → raw URL.
|
|
구강악안면영상의학 카테고리 미매핑 term에 파노라마 이미지 매핑(비전 검증)."""
|
|
API = "https://api.github.com/repos/melihoz/akudental/git/trees/main?recursive=1"
|
|
RAW = "https://raw.githubusercontent.com/melihoz/akudental/main/"
|
|
try:
|
|
d = json.loads(urllib.request.urlopen(
|
|
urllib.request.Request(API, headers={"User-Agent": "DentalDictBot/2.0"}),
|
|
timeout=20).read())
|
|
except Exception as e:
|
|
print(f" AKUDENTAL tree 오류: {e}", flush=True)
|
|
return 0
|
|
# 매니페스트에서 이미지 파일명 추출
|
|
manifest = None
|
|
img_paths = []
|
|
for t in d.get("tree", []):
|
|
p = t.get("path", "")
|
|
if p.endswith("akudental_instances.json"):
|
|
manifest = p
|
|
elif p.lower().endswith((".jpg", ".jpeg", ".png")) and "image" in p.lower():
|
|
img_paths.append(p)
|
|
if manifest:
|
|
try:
|
|
raw = urllib.request.urlopen(urllib.request.Request(
|
|
RAW + urllib.parse.quote(manifest),
|
|
headers={"User-Agent": "DentalDictBot/2.0"}), timeout=30).read()
|
|
data = json.loads(raw)
|
|
# 매니페스트 구조 가변 — image 경로 키 탐색
|
|
extra = []
|
|
if isinstance(data, list):
|
|
for item in data[:50]:
|
|
if isinstance(item, dict):
|
|
for v in item.values():
|
|
if isinstance(v, str) and v.lower().endswith((".jpg", ".jpeg", ".png")):
|
|
extra.append(v)
|
|
elif isinstance(data, dict):
|
|
for v in data.values():
|
|
if isinstance(v, str) and v.lower().endswith((".jpg", ".jpeg", ".png")):
|
|
extra.append(v)
|
|
img_paths = extra or img_paths
|
|
except Exception as e:
|
|
print(f" AKUDENTAL manifest 파싱 오류: {e}", flush=True)
|
|
print(f"AKUDENTAL: {len(img_paths)} panoramic images (model={model})", flush=True)
|
|
if not img_paths:
|
|
return 0
|
|
|
|
# 파노라마/방사선 관련 미매핑 term (영상의학 카테고리 우선)
|
|
if category:
|
|
rows = conn.execute(
|
|
"SELECT id,korean,english,category FROM terms "
|
|
"WHERE (image_url IS NULL OR image_url='') AND category=? "
|
|
"AND (english LIKE '%panoramic%' OR english LIKE '%radiograph%' "
|
|
"OR korean LIKE '%파노라마%' OR korean LIKE '%방사선%') "
|
|
"ORDER BY id LIMIT ?", (category, max_terms)).fetchall()
|
|
else:
|
|
rows = conn.execute(
|
|
"SELECT id,korean,english,category FROM terms "
|
|
"WHERE (image_url IS NULL OR image_url='') "
|
|
"AND (english LIKE '%panoramic%' OR english LIKE '%radiograph%' "
|
|
"OR korean LIKE '%파노라마%' OR korean LIKE '%방사선%') "
|
|
"ORDER BY id LIMIT ?", (max_terms,)).fetchall()
|
|
print(f"AKUDENTAL: {len(rows)} 후보 term", flush=True)
|
|
|
|
updates = []
|
|
used = set()
|
|
for tid, ko, en, cat in rows:
|
|
assigned = False
|
|
for p in img_paths[:10]:
|
|
img_url = RAW + urllib.parse.quote(p)
|
|
time.sleep(delay)
|
|
res = validate_image(img_url, ko, en, cat, hint="파노라마 방사선", model=model)
|
|
if res["ok"]:
|
|
local_url = localize_image(tid, img_url, korean=ko)
|
|
updates.append((local_url or img_url, tid))
|
|
used.add(tid)
|
|
print(f" ✓ {ko}", flush=True)
|
|
assigned = True
|
|
break
|
|
if not assigned:
|
|
print(f" — {ko} 적합 없음", flush=True)
|
|
if len(used) >= max_terms:
|
|
break
|
|
|
|
for url, tid in updates:
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
|
|
conn.commit()
|
|
print(f"AKUDENTAL: {len(updates)} mappings committed", flush=True)
|
|
return len(updates)
|
|
|
|
|
|
# ============================================================
|
|
# 16. COde — HuggingFace zirak-ai/COde (964MB zip 로컬 다운로드 필요)
|
|
# ============================================================
|
|
def _llm_label_match_batch(batch, label_list, model=VERIFY_MODEL):
|
|
"""Text-only LLM call: map each (tid, ko, en, cat) in batch to the best
|
|
COde label name, or None if no clinical photo could represent it.
|
|
Returns {tid: label_str or None}."""
|
|
label_block = ", ".join(sorted(label_list))
|
|
lines = "\n".join(
|
|
f"{i+1}. {en} ({ko}) [{cat}]"
|
|
for i, (tid, ko, en, cat) in enumerate(batch)
|
|
)
|
|
prompt = (
|
|
"Pick the best COde label for each dental term, or 'none' if no clinical oral "
|
|
"photo could represent it (e.g. bacteria names, histology, classification systems, "
|
|
"procedure names with no visual signature).\n"
|
|
f"Labels: {label_block}\n\n"
|
|
"Format: <number>. <exact label name or none> — one line per term, no explanations.\n\n"
|
|
f"Terms:\n{lines}"
|
|
)
|
|
payload = {"model": model, "messages": [{"role": "user", "content": prompt}], "stream": False}
|
|
try:
|
|
req = urllib.request.Request(
|
|
OLLAMA_URL, data=json.dumps(payload).encode(),
|
|
headers={"Content-Type": "application/json"},
|
|
)
|
|
with urllib.request.urlopen(req, timeout=180) as r:
|
|
text = json.loads(r.read()).get("message", {}).get("content", "")
|
|
except Exception as e:
|
|
print(f" LLM label match error: {e}", flush=True)
|
|
return {}
|
|
|
|
label_lower = {l.lower(): l for l in label_list}
|
|
result = {}
|
|
for line in text.splitlines():
|
|
m = re.match(r"^(\d+)\.\s*(.+)$", line.strip())
|
|
if not m:
|
|
continue
|
|
idx = int(m.group(1)) - 1
|
|
raw = m.group(2).strip().lower().rstrip(".")
|
|
if idx < 0 or idx >= len(batch):
|
|
continue
|
|
tid = batch[idx][0]
|
|
result[tid] = label_lower.get(raw) # None if "none" or unrecognised
|
|
return result
|
|
|
|
|
|
def map_code(conn, verify=True, max_terms=None, category=None, llm_match=False):
|
|
"""COde (HuggingFace zirak-ai/COde). ~50k 구강내사진 + 8k 방사선 + 진단.
|
|
HF datasets-server 미지원 → COde-Dataset.zip(964MB)을 dental_images/code/에
|
|
다운로드 후 풀어야 함.
|
|
|
|
구조: complete_dataset.csv 매니페스트 (8775행).
|
|
- photographs : Images/Photographs/ 내 파일명 (쉼표 구분 다중)
|
|
- anomalies_en : 영문 카테고리 라벨 (Gingivitis, Dental Caries, Pulpitis,
|
|
Periodontitis, Class II Malocclusion, ... 쉼표 구분)
|
|
라벨 → DB term english 매칭 → term별 최대 3장 후보 비전 검증 → 첫 OK 저장.
|
|
image_url = /api/files/dental_images/code/Images/Photographs/<fn> (게이트웨이 서빙).
|
|
|
|
max_terms: 최대 매핑 term 수 (None=전체). category: 해당 카테고리 term만 매핑."""
|
|
import csv as _csv
|
|
root = os.path.join(IMG_BASE, "code")
|
|
csv_path = os.path.join(root, "complete_dataset.csv")
|
|
photo_dir = os.path.join(root, "Images", "Photographs")
|
|
if not os.path.exists(csv_path):
|
|
print("COde: complete_dataset.csv 없음 — COde-Dataset.zip(964MB) 다운로드 후 "
|
|
"dental_images/code/ 에 해제하세요.", flush=True)
|
|
print(" https://huggingface.co/datasets/zirak-ai/COde/resolve/main/COde-Dataset.zip",
|
|
flush=True)
|
|
return 0
|
|
|
|
# 1. CSV → label(소문자) → [절대경로 이미지 후보]
|
|
# 단일-anomaly 행의 사진이 그 anomaly를 정확히 담을 확률이 높으므로
|
|
# 단일-anomaly 후보를 먼저 배치, 부족 시 다중-anomaly 행 후보로 보충.
|
|
single = {} # label -> [imgp] (단일-anomaly 행)
|
|
multi = {} # label -> [imgp] (다중-anomaly 행)
|
|
n_rows = 0
|
|
with open(csv_path, encoding="utf-8-sig") as f:
|
|
for row in _csv.DictReader(f):
|
|
n_rows += 1
|
|
photos = (row.get("photographs") or "").strip()
|
|
anom = (row.get("anomalies_en") or "").strip()
|
|
if not photos or not anom:
|
|
continue
|
|
labs = [x.strip().lower() for x in anom.replace('"', "").split(",") if x.strip()]
|
|
is_single = len(labs) == 1
|
|
for fn in photos.split(","):
|
|
fn = fn.strip()
|
|
if not fn:
|
|
continue
|
|
imgp = os.path.join(photo_dir, fn)
|
|
if not os.path.exists(imgp):
|
|
continue
|
|
for lab in labs:
|
|
(single if is_single else multi).setdefault(lab, []).append(imgp)
|
|
label_to_images = {}
|
|
for k in set(single) | set(multi):
|
|
label_to_images[k] = list(dict.fromkeys(single.get(k, []) + multi.get(k, [])))[:50]
|
|
print(f"COde: {n_rows} rows, {len(label_to_images)} labels, "
|
|
f"{sum(len(v) for v in label_to_images.values())} candidate images "
|
|
f"(single-anomaly 우선)", flush=True)
|
|
if not label_to_images:
|
|
return 0
|
|
|
|
# 2. 미매핑 term → 매칭 라벨 → 후보 검증
|
|
where = "(image_url IS NULL OR image_url='') AND english IS NOT NULL AND english!=''"
|
|
params = []
|
|
if category:
|
|
where += " AND category=?"
|
|
params.append(category)
|
|
terms = conn.execute(
|
|
"SELECT id,korean,english,category FROM terms WHERE " + where, params
|
|
).fetchall()
|
|
print(f"COde: {len(terms)} unmapped terms"
|
|
+ (f" (category={category})" if category else ""), flush=True)
|
|
|
|
def _word_match(enl, lab):
|
|
"""enl == lab 또는 한쪽이 단어경계 내에서 다른쪽에 포함.
|
|
부분문자열 사고(pit↔pulpitis 등) 방지."""
|
|
if enl == lab:
|
|
return True
|
|
if re.search(r"\b" + re.escape(lab) + r"\b", enl):
|
|
return True
|
|
if re.search(r"\b" + re.escape(enl) + r"\b", lab):
|
|
return True
|
|
return False
|
|
|
|
# COde는 질환 '묘사' 사진만 있으므로 예방/역학/병인 등 비-묘사 개념 term은 스킵
|
|
NON_DEPICT = re.compile(
|
|
r"\b(prevention|epidemiology|etiology|pathogenesis|classification|"
|
|
r"prognosis|prevalence|incidence|management of|treatment of|"
|
|
r"diagnosis of|definition|overview|introduction|history of)\b"
|
|
)
|
|
|
|
label_ptrs = {k: 0 for k in label_to_images}
|
|
updates = []
|
|
used = set()
|
|
MAX_TRIES = 3
|
|
|
|
def _assign(tid, ko, en, cat, best):
|
|
"""Try to assign an image from label_to_images[best]. Returns True on success."""
|
|
cands = label_to_images[best]
|
|
ptr = label_ptrs[best]
|
|
tries = 0
|
|
while ptr < len(cands) and tries < MAX_TRIES:
|
|
imgp = cands[ptr]
|
|
ptr += 1
|
|
tries += 1
|
|
if not os.path.exists(imgp):
|
|
continue
|
|
if verify:
|
|
res = validate_image(imgp, ko, en, cat,
|
|
hint=f"COde label: {best}", model=VERIFY_MODEL)
|
|
if not res["ok"]:
|
|
continue
|
|
fn = os.path.basename(imgp)
|
|
url = f"{URL_PREFIX}/code/Images/Photographs/{fn}"
|
|
updates.append((url, tid))
|
|
used.add(tid)
|
|
label_ptrs[best] = ptr
|
|
print(f" ✓ {ko} ← {best}", flush=True)
|
|
return True
|
|
label_ptrs[best] = ptr
|
|
return False
|
|
|
|
# Pass 1: 단어경계 문자열 매칭
|
|
unmatched = []
|
|
for tid, ko, en, cat in terms:
|
|
if max_terms and len(updates) >= max_terms:
|
|
break
|
|
enl = en.strip().lower()
|
|
if NON_DEPICT.search(enl):
|
|
continue
|
|
matches = [lab for lab in label_to_images if _word_match(enl, lab)]
|
|
if not matches:
|
|
unmatched.append((tid, ko, en, cat))
|
|
continue
|
|
best = max(matches, key=len)
|
|
_assign(tid, ko, en, cat, best)
|
|
|
|
# Pass 2: LLM 의미론적 라벨 매칭 (문자열 매칭 실패 term)
|
|
if llm_match and unmatched:
|
|
# 이미 매핑된 term은 제외
|
|
pending = [(tid, ko, en, cat) for tid, ko, en, cat in unmatched if tid not in used]
|
|
print(f" LLM 라벨 매칭: {len(pending)}개 term → 배치 처리 중...", flush=True)
|
|
BATCH = 20
|
|
label_list = list(label_to_images.keys())
|
|
for i in range(0, len(pending), BATCH):
|
|
if max_terms and len(updates) >= max_terms:
|
|
break
|
|
batch = pending[i:i + BATCH]
|
|
print(f" 배치 {i // BATCH + 1}/{(len(pending) + BATCH - 1) // BATCH} "
|
|
f"({len(batch)}개)", flush=True)
|
|
label_map = _llm_label_match_batch(batch, label_list)
|
|
for tid, ko, en, cat in batch:
|
|
if max_terms and len(updates) >= max_terms:
|
|
break
|
|
if tid in used:
|
|
continue
|
|
best = label_map.get(tid)
|
|
if not best:
|
|
continue
|
|
_assign(tid, ko, en, cat, best)
|
|
|
|
for url, tid in updates:
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (url, tid))
|
|
conn.commit()
|
|
print(f"COde: {len(updates)} mappings committed", flush=True)
|
|
return len(updates)
|
|
|
|
|
|
# ============================================================
|
|
# 17. Stock Image APIs — Pixabay (CC0) + Pexels + Unsplash
|
|
# ============================================================
|
|
def map_stock(conn, max_terms=50, delay=0.5, category=None,
|
|
model=VERIFY_MODEL, verify=True):
|
|
"""Pixabay(CC0) → Pexels → Unsplash 순으로 term별 이미지 검색.
|
|
키 파일: .smallclaw/pixabay_api_key.txt, pexels_api_key.txt, unsplash_api_key.txt
|
|
verify=True: 비전 모델 검증 후 저장. False: 첫 결과 바로 저장(빠름).
|
|
"""
|
|
|
|
def _key(fname):
|
|
try:
|
|
with open(os.path.join(SMALLCLAW, fname)) as f:
|
|
return f.read().strip()
|
|
except FileNotFoundError:
|
|
return None
|
|
|
|
PIXABAY = _key("pixabay_api_key.txt")
|
|
PEXELS = _key("pexels_api_key.txt")
|
|
UNSPLASH = _key("unsplash_api_key.txt")
|
|
|
|
if not any([PIXABAY, PEXELS, UNSPLASH]):
|
|
print("Stock: API 키 없음 — skipping", flush=True)
|
|
return 0
|
|
|
|
UA = "DentalDictBot/2.0 (educational)"
|
|
|
|
def _get(url, headers={}):
|
|
req = urllib.request.Request(url, headers={"User-Agent": UA, **headers})
|
|
with urllib.request.urlopen(req, timeout=15) as r:
|
|
return json.loads(r.read())
|
|
|
|
def _pixabay(q, n=5):
|
|
if not PIXABAY:
|
|
return []
|
|
try:
|
|
d = _get(f"https://pixabay.com/api/?key={PIXABAY}"
|
|
f"&q={urllib.parse.quote(q)}&image_type=photo"
|
|
f"&per_page={n}&safesearch=true&category=science")
|
|
return [h["webformatURL"] for h in d.get("hits", [])]
|
|
except Exception as e:
|
|
print(f" Pixabay: {e}", flush=True)
|
|
return []
|
|
|
|
def _pexels(q, n=5):
|
|
if not PEXELS:
|
|
return []
|
|
try:
|
|
d = _get(f"https://api.pexels.com/v1/search"
|
|
f"?query={urllib.parse.quote(q)}&per_page={n}",
|
|
{"Authorization": PEXELS})
|
|
return [p["src"]["medium"] for p in d.get("photos", [])]
|
|
except Exception as e:
|
|
print(f" Pexels: {e}", flush=True)
|
|
return []
|
|
|
|
def _unsplash(q, n=5):
|
|
if not UNSPLASH:
|
|
return []
|
|
try:
|
|
d = _get(f"https://api.unsplash.com/search/photos"
|
|
f"?query={urllib.parse.quote(q)}&per_page={n}",
|
|
{"Authorization": f"Client-ID {UNSPLASH}"})
|
|
return [r["urls"]["regular"] for r in d.get("results", [])]
|
|
except Exception as e:
|
|
print(f" Unsplash: {e}", flush=True)
|
|
return []
|
|
|
|
if category:
|
|
terms = conn.execute("""
|
|
SELECT id, korean, english, category FROM terms
|
|
WHERE (image_url IS NULL OR image_url='') AND category=?
|
|
AND english IS NOT NULL AND english != ''
|
|
ORDER BY id LIMIT ?
|
|
""", (category, max_terms)).fetchall()
|
|
else:
|
|
terms = conn.execute("""
|
|
SELECT id, korean, english, category FROM terms
|
|
WHERE (image_url IS NULL OR image_url='')
|
|
AND english IS NOT NULL AND english != ''
|
|
ORDER BY category, id LIMIT ?
|
|
""", (max_terms,)).fetchall()
|
|
|
|
keys_info = f"Pixabay={'✓' if PIXABAY else '✗'} Pexels={'✓' if PEXELS else '✗'} Unsplash={'✓' if UNSPLASH else '✗'}"
|
|
print(f"Stock: {len(terms)} terms"
|
|
f"{f' in {category}' if category else ''} ({keys_info}, verify={'on' if verify else 'off'})", flush=True)
|
|
|
|
updates = []
|
|
for i, (tid, ko, en, cat) in enumerate(terms):
|
|
time.sleep(delay)
|
|
q = f"dental {en}" if "dental" not in en.lower() else en
|
|
|
|
# Pixabay → Pexels → Unsplash
|
|
candidates = (
|
|
[("Pixabay", u) for u in _pixabay(q)] +
|
|
[("Pexels", u) for u in _pexels(q)] +
|
|
[("Unsplash", u) for u in _unsplash(q)]
|
|
)
|
|
if not candidates:
|
|
print(f" [{i+1}/{len(terms)}] {ko} — 결과 없음", flush=True)
|
|
continue
|
|
|
|
print(f" [{i+1}/{len(terms)}] {ko} — {len(candidates)} 후보", flush=True)
|
|
|
|
assigned = False
|
|
for src, img_url in candidates:
|
|
if verify:
|
|
res = validate_image(img_url, ko, en, cat,
|
|
hint=f"Stock photo ({src})", model=model)
|
|
if not res["ok"]:
|
|
continue
|
|
local_url = localize_image(tid, img_url, korean=ko)
|
|
final_url = local_url or img_url
|
|
conn.execute("UPDATE terms SET image_url=? WHERE id=?", (final_url, tid))
|
|
conn.commit()
|
|
updates.append((final_url, tid))
|
|
print(f" ✓ {src}", flush=True)
|
|
assigned = True
|
|
break
|
|
|
|
if not assigned:
|
|
print(f" — 적합 이미지 없음", flush=True)
|
|
|
|
print(f"Stock: {len(updates)} mappings committed", flush=True)
|
|
return len(updates)
|
|
|
|
|
|
# ============================================================
|
|
# 18. BRAR / AlphaDent — zip-only 데이터셋 (로컬 다운로드 필요)
|
|
# ============================================================
|
|
def map_brar(conn, verify=True):
|
|
"""BRAR-anchored multimodal (Figshare 30155974). 1,104 파노라마 방사선, CC-BY-4.0.
|
|
Figshare에 467MB zip만 있어 로컬 다운로드 필요."""
|
|
root = os.path.join(IMG_BASE, "brar")
|
|
if not os.path.isdir(root) or not any(os.scandir(root)):
|
|
print("BRAR: 로컬 다운로드 필요 — Figshare 30155974 (467MB zip)", flush=True)
|
|
print(" https://api.figshare.com/v2/articles/30155974/files", flush=True)
|
|
print(" dental_images/brar/ 에 풀고 재실행.", flush=True)
|
|
return 0
|
|
print("BRAR: 로컬 매핑은 미구현 (파노라마 radiograph zip) — 필요시 구현 요청", flush=True)
|
|
return 0
|
|
|
|
|
|
def map_alphadent(conn, verify=True):
|
|
"""AlphaDent (Zenodo 16582489). 치아 병변 탐지, Apache-2.0.
|
|
Zenodo에 4.8GB zip만 있어 로컬 다운로드 필요."""
|
|
root = os.path.join(IMG_BASE, "alphadent")
|
|
if not os.path.isdir(root) or not any(os.scandir(root)):
|
|
print("AlphaDent: 로컬 다운로드 필요 — Zenodo 16582489 (4.8GB zip)", flush=True)
|
|
print(" https://zenodo.org/api/records/16582489", flush=True)
|
|
print(" dental_images/alphadent/ 에 풀고 재실행.", flush=True)
|
|
return 0
|
|
print("AlphaDent: 로컬 매핑 미구현 (zip 구조 미확정) — 필요시 구현 요청", flush=True)
|
|
return 0
|
|
|
|
|
|
# ============================================================
|
|
# 18. OOPID / STS-3D-Tooth — 대용량 niche (수동 다운로드)
|
|
# ============================================================
|
|
def map_oopid(conn, verify=True):
|
|
"""OOPID (oopid.jp). 구강 세포병리 9,593 patch, 17GB, 커스텀 라이선스.
|
|
cytology niche — 수동 다운로드 필요."""
|
|
print("OOPID: 수동 다운로드 필요 — https://oopid.jp/ (17GB, 커스텀 라이선스)", flush=True)
|
|
print(" 구강 세포병리(cytology) niche — 사전 term 매핑 가치 낮음.", flush=True)
|
|
return 0
|
|
|
|
|
|
def map_sts3d(conn, verify=True):
|
|
"""STS-3D-Tooth (Zenodo 10597292). 3D CBCT 371볼륨, 31GB, CC-BY-4.0.
|
|
3D 볼륨 데이터 — 2D term 매핑 부적합."""
|
|
print("STS-3D-Tooth: 수동 다운로드 필요 — Zenodo 10597292 (31GB 3D CBCT)", flush=True)
|
|
print(" 3D 볼륨 데이터로 2D term 매핑 부적합 — niche.", flush=True)
|
|
return 0
|
|
|
|
|
|
# ============================================================
|
|
# Main
|
|
# ============================================================
|
|
# Image sources, in execution order for source="all". See docs/howto.md.
|
|
SOURCES = [
|
|
"kaggle", "roboflow", "huggingface", "radiopaedia", "itu",
|
|
"roboflow_api", "zenodo", "mendeley", "openi", "wikimedia",
|
|
"figshare", "pmc",
|
|
# 신규 (2026-06): 웹 스크랩/API + 로컬 다운로드 데이터셋
|
|
"vcu", "oralsd", "akudental", "code", "alphadent", "brar", "oopid", "sts3d",
|
|
# 스톡 이미지 API (Pixabay CC0 / Pexels / Unsplash)
|
|
"stock",
|
|
]
|
|
|
|
|
|
def run(source="all", verify=True,
|
|
max_wikimedia=50, wikimedia_category=None, wikimedia_model=VERIFY_MODEL, delay=5,
|
|
max_openi=50, openi_delay=0.5, openi_category=None, openi_model=VERIFY_MODEL,
|
|
max_figshare=50, figshare_category=None, figshare_model=VERIFY_MODEL, figshare_delay=1,
|
|
max_pmc=50, pmc_category=None, pmc_model=VERIFY_MODEL, pmc_delay=1,
|
|
max_vcu=60, vcu_category=None, vcu_model=VERIFY_MODEL, vcu_delay=1,
|
|
max_oralsd=50, oralsd_category=None, oralsd_model=VERIFY_MODEL, oralsd_delay=0.5,
|
|
max_akudental=20, akudental_category=None, akudental_model=VERIFY_MODEL, akudental_delay=0.5,
|
|
max_code=None, code_category=None, code_llm_match=False,
|
|
max_stock=50, stock_category=None, stock_model=VERIFY_MODEL, stock_delay=0.5):
|
|
"""Map dental dictionary terms to images from one or all sources.
|
|
|
|
Called by manage.py's `add-images` subcommand. `source` is "all" or one
|
|
of SOURCES. When `verify` is True every source validates each candidate
|
|
image with the vision model before saving (openi/wikimedia/figshare/pmc always do).
|
|
Returns the number of new mappings committed.
|
|
"""
|
|
conn = connect_db()
|
|
if source == "all":
|
|
want = SOURCES
|
|
elif "," in source:
|
|
want = [s.strip() for s in source.split(",") if s.strip() in SOURCES]
|
|
else:
|
|
want = [source]
|
|
|
|
total_before = conn.execute(
|
|
"SELECT COUNT(*) FROM terms WHERE image_url IS NOT NULL AND image_url != ''"
|
|
).fetchone()[0]
|
|
print(f"Images before: {total_before} (verify={'on' if verify else 'off'})")
|
|
|
|
if "kaggle" in want:
|
|
print("\n=== Mapping Kaggle datasets ===")
|
|
print(f"Kaggle: {map_kaggle_datasets(conn, verify=verify)} mapped")
|
|
|
|
if "roboflow" in want:
|
|
print("\n=== Mapping Roboflow datasets ===")
|
|
print(f"Roboflow: {map_roboflow_datasets(conn, verify=verify)} mapped")
|
|
|
|
if "huggingface" in want:
|
|
print("\n=== Mapping Hugging Face datasets ===")
|
|
print(f"Hugging Face: {map_huggingface_datasets(conn, verify=verify)} mapped")
|
|
|
|
if "radiopaedia" in want:
|
|
print("\n=== Mapping Radiopaedia URLs ===")
|
|
print(f"Radiopaedia: {map_radiopaedia_urls(conn)} mapped")
|
|
|
|
if "itu" in want:
|
|
print("\n=== Mapping ITU/HF datasets ===")
|
|
print(f"ITU/HF: {map_itu_datasets(conn)} mapped")
|
|
|
|
if "roboflow_api" in want:
|
|
print("\n=== Mapping Roboflow API (additional projects) ===")
|
|
print(f"Roboflow API: {map_roboflow_api_projects(conn, verify=verify)} mapped")
|
|
|
|
if "zenodo" in want:
|
|
print("\n=== Mapping Zenodo records ===")
|
|
print(f"Zenodo: {map_zenodo_records(conn, verify=verify)} mapped")
|
|
|
|
if "mendeley" in want:
|
|
print("\n=== Mapping Mendeley Data ===")
|
|
print(f"Mendeley: {map_mendeley_dataset(conn, verify=verify)} mapped")
|
|
|
|
if "openi" in want:
|
|
print("\n=== Mapping via NLM Open-i ===")
|
|
n = map_openi(conn, max_terms=max_openi, delay=openi_delay,
|
|
category=openi_category, model=openi_model)
|
|
print(f"Open-i: {n} mapped")
|
|
|
|
if "wikimedia" in want:
|
|
print("\n=== Searching Wikimedia Commons ===")
|
|
n = search_wikimedia_commons(conn, max_terms=max_wikimedia,
|
|
delay=delay, category=wikimedia_category,
|
|
model=wikimedia_model)
|
|
print(f"Wikimedia: {n} mapped")
|
|
|
|
if "figshare" in want:
|
|
print("\n=== Figshare 공개 그림 ===")
|
|
n = map_figshare(conn, max_terms=max_figshare, delay=figshare_delay,
|
|
category=figshare_category, model=figshare_model)
|
|
print(f"Figshare: {n} mapped")
|
|
|
|
if "pmc" in want:
|
|
print("\n=== PMC Entrez Figure 추출 ===")
|
|
n = map_pmc_entrez(conn, max_terms=max_pmc, delay=pmc_delay,
|
|
category=pmc_category, model=pmc_model)
|
|
print(f"PMC Entrez: {n} mapped")
|
|
|
|
if "vcu" in want:
|
|
print("\n=== VCU Oral Pathology Review ===")
|
|
n = map_vcu(conn, max_terms=max_vcu, delay=vcu_delay,
|
|
category=vcu_category, model=vcu_model)
|
|
print(f"VCU: {n} mapped")
|
|
|
|
if "oralsd" in want:
|
|
print("\n=== OralSDv1 (GitHub) ===")
|
|
n = map_oralsd(conn, max_terms=max_oralsd, delay=oralsd_delay,
|
|
category=oralsd_category, model=oralsd_model)
|
|
print(f"OralSDv1: {n} mapped")
|
|
|
|
if "akudental" in want:
|
|
print("\n=== AKUDENTAL 파노라마 ===")
|
|
n = map_akudental(conn, max_terms=max_akudental, delay=akudental_delay,
|
|
category=akudental_category, model=akudental_model)
|
|
print(f"AKUDENTAL: {n} mapped")
|
|
|
|
if "code" in want:
|
|
print("\n=== COde (HuggingFace 로컬) ===")
|
|
print(f"COde: {map_code(conn, verify=verify, max_terms=max_code, category=code_category, llm_match=code_llm_match)} mapped")
|
|
|
|
if "alphadent" in want:
|
|
print("\n=== AlphaDent (로컬 다운로드) ===")
|
|
print(f"AlphaDent: {map_alphadent(conn, verify=verify)} mapped")
|
|
|
|
if "brar" in want:
|
|
print("\n=== BRAR 파노라마 (로컬 다운로드) ===")
|
|
print(f"BRAR: {map_brar(conn, verify=verify)} mapped")
|
|
|
|
if "oopid" in want:
|
|
print("\n=== OOPID 세포병리 ===")
|
|
print(f"OOPID: {map_oopid(conn, verify=verify)} mapped")
|
|
|
|
if "sts3d" in want:
|
|
print("\n=== STS-3D-Tooth CBCT ===")
|
|
print(f"STS-3D: {map_sts3d(conn, verify=verify)} mapped")
|
|
|
|
if "stock" in want:
|
|
print("\n=== Stock Images (Pixabay / Pexels / Unsplash) ===")
|
|
n = map_stock(conn, max_terms=max_stock, delay=stock_delay,
|
|
category=stock_category, model=stock_model, verify=verify)
|
|
print(f"Stock: {n} mapped")
|
|
|
|
total_after = conn.execute(
|
|
"SELECT COUNT(*) FROM terms WHERE image_url IS NOT NULL AND image_url != ''"
|
|
).fetchone()[0]
|
|
|
|
print(f"\n=== Summary ===")
|
|
print(f"Images before: {total_before}")
|
|
print(f"Images after: {total_after}")
|
|
print(f"New mappings: {total_after - total_before}")
|
|
conn.close()
|
|
return total_after - total_before |