Reorganize .smallclaw/databases/ around a single entry point with
subcommands (add-terms / add-images / verify / reassign / stats),
absorb pmc_reassign.py into reassign, and route every image source
through inline vision-model verification before commit.
Layout
config.py paths, endpoints, API-key locations (no more
hard-coded absolute paths in source files)
manage.py argparse dispatcher
workflow_terms.py seed-JSON import (JSONC supported, dedupes on
korean/english)
workflow_images.py renamed from dental_image_workflow.py; main()
converted to run(verify=True, ...)
workflow_verify.py validate_image() / verify_db() + verify_updates()
gate used by every add-images source
seeds/ recovered seed_all.json, seed_periodontics.json
scratch/ ad-hoc work area replacing the /tmp habit
(only README.md is tracked)
docs/howto.md moved + expanded
docs/archive_image_rounds/ one-off round scripts + their results
docs/archive_validation/ model-comparison + validation history
Verification
- VERIFY_MODEL defaults to qwen3.5:397b-cloud (50-image test on
procedure-heavy sample: 9.8s/img avg, zero timeouts, Kimi-level
rigor — see archive_validation/).
- Prompt strengthened: in-image text/captions are not valid grounds
for "적합"; technique terms require visible procedure steps, not
generic device/anatomy photos.
- All 9 add-images sources gated through verify_updates() before
commit; --no-verify escape hatch for bulk runs.
.gitignore: API key files, dental_images/, __pycache__/, scratch/*
(README.md kept).
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
107 lines
3.3 KiB
Python
107 lines
3.3 KiB
Python
"""Term-addition workflow for the dental dictionary.
|
|
|
|
Imports terms from JSON seed files in seeds/ into dental_dict.db. Seed files
|
|
may contain `//` line comments (JSONC) — they are stripped before parsing.
|
|
|
|
A seed entry is an object with at least `korean`; any of these keys are used:
|
|
korean, english, latin, abbreviation, definition, category,
|
|
synonyms, related_ids, pmids, icd_code, notes
|
|
|
|
Duplicates (same korean OR same english) are skipped, so re-running a seed
|
|
file is safe. Called by manage.py's `add-terms` subcommand.
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import re
|
|
import sqlite3
|
|
from datetime import datetime
|
|
|
|
from config import DB_PATH, SEEDS_DIR
|
|
|
|
FIELDS = [
|
|
"korean", "english", "latin", "abbreviation", "definition",
|
|
"category", "synonyms", "related_ids", "pmids", "icd_code", "notes",
|
|
]
|
|
|
|
|
|
def load_jsonc(path):
|
|
"""Load a JSON file that may contain `//` line comments."""
|
|
with open(path, encoding="utf-8") as f:
|
|
text = f.read()
|
|
# Strip // comments that are not inside a string. Seed files only use
|
|
# // at the start of (optionally indented) lines, so a line-wise strip
|
|
# is sufficient and avoids a full JSON tokenizer.
|
|
cleaned = re.sub(r'^\s*//.*$', '', text, flags=re.MULTILINE)
|
|
return json.loads(cleaned)
|
|
|
|
|
|
def import_file(conn, path):
|
|
"""Import one seed file. Returns (inserted, skipped)."""
|
|
terms = load_jsonc(path)
|
|
now = datetime.now().isoformat()
|
|
inserted = skipped = 0
|
|
|
|
for t in terms:
|
|
korean = (t.get("korean") or "").strip()
|
|
english = (t.get("english") or "").strip()
|
|
if not korean:
|
|
continue
|
|
existing = conn.execute(
|
|
"SELECT id FROM terms WHERE korean = ? OR (english != '' AND english = ?)",
|
|
(korean, english),
|
|
).fetchone()
|
|
if existing:
|
|
skipped += 1
|
|
continue
|
|
|
|
values = [t.get(f, "") or "" for f in FIELDS]
|
|
conn.execute(
|
|
f"INSERT INTO terms ({', '.join(FIELDS)}, created_at, updated_at) "
|
|
f"VALUES ({', '.join('?' * len(FIELDS))}, ?, ?)",
|
|
(*values, now, now),
|
|
)
|
|
inserted += 1
|
|
|
|
conn.commit()
|
|
return inserted, skipped
|
|
|
|
|
|
def run(seed_files=None, all_seeds=False):
|
|
"""Import seed files into the DB.
|
|
|
|
seed_files: explicit list of paths (or names relative to seeds/).
|
|
all_seeds: if True, import every *.json under seeds/.
|
|
Returns total inserted count.
|
|
"""
|
|
if all_seeds:
|
|
paths = sorted(
|
|
os.path.join(SEEDS_DIR, f)
|
|
for f in os.listdir(SEEDS_DIR)
|
|
if f.endswith(".json")
|
|
)
|
|
else:
|
|
paths = []
|
|
for s in (seed_files or []):
|
|
paths.append(s if os.path.isabs(s) or os.path.exists(s)
|
|
else os.path.join(SEEDS_DIR, s))
|
|
|
|
if not paths:
|
|
print("No seed files given. Use --all or pass file names from seeds/.")
|
|
return 0
|
|
|
|
conn = sqlite3.connect(DB_PATH)
|
|
total_in = total_skip = 0
|
|
for path in paths:
|
|
if not os.path.exists(path):
|
|
print(f" ! not found: {path}")
|
|
continue
|
|
ins, skip = import_file(conn, path)
|
|
print(f" {os.path.basename(path)}: +{ins} inserted, {skip} skipped")
|
|
total_in += ins
|
|
total_skip += skip
|
|
conn.close()
|
|
|
|
print(f"\nTotal: {total_in} inserted, {total_skip} skipped")
|
|
return total_in
|