Files
homeclaw/.smallclaw/databases/workflow_terms.py
T
kimandClaude Opus 4.7 e11ddd7f3e Refactor dental dictionary into manage.py CLI
Reorganize .smallclaw/databases/ around a single entry point with
subcommands (add-terms / add-images / verify / reassign / stats),
absorb pmc_reassign.py into reassign, and route every image source
through inline vision-model verification before commit.

Layout
  config.py          paths, endpoints, API-key locations (no more
                     hard-coded absolute paths in source files)
  manage.py          argparse dispatcher
  workflow_terms.py  seed-JSON import (JSONC supported, dedupes on
                     korean/english)
  workflow_images.py renamed from dental_image_workflow.py; main()
                     converted to run(verify=True, ...)
  workflow_verify.py validate_image() / verify_db() + verify_updates()
                     gate used by every add-images source
  seeds/             recovered seed_all.json, seed_periodontics.json
  scratch/           ad-hoc work area replacing the /tmp habit
                     (only README.md is tracked)
  docs/howto.md      moved + expanded
  docs/archive_image_rounds/   one-off round scripts + their results
  docs/archive_validation/     model-comparison + validation history

Verification
  - VERIFY_MODEL defaults to qwen3.5:397b-cloud (50-image test on
    procedure-heavy sample: 9.8s/img avg, zero timeouts, Kimi-level
    rigor — see archive_validation/).
  - Prompt strengthened: in-image text/captions are not valid grounds
    for "적합"; technique terms require visible procedure steps, not
    generic device/anatomy photos.
  - All 9 add-images sources gated through verify_updates() before
    commit; --no-verify escape hatch for bulk runs.

.gitignore: API key files, dental_images/, __pycache__/, scratch/*
(README.md kept).

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-05-15 11:58:53 +09:00

107 lines
3.3 KiB
Python

"""Term-addition workflow for the dental dictionary.
Imports terms from JSON seed files in seeds/ into dental_dict.db. Seed files
may contain `//` line comments (JSONC) — they are stripped before parsing.
A seed entry is an object with at least `korean`; any of these keys are used:
korean, english, latin, abbreviation, definition, category,
synonyms, related_ids, pmids, icd_code, notes
Duplicates (same korean OR same english) are skipped, so re-running a seed
file is safe. Called by manage.py's `add-terms` subcommand.
"""
import json
import os
import re
import sqlite3
from datetime import datetime
from config import DB_PATH, SEEDS_DIR
FIELDS = [
"korean", "english", "latin", "abbreviation", "definition",
"category", "synonyms", "related_ids", "pmids", "icd_code", "notes",
]
def load_jsonc(path):
"""Load a JSON file that may contain `//` line comments."""
with open(path, encoding="utf-8") as f:
text = f.read()
# Strip // comments that are not inside a string. Seed files only use
# // at the start of (optionally indented) lines, so a line-wise strip
# is sufficient and avoids a full JSON tokenizer.
cleaned = re.sub(r'^\s*//.*$', '', text, flags=re.MULTILINE)
return json.loads(cleaned)
def import_file(conn, path):
"""Import one seed file. Returns (inserted, skipped)."""
terms = load_jsonc(path)
now = datetime.now().isoformat()
inserted = skipped = 0
for t in terms:
korean = (t.get("korean") or "").strip()
english = (t.get("english") or "").strip()
if not korean:
continue
existing = conn.execute(
"SELECT id FROM terms WHERE korean = ? OR (english != '' AND english = ?)",
(korean, english),
).fetchone()
if existing:
skipped += 1
continue
values = [t.get(f, "") or "" for f in FIELDS]
conn.execute(
f"INSERT INTO terms ({', '.join(FIELDS)}, created_at, updated_at) "
f"VALUES ({', '.join('?' * len(FIELDS))}, ?, ?)",
(*values, now, now),
)
inserted += 1
conn.commit()
return inserted, skipped
def run(seed_files=None, all_seeds=False):
"""Import seed files into the DB.
seed_files: explicit list of paths (or names relative to seeds/).
all_seeds: if True, import every *.json under seeds/.
Returns total inserted count.
"""
if all_seeds:
paths = sorted(
os.path.join(SEEDS_DIR, f)
for f in os.listdir(SEEDS_DIR)
if f.endswith(".json")
)
else:
paths = []
for s in (seed_files or []):
paths.append(s if os.path.isabs(s) or os.path.exists(s)
else os.path.join(SEEDS_DIR, s))
if not paths:
print("No seed files given. Use --all or pass file names from seeds/.")
return 0
conn = sqlite3.connect(DB_PATH)
total_in = total_skip = 0
for path in paths:
if not os.path.exists(path):
print(f" ! not found: {path}")
continue
ins, skip = import_file(conn, path)
print(f" {os.path.basename(path)}: +{ins} inserted, {skip} skipped")
total_in += ins
total_skip += skip
conn.close()
print(f"\nTotal: {total_in} inserted, {total_skip} skipped")
return total_in