- cherry: orofacial/TMD PPTX presentations + 음악가 구강건강 가이드 - papa: sinus/infraoccluded PPTX, garden-monitor Arduino, MIDI files - databases: PCSP scraper, translation scripts, vision benchmark, batch JSONs - web-ui: code.js/companion.py/styles.css major updates, autostart installer - src: ollama-client, session, LLMProvider, ollama-adapter patches - .gitignore: add claude-key, SQLite WAL/SHM, __pycache__, nohup.out Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
398 lines
14 KiB
Python
398 lines
14 KiB
Python
#!/usr/bin/env python3
|
|
"""Scrape PCSP (Pragmatic Case Studies in Psychotherapy) and build SQLite DB."""
|
|
|
|
import sqlite3
|
|
import json
|
|
import time
|
|
import re
|
|
import sys
|
|
from urllib.request import urlopen, Request
|
|
from html.parser import HTMLParser
|
|
|
|
BASE = "https://pcsp.nationalregister.org/index.php/pcsp"
|
|
DB_PATH = "/home/kim/homeclaw/.smallclaw/databases/psychotherapy_cases.db"
|
|
HEADERS = {"User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36"}
|
|
|
|
# ── HTML text extraction ──────────────────────────────────────────────
|
|
|
|
class TextExtractor(HTMLParser):
|
|
def __init__(self):
|
|
super().__init__()
|
|
self._pieces = []
|
|
self._skip = False
|
|
def handle_starttag(self, tag, attrs):
|
|
if tag in ("script", "style", "nav", "header", "footer"):
|
|
self._skip = True
|
|
def handle_endtag(self, tag):
|
|
if tag in ("script", "style", "nav", "header", "footer"):
|
|
self._skip = False
|
|
def handle_data(self, data):
|
|
if not self._skip:
|
|
self._pieces.append(data)
|
|
def get_text(self):
|
|
return " ".join(self._pieces)
|
|
|
|
def fetch(url, retries=3):
|
|
for i in range(retries):
|
|
try:
|
|
req = Request(url, headers=HEADERS)
|
|
with urlopen(req, timeout=30) as r:
|
|
return r.read().decode("utf-8", errors="replace")
|
|
except Exception as e:
|
|
print(f" Retry {i+1}/{retries} for {url}: {e}", file=sys.stderr)
|
|
time.sleep(5 * (i + 1))
|
|
return ""
|
|
|
|
def html_to_text(html):
|
|
p = TextExtractor()
|
|
p.feed(html)
|
|
return p.get_text().strip()
|
|
|
|
# ── Issue list from archive ──────────────────────────────────────────
|
|
|
|
def get_all_issues():
|
|
"""Get all issue URLs from archive pages."""
|
|
issues = []
|
|
for page in range(1, 5):
|
|
url = f"{BASE}/issue/archive/{page}" if page > 1 else f"{BASE}/issue/archive"
|
|
html = fetch(url)
|
|
# Find issue view links
|
|
for m in re.finditer(r'href="(https?://[^"]*/issue/view/(\d+))"', html):
|
|
issue_url, issue_id = m.group(1), m.group(2)
|
|
# Extract volume info from nearby text
|
|
issues.append({"url": issue_url, "issue_id": int(issue_id)})
|
|
# Deduplicate
|
|
seen = set()
|
|
unique = []
|
|
for i in issues:
|
|
if i["issue_id"] not in seen:
|
|
seen.add(i["issue_id"])
|
|
unique.append(i)
|
|
return sorted(unique, key=lambda x: x["issue_id"], reverse=True)
|
|
|
|
# ── Article extraction from issue page ────────────────────────────────
|
|
|
|
def extract_articles_from_issue(issue_url, issue_id):
|
|
"""Extract article links and metadata from an issue page."""
|
|
html = fetch(issue_url)
|
|
if not html:
|
|
return []
|
|
|
|
# Find volume/issue info
|
|
vol_match = re.search(r'Vol\s*(\d+)[\s,]+No\s*(\d+)\s*\((\d{4})\)', html)
|
|
vol = int(vol_match.group(1)) if vol_match else 0
|
|
num = int(vol_match.group(2)) if vol_match else 0
|
|
year = int(vol_match.group(3)) if vol_match else 0
|
|
|
|
articles = []
|
|
# Find article view links
|
|
seen_ids = set()
|
|
for m in re.finditer(r'href="(https?://[^"]*/article/view/(\d+))"', html):
|
|
art_url, art_id = m.group(1), m.group(2)
|
|
art_id = int(art_id)
|
|
if art_id in seen_ids:
|
|
continue
|
|
seen_ids.add(art_id)
|
|
|
|
# Try to find title near the link
|
|
# The title is usually in the next <a> or nearby text
|
|
# We'll get the title from the article page itself
|
|
articles.append({
|
|
"url": art_url,
|
|
"article_id": art_id,
|
|
"issue_id": issue_id,
|
|
"vol": vol,
|
|
"num": num,
|
|
"year": year,
|
|
})
|
|
|
|
return articles
|
|
|
|
# ── Article detail extraction ────────────────────────────────────────
|
|
|
|
def extract_article_detail(art):
|
|
"""Get metadata from an article page."""
|
|
url = art["url"]
|
|
html = fetch(url)
|
|
if not html:
|
|
return None
|
|
|
|
text = html_to_text(html)
|
|
|
|
# Title: usually in <h1> or meta tag
|
|
title = ""
|
|
h1_match = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
|
|
if h1_match:
|
|
title = re.sub(r'<[^>]+>', '', h1_match.group(1)).strip()
|
|
if not title:
|
|
meta_match = re.search(r'<meta\s+name="citation_title"\s+content="([^"]+)"', html)
|
|
if meta_match:
|
|
title = meta_match.group(1)
|
|
if not title:
|
|
meta_match = re.search(r'<meta\s+property="og:title"\s+content="([^"]+)"', html)
|
|
if meta_match:
|
|
title = meta_match.group(1)
|
|
|
|
# Authors
|
|
authors = ""
|
|
author_matches = re.findall(r'<meta\s+name="citation_author"\s+content="([^"]+)"', html)
|
|
if author_matches:
|
|
authors = "; ".join(author_matches)
|
|
|
|
# DOI
|
|
doi = ""
|
|
doi_match = re.search(r'<meta\s+name="citation_doi"\s+content="([^"]+)"', html)
|
|
if doi_match:
|
|
doi = doi_match.group(1)
|
|
|
|
# Keywords
|
|
keywords = ""
|
|
kw_match = re.search(r'<meta\s+name="citation_keywords"\s+content="([^"]+)"', html)
|
|
if kw_match:
|
|
keywords = kw_match.group(1)
|
|
# Also try to find keywords section in text
|
|
kw_section = re.search(r'Keywords[:\s]+(.*?)(?:\n|<|Introduction|Abstract)', text, re.IGNORECASE)
|
|
if kw_section and not keywords:
|
|
keywords = kw_section.group(1).strip().rstrip(".")
|
|
|
|
# Abstract
|
|
abstract = ""
|
|
# Try meta description first
|
|
abs_match = re.search(r'<meta\s+name="description"\s+content="([^"]+)"', html)
|
|
if abs_match:
|
|
abstract = abs_match.group(1)
|
|
# Try to find abstract section in the page
|
|
if not abstract:
|
|
abs_match = re.search(r'Abstract[:\s]+(.*?)(?:(?:Introduction|Keywords|1\.\s))', text, re.DOTALL | re.IGNORECASE)
|
|
if abs_match:
|
|
abstract = abs_match.group(1).strip()[:2000]
|
|
if not abstract:
|
|
# Try dc.description
|
|
abs_match = re.search(r'<meta\s+name="DC\.Description"\s+content="([^"]+)"', html)
|
|
if abs_match:
|
|
abstract = abs_match.group(1)
|
|
|
|
# Article type
|
|
article_type = "article"
|
|
if "commentary" in title.lower():
|
|
article_type = "commentary"
|
|
elif "response" in title.lower() and "comment" in title.lower():
|
|
article_type = "response"
|
|
elif "case study" in title.lower() or "case " in title.lower():
|
|
article_type = "case_study"
|
|
|
|
# Determine category from keywords/title
|
|
category = classify_article(title, keywords)
|
|
|
|
result = {
|
|
**art,
|
|
"title_en": title,
|
|
"authors": authors,
|
|
"doi": doi,
|
|
"keywords": keywords,
|
|
"abstract_en": abstract[:3000] if abstract else "",
|
|
"article_type": article_type,
|
|
"category_en": category,
|
|
}
|
|
return result
|
|
|
|
def classify_article(title, keywords):
|
|
"""Classify article into therapy approach category."""
|
|
text = (title + " " + keywords).lower()
|
|
|
|
categories = {
|
|
"CBT": ["cognitive behavio", "cbt", "cognitive-behavio", "cognitive therapy"],
|
|
"DBT": ["dialectical behavio", "dbt"],
|
|
"Psychodynamic": ["psychodynamic", "dynamic therapy", "psychoanalytic", "psychoanalysis"],
|
|
"Humanistic": ["humanistic", "client-centered", "person-centered", "rogerian"],
|
|
"AEDP": ["aedp", "accelerated experiential dynamic"],
|
|
"ACT": ["acceptance and commitment", "act ", "act therapy"],
|
|
"Schema Therapy": ["schema therapy"],
|
|
"EMDR": ["emdr", "eye movement desensitization"],
|
|
"Narrative Therapy": ["narrative therapy"],
|
|
"Exposure Therapy": ["exposure therapy", "exposure and response prevention", "erp"],
|
|
"Hypnosis": ["hypnosis", "hypnotherapy", "hypnotic"],
|
|
"Integrative": ["integrative", "eclectic", "multimodal"],
|
|
"EFT": ["emotionally focused", "eft"],
|
|
"Mindfulness": ["mindfulness", "mbct", "mbsr"],
|
|
"SFBT": ["solution-focused", "solution focused", "sfbt"],
|
|
}
|
|
|
|
for cat, terms in categories.items():
|
|
if any(t in text for t in terms):
|
|
return cat
|
|
|
|
return "Other"
|
|
|
|
# ── Korean translation mapping ────────────────────────────────────────
|
|
|
|
THERAPY_KO = {
|
|
"CBT": "인지행동치료",
|
|
"DBT": "변증법적행동치료",
|
|
"Psychodynamic": "정신역동치료",
|
|
"Humanistic": "인간중심치료",
|
|
"AEDP": "가속적 경험적 역동치료",
|
|
"ACT": "수용전념치료",
|
|
"Schema Therapy": "도식치료",
|
|
"EMDR": "안구운동 민감소실 재처리",
|
|
"Narrative Therapy": "내러티브치료",
|
|
"Exposure Therapy": "노출치료",
|
|
"Hypnosis": "최면치료",
|
|
"Integrative": "통합치료",
|
|
"EFT": "정서초점치료",
|
|
"Mindfulness": "마음챙김치료",
|
|
"SFBT": "해결중심단기치료",
|
|
"Other": "기타",
|
|
}
|
|
|
|
CATEGORY_KO = {
|
|
"CBT": "인지행동치료",
|
|
"DBT": "변증법적행동치료",
|
|
"Psychodynamic": "정신역동치료",
|
|
"Humanistic": "인간중심치료",
|
|
"AEDP": "가속적경험적역동치료",
|
|
"ACT": "수용전념치료",
|
|
"Schema Therapy": "도식치료",
|
|
"EMDR": "안구운동민감소실재처리",
|
|
"Narrative Therapy": "내러티브치료",
|
|
"Exposure Therapy": "노출치료",
|
|
"Hypnosis": "최면치료",
|
|
"Integrative": "통합치료",
|
|
"EFT": "정서초점치료",
|
|
"Mindfulness": "마음챙김치료",
|
|
"SFBT": "해결중심단기치료",
|
|
"Other": "기타",
|
|
}
|
|
|
|
# ── Build database ────────────────────────────────────────────────────
|
|
|
|
def create_db():
|
|
conn = sqlite3.connect(DB_PATH)
|
|
c = conn.cursor()
|
|
c.execute("DROP TABLE IF EXISTS cases")
|
|
c.execute("""CREATE TABLE cases (
|
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
article_id INTEGER,
|
|
issue_id INTEGER,
|
|
vol INTEGER,
|
|
num INTEGER,
|
|
year INTEGER,
|
|
title_en TEXT,
|
|
title_ko TEXT,
|
|
authors TEXT,
|
|
article_type TEXT,
|
|
category_en TEXT,
|
|
category_ko TEXT,
|
|
therapy_approach_en TEXT,
|
|
therapy_approach_ko TEXT,
|
|
keywords TEXT,
|
|
abstract_en TEXT,
|
|
abstract_ko TEXT,
|
|
doi TEXT,
|
|
url TEXT,
|
|
source_url TEXT,
|
|
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
|
|
)""")
|
|
conn.commit()
|
|
return conn
|
|
|
|
def save_articles(conn, articles):
|
|
c = conn.cursor()
|
|
for art in articles:
|
|
if not art or not art.get("title_en"):
|
|
continue
|
|
cat_en = art.get("category_en", "Other")
|
|
cat_ko = CATEGORY_KO.get(cat_en, "기타")
|
|
approach_en = cat_en
|
|
approach_ko = THERAPY_KO.get(cat_en, "기타")
|
|
|
|
c.execute("""INSERT INTO cases (
|
|
article_id, issue_id, vol, num, year,
|
|
title_en, title_ko, authors, article_type,
|
|
category_en, category_ko,
|
|
therapy_approach_en, therapy_approach_ko,
|
|
keywords, abstract_en, abstract_ko,
|
|
doi, url, source_url
|
|
) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)""",
|
|
(
|
|
art.get("article_id"),
|
|
art.get("issue_id"),
|
|
art.get("vol"),
|
|
art.get("num"),
|
|
art.get("year"),
|
|
art.get("title_en", ""),
|
|
"", # title_ko - will be translated later
|
|
art.get("authors", ""),
|
|
art.get("article_type", "article"),
|
|
cat_en, cat_ko,
|
|
approach_en, approach_ko,
|
|
art.get("keywords", ""),
|
|
art.get("abstract_en", ""),
|
|
"", # abstract_ko - will be translated later
|
|
art.get("doi", ""),
|
|
art.get("url", ""),
|
|
art.get("url", ""),
|
|
))
|
|
conn.commit()
|
|
|
|
# ── Main ──────────────────────────────────────────────────────────────
|
|
|
|
def main():
|
|
print("=== PCSP Psychotherapy Case Studies DB Builder ===")
|
|
print()
|
|
|
|
# Step 1: Get all issues
|
|
print("[1/4] Fetching issue list...")
|
|
issues = get_all_issues()
|
|
print(f" Found {len(issues)} issues")
|
|
|
|
# Step 2: Get article links from each issue
|
|
print("[2/4] Extracting article links from issues...")
|
|
all_articles = []
|
|
for i, issue in enumerate(issues):
|
|
print(f" Issue {issue['issue_id']} ({i+1}/{len(issues)})...", end=" ", flush=True)
|
|
arts = extract_articles_from_issue(issue["url"], issue["issue_id"])
|
|
print(f"{len(arts)} articles")
|
|
all_articles.extend(arts)
|
|
time.sleep(1) # Be polite
|
|
print(f" Total articles found: {len(all_articles)}")
|
|
|
|
# Step 3: Get details for each article
|
|
print("[3/4] Extracting article details...")
|
|
detailed_articles = []
|
|
for i, art in enumerate(all_articles):
|
|
print(f" Article {art['article_id']} ({i+1}/{len(all_articles)})...", end=" ", flush=True)
|
|
detail = extract_article_detail(art)
|
|
if detail:
|
|
title_preview = detail.get("title_en", "")[:60]
|
|
print(f"OK - {title_preview}...")
|
|
detailed_articles.append(detail)
|
|
else:
|
|
print("SKIP (no data)")
|
|
time.sleep(0.5) # Be polite
|
|
|
|
# Step 4: Save to database
|
|
print(f"[4/4] Saving {len(detailed_articles)} articles to database...")
|
|
conn = create_db()
|
|
save_articles(conn, detailed_articles)
|
|
|
|
# Stats
|
|
c = conn.cursor()
|
|
c.execute("SELECT COUNT(*) FROM cases")
|
|
total = c.fetchone()[0]
|
|
c.execute("SELECT article_type, COUNT(*) FROM cases GROUP BY article_type")
|
|
types = dict(c.fetchall())
|
|
c.execute("SELECT category_en, COUNT(*) FROM cases GROUP BY category_en ORDER BY COUNT(*) DESC")
|
|
cats = c.fetchall()
|
|
|
|
print()
|
|
print(f"=== Done! {total} articles saved to {DB_PATH} ===")
|
|
print(f" Types: {types}")
|
|
print(f" Categories:")
|
|
for cat, cnt in cats:
|
|
print(f" {cat}: {cnt}")
|
|
|
|
conn.close()
|
|
|
|
if __name__ == "__main__":
|
|
main() |