Files
homeclaw/.smallclaw/databases/scrape_pcsp.py
T
kimandClaude Sonnet 4.6 508d11615a Add user workspaces, dental DB scripts, server memory + code editor improvements
- cherry: orofacial/TMD PPTX presentations + 음악가 구강건강 가이드
- papa: sinus/infraoccluded PPTX, garden-monitor Arduino, MIDI files
- databases: PCSP scraper, translation scripts, vision benchmark, batch JSONs
- web-ui: code.js/companion.py/styles.css major updates, autostart installer
- src: ollama-client, session, LLMProvider, ollama-adapter patches
- .gitignore: add claude-key, SQLite WAL/SHM, __pycache__, nohup.out

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-07 23:13:03 +09:00

398 lines
14 KiB
Python

#!/usr/bin/env python3
"""Scrape PCSP (Pragmatic Case Studies in Psychotherapy) and build SQLite DB."""
import sqlite3
import json
import time
import re
import sys
from urllib.request import urlopen, Request
from html.parser import HTMLParser
BASE = "https://pcsp.nationalregister.org/index.php/pcsp"
DB_PATH = "/home/kim/homeclaw/.smallclaw/databases/psychotherapy_cases.db"
HEADERS = {"User-Agent": "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36"}
# ── HTML text extraction ──────────────────────────────────────────────
class TextExtractor(HTMLParser):
def __init__(self):
super().__init__()
self._pieces = []
self._skip = False
def handle_starttag(self, tag, attrs):
if tag in ("script", "style", "nav", "header", "footer"):
self._skip = True
def handle_endtag(self, tag):
if tag in ("script", "style", "nav", "header", "footer"):
self._skip = False
def handle_data(self, data):
if not self._skip:
self._pieces.append(data)
def get_text(self):
return " ".join(self._pieces)
def fetch(url, retries=3):
for i in range(retries):
try:
req = Request(url, headers=HEADERS)
with urlopen(req, timeout=30) as r:
return r.read().decode("utf-8", errors="replace")
except Exception as e:
print(f" Retry {i+1}/{retries} for {url}: {e}", file=sys.stderr)
time.sleep(5 * (i + 1))
return ""
def html_to_text(html):
p = TextExtractor()
p.feed(html)
return p.get_text().strip()
# ── Issue list from archive ──────────────────────────────────────────
def get_all_issues():
"""Get all issue URLs from archive pages."""
issues = []
for page in range(1, 5):
url = f"{BASE}/issue/archive/{page}" if page > 1 else f"{BASE}/issue/archive"
html = fetch(url)
# Find issue view links
for m in re.finditer(r'href="(https?://[^"]*/issue/view/(\d+))"', html):
issue_url, issue_id = m.group(1), m.group(2)
# Extract volume info from nearby text
issues.append({"url": issue_url, "issue_id": int(issue_id)})
# Deduplicate
seen = set()
unique = []
for i in issues:
if i["issue_id"] not in seen:
seen.add(i["issue_id"])
unique.append(i)
return sorted(unique, key=lambda x: x["issue_id"], reverse=True)
# ── Article extraction from issue page ────────────────────────────────
def extract_articles_from_issue(issue_url, issue_id):
"""Extract article links and metadata from an issue page."""
html = fetch(issue_url)
if not html:
return []
# Find volume/issue info
vol_match = re.search(r'Vol\s*(\d+)[\s,]+No\s*(\d+)\s*\((\d{4})\)', html)
vol = int(vol_match.group(1)) if vol_match else 0
num = int(vol_match.group(2)) if vol_match else 0
year = int(vol_match.group(3)) if vol_match else 0
articles = []
# Find article view links
seen_ids = set()
for m in re.finditer(r'href="(https?://[^"]*/article/view/(\d+))"', html):
art_url, art_id = m.group(1), m.group(2)
art_id = int(art_id)
if art_id in seen_ids:
continue
seen_ids.add(art_id)
# Try to find title near the link
# The title is usually in the next <a> or nearby text
# We'll get the title from the article page itself
articles.append({
"url": art_url,
"article_id": art_id,
"issue_id": issue_id,
"vol": vol,
"num": num,
"year": year,
})
return articles
# ── Article detail extraction ────────────────────────────────────────
def extract_article_detail(art):
"""Get metadata from an article page."""
url = art["url"]
html = fetch(url)
if not html:
return None
text = html_to_text(html)
# Title: usually in <h1> or meta tag
title = ""
h1_match = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
if h1_match:
title = re.sub(r'<[^>]+>', '', h1_match.group(1)).strip()
if not title:
meta_match = re.search(r'<meta\s+name="citation_title"\s+content="([^"]+)"', html)
if meta_match:
title = meta_match.group(1)
if not title:
meta_match = re.search(r'<meta\s+property="og:title"\s+content="([^"]+)"', html)
if meta_match:
title = meta_match.group(1)
# Authors
authors = ""
author_matches = re.findall(r'<meta\s+name="citation_author"\s+content="([^"]+)"', html)
if author_matches:
authors = "; ".join(author_matches)
# DOI
doi = ""
doi_match = re.search(r'<meta\s+name="citation_doi"\s+content="([^"]+)"', html)
if doi_match:
doi = doi_match.group(1)
# Keywords
keywords = ""
kw_match = re.search(r'<meta\s+name="citation_keywords"\s+content="([^"]+)"', html)
if kw_match:
keywords = kw_match.group(1)
# Also try to find keywords section in text
kw_section = re.search(r'Keywords[:\s]+(.*?)(?:\n|<|Introduction|Abstract)', text, re.IGNORECASE)
if kw_section and not keywords:
keywords = kw_section.group(1).strip().rstrip(".")
# Abstract
abstract = ""
# Try meta description first
abs_match = re.search(r'<meta\s+name="description"\s+content="([^"]+)"', html)
if abs_match:
abstract = abs_match.group(1)
# Try to find abstract section in the page
if not abstract:
abs_match = re.search(r'Abstract[:\s]+(.*?)(?:(?:Introduction|Keywords|1\.\s))', text, re.DOTALL | re.IGNORECASE)
if abs_match:
abstract = abs_match.group(1).strip()[:2000]
if not abstract:
# Try dc.description
abs_match = re.search(r'<meta\s+name="DC\.Description"\s+content="([^"]+)"', html)
if abs_match:
abstract = abs_match.group(1)
# Article type
article_type = "article"
if "commentary" in title.lower():
article_type = "commentary"
elif "response" in title.lower() and "comment" in title.lower():
article_type = "response"
elif "case study" in title.lower() or "case " in title.lower():
article_type = "case_study"
# Determine category from keywords/title
category = classify_article(title, keywords)
result = {
**art,
"title_en": title,
"authors": authors,
"doi": doi,
"keywords": keywords,
"abstract_en": abstract[:3000] if abstract else "",
"article_type": article_type,
"category_en": category,
}
return result
def classify_article(title, keywords):
"""Classify article into therapy approach category."""
text = (title + " " + keywords).lower()
categories = {
"CBT": ["cognitive behavio", "cbt", "cognitive-behavio", "cognitive therapy"],
"DBT": ["dialectical behavio", "dbt"],
"Psychodynamic": ["psychodynamic", "dynamic therapy", "psychoanalytic", "psychoanalysis"],
"Humanistic": ["humanistic", "client-centered", "person-centered", "rogerian"],
"AEDP": ["aedp", "accelerated experiential dynamic"],
"ACT": ["acceptance and commitment", "act ", "act therapy"],
"Schema Therapy": ["schema therapy"],
"EMDR": ["emdr", "eye movement desensitization"],
"Narrative Therapy": ["narrative therapy"],
"Exposure Therapy": ["exposure therapy", "exposure and response prevention", "erp"],
"Hypnosis": ["hypnosis", "hypnotherapy", "hypnotic"],
"Integrative": ["integrative", "eclectic", "multimodal"],
"EFT": ["emotionally focused", "eft"],
"Mindfulness": ["mindfulness", "mbct", "mbsr"],
"SFBT": ["solution-focused", "solution focused", "sfbt"],
}
for cat, terms in categories.items():
if any(t in text for t in terms):
return cat
return "Other"
# ── Korean translation mapping ────────────────────────────────────────
THERAPY_KO = {
"CBT": "인지행동치료",
"DBT": "변증법적행동치료",
"Psychodynamic": "정신역동치료",
"Humanistic": "인간중심치료",
"AEDP": "가속적 경험적 역동치료",
"ACT": "수용전념치료",
"Schema Therapy": "도식치료",
"EMDR": "안구운동 민감소실 재처리",
"Narrative Therapy": "내러티브치료",
"Exposure Therapy": "노출치료",
"Hypnosis": "최면치료",
"Integrative": "통합치료",
"EFT": "정서초점치료",
"Mindfulness": "마음챙김치료",
"SFBT": "해결중심단기치료",
"Other": "기타",
}
CATEGORY_KO = {
"CBT": "인지행동치료",
"DBT": "변증법적행동치료",
"Psychodynamic": "정신역동치료",
"Humanistic": "인간중심치료",
"AEDP": "가속적경험적역동치료",
"ACT": "수용전념치료",
"Schema Therapy": "도식치료",
"EMDR": "안구운동민감소실재처리",
"Narrative Therapy": "내러티브치료",
"Exposure Therapy": "노출치료",
"Hypnosis": "최면치료",
"Integrative": "통합치료",
"EFT": "정서초점치료",
"Mindfulness": "마음챙김치료",
"SFBT": "해결중심단기치료",
"Other": "기타",
}
# ── Build database ────────────────────────────────────────────────────
def create_db():
conn = sqlite3.connect(DB_PATH)
c = conn.cursor()
c.execute("DROP TABLE IF EXISTS cases")
c.execute("""CREATE TABLE cases (
id INTEGER PRIMARY KEY AUTOINCREMENT,
article_id INTEGER,
issue_id INTEGER,
vol INTEGER,
num INTEGER,
year INTEGER,
title_en TEXT,
title_ko TEXT,
authors TEXT,
article_type TEXT,
category_en TEXT,
category_ko TEXT,
therapy_approach_en TEXT,
therapy_approach_ko TEXT,
keywords TEXT,
abstract_en TEXT,
abstract_ko TEXT,
doi TEXT,
url TEXT,
source_url TEXT,
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
)""")
conn.commit()
return conn
def save_articles(conn, articles):
c = conn.cursor()
for art in articles:
if not art or not art.get("title_en"):
continue
cat_en = art.get("category_en", "Other")
cat_ko = CATEGORY_KO.get(cat_en, "기타")
approach_en = cat_en
approach_ko = THERAPY_KO.get(cat_en, "기타")
c.execute("""INSERT INTO cases (
article_id, issue_id, vol, num, year,
title_en, title_ko, authors, article_type,
category_en, category_ko,
therapy_approach_en, therapy_approach_ko,
keywords, abstract_en, abstract_ko,
doi, url, source_url
) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)""",
(
art.get("article_id"),
art.get("issue_id"),
art.get("vol"),
art.get("num"),
art.get("year"),
art.get("title_en", ""),
"", # title_ko - will be translated later
art.get("authors", ""),
art.get("article_type", "article"),
cat_en, cat_ko,
approach_en, approach_ko,
art.get("keywords", ""),
art.get("abstract_en", ""),
"", # abstract_ko - will be translated later
art.get("doi", ""),
art.get("url", ""),
art.get("url", ""),
))
conn.commit()
# ── Main ──────────────────────────────────────────────────────────────
def main():
print("=== PCSP Psychotherapy Case Studies DB Builder ===")
print()
# Step 1: Get all issues
print("[1/4] Fetching issue list...")
issues = get_all_issues()
print(f" Found {len(issues)} issues")
# Step 2: Get article links from each issue
print("[2/4] Extracting article links from issues...")
all_articles = []
for i, issue in enumerate(issues):
print(f" Issue {issue['issue_id']} ({i+1}/{len(issues)})...", end=" ", flush=True)
arts = extract_articles_from_issue(issue["url"], issue["issue_id"])
print(f"{len(arts)} articles")
all_articles.extend(arts)
time.sleep(1) # Be polite
print(f" Total articles found: {len(all_articles)}")
# Step 3: Get details for each article
print("[3/4] Extracting article details...")
detailed_articles = []
for i, art in enumerate(all_articles):
print(f" Article {art['article_id']} ({i+1}/{len(all_articles)})...", end=" ", flush=True)
detail = extract_article_detail(art)
if detail:
title_preview = detail.get("title_en", "")[:60]
print(f"OK - {title_preview}...")
detailed_articles.append(detail)
else:
print("SKIP (no data)")
time.sleep(0.5) # Be polite
# Step 4: Save to database
print(f"[4/4] Saving {len(detailed_articles)} articles to database...")
conn = create_db()
save_articles(conn, detailed_articles)
# Stats
c = conn.cursor()
c.execute("SELECT COUNT(*) FROM cases")
total = c.fetchone()[0]
c.execute("SELECT article_type, COUNT(*) FROM cases GROUP BY article_type")
types = dict(c.fetchall())
c.execute("SELECT category_en, COUNT(*) FROM cases GROUP BY category_en ORDER BY COUNT(*) DESC")
cats = c.fetchall()
print()
print(f"=== Done! {total} articles saved to {DB_PATH} ===")
print(f" Types: {types}")
print(f" Categories:")
for cat, cnt in cats:
print(f" {cat}: {cnt}")
conn.close()
if __name__ == "__main__":
main()