Files
homeclaw/src/tools/pubmed.ts
T
kimandClaude Sonnet 4.6 43942d4b2c Refactor code editor out of main UI + PubMed FTP PDF + login token fix
- Move code editor to standalone page; remove from main index/app mode switcher
- Add diff-active tab highlight and diff bar CSS to styles
- Add markdown table styles for chat messages
- PubMed: try NCBI FTP link before Unpaywall for PDF download
- Login: also persist token to localStorage for cross-tab auth
- Update skills state and config

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-06 15:28:03 +09:00

533 lines
21 KiB
TypeScript

import path from 'path';
import fs from 'fs';
import { execFile } from 'child_process';
import { promisify } from 'util';
import { ToolResult } from '../types.js';
import { getConfig } from '../config/config.js';
import { getWorkspacePath } from '../config/paths.js';
const execFileAsync = promisify(execFile);
const EUTILS_BASE = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils';
const PMC_OA_BASE = 'https://www.ncbi.nlm.nih.gov/pmc/utils/oa/oa.fcgi';
function getPubMedApiKey(): string | undefined {
try {
const cm = getConfig();
const cfg = cm.getConfig();
return cm.resolveSecret(cfg.pubmed?.api_key || cfg.search?.pubmed_api_key) || undefined;
} catch {}
return undefined;
}
function buildUrl(endpoint: string, params: Record<string, string | number>): string {
const apiKey = getPubMedApiKey();
const p = new URLSearchParams();
for (const [k, v] of Object.entries(params)) p.set(k, String(v));
if (apiKey) p.set('api_key', apiKey);
return `${EUTILS_BASE}/${endpoint}?${p.toString()}`;
}
async function fetchJson(url: string): Promise<any> {
const res = await fetch(url, { signal: AbortSignal.timeout(20_000) });
if (!res.ok) throw new Error(`HTTP ${res.status}: ${res.statusText}`);
return res.json();
}
async function fetchText(url: string): Promise<string> {
const res = await fetch(url, { signal: AbortSignal.timeout(30_000) });
if (!res.ok) throw new Error(`HTTP ${res.status}: ${res.statusText}`);
return res.text();
}
function extractXmlField(xml: string, tag: string): string {
const m = xml.match(new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`, 'i'));
return m ? m[1].replace(/<[^>]+>/g, '').trim() : '';
}
function extractAllXmlFields(xml: string, tag: string): string[] {
const re = new RegExp(`<${tag}[^>]*>([\\s\\S]*?)</${tag}>`, 'gi');
const results: string[] = [];
let m: RegExpExecArray | null;
while ((m = re.exec(xml)) !== null) {
const text = m[1].replace(/<[^>]+>/g, '').trim();
if (text) results.push(text);
}
return results;
}
// ── pubmed_search ─────────────────────────────────────────────────────────────
export const pubmedSearchTool = {
name: 'pubmed_search',
description: 'Search PubMed for biomedical literature. Returns article IDs, titles, authors, journal, year, and abstract snippets. Supports boolean operators (AND, OR, NOT) and PubMed field tags (e.g. [tiab], [au], [mesh]).',
schema: {
query: 'PubMed search query string (supports boolean operators and field tags)',
max_results: 'Maximum number of results to return (default: 10, max: 50)',
sort: 'Sort order: "relevance" (default) or "date" (most recent first)',
min_year: 'Optional: filter articles from this year onwards (e.g. 2020)',
},
jsonSchema: {
type: 'object',
properties: {
query: { type: 'string', description: 'PubMed search query' },
max_results: { type: 'number', description: 'Max results (default 10, max 50)' },
sort: { type: 'string', description: '"relevance" or "date"' },
min_year: { type: 'number', description: 'Filter articles from this year onwards' },
},
required: ['query'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const query = String(args?.query || '').trim();
const maxRet = Math.min(Number(args?.max_results) || 10, 50);
const sort = args?.sort === 'date' ? 'pub+date' : 'relevance';
const minYear = args?.min_year ? Number(args.min_year) : null;
if (!query) return { success: false, error: 'query is required' };
const fullQuery = minYear ? `${query} AND ${minYear}:3000[pdat]` : query;
// 1. esearch → get PMIDs
const searchUrl = buildUrl('esearch.fcgi', {
db: 'pubmed', term: fullQuery, retmode: 'json',
retmax: maxRet, sort,
usehistory: 'y',
});
const searchData = await fetchJson(searchUrl);
const esResult = searchData?.esearchresult;
const pmids: string[] = esResult?.idlist || [];
if (!pmids.length) {
return {
success: true,
stdout: `No results found for query: "${query}"`,
data: { query, count: 0, results: [] },
};
}
const totalCount = Number(esResult?.count || 0);
// 2. esummary → get titles, authors, journal, year
const summaryUrl = buildUrl('esummary.fcgi', {
db: 'pubmed', id: pmids.join(','), retmode: 'json',
});
const summaryData = await fetchJson(summaryUrl);
const summaries = summaryData?.result || {};
const results = pmids.map((pmid: string) => {
const s = summaries[pmid] || {};
const authors = (s.authors || []).slice(0, 3).map((a: any) => a.name).join(', ');
const hasMore = (s.authors || []).length > 3;
return {
pmid,
title: s.title || '(no title)',
authors: hasMore ? `${authors} et al.` : authors,
journal: s.fulljournalname || s.source || '',
year: s.pubdate ? s.pubdate.split(' ')[0] : '',
doi: s.elocationid || '',
pmcid: s.pmcrefcount !== undefined ? (s.articleids || []).find((a: any) => a.idtype === 'pmc')?.value || '' : '',
pub_types: (s.pubtype || []).join(', '),
};
});
const lines = [
`PubMed Search: "${query}"`,
`Found ${totalCount} total results (showing ${results.length})\n`,
...results.map((r, i) => [
`[${i + 1}] PMID: ${r.pmid}${r.pmcid ? ` | PMC: ${r.pmcid}` : ''}`,
` Title: ${r.title}`,
` Authors: ${r.authors || '(unknown)'}`,
` Journal: ${r.journal} (${r.year})`,
r.doi ? ` DOI: https://doi.org/${r.doi}` : '',
` PubMed: https://pubmed.ncbi.nlm.nih.gov/${r.pmid}/`,
r.pmcid ? ` Full text (PMC): https://www.ncbi.nlm.nih.gov/pmc/articles/${r.pmcid}/` : '',
r.pub_types ? ` Type: ${r.pub_types}` : '',
].filter(Boolean).join('\n')),
];
return {
success: true,
stdout: lines.join('\n'),
data: { query, total_count: totalCount, results },
};
},
};
// ── pubmed_fetch ──────────────────────────────────────────────────────────────
export const pubmedFetchTool = {
name: 'pubmed_fetch',
description: 'Fetch full metadata and abstract for one or more PubMed articles by PMID. Returns title, authors, abstract, keywords, MeSH terms, DOI, and PMC ID if available.',
schema: {
pmids: 'Comma-separated PubMed IDs (PMIDs), e.g. "12345678,23456789"',
},
jsonSchema: {
type: 'object',
properties: {
pmids: { type: 'string', description: 'Comma-separated PMIDs' },
},
required: ['pmids'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
const pmids = String(args?.pmids || '').replace(/\s+/g, '');
if (!pmids) return { success: false, error: 'pmids is required' };
const url = buildUrl('efetch.fcgi', {
db: 'pubmed', id: pmids, rettype: 'xml', retmode: 'xml',
});
const xml = await fetchText(url);
// Parse each PubmedArticle block
const articleBlocks = xml.match(/<PubmedArticle>[\s\S]*?<\/PubmedArticle>/gi) || [];
const articles = articleBlocks.map((block) => {
const pmid = extractXmlField(block, 'PMID');
const title = extractXmlField(block, 'ArticleTitle');
const journal = extractXmlField(block, 'Title');
const year = extractXmlField(block, 'Year') || extractXmlField(block, 'MedlineDate').slice(0, 4);
// Abstract — may have multiple AbstractText sections
const abstractParts = extractAllXmlFields(block, 'AbstractText');
const abstract = abstractParts.join(' ');
// Authors
const lastNames = extractAllXmlFields(block, 'LastName');
const initials = extractAllXmlFields(block, 'Initials');
const authors = lastNames.map((ln, i) => `${ln} ${initials[i] || ''}`.trim()).join(', ');
// DOI
const doiMatch = block.match(/<ArticleId IdType="doi">([^<]+)<\/ArticleId>/i);
const doi = doiMatch ? doiMatch[1].trim() : '';
// PMC ID
const pmcMatch = block.match(/<ArticleId IdType="pmc">(PMC\d+)<\/ArticleId>/i);
const pmcid = pmcMatch ? pmcMatch[1].trim() : '';
// Keywords
const keywords = extractAllXmlFields(block, 'Keyword');
// MeSH terms
const meshTerms = extractAllXmlFields(block, 'DescriptorName');
return { pmid, title, journal, year, authors, doi, pmcid, abstract, keywords, mesh_terms: meshTerms };
});
if (!articles.length) {
return { success: false, error: `No articles found for PMIDs: ${pmids}` };
}
const lines = articles.flatMap((a) => [
`═══════════════════════════════════════`,
`PMID: ${a.pmid}${a.pmcid ? ` | ${a.pmcid}` : ''}`,
`Title: ${a.title}`,
`Authors: ${a.authors}`,
`Journal: ${a.journal} (${a.year})`,
a.doi ? `DOI: https://doi.org/${a.doi}` : '',
a.pmcid ? `Full text (PMC): https://www.ncbi.nlm.nih.gov/pmc/articles/${a.pmcid}/` : '',
'',
a.abstract ? `Abstract:\n${a.abstract}` : '(No abstract available)',
'',
a.keywords.length ? `Keywords: ${a.keywords.join('; ')}` : '',
a.mesh_terms.length ? `MeSH: ${a.mesh_terms.slice(0, 10).join('; ')}` : '',
].filter((l) => l !== undefined));
return {
success: true,
stdout: lines.join('\n'),
data: { articles },
};
},
};
// ── pubmed_fulltext ───────────────────────────────────────────────────────────
export const pubmedFulltextTool = {
name: 'pubmed_fulltext',
description: 'Download the full text of an open-access PubMed Central (PMC) article. Saves to workspace as text (default) or PDF. Requires the PMC ID (e.g. "PMC1234567"). Only works for open-access articles.',
schema: {
pmcid: 'PMC ID of the article (e.g. "PMC1234567" or just "1234567")',
format: 'Download format: "text" (default, extracts plain text from XML) or "pdf" (downloads the actual PDF file)',
save_path: 'Optional file path within workspace to save the file (default: pubmed/<pmcid>.txt or pubmed/<pmcid>.pdf)',
},
jsonSchema: {
type: 'object',
properties: {
pmcid: { type: 'string', description: 'PMC ID (e.g. PMC1234567)' },
format: { type: 'string', description: '"text" (default) or "pdf"' },
save_path: { type: 'string', description: 'Workspace-relative save path (optional)' },
},
required: ['pmcid'],
additionalProperties: false,
},
execute: async (args: any): Promise<ToolResult> => {
let pmcid = String(args?.pmcid || '').trim().toUpperCase();
if (!pmcid) return { success: false, error: 'pmcid is required' };
if (!pmcid.startsWith('PMC')) pmcid = `PMC${pmcid}`;
// Guard: save_path extension must match format. format="text" writes plaintext/markdown,
// format="pdf" writes binary PDF — mismatched extensions produce broken download buttons.
if (args?.save_path) {
const savePathExt = path.extname(String(args.save_path)).toLowerCase();
const isPdfFormat = args?.format === 'pdf';
if (isPdfFormat && savePathExt && savePathExt !== '.pdf') {
return { success: false, error: `format="pdf" requires save_path ending in .pdf (got "${savePathExt}"). Use a .pdf extension or omit save_path.` };
}
if (!isPdfFormat && savePathExt === '.pdf') {
return { success: false, error: `format="text" writes plain text/markdown — save_path must NOT end in .pdf (got "${args.save_path}"). Use .md or .txt, or set format="pdf" to download the actual PDF binary.` };
}
}
const numericId = pmcid.replace('PMC', '');
// 1. Check PMC Open Access availability via OA API (XML response)
try {
const oaXml = await fetchText(`${PMC_OA_BASE}?id=${pmcid}`);
if (oaXml.includes('idIsNotOpenAccess')) {
// Article is not OA — get PMID and DOI via single PMC esummary call
let altInfo = '';
try {
const apiKey = getPubMedApiKey();
const sumParams = new URLSearchParams({ db: 'pmc', id: numericId, retmode: 'json' });
if (apiKey) sumParams.set('api_key', apiKey);
const summary = await fetchJson(`${EUTILS_BASE}/esummary.fcgi?${sumParams}`);
const ids: any[] = summary?.result?.[numericId]?.articleids || [];
const pmid = ids.find((a: any) => a.idtype === 'pmid')?.value || '';
const doi = ids.find((a: any) => a.idtype === 'doi')?.value || '';
if (pmid) altInfo += `\nPMID: ${pmid}\nPubMed page: https://pubmed.ncbi.nlm.nih.gov/${pmid}/`;
if (doi) altInfo += `\nDOI: https://doi.org/${doi}`;
if (doi) {
try {
const uw = await fetchJson(`https://api.unpaywall.org/v2/${doi}?email=pubmed@smallclaw.local`);
const oaUrl = uw?.best_oa_location?.url_for_pdf || uw?.best_oa_location?.url || '';
if (oaUrl) altInfo += `\nUnpaywall OA link: ${oaUrl}`;
} catch {}
}
} catch {}
return {
success: false,
error: [
`${pmcid} is not Open Access — full text cannot be downloaded.`,
`The abstract is available via pubmed_fetch.`,
altInfo,
].filter(Boolean).join('\n'),
};
}
if (args?.format === 'pdf') {
// Extract FTP PDF link directly from OA API XML (most reliable source)
let ftpPdfUrl = '';
const ftpMatch = oaXml.match(/href="(ftp:\/\/[^"]+\.pdf)"/i);
if (ftpMatch) {
ftpPdfUrl = ftpMatch[1].replace('ftp://ftp.ncbi.nlm.nih.gov/', 'https://ftp.ncbi.nlm.nih.gov/');
}
// PDF mode: get DOI via single PMC esummary call, then Unpaywall
const apiKey = getPubMedApiKey();
const sumParams = new URLSearchParams({ db: 'pmc', id: numericId, retmode: 'json' });
if (apiKey) sumParams.set('api_key', apiKey);
const summary = await fetchJson(`${EUTILS_BASE}/esummary.fcgi?${sumParams}`);
const ids: any[] = summary?.result?.[numericId]?.articleids || [];
const doi = ids.find((a: any) => a.idtype === 'doi')?.value || '';
// Query Unpaywall for PDF URL
let pdfUrl = '';
if (doi) {
try {
const uw = await fetchJson(`https://api.unpaywall.org/v2/${doi}?email=pubmed@smallclaw.local`);
pdfUrl = (uw?.oa_locations || [])
.map((l: any) => l.url_for_pdf)
.find((u: any) => u) || '';
} catch {}
}
// Download PDF — try NCBI FTP first, then Unpaywall, then Europe PMC
const workspaceDir = getWorkspacePath(args);
const pdfDir = path.join(workspaceDir, 'pubmed');
fs.mkdirSync(pdfDir, { recursive: true });
const pdfDest = args?.save_path
? path.join(workspaceDir, args.save_path)
: path.join(pdfDir, `${pmcid}.pdf`);
const europePmcUrl = `https://europepmc.org/api/getPdf?pmcid=${pmcid}`;
const urlsToTry = [ftpPdfUrl, pdfUrl, europePmcUrl].filter(Boolean);
let pdfBuf: Buffer | null = null;
let usedUrl = '';
for (const tryUrl of urlsToTry) {
try {
const res = await fetch(tryUrl, {
signal: AbortSignal.timeout(60_000),
headers: {
'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36',
'Accept': 'application/pdf,*/*',
'Accept-Language': 'en-US,en;q=0.9',
'Referer': 'https://europepmc.org/',
},
});
if (!res.ok) continue;
const buf = Buffer.from(await res.arrayBuffer());
// Reject HTML responses (PMC sometimes returns HTML instead of PDF)
if (!buf.slice(0, 5).toString('ascii').startsWith('%PDF')) {
console.log(`[pubmed] ${tryUrl} returned non-PDF content (HTML?), trying next`);
continue;
}
pdfBuf = buf;
usedUrl = tryUrl;
break;
} catch {}
}
if (!pdfBuf) {
return {
success: false,
error: [
`PDF download failed for ${pmcid}: all sources returned non-PDF content.`,
`Tried: ${urlsToTry.join(', ')}`,
`Try format "text" to get the plain-text version instead.`,
].join('\n'),
};
}
fs.writeFileSync(pdfDest, pdfBuf);
const sizeKb = (pdfBuf.length / 1024).toFixed(1);
const relPath = path.relative(workspaceDir, pdfDest);
return {
success: true,
stdout: [
`PDF downloaded: ${pmcid}`,
doi ? `DOI: ${doi}` : '',
`Saved to: ${relPath} (${sizeKb} KB)`,
`File link: /api/files/${relPath}`,
].filter(Boolean).join('\n'),
data: { pmcid, doi, pdf_url: usedUrl, save_path: pdfDest, size_kb: Number(sizeKb) },
};
}
} catch (oaErr: any) {
// OA check failed — continue to efetch attempt
}
// 2. Fetch full text via efetch (PMC XML → plain text extraction)
const apiKey = getPubMedApiKey();
const fetchParams = new URLSearchParams({
db: 'pmc', id: numericId, rettype: 'full', retmode: 'xml',
});
if (apiKey) fetchParams.set('api_key', apiKey);
const fulltextUrl = `${EUTILS_BASE}/efetch.fcgi?${fetchParams.toString()}`;
let xml: string;
try {
xml = await fetchText(fulltextUrl);
} catch (err: any) {
return { success: false, error: `Failed to fetch fulltext for ${pmcid}: ${err.message}` };
}
if (xml.includes('<error>') || xml.includes('Error occurred')) {
const errMsg = extractXmlField(xml, 'error') || 'Unknown error from PMC API';
return {
success: false,
error: `PMC returned an error for ${pmcid}: ${errMsg}. Article may not be open access.`,
};
}
// Extract structured sections from PMC XML
// Limit author extraction to <front> to avoid pulling reference authors
const frontXml = xml.match(/<front[\s\S]*?<\/front>/i)?.[0] || xml;
const title = extractXmlField(frontXml, 'article-title');
const journal = extractXmlField(frontXml, 'journal-title');
const year = extractXmlField(frontXml, 'year');
// Authors — extract only from article metadata (contrib-group inside front)
const contribGroupXml = frontXml.match(/<contrib-group[\s\S]*?<\/contrib-group>/i)?.[0] || frontXml;
const surnameList = extractAllXmlFields(contribGroupXml, 'surname');
const givenList = extractAllXmlFields(contribGroupXml, 'given-names');
const MAX_AUTHORS = 10;
const authorList = surnameList.slice(0, MAX_AUTHORS).map((s, i) => `${s} ${givenList[i] || ''}`.trim());
const authors = surnameList.length > MAX_AUTHORS
? `${authorList.join(', ')} et al. (${surnameList.length} authors total)`
: authorList.join(', ');
// Abstract
const abstractXml = xml.match(/<abstract[\s\S]*?<\/abstract>/i)?.[0] || '';
const abstractText = abstractXml.replace(/<[^>]+>/g, ' ').replace(/\s+/g, ' ').trim();
// Body sections
const bodyXml = xml.match(/<body[\s\S]*?<\/body>/i)?.[0] || '';
const bodyText = bodyXml
.replace(/<title>/gi, '\n\n## ')
.replace(/<\/title>/gi, '\n')
.replace(/<p>/gi, '\n')
.replace(/<\/p>/gi, '')
.replace(/<[^>]+>/g, '')
.replace(/\s*\n\s*/g, '\n')
.replace(/\n{3,}/g, '\n\n')
.trim();
if (!bodyText) {
return {
success: false,
error: [
`${pmcid} is indexed in PMC but its full-text XML body is not available via the API.`,
`Abstract is available — use pubmed_fetch to retrieve it.`,
`PMC page: https://www.ncbi.nlm.nih.gov/pmc/articles/${pmcid}/`,
].join('\n'),
};
}
const fullContent = [
`# ${title}`,
`${authors}`,
`${journal} (${year})`,
`PMC ID: ${pmcid}`,
'',
`## Abstract`,
abstractText,
'',
`## Full Text`,
bodyText,
].join('\n');
// Save to workspace (prefer per-user path injected by v2 executeTool)
const config = getConfig().getConfig();
const workspaceDir = getWorkspacePath(args);
const savePath = args?.save_path
? path.join(workspaceDir, args.save_path)
: path.join(workspaceDir, 'pubmed', `${pmcid}.txt`);
fs.mkdirSync(path.dirname(savePath), { recursive: true });
fs.writeFileSync(savePath, fullContent, 'utf-8');
const relPath = path.relative(workspaceDir, savePath);
const charCount = fullContent.length;
const wordCount = fullContent.split(/\s+/).length;
return {
success: true,
stdout: [
`Full text downloaded: ${pmcid}`,
`Title: ${title}`,
`Authors: ${authors}`,
`Journal: ${journal} (${year})`,
``,
`Saved to: ${relPath} (${wordCount.toLocaleString()} words, ${charCount.toLocaleString()} chars)`,
``,
`Abstract preview:`,
abstractText.slice(0, 500) + (abstractText.length > 500 ? '...' : ''),
].join('\n'),
data: {
pmcid,
title,
authors,
journal,
year,
save_path: savePath,
word_count: wordCount,
char_count: charCount,
},
};
},
};