From 72c0401ea0f2a38559b97b8015c2414168f5cc7f Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 29 Dec 2025 04:44:48 +0000 Subject: [PATCH 1/2] Add enhanced fulltext retrieval strategies Implement multiple strategies for pulling fulltext from publications: - BioC XML API (NCBI BioNLP) for cleanest structured text - Europe PMC for wider open access coverage - Unpaywall API for finding OA versions via DOI - Identifier conversion utilities (DOI <-> PMID <-> PMCID) Integrate strategies into PMIDSource and DOISource with fallback: 1. BioC XML (fastest, cleanest) 2. Europe PMC (broader coverage) 3. Traditional PMC XML/HTML Based on patterns from aurelian and artl-mcp repositories. --- .../etl/fulltext_strategies.py | 740 ++++++++++++++++++ .../etl/sources/doi.py | 75 +- .../etl/sources/pmid.py | 31 +- tests/test_fulltext_strategies.py | 345 ++++++++ 4 files changed, 1188 insertions(+), 3 deletions(-) create mode 100644 src/linkml_reference_validator/etl/fulltext_strategies.py create mode 100644 tests/test_fulltext_strategies.py diff --git a/src/linkml_reference_validator/etl/fulltext_strategies.py b/src/linkml_reference_validator/etl/fulltext_strategies.py new file mode 100644 index 0000000..d42711b --- /dev/null +++ b/src/linkml_reference_validator/etl/fulltext_strategies.py @@ -0,0 +1,740 @@ +"""Enhanced fulltext retrieval strategies. + +This module provides multiple strategies for fetching fulltext content +from scientific publications, including: +- BioC XML API (NCBI BioNLP) +- Europe PMC +- Unpaywall (open access papers via DOI) +- Identifier conversion utilities (DOI <-> PMID <-> PMCID) + +Examples: + >>> from linkml_reference_validator.etl.fulltext_strategies import FulltextFetcher + >>> fetcher = FulltextFetcher(email="user@example.com") + >>> # result = fetcher.fetch_fulltext_for_pmid("12345678") + >>> # result = fetcher.fetch_fulltext_for_doi("10.1234/example") +""" + +import logging +import time +from abc import ABC, abstractmethod +from dataclasses import dataclass, field +from typing import Optional + +import requests # type: ignore +from bs4 import BeautifulSoup # type: ignore + +logger = logging.getLogger(__name__) + +# API URLs +BIOC_URL = "https://www.ncbi.nlm.nih.gov/research/bionlp/RESTful/pmcoa.cgi/BioC_xml/{pmid}/ascii" +EUROPEPMC_SEARCH_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest/search" +EUROPEPMC_FULLTEXT_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest/{source}/{id}/fullTextXML" +UNPAYWALL_URL = "https://api.unpaywall.org/v2/{doi}" +NCBI_IDCONV_URL = "https://www.ncbi.nlm.nih.gov/pmc/utils/idconv/v1.0/" +NCBI_ESUMMARY_URL = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esummary.fcgi" + + +@dataclass +class FulltextResult: + """Result from a fulltext retrieval attempt. + + Attributes: + content: The fulltext content if found + source: The source that provided the content (bioc, europepmc, unpaywall) + content_type: Type of content (full_text_bioc, full_text_europepmc, etc.) + success: Whether the retrieval was successful + error_message: Error message if retrieval failed + metadata: Additional metadata from the source + + Examples: + >>> result = FulltextResult( + ... content="Full article text here.", + ... source="bioc", + ... content_type="full_text_bioc", + ... success=True, + ... ) + >>> result.success + True + >>> result.source + 'bioc' + """ + + content: Optional[str] = None + source: str = "unknown" + content_type: str = "unknown" + success: bool = True + error_message: Optional[str] = None + metadata: dict = field(default_factory=dict) + + +class FulltextStrategy(ABC): + """Abstract base class for fulltext retrieval strategies. + + Subclasses must implement the fetch method to retrieve fulltext + content from their respective sources. + + Examples: + >>> class MyStrategy(FulltextStrategy): + ... def fetch(self, identifier): + ... return FulltextResult(content="test", source="my_source") + >>> strategy = MyStrategy() + >>> strategy.name + 'MyStrategy' + """ + + @property + def name(self) -> str: + """Return the strategy name.""" + return self.__class__.__name__ + + @abstractmethod + def fetch(self, identifier: str, rate_limit_delay: float = 0.5) -> FulltextResult: + """Fetch fulltext for the given identifier. + + Args: + identifier: The identifier to fetch (PMID, DOI, etc.) + rate_limit_delay: Delay between API requests + + Returns: + FulltextResult with content if successful + """ + ... + + +class BioCStrategy(FulltextStrategy): + """Fetch fulltext using NCBI BioC XML API. + + The BioC API provides structured fulltext for articles in the + PubMed Central Open Access subset. This is often the most reliable + source for clean fulltext. + + Examples: + >>> strategy = BioCStrategy() + >>> url = strategy._build_url("12345678") + >>> "12345678" in url + True + >>> "BioC_xml" in url + True + """ + + def _build_url(self, pmid: str) -> str: + """Build the BioC API URL for a PMID. + + Args: + pmid: The PubMed ID + + Returns: + The API URL + + Examples: + >>> strategy = BioCStrategy() + >>> url = strategy._build_url("12345") + >>> "12345" in url and "BioC_xml" in url + True + """ + return BIOC_URL.format(pmid=pmid) + + def fetch(self, identifier: str, rate_limit_delay: float = 0.5) -> FulltextResult: + """Fetch fulltext from BioC API. + + Args: + identifier: PMID to fetch + rate_limit_delay: Delay before request + + Returns: + FulltextResult with fulltext if available + """ + pmid = identifier.strip() + if ":" in pmid: + pmid = pmid.split(":")[-1] + + time.sleep(rate_limit_delay) + url = self._build_url(pmid) + + try: + response = requests.get(url, timeout=30) + except requests.RequestException as e: + logger.warning(f"BioC request failed for PMID:{pmid}: {e}") + return FulltextResult( + success=False, + source="bioc", + content_type="unavailable", + error_message=str(e), + ) + + if response.status_code != 200: + logger.debug(f"BioC not available for PMID:{pmid} (status {response.status_code})") + return FulltextResult( + success=False, + source="bioc", + content_type="unavailable", + error_message=f"HTTP {response.status_code}", + ) + + # Parse BioC XML + soup = BeautifulSoup(response.text, "xml") + text_sections = [text_tag.get_text() for text_tag in soup.find_all("text")] + + if not text_sections: + return FulltextResult( + success=False, + source="bioc", + content_type="unavailable", + error_message="No text sections found in BioC response", + ) + + full_text = "\n\n".join(text_sections).strip() + + if len(full_text) < 500: + return FulltextResult( + success=False, + source="bioc", + content_type="unavailable", + error_message="BioC response too short", + ) + + return FulltextResult( + content=full_text, + source="bioc", + content_type="full_text_bioc", + success=True, + ) + + +class EuropePMCStrategy(FulltextStrategy): + """Fetch fulltext from Europe PMC. + + Europe PMC provides fulltext for many open access articles, + including some not available in the US PubMed Central. + + Examples: + >>> strategy = EuropePMCStrategy() + >>> url = strategy._build_search_url("12345678") + >>> "europepmc" in url + True + """ + + def _build_search_url(self, pmid: str) -> str: + """Build Europe PMC search URL for a PMID. + + Args: + pmid: The PubMed ID + + Returns: + The search API URL + """ + return f"{EUROPEPMC_SEARCH_URL}?query=ext_id:{pmid}&format=json" + + def _build_fulltext_url(self, pmcid: str) -> str: + """Build Europe PMC fulltext URL for a PMCID. + + Args: + pmcid: The PMC ID (with or without PMC prefix) + + Returns: + The fulltext API URL + """ + # Strip PMC prefix if present + pmc_id = pmcid.replace("PMC", "") + return EUROPEPMC_FULLTEXT_URL.format(source="PMC", id=pmc_id) + + def fetch(self, identifier: str, rate_limit_delay: float = 0.5) -> FulltextResult: + """Fetch fulltext from Europe PMC. + + Args: + identifier: PMID to fetch + rate_limit_delay: Delay before request + + Returns: + FulltextResult with fulltext if available + """ + pmid = identifier.strip() + if ":" in pmid: + pmid = pmid.split(":")[-1] + + time.sleep(rate_limit_delay) + + # First, search for the article to get PMCID + search_url = self._build_search_url(pmid) + try: + response = requests.get(search_url, timeout=30) + except requests.RequestException as e: + logger.warning(f"Europe PMC search failed for PMID:{pmid}: {e}") + return FulltextResult( + success=False, + source="europepmc", + content_type="unavailable", + error_message=str(e), + ) + + if response.status_code != 200: + return FulltextResult( + success=False, + source="europepmc", + content_type="unavailable", + error_message=f"Search HTTP {response.status_code}", + ) + + data = response.json() + results = data.get("resultList", {}).get("result", []) + + if not results: + return FulltextResult( + success=False, + source="europepmc", + content_type="unavailable", + error_message="Article not found in Europe PMC", + ) + + article = results[0] + pmcid = article.get("pmcid") + is_oa = article.get("isOpenAccess") == "Y" + + if not pmcid or not is_oa: + return FulltextResult( + success=False, + source="europepmc", + content_type="unavailable", + error_message="Article not open access or no PMC ID", + ) + + # Fetch the fulltext XML + time.sleep(rate_limit_delay) + fulltext_url = self._build_fulltext_url(pmcid) + + try: + ft_response = requests.get(fulltext_url, timeout=30) + except requests.RequestException as e: + logger.warning(f"Europe PMC fulltext fetch failed: {e}") + return FulltextResult( + success=False, + source="europepmc", + content_type="unavailable", + error_message=str(e), + ) + + if ft_response.status_code != 200: + return FulltextResult( + success=False, + source="europepmc", + content_type="unavailable", + error_message=f"Fulltext HTTP {ft_response.status_code}", + ) + + # Parse the fulltext XML + soup = BeautifulSoup(ft_response.text, "xml") + body = soup.find("body") + + if body: + paragraphs = body.find_all("p") + if paragraphs: + text = "\n\n".join(p.get_text() for p in paragraphs) + if len(text) > 500: + return FulltextResult( + content=text, + source="europepmc", + content_type="full_text_europepmc", + success=True, + metadata={"pmcid": pmcid}, + ) + + return FulltextResult( + success=False, + source="europepmc", + content_type="unavailable", + error_message="Could not extract text from Europe PMC XML", + ) + + +class UnpaywallStrategy(FulltextStrategy): + """Fetch open access papers via Unpaywall API. + + Unpaywall provides access to legal open access versions of papers. + Requires a valid email address for API access. + + Examples: + >>> strategy = UnpaywallStrategy(email="test@example.com") + >>> url = strategy._build_url("10.1234/example") + >>> "api.unpaywall.org" in url + True + """ + + def __init__(self, email: str = "linkml-reference-validator@example.com"): + """Initialize with email for API access. + + Args: + email: Email address for Unpaywall API + """ + self.email = email + + def _build_url(self, doi: str) -> str: + """Build Unpaywall API URL for a DOI. + + Args: + doi: The DOI + + Returns: + The API URL + + Examples: + >>> strategy = UnpaywallStrategy(email="test@example.com") + >>> url = strategy._build_url("10.1234/test") + >>> "10.1234/test" in url and "test@example.com" in url + True + """ + return f"{UNPAYWALL_URL.format(doi=doi)}?email={self.email}" + + def fetch(self, identifier: str, rate_limit_delay: float = 0.5) -> FulltextResult: + """Fetch open access info from Unpaywall. + + Note: This does not fetch the actual fulltext, but provides + information about where to find open access versions. + + Args: + identifier: DOI to look up + rate_limit_delay: Delay before request + + Returns: + FulltextResult with OA location info + """ + doi = identifier.strip() + if doi.lower().startswith("doi:"): + doi = doi[4:] + + time.sleep(rate_limit_delay) + url = self._build_url(doi) + + try: + response = requests.get(url, timeout=30) + except requests.RequestException as e: + logger.warning(f"Unpaywall request failed for DOI:{doi}: {e}") + return FulltextResult( + success=False, + source="unpaywall", + content_type="unavailable", + error_message=str(e), + ) + + if response.status_code != 200: + return FulltextResult( + success=False, + source="unpaywall", + content_type="unavailable", + error_message=f"HTTP {response.status_code}", + ) + + data = response.json() + is_oa = data.get("is_oa", False) + + if not is_oa: + return FulltextResult( + success=False, + source="unpaywall", + content_type="unavailable", + error_message="Article is not open access", + ) + + best_location = data.get("best_oa_location", {}) or {} + pdf_url = best_location.get("url_for_pdf") or best_location.get("url") + oa_locations = data.get("oa_locations", []) + + # Look for PMC source in OA locations + pmcid = None + for loc in oa_locations: + pmh_id = loc.get("pmh_id", "") + if "pubmedcentral" in pmh_id.lower(): + # Extract PMC ID from pmh_id like "oai:pubmedcentral.nih.gov:123456" + parts = pmh_id.split(":") + if len(parts) >= 3: + pmcid = f"PMC{parts[-1]}" + break + + return FulltextResult( + content=None, # Unpaywall doesn't provide content directly + source="unpaywall", + content_type="oa_location", + success=True, + metadata={ + "is_oa": True, + "pdf_url": pdf_url, + "pmcid": pmcid, + "license": best_location.get("license"), + "version": best_location.get("version"), + }, + ) + + +class IdentifierConverter: + """Convert between DOI, PMID, and PMCID identifiers. + + Uses NCBI ID Converter and E-utilities APIs. + + Examples: + >>> converter = IdentifierConverter() + >>> # pmid = converter.doi_to_pmid("10.1234/example") + >>> # doi = converter.pmid_to_doi("12345678") + """ + + def __init__(self, email: str = "linkml-reference-validator@example.com"): + """Initialize with email for NCBI API. + + Args: + email: Email for NCBI API access + """ + self.email = email + + def doi_to_pmid(self, doi: str, rate_limit_delay: float = 0.5) -> Optional[str]: + """Convert DOI to PMID. + + Args: + doi: The DOI to convert + rate_limit_delay: Delay before request + + Returns: + PMID if found, None otherwise + + Examples: + >>> converter = IdentifierConverter() + >>> # This would make an API call in real usage + >>> # pmid = converter.doi_to_pmid("10.1234/example") + """ + if doi.lower().startswith("doi:"): + doi = doi[4:] + + time.sleep(rate_limit_delay) + url = f"{NCBI_IDCONV_URL}?ids={doi}&format=json" + + try: + response = requests.get(url, timeout=30) + except requests.RequestException as e: + logger.warning(f"ID conversion failed for DOI:{doi}: {e}") + return None + + if response.status_code != 200: + return None + + data = response.json() + records = data.get("records", []) + + if records: + return records[0].get("pmid") + return None + + def pmid_to_doi(self, pmid: str, rate_limit_delay: float = 0.5) -> Optional[str]: + """Convert PMID to DOI. + + Args: + pmid: The PMID to convert + rate_limit_delay: Delay before request + + Returns: + DOI if found, None otherwise + """ + if ":" in pmid: + pmid = pmid.split(":")[-1] + + time.sleep(rate_limit_delay) + url = f"{NCBI_ESUMMARY_URL}?db=pubmed&id={pmid}&retmode=json" + + try: + response = requests.get(url, timeout=30) + except requests.RequestException as e: + logger.warning(f"PMID to DOI conversion failed for {pmid}: {e}") + return None + + if response.status_code != 200: + return None + + data = response.json() + + try: + article_info = data["result"][str(pmid)] + for aid in article_info.get("articleids", []): + if aid.get("idtype") == "doi": + return aid.get("value") + # Check elocationid as fallback + elocationid = article_info.get("elocationid", "") + if elocationid.startswith("10."): + return elocationid + except KeyError: + pass + + return None + + def pmid_to_pmcid(self, pmid: str, rate_limit_delay: float = 0.5) -> Optional[str]: + """Convert PMID to PMCID. + + Args: + pmid: The PMID to convert + rate_limit_delay: Delay before request + + Returns: + PMCID if found, None otherwise + """ + if ":" in pmid: + pmid = pmid.split(":")[-1] + + time.sleep(rate_limit_delay) + url = f"{NCBI_IDCONV_URL}?ids={pmid}&format=json" + + try: + response = requests.get(url, timeout=30) + except requests.RequestException as e: + logger.warning(f"PMID to PMCID conversion failed for {pmid}: {e}") + return None + + if response.status_code != 200: + return None + + data = response.json() + records = data.get("records", []) + + if records: + return records[0].get("pmcid") + return None + + def pmcid_to_pmid(self, pmcid: str, rate_limit_delay: float = 0.5) -> Optional[str]: + """Convert PMCID to PMID. + + Args: + pmcid: The PMCID to convert (with or without PMC prefix) + rate_limit_delay: Delay before request + + Returns: + PMID if found, None otherwise + """ + # Strip PMC prefix if present + pmc_id = pmcid.replace("PMC", "").replace("pmc", "") + if ":" in pmc_id: + pmc_id = pmc_id.split(":")[-1] + + time.sleep(rate_limit_delay) + url = f"{NCBI_ESUMMARY_URL}?db=pmc&id={pmc_id}&retmode=json" + + try: + response = requests.get(url, timeout=30) + except requests.RequestException as e: + logger.warning(f"PMCID to PMID conversion failed for {pmcid}: {e}") + return None + + if response.status_code != 200: + return None + + data = response.json() + + try: + uids = data["result"]["uids"] + if uids: + uid = uids[0] + article_ids = data["result"][uid].get("articleids", []) + for item in article_ids: + if item.get("idtype") == "pmid": + return item.get("value") + except KeyError: + pass + + return None + + +class FulltextFetcher: + """Orchestrates multiple fulltext strategies with fallback. + + Tries strategies in order until one succeeds. + + Examples: + >>> fetcher = FulltextFetcher(email="test@example.com") + >>> # result = fetcher.fetch_fulltext_for_pmid("12345678") + >>> # result = fetcher.fetch_fulltext_for_doi("10.1234/example") + """ + + def __init__( + self, + email: str = "linkml-reference-validator@example.com", + rate_limit_delay: float = 0.5, + ): + """Initialize the fetcher with strategies. + + Args: + email: Email for API access + rate_limit_delay: Delay between requests + """ + self.email = email + self.rate_limit_delay = rate_limit_delay + self.converter = IdentifierConverter(email=email) + + # Strategies in priority order for PMIDs + self.pmid_strategies: list[FulltextStrategy] = [ + BioCStrategy(), + EuropePMCStrategy(), + ] + + self.unpaywall = UnpaywallStrategy(email=email) + + def fetch_fulltext_for_pmid(self, pmid: str) -> FulltextResult: + """Fetch fulltext for a PMID using all strategies. + + Tries strategies in order: + 1. BioC XML API + 2. Europe PMC + + Args: + pmid: The PubMed ID + + Returns: + FulltextResult from the first successful strategy + """ + for strategy in self.pmid_strategies: + result = strategy.fetch(pmid, self.rate_limit_delay) + if result.success and result.content: + logger.info(f"Fetched fulltext for PMID:{pmid} via {strategy.name}") + return result + + # All strategies failed + return FulltextResult( + success=False, + source="none", + content_type="unavailable", + error_message="No fulltext available from any source", + ) + + def fetch_fulltext_for_doi(self, doi: str) -> FulltextResult: + """Fetch fulltext for a DOI. + + Tries to convert DOI to PMID first, then uses PMID strategies. + Falls back to Unpaywall for OA location. + + Args: + doi: The DOI + + Returns: + FulltextResult + """ + # Try to convert DOI to PMID + pmid = self.converter.doi_to_pmid(doi, self.rate_limit_delay) + + if pmid: + result = self.fetch_fulltext_for_pmid(pmid) + if result.success: + return result + + # Try Unpaywall for OA location + unpaywall_result = self.unpaywall.fetch(doi, self.rate_limit_delay) + if unpaywall_result.success: + # If Unpaywall found a PMCID, try to fetch that + pmcid = unpaywall_result.metadata.get("pmcid") + if pmcid: + pmid_from_pmc = self.converter.pmcid_to_pmid(pmcid, self.rate_limit_delay) + if pmid_from_pmc: + result = self.fetch_fulltext_for_pmid(pmid_from_pmc) + if result.success: + return result + + # Return the Unpaywall result with OA location info + return unpaywall_result + + return FulltextResult( + success=False, + source="none", + content_type="unavailable", + error_message="No fulltext available from any source", + ) diff --git a/src/linkml_reference_validator/etl/sources/doi.py b/src/linkml_reference_validator/etl/sources/doi.py index 68e795a..db39c93 100644 --- a/src/linkml_reference_validator/etl/sources/doi.py +++ b/src/linkml_reference_validator/etl/sources/doi.py @@ -2,6 +2,9 @@ Fetches publication metadata from Crossref API, with fallback to DataCite for DOIs not found in Crossref (e.g., Zenodo, Figshare, Dryad). +Also attempts fulltext retrieval via: +- Unpaywall (open access papers) +- Identifier conversion to PMID for PMC access Examples: >>> from linkml_reference_validator.etl.sources.doi import DOISource @@ -28,6 +31,11 @@ extract_extra_fields, format_extra_fields_for_content, ) +from linkml_reference_validator.etl.fulltext_strategies import ( + UnpaywallStrategy, + IdentifierConverter, + FulltextFetcher, +) logger = logging.getLogger(__name__) @@ -63,6 +71,10 @@ def fetch( ) -> Optional[ReferenceContent]: """Fetch a publication by DOI from Crossref, falling back to DataCite. + Also attempts fulltext retrieval via: + 1. Convert DOI to PMID, then use PMC strategies + 2. Unpaywall for open access versions + Args: identifier: DOI (without prefix) config: Configuration including rate limiting and email @@ -142,11 +154,20 @@ def _fetch_from_crossref( is_preprint = self._crossref_preprint_status(message) + # Try to get fulltext via enhanced strategies + fulltext, content_type = self._fetch_fulltext(doi, config) + + if fulltext: + content = f"{abstract}\n\n{fulltext}" if abstract else fulltext + else: + content = abstract if abstract else None + content_type = "abstract_only" if abstract else "unavailable" + return ReferenceContent( reference_id=f"DOI:{doi}", title=title, - content=abstract if abstract else None, - content_type="abstract_only" if abstract else "unavailable", + content=content, + content_type=content_type, authors=authors, journal=journal, year=year, @@ -432,6 +453,56 @@ def _parse_datacite_subjects(self, subjects: list) -> Optional[list[str]]: result.append(subj) return result if result else None + def _fetch_fulltext( + self, doi: str, config: ReferenceValidationConfig + ) -> tuple[Optional[str], str]: + """Attempt to fetch fulltext for a DOI. + + Tries: + 1. Convert DOI to PMID and use FulltextFetcher + 2. Check Unpaywall for OA location info + + Args: + doi: The DOI + config: Configuration for rate limiting + + Returns: + Tuple of (fulltext, content_type) + """ + converter = IdentifierConverter(email=config.email) + + # Try to convert DOI to PMID + pmid = converter.doi_to_pmid(doi, config.rate_limit_delay) + if pmid: + fetcher = FulltextFetcher( + email=config.email, + rate_limit_delay=config.rate_limit_delay, + ) + result = fetcher.fetch_fulltext_for_pmid(pmid) + if result.success and result.content: + logger.info(f"Fetched fulltext for DOI:{doi} via PMID:{pmid}") + return result.content, result.content_type + + # Try Unpaywall - it may provide PMCID even without direct PMID + unpaywall = UnpaywallStrategy(email=config.email) + oa_result = unpaywall.fetch(doi, config.rate_limit_delay) + if oa_result.success: + pmcid = oa_result.metadata.get("pmcid") + if pmcid: + # Try to get PMID from PMCID + pmid_from_pmc = converter.pmcid_to_pmid(pmcid, config.rate_limit_delay) + if pmid_from_pmc: + fetcher = FulltextFetcher( + email=config.email, + rate_limit_delay=config.rate_limit_delay, + ) + result = fetcher.fetch_fulltext_for_pmid(pmid_from_pmc) + if result.success and result.content: + logger.info(f"Fetched fulltext for DOI:{doi} via Unpaywall PMCID") + return result.content, result.content_type + + return None, "abstract_only" + def _parse_crossref_authors(self, authors: list) -> list[str]: """Parse author list from Crossref response. diff --git a/src/linkml_reference_validator/etl/sources/pmid.py b/src/linkml_reference_validator/etl/sources/pmid.py index fff79e4..f7eee71 100644 --- a/src/linkml_reference_validator/etl/sources/pmid.py +++ b/src/linkml_reference_validator/etl/sources/pmid.py @@ -1,6 +1,10 @@ """PMID (PubMed ID) reference source. Fetches publication content from PubMed/NCBI using the Entrez API. +Supports multiple fulltext retrieval strategies: +- BioC XML API (NCBI BioNLP) +- Europe PMC +- PMC XML/HTML (traditional approach) Examples: >>> from linkml_reference_validator.etl.sources.pmid import PMIDSource @@ -33,6 +37,10 @@ extract_extra_fields, format_extra_fields_for_content, ) +from linkml_reference_validator.etl.fulltext_strategies import ( + BioCStrategy, + EuropePMCStrategy, +) logger = logging.getLogger(__name__) @@ -466,7 +474,13 @@ def _parse_publication_types( def _fetch_pmc_fulltext( self, pmid: str, config: ReferenceValidationConfig ) -> tuple[Optional[str], str]: - """Attempt to fetch full text from PMC. + """Attempt to fetch full text using multiple strategies. + + Tries strategies in order: + 1. BioC XML API (cleanest structured text) + 2. Europe PMC (wider coverage) + 3. PMC XML (traditional approach) + 4. PMC HTML (fallback) Args: pmid: PubMed ID @@ -475,6 +489,21 @@ def _fetch_pmc_fulltext( Returns: Tuple of (full_text, content_type) """ + # Strategy 1: Try BioC XML API (cleanest fulltext) + bioc = BioCStrategy() + bioc_result = bioc.fetch(pmid, config.rate_limit_delay) + if bioc_result.success and bioc_result.content: + logger.info(f"Fetched fulltext for PMID:{pmid} via BioC API") + return bioc_result.content, "full_text_bioc" + + # Strategy 2: Try Europe PMC + europepmc = EuropePMCStrategy() + europepmc_result = europepmc.fetch(pmid, config.rate_limit_delay) + if europepmc_result.success and europepmc_result.content: + logger.info(f"Fetched fulltext for PMID:{pmid} via Europe PMC") + return europepmc_result.content, "full_text_europepmc" + + # Strategy 3 & 4: Traditional PMC approach pmcid = self._get_pmcid(pmid, config) if not pmcid: return None, "no_pmc" diff --git a/tests/test_fulltext_strategies.py b/tests/test_fulltext_strategies.py new file mode 100644 index 0000000..61a9e88 --- /dev/null +++ b/tests/test_fulltext_strategies.py @@ -0,0 +1,345 @@ +"""Tests for enhanced fulltext retrieval strategies. + +These tests verify the BioC XML, Europe PMC, Unpaywall, and identifier +conversion utilities. +""" + +import pytest +from unittest.mock import Mock, patch + +from linkml_reference_validator.etl.fulltext_strategies import ( + FulltextStrategy, + BioCStrategy, + EuropePMCStrategy, + UnpaywallStrategy, + IdentifierConverter, + FulltextResult, +) + + +class TestFulltextResult: + """Test the FulltextResult data structure.""" + + def test_fulltext_result_creation(self): + """Test creating a basic FulltextResult.""" + result = FulltextResult( + content="This is the full text content.", + source="bioc", + content_type="full_text", + ) + assert result.content == "This is the full text content." + assert result.source == "bioc" + assert result.content_type == "full_text" + assert result.success is True + + def test_fulltext_result_failure(self): + """Test creating a failed FulltextResult.""" + result = FulltextResult( + content=None, + source="unpaywall", + content_type="unavailable", + success=False, + error_message="No open access version found", + ) + assert result.content is None + assert result.success is False + assert "No open access" in result.error_message + + +class TestBioCStrategy: + """Test BioC XML fulltext retrieval.""" + + def test_bioc_url_construction(self): + """Test the BioC URL is correctly constructed.""" + strategy = BioCStrategy() + url = strategy._build_url("12345678") + assert "12345678" in url + assert "BioC_xml" in url + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_bioc_fetch_success(self, mock_get): + """Test successful BioC fulltext fetch.""" + # Need to provide enough text to pass the 500 char minimum + long_intro = "This is the introduction paragraph. " * 20 + long_results = "This is the results section with detailed findings. " * 20 + mock_response = Mock() + mock_response.status_code = 200 + mock_response.text = f""" + + + + {long_intro} + + + {long_results} + + + """ + mock_get.return_value = mock_response + + strategy = BioCStrategy() + result = strategy.fetch("12345678") + + assert result.success is True + assert "introduction paragraph" in result.content + assert "results section" in result.content + assert result.source == "bioc" + assert result.content_type == "full_text_bioc" + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_bioc_fetch_not_found(self, mock_get): + """Test BioC fetch when article not in OA subset.""" + mock_response = Mock() + mock_response.status_code = 404 + mock_get.return_value = mock_response + + strategy = BioCStrategy() + result = strategy.fetch("99999999") + + assert result.success is False + assert result.content is None + + +class TestEuropePMCStrategy: + """Test Europe PMC fulltext retrieval.""" + + def test_europepmc_api_url(self): + """Test Europe PMC API URL construction.""" + strategy = EuropePMCStrategy() + url = strategy._build_search_url("12345678") + assert "europepmc" in url + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_europepmc_fetch_by_pmid(self, mock_get): + """Test fetching fulltext via Europe PMC by PMID.""" + # Mock search response + search_response = Mock() + search_response.status_code = 200 + search_response.json.return_value = { + "resultList": { + "result": [{ + "pmid": "12345678", + "pmcid": "PMC123456", + "isOpenAccess": "Y", + }] + } + } + + # Mock fulltext response - needs enough text to pass 500 char minimum + long_text = "This is the full article text from Europe PMC. " * 20 + fulltext_response = Mock() + fulltext_response.status_code = 200 + fulltext_response.text = f""" +
+ + +

{long_text}

+
+ +
""" + + mock_get.side_effect = [search_response, fulltext_response] + + strategy = EuropePMCStrategy() + result = strategy.fetch("12345678") + + assert result.success is True + assert "Europe PMC" in result.content + assert result.source == "europepmc" + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_europepmc_not_open_access(self, mock_get): + """Test Europe PMC when article is not open access.""" + mock_response = Mock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "resultList": { + "result": [{ + "pmid": "12345678", + "isOpenAccess": "N", + }] + } + } + mock_get.return_value = mock_response + + strategy = EuropePMCStrategy() + result = strategy.fetch("12345678") + + assert result.success is False + + +class TestUnpaywallStrategy: + """Test Unpaywall API for open access papers.""" + + def test_unpaywall_url_construction(self): + """Test Unpaywall API URL is correctly constructed.""" + strategy = UnpaywallStrategy(email="test@example.com") + url = strategy._build_url("10.1234/example.doi") + assert "api.unpaywall.org" in url + assert "10.1234/example.doi" in url + assert "test@example.com" in url + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_unpaywall_fetch_open_access(self, mock_get): + """Test Unpaywall finding an open access version.""" + mock_response = Mock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "is_oa": True, + "best_oa_location": { + "url": "https://example.com/paper.pdf", + "url_for_pdf": "https://example.com/paper.pdf", + "license": "cc-by", + "version": "publishedVersion", + }, + "oa_locations": [ + { + "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC123456/", + "pmh_id": "oai:pubmedcentral.nih.gov:123456", + } + ] + } + mock_get.return_value = mock_response + + strategy = UnpaywallStrategy(email="test@example.com") + result = strategy.fetch("10.1234/example.doi") + + assert result.success is True + assert result.source == "unpaywall" + assert result.metadata["is_oa"] is True + assert "pdf" in result.metadata.get("pdf_url", "") + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_unpaywall_not_open_access(self, mock_get): + """Test Unpaywall when article is not open access.""" + mock_response = Mock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "is_oa": False, + "best_oa_location": None, + } + mock_get.return_value = mock_response + + strategy = UnpaywallStrategy(email="test@example.com") + result = strategy.fetch("10.1234/closed.doi") + + assert result.success is False + assert "not open access" in result.error_message.lower() + + +class TestIdentifierConverter: + """Test identifier conversion utilities.""" + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_doi_to_pmid(self, mock_get): + """Test converting DOI to PMID.""" + mock_response = Mock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "records": [{ + "pmid": "12345678", + "pmcid": "PMC654321", + "doi": "10.1234/example", + }] + } + mock_get.return_value = mock_response + + converter = IdentifierConverter() + pmid = converter.doi_to_pmid("10.1234/example") + assert pmid == "12345678" + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_pmid_to_doi(self, mock_get): + """Test converting PMID to DOI.""" + mock_response = Mock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "result": { + "12345678": { + "articleids": [ + {"idtype": "pubmed", "value": "12345678"}, + {"idtype": "doi", "value": "10.1234/example"}, + ] + } + } + } + mock_get.return_value = mock_response + + converter = IdentifierConverter() + doi = converter.pmid_to_doi("12345678") + assert doi == "10.1234/example" + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_pmid_to_pmcid(self, mock_get): + """Test converting PMID to PMCID.""" + mock_response = Mock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "records": [{ + "pmid": "12345678", + "pmcid": "PMC654321", + }] + } + mock_get.return_value = mock_response + + converter = IdentifierConverter() + pmcid = converter.pmid_to_pmcid("12345678") + assert pmcid == "PMC654321" + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_pmcid_to_pmid(self, mock_get): + """Test converting PMCID to PMID.""" + mock_response = Mock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "result": { + "uids": ["654321"], + "654321": { + "articleids": [ + {"idtype": "pmid", "value": "12345678"}, + ] + } + } + } + mock_get.return_value = mock_response + + converter = IdentifierConverter() + pmid = converter.pmcid_to_pmid("PMC654321") + assert pmid == "12345678" + + +class TestFulltextStrategyChain: + """Test chaining multiple fulltext strategies.""" + + @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") + def test_strategy_chain_fallback(self, mock_get): + """Test that strategies fall back when one fails.""" + # First strategy (BioC) fails + bioc_response = Mock() + bioc_response.status_code = 404 + + # Second strategy (Europe PMC) succeeds + long_text = "Europe PMC text with enough content to pass validation. " * 20 + europepmc_search = Mock() + europepmc_search.status_code = 200 + europepmc_search.json.return_value = { + "resultList": { + "result": [{ + "pmid": "12345678", + "pmcid": "PMC123456", + "isOpenAccess": "Y", + }] + } + } + europepmc_fulltext = Mock() + europepmc_fulltext.status_code = 200 + europepmc_fulltext.text = f"

{long_text}

" + + mock_get.side_effect = [bioc_response, europepmc_search, europepmc_fulltext] + + from linkml_reference_validator.etl.fulltext_strategies import FulltextFetcher + fetcher = FulltextFetcher(email="test@example.com") + result = fetcher.fetch_fulltext_for_pmid("12345678") + + assert result.success is True + assert result.source == "europepmc" From 5d90b3331dc2c5b2e83c86858c27538408b31019 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 19:32:47 +0000 Subject: [PATCH 2/2] Refactor fulltext strategies into proper FullTextProvider classes Add BioCFullTextProvider and EuropePMCFullTextProvider as proper FullTextProvider implementations following the existing architecture. - bioc.py: Fetches structured text from NCBI BioNLP BioC XML API - epmc.py: Fetches fulltext XML for OA articles from Europe PMC (complements epmc_preprint.py which handles preprints only) Remove fulltext_strategies.py which bypassed the provider chain by fetching fulltext inside source classes. The provider architecture uses _enrich_with_full_text() in reference_fetcher.py to call providers after sources return content. Revert doi.py and pmid.py to main branch versions - fulltext enrichment happens via the provider chain, not inside sources. Co-Authored-By: Claude Opus 4.5 Claude-Session: https://claude.ai/code/session_01PYbfu1JPi9rr7v5XU6qbJj --- .../etl/fulltext/__init__.py | 12 +- .../etl/fulltext/bioc.py | 117 +++ .../etl/fulltext/epmc.py | 182 +++++ .../etl/fulltext_strategies.py | 740 ------------------ .../etl/sources/doi.py | 75 +- .../etl/sources/pmid.py | 31 +- tests/test_fulltext_providers.py | 150 ++++ tests/test_fulltext_strategies.py | 345 -------- 8 files changed, 460 insertions(+), 1192 deletions(-) create mode 100644 src/linkml_reference_validator/etl/fulltext/bioc.py create mode 100644 src/linkml_reference_validator/etl/fulltext/epmc.py delete mode 100644 src/linkml_reference_validator/etl/fulltext_strategies.py delete mode 100644 tests/test_fulltext_strategies.py diff --git a/src/linkml_reference_validator/etl/fulltext/__init__.py b/src/linkml_reference_validator/etl/fulltext/__init__.py index f309ef9..4136b8b 100644 --- a/src/linkml_reference_validator/etl/fulltext/__init__.py +++ b/src/linkml_reference_validator/etl/fulltext/__init__.py @@ -6,18 +6,22 @@ ) # Import providers to register them -from linkml_reference_validator.etl.fulltext.pmc import PMCFullTextProvider +from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider +from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider from linkml_reference_validator.etl.fulltext.epmc_preprint import EuropePMCPreprintProvider -from linkml_reference_validator.etl.fulltext.unpaywall import UnpaywallProvider from linkml_reference_validator.etl.fulltext.openalex import OpenAlexProvider +from linkml_reference_validator.etl.fulltext.pmc import PMCFullTextProvider +from linkml_reference_validator.etl.fulltext.unpaywall import UnpaywallProvider from linkml_reference_validator.etl.fulltext.zotero import ZoteroFullTextProvider __all__ = [ + "BioCFullTextProvider", + "EuropePMCFullTextProvider", + "EuropePMCPreprintProvider", "FullTextProvider", "FullTextProviderRegistry", + "OpenAlexProvider", "PMCFullTextProvider", - "EuropePMCPreprintProvider", "UnpaywallProvider", - "OpenAlexProvider", "ZoteroFullTextProvider", ] diff --git a/src/linkml_reference_validator/etl/fulltext/bioc.py b/src/linkml_reference_validator/etl/fulltext/bioc.py new file mode 100644 index 0000000..f578ec9 --- /dev/null +++ b/src/linkml_reference_validator/etl/fulltext/bioc.py @@ -0,0 +1,117 @@ +"""BioC XML full-text provider. + +Fetches structured full text from the NCBI BioNLP BioC API, which provides +clean paragraph-level text for articles in the PMC Open Access subset. +""" + +import logging +import time +from typing import Optional + +import requests # type: ignore +from bs4 import BeautifulSoup # type: ignore + +from linkml_reference_validator.models import ( + FullTextLocation, + ReferenceIdentifiers, + ReferenceValidationConfig, +) +from linkml_reference_validator.etl.fulltext.base import ( + FullTextProvider, + FullTextProviderRegistry, +) +from linkml_reference_validator.etl.extract import MIN_FULLTEXT_CHARS + +logger = logging.getLogger(__name__) + +BIOC_URL = "https://www.ncbi.nlm.nih.gov/research/bionlp/RESTful/pmcoa.cgi/BioC_xml/{pmid}/ascii" + + +@FullTextProviderRegistry.register +class BioCFullTextProvider(FullTextProvider): + """Fetch full text via the NCBI BioNLP BioC XML API. + + The BioC API provides structured, clean paragraph text for articles in + the PMC Open Access subset. It returns text with passage-level structure + (title, abstract, body sections) without inline markup. + + Examples: + >>> BioCFullTextProvider.name() + 'bioc' + """ + + @classmethod + def name(cls) -> str: + return "bioc" + + def locate( + self, ids: ReferenceIdentifiers, config: ReferenceValidationConfig + ) -> Optional[FullTextLocation]: + if not ids.pmid: + return None + + time.sleep(config.rate_limit_delay) + + url = BIOC_URL.format(pmid=ids.pmid) + try: + response = requests.get(url, timeout=30) + except requests.RequestException as exc: + logger.debug(f"BioC request failed for PMID:{ids.pmid}: {exc}") + return None + + if response.status_code == 404: + logger.debug(f"PMID:{ids.pmid} not in PMC Open Access subset") + return None + if response.status_code != 200: + logger.debug(f"BioC returned {response.status_code} for PMID:{ids.pmid}") + return None + + text = self._extract_text(response.text) + if not text or len(text) < MIN_FULLTEXT_CHARS: + logger.debug(f"BioC returned insufficient text for PMID:{ids.pmid}") + return None + + return FullTextLocation( + text=text, + format_hint="text", + oa_status="green", + provider="bioc", + ) + + def _extract_text(self, xml_content: str) -> Optional[str]: + """Extract paragraph text from BioC XML. + + Args: + xml_content: BioC XML response + + Returns: + Concatenated passage text, or None if parsing fails + + Examples: + >>> provider = BioCFullTextProvider() + >>> xml = ''' + ... Introduction text. + ... Methods section. + ... ''' + >>> provider._extract_text(xml) + 'Introduction text.\\n\\nMethods section.' + """ + try: + soup = BeautifulSoup(xml_content, "xml") + except Exception as exc: + logger.debug(f"Failed to parse BioC XML: {exc}") + return None + + passages = soup.find_all("passage") + if not passages: + return None + + texts = [] + for passage in passages: + text_elem = passage.find("text") + if text_elem and text_elem.string: + text = text_elem.string.strip() + if text: + texts.append(text) + + return "\n\n".join(texts) if texts else None diff --git a/src/linkml_reference_validator/etl/fulltext/epmc.py b/src/linkml_reference_validator/etl/fulltext/epmc.py new file mode 100644 index 0000000..a494b69 --- /dev/null +++ b/src/linkml_reference_validator/etl/fulltext/epmc.py @@ -0,0 +1,182 @@ +"""Europe PMC full-text provider. + +Fetches full-text XML for open access articles from the Europe PMC API. +This complements EuropePMCPreprintProvider, which handles preprints only. +""" + +import logging +import time +from typing import Optional + +import requests # type: ignore +from bs4 import BeautifulSoup # type: ignore + +from linkml_reference_validator.models import ( + FullTextLocation, + ReferenceIdentifiers, + ReferenceValidationConfig, +) +from linkml_reference_validator.etl.fulltext.base import ( + FullTextProvider, + FullTextProviderRegistry, +) +from linkml_reference_validator.etl.extract import MIN_FULLTEXT_CHARS + +logger = logging.getLogger(__name__) + +EUROPEPMC_SEARCH_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest/search" +EUROPEPMC_FULLTEXT_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest/{source}/{id}/fullTextXML" + + +@FullTextProviderRegistry.register +class EuropePMCFullTextProvider(FullTextProvider): + """Fetch full-text XML for open access articles from Europe PMC. + + Queries the Europe PMC search API to find open access articles by PMID, + then fetches full-text XML when available. This provider handles regular + OA articles; preprints are handled by EuropePMCPreprintProvider. + + Examples: + >>> EuropePMCFullTextProvider.name() + 'epmc' + """ + + @classmethod + def name(cls) -> str: + return "epmc" + + def locate( + self, ids: ReferenceIdentifiers, config: ReferenceValidationConfig + ) -> Optional[FullTextLocation]: + if not ids.pmid: + return None + + time.sleep(config.rate_limit_delay) + + params = { + "query": f"EXT_ID:{ids.pmid} AND SRC:MED", + "format": "json", + "resultType": "core", + "pageSize": "1", + "email": config.email, + } + + try: + response = requests.get(EUROPEPMC_SEARCH_URL, params=params, timeout=30) + except requests.RequestException as exc: + logger.debug(f"Europe PMC search failed for PMID:{ids.pmid}: {exc}") + return None + + if response.status_code != 200: + logger.debug(f"Europe PMC returned {response.status_code} for PMID:{ids.pmid}") + return None + + try: + data = response.json() + except ValueError: + logger.debug(f"Invalid JSON from Europe PMC for PMID:{ids.pmid}") + return None + + result = self._find_oa_result(data) + if not result: + return None + + pmcid = result.get("pmcid") + if not pmcid: + logger.debug(f"No PMCID in Europe PMC result for PMID:{ids.pmid}") + return None + + text = self._fetch_fulltext_xml(pmcid, config) + if not text or len(text) < MIN_FULLTEXT_CHARS: + return None + + return FullTextLocation( + text=text, + format_hint="text", + oa_status="green", + license=result.get("license"), + provider="epmc", + ) + + def _find_oa_result(self, data: dict) -> Optional[dict]: + """Find an open access result from search response. + + Args: + data: Europe PMC search response JSON + + Returns: + First OA result dict, or None + """ + results = data.get("resultList", {}).get("result", []) + for result in results: + if not isinstance(result, dict): + continue + if result.get("isOpenAccess") == "Y": + return result + return None + + def _fetch_fulltext_xml( + self, pmcid: str, config: ReferenceValidationConfig + ) -> Optional[str]: + """Fetch and extract text from Europe PMC fullTextXML endpoint. + + Args: + pmcid: PMC ID (with or without PMC prefix) + config: Configuration for rate limiting + + Returns: + Extracted body text, or None + """ + pmcid_clean = pmcid.replace("PMC", "") + url = EUROPEPMC_FULLTEXT_URL.format(source="PMC", id=pmcid_clean) + + time.sleep(config.rate_limit_delay) + + try: + response = requests.get(url, timeout=30) + except requests.RequestException as exc: + logger.debug(f"Europe PMC fulltext fetch failed for {pmcid}: {exc}") + return None + + if response.status_code != 200: + logger.debug(f"Europe PMC fulltext returned {response.status_code} for {pmcid}") + return None + + return self._extract_body_text(response.text) + + def _extract_body_text(self, xml_content: str) -> Optional[str]: + """Extract body paragraphs from JATS XML. + + Args: + xml_content: JATS XML content + + Returns: + Concatenated paragraph text, or None + + Examples: + >>> provider = EuropePMCFullTextProvider() + >>> xml = '

Text here.

' + >>> provider._extract_body_text(xml) + 'Text here.' + """ + try: + soup = BeautifulSoup(xml_content, "xml") + except Exception as exc: + logger.debug(f"Failed to parse Europe PMC XML: {exc}") + return None + + body = soup.find("body") + if not body: + return None + + paragraphs = body.find_all("p") + if not paragraphs: + return None + + texts = [] + for p in paragraphs: + text = p.get_text().strip() + if text: + texts.append(text) + + return "\n\n".join(texts) if texts else None diff --git a/src/linkml_reference_validator/etl/fulltext_strategies.py b/src/linkml_reference_validator/etl/fulltext_strategies.py deleted file mode 100644 index d42711b..0000000 --- a/src/linkml_reference_validator/etl/fulltext_strategies.py +++ /dev/null @@ -1,740 +0,0 @@ -"""Enhanced fulltext retrieval strategies. - -This module provides multiple strategies for fetching fulltext content -from scientific publications, including: -- BioC XML API (NCBI BioNLP) -- Europe PMC -- Unpaywall (open access papers via DOI) -- Identifier conversion utilities (DOI <-> PMID <-> PMCID) - -Examples: - >>> from linkml_reference_validator.etl.fulltext_strategies import FulltextFetcher - >>> fetcher = FulltextFetcher(email="user@example.com") - >>> # result = fetcher.fetch_fulltext_for_pmid("12345678") - >>> # result = fetcher.fetch_fulltext_for_doi("10.1234/example") -""" - -import logging -import time -from abc import ABC, abstractmethod -from dataclasses import dataclass, field -from typing import Optional - -import requests # type: ignore -from bs4 import BeautifulSoup # type: ignore - -logger = logging.getLogger(__name__) - -# API URLs -BIOC_URL = "https://www.ncbi.nlm.nih.gov/research/bionlp/RESTful/pmcoa.cgi/BioC_xml/{pmid}/ascii" -EUROPEPMC_SEARCH_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest/search" -EUROPEPMC_FULLTEXT_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest/{source}/{id}/fullTextXML" -UNPAYWALL_URL = "https://api.unpaywall.org/v2/{doi}" -NCBI_IDCONV_URL = "https://www.ncbi.nlm.nih.gov/pmc/utils/idconv/v1.0/" -NCBI_ESUMMARY_URL = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esummary.fcgi" - - -@dataclass -class FulltextResult: - """Result from a fulltext retrieval attempt. - - Attributes: - content: The fulltext content if found - source: The source that provided the content (bioc, europepmc, unpaywall) - content_type: Type of content (full_text_bioc, full_text_europepmc, etc.) - success: Whether the retrieval was successful - error_message: Error message if retrieval failed - metadata: Additional metadata from the source - - Examples: - >>> result = FulltextResult( - ... content="Full article text here.", - ... source="bioc", - ... content_type="full_text_bioc", - ... success=True, - ... ) - >>> result.success - True - >>> result.source - 'bioc' - """ - - content: Optional[str] = None - source: str = "unknown" - content_type: str = "unknown" - success: bool = True - error_message: Optional[str] = None - metadata: dict = field(default_factory=dict) - - -class FulltextStrategy(ABC): - """Abstract base class for fulltext retrieval strategies. - - Subclasses must implement the fetch method to retrieve fulltext - content from their respective sources. - - Examples: - >>> class MyStrategy(FulltextStrategy): - ... def fetch(self, identifier): - ... return FulltextResult(content="test", source="my_source") - >>> strategy = MyStrategy() - >>> strategy.name - 'MyStrategy' - """ - - @property - def name(self) -> str: - """Return the strategy name.""" - return self.__class__.__name__ - - @abstractmethod - def fetch(self, identifier: str, rate_limit_delay: float = 0.5) -> FulltextResult: - """Fetch fulltext for the given identifier. - - Args: - identifier: The identifier to fetch (PMID, DOI, etc.) - rate_limit_delay: Delay between API requests - - Returns: - FulltextResult with content if successful - """ - ... - - -class BioCStrategy(FulltextStrategy): - """Fetch fulltext using NCBI BioC XML API. - - The BioC API provides structured fulltext for articles in the - PubMed Central Open Access subset. This is often the most reliable - source for clean fulltext. - - Examples: - >>> strategy = BioCStrategy() - >>> url = strategy._build_url("12345678") - >>> "12345678" in url - True - >>> "BioC_xml" in url - True - """ - - def _build_url(self, pmid: str) -> str: - """Build the BioC API URL for a PMID. - - Args: - pmid: The PubMed ID - - Returns: - The API URL - - Examples: - >>> strategy = BioCStrategy() - >>> url = strategy._build_url("12345") - >>> "12345" in url and "BioC_xml" in url - True - """ - return BIOC_URL.format(pmid=pmid) - - def fetch(self, identifier: str, rate_limit_delay: float = 0.5) -> FulltextResult: - """Fetch fulltext from BioC API. - - Args: - identifier: PMID to fetch - rate_limit_delay: Delay before request - - Returns: - FulltextResult with fulltext if available - """ - pmid = identifier.strip() - if ":" in pmid: - pmid = pmid.split(":")[-1] - - time.sleep(rate_limit_delay) - url = self._build_url(pmid) - - try: - response = requests.get(url, timeout=30) - except requests.RequestException as e: - logger.warning(f"BioC request failed for PMID:{pmid}: {e}") - return FulltextResult( - success=False, - source="bioc", - content_type="unavailable", - error_message=str(e), - ) - - if response.status_code != 200: - logger.debug(f"BioC not available for PMID:{pmid} (status {response.status_code})") - return FulltextResult( - success=False, - source="bioc", - content_type="unavailable", - error_message=f"HTTP {response.status_code}", - ) - - # Parse BioC XML - soup = BeautifulSoup(response.text, "xml") - text_sections = [text_tag.get_text() for text_tag in soup.find_all("text")] - - if not text_sections: - return FulltextResult( - success=False, - source="bioc", - content_type="unavailable", - error_message="No text sections found in BioC response", - ) - - full_text = "\n\n".join(text_sections).strip() - - if len(full_text) < 500: - return FulltextResult( - success=False, - source="bioc", - content_type="unavailable", - error_message="BioC response too short", - ) - - return FulltextResult( - content=full_text, - source="bioc", - content_type="full_text_bioc", - success=True, - ) - - -class EuropePMCStrategy(FulltextStrategy): - """Fetch fulltext from Europe PMC. - - Europe PMC provides fulltext for many open access articles, - including some not available in the US PubMed Central. - - Examples: - >>> strategy = EuropePMCStrategy() - >>> url = strategy._build_search_url("12345678") - >>> "europepmc" in url - True - """ - - def _build_search_url(self, pmid: str) -> str: - """Build Europe PMC search URL for a PMID. - - Args: - pmid: The PubMed ID - - Returns: - The search API URL - """ - return f"{EUROPEPMC_SEARCH_URL}?query=ext_id:{pmid}&format=json" - - def _build_fulltext_url(self, pmcid: str) -> str: - """Build Europe PMC fulltext URL for a PMCID. - - Args: - pmcid: The PMC ID (with or without PMC prefix) - - Returns: - The fulltext API URL - """ - # Strip PMC prefix if present - pmc_id = pmcid.replace("PMC", "") - return EUROPEPMC_FULLTEXT_URL.format(source="PMC", id=pmc_id) - - def fetch(self, identifier: str, rate_limit_delay: float = 0.5) -> FulltextResult: - """Fetch fulltext from Europe PMC. - - Args: - identifier: PMID to fetch - rate_limit_delay: Delay before request - - Returns: - FulltextResult with fulltext if available - """ - pmid = identifier.strip() - if ":" in pmid: - pmid = pmid.split(":")[-1] - - time.sleep(rate_limit_delay) - - # First, search for the article to get PMCID - search_url = self._build_search_url(pmid) - try: - response = requests.get(search_url, timeout=30) - except requests.RequestException as e: - logger.warning(f"Europe PMC search failed for PMID:{pmid}: {e}") - return FulltextResult( - success=False, - source="europepmc", - content_type="unavailable", - error_message=str(e), - ) - - if response.status_code != 200: - return FulltextResult( - success=False, - source="europepmc", - content_type="unavailable", - error_message=f"Search HTTP {response.status_code}", - ) - - data = response.json() - results = data.get("resultList", {}).get("result", []) - - if not results: - return FulltextResult( - success=False, - source="europepmc", - content_type="unavailable", - error_message="Article not found in Europe PMC", - ) - - article = results[0] - pmcid = article.get("pmcid") - is_oa = article.get("isOpenAccess") == "Y" - - if not pmcid or not is_oa: - return FulltextResult( - success=False, - source="europepmc", - content_type="unavailable", - error_message="Article not open access or no PMC ID", - ) - - # Fetch the fulltext XML - time.sleep(rate_limit_delay) - fulltext_url = self._build_fulltext_url(pmcid) - - try: - ft_response = requests.get(fulltext_url, timeout=30) - except requests.RequestException as e: - logger.warning(f"Europe PMC fulltext fetch failed: {e}") - return FulltextResult( - success=False, - source="europepmc", - content_type="unavailable", - error_message=str(e), - ) - - if ft_response.status_code != 200: - return FulltextResult( - success=False, - source="europepmc", - content_type="unavailable", - error_message=f"Fulltext HTTP {ft_response.status_code}", - ) - - # Parse the fulltext XML - soup = BeautifulSoup(ft_response.text, "xml") - body = soup.find("body") - - if body: - paragraphs = body.find_all("p") - if paragraphs: - text = "\n\n".join(p.get_text() for p in paragraphs) - if len(text) > 500: - return FulltextResult( - content=text, - source="europepmc", - content_type="full_text_europepmc", - success=True, - metadata={"pmcid": pmcid}, - ) - - return FulltextResult( - success=False, - source="europepmc", - content_type="unavailable", - error_message="Could not extract text from Europe PMC XML", - ) - - -class UnpaywallStrategy(FulltextStrategy): - """Fetch open access papers via Unpaywall API. - - Unpaywall provides access to legal open access versions of papers. - Requires a valid email address for API access. - - Examples: - >>> strategy = UnpaywallStrategy(email="test@example.com") - >>> url = strategy._build_url("10.1234/example") - >>> "api.unpaywall.org" in url - True - """ - - def __init__(self, email: str = "linkml-reference-validator@example.com"): - """Initialize with email for API access. - - Args: - email: Email address for Unpaywall API - """ - self.email = email - - def _build_url(self, doi: str) -> str: - """Build Unpaywall API URL for a DOI. - - Args: - doi: The DOI - - Returns: - The API URL - - Examples: - >>> strategy = UnpaywallStrategy(email="test@example.com") - >>> url = strategy._build_url("10.1234/test") - >>> "10.1234/test" in url and "test@example.com" in url - True - """ - return f"{UNPAYWALL_URL.format(doi=doi)}?email={self.email}" - - def fetch(self, identifier: str, rate_limit_delay: float = 0.5) -> FulltextResult: - """Fetch open access info from Unpaywall. - - Note: This does not fetch the actual fulltext, but provides - information about where to find open access versions. - - Args: - identifier: DOI to look up - rate_limit_delay: Delay before request - - Returns: - FulltextResult with OA location info - """ - doi = identifier.strip() - if doi.lower().startswith("doi:"): - doi = doi[4:] - - time.sleep(rate_limit_delay) - url = self._build_url(doi) - - try: - response = requests.get(url, timeout=30) - except requests.RequestException as e: - logger.warning(f"Unpaywall request failed for DOI:{doi}: {e}") - return FulltextResult( - success=False, - source="unpaywall", - content_type="unavailable", - error_message=str(e), - ) - - if response.status_code != 200: - return FulltextResult( - success=False, - source="unpaywall", - content_type="unavailable", - error_message=f"HTTP {response.status_code}", - ) - - data = response.json() - is_oa = data.get("is_oa", False) - - if not is_oa: - return FulltextResult( - success=False, - source="unpaywall", - content_type="unavailable", - error_message="Article is not open access", - ) - - best_location = data.get("best_oa_location", {}) or {} - pdf_url = best_location.get("url_for_pdf") or best_location.get("url") - oa_locations = data.get("oa_locations", []) - - # Look for PMC source in OA locations - pmcid = None - for loc in oa_locations: - pmh_id = loc.get("pmh_id", "") - if "pubmedcentral" in pmh_id.lower(): - # Extract PMC ID from pmh_id like "oai:pubmedcentral.nih.gov:123456" - parts = pmh_id.split(":") - if len(parts) >= 3: - pmcid = f"PMC{parts[-1]}" - break - - return FulltextResult( - content=None, # Unpaywall doesn't provide content directly - source="unpaywall", - content_type="oa_location", - success=True, - metadata={ - "is_oa": True, - "pdf_url": pdf_url, - "pmcid": pmcid, - "license": best_location.get("license"), - "version": best_location.get("version"), - }, - ) - - -class IdentifierConverter: - """Convert between DOI, PMID, and PMCID identifiers. - - Uses NCBI ID Converter and E-utilities APIs. - - Examples: - >>> converter = IdentifierConverter() - >>> # pmid = converter.doi_to_pmid("10.1234/example") - >>> # doi = converter.pmid_to_doi("12345678") - """ - - def __init__(self, email: str = "linkml-reference-validator@example.com"): - """Initialize with email for NCBI API. - - Args: - email: Email for NCBI API access - """ - self.email = email - - def doi_to_pmid(self, doi: str, rate_limit_delay: float = 0.5) -> Optional[str]: - """Convert DOI to PMID. - - Args: - doi: The DOI to convert - rate_limit_delay: Delay before request - - Returns: - PMID if found, None otherwise - - Examples: - >>> converter = IdentifierConverter() - >>> # This would make an API call in real usage - >>> # pmid = converter.doi_to_pmid("10.1234/example") - """ - if doi.lower().startswith("doi:"): - doi = doi[4:] - - time.sleep(rate_limit_delay) - url = f"{NCBI_IDCONV_URL}?ids={doi}&format=json" - - try: - response = requests.get(url, timeout=30) - except requests.RequestException as e: - logger.warning(f"ID conversion failed for DOI:{doi}: {e}") - return None - - if response.status_code != 200: - return None - - data = response.json() - records = data.get("records", []) - - if records: - return records[0].get("pmid") - return None - - def pmid_to_doi(self, pmid: str, rate_limit_delay: float = 0.5) -> Optional[str]: - """Convert PMID to DOI. - - Args: - pmid: The PMID to convert - rate_limit_delay: Delay before request - - Returns: - DOI if found, None otherwise - """ - if ":" in pmid: - pmid = pmid.split(":")[-1] - - time.sleep(rate_limit_delay) - url = f"{NCBI_ESUMMARY_URL}?db=pubmed&id={pmid}&retmode=json" - - try: - response = requests.get(url, timeout=30) - except requests.RequestException as e: - logger.warning(f"PMID to DOI conversion failed for {pmid}: {e}") - return None - - if response.status_code != 200: - return None - - data = response.json() - - try: - article_info = data["result"][str(pmid)] - for aid in article_info.get("articleids", []): - if aid.get("idtype") == "doi": - return aid.get("value") - # Check elocationid as fallback - elocationid = article_info.get("elocationid", "") - if elocationid.startswith("10."): - return elocationid - except KeyError: - pass - - return None - - def pmid_to_pmcid(self, pmid: str, rate_limit_delay: float = 0.5) -> Optional[str]: - """Convert PMID to PMCID. - - Args: - pmid: The PMID to convert - rate_limit_delay: Delay before request - - Returns: - PMCID if found, None otherwise - """ - if ":" in pmid: - pmid = pmid.split(":")[-1] - - time.sleep(rate_limit_delay) - url = f"{NCBI_IDCONV_URL}?ids={pmid}&format=json" - - try: - response = requests.get(url, timeout=30) - except requests.RequestException as e: - logger.warning(f"PMID to PMCID conversion failed for {pmid}: {e}") - return None - - if response.status_code != 200: - return None - - data = response.json() - records = data.get("records", []) - - if records: - return records[0].get("pmcid") - return None - - def pmcid_to_pmid(self, pmcid: str, rate_limit_delay: float = 0.5) -> Optional[str]: - """Convert PMCID to PMID. - - Args: - pmcid: The PMCID to convert (with or without PMC prefix) - rate_limit_delay: Delay before request - - Returns: - PMID if found, None otherwise - """ - # Strip PMC prefix if present - pmc_id = pmcid.replace("PMC", "").replace("pmc", "") - if ":" in pmc_id: - pmc_id = pmc_id.split(":")[-1] - - time.sleep(rate_limit_delay) - url = f"{NCBI_ESUMMARY_URL}?db=pmc&id={pmc_id}&retmode=json" - - try: - response = requests.get(url, timeout=30) - except requests.RequestException as e: - logger.warning(f"PMCID to PMID conversion failed for {pmcid}: {e}") - return None - - if response.status_code != 200: - return None - - data = response.json() - - try: - uids = data["result"]["uids"] - if uids: - uid = uids[0] - article_ids = data["result"][uid].get("articleids", []) - for item in article_ids: - if item.get("idtype") == "pmid": - return item.get("value") - except KeyError: - pass - - return None - - -class FulltextFetcher: - """Orchestrates multiple fulltext strategies with fallback. - - Tries strategies in order until one succeeds. - - Examples: - >>> fetcher = FulltextFetcher(email="test@example.com") - >>> # result = fetcher.fetch_fulltext_for_pmid("12345678") - >>> # result = fetcher.fetch_fulltext_for_doi("10.1234/example") - """ - - def __init__( - self, - email: str = "linkml-reference-validator@example.com", - rate_limit_delay: float = 0.5, - ): - """Initialize the fetcher with strategies. - - Args: - email: Email for API access - rate_limit_delay: Delay between requests - """ - self.email = email - self.rate_limit_delay = rate_limit_delay - self.converter = IdentifierConverter(email=email) - - # Strategies in priority order for PMIDs - self.pmid_strategies: list[FulltextStrategy] = [ - BioCStrategy(), - EuropePMCStrategy(), - ] - - self.unpaywall = UnpaywallStrategy(email=email) - - def fetch_fulltext_for_pmid(self, pmid: str) -> FulltextResult: - """Fetch fulltext for a PMID using all strategies. - - Tries strategies in order: - 1. BioC XML API - 2. Europe PMC - - Args: - pmid: The PubMed ID - - Returns: - FulltextResult from the first successful strategy - """ - for strategy in self.pmid_strategies: - result = strategy.fetch(pmid, self.rate_limit_delay) - if result.success and result.content: - logger.info(f"Fetched fulltext for PMID:{pmid} via {strategy.name}") - return result - - # All strategies failed - return FulltextResult( - success=False, - source="none", - content_type="unavailable", - error_message="No fulltext available from any source", - ) - - def fetch_fulltext_for_doi(self, doi: str) -> FulltextResult: - """Fetch fulltext for a DOI. - - Tries to convert DOI to PMID first, then uses PMID strategies. - Falls back to Unpaywall for OA location. - - Args: - doi: The DOI - - Returns: - FulltextResult - """ - # Try to convert DOI to PMID - pmid = self.converter.doi_to_pmid(doi, self.rate_limit_delay) - - if pmid: - result = self.fetch_fulltext_for_pmid(pmid) - if result.success: - return result - - # Try Unpaywall for OA location - unpaywall_result = self.unpaywall.fetch(doi, self.rate_limit_delay) - if unpaywall_result.success: - # If Unpaywall found a PMCID, try to fetch that - pmcid = unpaywall_result.metadata.get("pmcid") - if pmcid: - pmid_from_pmc = self.converter.pmcid_to_pmid(pmcid, self.rate_limit_delay) - if pmid_from_pmc: - result = self.fetch_fulltext_for_pmid(pmid_from_pmc) - if result.success: - return result - - # Return the Unpaywall result with OA location info - return unpaywall_result - - return FulltextResult( - success=False, - source="none", - content_type="unavailable", - error_message="No fulltext available from any source", - ) diff --git a/src/linkml_reference_validator/etl/sources/doi.py b/src/linkml_reference_validator/etl/sources/doi.py index db39c93..68e795a 100644 --- a/src/linkml_reference_validator/etl/sources/doi.py +++ b/src/linkml_reference_validator/etl/sources/doi.py @@ -2,9 +2,6 @@ Fetches publication metadata from Crossref API, with fallback to DataCite for DOIs not found in Crossref (e.g., Zenodo, Figshare, Dryad). -Also attempts fulltext retrieval via: -- Unpaywall (open access papers) -- Identifier conversion to PMID for PMC access Examples: >>> from linkml_reference_validator.etl.sources.doi import DOISource @@ -31,11 +28,6 @@ extract_extra_fields, format_extra_fields_for_content, ) -from linkml_reference_validator.etl.fulltext_strategies import ( - UnpaywallStrategy, - IdentifierConverter, - FulltextFetcher, -) logger = logging.getLogger(__name__) @@ -71,10 +63,6 @@ def fetch( ) -> Optional[ReferenceContent]: """Fetch a publication by DOI from Crossref, falling back to DataCite. - Also attempts fulltext retrieval via: - 1. Convert DOI to PMID, then use PMC strategies - 2. Unpaywall for open access versions - Args: identifier: DOI (without prefix) config: Configuration including rate limiting and email @@ -154,20 +142,11 @@ def _fetch_from_crossref( is_preprint = self._crossref_preprint_status(message) - # Try to get fulltext via enhanced strategies - fulltext, content_type = self._fetch_fulltext(doi, config) - - if fulltext: - content = f"{abstract}\n\n{fulltext}" if abstract else fulltext - else: - content = abstract if abstract else None - content_type = "abstract_only" if abstract else "unavailable" - return ReferenceContent( reference_id=f"DOI:{doi}", title=title, - content=content, - content_type=content_type, + content=abstract if abstract else None, + content_type="abstract_only" if abstract else "unavailable", authors=authors, journal=journal, year=year, @@ -453,56 +432,6 @@ def _parse_datacite_subjects(self, subjects: list) -> Optional[list[str]]: result.append(subj) return result if result else None - def _fetch_fulltext( - self, doi: str, config: ReferenceValidationConfig - ) -> tuple[Optional[str], str]: - """Attempt to fetch fulltext for a DOI. - - Tries: - 1. Convert DOI to PMID and use FulltextFetcher - 2. Check Unpaywall for OA location info - - Args: - doi: The DOI - config: Configuration for rate limiting - - Returns: - Tuple of (fulltext, content_type) - """ - converter = IdentifierConverter(email=config.email) - - # Try to convert DOI to PMID - pmid = converter.doi_to_pmid(doi, config.rate_limit_delay) - if pmid: - fetcher = FulltextFetcher( - email=config.email, - rate_limit_delay=config.rate_limit_delay, - ) - result = fetcher.fetch_fulltext_for_pmid(pmid) - if result.success and result.content: - logger.info(f"Fetched fulltext for DOI:{doi} via PMID:{pmid}") - return result.content, result.content_type - - # Try Unpaywall - it may provide PMCID even without direct PMID - unpaywall = UnpaywallStrategy(email=config.email) - oa_result = unpaywall.fetch(doi, config.rate_limit_delay) - if oa_result.success: - pmcid = oa_result.metadata.get("pmcid") - if pmcid: - # Try to get PMID from PMCID - pmid_from_pmc = converter.pmcid_to_pmid(pmcid, config.rate_limit_delay) - if pmid_from_pmc: - fetcher = FulltextFetcher( - email=config.email, - rate_limit_delay=config.rate_limit_delay, - ) - result = fetcher.fetch_fulltext_for_pmid(pmid_from_pmc) - if result.success and result.content: - logger.info(f"Fetched fulltext for DOI:{doi} via Unpaywall PMCID") - return result.content, result.content_type - - return None, "abstract_only" - def _parse_crossref_authors(self, authors: list) -> list[str]: """Parse author list from Crossref response. diff --git a/src/linkml_reference_validator/etl/sources/pmid.py b/src/linkml_reference_validator/etl/sources/pmid.py index f7eee71..fff79e4 100644 --- a/src/linkml_reference_validator/etl/sources/pmid.py +++ b/src/linkml_reference_validator/etl/sources/pmid.py @@ -1,10 +1,6 @@ """PMID (PubMed ID) reference source. Fetches publication content from PubMed/NCBI using the Entrez API. -Supports multiple fulltext retrieval strategies: -- BioC XML API (NCBI BioNLP) -- Europe PMC -- PMC XML/HTML (traditional approach) Examples: >>> from linkml_reference_validator.etl.sources.pmid import PMIDSource @@ -37,10 +33,6 @@ extract_extra_fields, format_extra_fields_for_content, ) -from linkml_reference_validator.etl.fulltext_strategies import ( - BioCStrategy, - EuropePMCStrategy, -) logger = logging.getLogger(__name__) @@ -474,13 +466,7 @@ def _parse_publication_types( def _fetch_pmc_fulltext( self, pmid: str, config: ReferenceValidationConfig ) -> tuple[Optional[str], str]: - """Attempt to fetch full text using multiple strategies. - - Tries strategies in order: - 1. BioC XML API (cleanest structured text) - 2. Europe PMC (wider coverage) - 3. PMC XML (traditional approach) - 4. PMC HTML (fallback) + """Attempt to fetch full text from PMC. Args: pmid: PubMed ID @@ -489,21 +475,6 @@ def _fetch_pmc_fulltext( Returns: Tuple of (full_text, content_type) """ - # Strategy 1: Try BioC XML API (cleanest fulltext) - bioc = BioCStrategy() - bioc_result = bioc.fetch(pmid, config.rate_limit_delay) - if bioc_result.success and bioc_result.content: - logger.info(f"Fetched fulltext for PMID:{pmid} via BioC API") - return bioc_result.content, "full_text_bioc" - - # Strategy 2: Try Europe PMC - europepmc = EuropePMCStrategy() - europepmc_result = europepmc.fetch(pmid, config.rate_limit_delay) - if europepmc_result.success and europepmc_result.content: - logger.info(f"Fetched fulltext for PMID:{pmid} via Europe PMC") - return europepmc_result.content, "full_text_europepmc" - - # Strategy 3 & 4: Traditional PMC approach pmcid = self._get_pmcid(pmid, config) if not pmcid: return None, "no_pmc" diff --git a/tests/test_fulltext_providers.py b/tests/test_fulltext_providers.py index bf628d9..7c9f868 100644 --- a/tests/test_fulltext_providers.py +++ b/tests/test_fulltext_providers.py @@ -354,3 +354,153 @@ def test_the_floor_does_not_follow_the_notice_length_down(): gates silently start admitting the band they just vacated. """ assert MIN_FULLTEXT_CHARS >= 1000 + + +class TestBioCFullTextProvider: + """Test BioC XML fulltext provider.""" + + def test_provider_name(self): + """Test the provider name.""" + from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider + + assert BioCFullTextProvider.name() == "bioc" + + def test_requires_pmid(self, config): + """Test that provider requires PMID.""" + from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider + + result = BioCFullTextProvider().locate(ReferenceIdentifiers(doi="10.1234/test"), config) + assert result is None + + @patch("linkml_reference_validator.etl.fulltext.bioc.requests.get") + def test_bioc_fetch_success(self, mock_get, config): + """Test successful BioC fulltext fetch.""" + from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider + + long_intro = "This is the introduction paragraph. " * 20 + long_results = "This is the results section with detailed findings. " * 20 + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.text = f""" + + + + {long_intro} + + + {long_results} + + + """ + mock_get.return_value = mock_response + + result = BioCFullTextProvider().locate(ReferenceIdentifiers(pmid="12345678"), config) + + assert result is not None + assert result.text is not None + assert "introduction paragraph" in result.text + assert "results section" in result.text + assert result.provider == "bioc" + + @patch("linkml_reference_validator.etl.fulltext.bioc.requests.get") + def test_bioc_fetch_not_found(self, mock_get, config): + """Test BioC fetch when article not in OA subset.""" + from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider + + mock_response = MagicMock() + mock_response.status_code = 404 + mock_get.return_value = mock_response + + result = BioCFullTextProvider().locate(ReferenceIdentifiers(pmid="99999999"), config) + assert result is None + + +class TestEuropePMCFullTextProvider: + """Test Europe PMC fulltext provider.""" + + def test_provider_name(self): + """Test the provider name.""" + from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider + + assert EuropePMCFullTextProvider.name() == "epmc" + + def test_requires_pmid(self, config): + """Test that provider requires PMID.""" + from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider + + result = EuropePMCFullTextProvider().locate(ReferenceIdentifiers(doi="10.1234/test"), config) + assert result is None + + @patch("linkml_reference_validator.etl.fulltext.epmc.requests.get") + def test_europepmc_fetch_success(self, mock_get, config): + """Test fetching fulltext via Europe PMC.""" + from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider + + long_text = "This is the full article text from Europe PMC. " * 25 + + search_response = MagicMock() + search_response.status_code = 200 + search_response.json.return_value = { + "resultList": { + "result": [{ + "pmid": "12345678", + "pmcid": "PMC123456", + "isOpenAccess": "Y", + "license": "cc-by", + }] + } + } + + fulltext_response = MagicMock() + fulltext_response.status_code = 200 + fulltext_response.text = f""" +
+ + +

{long_text}

+
+ +
""" + + mock_get.side_effect = [search_response, fulltext_response] + + result = EuropePMCFullTextProvider().locate(ReferenceIdentifiers(pmid="12345678"), config) + + assert result is not None + assert result.text is not None + assert "Europe PMC" in result.text + assert result.provider == "epmc" + + @patch("linkml_reference_validator.etl.fulltext.epmc.requests.get") + def test_europepmc_not_open_access(self, mock_get, config): + """Test Europe PMC when article is not open access.""" + from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider + + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "resultList": { + "result": [{ + "pmid": "12345678", + "isOpenAccess": "N", + }] + } + } + mock_get.return_value = mock_response + + result = EuropePMCFullTextProvider().locate(ReferenceIdentifiers(pmid="12345678"), config) + assert result is None + + +class TestProviderRegistration: + """Test that new providers are properly registered.""" + + def test_bioc_registered(self): + """Test BioC provider is registered.""" + provider = FullTextProviderRegistry.get("bioc") + assert provider is not None + + def test_epmc_registered(self): + """Test Europe PMC provider is registered.""" + provider = FullTextProviderRegistry.get("epmc") + assert provider is not None diff --git a/tests/test_fulltext_strategies.py b/tests/test_fulltext_strategies.py deleted file mode 100644 index 61a9e88..0000000 --- a/tests/test_fulltext_strategies.py +++ /dev/null @@ -1,345 +0,0 @@ -"""Tests for enhanced fulltext retrieval strategies. - -These tests verify the BioC XML, Europe PMC, Unpaywall, and identifier -conversion utilities. -""" - -import pytest -from unittest.mock import Mock, patch - -from linkml_reference_validator.etl.fulltext_strategies import ( - FulltextStrategy, - BioCStrategy, - EuropePMCStrategy, - UnpaywallStrategy, - IdentifierConverter, - FulltextResult, -) - - -class TestFulltextResult: - """Test the FulltextResult data structure.""" - - def test_fulltext_result_creation(self): - """Test creating a basic FulltextResult.""" - result = FulltextResult( - content="This is the full text content.", - source="bioc", - content_type="full_text", - ) - assert result.content == "This is the full text content." - assert result.source == "bioc" - assert result.content_type == "full_text" - assert result.success is True - - def test_fulltext_result_failure(self): - """Test creating a failed FulltextResult.""" - result = FulltextResult( - content=None, - source="unpaywall", - content_type="unavailable", - success=False, - error_message="No open access version found", - ) - assert result.content is None - assert result.success is False - assert "No open access" in result.error_message - - -class TestBioCStrategy: - """Test BioC XML fulltext retrieval.""" - - def test_bioc_url_construction(self): - """Test the BioC URL is correctly constructed.""" - strategy = BioCStrategy() - url = strategy._build_url("12345678") - assert "12345678" in url - assert "BioC_xml" in url - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_bioc_fetch_success(self, mock_get): - """Test successful BioC fulltext fetch.""" - # Need to provide enough text to pass the 500 char minimum - long_intro = "This is the introduction paragraph. " * 20 - long_results = "This is the results section with detailed findings. " * 20 - mock_response = Mock() - mock_response.status_code = 200 - mock_response.text = f""" - - - - {long_intro} - - - {long_results} - - - """ - mock_get.return_value = mock_response - - strategy = BioCStrategy() - result = strategy.fetch("12345678") - - assert result.success is True - assert "introduction paragraph" in result.content - assert "results section" in result.content - assert result.source == "bioc" - assert result.content_type == "full_text_bioc" - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_bioc_fetch_not_found(self, mock_get): - """Test BioC fetch when article not in OA subset.""" - mock_response = Mock() - mock_response.status_code = 404 - mock_get.return_value = mock_response - - strategy = BioCStrategy() - result = strategy.fetch("99999999") - - assert result.success is False - assert result.content is None - - -class TestEuropePMCStrategy: - """Test Europe PMC fulltext retrieval.""" - - def test_europepmc_api_url(self): - """Test Europe PMC API URL construction.""" - strategy = EuropePMCStrategy() - url = strategy._build_search_url("12345678") - assert "europepmc" in url - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_europepmc_fetch_by_pmid(self, mock_get): - """Test fetching fulltext via Europe PMC by PMID.""" - # Mock search response - search_response = Mock() - search_response.status_code = 200 - search_response.json.return_value = { - "resultList": { - "result": [{ - "pmid": "12345678", - "pmcid": "PMC123456", - "isOpenAccess": "Y", - }] - } - } - - # Mock fulltext response - needs enough text to pass 500 char minimum - long_text = "This is the full article text from Europe PMC. " * 20 - fulltext_response = Mock() - fulltext_response.status_code = 200 - fulltext_response.text = f""" -
- - -

{long_text}

-
- -
""" - - mock_get.side_effect = [search_response, fulltext_response] - - strategy = EuropePMCStrategy() - result = strategy.fetch("12345678") - - assert result.success is True - assert "Europe PMC" in result.content - assert result.source == "europepmc" - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_europepmc_not_open_access(self, mock_get): - """Test Europe PMC when article is not open access.""" - mock_response = Mock() - mock_response.status_code = 200 - mock_response.json.return_value = { - "resultList": { - "result": [{ - "pmid": "12345678", - "isOpenAccess": "N", - }] - } - } - mock_get.return_value = mock_response - - strategy = EuropePMCStrategy() - result = strategy.fetch("12345678") - - assert result.success is False - - -class TestUnpaywallStrategy: - """Test Unpaywall API for open access papers.""" - - def test_unpaywall_url_construction(self): - """Test Unpaywall API URL is correctly constructed.""" - strategy = UnpaywallStrategy(email="test@example.com") - url = strategy._build_url("10.1234/example.doi") - assert "api.unpaywall.org" in url - assert "10.1234/example.doi" in url - assert "test@example.com" in url - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_unpaywall_fetch_open_access(self, mock_get): - """Test Unpaywall finding an open access version.""" - mock_response = Mock() - mock_response.status_code = 200 - mock_response.json.return_value = { - "is_oa": True, - "best_oa_location": { - "url": "https://example.com/paper.pdf", - "url_for_pdf": "https://example.com/paper.pdf", - "license": "cc-by", - "version": "publishedVersion", - }, - "oa_locations": [ - { - "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC123456/", - "pmh_id": "oai:pubmedcentral.nih.gov:123456", - } - ] - } - mock_get.return_value = mock_response - - strategy = UnpaywallStrategy(email="test@example.com") - result = strategy.fetch("10.1234/example.doi") - - assert result.success is True - assert result.source == "unpaywall" - assert result.metadata["is_oa"] is True - assert "pdf" in result.metadata.get("pdf_url", "") - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_unpaywall_not_open_access(self, mock_get): - """Test Unpaywall when article is not open access.""" - mock_response = Mock() - mock_response.status_code = 200 - mock_response.json.return_value = { - "is_oa": False, - "best_oa_location": None, - } - mock_get.return_value = mock_response - - strategy = UnpaywallStrategy(email="test@example.com") - result = strategy.fetch("10.1234/closed.doi") - - assert result.success is False - assert "not open access" in result.error_message.lower() - - -class TestIdentifierConverter: - """Test identifier conversion utilities.""" - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_doi_to_pmid(self, mock_get): - """Test converting DOI to PMID.""" - mock_response = Mock() - mock_response.status_code = 200 - mock_response.json.return_value = { - "records": [{ - "pmid": "12345678", - "pmcid": "PMC654321", - "doi": "10.1234/example", - }] - } - mock_get.return_value = mock_response - - converter = IdentifierConverter() - pmid = converter.doi_to_pmid("10.1234/example") - assert pmid == "12345678" - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_pmid_to_doi(self, mock_get): - """Test converting PMID to DOI.""" - mock_response = Mock() - mock_response.status_code = 200 - mock_response.json.return_value = { - "result": { - "12345678": { - "articleids": [ - {"idtype": "pubmed", "value": "12345678"}, - {"idtype": "doi", "value": "10.1234/example"}, - ] - } - } - } - mock_get.return_value = mock_response - - converter = IdentifierConverter() - doi = converter.pmid_to_doi("12345678") - assert doi == "10.1234/example" - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_pmid_to_pmcid(self, mock_get): - """Test converting PMID to PMCID.""" - mock_response = Mock() - mock_response.status_code = 200 - mock_response.json.return_value = { - "records": [{ - "pmid": "12345678", - "pmcid": "PMC654321", - }] - } - mock_get.return_value = mock_response - - converter = IdentifierConverter() - pmcid = converter.pmid_to_pmcid("12345678") - assert pmcid == "PMC654321" - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_pmcid_to_pmid(self, mock_get): - """Test converting PMCID to PMID.""" - mock_response = Mock() - mock_response.status_code = 200 - mock_response.json.return_value = { - "result": { - "uids": ["654321"], - "654321": { - "articleids": [ - {"idtype": "pmid", "value": "12345678"}, - ] - } - } - } - mock_get.return_value = mock_response - - converter = IdentifierConverter() - pmid = converter.pmcid_to_pmid("PMC654321") - assert pmid == "12345678" - - -class TestFulltextStrategyChain: - """Test chaining multiple fulltext strategies.""" - - @patch("linkml_reference_validator.etl.fulltext_strategies.requests.get") - def test_strategy_chain_fallback(self, mock_get): - """Test that strategies fall back when one fails.""" - # First strategy (BioC) fails - bioc_response = Mock() - bioc_response.status_code = 404 - - # Second strategy (Europe PMC) succeeds - long_text = "Europe PMC text with enough content to pass validation. " * 20 - europepmc_search = Mock() - europepmc_search.status_code = 200 - europepmc_search.json.return_value = { - "resultList": { - "result": [{ - "pmid": "12345678", - "pmcid": "PMC123456", - "isOpenAccess": "Y", - }] - } - } - europepmc_fulltext = Mock() - europepmc_fulltext.status_code = 200 - europepmc_fulltext.text = f"

{long_text}

" - - mock_get.side_effect = [bioc_response, europepmc_search, europepmc_fulltext] - - from linkml_reference_validator.etl.fulltext_strategies import FulltextFetcher - fetcher = FulltextFetcher(email="test@example.com") - result = fetcher.fetch_fulltext_for_pmid("12345678") - - assert result.success is True - assert result.source == "europepmc"