diff --git a/src/linkml_reference_validator/etl/fulltext/__init__.py b/src/linkml_reference_validator/etl/fulltext/__init__.py index f309ef9..4136b8b 100644 --- a/src/linkml_reference_validator/etl/fulltext/__init__.py +++ b/src/linkml_reference_validator/etl/fulltext/__init__.py @@ -6,18 +6,22 @@ ) # Import providers to register them -from linkml_reference_validator.etl.fulltext.pmc import PMCFullTextProvider +from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider +from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider from linkml_reference_validator.etl.fulltext.epmc_preprint import EuropePMCPreprintProvider -from linkml_reference_validator.etl.fulltext.unpaywall import UnpaywallProvider from linkml_reference_validator.etl.fulltext.openalex import OpenAlexProvider +from linkml_reference_validator.etl.fulltext.pmc import PMCFullTextProvider +from linkml_reference_validator.etl.fulltext.unpaywall import UnpaywallProvider from linkml_reference_validator.etl.fulltext.zotero import ZoteroFullTextProvider __all__ = [ + "BioCFullTextProvider", + "EuropePMCFullTextProvider", + "EuropePMCPreprintProvider", "FullTextProvider", "FullTextProviderRegistry", + "OpenAlexProvider", "PMCFullTextProvider", - "EuropePMCPreprintProvider", "UnpaywallProvider", - "OpenAlexProvider", "ZoteroFullTextProvider", ] diff --git a/src/linkml_reference_validator/etl/fulltext/bioc.py b/src/linkml_reference_validator/etl/fulltext/bioc.py new file mode 100644 index 0000000..f578ec9 --- /dev/null +++ b/src/linkml_reference_validator/etl/fulltext/bioc.py @@ -0,0 +1,117 @@ +"""BioC XML full-text provider. + +Fetches structured full text from the NCBI BioNLP BioC API, which provides +clean paragraph-level text for articles in the PMC Open Access subset. +""" + +import logging +import time +from typing import Optional + +import requests # type: ignore +from bs4 import BeautifulSoup # type: ignore + +from linkml_reference_validator.models import ( + FullTextLocation, + ReferenceIdentifiers, + ReferenceValidationConfig, +) +from linkml_reference_validator.etl.fulltext.base import ( + FullTextProvider, + FullTextProviderRegistry, +) +from linkml_reference_validator.etl.extract import MIN_FULLTEXT_CHARS + +logger = logging.getLogger(__name__) + +BIOC_URL = "https://www.ncbi.nlm.nih.gov/research/bionlp/RESTful/pmcoa.cgi/BioC_xml/{pmid}/ascii" + + +@FullTextProviderRegistry.register +class BioCFullTextProvider(FullTextProvider): + """Fetch full text via the NCBI BioNLP BioC XML API. + + The BioC API provides structured, clean paragraph text for articles in + the PMC Open Access subset. It returns text with passage-level structure + (title, abstract, body sections) without inline markup. + + Examples: + >>> BioCFullTextProvider.name() + 'bioc' + """ + + @classmethod + def name(cls) -> str: + return "bioc" + + def locate( + self, ids: ReferenceIdentifiers, config: ReferenceValidationConfig + ) -> Optional[FullTextLocation]: + if not ids.pmid: + return None + + time.sleep(config.rate_limit_delay) + + url = BIOC_URL.format(pmid=ids.pmid) + try: + response = requests.get(url, timeout=30) + except requests.RequestException as exc: + logger.debug(f"BioC request failed for PMID:{ids.pmid}: {exc}") + return None + + if response.status_code == 404: + logger.debug(f"PMID:{ids.pmid} not in PMC Open Access subset") + return None + if response.status_code != 200: + logger.debug(f"BioC returned {response.status_code} for PMID:{ids.pmid}") + return None + + text = self._extract_text(response.text) + if not text or len(text) < MIN_FULLTEXT_CHARS: + logger.debug(f"BioC returned insufficient text for PMID:{ids.pmid}") + return None + + return FullTextLocation( + text=text, + format_hint="text", + oa_status="green", + provider="bioc", + ) + + def _extract_text(self, xml_content: str) -> Optional[str]: + """Extract paragraph text from BioC XML. + + Args: + xml_content: BioC XML response + + Returns: + Concatenated passage text, or None if parsing fails + + Examples: + >>> provider = BioCFullTextProvider() + >>> xml = ''' + ... Introduction text. + ... Methods section. + ... ''' + >>> provider._extract_text(xml) + 'Introduction text.\\n\\nMethods section.' + """ + try: + soup = BeautifulSoup(xml_content, "xml") + except Exception as exc: + logger.debug(f"Failed to parse BioC XML: {exc}") + return None + + passages = soup.find_all("passage") + if not passages: + return None + + texts = [] + for passage in passages: + text_elem = passage.find("text") + if text_elem and text_elem.string: + text = text_elem.string.strip() + if text: + texts.append(text) + + return "\n\n".join(texts) if texts else None diff --git a/src/linkml_reference_validator/etl/fulltext/epmc.py b/src/linkml_reference_validator/etl/fulltext/epmc.py new file mode 100644 index 0000000..a494b69 --- /dev/null +++ b/src/linkml_reference_validator/etl/fulltext/epmc.py @@ -0,0 +1,182 @@ +"""Europe PMC full-text provider. + +Fetches full-text XML for open access articles from the Europe PMC API. +This complements EuropePMCPreprintProvider, which handles preprints only. +""" + +import logging +import time +from typing import Optional + +import requests # type: ignore +from bs4 import BeautifulSoup # type: ignore + +from linkml_reference_validator.models import ( + FullTextLocation, + ReferenceIdentifiers, + ReferenceValidationConfig, +) +from linkml_reference_validator.etl.fulltext.base import ( + FullTextProvider, + FullTextProviderRegistry, +) +from linkml_reference_validator.etl.extract import MIN_FULLTEXT_CHARS + +logger = logging.getLogger(__name__) + +EUROPEPMC_SEARCH_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest/search" +EUROPEPMC_FULLTEXT_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest/{source}/{id}/fullTextXML" + + +@FullTextProviderRegistry.register +class EuropePMCFullTextProvider(FullTextProvider): + """Fetch full-text XML for open access articles from Europe PMC. + + Queries the Europe PMC search API to find open access articles by PMID, + then fetches full-text XML when available. This provider handles regular + OA articles; preprints are handled by EuropePMCPreprintProvider. + + Examples: + >>> EuropePMCFullTextProvider.name() + 'epmc' + """ + + @classmethod + def name(cls) -> str: + return "epmc" + + def locate( + self, ids: ReferenceIdentifiers, config: ReferenceValidationConfig + ) -> Optional[FullTextLocation]: + if not ids.pmid: + return None + + time.sleep(config.rate_limit_delay) + + params = { + "query": f"EXT_ID:{ids.pmid} AND SRC:MED", + "format": "json", + "resultType": "core", + "pageSize": "1", + "email": config.email, + } + + try: + response = requests.get(EUROPEPMC_SEARCH_URL, params=params, timeout=30) + except requests.RequestException as exc: + logger.debug(f"Europe PMC search failed for PMID:{ids.pmid}: {exc}") + return None + + if response.status_code != 200: + logger.debug(f"Europe PMC returned {response.status_code} for PMID:{ids.pmid}") + return None + + try: + data = response.json() + except ValueError: + logger.debug(f"Invalid JSON from Europe PMC for PMID:{ids.pmid}") + return None + + result = self._find_oa_result(data) + if not result: + return None + + pmcid = result.get("pmcid") + if not pmcid: + logger.debug(f"No PMCID in Europe PMC result for PMID:{ids.pmid}") + return None + + text = self._fetch_fulltext_xml(pmcid, config) + if not text or len(text) < MIN_FULLTEXT_CHARS: + return None + + return FullTextLocation( + text=text, + format_hint="text", + oa_status="green", + license=result.get("license"), + provider="epmc", + ) + + def _find_oa_result(self, data: dict) -> Optional[dict]: + """Find an open access result from search response. + + Args: + data: Europe PMC search response JSON + + Returns: + First OA result dict, or None + """ + results = data.get("resultList", {}).get("result", []) + for result in results: + if not isinstance(result, dict): + continue + if result.get("isOpenAccess") == "Y": + return result + return None + + def _fetch_fulltext_xml( + self, pmcid: str, config: ReferenceValidationConfig + ) -> Optional[str]: + """Fetch and extract text from Europe PMC fullTextXML endpoint. + + Args: + pmcid: PMC ID (with or without PMC prefix) + config: Configuration for rate limiting + + Returns: + Extracted body text, or None + """ + pmcid_clean = pmcid.replace("PMC", "") + url = EUROPEPMC_FULLTEXT_URL.format(source="PMC", id=pmcid_clean) + + time.sleep(config.rate_limit_delay) + + try: + response = requests.get(url, timeout=30) + except requests.RequestException as exc: + logger.debug(f"Europe PMC fulltext fetch failed for {pmcid}: {exc}") + return None + + if response.status_code != 200: + logger.debug(f"Europe PMC fulltext returned {response.status_code} for {pmcid}") + return None + + return self._extract_body_text(response.text) + + def _extract_body_text(self, xml_content: str) -> Optional[str]: + """Extract body paragraphs from JATS XML. + + Args: + xml_content: JATS XML content + + Returns: + Concatenated paragraph text, or None + + Examples: + >>> provider = EuropePMCFullTextProvider() + >>> xml = '

Text here.

' + >>> provider._extract_body_text(xml) + 'Text here.' + """ + try: + soup = BeautifulSoup(xml_content, "xml") + except Exception as exc: + logger.debug(f"Failed to parse Europe PMC XML: {exc}") + return None + + body = soup.find("body") + if not body: + return None + + paragraphs = body.find_all("p") + if not paragraphs: + return None + + texts = [] + for p in paragraphs: + text = p.get_text().strip() + if text: + texts.append(text) + + return "\n\n".join(texts) if texts else None diff --git a/tests/test_fulltext_providers.py b/tests/test_fulltext_providers.py index bf628d9..7c9f868 100644 --- a/tests/test_fulltext_providers.py +++ b/tests/test_fulltext_providers.py @@ -354,3 +354,153 @@ def test_the_floor_does_not_follow_the_notice_length_down(): gates silently start admitting the band they just vacated. """ assert MIN_FULLTEXT_CHARS >= 1000 + + +class TestBioCFullTextProvider: + """Test BioC XML fulltext provider.""" + + def test_provider_name(self): + """Test the provider name.""" + from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider + + assert BioCFullTextProvider.name() == "bioc" + + def test_requires_pmid(self, config): + """Test that provider requires PMID.""" + from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider + + result = BioCFullTextProvider().locate(ReferenceIdentifiers(doi="10.1234/test"), config) + assert result is None + + @patch("linkml_reference_validator.etl.fulltext.bioc.requests.get") + def test_bioc_fetch_success(self, mock_get, config): + """Test successful BioC fulltext fetch.""" + from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider + + long_intro = "This is the introduction paragraph. " * 20 + long_results = "This is the results section with detailed findings. " * 20 + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.text = f""" + + + + {long_intro} + + + {long_results} + + + """ + mock_get.return_value = mock_response + + result = BioCFullTextProvider().locate(ReferenceIdentifiers(pmid="12345678"), config) + + assert result is not None + assert result.text is not None + assert "introduction paragraph" in result.text + assert "results section" in result.text + assert result.provider == "bioc" + + @patch("linkml_reference_validator.etl.fulltext.bioc.requests.get") + def test_bioc_fetch_not_found(self, mock_get, config): + """Test BioC fetch when article not in OA subset.""" + from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider + + mock_response = MagicMock() + mock_response.status_code = 404 + mock_get.return_value = mock_response + + result = BioCFullTextProvider().locate(ReferenceIdentifiers(pmid="99999999"), config) + assert result is None + + +class TestEuropePMCFullTextProvider: + """Test Europe PMC fulltext provider.""" + + def test_provider_name(self): + """Test the provider name.""" + from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider + + assert EuropePMCFullTextProvider.name() == "epmc" + + def test_requires_pmid(self, config): + """Test that provider requires PMID.""" + from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider + + result = EuropePMCFullTextProvider().locate(ReferenceIdentifiers(doi="10.1234/test"), config) + assert result is None + + @patch("linkml_reference_validator.etl.fulltext.epmc.requests.get") + def test_europepmc_fetch_success(self, mock_get, config): + """Test fetching fulltext via Europe PMC.""" + from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider + + long_text = "This is the full article text from Europe PMC. " * 25 + + search_response = MagicMock() + search_response.status_code = 200 + search_response.json.return_value = { + "resultList": { + "result": [{ + "pmid": "12345678", + "pmcid": "PMC123456", + "isOpenAccess": "Y", + "license": "cc-by", + }] + } + } + + fulltext_response = MagicMock() + fulltext_response.status_code = 200 + fulltext_response.text = f""" +
+ + +

{long_text}

+
+ +
""" + + mock_get.side_effect = [search_response, fulltext_response] + + result = EuropePMCFullTextProvider().locate(ReferenceIdentifiers(pmid="12345678"), config) + + assert result is not None + assert result.text is not None + assert "Europe PMC" in result.text + assert result.provider == "epmc" + + @patch("linkml_reference_validator.etl.fulltext.epmc.requests.get") + def test_europepmc_not_open_access(self, mock_get, config): + """Test Europe PMC when article is not open access.""" + from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider + + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.json.return_value = { + "resultList": { + "result": [{ + "pmid": "12345678", + "isOpenAccess": "N", + }] + } + } + mock_get.return_value = mock_response + + result = EuropePMCFullTextProvider().locate(ReferenceIdentifiers(pmid="12345678"), config) + assert result is None + + +class TestProviderRegistration: + """Test that new providers are properly registered.""" + + def test_bioc_registered(self): + """Test BioC provider is registered.""" + provider = FullTextProviderRegistry.get("bioc") + assert provider is not None + + def test_epmc_registered(self): + """Test Europe PMC provider is registered.""" + provider = FullTextProviderRegistry.get("epmc") + assert provider is not None