diff --git a/src/linkml_reference_validator/etl/fulltext/__init__.py b/src/linkml_reference_validator/etl/fulltext/__init__.py
index f309ef9..4136b8b 100644
--- a/src/linkml_reference_validator/etl/fulltext/__init__.py
+++ b/src/linkml_reference_validator/etl/fulltext/__init__.py
@@ -6,18 +6,22 @@
)
# Import providers to register them
-from linkml_reference_validator.etl.fulltext.pmc import PMCFullTextProvider
+from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider
+from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider
from linkml_reference_validator.etl.fulltext.epmc_preprint import EuropePMCPreprintProvider
-from linkml_reference_validator.etl.fulltext.unpaywall import UnpaywallProvider
from linkml_reference_validator.etl.fulltext.openalex import OpenAlexProvider
+from linkml_reference_validator.etl.fulltext.pmc import PMCFullTextProvider
+from linkml_reference_validator.etl.fulltext.unpaywall import UnpaywallProvider
from linkml_reference_validator.etl.fulltext.zotero import ZoteroFullTextProvider
__all__ = [
+ "BioCFullTextProvider",
+ "EuropePMCFullTextProvider",
+ "EuropePMCPreprintProvider",
"FullTextProvider",
"FullTextProviderRegistry",
+ "OpenAlexProvider",
"PMCFullTextProvider",
- "EuropePMCPreprintProvider",
"UnpaywallProvider",
- "OpenAlexProvider",
"ZoteroFullTextProvider",
]
diff --git a/src/linkml_reference_validator/etl/fulltext/bioc.py b/src/linkml_reference_validator/etl/fulltext/bioc.py
new file mode 100644
index 0000000..f578ec9
--- /dev/null
+++ b/src/linkml_reference_validator/etl/fulltext/bioc.py
@@ -0,0 +1,117 @@
+"""BioC XML full-text provider.
+
+Fetches structured full text from the NCBI BioNLP BioC API, which provides
+clean paragraph-level text for articles in the PMC Open Access subset.
+"""
+
+import logging
+import time
+from typing import Optional
+
+import requests # type: ignore
+from bs4 import BeautifulSoup # type: ignore
+
+from linkml_reference_validator.models import (
+ FullTextLocation,
+ ReferenceIdentifiers,
+ ReferenceValidationConfig,
+)
+from linkml_reference_validator.etl.fulltext.base import (
+ FullTextProvider,
+ FullTextProviderRegistry,
+)
+from linkml_reference_validator.etl.extract import MIN_FULLTEXT_CHARS
+
+logger = logging.getLogger(__name__)
+
+BIOC_URL = "https://www.ncbi.nlm.nih.gov/research/bionlp/RESTful/pmcoa.cgi/BioC_xml/{pmid}/ascii"
+
+
+@FullTextProviderRegistry.register
+class BioCFullTextProvider(FullTextProvider):
+ """Fetch full text via the NCBI BioNLP BioC XML API.
+
+ The BioC API provides structured, clean paragraph text for articles in
+ the PMC Open Access subset. It returns text with passage-level structure
+ (title, abstract, body sections) without inline markup.
+
+ Examples:
+ >>> BioCFullTextProvider.name()
+ 'bioc'
+ """
+
+ @classmethod
+ def name(cls) -> str:
+ return "bioc"
+
+ def locate(
+ self, ids: ReferenceIdentifiers, config: ReferenceValidationConfig
+ ) -> Optional[FullTextLocation]:
+ if not ids.pmid:
+ return None
+
+ time.sleep(config.rate_limit_delay)
+
+ url = BIOC_URL.format(pmid=ids.pmid)
+ try:
+ response = requests.get(url, timeout=30)
+ except requests.RequestException as exc:
+ logger.debug(f"BioC request failed for PMID:{ids.pmid}: {exc}")
+ return None
+
+ if response.status_code == 404:
+ logger.debug(f"PMID:{ids.pmid} not in PMC Open Access subset")
+ return None
+ if response.status_code != 200:
+ logger.debug(f"BioC returned {response.status_code} for PMID:{ids.pmid}")
+ return None
+
+ text = self._extract_text(response.text)
+ if not text or len(text) < MIN_FULLTEXT_CHARS:
+ logger.debug(f"BioC returned insufficient text for PMID:{ids.pmid}")
+ return None
+
+ return FullTextLocation(
+ text=text,
+ format_hint="text",
+ oa_status="green",
+ provider="bioc",
+ )
+
+ def _extract_text(self, xml_content: str) -> Optional[str]:
+ """Extract paragraph text from BioC XML.
+
+ Args:
+ xml_content: BioC XML response
+
+ Returns:
+ Concatenated passage text, or None if parsing fails
+
+ Examples:
+ >>> provider = BioCFullTextProvider()
+ >>> xml = '''
+ ... Introduction text.
+ ... Methods section.
+ ... '''
+ >>> provider._extract_text(xml)
+ 'Introduction text.\\n\\nMethods section.'
+ """
+ try:
+ soup = BeautifulSoup(xml_content, "xml")
+ except Exception as exc:
+ logger.debug(f"Failed to parse BioC XML: {exc}")
+ return None
+
+ passages = soup.find_all("passage")
+ if not passages:
+ return None
+
+ texts = []
+ for passage in passages:
+ text_elem = passage.find("text")
+ if text_elem and text_elem.string:
+ text = text_elem.string.strip()
+ if text:
+ texts.append(text)
+
+ return "\n\n".join(texts) if texts else None
diff --git a/src/linkml_reference_validator/etl/fulltext/epmc.py b/src/linkml_reference_validator/etl/fulltext/epmc.py
new file mode 100644
index 0000000..a494b69
--- /dev/null
+++ b/src/linkml_reference_validator/etl/fulltext/epmc.py
@@ -0,0 +1,182 @@
+"""Europe PMC full-text provider.
+
+Fetches full-text XML for open access articles from the Europe PMC API.
+This complements EuropePMCPreprintProvider, which handles preprints only.
+"""
+
+import logging
+import time
+from typing import Optional
+
+import requests # type: ignore
+from bs4 import BeautifulSoup # type: ignore
+
+from linkml_reference_validator.models import (
+ FullTextLocation,
+ ReferenceIdentifiers,
+ ReferenceValidationConfig,
+)
+from linkml_reference_validator.etl.fulltext.base import (
+ FullTextProvider,
+ FullTextProviderRegistry,
+)
+from linkml_reference_validator.etl.extract import MIN_FULLTEXT_CHARS
+
+logger = logging.getLogger(__name__)
+
+EUROPEPMC_SEARCH_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest/search"
+EUROPEPMC_FULLTEXT_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest/{source}/{id}/fullTextXML"
+
+
+@FullTextProviderRegistry.register
+class EuropePMCFullTextProvider(FullTextProvider):
+ """Fetch full-text XML for open access articles from Europe PMC.
+
+ Queries the Europe PMC search API to find open access articles by PMID,
+ then fetches full-text XML when available. This provider handles regular
+ OA articles; preprints are handled by EuropePMCPreprintProvider.
+
+ Examples:
+ >>> EuropePMCFullTextProvider.name()
+ 'epmc'
+ """
+
+ @classmethod
+ def name(cls) -> str:
+ return "epmc"
+
+ def locate(
+ self, ids: ReferenceIdentifiers, config: ReferenceValidationConfig
+ ) -> Optional[FullTextLocation]:
+ if not ids.pmid:
+ return None
+
+ time.sleep(config.rate_limit_delay)
+
+ params = {
+ "query": f"EXT_ID:{ids.pmid} AND SRC:MED",
+ "format": "json",
+ "resultType": "core",
+ "pageSize": "1",
+ "email": config.email,
+ }
+
+ try:
+ response = requests.get(EUROPEPMC_SEARCH_URL, params=params, timeout=30)
+ except requests.RequestException as exc:
+ logger.debug(f"Europe PMC search failed for PMID:{ids.pmid}: {exc}")
+ return None
+
+ if response.status_code != 200:
+ logger.debug(f"Europe PMC returned {response.status_code} for PMID:{ids.pmid}")
+ return None
+
+ try:
+ data = response.json()
+ except ValueError:
+ logger.debug(f"Invalid JSON from Europe PMC for PMID:{ids.pmid}")
+ return None
+
+ result = self._find_oa_result(data)
+ if not result:
+ return None
+
+ pmcid = result.get("pmcid")
+ if not pmcid:
+ logger.debug(f"No PMCID in Europe PMC result for PMID:{ids.pmid}")
+ return None
+
+ text = self._fetch_fulltext_xml(pmcid, config)
+ if not text or len(text) < MIN_FULLTEXT_CHARS:
+ return None
+
+ return FullTextLocation(
+ text=text,
+ format_hint="text",
+ oa_status="green",
+ license=result.get("license"),
+ provider="epmc",
+ )
+
+ def _find_oa_result(self, data: dict) -> Optional[dict]:
+ """Find an open access result from search response.
+
+ Args:
+ data: Europe PMC search response JSON
+
+ Returns:
+ First OA result dict, or None
+ """
+ results = data.get("resultList", {}).get("result", [])
+ for result in results:
+ if not isinstance(result, dict):
+ continue
+ if result.get("isOpenAccess") == "Y":
+ return result
+ return None
+
+ def _fetch_fulltext_xml(
+ self, pmcid: str, config: ReferenceValidationConfig
+ ) -> Optional[str]:
+ """Fetch and extract text from Europe PMC fullTextXML endpoint.
+
+ Args:
+ pmcid: PMC ID (with or without PMC prefix)
+ config: Configuration for rate limiting
+
+ Returns:
+ Extracted body text, or None
+ """
+ pmcid_clean = pmcid.replace("PMC", "")
+ url = EUROPEPMC_FULLTEXT_URL.format(source="PMC", id=pmcid_clean)
+
+ time.sleep(config.rate_limit_delay)
+
+ try:
+ response = requests.get(url, timeout=30)
+ except requests.RequestException as exc:
+ logger.debug(f"Europe PMC fulltext fetch failed for {pmcid}: {exc}")
+ return None
+
+ if response.status_code != 200:
+ logger.debug(f"Europe PMC fulltext returned {response.status_code} for {pmcid}")
+ return None
+
+ return self._extract_body_text(response.text)
+
+ def _extract_body_text(self, xml_content: str) -> Optional[str]:
+ """Extract body paragraphs from JATS XML.
+
+ Args:
+ xml_content: JATS XML content
+
+ Returns:
+ Concatenated paragraph text, or None
+
+ Examples:
+ >>> provider = EuropePMCFullTextProvider()
+ >>> xml = 'Text here.
'
+ >>> provider._extract_body_text(xml)
+ 'Text here.'
+ """
+ try:
+ soup = BeautifulSoup(xml_content, "xml")
+ except Exception as exc:
+ logger.debug(f"Failed to parse Europe PMC XML: {exc}")
+ return None
+
+ body = soup.find("body")
+ if not body:
+ return None
+
+ paragraphs = body.find_all("p")
+ if not paragraphs:
+ return None
+
+ texts = []
+ for p in paragraphs:
+ text = p.get_text().strip()
+ if text:
+ texts.append(text)
+
+ return "\n\n".join(texts) if texts else None
diff --git a/tests/test_fulltext_providers.py b/tests/test_fulltext_providers.py
index bf628d9..7c9f868 100644
--- a/tests/test_fulltext_providers.py
+++ b/tests/test_fulltext_providers.py
@@ -354,3 +354,153 @@ def test_the_floor_does_not_follow_the_notice_length_down():
gates silently start admitting the band they just vacated.
"""
assert MIN_FULLTEXT_CHARS >= 1000
+
+
+class TestBioCFullTextProvider:
+ """Test BioC XML fulltext provider."""
+
+ def test_provider_name(self):
+ """Test the provider name."""
+ from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider
+
+ assert BioCFullTextProvider.name() == "bioc"
+
+ def test_requires_pmid(self, config):
+ """Test that provider requires PMID."""
+ from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider
+
+ result = BioCFullTextProvider().locate(ReferenceIdentifiers(doi="10.1234/test"), config)
+ assert result is None
+
+ @patch("linkml_reference_validator.etl.fulltext.bioc.requests.get")
+ def test_bioc_fetch_success(self, mock_get, config):
+ """Test successful BioC fulltext fetch."""
+ from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider
+
+ long_intro = "This is the introduction paragraph. " * 20
+ long_results = "This is the results section with detailed findings. " * 20
+ mock_response = MagicMock()
+ mock_response.status_code = 200
+ mock_response.text = f"""
+
+
+
+ {long_intro}
+
+
+ {long_results}
+
+
+ """
+ mock_get.return_value = mock_response
+
+ result = BioCFullTextProvider().locate(ReferenceIdentifiers(pmid="12345678"), config)
+
+ assert result is not None
+ assert result.text is not None
+ assert "introduction paragraph" in result.text
+ assert "results section" in result.text
+ assert result.provider == "bioc"
+
+ @patch("linkml_reference_validator.etl.fulltext.bioc.requests.get")
+ def test_bioc_fetch_not_found(self, mock_get, config):
+ """Test BioC fetch when article not in OA subset."""
+ from linkml_reference_validator.etl.fulltext.bioc import BioCFullTextProvider
+
+ mock_response = MagicMock()
+ mock_response.status_code = 404
+ mock_get.return_value = mock_response
+
+ result = BioCFullTextProvider().locate(ReferenceIdentifiers(pmid="99999999"), config)
+ assert result is None
+
+
+class TestEuropePMCFullTextProvider:
+ """Test Europe PMC fulltext provider."""
+
+ def test_provider_name(self):
+ """Test the provider name."""
+ from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider
+
+ assert EuropePMCFullTextProvider.name() == "epmc"
+
+ def test_requires_pmid(self, config):
+ """Test that provider requires PMID."""
+ from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider
+
+ result = EuropePMCFullTextProvider().locate(ReferenceIdentifiers(doi="10.1234/test"), config)
+ assert result is None
+
+ @patch("linkml_reference_validator.etl.fulltext.epmc.requests.get")
+ def test_europepmc_fetch_success(self, mock_get, config):
+ """Test fetching fulltext via Europe PMC."""
+ from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider
+
+ long_text = "This is the full article text from Europe PMC. " * 25
+
+ search_response = MagicMock()
+ search_response.status_code = 200
+ search_response.json.return_value = {
+ "resultList": {
+ "result": [{
+ "pmid": "12345678",
+ "pmcid": "PMC123456",
+ "isOpenAccess": "Y",
+ "license": "cc-by",
+ }]
+ }
+ }
+
+ fulltext_response = MagicMock()
+ fulltext_response.status_code = 200
+ fulltext_response.text = f"""
+
+
+
+ {long_text}
+
+
+ """
+
+ mock_get.side_effect = [search_response, fulltext_response]
+
+ result = EuropePMCFullTextProvider().locate(ReferenceIdentifiers(pmid="12345678"), config)
+
+ assert result is not None
+ assert result.text is not None
+ assert "Europe PMC" in result.text
+ assert result.provider == "epmc"
+
+ @patch("linkml_reference_validator.etl.fulltext.epmc.requests.get")
+ def test_europepmc_not_open_access(self, mock_get, config):
+ """Test Europe PMC when article is not open access."""
+ from linkml_reference_validator.etl.fulltext.epmc import EuropePMCFullTextProvider
+
+ mock_response = MagicMock()
+ mock_response.status_code = 200
+ mock_response.json.return_value = {
+ "resultList": {
+ "result": [{
+ "pmid": "12345678",
+ "isOpenAccess": "N",
+ }]
+ }
+ }
+ mock_get.return_value = mock_response
+
+ result = EuropePMCFullTextProvider().locate(ReferenceIdentifiers(pmid="12345678"), config)
+ assert result is None
+
+
+class TestProviderRegistration:
+ """Test that new providers are properly registered."""
+
+ def test_bioc_registered(self):
+ """Test BioC provider is registered."""
+ provider = FullTextProviderRegistry.get("bioc")
+ assert provider is not None
+
+ def test_epmc_registered(self):
+ """Test Europe PMC provider is registered."""
+ provider = FullTextProviderRegistry.get("epmc")
+ assert provider is not None