From 3959cd7b03e0050366c1867d33ff60d5103e951f Mon Sep 17 00:00:00 2001 From: engkinandatama Date: Wed, 30 Sep 2026 06:47:59 +0700 Subject: [PATCH] feat(literature): implement Europe PMC adapter, literature MCP tools, and CLI (v0.7.0) --- README.md | 5 +- docs/API.md | 3 + pyproject.toml | 2 +- src/biocontext/adapters.py | 168 +++++++++++++++++++++++++++++++ src/biocontext/cli.py | 16 ++- src/biocontext/config.py | 18 ++++ src/biocontext/resolver.py | 26 +++++ src/biocontext/schemas.py | 23 +++++ src/biocontext/server.py | 32 ++++++ tests/test_literature_adapter.py | 107 ++++++++++++++++++++ 10 files changed, 397 insertions(+), 3 deletions(-) create mode 100644 tests/test_literature_adapter.py diff --git a/README.md b/README.md index d3c68d6..1c7f146 100644 --- a/README.md +++ b/README.md @@ -137,7 +137,7 @@ biocontext cache clear ## 🔌 Available MCP Tools -When connected via MCP, BioContext exposes 10 production-ready biological & clinical tools: +When connected via MCP, BioContext exposes 12 production-ready biological, clinical & literature tools: | MCP Tool | Signature & Parameters | Description | | :--- | :--- | :--- | @@ -150,8 +150,11 @@ When connected via MCP, BioContext exposes 10 production-ready biological & clin | `get_pathway_details`| `st_id: str` | Retrieves descriptive summary and metadata for a Reactome pathway. | | `resolve_disease` | `query: str`, `limit: int = 5` | Resolves disease names, synonyms, or IDs to canonical MONDO Disease Ontology entities. | | `get_target_diseases` | `gene: str`, `limit: int = 10` | Retrieves evidence-backed therapeutic target-disease associations from Open Targets Platform. | +| `get_supporting_publications` | `query: str`, `limit: int = 5` | Retrieves peer-reviewed supporting scientific publications and citations from Europe PMC / PubMed. | +| `get_publication_details` | `identifier: str` | Fetches detailed publication metadata, abstract, and citation counts by PMID, PMCID, or DOI. | | `get_mouse_gene` | `mgi_id: str` | Direct lookup of mouse gene models from MGI. | + --- ## 📊 Empirical Accuracy Benchmark diff --git a/docs/API.md b/docs/API.md index 64640c0..926d7f1 100644 --- a/docs/API.md +++ b/docs/API.md @@ -148,7 +148,10 @@ asyncio.run(main()) | `pathway-info` | `biocontext pathway-info ` | Retrieve Reactome pathway summation. | | `disease` | `biocontext disease [-l INT]` | Resolve disease name or MONDO identifier. | | `targets` | `biocontext targets [-l INT]` | Retrieve Open Targets evidence-backed disease associations. | +| `literature` | `biocontext literature [-l INT]` | Search supporting peer-reviewed publications from Europe PMC / PubMed. | +| `paper` | `biocontext paper ` | Fetch detailed paper metadata and abstract by PMID, PMCID, or DOI. | | `mouse` | `biocontext mouse ` | Lookup mouse gene model via MGI. | | `cache` | `biocontext cache [stats\|clear]` | Inspect or clear SQLite cache. | | `serve` | `biocontext serve` | Start stdio MCP server for AI clients. | + diff --git a/pyproject.toml b/pyproject.toml index b306666..f292011 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "biocontext-mcp" -version = "0.6.0" +version = "0.7.0" description = "Authoritative Biological Entity Resolution & Contextual Intelligence Framework" readme = "README.md" authors = [ diff --git a/src/biocontext/adapters.py b/src/biocontext/adapters.py index 8b90805..73e1bf3 100644 --- a/src/biocontext/adapters.py +++ b/src/biocontext/adapters.py @@ -21,6 +21,8 @@ PathwayContext, PathwayEntity, ProteinEntity, + PublicationEntity, + LiteratureContext, TargetAssociationContext, TargetDiseaseAssociation, TranscriptEntity, @@ -1462,6 +1464,172 @@ async def fetch_target_diseases( return context +class LiteratureAdapter(BaseBioAdapter): + """Adapter for scientific publications via Europe PMC REST API. + Provides authoritative citation metadata, PMID/PMCID/DOI cross-mapping, + and open-access abstracts for biological and clinical research. + """ + + BASE_URL = "https://www.ebi.ac.uk/europepmc/webservices/rest" + + def __init__(self, cache: Optional[SQLiteCache] = None, email: Optional[str] = None): + super().__init__(name="EuropePMC", cache=cache) + self.email = ClientConfig.get_email(email) + self.headers = ClientConfig.get_headers(self.email) + self.rate_limiter = AsyncRateLimiter(requests_per_second=RateLimitConfig.EUROPEPMC_RPS) + + async def resolve_gene(self, query: str, taxon_id: int = 9606) -> Optional[Dict[str, Any]]: + """Literature adapter queries publications rather than gene symbol models.""" + return None + + def _parse_publication(self, item: Dict[str, Any]) -> PublicationEntity: + """Parse Europe PMC raw result item into PublicationEntity.""" + pmid = item.get("pmid") or item.get("id") + pmcid = item.get("pmcid") + doi = item.get("doi") + title = item.get("title", "").strip().rstrip(".") + + # Author parsing + author_string = item.get("authorString", "") + authors = [a.strip() for a in author_string.split(",") if a.strip()] if author_string else [] + + journal_title = item.get("journalTitle") + pub_year = None + if "pubYear" in item and item["pubYear"]: + try: + pub_year = int(item["pubYear"]) + except (ValueError, TypeError): + pass + + abstract = item.get("abstractText") + cited_by = None + if "citedByCount" in item and item["citedByCount"] is not None: + try: + cited_by = int(item["citedByCount"]) + except (ValueError, TypeError): + pass + + url = f"https://europepmc.org/article/MED/{pmid}" if pmid else ( + f"https://doi.org/{doi}" if doi else None + ) + + return PublicationEntity( + pmid=str(pmid) if pmid else None, + pmcid=str(pmcid) if pmcid else None, + doi=str(doi) if doi else None, + title=title or "Untitled Publication", + authors=authors, + journal=journal_title, + pub_year=pub_year, + abstract_text=abstract, + cited_by_count=cited_by, + url=url + ) + + async def fetch_by_id(self, identifier: str) -> Optional[PublicationEntity]: + """Fetch publication metadata by PMID, PMCID, or DOI.""" + clean_id = identifier.strip() + if not clean_id: + return None + + cache_key = f"europepmc:id:{clean_id.lower()}" + cached = self.cache.get("europepmc", cache_key) + if cached: + return PublicationEntity(**cached) + + # Formulate query based on identifier pattern + if clean_id.upper().startswith("PMC"): + query_str = clean_id.upper() + elif clean_id.startswith("10."): + query_str = f"DOI:{clean_id}" + elif clean_id.isdigit(): + query_str = f"EXT_ID:{clean_id} AND SRC:MED" + else: + query_str = f"EXT_ID:{clean_id}" + + + url = f"{self.BASE_URL}/search" + params = { + "query": query_str, + "format": "json", + "pageSize": 1, + "resultType": "core" + } + + await self.rate_limiter.acquire() + try: + async with httpx.AsyncClient(timeout=RateLimitConfig.EUROPEPMC_TIMEOUT_SEC) as client: + resp = await client.get(url, params=params, headers=self.headers) + if resp.status_code != 200: + logger.warning("Europe PMC id query failed | query=%s status=%d", query_str, resp.status_code) + return None + data = resp.json() + except Exception as e: + logger.warning("Europe PMC id query exception | query=%s error=%s", query_str, str(e)) + return None + + results = data.get("resultList", {}).get("result", []) + if not results: + return None + + pub = self._parse_publication(results[0]) + self.cache.set("europepmc", cache_key, pub.model_dump()) + return pub + + async def search_publications(self, query: str, limit: int = 5) -> LiteratureContext: + """Search Europe PMC publications supporting an entity query.""" + clean_query = query.strip() + if not clean_query: + return LiteratureContext(query=query, total_hits=0, publications=[]) + + cache_key = f"europepmc:search:{clean_query.lower()}:{limit}" + cached = self.cache.get("europepmc", cache_key) + if cached: + return LiteratureContext(**cached) + + # Enrich biological search query (prefer peer-reviewed MEDLINE citations) + if clean_query.isdigit(): + formatted_query = f"EXT_ID:{clean_query} AND SRC:MED" + else: + formatted_query = f"({clean_query}) AND (SRC:MED OR SRC:PMC)" + + url = f"{self.BASE_URL}/search" + params = { + "query": formatted_query, + "format": "json", + "pageSize": limit, + "resultType": "core", + "sort": "CITED desc" + } + + await self.rate_limiter.acquire() + try: + async with httpx.AsyncClient(timeout=RateLimitConfig.EUROPEPMC_TIMEOUT_SEC) as client: + resp = await client.get(url, params=params, headers=self.headers) + if resp.status_code != 200: + logger.warning("Europe PMC search failed | query=%s status=%d", formatted_query, resp.status_code) + return LiteratureContext(query=clean_query, total_hits=0, publications=[]) + data = resp.json() + except Exception as e: + logger.warning("Europe PMC search exception | query=%s error=%s", formatted_query, str(e)) + return LiteratureContext(query=clean_query, total_hits=0, publications=[]) + + total_count = int(data.get("hitCount", 0)) + raw_items = data.get("resultList", {}).get("result", []) + publications = [self._parse_publication(item) for item in raw_items] + + context = LiteratureContext( + query=clean_query, + source="Europe PMC", + total_hits=total_count, + publications=publications + ) + + self.cache.set("europepmc", cache_key, context.model_dump()) + return context + + + diff --git a/src/biocontext/cli.py b/src/biocontext/cli.py index 4517a56..3039ab6 100644 --- a/src/biocontext/cli.py +++ b/src/biocontext/cli.py @@ -222,6 +222,19 @@ async def run_cli_async(args: argparse.Namespace) -> int: print(context.model_dump_json(indent=2)) return 0 + elif args.command == "literature": + context = await resolver.get_supporting_publications(query=args.query, limit=args.limit) + print(context.model_dump_json(indent=2)) + return 0 + + elif args.command == "paper": + pub = await resolver.get_publication_details(identifier=args.identifier) + if not pub: + print(json.dumps({"status": "not_found", "identifier": args.identifier}, indent=2)) + return 1 + print(pub.model_dump_json(indent=2)) + return 0 + elif args.command == "cache": cache = resolver.cache if args.cache_action == "clear": @@ -256,7 +269,8 @@ def main(): sys.exit(pytest.main(["tests/", "-v"])) elif args.command in ( "resolve", "batch", "protein", "transcripts", "ortholog", "mouse", - "annotate", "go", "pathway", "pathway-info", "disease", "targets", "cache" + "annotate", "go", "pathway", "pathway-info", "disease", "targets", + "literature", "paper", "cache" ): sys.exit(asyncio.run(run_cli_async(args))) else: diff --git a/src/biocontext/config.py b/src/biocontext/config.py index ec986b0..121f10e 100644 --- a/src/biocontext/config.py +++ b/src/biocontext/config.py @@ -42,6 +42,9 @@ class RateLimitConfig: MONDO_TIMEOUT_SEC: float = 15.0 OPENTARGETS_RPS: float = 10.0 OPENTARGETS_TIMEOUT_SEC: float = 15.0 + EUROPEPMC_RPS: float = 10.0 + EUROPEPMC_TIMEOUT_SEC: float = 15.0 + @@ -179,6 +182,21 @@ def get_headers(cls, email: Optional[str] = None) -> Dict[str, str]: {"flags": ["--limit"], "type": int, "default": 10, "help": "Maximum associations to return (default: 10)"} ] }, + { + "name": "literature", + "help": "Search supporting scientific publications from Europe PMC / PubMed", + "arguments": [ + {"flags": ["query"], "help": "Gene symbol, disease name, or scientific query (e.g. TP53)"}, + {"flags": ["--limit"], "type": int, "default": 5, "help": "Maximum publications to return (default: 5)"} + ] + }, + { + "name": "paper", + "help": "Fetch detailed scientific paper metadata by PMID, PMCID, or DOI", + "arguments": [ + {"flags": ["identifier"], "help": "Publication ID (e.g. 30514107, PMC6280721, or 10.1038/...)"} + ] + }, { diff --git a/src/biocontext/resolver.py b/src/biocontext/resolver.py index b976ffa..3f04bb2 100644 --- a/src/biocontext/resolver.py +++ b/src/biocontext/resolver.py @@ -6,6 +6,7 @@ EnsemblAdapter, GeneOntologyAdapter, HGNCAdapter, + LiteratureAdapter, MGIAdapter, MondoAdapter, NCBIAdapter, @@ -21,9 +22,11 @@ DiseaseEntity, FunctionalAnnotation, GOAnnotation, + LiteratureContext, MatchReason, PathwayContext, PathwayEntity, + PublicationEntity, ResolutionContext, ResolutionResult, TargetAssociationContext, @@ -53,6 +56,8 @@ def __init__( self.reactome = ReactomeAdapter(cache=self.cache, email=email) self.mondo = MondoAdapter(cache=self.cache, email=email) self.opentargets = OpenTargetsAdapter(cache=self.cache, email=email) + self.literature = LiteratureAdapter(cache=self.cache, email=email) + @@ -573,4 +578,25 @@ async def get_target_diseases(self, gene_query: str, limit: int = 10) -> Optiona limit=limit ) + async def get_supporting_publications(self, query: str, limit: int = 5) -> LiteratureContext: + """Fetch authoritative supporting scientific publications for a gene, disease, or biomedical query.""" + clean_q = query.strip() + if not clean_q: + return LiteratureContext(query=query, total_hits=0, publications=[]) + + # If query is a gene, try resolving to approved symbol to enrich query + resolved_sym = None + if not clean_q.isdigit() and not clean_q.upper().startswith("PMC") and not clean_q.startswith("10."): + res = await self.resolve(clean_q) + if res and res.resolved_entity: + resolved_sym = res.resolved_entity.symbol + + search_term = resolved_sym or clean_q + return await self.literature.search_publications(query=search_term, limit=limit) + + async def get_publication_details(self, identifier: str) -> Optional[PublicationEntity]: + """Fetch detailed publication metadata by PMID, PMCID, or DOI.""" + return await self.literature.fetch_by_id(identifier) + + diff --git a/src/biocontext/schemas.py b/src/biocontext/schemas.py index 59d40e8..dce1529 100644 --- a/src/biocontext/schemas.py +++ b/src/biocontext/schemas.py @@ -186,3 +186,26 @@ class TargetAssociationContext(BaseModel): total_associations: int = Field(0, description="Total number of associated diseases found") associations: List[TargetDiseaseAssociation] = Field(default_factory=list, description="Top ranked disease associations") + +class PublicationEntity(BaseModel): + """Authoritative scientific literature publication from Europe PMC / PubMed.""" + pmid: Optional[str] = Field(None, description="PubMed Identifier (e.g. '30514107')") + pmcid: Optional[str] = Field(None, description="PubMed Central Open Access ID (e.g. 'PMC6280721')") + doi: Optional[str] = Field(None, description="Digital Object Identifier (e.g. '10.1038/s41586-018-0774-4')") + title: str = Field(..., description="Title of the research publication") + authors: List[str] = Field(default_factory=list, description="List of author names") + journal: Optional[str] = Field(None, description="Journal title / abbreviation") + pub_year: Optional[int] = Field(None, description="Year of publication") + abstract_text: Optional[str] = Field(None, description="Abstract summary text") + cited_by_count: Optional[int] = Field(None, description="Number of scientific citations") + url: Optional[str] = Field(None, description="Direct URL to Europe PMC / PubMed entry") + + +class LiteratureContext(BaseModel): + """Curated collection of supporting scientific publications for an entity or query.""" + query: str = Field(..., description="Input gene symbol, disease name, or query string") + source: str = Field("Europe PMC", description="Authoritative literature index queried") + total_hits: int = Field(0, description="Total scientific publications matching query") + publications: List[PublicationEntity] = Field(default_factory=list, description="Top ranked supporting publications") + + diff --git a/src/biocontext/server.py b/src/biocontext/server.py index 7201124..ecb2aa2 100644 --- a/src/biocontext/server.py +++ b/src/biocontext/server.py @@ -273,6 +273,38 @@ async def get_target_diseases(gene: str, limit: int = 10) -> str: return context.model_dump_json(indent=2) +@mcp.tool() +async def get_supporting_publications(query: str, limit: int = 5) -> str: + """Fetch authoritative supporting scientific literature from Europe PMC / PubMed. + + Args: + query: Gene symbol (e.g. 'TP53'), disease name, or scientific query string. + limit: Maximum number of peer-reviewed publications to return (default: 5). + + Returns: + JSON string containing publication titles, journal, year, authors, abstract, and citation counts. + """ + context = await resolver.get_supporting_publications(query=query, limit=limit) + return context.model_dump_json(indent=2) + + +@mcp.tool() +async def get_publication_details(identifier: str) -> str: + """Fetch detailed metadata for a specific scientific paper by PMID, PMCID, or DOI. + + Args: + identifier: Publication identifier (e.g. PMID '30514107', PMCID 'PMC6280721', or DOI '10.1038/s41586-018-0774-4'). + + Returns: + JSON string containing full paper metadata, abstract, and direct web links. + """ + pub = await resolver.get_publication_details(identifier=identifier) + if not pub: + return '{"status": "not_found", "identifier": "%s"}' % identifier + + return pub.model_dump_json(indent=2) + + def main(): """Run MCP server over stdio.""" mcp.run(transport="stdio") diff --git a/tests/test_literature_adapter.py b/tests/test_literature_adapter.py new file mode 100644 index 0000000..d7fa0e8 --- /dev/null +++ b/tests/test_literature_adapter.py @@ -0,0 +1,107 @@ +"""Tests for Europe PMC Literature Adapter, Literature Context, and Server Tools.""" + +import json +import pytest +from biocontext.adapters import LiteratureAdapter +from biocontext.base import SQLiteCache +from biocontext.resolver import EntityResolver +from biocontext.schemas import LiteratureContext, PublicationEntity +from biocontext.server import get_publication_details, get_supporting_publications + + +@pytest.fixture +def temp_cache(tmp_path): + db_file = tmp_path / "test_literature_cache.db" + return SQLiteCache(db_path=str(db_file)) + + +@pytest.fixture +def literature_adapter(temp_cache): + return LiteratureAdapter(cache=temp_cache) + + +@pytest.fixture +def resolver(temp_cache): + return EntityResolver(cache=temp_cache) + + +@pytest.mark.asyncio +async def test_fetch_publication_by_pmid(literature_adapter): + """Test fetching publication metadata by PubMed ID (PMID: 30514107).""" + pub = await literature_adapter.fetch_by_id("30514107") + assert pub is not None + assert isinstance(pub, PublicationEntity) + assert pub.pmid == "30514107" + assert pub.title is not None + assert len(pub.authors) > 0 + assert pub.pub_year is not None + + +@pytest.mark.asyncio +async def test_fetch_publication_by_pmcid(literature_adapter): + """Test fetching publication metadata by PMC ID (PMC13589805).""" + pub = await literature_adapter.fetch_by_id("PMC13589805") + assert pub is not None + assert isinstance(pub, PublicationEntity) + assert pub.pmcid == "PMC13589805" + + +@pytest.mark.asyncio +async def test_fetch_publication_by_doi(literature_adapter): + """Test fetching publication metadata by DOI.""" + pub = await literature_adapter.fetch_by_id("10.1096/fj.201801695r") + assert pub is not None + assert isinstance(pub, PublicationEntity) + assert pub.doi == "10.1096/fj.201801695r" + assert "CELF1" in pub.title or "p53" in pub.title + + + +@pytest.mark.asyncio +async def test_search_publications_tp53(literature_adapter): + """Test searching Europe PMC for TP53 citations.""" + context = await literature_adapter.search_publications("TP53", limit=3) + assert context is not None + assert isinstance(context, LiteratureContext) + assert context.total_hits > 1000 + assert len(context.publications) == 3 + + first = context.publications[0] + assert first.title is not None + assert first.cited_by_count is not None and first.cited_by_count >= 0 + + +@pytest.mark.asyncio +async def test_resolver_get_supporting_publications(resolver): + """Test EntityResolver.get_supporting_publications for gene symbol BRCA1.""" + context = await resolver.get_supporting_publications("BRCA1", limit=3) + assert context is not None + assert isinstance(context, LiteratureContext) + assert len(context.publications) > 0 + + +@pytest.mark.asyncio +async def test_resolver_get_publication_details(resolver): + """Test EntityResolver.get_publication_details.""" + pub = await resolver.get_publication_details("30514107") + assert pub is not None + assert pub.pmid == "30514107" + + +@pytest.mark.asyncio +async def test_mcp_tool_get_supporting_publications(): + """Test MCP get_supporting_publications tool serialization.""" + res_str = await get_supporting_publications("TP53", limit=2) + assert res_str is not None + data = json.loads(res_str) + assert "publications" in data + assert len(data["publications"]) > 0 + + +@pytest.mark.asyncio +async def test_mcp_tool_get_publication_details(): + """Test MCP get_publication_details tool serialization.""" + res_str = await get_publication_details("30514107") + assert res_str is not None + data = json.loads(res_str) + assert data.get("pmid") == "30514107"