Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -137,7 +137,7 @@ biocontext cache clear

## 🔌 Available MCP Tools

When connected via MCP, BioContext exposes 8 production-ready biological tools:
When connected via MCP, BioContext exposes 10 production-ready biological & clinical tools:

| MCP Tool | Signature & Parameters | Description |
| :--- | :--- | :--- |
Expand All @@ -148,6 +148,8 @@ When connected via MCP, BioContext exposes 8 production-ready biological tools:
| `get_go_term` | `go_id: str` | Inspects a specific Gene Ontology term definition and aspect. |
| `get_pathways` | `query: str`, `taxon_id: int = 9606`, `species: str = "Homo sapiens"`, `limit: int = 10` | Maps genes/proteins to biological pathways via Reactome. |
| `get_pathway_details`| `st_id: str` | Retrieves descriptive summary and metadata for a Reactome pathway. |
| `resolve_disease` | `query: str`, `limit: int = 5` | Resolves disease names, synonyms, or IDs to canonical MONDO Disease Ontology entities. |
| `get_target_diseases` | `gene: str`, `limit: int = 10` | Retrieves evidence-backed therapeutic target-disease associations from Open Targets Platform. |
| `get_mouse_gene` | `mgi_id: str` | Direct lookup of mouse gene models from MGI. |

---
Expand Down
10 changes: 10 additions & 0 deletions docs/API.md
Original file line number Diff line number Diff line change
Expand Up @@ -124,6 +124,13 @@ async def main():
pathways = await resolver.get_pathways("TP53", limit=5)
print("Reactome Pathways:", [p.name for p in pathways.pathways])

# 5. Disease Ontology (MONDO) & Open Targets
diseases = await resolver.resolve_disease("Li-Fraumeni", limit=2)
print("MONDO Disease:", diseases[0].mondo_id, diseases[0].name)

target_assocs = await resolver.get_target_diseases("TP53", limit=3)
print("Associated Diseases:", [(a.disease_name, a.score) for a in target_assocs.associations])

asyncio.run(main())
```

Expand All @@ -139,6 +146,9 @@ asyncio.run(main())
| `go` | `biocontext go <GO_ID>` | Inspect GO term metadata. |
| `pathway` | `biocontext pathway <query> [-l INT]` | Retrieve Reactome pathways. |
| `pathway-info` | `biocontext pathway-info <ST_ID>` | Retrieve Reactome pathway summation. |
| `disease` | `biocontext disease <query> [-l INT]` | Resolve disease name or MONDO identifier. |
| `targets` | `biocontext targets <gene> [-l INT]` | Retrieve Open Targets evidence-backed disease associations. |
| `mouse` | `biocontext mouse <MGI_ID>` | Lookup mouse gene model via MGI. |
| `cache` | `biocontext cache [stats\|clear]` | Inspect or clear SQLite cache. |
| `serve` | `biocontext serve` | Start stdio MCP server for AI clients. |

2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
[project]
name = "biocontext-mcp"
version = "0.5.1"
version = "0.6.0"
description = "Authoritative Biological Entity Resolution & Contextual Intelligence Framework"
readme = "README.md"
authors = [
Expand Down
252 changes: 252 additions & 0 deletions src/biocontext/adapters.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
from biocontext.config import ClientConfig, RateLimitConfig
from biocontext.logging import get_logger
from biocontext.schemas import (
DiseaseEntity,
ExonEntity,
FunctionalAnnotation,
GOAnnotation,
Expand All @@ -20,6 +21,8 @@
PathwayContext,
PathwayEntity,
ProteinEntity,
TargetAssociationContext,
TargetDiseaseAssociation,
TranscriptEntity,
)

Expand Down Expand Up @@ -1211,6 +1214,255 @@ async def fetch_pathway_details(self, st_id: str) -> Optional[Dict[str, Any]]:
return details


class MondoAdapter(BaseBioAdapter):
"""Adapter for MONDO Disease Ontology via EBI OLS4 API.
Primary authority for canonical disease identifiers (MONDO:xxxxxxx), preferred names,
synonyms, and cross-references (OMIM, Orphanet, DOID, UMLS).
"""

BASE_URL = "https://www.ebi.ac.uk/ols4/api"

def __init__(self, cache: Optional[SQLiteCache] = None, email: Optional[str] = None):
super().__init__(name="MONDO", cache=cache)
self.email = ClientConfig.get_email(email)
self.headers = ClientConfig.get_headers(self.email)
self.rate_limiter = AsyncRateLimiter(requests_per_second=RateLimitConfig.MONDO_RPS)

async def resolve_gene(self, query: str, taxon_id: int = 9606) -> Optional[Dict[str, Any]]:
"""MONDO resolves disease entities rather than genes."""
return None

async def fetch_by_id(self, mondo_id: str) -> Optional[DiseaseEntity]:
"""Fetch canonical disease entity by MONDO ID (e.g. 'MONDO:0018875' or 'MONDO_0018875')."""
clean_id = mondo_id.strip()
if clean_id.startswith("MONDO:"):
iri_id = clean_id.replace(":", "_")
elif clean_id.startswith("MONDO_"):
iri_id = clean_id
clean_id = clean_id.replace("_", ":")
else:
return None

cache_key = f"mondo:id:{clean_id.upper()}"
cached = self.cache.get("mondo", cache_key)
if cached:
return DiseaseEntity(**cached)

iri = f"http://purl.obolibrary.org/obo/{iri_id}"
url = f"{self.BASE_URL}/ontologies/mondo/terms"
params = {"iri": iri}

await self.rate_limiter.acquire()
try:
async with httpx.AsyncClient(timeout=RateLimitConfig.MONDO_TIMEOUT_SEC) as client:
resp = await client.get(url, params=params, headers=self.headers)
if resp.status_code != 200:
return None
data = resp.json()
except Exception as e:
logger.warning("Mondo term lookup exception | id=%s error=%s", clean_id, str(e))
return None

terms = data.get("_embedded", {}).get("terms", [])
if not terms:
return None

term = terms[0]
name = term.get("label") or clean_id
descriptions = term.get("description", [])
desc = descriptions[0] if descriptions else None
synonyms = term.get("synonyms") or []

# Cross references / dbxrefs
xrefs = []
annotation = term.get("annotation", {})
dbxrefs = annotation.get("database_cross_reference", [])
if isinstance(dbxrefs, list):
xrefs = [str(x) for x in dbxrefs]

disease = DiseaseEntity(
mondo_id=clean_id,
name=name,
description=desc,
synonyms=synonyms,
cross_references=xrefs
)
self.cache.set("mondo", cache_key, disease.model_dump())
return disease

async def search_disease(self, query: str, limit: int = 5) -> List[DiseaseEntity]:
"""Search diseases by name, synonym, or keyword within MONDO."""
clean_query = query.strip()
if not clean_query:
return []

cache_key = f"mondo:search:{clean_query.lower()}:{limit}"
cached = self.cache.get("mondo", cache_key)
if cached and "items" in cached:
return [DiseaseEntity(**item) for item in cached["items"]]

url = f"{self.BASE_URL}/search"
params = {
"q": clean_query,
"ontology": "mondo",
"rows": limit,
"queryFields": "label,synonym"
}

await self.rate_limiter.acquire()
try:
async with httpx.AsyncClient(timeout=RateLimitConfig.MONDO_TIMEOUT_SEC) as client:
resp = await client.get(url, params=params, headers=self.headers)
if resp.status_code != 200:
return []
data = resp.json()
except Exception as e:
logger.warning("Mondo search exception | query=%s error=%s", clean_query, str(e))
return []

docs = data.get("response", {}).get("docs", [])
results: List[DiseaseEntity] = []
for doc in docs:
short_form = doc.get("short_form", "")
if not short_form.startswith("MONDO_") and not short_form.startswith("MONDO:"):
continue
mondo_id = short_form.replace("_", ":")
name = doc.get("label") or mondo_id
descriptions = doc.get("description", [])
desc = descriptions[0] if descriptions else None
synonyms = doc.get("synonym", []) or []

results.append(DiseaseEntity(
mondo_id=mondo_id,
name=name,
description=desc,
synonyms=synonyms,
cross_references=[]
))

self.cache.set("mondo", cache_key, {"items": [r.model_dump() for r in results]})
return results


class OpenTargetsAdapter(BaseBioAdapter):
"""Adapter for Open Targets Platform GraphQL API.
Primary authority for target-disease association scores, clinical pipeline evidence,
and genetic target tractability.
"""

GRAPHQL_URL = "https://api.platform.opentargets.org/api/v4/graphql"

def __init__(self, cache: Optional[SQLiteCache] = None, email: Optional[str] = None):
super().__init__(name="OpenTargets", cache=cache)
self.email = ClientConfig.get_email(email)
self.headers = ClientConfig.get_headers(self.email)
self.headers["Content-Type"] = "application/json"
self.rate_limiter = AsyncRateLimiter(requests_per_second=RateLimitConfig.OPENTARGETS_RPS)

async def resolve_gene(self, query: str, taxon_id: int = 9606) -> Optional[Dict[str, Any]]:
"""OpenTargets is queried via resolved Ensembl ID rather than as a primary gene symbol resolver."""
return None

async def fetch_target_diseases(
self,
ensembl_gene_id: str,
symbol: Optional[str] = None,
limit: int = 10
) -> TargetAssociationContext:
"""Fetch top disease associations for a target gene by Ensembl Gene ID."""
clean_id = ensembl_gene_id.strip().upper()
cache_key = f"opentargets:target:{clean_id}:{limit}"
cached = self.cache.get("opentargets", cache_key)
if cached:
return TargetAssociationContext(**cached)

query = """
query TargetDiseases($ensemblId: String!, $size: Int!) {
target(ensemblId: $ensemblId) {
id
approvedSymbol
approvedName
associatedDiseases(page: {size: $size, index: 0}) {
count
rows {
score
datatypeScores {
id
score
}
disease {
id
name
}
}
}
}
}
"""

payload = {
"query": query,
"variables": {
"ensemblId": clean_id,
"size": limit
}
}

await self.rate_limiter.acquire()
try:
async with httpx.AsyncClient(timeout=RateLimitConfig.OPENTARGETS_TIMEOUT_SEC) as client:
resp = await client.post(self.GRAPHQL_URL, json=payload, headers=self.headers)
if resp.status_code != 200:
logger.warning("OpenTargets GraphQL error | status=%d body=%s", resp.status_code, resp.text[:200])
return TargetAssociationContext(query=clean_id, ensembl_gene_id=clean_id, symbol=symbol)
res_data = resp.json()
except Exception as e:
logger.warning("OpenTargets request exception | target=%s error=%s", clean_id, str(e))
return TargetAssociationContext(query=clean_id, ensembl_gene_id=clean_id, symbol=symbol)

target_data = res_data.get("data", {}).get("target")
if not target_data:
return TargetAssociationContext(query=clean_id, ensembl_gene_id=clean_id, symbol=symbol)

approved_symbol = target_data.get("approvedSymbol") or symbol
assoc_block = target_data.get("associatedDiseases", {})
total_count = assoc_block.get("count", 0)
rows = assoc_block.get("rows", [])

associations: List[TargetDiseaseAssociation] = []
for r in rows:
disease_info = r.get("disease", {})
d_id = disease_info.get("id", "")
d_name = disease_info.get("name", d_id)
score = float(r.get("score", 0.0))

dt_scores_dict: Dict[str, float] = {}
for dt in r.get("datatypeScores", []):
dt_id = dt.get("id")
dt_val = dt.get("score")
if dt_id and dt_val is not None:
dt_scores_dict[dt_id] = float(dt_val)

associations.append(TargetDiseaseAssociation(
disease_id=d_id.replace("_", ":") if d_id.startswith("MONDO_") else d_id,
disease_name=d_name,
score=round(score, 4),
datatype_scores=dt_scores_dict if dt_scores_dict else None
))

context = TargetAssociationContext(
query=clean_id,
ensembl_gene_id=clean_id,
symbol=approved_symbol,
total_associations=total_count,
associations=associations
)

self.cache.set("opentargets", cache_key, context.model_dump())
return context






15 changes: 14 additions & 1 deletion src/biocontext/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -209,6 +209,19 @@ async def run_cli_async(args: argparse.Namespace) -> int:
print(json.dumps(details, indent=2))
return 0

elif args.command == "disease":
diseases = await resolver.resolve_disease(query=args.query, limit=args.limit)
print(json.dumps([d.model_dump() for d in diseases], indent=2))
return 0

elif args.command == "targets":
context = await resolver.get_target_diseases(gene_query=args.gene, limit=args.limit)
if not context:
print(json.dumps({"status": "not_found", "gene": args.gene}, indent=2))
return 1
print(context.model_dump_json(indent=2))
return 0

elif args.command == "cache":
cache = resolver.cache
if args.cache_action == "clear":
Expand Down Expand Up @@ -243,7 +256,7 @@ def main():
sys.exit(pytest.main(["tests/", "-v"]))
elif args.command in (
"resolve", "batch", "protein", "transcripts", "ortholog", "mouse",
"annotate", "go", "pathway", "pathway-info", "cache"
"annotate", "go", "pathway", "pathway-info", "disease", "targets", "cache"
):
sys.exit(asyncio.run(run_cli_async(args)))
else:
Expand Down
21 changes: 21 additions & 0 deletions src/biocontext/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,11 @@ class RateLimitConfig:
QUICKGO_TIMEOUT_SEC: float = 15.0
REACTOME_RPS: float = 5.0
REACTOME_TIMEOUT_SEC: float = 15.0
MONDO_RPS: float = 10.0
MONDO_TIMEOUT_SEC: float = 15.0
OPENTARGETS_RPS: float = 10.0
OPENTARGETS_TIMEOUT_SEC: float = 15.0




Expand Down Expand Up @@ -158,6 +163,22 @@ def get_headers(cls, email: Optional[str] = None) -> Dict[str, str]:
{"flags": ["st_id"], "help": "Reactome stable ID (e.g. R-HSA-5357801)"}
]
},
{
"name": "disease",
"help": "Resolve disease name, synonym, or ID against MONDO Disease Ontology",
"arguments": [
{"flags": ["query"], "help": "Disease name (e.g. 'Li-Fraumeni syndrome') or MONDO ID ('MONDO:0018875')"},
{"flags": ["--limit"], "type": int, "default": 5, "help": "Maximum matches to return (default: 5)"}
]
},
{
"name": "targets",
"help": "Fetch evidence-backed target-disease associations from Open Targets Platform",
"arguments": [
{"flags": ["gene"], "help": "Gene symbol (e.g. TP53) or Ensembl Gene ID"},
{"flags": ["--limit"], "type": int, "default": 10, "help": "Maximum associations to return (default: 10)"}
]
},


{
Expand Down
Loading
Loading