File size: 7,259 Bytes
f400521 9965499 eaba1ed f400521 874a822 86771dc f400521 874a822 f400521 874a822 f400521 eaba1ed 874a822 f400521 eaba1ed 874a822 eaba1ed f400521 eaba1ed 874a822 eaba1ed f400521 874a822 f400521 eaba1ed f400521 eaba1ed f400521 c72a1cd 3637999 e33dfeb f400521 eaba1ed f400521 eaba1ed e33dfeb f400521 eaba1ed 9965499 eaba1ed f400521 eaba1ed f400521 eaba1ed f400521 eaba1ed 3d539ca 874a822 f400521 eaba1ed f400521 874a822 f400521 86771dc f400521 eaba1ed 874a822 f400521 eaba1ed e33dfeb f400521 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 | #!/usr/bin/env python3
"""
mcp/orchestrator.py β MedGenesis v5
βββββββββββββββββββββββββββββββββββ
Asynchronously fan-outs across >10 open biomedical APIs, then returns
one consolidated dictionary for the Streamlit UI.
Public-keyβfree by default:
β’ MyGene.info, Ensembl REST, Open Targets GraphQL
β’ PubMed (E-utils), arXiv
β’ UMLS, openFDA, DisGeNET
β’ Expression Atlas, ClinicalTrials.gov (+ WHO ICTRP fallback)
β’ cBioPortal, DrugCentral, PubChem
If you add secrets **MYGENE_KEY**, **OT_KEY**, **CBIO_KEY** or
**NCBI_EUTILS_KEY**, they are auto-detected and used β otherwise the code
runs key-less.
Returned payload keys
βββββββββββββββββββββ
papers, ai_summary, llm_used, umls, drug_safety,
genes_rich, expr_atlas, drug_meta, chem_info,
gene_disease, clinical_trials, cbio_variants
"""
from __future__ import annotations
import asyncio
from typing import Dict, Any, List
# ββ Literature ββββββββββββββββββββββββββββββββββββββββββββββββββββββ
from mcp.arxiv import fetch_arxiv
from mcp.pubmed import fetch_pubmed
# ββ NLP + enrichment ββββββββββββββββββββββββββββββββββββββββββββββββ
from mcp.nlp import extract_keywords
from mcp.umls import lookup_umls
from mcp.openfda import fetch_drug_safety
from mcp.disgenet import disease_to_genes
from mcp.clinicaltrials import search_trials
# Gene / expression modules
from mcp.gene_hub import resolve_gene # MyGene β Ensembl β OT
from mcp.atlas import fetch_expression
from mcp.cbio import fetch_cbio # cancer variants
# Drug metadata & chemistry
from mcp.drugcentral_ext import fetch_drugcentral
from mcp.pubchem_ext import fetch_compound
# ββ Large-language model helpers ββββββββββββββββββββββββββββββββββββ
from mcp.openai_utils import ai_summarize, ai_qa
from mcp.gemini import gemini_summarize, gemini_qa
_LLM_DEFAULT = "openai"
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# LLM router
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
def _llm_router(name: str):
"""Return (summarise_fn, qa_fn, engine_tag)."""
if name.lower() == "gemini":
return gemini_summarize, gemini_qa, "gemini"
return ai_summarize, ai_qa, "openai"
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# Main orchestrator
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
async def orchestrate_search(query: str,
llm: str = _LLM_DEFAULT) -> Dict[str, Any]:
"""Run the complete async pipeline; always resolves without raising."""
# 1 Literature ---------------------------------------------------
arxiv_f = asyncio.create_task(fetch_arxiv(query, max_results=10))
pubmed_f = asyncio.create_task(fetch_pubmed(query, max_results=10))
papers: List[Dict] = []
for res in await asyncio.gather(arxiv_f, pubmed_f, return_exceptions=True):
if not isinstance(res, Exception):
papers.extend(res)
# 2 Keyword extraction ------------------------------------------
corpus = " ".join(p.get("summary", "") for p in papers)
keywords = extract_keywords(corpus)[:10]
# 3 Parallel enrichment -----------------------------------------
umls_jobs = [lookup_umls(k) for k in keywords]
fda_jobs = [fetch_drug_safety(k) for k in keywords]
gene_jobs = [resolve_gene(k) for k in keywords]
expr_jobs = [fetch_expression(k) for k in keywords]
drug_jobs = [fetch_drugcentral(k) for k in keywords]
chem_jobs = [fetch_compound(k) for k in keywords]
umls, fda, genes, exprs, drugs, chems = await asyncio.gather(
asyncio.gather(*umls_jobs, return_exceptions=True),
asyncio.gather(*fda_jobs, return_exceptions=True),
asyncio.gather(*gene_jobs, return_exceptions=True),
asyncio.gather(*expr_jobs, return_exceptions=True),
asyncio.gather(*drug_jobs, return_exceptions=True),
asyncio.gather(*chem_jobs, return_exceptions=True),
)
# filter out errors / empty payloads
umls = [u for u in umls if isinstance(u, dict)]
fda = [d for d in fda if d]
genes = [g for g in genes if g]
exprs = [e for e in exprs if e]
drugs = [d for d in drugs if d]
chems = [c for c in chems if c]
# 4 Other single-shot APIs --------------------------------------
gene_dis = await disease_to_genes(query)
trials = await search_trials(query, max_studies=20)
# Cancer variants for first 3 gene symbols (quota safety)
cbio_jobs = [fetch_cbio(g.get("symbol", "")) for g in genes[:3]]
cbio_vars = []
if cbio_jobs:
tmp = await asyncio.gather(*cbio_jobs, return_exceptions=True)
cbio_vars = [v for v in tmp if v]
# 5 AI summary ---------------------------------------------------
summarise, _, engine_tag = _llm_router(llm)
ai_summary = await summarise(corpus) if corpus else ""
# 6 Return payload ----------------------------------------------
return {
"papers" : papers,
"ai_summary" : ai_summary,
"llm_used" : engine_tag,
"umls" : umls,
"drug_safety" : fda,
"genes_rich" : genes,
"expr_atlas" : exprs,
"drug_meta" : drugs,
"chem_info" : chems,
"gene_disease" : gene_dis,
"clinical_trials" : trials,
"cbio_variants" : cbio_vars,
}
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
# Follow-up question-answer
# ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
async def answer_ai_question(question: str, *,
context: str,
llm: str = _LLM_DEFAULT) -> Dict[str, str]:
"""Return {"answer": str} using chosen LLM."""
_, qa_fn, _ = _llm_router(llm)
return {"answer": await qa_fn(question, context=context)}
|