S4MPL3BI4S commited on
Commit
668f8f4
·
verified ·
1 Parent(s): cbc3552

Deploy MSC Aging Corpus research API: REST /research, OLS reruns, chat UI

Browse files
Files changed (7) hide show
  1. Dockerfile +16 -0
  2. README.md +52 -4
  3. app.py +235 -0
  4. msc_corpus/__init__.py +5 -0
  5. msc_corpus/__main__.py +70 -0
  6. msc_corpus/client.py +290 -0
  7. requirements.txt +7 -0
Dockerfile ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ FROM python:3.11-slim
2
+
3
+ WORKDIR /app
4
+
5
+ COPY requirements.txt .
6
+ RUN pip install --no-cache-dir -r requirements.txt
7
+
8
+ COPY msc_corpus ./msc_corpus
9
+ COPY app.py ./app.py
10
+
11
+ ENV CORPUS_REVISION=corpus-v2026.06.6
12
+ ENV CORPUS_CACHE_DIR=/tmp/msc_corpus_cache
13
+
14
+ EXPOSE 7860
15
+
16
+ CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "7860"]
README.md CHANGED
@@ -1,10 +1,58 @@
1
  ---
2
- title: Msc Aging Corpus Api
3
- emoji: 🌖
4
- colorFrom: pink
5
  colorTo: green
6
  sdk: docker
 
7
  pinned: false
 
 
8
  ---
9
 
10
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
+ title: MSC Aging Corpus API
3
+ emoji: 🧬
4
+ colorFrom: blue
5
  colorTo: green
6
  sdk: docker
7
+ app_port: 7860
8
  pinned: false
9
+ license: cc-by-4.0
10
+ short_description: REST + chat research API for MSC aging transcriptome corpus
11
  ---
12
 
13
+ # MSC Aging Corpus Research API
14
+
15
+ **Single live connection point** for agents and researchers querying the
16
+ [MSC Aging Agent Corpus](https://huggingface.co/datasets/S4MPL3BI4S/msc-aging-agent-corpus).
17
+
18
+ Curated by **Dr. James Utley, PhD** · **Syndicate Laboratories**
19
+
20
+ ## Agent URL (use this)
21
+
22
+ **API base:** `https://S4MPL3BI4S-msc-aging-corpus-api.hf.space`
23
+
24
+ | Endpoint | Purpose |
25
+ | --- | --- |
26
+ | `GET /` | Service info |
27
+ | `GET /docs` | OpenAPI — **agents start here** |
28
+ | `POST /research` | Multi-step hypothesis workflow |
29
+ | `GET /search?gene=NDRG1` | Cross-dataset biomarker hits |
30
+ | `GET /rerun?gene=NDRG1&dataset_id=GSE39540` | Fresh OLS statistics |
31
+ | `GET /expression?gene=NDRG1&dataset_id=GSE39540` | Sample-level data |
32
+ | `GET /manifest` | Cohort scope |
33
+ | `GET /cite?gene=NDRG1&datasets=GSE39540` | APA citations |
34
+ | `/chat` | Structured chat UI |
35
+
36
+ ## Example (curl)
37
+
38
+ ```bash
39
+ curl -X POST "https://S4MPL3BI4S-msc-aging-corpus-api.hf.space/research" \
40
+ -H "Content-Type: application/json" \
41
+ -d '{"gene":"NDRG1","dataset_id":"GSE39540","species":"Homo sapiens","actions":["search","rerun","cite"]}'
42
+ ```
43
+
44
+ ## Example (Python)
45
+
46
+ ```python
47
+ import requests
48
+
49
+ API = "https://S4MPL3BI4S-msc-aging-corpus-api.hf.space"
50
+ r = requests.post(f"{API}/research", json={
51
+ "gene": "NDRG1",
52
+ "dataset_id": "GSE39540",
53
+ "actions": ["search", "rerun", "cite"],
54
+ })
55
+ print(r.json())
56
+ ```
57
+
58
+ Data is pulled from Dataset `S4MPL3BI4S/msc-aging-agent-corpus` @ `corpus-v2026.06.6`.
app.py ADDED
@@ -0,0 +1,235 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """FastAPI research API for the MSC Aging Agent Corpus (Hugging Face Space)."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import json
7
+ import os
8
+ import re
9
+ from contextlib import asynccontextmanager
10
+ from typing import Any
11
+
12
+ import gradio as gr
13
+ import pandas as pd
14
+ from fastapi import FastAPI, HTTPException, Query
15
+ from pydantic import BaseModel, Field
16
+
17
+ from msc_corpus import connect
18
+
19
+ DEFAULT_REVISION = os.environ.get("CORPUS_REVISION", "corpus-v2026.06.6")
20
+ CACHE_DIR = os.environ.get("CORPUS_CACHE_DIR", "/tmp/msc_corpus_cache")
21
+
22
+ corpus = None
23
+
24
+
25
+ @asynccontextmanager
26
+ async def lifespan(_app: FastAPI):
27
+ global corpus
28
+ corpus = connect(remote_only=True, revision=DEFAULT_REVISION, cache_dir=CACHE_DIR)
29
+ yield
30
+
31
+
32
+ app = FastAPI(
33
+ title="MSC Aging Corpus Research API",
34
+ description=(
35
+ "Query the Syndicate Laboratories MSC Aging Agent Corpus: biomarker search, "
36
+ "sample-level expression, OLS statistical reruns, and APA citations. "
37
+ "Data source: Hugging Face Dataset S4MPL3BI4S/msc-aging-agent-corpus."
38
+ ),
39
+ version="1.0.0",
40
+ lifespan=lifespan,
41
+ )
42
+
43
+
44
+ class ResearchRequest(BaseModel):
45
+ gene: str = Field(..., description="Gene symbol, e.g. NDRG1")
46
+ dataset_id: str | None = Field(None, description="Optional GEO accession, e.g. GSE39540")
47
+ species: str | None = Field(None, description="Optional species filter, e.g. Homo sapiens")
48
+ fdr_only: bool = False
49
+ actions: list[str] = Field(
50
+ default=["search", "rerun", "cite"],
51
+ description="Steps: search, expression, rerun, mechanism, cite",
52
+ )
53
+
54
+
55
+ def _require_corpus():
56
+ if corpus is None:
57
+ raise HTTPException(status_code=503, detail="Corpus not loaded yet.")
58
+ return corpus
59
+
60
+
61
+ def _records(frame: pd.DataFrame, limit: int = 500) -> dict[str, Any]:
62
+ if frame.empty:
63
+ return {"count": 0, "rows": []}
64
+ trimmed = frame.head(limit)
65
+ return {
66
+ "count": int(len(frame)),
67
+ "truncated": len(frame) > limit,
68
+ "rows": json.loads(trimmed.to_json(orient="records")),
69
+ }
70
+
71
+
72
+ @app.get("/")
73
+ def root() -> dict[str, str]:
74
+ return {
75
+ "service": "MSC Aging Corpus Research API",
76
+ "curator": "James Utley, PhD · Syndicate Laboratories",
77
+ "dataset": "S4MPL3BI4S/msc-aging-agent-corpus",
78
+ "revision": DEFAULT_REVISION,
79
+ "openapi_docs": "/docs",
80
+ "chat_ui": "/chat",
81
+ "research_endpoint": "POST /research",
82
+ }
83
+
84
+
85
+ @app.get("/health")
86
+ def health() -> dict[str, str]:
87
+ _require_corpus()
88
+ return {"status": "ok", "revision": DEFAULT_REVISION}
89
+
90
+
91
+ @app.get("/connect")
92
+ def connection_manifest() -> dict[str, Any]:
93
+ return _require_corpus().connection_info()
94
+
95
+
96
+ @app.get("/manifest")
97
+ def manifest(
98
+ species: str | None = None,
99
+ limit: int = Query(100, ge=1, le=500),
100
+ ) -> dict[str, Any]:
101
+ frame = _require_corpus().manifest()
102
+ if species:
103
+ frame = frame.loc[frame["species"] == species]
104
+ return _records(frame, limit=limit)
105
+
106
+
107
+ @app.get("/search")
108
+ def search_gene(
109
+ gene: str,
110
+ species: str | None = None,
111
+ dataset_id: str | None = None,
112
+ fdr_only: bool = False,
113
+ limit: int = Query(200, ge=1, le=1000),
114
+ ) -> dict[str, Any]:
115
+ frame = _require_corpus().search_gene(
116
+ gene,
117
+ species=species,
118
+ dataset_id=dataset_id,
119
+ fdr_only=fdr_only,
120
+ )
121
+ return _records(frame, limit=limit)
122
+
123
+
124
+ @app.get("/expression")
125
+ def expression(
126
+ gene: str,
127
+ dataset_id: str,
128
+ limit: int = Query(500, ge=1, le=5000),
129
+ ) -> dict[str, Any]:
130
+ try:
131
+ frame = _require_corpus().expression(gene, dataset_id)
132
+ except Exception as exc: # noqa: BLE001
133
+ raise HTTPException(status_code=404, detail=str(exc)) from exc
134
+ return _records(frame, limit=limit)
135
+
136
+
137
+ @app.get("/rerun")
138
+ def rerun_model(gene: str, dataset_id: str) -> dict[str, Any]:
139
+ try:
140
+ result = _require_corpus().rerun_model(gene, dataset_id)
141
+ except ValueError as exc:
142
+ raise HTTPException(status_code=404, detail=str(exc)) from exc
143
+ result.pop("summary", None)
144
+ return result
145
+
146
+
147
+ @app.get("/cite")
148
+ def cite(gene: str, datasets: str = Query(..., description="Comma-separated GSE IDs")) -> dict[str, str]:
149
+ used = [item.strip() for item in datasets.split(",") if item.strip()]
150
+ return _require_corpus().cite_gene(gene, datasets_used=used)
151
+
152
+
153
+ @app.post("/research")
154
+ def research(body: ResearchRequest) -> dict[str, Any]:
155
+ try:
156
+ return _require_corpus().research(
157
+ body.gene,
158
+ dataset_id=body.dataset_id,
159
+ species=body.species,
160
+ actions=body.actions,
161
+ fdr_only=body.fdr_only,
162
+ )
163
+ except ValueError as exc:
164
+ raise HTTPException(status_code=404, detail=str(exc)) from exc
165
+
166
+
167
+ def _parse_chat(message: str) -> tuple[str, dict[str, Any] | None]:
168
+ text = message.strip()
169
+ lower = text.lower()
170
+
171
+ if lower in {"help", "?"}:
172
+ return (
173
+ "Commands:\n"
174
+ "- `search NDRG1` or `search NDRG1 human`\n"
175
+ "- `rerun NDRG1 GSE39540`\n"
176
+ "- `research NDRG1 GSE39540` (search + rerun + cite)\n"
177
+ "- `manifest` or `manifest human`\n"
178
+ "Agents should prefer REST: POST /research or GET /docs",
179
+ None,
180
+ )
181
+
182
+ if lower.startswith("manifest"):
183
+ species = "Homo sapiens" if "human" in lower else None
184
+ payload = manifest(species=species)
185
+ return json.dumps(payload, indent=2), None
186
+
187
+ rerun_match = re.match(r"rerun\s+(\S+)\s+(GSE\d+)", text, re.I)
188
+ if rerun_match:
189
+ gene, gse = rerun_match.groups()
190
+ return json.dumps(rerun_model(gene, gse), indent=2), None
191
+
192
+ research_match = re.match(r"research\s+(\S+)(?:\s+(GSE\d+))?", text, re.I)
193
+ if research_match:
194
+ gene, gse = research_match.groups()
195
+ body = ResearchRequest(gene=gene, dataset_id=gse)
196
+ return json.dumps(research(body), indent=2), None
197
+
198
+ search_match = re.match(r"search\s+(\S+)(?:\s+(human|mouse|rat))?", text, re.I)
199
+ if search_match:
200
+ gene, sp = search_match.groups()
201
+ species = {"human": "Homo sapiens", "mouse": "Mus musculus", "rat": "Rattus norvegicus"}.get(
202
+ (sp or "").lower()
203
+ )
204
+ return json.dumps(search_gene(gene, species=species), indent=2), None
205
+
206
+ return (
207
+ "I did not understand that. Try `help`, `search NDRG1`, `rerun NDRG1 GSE39540`, "
208
+ "or `research NDRG1 GSE39540`. For full control use /docs.",
209
+ None,
210
+ )
211
+
212
+
213
+ def chat_fn(message: str, history: list[dict[str, str]]) -> str:
214
+ del history
215
+ reply, _ = _parse_chat(message)
216
+ return reply
217
+
218
+
219
+ demo = gr.ChatInterface(
220
+ fn=chat_fn,
221
+ title="MSC Aging Corpus Research Chat",
222
+ description=(
223
+ "Structured queries against the MSC Aging Agent Corpus. "
224
+ "For programmatic access open **/docs** (REST API)."
225
+ ),
226
+ examples=[
227
+ "help",
228
+ "search NDRG1 human",
229
+ "rerun NDRG1 GSE39540",
230
+ "research NDRG1 GSE39540",
231
+ "manifest human",
232
+ ],
233
+ )
234
+
235
+ app = gr.mount_gradio_app(app, demo, path="/chat")
msc_corpus/__init__.py ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ """Single-connection Python client for the MSC Aging Agent Corpus."""
2
+
3
+ from msc_corpus.client import MSCAgingCorpus, connect
4
+
5
+ __all__ = ["MSCAgingCorpus", "connect"]
msc_corpus/__main__.py ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """CLI entry point: uv run python -m msc_corpus search NDRG1 --dataset GSE39540 --rerun"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+
8
+ from msc_corpus import connect
9
+
10
+
11
+ def main() -> None:
12
+ parser = argparse.ArgumentParser(
13
+ description="MSC Aging Agent Corpus — single-connection research CLI",
14
+ )
15
+ sub = parser.add_subparsers(dest="command", required=True)
16
+
17
+ sub.add_parser("info", help="Print connection manifest")
18
+
19
+ search = sub.add_parser("search", help="Search biomarker hits for a gene")
20
+ search.add_argument("gene")
21
+ search.add_argument("--species", default=None)
22
+ search.add_argument("--dataset", default=None)
23
+ search.add_argument("--fdr-only", action="store_true")
24
+
25
+ rerun = sub.add_parser("rerun", help="Rerun linear model for a gene in one GSE")
26
+ rerun.add_argument("gene")
27
+ rerun.add_argument("dataset")
28
+
29
+ expr = sub.add_parser("expression", help="Fetch sample-level expression rows")
30
+ expr.add_argument("gene")
31
+ expr.add_argument("dataset")
32
+
33
+ args = parser.parse_args()
34
+ corpus = connect()
35
+
36
+ if args.command == "info":
37
+ print(json.dumps(corpus.connection_info(), indent=2))
38
+ return
39
+
40
+ if args.command == "search":
41
+ hits = corpus.search_gene(
42
+ args.gene,
43
+ species=args.species,
44
+ dataset_id=args.dataset,
45
+ fdr_only=args.fdr_only,
46
+ )
47
+ print(hits.to_string(index=False) if not hits.empty else "No hits.")
48
+ return
49
+
50
+ if args.command == "rerun":
51
+ result = corpus.rerun_model(args.gene, args.dataset)
52
+ print(result["summary"])
53
+ print(json.dumps(
54
+ {
55
+ "coefficients": result["coefficients"],
56
+ "p_values": result["p_values"],
57
+ "r_squared": result["r_squared"],
58
+ "n_samples": result["n_samples"],
59
+ },
60
+ indent=2,
61
+ ))
62
+ return
63
+
64
+ if args.command == "expression":
65
+ frame = corpus.expression(args.gene, args.dataset)
66
+ print(frame.to_string(index=False) if not frame.empty else "No expression rows.")
67
+
68
+
69
+ if __name__ == "__main__":
70
+ main()
msc_corpus/client.py ADDED
@@ -0,0 +1,290 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Research client for the MSC Aging Agent Corpus (Hugging Face Dataset)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import gzip
6
+ import io
7
+ import json
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ import pandas as pd
12
+
13
+ DEFAULT_REPO = "S4MPL3BI4S/msc-aging-agent-corpus"
14
+ DEFAULT_REVISION = "corpus-v2026.06.6"
15
+
16
+
17
+ def _repo_root() -> Path:
18
+ return Path(__file__).resolve().parents[1]
19
+
20
+
21
+ def connect(
22
+ *,
23
+ revision: str = DEFAULT_REVISION,
24
+ repo_id: str = DEFAULT_REPO,
25
+ local_root: Path | None = None,
26
+ cache_dir: str | None = None,
27
+ remote_only: bool = False,
28
+ ) -> MSCAgingCorpus:
29
+ """Connect to the corpus from Hugging Face or a local clone."""
30
+ return MSCAgingCorpus(
31
+ revision=revision,
32
+ repo_id=repo_id,
33
+ local_root=local_root,
34
+ cache_dir=cache_dir,
35
+ remote_only=remote_only,
36
+ )
37
+
38
+
39
+ class MSCAgingCorpus:
40
+ """Agent-first research interface to screened MSC aging transcriptome data."""
41
+
42
+ def __init__(
43
+ self,
44
+ *,
45
+ revision: str = DEFAULT_REVISION,
46
+ repo_id: str = DEFAULT_REPO,
47
+ local_root: Path | None = None,
48
+ cache_dir: str | None = None,
49
+ remote_only: bool = False,
50
+ ) -> None:
51
+ self.revision = revision
52
+ self.repo_id = repo_id
53
+ self.remote_only = remote_only
54
+ self.local_root = None if remote_only else (local_root or _repo_root())
55
+ self.cache_dir = cache_dir
56
+ self._frames: dict[str, pd.DataFrame] = {}
57
+ self._byte_cache: dict[str, bytes] = {}
58
+
59
+ def connection_info(self) -> dict[str, Any]:
60
+ return json.loads(self._read_text("reference/AGENT_CONNECTION.json"))
61
+
62
+ def provenance(self) -> dict[str, Any]:
63
+ return json.loads(self._read_text("datasets/CORPUS_PROVENANCE.json"))
64
+
65
+ def manifest(self) -> pd.DataFrame:
66
+ return self._load_csv("datasets/dataset_manifest.csv")
67
+
68
+ def biomarkers(self) -> pd.DataFrame:
69
+ return self._load_csv("datasets/aging_biomarker_candidates.csv")
70
+
71
+ def search_gene(
72
+ self,
73
+ gene_symbol: str,
74
+ *,
75
+ species: str | None = None,
76
+ fdr_only: bool = False,
77
+ dataset_id: str | None = None,
78
+ ) -> pd.DataFrame:
79
+ """Cross-dataset biomarker hits for a gene."""
80
+ df = self.biomarkers()
81
+ mask = df["gene_symbol"].str.upper() == gene_symbol.upper()
82
+ if species:
83
+ mask &= df["species"] == species
84
+ if dataset_id:
85
+ mask &= df["dataset_id"] == dataset_id
86
+ if fdr_only:
87
+ mask &= df["fdr_significant"].astype(str).str.upper().eq("TRUE")
88
+ return df.loc[mask].sort_values(["q_value", "p_value"], na_position="last")
89
+
90
+ def screened_hits(
91
+ self,
92
+ gene_symbol: str,
93
+ *,
94
+ dataset_id: str | None = None,
95
+ fdr_only: bool = False,
96
+ ) -> pd.DataFrame:
97
+ """Feature-level screened associations, optionally scoped to one GSE."""
98
+ if dataset_id:
99
+ datasets = [dataset_id]
100
+ else:
101
+ datasets = sorted(self.search_gene(gene_symbol)["dataset_id"].unique())
102
+ frames: list[pd.DataFrame] = []
103
+ for gse in datasets:
104
+ path = f"datasets/{gse}/screened_gene_associations.csv"
105
+ try:
106
+ df = self._load_csv(path)
107
+ except FileNotFoundError:
108
+ continue
109
+ mask = df["gene_symbol"].str.upper() == gene_symbol.upper()
110
+ if fdr_only and "fdr_significant" in df.columns:
111
+ mask &= df["fdr_significant"].astype(str).str.upper().eq("TRUE")
112
+ hit = df.loc[mask]
113
+ if not hit.empty:
114
+ frames.append(hit)
115
+ if not frames:
116
+ return pd.DataFrame()
117
+ return pd.concat(frames, ignore_index=True)
118
+
119
+ def expression(self, gene_symbol: str, dataset_id: str) -> pd.DataFrame:
120
+ """Sample-level expression plus metadata for modeling."""
121
+ path = f"datasets/{dataset_id}/expression_screened_long.csv.gz"
122
+ df = self._load_csv(path)
123
+ return df.loc[df["gene_symbol"].str.upper() == gene_symbol.upper()].copy()
124
+
125
+ def rerun_model(self, gene_symbol: str, dataset_id: str) -> dict[str, Any]:
126
+ """Rerun the documented linear model for hypothesis testing."""
127
+ import statsmodels.formula.api as smf
128
+
129
+ manifest_row = self.manifest().loc[self.manifest()["dataset_id"] == dataset_id]
130
+ if manifest_row.empty:
131
+ raise ValueError(f"Unknown dataset_id: {dataset_id}")
132
+ formula = manifest_row.iloc[0]["model_formula"]
133
+ if not str(formula).startswith("~"):
134
+ raise ValueError(f"Unexpected model_formula for {dataset_id}: {formula}")
135
+
136
+ expr = self.expression(gene_symbol, dataset_id)
137
+ if expr.empty:
138
+ raise ValueError(f"No expression rows for {gene_symbol} in {dataset_id}")
139
+
140
+ if "feature_id" in expr.columns:
141
+ best_feature = (
142
+ expr.groupby("feature_id")["q_value"].min().sort_values().index[0]
143
+ )
144
+ expr = expr.loc[expr["feature_id"] == best_feature].copy()
145
+
146
+ model_formula = f"expression_value {formula}"
147
+ model = smf.ols(model_formula, data=expr).fit()
148
+ coef = model.params.to_dict()
149
+ pvalues = model.pvalues.to_dict()
150
+ return {
151
+ "dataset_id": dataset_id,
152
+ "gene_symbol": gene_symbol,
153
+ "model_formula": model_formula,
154
+ "n_samples": int(model.nobs),
155
+ "r_squared": float(model.rsquared),
156
+ "adj_r_squared": float(model.rsquared_adj),
157
+ "coefficients": coef,
158
+ "p_values": pvalues,
159
+ "summary": str(model.summary()),
160
+ }
161
+
162
+ def clinical_studies(self, condition: str | None = None) -> pd.DataFrame:
163
+ df = self._load_csv("datasets/clinical_evidence/dvc_stem_study_manifest.csv")
164
+ if condition:
165
+ return df.loc[df["condition_category"].str.contains(condition, case=False, na=False)]
166
+ return df
167
+
168
+ def mechanism_hits(self, gene_symbol: str | None = None) -> pd.DataFrame:
169
+ df = self._load_csv("datasets/mechanistic_associations/secretome_transcriptome_hits.csv")
170
+ if gene_symbol:
171
+ return df.loc[df["gene_symbol"].str.upper() == gene_symbol.upper()]
172
+ return df
173
+
174
+ def cite_corpus(self) -> str:
175
+ return self.provenance().get("corpus_citation_apa_hf", "")
176
+
177
+ def cite_gene(self, gene_symbol: str, datasets_used: list[str]) -> dict[str, str]:
178
+ out = {"corpus": self.cite_corpus()}
179
+ try:
180
+ citations = json.loads(self._read_text("reference/citations_apa.json"))
181
+ out["corpus"] = citations["corpus_citation"]["hf_distribution"]
182
+ geo = citations.get("transcriptome_geo", {})
183
+ for gse in datasets_used:
184
+ if gse in geo:
185
+ out[gse] = geo[gse]["apa"]
186
+ except Exception:
187
+ for gse in datasets_used:
188
+ out[gse] = f"https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc={gse}"
189
+ return out
190
+
191
+ def research(
192
+ self,
193
+ gene_symbol: str,
194
+ *,
195
+ dataset_id: str | None = None,
196
+ species: str | None = None,
197
+ actions: list[str] | None = None,
198
+ fdr_only: bool = False,
199
+ ) -> dict[str, Any]:
200
+ """Run a multi-step research workflow (search, expression, rerun, cite)."""
201
+ steps = actions or ["search", "rerun", "cite"]
202
+ result: dict[str, Any] = {"gene_symbol": gene_symbol, "actions": steps}
203
+ datasets_used: list[str] = []
204
+
205
+ if "search" in steps:
206
+ search = self.search_gene(
207
+ gene_symbol,
208
+ species=species,
209
+ dataset_id=dataset_id,
210
+ fdr_only=fdr_only,
211
+ )
212
+ result["search"] = _frame_payload(search)
213
+ datasets_used = sorted(search["dataset_id"].unique()) if not search.empty else []
214
+
215
+ target_dataset = dataset_id
216
+ if not target_dataset and datasets_used:
217
+ target_dataset = datasets_used[0]
218
+
219
+ if "expression" in steps:
220
+ if not target_dataset:
221
+ result["expression"] = {"error": "No dataset_id and no search hits."}
222
+ else:
223
+ result["expression"] = _frame_payload(
224
+ self.expression(gene_symbol, target_dataset)
225
+ )
226
+
227
+ if "rerun" in steps:
228
+ if not target_dataset:
229
+ result["rerun"] = {"error": "No dataset_id and no search hits."}
230
+ else:
231
+ rerun = self.rerun_model(gene_symbol, target_dataset)
232
+ rerun.pop("summary", None)
233
+ result["rerun"] = rerun
234
+ if target_dataset not in datasets_used:
235
+ datasets_used.append(target_dataset)
236
+
237
+ if "mechanism" in steps:
238
+ result["mechanism"] = _frame_payload(self.mechanism_hits(gene_symbol))
239
+
240
+ if "cite" in steps:
241
+ result["citations"] = self.cite_gene(gene_symbol, datasets_used=datasets_used)
242
+
243
+ return result
244
+
245
+ def _load_csv(self, rel_path: str) -> pd.DataFrame:
246
+ if rel_path in self._frames:
247
+ return self._frames[rel_path]
248
+ raw = self._read_bytes(rel_path)
249
+ if rel_path.endswith(".gz"):
250
+ with gzip.open(io.BytesIO(raw), "rt", encoding="utf-8") as handle:
251
+ frame = pd.read_csv(handle)
252
+ else:
253
+ frame = pd.read_csv(io.BytesIO(raw))
254
+ self._frames[rel_path] = frame
255
+ return frame
256
+
257
+ def _read_text(self, rel_path: str) -> str:
258
+ return self._read_bytes(rel_path).decode("utf-8")
259
+
260
+ def _read_bytes(self, rel_path: str) -> bytes:
261
+ if self.local_root is not None:
262
+ local = self.local_root / rel_path
263
+ if local.is_file():
264
+ return local.read_bytes()
265
+ return self._download(rel_path)
266
+
267
+ def _download(self, rel_path: str) -> bytes:
268
+ if rel_path not in self._byte_cache:
269
+ from huggingface_hub import hf_hub_download
270
+
271
+ path = hf_hub_download(
272
+ repo_id=self.repo_id,
273
+ filename=rel_path,
274
+ repo_type="dataset",
275
+ revision=self.revision,
276
+ cache_dir=self.cache_dir,
277
+ )
278
+ self._byte_cache[rel_path] = Path(path).read_bytes()
279
+ return self._byte_cache[rel_path]
280
+
281
+
282
+ def _frame_payload(frame: pd.DataFrame, limit: int = 500) -> dict[str, Any]:
283
+ if frame.empty:
284
+ return {"count": 0, "rows": []}
285
+ trimmed = frame.head(limit)
286
+ return {
287
+ "count": int(len(frame)),
288
+ "truncated": len(frame) > limit,
289
+ "rows": json.loads(trimmed.to_json(orient="records")),
290
+ }
requirements.txt ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ fastapi>=0.110.0
2
+ uvicorn[standard]>=0.27.0
3
+ gradio>=4.44.0
4
+ huggingface_hub>=0.20.0
5
+ pandas>=2.1.0
6
+ statsmodels>=0.14.0
7
+ pydantic>=2.6.0