Spaces:
Sleeping
Sleeping
Sync main.py from blood-brain-omics repo
Browse files
main.py
CHANGED
|
@@ -33,7 +33,6 @@ import logging
|
|
| 33 |
import os
|
| 34 |
import time
|
| 35 |
import traceback
|
| 36 |
-
import urllib.request
|
| 37 |
from typing import Optional
|
| 38 |
|
| 39 |
import duckdb
|
|
@@ -51,13 +50,10 @@ logger = logging.getLogger("blood_brain_api")
|
|
| 51 |
# Configuration
|
| 52 |
# =============================================================================
|
| 53 |
|
| 54 |
-
#
|
| 55 |
-
|
| 56 |
-
|
| 57 |
-
|
| 58 |
-
HF_BASE_URL = f"https://huggingface.co/datasets/{HF_DATASET}/resolve/main"
|
| 59 |
-
|
| 60 |
-
# Local fallback for dev
|
| 61 |
DATA_DIR = os.environ.get(
|
| 62 |
"BENCHMARK_DATA_DIR",
|
| 63 |
os.path.join(os.path.dirname(__file__), "..", "data")
|
|
@@ -219,14 +215,6 @@ db = None
|
|
| 219 |
registry = None
|
| 220 |
|
| 221 |
|
| 222 |
-
def _resolve_parquet(filename: str) -> str:
|
| 223 |
-
"""Return remote URL or local path for a Parquet file."""
|
| 224 |
-
local_path = os.path.join(DATA_DIR, filename)
|
| 225 |
-
if os.path.exists(local_path):
|
| 226 |
-
return local_path
|
| 227 |
-
return f"{HF_BASE_URL}/{filename}"
|
| 228 |
-
|
| 229 |
-
|
| 230 |
@app.on_event("startup")
|
| 231 |
def startup():
|
| 232 |
global db, registry
|
|
@@ -236,21 +224,17 @@ def startup():
|
|
| 236 |
|
| 237 |
# Initialize DuckDB with resource limits suitable for CPU Basic.
|
| 238 |
db = duckdb.connect(":memory:")
|
| 239 |
-
db.execute("INSTALL httpfs; LOAD httpfs;")
|
| 240 |
db.execute(f"SET memory_limit='{DUCKDB_MEMORY_LIMIT}';")
|
| 241 |
db.execute(f"SET threads={DUCKDB_THREADS};")
|
| 242 |
db.execute(f"SET temp_directory='{DUCKDB_TEMP_DIR}';")
|
| 243 |
db.execute("SET preserve_insertion_order=false;")
|
| 244 |
|
| 245 |
-
|
| 246 |
-
|
| 247 |
-
features_src = _resolve_parquet("feature_importance.parquet")
|
| 248 |
|
| 249 |
-
is_remote = results_src.startswith("http")
|
| 250 |
-
mode = "remote httpfs" if is_remote else "local files"
|
| 251 |
logger.info(
|
| 252 |
-
"DuckDB settings — memory_limit=%s threads=%d temp_dir=%s
|
| 253 |
-
DUCKDB_MEMORY_LIMIT, DUCKDB_THREADS, DUCKDB_TEMP_DIR,
|
| 254 |
)
|
| 255 |
|
| 256 |
# Load both tables into memory for reliable, fast queries.
|
|
@@ -277,22 +261,15 @@ def startup():
|
|
| 277 |
n_features = db.execute("SELECT COUNT(*) FROM features").fetchone()[0]
|
| 278 |
logger.info("Tables loaded: %d results, %d features", n_results, n_features)
|
| 279 |
|
| 280 |
-
# Load registry JSON
|
| 281 |
registry_local = os.path.join(DATA_DIR, "benchmark_registry.json")
|
| 282 |
-
|
| 283 |
with open(registry_local) as f:
|
| 284 |
registry = json.load(f)
|
| 285 |
-
|
| 286 |
-
|
| 287 |
-
|
| 288 |
-
|
| 289 |
-
with urllib.request.urlopen(registry_url) as resp:
|
| 290 |
-
registry = json.loads(resp.read().decode())
|
| 291 |
-
logger.info("Downloaded registry from %s", registry_url)
|
| 292 |
-
except Exception as e:
|
| 293 |
-
logger.warning("Could not load registry: %s", e)
|
| 294 |
-
registry = {"project": {}, "blood_platforms": {},
|
| 295 |
-
"brain_targets": {}, "phases": {}, "models": {}}
|
| 296 |
|
| 297 |
|
| 298 |
@app.on_event("shutdown")
|
|
|
|
| 33 |
import os
|
| 34 |
import time
|
| 35 |
import traceback
|
|
|
|
| 36 |
from typing import Optional
|
| 37 |
|
| 38 |
import duckdb
|
|
|
|
| 50 |
# Configuration
|
| 51 |
# =============================================================================
|
| 52 |
|
| 53 |
+
# Local data directory. Holds parquets + registry. On the HF Space the
|
| 54 |
+
# start.sh entrypoint downloads these from the HF bucket
|
| 55 |
+
# (stasaking/blood-brain-benchmark) into /app/data before launching the
|
| 56 |
+
# API. In dev this points at webapp/data with the files already on disk.
|
|
|
|
|
|
|
|
|
|
| 57 |
DATA_DIR = os.environ.get(
|
| 58 |
"BENCHMARK_DATA_DIR",
|
| 59 |
os.path.join(os.path.dirname(__file__), "..", "data")
|
|
|
|
| 215 |
registry = None
|
| 216 |
|
| 217 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 218 |
@app.on_event("startup")
|
| 219 |
def startup():
|
| 220 |
global db, registry
|
|
|
|
| 224 |
|
| 225 |
# Initialize DuckDB with resource limits suitable for CPU Basic.
|
| 226 |
db = duckdb.connect(":memory:")
|
|
|
|
| 227 |
db.execute(f"SET memory_limit='{DUCKDB_MEMORY_LIMIT}';")
|
| 228 |
db.execute(f"SET threads={DUCKDB_THREADS};")
|
| 229 |
db.execute(f"SET temp_directory='{DUCKDB_TEMP_DIR}';")
|
| 230 |
db.execute("SET preserve_insertion_order=false;")
|
| 231 |
|
| 232 |
+
results_src = os.path.join(DATA_DIR, "benchmark_results.parquet")
|
| 233 |
+
features_src = os.path.join(DATA_DIR, "feature_importance.parquet")
|
|
|
|
| 234 |
|
|
|
|
|
|
|
| 235 |
logger.info(
|
| 236 |
+
"DuckDB settings — memory_limit=%s threads=%d temp_dir=%s data_dir=%s",
|
| 237 |
+
DUCKDB_MEMORY_LIMIT, DUCKDB_THREADS, DUCKDB_TEMP_DIR, DATA_DIR,
|
| 238 |
)
|
| 239 |
|
| 240 |
# Load both tables into memory for reliable, fast queries.
|
|
|
|
| 261 |
n_features = db.execute("SELECT COUNT(*) FROM features").fetchone()[0]
|
| 262 |
logger.info("Tables loaded: %d results, %d features", n_results, n_features)
|
| 263 |
|
| 264 |
+
# Load registry JSON from local data dir.
|
| 265 |
registry_local = os.path.join(DATA_DIR, "benchmark_registry.json")
|
| 266 |
+
try:
|
| 267 |
with open(registry_local) as f:
|
| 268 |
registry = json.load(f)
|
| 269 |
+
except Exception as e:
|
| 270 |
+
logger.warning("Could not load registry from %s: %s", registry_local, e)
|
| 271 |
+
registry = {"project": {}, "blood_platforms": {},
|
| 272 |
+
"brain_targets": {}, "phases": {}, "models": {}}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 273 |
|
| 274 |
|
| 275 |
@app.on_event("shutdown")
|