stasaking commited on
Commit
9a09205
·
verified ·
1 Parent(s): e2db6eb

Sync main.py from blood-brain-omics repo

Browse files
Files changed (1) hide show
  1. main.py +14 -37
main.py CHANGED
@@ -33,7 +33,6 @@ import logging
33
  import os
34
  import time
35
  import traceback
36
- import urllib.request
37
  from typing import Optional
38
 
39
  import duckdb
@@ -51,13 +50,10 @@ logger = logging.getLogger("blood_brain_api")
51
  # Configuration
52
  # =============================================================================
53
 
54
- # HF Dataset repo for remote Parquet queries
55
- HF_DATASET = os.environ.get(
56
- "HF_DATASET", "stasaking/blood-brain-benchmark"
57
- )
58
- HF_BASE_URL = f"https://huggingface.co/datasets/{HF_DATASET}/resolve/main"
59
-
60
- # Local fallback for dev
61
  DATA_DIR = os.environ.get(
62
  "BENCHMARK_DATA_DIR",
63
  os.path.join(os.path.dirname(__file__), "..", "data")
@@ -219,14 +215,6 @@ db = None
219
  registry = None
220
 
221
 
222
- def _resolve_parquet(filename: str) -> str:
223
- """Return remote URL or local path for a Parquet file."""
224
- local_path = os.path.join(DATA_DIR, filename)
225
- if os.path.exists(local_path):
226
- return local_path
227
- return f"{HF_BASE_URL}/{filename}"
228
-
229
-
230
  @app.on_event("startup")
231
  def startup():
232
  global db, registry
@@ -236,21 +224,17 @@ def startup():
236
 
237
  # Initialize DuckDB with resource limits suitable for CPU Basic.
238
  db = duckdb.connect(":memory:")
239
- db.execute("INSTALL httpfs; LOAD httpfs;")
240
  db.execute(f"SET memory_limit='{DUCKDB_MEMORY_LIMIT}';")
241
  db.execute(f"SET threads={DUCKDB_THREADS};")
242
  db.execute(f"SET temp_directory='{DUCKDB_TEMP_DIR}';")
243
  db.execute("SET preserve_insertion_order=false;")
244
 
245
- # Resolve data sources (remote or local)
246
- results_src = _resolve_parquet("benchmark_results.parquet")
247
- features_src = _resolve_parquet("feature_importance.parquet")
248
 
249
- is_remote = results_src.startswith("http")
250
- mode = "remote httpfs" if is_remote else "local files"
251
  logger.info(
252
- "DuckDB settings — memory_limit=%s threads=%d temp_dir=%s mode=%s",
253
- DUCKDB_MEMORY_LIMIT, DUCKDB_THREADS, DUCKDB_TEMP_DIR, mode,
254
  )
255
 
256
  # Load both tables into memory for reliable, fast queries.
@@ -277,22 +261,15 @@ def startup():
277
  n_features = db.execute("SELECT COUNT(*) FROM features").fetchone()[0]
278
  logger.info("Tables loaded: %d results, %d features", n_results, n_features)
279
 
280
- # Load registry JSON (small file, always downloaded)
281
  registry_local = os.path.join(DATA_DIR, "benchmark_registry.json")
282
- if os.path.exists(registry_local):
283
  with open(registry_local) as f:
284
  registry = json.load(f)
285
- else:
286
- # Download from HF
287
- registry_url = f"{HF_BASE_URL}/benchmark_registry.json"
288
- try:
289
- with urllib.request.urlopen(registry_url) as resp:
290
- registry = json.loads(resp.read().decode())
291
- logger.info("Downloaded registry from %s", registry_url)
292
- except Exception as e:
293
- logger.warning("Could not load registry: %s", e)
294
- registry = {"project": {}, "blood_platforms": {},
295
- "brain_targets": {}, "phases": {}, "models": {}}
296
 
297
 
298
  @app.on_event("shutdown")
 
33
  import os
34
  import time
35
  import traceback
 
36
  from typing import Optional
37
 
38
  import duckdb
 
50
  # Configuration
51
  # =============================================================================
52
 
53
+ # Local data directory. Holds parquets + registry. On the HF Space the
54
+ # start.sh entrypoint downloads these from the HF bucket
55
+ # (stasaking/blood-brain-benchmark) into /app/data before launching the
56
+ # API. In dev this points at webapp/data with the files already on disk.
 
 
 
57
  DATA_DIR = os.environ.get(
58
  "BENCHMARK_DATA_DIR",
59
  os.path.join(os.path.dirname(__file__), "..", "data")
 
215
  registry = None
216
 
217
 
 
 
 
 
 
 
 
 
218
  @app.on_event("startup")
219
  def startup():
220
  global db, registry
 
224
 
225
  # Initialize DuckDB with resource limits suitable for CPU Basic.
226
  db = duckdb.connect(":memory:")
 
227
  db.execute(f"SET memory_limit='{DUCKDB_MEMORY_LIMIT}';")
228
  db.execute(f"SET threads={DUCKDB_THREADS};")
229
  db.execute(f"SET temp_directory='{DUCKDB_TEMP_DIR}';")
230
  db.execute("SET preserve_insertion_order=false;")
231
 
232
+ results_src = os.path.join(DATA_DIR, "benchmark_results.parquet")
233
+ features_src = os.path.join(DATA_DIR, "feature_importance.parquet")
 
234
 
 
 
235
  logger.info(
236
+ "DuckDB settings — memory_limit=%s threads=%d temp_dir=%s data_dir=%s",
237
+ DUCKDB_MEMORY_LIMIT, DUCKDB_THREADS, DUCKDB_TEMP_DIR, DATA_DIR,
238
  )
239
 
240
  # Load both tables into memory for reliable, fast queries.
 
261
  n_features = db.execute("SELECT COUNT(*) FROM features").fetchone()[0]
262
  logger.info("Tables loaded: %d results, %d features", n_results, n_features)
263
 
264
+ # Load registry JSON from local data dir.
265
  registry_local = os.path.join(DATA_DIR, "benchmark_registry.json")
266
+ try:
267
  with open(registry_local) as f:
268
  registry = json.load(f)
269
+ except Exception as e:
270
+ logger.warning("Could not load registry from %s: %s", registry_local, e)
271
+ registry = {"project": {}, "blood_platforms": {},
272
+ "brain_targets": {}, "phases": {}, "models": {}}
 
 
 
 
 
 
 
273
 
274
 
275
  @app.on_event("shutdown")