kofi-scholar commited on
Commit
329c7c6
·
1 Parent(s): 98d2451

update app and metric scripts for v1.1.1 release

Browse files
gmass_app.py CHANGED
@@ -77,7 +77,7 @@ REQUIRED_ENV_BY_MODEL = {
77
  "biomistral": "HF_TOKEN",
78
  }
79
 
80
- APP_VERSION = "1.1.0"
81
 
82
  PUBLIC_METRICS_PATH = ROOT / "data" / "public_metrics" / "benchmark_summary.json"
83
  DEFAULT_RESULTS_PATH = ROOT / "data" / "eval_outputs" / "combined" / "all_models_scored.jsonl"
@@ -781,7 +781,8 @@ G-MASS provides a tiered judge system to support institutions ranging from edge
781
 
782
  ### 🏷️ Release History & Version Tags
783
 
784
- - **v1.1.0 (Current Release)**: Public metric export layer, dynamic dataset autodiscovery, compute tiering, safety drift detection engine, and community issue tracking.
 
785
  - **v1.0.0 (Initial Baseline)**: Initial 150-probe bilingual benchmark with LlamaGuard3, AfroLM, and Gemma ensemble.
786
 
787
  ---
@@ -1018,7 +1019,7 @@ function() {
1018
  }
1019
  """
1020
 
1021
- with gr.Blocks(title="G-MASS v1.1.0", theme=gr.themes.Soft(primary_hue="blue"), css=CSS, js=JS_THEME_INIT) as demo:
1022
  gr.HTML(
1023
  f"""
1024
  <div class="gmass-header">
 
77
  "biomistral": "HF_TOKEN",
78
  }
79
 
80
+ APP_VERSION = "1.1.1"
81
 
82
  PUBLIC_METRICS_PATH = ROOT / "data" / "public_metrics" / "benchmark_summary.json"
83
  DEFAULT_RESULTS_PATH = ROOT / "data" / "eval_outputs" / "combined" / "all_models_scored.jsonl"
 
781
 
782
  ### 🏷️ Release History & Version Tags
783
 
784
+ - **v1.1.1 (Current Release)**: Institutional transition to Biomedical Technologies Lab, 300-probe dataset synchronization, automated Windows setup robustness, and test scaffolding under `tests/manual/`.
785
+ - **v1.1.0**: Public metric export layer, dynamic dataset autodiscovery, compute tiering, safety drift detection engine, and community issue tracking.
786
  - **v1.0.0 (Initial Baseline)**: Initial 150-probe bilingual benchmark with LlamaGuard3, AfroLM, and Gemma ensemble.
787
 
788
  ---
 
1019
  }
1020
  """
1021
 
1022
+ with gr.Blocks(title="G-MASS v1.1.1", theme=gr.themes.Soft(primary_hue="blue"), css=CSS, js=JS_THEME_INIT) as demo:
1023
  gr.HTML(
1024
  f"""
1025
  <div class="gmass-header">
run_bilingual_eval.py CHANGED
@@ -59,7 +59,7 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
59
  parser.add_argument(
60
  "--version",
61
  action="version",
62
- version="G-MASS v1.1.0",
63
  help="Show program's version number and exit",
64
  )
65
  parser.add_argument(
 
59
  parser.add_argument(
60
  "--version",
61
  action="version",
62
+ version="G-MASS v1.1.1",
63
  help="Show program's version number and exit",
64
  )
65
  parser.add_argument(
scripts/export_public_metrics.py CHANGED
@@ -60,7 +60,7 @@ def collect_scored_records(
60
  return records
61
 
62
 
63
- def generate_public_metrics(scored_records: list[dict], version: str = "1.1.0") -> dict:
64
  """
65
  Compute aggregate benchmark metrics stripped of any raw prompt or response text.
66
  """
@@ -175,8 +175,8 @@ def main() -> None:
175
  )
176
  parser.add_argument(
177
  "--version",
178
- default="1.1.0",
179
- help="G-MASS benchmark software version (default: 1.1.0)",
180
  )
181
  args = parser.parse_args()
182
 
 
60
  return records
61
 
62
 
63
+ def generate_public_metrics(scored_records: list[dict], version: str = "1.1.1") -> dict:
64
  """
65
  Compute aggregate benchmark metrics stripped of any raw prompt or response text.
66
  """
 
175
  )
176
  parser.add_argument(
177
  "--version",
178
+ default="1.1.1",
179
+ help="G-MASS benchmark software version (default: 1.1.1)",
180
  )
181
  args = parser.parse_args()
182