Spaces:
Sleeping
Sleeping
anikfekr commited on
Commit ·
d220025
1
Parent(s): 36634df
refactor: streamline leaderboard functionality and enhance submission process
Browse files- __pycache__/envs.cpython-312.pyc +0 -0
- __pycache__/utils.cpython-312.pyc +0 -0
- app.py +48 -293
- envs.py +21 -0
- utils.py +195 -0
__pycache__/envs.cpython-312.pyc
ADDED
|
Binary file (722 Bytes). View file
|
|
|
__pycache__/utils.cpython-312.pyc
ADDED
|
Binary file (7.25 kB). View file
|
|
|
app.py
CHANGED
|
@@ -1,12 +1,14 @@
|
|
| 1 |
import gradio as gr
|
| 2 |
-
|
| 3 |
-
|
| 4 |
-
from
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 5 |
|
| 6 |
-
# HF Variables
|
| 7 |
-
USERNAME = "alinf"
|
| 8 |
-
DATASET_NAME = "crcis_multiple-choice_results"
|
| 9 |
-
SPLIT_NAME = "results"
|
| 10 |
|
| 11 |
# Dataset Variables
|
| 12 |
overall_score_column = "overall"
|
|
@@ -15,300 +17,53 @@ precision_column = "precision"
|
|
| 15 |
num_params_column = "#parameters (B)"
|
| 16 |
num_params_range = [0, 100] # in billions
|
| 17 |
|
| 18 |
-
#
|
| 19 |
-
|
| 20 |
-
<h1 align="center">🏆 {} Leaderboard</h1>
|
| 21 |
-
""".format(USERNAME)
|
| 22 |
|
| 23 |
-
|
| 24 |
-
|
| 25 |
-
<p align="center">
|
| 26 |
-
<a href="https://huggingface.co/datasets/{}/{}">Dataset</a> •
|
| 27 |
-
<a href="https://arxiv.org/abs/xxxx.xxxxx">Paper</a> •
|
| 28 |
-
<a href="https://github.com/your-username/your-benchmark">GitHub</a>
|
| 29 |
-
</p>
|
| 30 |
-
""".format(USERNAME, USERNAME, DATASET_NAME)
|
| 31 |
|
| 32 |
-
|
| 33 |
-
"""Load results from HuggingFace dataset."""
|
| 34 |
-
try:
|
| 35 |
-
dataset = load_dataset(
|
| 36 |
-
f"{USERNAME}/{DATASET_NAME}",
|
| 37 |
-
split=SPLIT_NAME,
|
| 38 |
-
download_mode="force_redownload"
|
| 39 |
-
)
|
| 40 |
-
print(f"Loaded {len(dataset)} entries from the dataset.")
|
| 41 |
-
return dataset
|
| 42 |
-
except Exception as e:
|
| 43 |
-
print(f"Error loading dataset: {e}")
|
| 44 |
-
# Return sample data if loading fails
|
| 45 |
-
return pd.DataFrame([
|
| 46 |
-
{
|
| 47 |
-
"rank": 1,
|
| 48 |
-
"model": "Sample Model",
|
| 49 |
-
"overall": 75.5,
|
| 50 |
-
"precision": "fp16",
|
| 51 |
-
"#parameters (B)": 7.0
|
| 52 |
-
}
|
| 53 |
-
])
|
| 54 |
|
| 55 |
-
|
| 56 |
-
|
| 57 |
-
|
| 58 |
-
|
| 59 |
-
|
| 60 |
-
|
| 61 |
-
|
| 62 |
-
|
| 63 |
-
#
|
| 64 |
-
|
| 65 |
|
| 66 |
-
|
| 67 |
-
|
|
|
|
| 68 |
|
| 69 |
-
|
| 70 |
-
|
| 71 |
-
|
| 72 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 73 |
|
| 74 |
-
|
| 75 |
-
|
| 76 |
-
# Custom CSS for styling
|
| 77 |
-
custom_css = """
|
| 78 |
-
.gradio-container {
|
| 79 |
-
font-family: 'IBM Plex Sans', sans-serif;
|
| 80 |
-
}
|
| 81 |
-
|
| 82 |
-
.leaderboard-table {
|
| 83 |
-
margin-top: 20px;
|
| 84 |
-
}
|
| 85 |
-
|
| 86 |
-
h1 {
|
| 87 |
-
color: #2c3e50;
|
| 88 |
-
margin-bottom: 10px;
|
| 89 |
-
}
|
| 90 |
-
|
| 91 |
-
.description {
|
| 92 |
-
color: #7f8c8d;
|
| 93 |
-
margin-bottom: 30px;
|
| 94 |
-
}
|
| 95 |
-
|
| 96 |
-
.tab-nav button {
|
| 97 |
-
font-size: 16px;
|
| 98 |
-
font-weight: 600;
|
| 99 |
-
}
|
| 100 |
-
"""
|
| 101 |
-
|
| 102 |
-
# Prepare the data
|
| 103 |
-
df = prepare_leaderboard_data()
|
| 104 |
-
|
| 105 |
-
print(df)
|
| 106 |
-
|
| 107 |
-
# Determine filterable columns
|
| 108 |
-
filterable_columns = []
|
| 109 |
-
filterable_columns.append(precision_column)
|
| 110 |
-
filterable_columns.append(
|
| 111 |
-
ColumnFilter(num_params_column, type="slider", min=num_params_range[0], max=num_params_range[1])
|
| 112 |
-
)
|
| 113 |
-
|
| 114 |
-
# Determine search columns
|
| 115 |
-
search_columns = []
|
| 116 |
-
search_columns.append(model_name_column)
|
| 117 |
-
|
| 118 |
-
# Columns to hide
|
| 119 |
-
hidden_columns = ["model_link"]
|
| 120 |
-
|
| 121 |
-
# Default visible columns (all except hidden ones)
|
| 122 |
-
visible_columns = [col for col in df.columns if col not in hidden_columns]
|
| 123 |
-
|
| 124 |
-
columns_datatype = df[visible_columns].dtypes.to_list()
|
| 125 |
-
|
| 126 |
-
# Create Gradio interface
|
| 127 |
-
with gr.Blocks(title=f"{USERNAME} Benchmark Leaderboard", css=custom_css) as demo:
|
| 128 |
-
gr.HTML(TITLE)
|
| 129 |
-
gr.Markdown(DESCRIPTION, elem_classes="description")
|
| 130 |
|
| 131 |
-
with gr.
|
| 132 |
-
|
| 133 |
-
|
| 134 |
-
|
| 135 |
-
|
| 136 |
-
|
| 137 |
-
- Click column headers to sort by different metrics
|
| 138 |
-
- Use filters to find models that match your requirements
|
| 139 |
-
- Search supports multiple queries separated by semicolons (;)
|
| 140 |
-
---
|
| 141 |
-
""")
|
| 142 |
-
|
| 143 |
-
# Create the leaderboard with gradio_leaderboard
|
| 144 |
-
Leaderboard(
|
| 145 |
-
value=df,
|
| 146 |
-
datatype=columns_datatype,
|
| 147 |
-
select_columns=SelectColumns(
|
| 148 |
-
default_selection=visible_columns,
|
| 149 |
-
cant_deselect=[model_name_column, "Rank"],
|
| 150 |
-
label="📊 Select Columns to Display",
|
| 151 |
-
info="Choose which metrics to show in the leaderboard"
|
| 152 |
-
),
|
| 153 |
-
search_columns=search_columns if search_columns else None,
|
| 154 |
-
hide_columns=hidden_columns,
|
| 155 |
-
filter_columns=filterable_columns if filterable_columns else None,
|
| 156 |
-
interactive=False,
|
| 157 |
-
elem_classes="leaderboard-table"
|
| 158 |
-
)
|
| 159 |
-
|
| 160 |
-
# About Tab
|
| 161 |
-
with gr.TabItem("📖 About"):
|
| 162 |
-
gr.Markdown(f"""
|
| 163 |
-
## About This Benchmark
|
| 164 |
-
|
| 165 |
-
This leaderboard tracks the performance of Large Language Models on the **{USERNAME} Benchmark**.
|
| 166 |
-
|
| 167 |
-
### Evaluation Details
|
| 168 |
-
- **Tasks**: Multiple-choice questions across multiple subjects
|
| 169 |
-
- **Evaluation Method**: 5-shot evaluation using lm-evaluation-harness
|
| 170 |
-
- **Metric**: Accuracy (%)
|
| 171 |
-
- **Dataset**: [{USERNAME}/{DATASET_NAME}](https://huggingface.co/datasets/{USERNAME}/{DATASET_NAME})
|
| 172 |
-
|
| 173 |
-
### How to Submit Your Model
|
| 174 |
-
|
| 175 |
-
1. **Evaluate your model** using lm-evaluation-harness:
|
| 176 |
-
```bash
|
| 177 |
-
lm_eval --model hf \\
|
| 178 |
-
--model_args pretrained=your-org/your-model \\
|
| 179 |
-
--tasks {DATASET_NAME} \\
|
| 180 |
-
--num_fewshot 5 \\
|
| 181 |
-
--output_path ./results
|
| 182 |
-
```
|
| 183 |
-
|
| 184 |
-
2. **Submit results** via [GitHub Issues](https://github.com/your-username/your-benchmark/issues)
|
| 185 |
-
|
| 186 |
-
3. **Include the following information**:
|
| 187 |
-
- Model name and organization
|
| 188 |
-
- All evaluation scores
|
| 189 |
-
- Evaluation logs or result files
|
| 190 |
-
- Model precision and parameter count
|
| 191 |
-
|
| 192 |
-
### Evaluation Criteria
|
| 193 |
-
|
| 194 |
-
We verify all submissions to ensure:
|
| 195 |
-
- Results are reproducible
|
| 196 |
-
- Evaluation was performed correctly
|
| 197 |
-
- Model information is accurate
|
| 198 |
-
|
| 199 |
-
Results are typically added within 48 hours of submission.
|
| 200 |
-
|
| 201 |
-
### Citation
|
| 202 |
-
|
| 203 |
-
If you use this benchmark in your research, please cite:
|
| 204 |
-
|
| 205 |
-
```bibtex
|
| 206 |
-
@article{{yourbenchmark2024,
|
| 207 |
-
title={{Your Benchmark Title}},
|
| 208 |
-
author={{Your Name}},
|
| 209 |
-
journal={{arXiv preprint arXiv:xxxx.xxxxx}},
|
| 210 |
-
year={{2024}}
|
| 211 |
-
}}
|
| 212 |
-
```
|
| 213 |
-
|
| 214 |
-
### Contact
|
| 215 |
-
|
| 216 |
-
For questions or issues, please:
|
| 217 |
-
- Open an issue on [GitHub](https://github.com/your-username/your-benchmark/issues)
|
| 218 |
-
- Contact us at your-email@example.com
|
| 219 |
-
""")
|
| 220 |
|
| 221 |
-
|
| 222 |
-
with gr.TabItem("📤 Submit"):
|
| 223 |
-
gr.Markdown("""
|
| 224 |
-
## Submit Your Model Results
|
| 225 |
-
|
| 226 |
-
Ready to add your model to the leaderboard? Follow these steps:
|
| 227 |
-
""")
|
| 228 |
-
|
| 229 |
-
with gr.Accordion("📝 Submission Form", open=True):
|
| 230 |
-
gr.Markdown("""
|
| 231 |
-
Please fill out the form below with your model information.
|
| 232 |
-
Note: This is for display purposes. Actual submissions should be made via GitHub Issues.
|
| 233 |
-
""")
|
| 234 |
-
|
| 235 |
-
with gr.Row():
|
| 236 |
-
model_name = gr.Textbox(
|
| 237 |
-
label="Model Name",
|
| 238 |
-
placeholder="e.g., gpt-4",
|
| 239 |
-
info="The name of your model"
|
| 240 |
-
)
|
| 241 |
-
organization = gr.Textbox(
|
| 242 |
-
label="Organization",
|
| 243 |
-
placeholder="e.g., OpenAI",
|
| 244 |
-
info="Your organization or username"
|
| 245 |
-
)
|
| 246 |
-
|
| 247 |
-
with gr.Row():
|
| 248 |
-
precision = gr.Dropdown(
|
| 249 |
-
label="Precision",
|
| 250 |
-
choices=["fp32", "fp16", "bf16", "int8", "int4"],
|
| 251 |
-
value="fp16",
|
| 252 |
-
info="Model precision used for evaluation"
|
| 253 |
-
)
|
| 254 |
-
param_count = gr.Number(
|
| 255 |
-
label="Parameters (Billions)",
|
| 256 |
-
info="Number of parameters in billions (e.g., 7.0)"
|
| 257 |
-
)
|
| 258 |
-
|
| 259 |
-
with gr.Row():
|
| 260 |
-
overall_score = gr.Number(
|
| 261 |
-
label="Overall Score",
|
| 262 |
-
info="Average score across all tasks (e.g., 75.5)"
|
| 263 |
-
)
|
| 264 |
-
|
| 265 |
-
results_file = gr.File(
|
| 266 |
-
label="Upload Results JSON",
|
| 267 |
-
file_types=[".json", ".jsonl"],
|
| 268 |
-
)
|
| 269 |
-
|
| 270 |
-
contact_email = gr.Textbox(
|
| 271 |
-
label="Contact Email",
|
| 272 |
-
placeholder="your-email@example.com",
|
| 273 |
-
info="We'll contact you about your submission"
|
| 274 |
-
)
|
| 275 |
-
|
| 276 |
-
submit_btn = gr.Button("📤 Submit for Review", variant="primary", size="lg")
|
| 277 |
-
|
| 278 |
-
output_message = gr.Markdown(visible=False)
|
| 279 |
-
|
| 280 |
-
def handle_submission(model, org, prec, params, score, file, email):
|
| 281 |
-
"""Handle submission form."""
|
| 282 |
-
if not all([model, org, score, email]):
|
| 283 |
-
return gr.Markdown(
|
| 284 |
-
"⚠️ Please fill in all required fields.",
|
| 285 |
-
visible=True
|
| 286 |
-
)
|
| 287 |
-
|
| 288 |
-
return gr.Markdown(
|
| 289 |
-
f"""
|
| 290 |
-
✅ **Thank you for your submission!**
|
| 291 |
-
|
| 292 |
-
We've received your submission for **{model}** from **{org}**.
|
| 293 |
-
|
| 294 |
-
**Next steps:**
|
| 295 |
-
1. Create an issue on our [GitHub repository](https://github.com/your-username/your-benchmark/issues)
|
| 296 |
-
2. Include all the information you provided here
|
| 297 |
-
3. Attach your results file
|
| 298 |
-
|
| 299 |
-
We'll review your submission and add it to the leaderboard within 48 hours.
|
| 300 |
-
|
| 301 |
-
You'll receive a notification at **{email}** when your model is added.
|
| 302 |
-
""",
|
| 303 |
-
visible=True
|
| 304 |
-
)
|
| 305 |
-
|
| 306 |
-
submit_btn.click(
|
| 307 |
-
fn=handle_submission,
|
| 308 |
-
inputs=[model_name, organization, precision, param_count,
|
| 309 |
-
overall_score, results_file, contact_email],
|
| 310 |
-
outputs=output_message
|
| 311 |
-
)
|
| 312 |
|
| 313 |
if __name__ == "__main__":
|
| 314 |
demo.launch()
|
|
|
|
| 1 |
import gradio as gr
|
| 2 |
+
from gradio_leaderboard import Leaderboard, SelectColumns
|
| 3 |
+
|
| 4 |
+
from utils import (
|
| 5 |
+
LLM_BENCHMARKS_ABOUT_TEXT,
|
| 6 |
+
LLM_BENCHMARKS_SUBMIT_TEXT,
|
| 7 |
+
custom_css,
|
| 8 |
+
prepare_leaderboard_data,
|
| 9 |
+
submit
|
| 10 |
+
)
|
| 11 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 12 |
|
| 13 |
# Dataset Variables
|
| 14 |
overall_score_column = "overall"
|
|
|
|
| 17 |
num_params_column = "#parameters (B)"
|
| 18 |
num_params_range = [0, 100] # in billions
|
| 19 |
|
| 20 |
+
# Prepare the data
|
| 21 |
+
leaderboard_df = prepare_leaderboard_data(overall_score_column, model_name_column)
|
|
|
|
|
|
|
| 22 |
|
| 23 |
+
# Determine columns data type (markdown for all to support clickable links)
|
| 24 |
+
columns_datatype = ["markdown" for _ in range(len(leaderboard_df.columns))]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 25 |
|
| 26 |
+
NUM_MODELS = len(leaderboard_df)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
|
| 28 |
+
with gr.Blocks(
|
| 29 |
+
css=custom_css,
|
| 30 |
+
theme=gr.themes.Default(
|
| 31 |
+
font=["sans-serif", "ui-sans-serif", "system-ui"],
|
| 32 |
+
font_mono=["monospace", "ui-monospace", "Consolas"]
|
| 33 |
+
)
|
| 34 |
+
) as demo:
|
| 35 |
+
gr.Markdown("""
|
| 36 |
+
# 🏆 CRCIS LLM Leaderboard
|
| 37 |
+
""")
|
| 38 |
|
| 39 |
+
gr.Markdown(f"""
|
| 40 |
+
- **Total Models**: {NUM_MODELS}
|
| 41 |
+
""")
|
| 42 |
|
| 43 |
+
with gr.Tab("🏆 Leaderboard"):
|
| 44 |
+
Leaderboard(
|
| 45 |
+
value=leaderboard_df,
|
| 46 |
+
datatype=columns_datatype,
|
| 47 |
+
select_columns=SelectColumns(
|
| 48 |
+
default_selection=leaderboard_df.columns.tolist(),
|
| 49 |
+
cant_deselect=[model_name_column],
|
| 50 |
+
label="Select Columns to Show",
|
| 51 |
+
),
|
| 52 |
+
search_columns=[model_name_column],
|
| 53 |
+
filter_columns=[precision_column, num_params_column],
|
| 54 |
+
)
|
| 55 |
|
| 56 |
+
with gr.TabItem("📖 About"):
|
| 57 |
+
gr.Markdown(LLM_BENCHMARKS_ABOUT_TEXT)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 58 |
|
| 59 |
+
with gr.Tab("📤 Submit"):
|
| 60 |
+
gr.Markdown(LLM_BENCHMARKS_SUBMIT_TEXT)
|
| 61 |
+
model_name = gr.Textbox(label="Model name")
|
| 62 |
+
model_id = gr.Textbox(label="Model ID (e.g., username/model-name)")
|
| 63 |
+
contact_email = gr.Textbox(label="Contact E-Mail")
|
| 64 |
+
submit_btn = gr.Button("Submit")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 65 |
|
| 66 |
+
submit_btn.click(submit, inputs=[model_name, model_id, contact_email], outputs=[])
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 67 |
|
| 68 |
if __name__ == "__main__":
|
| 69 |
demo.launch()
|
envs.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
|
| 3 |
+
from huggingface_hub import HfApi
|
| 4 |
+
|
| 5 |
+
# Info to change for your repository
|
| 6 |
+
# ----------------------------------
|
| 7 |
+
TOKEN = os.environ.get("HF_TOKEN")
|
| 8 |
+
|
| 9 |
+
OWNER = "alinf" # Change to your HF username
|
| 10 |
+
RESULT_DATASET_NAME = "crcis_multiple-choice_results"
|
| 11 |
+
# ----------------------------------
|
| 12 |
+
|
| 13 |
+
REQUEST_QUEUE_REPO = f"{OWNER}/crcis_request_queue"
|
| 14 |
+
|
| 15 |
+
# If you setup a cache later, just set/change 'HF_HOME' env variable
|
| 16 |
+
CACHE_PATH = os.getenv("HF_HOME", ".")
|
| 17 |
+
|
| 18 |
+
# Local caches
|
| 19 |
+
EVAL_REQUESTS_PATH = os.path.join(CACHE_PATH, "eval-queue")
|
| 20 |
+
|
| 21 |
+
API = HfApi(token=TOKEN)
|
utils.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import json
|
| 2 |
+
import os
|
| 3 |
+
from datetime import datetime
|
| 4 |
+
|
| 5 |
+
import gradio as gr
|
| 6 |
+
import pandas as pd
|
| 7 |
+
from datasets import load_dataset
|
| 8 |
+
|
| 9 |
+
from envs import API, EVAL_REQUESTS_PATH, REQUEST_QUEUE_REPO, OWNER, RESULT_DATASET_NAME
|
| 10 |
+
|
| 11 |
+
custom_css = """
|
| 12 |
+
.gradio-container {
|
| 13 |
+
font-family: 'IBM Plex Sans', sans-serif;
|
| 14 |
+
}
|
| 15 |
+
|
| 16 |
+
.leaderboard-table {
|
| 17 |
+
margin-top: 20px;
|
| 18 |
+
}
|
| 19 |
+
|
| 20 |
+
h1 {
|
| 21 |
+
color: #2c3e50;
|
| 22 |
+
margin-bottom: 10px;
|
| 23 |
+
}
|
| 24 |
+
|
| 25 |
+
.description {
|
| 26 |
+
color: #7f8c8d;
|
| 27 |
+
margin-bottom: 30px;
|
| 28 |
+
}
|
| 29 |
+
|
| 30 |
+
.tab-nav button {
|
| 31 |
+
font-size: 16px;
|
| 32 |
+
font-weight: 600;
|
| 33 |
+
}
|
| 34 |
+
|
| 35 |
+
#leaderboard-table {
|
| 36 |
+
margin-top: 15px;
|
| 37 |
+
text-align: center;
|
| 38 |
+
}
|
| 39 |
+
|
| 40 |
+
#leaderboard-table th,
|
| 41 |
+
#leaderboard-table td {
|
| 42 |
+
text-align: center;
|
| 43 |
+
vertical-align: middle;
|
| 44 |
+
}
|
| 45 |
+
|
| 46 |
+
#leaderboard-table td:first-child,
|
| 47 |
+
#leaderboard-table th:first-child {
|
| 48 |
+
text-align: left;
|
| 49 |
+
max-width: 500px;
|
| 50 |
+
}
|
| 51 |
+
"""
|
| 52 |
+
|
| 53 |
+
LLM_BENCHMARKS_ABOUT_TEXT = f"""
|
| 54 |
+
# CRCIS LLM Leaderboard
|
| 55 |
+
|
| 56 |
+
This leaderboard tracks the performance of Large Language Models on the **CRCIS Benchmark**.
|
| 57 |
+
|
| 58 |
+
## Evaluation Details
|
| 59 |
+
- **Tasks**: Multiple-choice questions across multiple subjects including:
|
| 60 |
+
- Humanities - History
|
| 61 |
+
- Quran - General Information
|
| 62 |
+
- Quran - Tafsir
|
| 63 |
+
- **Evaluation Method**: 5-shot evaluation using lm-evaluation-harness
|
| 64 |
+
- **Metric**: Accuracy (%)
|
| 65 |
+
- **Dataset**: [{OWNER}/{RESULT_DATASET_NAME}](https://huggingface.co/datasets/{OWNER}/{RESULT_DATASET_NAME})
|
| 66 |
+
|
| 67 |
+
## Background and Goals
|
| 68 |
+
|
| 69 |
+
This leaderboard provides a comprehensive benchmarking system for evaluating LLMs on CRCIS-specific tasks.
|
| 70 |
+
The evaluation framework is based on the open-source [LM Evaluation Harness](https://github.com/EleutherAI/lm-evaluation-harness),
|
| 71 |
+
offering a reliable platform for assessing model performance on specialized domain knowledge.
|
| 72 |
+
|
| 73 |
+
## Data Integrity
|
| 74 |
+
|
| 75 |
+
To maintain evaluation integrity and prevent overfitting, the full benchmark dataset is used for evaluation.
|
| 76 |
+
This approach ensures that results genuinely represent each model's capabilities.
|
| 77 |
+
|
| 78 |
+
"""
|
| 79 |
+
|
| 80 |
+
LLM_BENCHMARKS_SUBMIT_TEXT = """## Submitting a Model for Evaluation
|
| 81 |
+
|
| 82 |
+
To submit your model for evaluation, follow these steps:
|
| 83 |
+
|
| 84 |
+
1. **Ensure your model is on Hugging Face**: Your model must be publicly available on [Hugging Face](https://huggingface.co/).
|
| 85 |
+
|
| 86 |
+
2. **Submit Request**: Fill out the form below with your model's information and Hugging Face identifier.
|
| 87 |
+
|
| 88 |
+
3. **Evaluation Queue**: Submissions will be queued and processed. The evaluation may take some time depending on the queue.
|
| 89 |
+
|
| 90 |
+
4. **Results**: Once the evaluation is complete, your model's results will be E-mailed to you.
|
| 91 |
+
|
| 92 |
+
We appreciate your contributions to the LLM ecosystem!
|
| 93 |
+
"""
|
| 94 |
+
|
| 95 |
+
|
| 96 |
+
def load_results_from_hf_dataset():
|
| 97 |
+
"""Load results from HuggingFace dataset."""
|
| 98 |
+
try:
|
| 99 |
+
dataset = load_dataset(
|
| 100 |
+
f"{OWNER}/{RESULT_DATASET_NAME}",
|
| 101 |
+
split="results",
|
| 102 |
+
download_mode="force_redownload"
|
| 103 |
+
)
|
| 104 |
+
print(f"Loaded {len(dataset)} entries from the dataset.")
|
| 105 |
+
return pd.DataFrame(dataset)
|
| 106 |
+
except Exception as e:
|
| 107 |
+
print(f"Error loading dataset: {e}")
|
| 108 |
+
# Return sample data if loading fails
|
| 109 |
+
return pd.DataFrame([
|
| 110 |
+
{
|
| 111 |
+
"model": "Sample Model",
|
| 112 |
+
"overall": 75.5,
|
| 113 |
+
"precision": "fp16",
|
| 114 |
+
"#parameters (B)": 7.0
|
| 115 |
+
}
|
| 116 |
+
])
|
| 117 |
+
|
| 118 |
+
|
| 119 |
+
def sort_dataframe_by_column(df, column_name):
|
| 120 |
+
"""Sort dataframe by specified column in descending order."""
|
| 121 |
+
if column_name not in df.columns:
|
| 122 |
+
raise ValueError(f"Column '{column_name}' does not exist in the DataFrame.")
|
| 123 |
+
return df.sort_values(by=column_name, ascending=False).reset_index(drop=True)
|
| 124 |
+
|
| 125 |
+
|
| 126 |
+
def make_clickable_model(model_name):
|
| 127 |
+
"""Make model name clickable with link to HuggingFace."""
|
| 128 |
+
link = f"https://huggingface.co/{model_name}"
|
| 129 |
+
style = "color: var(--link-text-color); text-decoration: underline; text-decoration-style: dotted;"
|
| 130 |
+
return f'<a target="_blank" href="{link}" style="{style}">{model_name}</a>'
|
| 131 |
+
|
| 132 |
+
|
| 133 |
+
def prepare_leaderboard_data(overall_score_column="overall", model_name_column="model"):
|
| 134 |
+
"""Prepare and format the leaderboard DataFrame."""
|
| 135 |
+
df = load_results_from_hf_dataset()
|
| 136 |
+
|
| 137 |
+
# Remove columns with parentheses (except those with #)
|
| 138 |
+
df = df[[col for col in df.columns if ("(" not in col and ")" not in col) or "#" in col]]
|
| 139 |
+
|
| 140 |
+
# Sort by overall score (descending)
|
| 141 |
+
df = sort_dataframe_by_column(df, overall_score_column)
|
| 142 |
+
|
| 143 |
+
# Make model names clickable
|
| 144 |
+
df[model_name_column] = df[model_name_column].apply(make_clickable_model)
|
| 145 |
+
|
| 146 |
+
return df
|
| 147 |
+
|
| 148 |
+
|
| 149 |
+
def submit(model_name, model_id, contact_email):
|
| 150 |
+
"""Handle model submission to evaluation queue."""
|
| 151 |
+
if model_name == "" or model_id == "" or contact_email == "":
|
| 152 |
+
gr.Info("Please fill all the fields")
|
| 153 |
+
return
|
| 154 |
+
|
| 155 |
+
try:
|
| 156 |
+
user_name = ""
|
| 157 |
+
if "/" in model_id:
|
| 158 |
+
user_name = model_id.split("/")[0]
|
| 159 |
+
model_path = model_id.split("/")[1]
|
| 160 |
+
else:
|
| 161 |
+
gr.Error("Model ID must be in the format 'username/model-name'")
|
| 162 |
+
return
|
| 163 |
+
|
| 164 |
+
eval_entry = {
|
| 165 |
+
"model_name": model_name,
|
| 166 |
+
"model_id": model_id,
|
| 167 |
+
"contact_email": contact_email,
|
| 168 |
+
}
|
| 169 |
+
|
| 170 |
+
# Get the current timestamp to add to the filename
|
| 171 |
+
timestamp = datetime.now().strftime("%Y%m%d")
|
| 172 |
+
|
| 173 |
+
OUT_DIR = f"{EVAL_REQUESTS_PATH}/{user_name}"
|
| 174 |
+
os.makedirs(OUT_DIR, exist_ok=True)
|
| 175 |
+
|
| 176 |
+
# Add the timestamp to the filename
|
| 177 |
+
out_path = f"{OUT_DIR}/{user_name}_{model_path}_{timestamp}.json"
|
| 178 |
+
|
| 179 |
+
with open(out_path, "w") as f:
|
| 180 |
+
f.write(json.dumps(eval_entry))
|
| 181 |
+
|
| 182 |
+
print("Uploading eval file")
|
| 183 |
+
API.upload_file(
|
| 184 |
+
path_or_fileobj=out_path,
|
| 185 |
+
path_in_repo=out_path.split("eval-queue/")[1],
|
| 186 |
+
repo_id=REQUEST_QUEUE_REPO,
|
| 187 |
+
repo_type="dataset",
|
| 188 |
+
commit_message=f"Add {model_name} to eval queue",
|
| 189 |
+
)
|
| 190 |
+
|
| 191 |
+
gr.Info("Successfully submitted", duration=10)
|
| 192 |
+
# Remove the local file
|
| 193 |
+
os.remove(out_path)
|
| 194 |
+
except Exception as e:
|
| 195 |
+
gr.Error(f"Error submitting the model: {e}")
|