anikfekr commited on
Commit
d220025
·
1 Parent(s): 36634df

refactor: streamline leaderboard functionality and enhance submission process

Browse files
__pycache__/envs.cpython-312.pyc ADDED
Binary file (722 Bytes). View file
 
__pycache__/utils.cpython-312.pyc ADDED
Binary file (7.25 kB). View file
 
app.py CHANGED
@@ -1,12 +1,14 @@
1
  import gradio as gr
2
- import pandas as pd
3
- from datasets import load_dataset
4
- from gradio_leaderboard import Leaderboard, ColumnFilter, SelectColumns
 
 
 
 
 
 
5
 
6
- # HF Variables
7
- USERNAME = "alinf"
8
- DATASET_NAME = "crcis_multiple-choice_results"
9
- SPLIT_NAME = "results"
10
 
11
  # Dataset Variables
12
  overall_score_column = "overall"
@@ -15,300 +17,53 @@ precision_column = "precision"
15
  num_params_column = "#parameters (B)"
16
  num_params_range = [0, 100] # in billions
17
 
18
- # Leaderboard title and description
19
- TITLE = """
20
- <h1 align="center">🏆 {} Leaderboard</h1>
21
- """.format(USERNAME)
22
 
23
- DESCRIPTION = """
24
- <p align="center">Evaluating Large Language Models on {} Benchmark</p>
25
- <p align="center">
26
- <a href="https://huggingface.co/datasets/{}/{}">Dataset</a> •
27
- <a href="https://arxiv.org/abs/xxxx.xxxxx">Paper</a> •
28
- <a href="https://github.com/your-username/your-benchmark">GitHub</a>
29
- </p>
30
- """.format(USERNAME, USERNAME, DATASET_NAME)
31
 
32
- def load_results():
33
- """Load results from HuggingFace dataset."""
34
- try:
35
- dataset = load_dataset(
36
- f"{USERNAME}/{DATASET_NAME}",
37
- split=SPLIT_NAME,
38
- download_mode="force_redownload"
39
- )
40
- print(f"Loaded {len(dataset)} entries from the dataset.")
41
- return dataset
42
- except Exception as e:
43
- print(f"Error loading dataset: {e}")
44
- # Return sample data if loading fails
45
- return pd.DataFrame([
46
- {
47
- "rank": 1,
48
- "model": "Sample Model",
49
- "overall": 75.5,
50
- "precision": "fp16",
51
- "#parameters (B)": 7.0
52
- }
53
- ])
54
 
55
- def prepare_leaderboard_data():
56
- """Prepare and format the leaderboard DataFrame."""
57
- data = load_results()
58
- df = pd.DataFrame(data)
59
-
60
- # Remove columns with parentheses (except those with #)
61
- df = df[[col for col in df.columns if ("(" not in col and ")" not in col) or "#" in col]]
62
-
63
- # Sort by overall score (descending)
64
- df = df.sort_values(by=overall_score_column, ascending=False)
65
 
66
- # Add rank column
67
- df.insert(0, "Rank", range(1, len(df) + 1))
 
68
 
69
- # Add a hidden column for model links (useful for search)
70
- df['model_link'] = df[model_name_column].apply(
71
- lambda x: f"https://huggingface.co/{x}" if pd.notna(x) else ""
72
- )
 
 
 
 
 
 
 
 
73
 
74
- return df
75
-
76
- # Custom CSS for styling
77
- custom_css = """
78
- .gradio-container {
79
- font-family: 'IBM Plex Sans', sans-serif;
80
- }
81
-
82
- .leaderboard-table {
83
- margin-top: 20px;
84
- }
85
-
86
- h1 {
87
- color: #2c3e50;
88
- margin-bottom: 10px;
89
- }
90
-
91
- .description {
92
- color: #7f8c8d;
93
- margin-bottom: 30px;
94
- }
95
-
96
- .tab-nav button {
97
- font-size: 16px;
98
- font-weight: 600;
99
- }
100
- """
101
-
102
- # Prepare the data
103
- df = prepare_leaderboard_data()
104
-
105
- print(df)
106
-
107
- # Determine filterable columns
108
- filterable_columns = []
109
- filterable_columns.append(precision_column)
110
- filterable_columns.append(
111
- ColumnFilter(num_params_column, type="slider", min=num_params_range[0], max=num_params_range[1])
112
- )
113
-
114
- # Determine search columns
115
- search_columns = []
116
- search_columns.append(model_name_column)
117
-
118
- # Columns to hide
119
- hidden_columns = ["model_link"]
120
-
121
- # Default visible columns (all except hidden ones)
122
- visible_columns = [col for col in df.columns if col not in hidden_columns]
123
-
124
- columns_datatype = df[visible_columns].dtypes.to_list()
125
-
126
- # Create Gradio interface
127
- with gr.Blocks(title=f"{USERNAME} Benchmark Leaderboard", css=custom_css) as demo:
128
- gr.HTML(TITLE)
129
- gr.Markdown(DESCRIPTION, elem_classes="description")
130
 
131
- with gr.Tabs():
132
- # Main Leaderboard Tab
133
- with gr.TabItem("🏆 Leaderboard"):
134
- gr.Markdown("""
135
- **💡 Tips:**
136
- - Models are ranked by overall score by default
137
- - Click column headers to sort by different metrics
138
- - Use filters to find models that match your requirements
139
- - Search supports multiple queries separated by semicolons (;)
140
- ---
141
- """)
142
-
143
- # Create the leaderboard with gradio_leaderboard
144
- Leaderboard(
145
- value=df,
146
- datatype=columns_datatype,
147
- select_columns=SelectColumns(
148
- default_selection=visible_columns,
149
- cant_deselect=[model_name_column, "Rank"],
150
- label="📊 Select Columns to Display",
151
- info="Choose which metrics to show in the leaderboard"
152
- ),
153
- search_columns=search_columns if search_columns else None,
154
- hide_columns=hidden_columns,
155
- filter_columns=filterable_columns if filterable_columns else None,
156
- interactive=False,
157
- elem_classes="leaderboard-table"
158
- )
159
-
160
- # About Tab
161
- with gr.TabItem("📖 About"):
162
- gr.Markdown(f"""
163
- ## About This Benchmark
164
-
165
- This leaderboard tracks the performance of Large Language Models on the **{USERNAME} Benchmark**.
166
-
167
- ### Evaluation Details
168
- - **Tasks**: Multiple-choice questions across multiple subjects
169
- - **Evaluation Method**: 5-shot evaluation using lm-evaluation-harness
170
- - **Metric**: Accuracy (%)
171
- - **Dataset**: [{USERNAME}/{DATASET_NAME}](https://huggingface.co/datasets/{USERNAME}/{DATASET_NAME})
172
-
173
- ### How to Submit Your Model
174
-
175
- 1. **Evaluate your model** using lm-evaluation-harness:
176
- ```bash
177
- lm_eval --model hf \\
178
- --model_args pretrained=your-org/your-model \\
179
- --tasks {DATASET_NAME} \\
180
- --num_fewshot 5 \\
181
- --output_path ./results
182
- ```
183
-
184
- 2. **Submit results** via [GitHub Issues](https://github.com/your-username/your-benchmark/issues)
185
-
186
- 3. **Include the following information**:
187
- - Model name and organization
188
- - All evaluation scores
189
- - Evaluation logs or result files
190
- - Model precision and parameter count
191
-
192
- ### Evaluation Criteria
193
-
194
- We verify all submissions to ensure:
195
- - Results are reproducible
196
- - Evaluation was performed correctly
197
- - Model information is accurate
198
-
199
- Results are typically added within 48 hours of submission.
200
-
201
- ### Citation
202
-
203
- If you use this benchmark in your research, please cite:
204
-
205
- ```bibtex
206
- @article{{yourbenchmark2024,
207
- title={{Your Benchmark Title}},
208
- author={{Your Name}},
209
- journal={{arXiv preprint arXiv:xxxx.xxxxx}},
210
- year={{2024}}
211
- }}
212
- ```
213
-
214
- ### Contact
215
-
216
- For questions or issues, please:
217
- - Open an issue on [GitHub](https://github.com/your-username/your-benchmark/issues)
218
- - Contact us at your-email@example.com
219
- """)
220
 
221
- # Submit Tab
222
- with gr.TabItem("📤 Submit"):
223
- gr.Markdown("""
224
- ## Submit Your Model Results
225
-
226
- Ready to add your model to the leaderboard? Follow these steps:
227
- """)
228
-
229
- with gr.Accordion("📝 Submission Form", open=True):
230
- gr.Markdown("""
231
- Please fill out the form below with your model information.
232
- Note: This is for display purposes. Actual submissions should be made via GitHub Issues.
233
- """)
234
-
235
- with gr.Row():
236
- model_name = gr.Textbox(
237
- label="Model Name",
238
- placeholder="e.g., gpt-4",
239
- info="The name of your model"
240
- )
241
- organization = gr.Textbox(
242
- label="Organization",
243
- placeholder="e.g., OpenAI",
244
- info="Your organization or username"
245
- )
246
-
247
- with gr.Row():
248
- precision = gr.Dropdown(
249
- label="Precision",
250
- choices=["fp32", "fp16", "bf16", "int8", "int4"],
251
- value="fp16",
252
- info="Model precision used for evaluation"
253
- )
254
- param_count = gr.Number(
255
- label="Parameters (Billions)",
256
- info="Number of parameters in billions (e.g., 7.0)"
257
- )
258
-
259
- with gr.Row():
260
- overall_score = gr.Number(
261
- label="Overall Score",
262
- info="Average score across all tasks (e.g., 75.5)"
263
- )
264
-
265
- results_file = gr.File(
266
- label="Upload Results JSON",
267
- file_types=[".json", ".jsonl"],
268
- )
269
-
270
- contact_email = gr.Textbox(
271
- label="Contact Email",
272
- placeholder="your-email@example.com",
273
- info="We'll contact you about your submission"
274
- )
275
-
276
- submit_btn = gr.Button("📤 Submit for Review", variant="primary", size="lg")
277
-
278
- output_message = gr.Markdown(visible=False)
279
-
280
- def handle_submission(model, org, prec, params, score, file, email):
281
- """Handle submission form."""
282
- if not all([model, org, score, email]):
283
- return gr.Markdown(
284
- "⚠️ Please fill in all required fields.",
285
- visible=True
286
- )
287
-
288
- return gr.Markdown(
289
- f"""
290
- ✅ **Thank you for your submission!**
291
-
292
- We've received your submission for **{model}** from **{org}**.
293
-
294
- **Next steps:**
295
- 1. Create an issue on our [GitHub repository](https://github.com/your-username/your-benchmark/issues)
296
- 2. Include all the information you provided here
297
- 3. Attach your results file
298
-
299
- We'll review your submission and add it to the leaderboard within 48 hours.
300
-
301
- You'll receive a notification at **{email}** when your model is added.
302
- """,
303
- visible=True
304
- )
305
-
306
- submit_btn.click(
307
- fn=handle_submission,
308
- inputs=[model_name, organization, precision, param_count,
309
- overall_score, results_file, contact_email],
310
- outputs=output_message
311
- )
312
 
313
  if __name__ == "__main__":
314
  demo.launch()
 
1
  import gradio as gr
2
+ from gradio_leaderboard import Leaderboard, SelectColumns
3
+
4
+ from utils import (
5
+ LLM_BENCHMARKS_ABOUT_TEXT,
6
+ LLM_BENCHMARKS_SUBMIT_TEXT,
7
+ custom_css,
8
+ prepare_leaderboard_data,
9
+ submit
10
+ )
11
 
 
 
 
 
12
 
13
  # Dataset Variables
14
  overall_score_column = "overall"
 
17
  num_params_column = "#parameters (B)"
18
  num_params_range = [0, 100] # in billions
19
 
20
+ # Prepare the data
21
+ leaderboard_df = prepare_leaderboard_data(overall_score_column, model_name_column)
 
 
22
 
23
+ # Determine columns data type (markdown for all to support clickable links)
24
+ columns_datatype = ["markdown" for _ in range(len(leaderboard_df.columns))]
 
 
 
 
 
 
25
 
26
+ NUM_MODELS = len(leaderboard_df)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
27
 
28
+ with gr.Blocks(
29
+ css=custom_css,
30
+ theme=gr.themes.Default(
31
+ font=["sans-serif", "ui-sans-serif", "system-ui"],
32
+ font_mono=["monospace", "ui-monospace", "Consolas"]
33
+ )
34
+ ) as demo:
35
+ gr.Markdown("""
36
+ # 🏆 CRCIS LLM Leaderboard
37
+ """)
38
 
39
+ gr.Markdown(f"""
40
+ - **Total Models**: {NUM_MODELS}
41
+ """)
42
 
43
+ with gr.Tab("🏆 Leaderboard"):
44
+ Leaderboard(
45
+ value=leaderboard_df,
46
+ datatype=columns_datatype,
47
+ select_columns=SelectColumns(
48
+ default_selection=leaderboard_df.columns.tolist(),
49
+ cant_deselect=[model_name_column],
50
+ label="Select Columns to Show",
51
+ ),
52
+ search_columns=[model_name_column],
53
+ filter_columns=[precision_column, num_params_column],
54
+ )
55
 
56
+ with gr.TabItem("📖 About"):
57
+ gr.Markdown(LLM_BENCHMARKS_ABOUT_TEXT)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
58
 
59
+ with gr.Tab("📤 Submit"):
60
+ gr.Markdown(LLM_BENCHMARKS_SUBMIT_TEXT)
61
+ model_name = gr.Textbox(label="Model name")
62
+ model_id = gr.Textbox(label="Model ID (e.g., username/model-name)")
63
+ contact_email = gr.Textbox(label="Contact E-Mail")
64
+ submit_btn = gr.Button("Submit")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
65
 
66
+ submit_btn.click(submit, inputs=[model_name, model_id, contact_email], outputs=[])
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
67
 
68
  if __name__ == "__main__":
69
  demo.launch()
envs.py ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+
3
+ from huggingface_hub import HfApi
4
+
5
+ # Info to change for your repository
6
+ # ----------------------------------
7
+ TOKEN = os.environ.get("HF_TOKEN")
8
+
9
+ OWNER = "alinf" # Change to your HF username
10
+ RESULT_DATASET_NAME = "crcis_multiple-choice_results"
11
+ # ----------------------------------
12
+
13
+ REQUEST_QUEUE_REPO = f"{OWNER}/crcis_request_queue"
14
+
15
+ # If you setup a cache later, just set/change 'HF_HOME' env variable
16
+ CACHE_PATH = os.getenv("HF_HOME", ".")
17
+
18
+ # Local caches
19
+ EVAL_REQUESTS_PATH = os.path.join(CACHE_PATH, "eval-queue")
20
+
21
+ API = HfApi(token=TOKEN)
utils.py ADDED
@@ -0,0 +1,195 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import json
2
+ import os
3
+ from datetime import datetime
4
+
5
+ import gradio as gr
6
+ import pandas as pd
7
+ from datasets import load_dataset
8
+
9
+ from envs import API, EVAL_REQUESTS_PATH, REQUEST_QUEUE_REPO, OWNER, RESULT_DATASET_NAME
10
+
11
+ custom_css = """
12
+ .gradio-container {
13
+ font-family: 'IBM Plex Sans', sans-serif;
14
+ }
15
+
16
+ .leaderboard-table {
17
+ margin-top: 20px;
18
+ }
19
+
20
+ h1 {
21
+ color: #2c3e50;
22
+ margin-bottom: 10px;
23
+ }
24
+
25
+ .description {
26
+ color: #7f8c8d;
27
+ margin-bottom: 30px;
28
+ }
29
+
30
+ .tab-nav button {
31
+ font-size: 16px;
32
+ font-weight: 600;
33
+ }
34
+
35
+ #leaderboard-table {
36
+ margin-top: 15px;
37
+ text-align: center;
38
+ }
39
+
40
+ #leaderboard-table th,
41
+ #leaderboard-table td {
42
+ text-align: center;
43
+ vertical-align: middle;
44
+ }
45
+
46
+ #leaderboard-table td:first-child,
47
+ #leaderboard-table th:first-child {
48
+ text-align: left;
49
+ max-width: 500px;
50
+ }
51
+ """
52
+
53
+ LLM_BENCHMARKS_ABOUT_TEXT = f"""
54
+ # CRCIS LLM Leaderboard
55
+
56
+ This leaderboard tracks the performance of Large Language Models on the **CRCIS Benchmark**.
57
+
58
+ ## Evaluation Details
59
+ - **Tasks**: Multiple-choice questions across multiple subjects including:
60
+ - Humanities - History
61
+ - Quran - General Information
62
+ - Quran - Tafsir
63
+ - **Evaluation Method**: 5-shot evaluation using lm-evaluation-harness
64
+ - **Metric**: Accuracy (%)
65
+ - **Dataset**: [{OWNER}/{RESULT_DATASET_NAME}](https://huggingface.co/datasets/{OWNER}/{RESULT_DATASET_NAME})
66
+
67
+ ## Background and Goals
68
+
69
+ This leaderboard provides a comprehensive benchmarking system for evaluating LLMs on CRCIS-specific tasks.
70
+ The evaluation framework is based on the open-source [LM Evaluation Harness](https://github.com/EleutherAI/lm-evaluation-harness),
71
+ offering a reliable platform for assessing model performance on specialized domain knowledge.
72
+
73
+ ## Data Integrity
74
+
75
+ To maintain evaluation integrity and prevent overfitting, the full benchmark dataset is used for evaluation.
76
+ This approach ensures that results genuinely represent each model's capabilities.
77
+
78
+ """
79
+
80
+ LLM_BENCHMARKS_SUBMIT_TEXT = """## Submitting a Model for Evaluation
81
+
82
+ To submit your model for evaluation, follow these steps:
83
+
84
+ 1. **Ensure your model is on Hugging Face**: Your model must be publicly available on [Hugging Face](https://huggingface.co/).
85
+
86
+ 2. **Submit Request**: Fill out the form below with your model's information and Hugging Face identifier.
87
+
88
+ 3. **Evaluation Queue**: Submissions will be queued and processed. The evaluation may take some time depending on the queue.
89
+
90
+ 4. **Results**: Once the evaluation is complete, your model's results will be E-mailed to you.
91
+
92
+ We appreciate your contributions to the LLM ecosystem!
93
+ """
94
+
95
+
96
+ def load_results_from_hf_dataset():
97
+ """Load results from HuggingFace dataset."""
98
+ try:
99
+ dataset = load_dataset(
100
+ f"{OWNER}/{RESULT_DATASET_NAME}",
101
+ split="results",
102
+ download_mode="force_redownload"
103
+ )
104
+ print(f"Loaded {len(dataset)} entries from the dataset.")
105
+ return pd.DataFrame(dataset)
106
+ except Exception as e:
107
+ print(f"Error loading dataset: {e}")
108
+ # Return sample data if loading fails
109
+ return pd.DataFrame([
110
+ {
111
+ "model": "Sample Model",
112
+ "overall": 75.5,
113
+ "precision": "fp16",
114
+ "#parameters (B)": 7.0
115
+ }
116
+ ])
117
+
118
+
119
+ def sort_dataframe_by_column(df, column_name):
120
+ """Sort dataframe by specified column in descending order."""
121
+ if column_name not in df.columns:
122
+ raise ValueError(f"Column '{column_name}' does not exist in the DataFrame.")
123
+ return df.sort_values(by=column_name, ascending=False).reset_index(drop=True)
124
+
125
+
126
+ def make_clickable_model(model_name):
127
+ """Make model name clickable with link to HuggingFace."""
128
+ link = f"https://huggingface.co/{model_name}"
129
+ style = "color: var(--link-text-color); text-decoration: underline; text-decoration-style: dotted;"
130
+ return f'<a target="_blank" href="{link}" style="{style}">{model_name}</a>'
131
+
132
+
133
+ def prepare_leaderboard_data(overall_score_column="overall", model_name_column="model"):
134
+ """Prepare and format the leaderboard DataFrame."""
135
+ df = load_results_from_hf_dataset()
136
+
137
+ # Remove columns with parentheses (except those with #)
138
+ df = df[[col for col in df.columns if ("(" not in col and ")" not in col) or "#" in col]]
139
+
140
+ # Sort by overall score (descending)
141
+ df = sort_dataframe_by_column(df, overall_score_column)
142
+
143
+ # Make model names clickable
144
+ df[model_name_column] = df[model_name_column].apply(make_clickable_model)
145
+
146
+ return df
147
+
148
+
149
+ def submit(model_name, model_id, contact_email):
150
+ """Handle model submission to evaluation queue."""
151
+ if model_name == "" or model_id == "" or contact_email == "":
152
+ gr.Info("Please fill all the fields")
153
+ return
154
+
155
+ try:
156
+ user_name = ""
157
+ if "/" in model_id:
158
+ user_name = model_id.split("/")[0]
159
+ model_path = model_id.split("/")[1]
160
+ else:
161
+ gr.Error("Model ID must be in the format 'username/model-name'")
162
+ return
163
+
164
+ eval_entry = {
165
+ "model_name": model_name,
166
+ "model_id": model_id,
167
+ "contact_email": contact_email,
168
+ }
169
+
170
+ # Get the current timestamp to add to the filename
171
+ timestamp = datetime.now().strftime("%Y%m%d")
172
+
173
+ OUT_DIR = f"{EVAL_REQUESTS_PATH}/{user_name}"
174
+ os.makedirs(OUT_DIR, exist_ok=True)
175
+
176
+ # Add the timestamp to the filename
177
+ out_path = f"{OUT_DIR}/{user_name}_{model_path}_{timestamp}.json"
178
+
179
+ with open(out_path, "w") as f:
180
+ f.write(json.dumps(eval_entry))
181
+
182
+ print("Uploading eval file")
183
+ API.upload_file(
184
+ path_or_fileobj=out_path,
185
+ path_in_repo=out_path.split("eval-queue/")[1],
186
+ repo_id=REQUEST_QUEUE_REPO,
187
+ repo_type="dataset",
188
+ commit_message=f"Add {model_name} to eval queue",
189
+ )
190
+
191
+ gr.Info("Successfully submitted", duration=10)
192
+ # Remove the local file
193
+ os.remove(out_path)
194
+ except Exception as e:
195
+ gr.Error(f"Error submitting the model: {e}")