anikfekr commited on
Commit
dae933d
Β·
1 Parent(s): 4f00cf8
Files changed (2) hide show
  1. app.py +321 -0
  2. requirements.txt +2 -0
app.py ADDED
@@ -0,0 +1,321 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import gradio as gr
2
+ import pandas as pd
3
+ from datasets import load_dataset
4
+ from gradio_leaderboard import Leaderboard, ColumnFilter, SelectColumns
5
+
6
+ # HF Variables
7
+ USERNAME = "alinf"
8
+ DATASET_NAME = "crcis_multiple-choice_results"
9
+ SPLIT_NAME = "results"
10
+
11
+ # Dataset Variables
12
+ overall_score_column = "overall"
13
+ model_name_column = "model"
14
+ precision_column = "precision"
15
+ num_params_column = "#parameters (B)"
16
+ num_params_range = [0, 100] # in billions
17
+
18
+ # Leaderboard title and description
19
+ TITLE = """
20
+ <h1 align="center">πŸ† {} Leaderboard</h1>
21
+ """.format(USERNAME)
22
+
23
+ DESCRIPTION = """
24
+ <p align="center">Evaluating Large Language Models on {} Benchmark</p>
25
+ <p align="center">
26
+ <a href="https://huggingface.co/datasets/{}/{}">Dataset</a> β€’
27
+ <a href="https://arxiv.org/abs/xxxx.xxxxx">Paper</a> β€’
28
+ <a href="https://github.com/your-username/your-benchmark">GitHub</a>
29
+ </p>
30
+ """.format(USERNAME, USERNAME, DATASET_NAME)
31
+
32
+ def load_results():
33
+ """Load results from HuggingFace dataset."""
34
+ try:
35
+ dataset = load_dataset(
36
+ f"{USERNAME}/{DATASET_NAME}",
37
+ split=SPLIT_NAME,
38
+ download_mode="force_redownload"
39
+ )
40
+ print(f"Loaded {len(dataset)} entries from the dataset.")
41
+ return dataset
42
+ except Exception as e:
43
+ print(f"Error loading dataset: {e}")
44
+ # Return sample data if loading fails
45
+ return pd.DataFrame([
46
+ {
47
+ "rank": 1,
48
+ "model": "Sample Model",
49
+ "overall": 75.5,
50
+ "precision": "fp16",
51
+ "#parameters (B)": 7.0
52
+ }
53
+ ])
54
+
55
+ def prepare_leaderboard_data():
56
+ """Prepare and format the leaderboard DataFrame."""
57
+ data = load_results()
58
+ df = pd.DataFrame(data)
59
+
60
+ # Remove columns with parentheses (except those with #)
61
+ df = df[[col for col in df.columns if ("(" not in col and ")" not in col) or "#" in col]]
62
+
63
+ # Sort by overall score (descending)
64
+ df = df.sort_values(by=overall_score_column, ascending=False)
65
+
66
+ # Add rank column
67
+ df.insert(0, "Rank", range(1, len(df) + 1))
68
+
69
+ # Add a hidden column for model links (useful for search)
70
+ df['model_link'] = df[model_name_column].apply(
71
+ lambda x: f"https://huggingface.co/{x}" if pd.notna(x) else ""
72
+ )
73
+
74
+ return df
75
+
76
+ # Custom CSS for styling
77
+ custom_css = """
78
+ .gradio-container {
79
+ font-family: 'IBM Plex Sans', sans-serif;
80
+ }
81
+
82
+ .leaderboard-table {
83
+ margin-top: 20px;
84
+ }
85
+
86
+ h1 {
87
+ color: #2c3e50;
88
+ margin-bottom: 10px;
89
+ }
90
+
91
+ .description {
92
+ color: #7f8c8d;
93
+ margin-bottom: 30px;
94
+ }
95
+
96
+ .tab-nav button {
97
+ font-size: 16px;
98
+ font-weight: 600;
99
+ }
100
+ """
101
+
102
+ # Prepare the data
103
+ df = prepare_leaderboard_data()
104
+
105
+ # Determine filterable columns
106
+ filterable_columns = []
107
+ filterable_columns.append(precision_column)
108
+ filterable_columns.append(
109
+ ColumnFilter(num_params_column, type="slider", min=num_params_range[0], max=num_params_range[1])
110
+ )
111
+
112
+ # Determine search columns
113
+ search_columns = []
114
+ search_columns.append(model_name_column)
115
+
116
+ # Columns to hide
117
+ hidden_columns = ["model_link"]
118
+
119
+ # Default visible columns (all except hidden ones)
120
+ visible_columns = [col for col in df.columns if col not in hidden_columns]
121
+
122
+ columns_datatype = df[visible_columns].dtypes.to_list()
123
+
124
+ # Create Gradio interface
125
+ with gr.Blocks(title=f"{USERNAME} Benchmark Leaderboard", css=custom_css) as demo:
126
+ gr.HTML(TITLE)
127
+ gr.Markdown(DESCRIPTION, elem_classes="description")
128
+
129
+ with gr.Tabs():
130
+ # Main Leaderboard Tab
131
+ with gr.TabItem("πŸ† Leaderboard"):
132
+ gr.Markdown("""
133
+ ### How to use this leaderboard:
134
+ - **Search**: Use the search box to find models by name
135
+ - **Filter**: Use the filters below to narrow down results by precision, parameters, etc.
136
+ - **Sort**: Click on column headers to sort by that metric
137
+ - **Select Columns**: Choose which columns to display using the column selector
138
+ """)
139
+
140
+ # Create the leaderboard with gradio_leaderboard
141
+ leaderboard = Leaderboard(
142
+ value=df,
143
+ datatype=columns_datatype,
144
+ select_columns=SelectColumns(
145
+ default_selection=visible_columns,
146
+ cant_deselect=[model_name_column, "Rank"],
147
+ label="πŸ“Š Select Columns to Display",
148
+ info="Choose which metrics to show in the leaderboard"
149
+ ),
150
+ search_columns=search_columns if search_columns else None,
151
+ hide_columns=hidden_columns,
152
+ filter_columns=filterable_columns if filterable_columns else None,
153
+ interactive=False,
154
+ elem_classes="leaderboard-table"
155
+ )
156
+
157
+ gr.Markdown("""
158
+ ---
159
+ **πŸ’‘ Tips:**
160
+ - Models are ranked by overall score by default
161
+ - Click column headers to sort by different metrics
162
+ - Use filters to find models that match your requirements
163
+ - Search supports multiple queries separated by semicolons (;)
164
+ - To search in specific columns, use `column_name: query`
165
+ """)
166
+
167
+ # About Tab
168
+ with gr.TabItem("πŸ“– About"):
169
+ gr.Markdown(f"""
170
+ ## About This Benchmark
171
+
172
+ This leaderboard tracks the performance of Large Language Models on the **{USERNAME} Benchmark**.
173
+
174
+ ### Evaluation Details
175
+ - **Tasks**: Multiple-choice questions across multiple subjects
176
+ - **Evaluation Method**: 5-shot evaluation using lm-evaluation-harness
177
+ - **Metric**: Accuracy (%)
178
+ - **Dataset**: [{USERNAME}/{DATASET_NAME}](https://huggingface.co/datasets/{USERNAME}/{DATASET_NAME})
179
+
180
+ ### How to Submit Your Model
181
+
182
+ 1. **Evaluate your model** using lm-evaluation-harness:
183
+ ```bash
184
+ lm_eval --model hf \\
185
+ --model_args pretrained=your-org/your-model \\
186
+ --tasks {DATASET_NAME} \\
187
+ --num_fewshot 5 \\
188
+ --output_path ./results
189
+ ```
190
+
191
+ 2. **Submit results** via [GitHub Issues](https://github.com/your-username/your-benchmark/issues)
192
+
193
+ 3. **Include the following information**:
194
+ - Model name and organization
195
+ - All evaluation scores
196
+ - Evaluation logs or result files
197
+ - Model precision and parameter count
198
+
199
+ ### Evaluation Criteria
200
+
201
+ We verify all submissions to ensure:
202
+ - Results are reproducible
203
+ - Evaluation was performed correctly
204
+ - Model information is accurate
205
+
206
+ Results are typically added within 48 hours of submission.
207
+
208
+ ### Citation
209
+
210
+ If you use this benchmark in your research, please cite:
211
+
212
+ ```bibtex
213
+ @article{{yourbenchmark2024,
214
+ title={{Your Benchmark Title}},
215
+ author={{Your Name}},
216
+ journal={{arXiv preprint arXiv:xxxx.xxxxx}},
217
+ year={{2024}}
218
+ }}
219
+ ```
220
+
221
+ ### Contact
222
+
223
+ For questions or issues, please:
224
+ - Open an issue on [GitHub](https://github.com/your-username/your-benchmark/issues)
225
+ - Contact us at your-email@example.com
226
+ """)
227
+
228
+ # Submit Tab
229
+ with gr.TabItem("πŸ“€ Submit"):
230
+ gr.Markdown("""
231
+ ## Submit Your Model Results
232
+
233
+ Ready to add your model to the leaderboard? Follow these steps:
234
+ """)
235
+
236
+ with gr.Accordion("πŸ“ Submission Form", open=True):
237
+ gr.Markdown("""
238
+ Please fill out the form below with your model information.
239
+ Note: This is for display purposes. Actual submissions should be made via GitHub Issues.
240
+ """)
241
+
242
+ with gr.Row():
243
+ model_name = gr.Textbox(
244
+ label="Model Name",
245
+ placeholder="e.g., gpt-4",
246
+ info="The name of your model"
247
+ )
248
+ organization = gr.Textbox(
249
+ label="Organization",
250
+ placeholder="e.g., OpenAI",
251
+ info="Your organization or username"
252
+ )
253
+
254
+ with gr.Row():
255
+ precision = gr.Dropdown(
256
+ label="Precision",
257
+ choices=["fp32", "fp16", "bf16", "int8", "int4"],
258
+ value="fp16",
259
+ info="Model precision used for evaluation"
260
+ )
261
+ param_count = gr.Number(
262
+ label="Parameters (Billions)",
263
+ info="Number of parameters in billions (e.g., 7.0)"
264
+ )
265
+
266
+ with gr.Row():
267
+ overall_score = gr.Number(
268
+ label="Overall Score",
269
+ info="Average score across all tasks (e.g., 75.5)"
270
+ )
271
+
272
+ results_file = gr.File(
273
+ label="Upload Results JSON",
274
+ file_types=[".json", ".jsonl"],
275
+ )
276
+
277
+ contact_email = gr.Textbox(
278
+ label="Contact Email",
279
+ placeholder="your-email@example.com",
280
+ info="We'll contact you about your submission"
281
+ )
282
+
283
+ submit_btn = gr.Button("πŸ“€ Submit for Review", variant="primary", size="lg")
284
+
285
+ output_message = gr.Markdown(visible=False)
286
+
287
+ def handle_submission(model, org, prec, params, score, file, email):
288
+ """Handle submission form."""
289
+ if not all([model, org, score, email]):
290
+ return gr.Markdown(
291
+ "⚠️ Please fill in all required fields.",
292
+ visible=True
293
+ )
294
+
295
+ return gr.Markdown(
296
+ f"""
297
+ βœ… **Thank you for your submission!**
298
+
299
+ We've received your submission for **{model}** from **{org}**.
300
+
301
+ **Next steps:**
302
+ 1. Create an issue on our [GitHub repository](https://github.com/your-username/your-benchmark/issues)
303
+ 2. Include all the information you provided here
304
+ 3. Attach your results file
305
+
306
+ We'll review your submission and add it to the leaderboard within 48 hours.
307
+
308
+ You'll receive a notification at **{email}** when your model is added.
309
+ """,
310
+ visible=True
311
+ )
312
+
313
+ submit_btn.click(
314
+ fn=handle_submission,
315
+ inputs=[model_name, organization, precision, param_count,
316
+ overall_score, results_file, contact_email],
317
+ outputs=output_message
318
+ )
319
+
320
+ if __name__ == "__main__":
321
+ demo.launch()
requirements.txt ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ gradio_leaderboard
2
+ datasets