Pedro Dias commited on
Commit
5967a6e
·
1 Parent(s): bba5081
src/about.py CHANGED
@@ -1,6 +1,7 @@
1
  from dataclasses import dataclass
2
  from enum import Enum
3
 
 
4
  @dataclass
5
  class Task:
6
  benchmark: str
@@ -11,13 +12,13 @@ class Task:
11
  # Select your tasks here
12
  # ---------------------------------------------------
13
  class Tasks(Enum):
14
- # task_key in the json file, metric_key in the json file, name to display in the leaderboard
15
  task0 = Task("anli_r1", "acc", "ANLI")
16
  task1 = Task("logiqa", "acc_norm", "LogiQA")
17
 
18
- NUM_FEWSHOT = 0 # Change with your few shot
19
- # ---------------------------------------------------
20
 
 
 
21
 
22
 
23
  # Your leaderboard name
 
1
  from dataclasses import dataclass
2
  from enum import Enum
3
 
4
+
5
  @dataclass
6
  class Task:
7
  benchmark: str
 
12
  # Select your tasks here
13
  # ---------------------------------------------------
14
  class Tasks(Enum):
15
+ # task_key in the json file, metric_key in the json file, name to display in the leaderboard
16
  task0 = Task("anli_r1", "acc", "ANLI")
17
  task1 = Task("logiqa", "acc_norm", "LogiQA")
18
 
 
 
19
 
20
+ NUM_FEWSHOT = 0 # Change with your few shot
21
+ # ---------------------------------------------------
22
 
23
 
24
  # Your leaderboard name
src/display/utils.py CHANGED
@@ -1,10 +1,12 @@
1
- from dataclasses import dataclass, make_dataclass, field
2
  from enum import Enum
3
- from typing import List, Tuple, Any
 
4
  import pandas as pd
5
 
6
  from src.about import Tasks
7
 
 
8
  def fields(raw_class):
9
  return [v for k, v in raw_class.__dict__.items() if k[:2] != "__" and k[-2:] != "__"]
10
 
@@ -20,6 +22,7 @@ class ColumnContent:
20
  hidden: bool = False
21
  never_hidden: bool = False
22
 
 
23
  # Create the dataclass dynamically with all the expected fields
24
  field_specs = [
25
  ("model_type_symbol", ColumnContent),
@@ -32,17 +35,19 @@ for task in Tasks:
32
  field_specs.append((task.name, ColumnContent))
33
 
34
  # Add model information
35
- field_specs.extend([
36
- ("model_type", ColumnContent),
37
- ("architecture", ColumnContent),
38
- ("weight_type", ColumnContent),
39
- ("precision", ColumnContent),
40
- ("license", ColumnContent),
41
- ("params", ColumnContent),
42
- ("likes", ColumnContent),
43
- ("still_on_hub", ColumnContent),
44
- ("revision", ColumnContent),
45
- ])
 
 
46
 
47
  # Create the dataclass
48
  AutoEvalColumn = make_dataclass("AutoEvalColumn", field_specs, frozen=True)
@@ -67,6 +72,7 @@ AutoEvalColumn.likes = ColumnContent("Hub ❤️", "number", False)
67
  AutoEvalColumn.still_on_hub = ColumnContent("Available on the hub", "bool", False)
68
  AutoEvalColumn.revision = ColumnContent("Model sha", "str", False, False)
69
 
 
70
  ## For the queue columns in the submission tab
71
  @dataclass(frozen=True)
72
  class EvalQueueColumn: # Queue column
@@ -77,12 +83,13 @@ class EvalQueueColumn: # Queue column
77
  weight_type = ColumnContent("weight_type", "str", "Original")
78
  status = ColumnContent("status", "str", True)
79
 
 
80
  ## All the model information that we might need
81
  @dataclass
82
  class ModelDetails:
83
  name: str
84
  display_name: str = ""
85
- symbol: str = "" # emoji
86
 
87
 
88
  class ModelType(Enum):
@@ -107,11 +114,13 @@ class ModelType(Enum):
107
  return ModelType.IFT
108
  return ModelType.Unknown
109
 
 
110
  class WeightType(Enum):
111
  Adapter = ModelDetails("Adapter")
112
  Original = ModelDetails("Original")
113
  Delta = ModelDetails("Delta")
114
 
 
115
  class Precision(Enum):
116
  float16 = ModelDetails("float16")
117
  bfloat16 = ModelDetails("bfloat16")
@@ -124,6 +133,7 @@ class Precision(Enum):
124
  return Precision.bfloat16
125
  return Precision.Unknown
126
 
 
127
  # Column selection
128
  COLS = [c.name for c in fields(AutoEvalColumn) if not c.hidden]
129
 
@@ -131,4 +141,3 @@ EVAL_COLS = [c.name for c in fields(EvalQueueColumn)]
131
  EVAL_TYPES = [c.type for c in fields(EvalQueueColumn)]
132
 
133
  BENCHMARK_COLS = [t.value.col_name for t in Tasks]
134
-
 
1
+ from dataclasses import dataclass, field, make_dataclass
2
  from enum import Enum
3
+ from typing import Any, List, Tuple
4
+
5
  import pandas as pd
6
 
7
  from src.about import Tasks
8
 
9
+
10
  def fields(raw_class):
11
  return [v for k, v in raw_class.__dict__.items() if k[:2] != "__" and k[-2:] != "__"]
12
 
 
22
  hidden: bool = False
23
  never_hidden: bool = False
24
 
25
+
26
  # Create the dataclass dynamically with all the expected fields
27
  field_specs = [
28
  ("model_type_symbol", ColumnContent),
 
35
  field_specs.append((task.name, ColumnContent))
36
 
37
  # Add model information
38
+ field_specs.extend(
39
+ [
40
+ ("model_type", ColumnContent),
41
+ ("architecture", ColumnContent),
42
+ ("weight_type", ColumnContent),
43
+ ("precision", ColumnContent),
44
+ ("license", ColumnContent),
45
+ ("params", ColumnContent),
46
+ ("likes", ColumnContent),
47
+ ("still_on_hub", ColumnContent),
48
+ ("revision", ColumnContent),
49
+ ]
50
+ )
51
 
52
  # Create the dataclass
53
  AutoEvalColumn = make_dataclass("AutoEvalColumn", field_specs, frozen=True)
 
72
  AutoEvalColumn.still_on_hub = ColumnContent("Available on the hub", "bool", False)
73
  AutoEvalColumn.revision = ColumnContent("Model sha", "str", False, False)
74
 
75
+
76
  ## For the queue columns in the submission tab
77
  @dataclass(frozen=True)
78
  class EvalQueueColumn: # Queue column
 
83
  weight_type = ColumnContent("weight_type", "str", "Original")
84
  status = ColumnContent("status", "str", True)
85
 
86
+
87
  ## All the model information that we might need
88
  @dataclass
89
  class ModelDetails:
90
  name: str
91
  display_name: str = ""
92
+ symbol: str = "" # emoji
93
 
94
 
95
  class ModelType(Enum):
 
114
  return ModelType.IFT
115
  return ModelType.Unknown
116
 
117
+
118
  class WeightType(Enum):
119
  Adapter = ModelDetails("Adapter")
120
  Original = ModelDetails("Original")
121
  Delta = ModelDetails("Delta")
122
 
123
+
124
  class Precision(Enum):
125
  float16 = ModelDetails("float16")
126
  bfloat16 = ModelDetails("bfloat16")
 
133
  return Precision.bfloat16
134
  return Precision.Unknown
135
 
136
+
137
  # Column selection
138
  COLS = [c.name for c in fields(AutoEvalColumn) if not c.hidden]
139
 
 
141
  EVAL_TYPES = [c.type for c in fields(EvalQueueColumn)]
142
 
143
  BENCHMARK_COLS = [t.value.col_name for t in Tasks]
 
src/envs.py CHANGED
@@ -4,9 +4,9 @@ from huggingface_hub import HfApi
4
 
5
  # Info to change for your repository
6
  # ----------------------------------
7
- TOKEN = os.environ.get("HF_TOKEN") # A read/write token for your org
8
 
9
- OWNER = "lokahq" # Change to your org - don't forget to create a results and request dataset, with the correct format!
10
  # ----------------------------------
11
 
12
  REPO_ID = f"{OWNER}/dna-benchmark"
@@ -14,7 +14,7 @@ QUEUE_REPO = f"{OWNER}/requests"
14
  RESULTS_REPO = f"{OWNER}/bench-dna-results"
15
 
16
  # If you setup a cache later, just change HF_HOME
17
- CACHE_PATH=os.getenv("HF_HOME", ".")
18
 
19
  # Local caches
20
  EVAL_REQUESTS_PATH = os.path.join(CACHE_PATH, "eval-queue")
 
4
 
5
  # Info to change for your repository
6
  # ----------------------------------
7
+ TOKEN = os.environ.get("HF_TOKEN") # A read/write token for your org
8
 
9
+ OWNER = "lokahq" # Change to your org - don't forget to create a results and request dataset, with the correct format!
10
  # ----------------------------------
11
 
12
  REPO_ID = f"{OWNER}/dna-benchmark"
 
14
  RESULTS_REPO = f"{OWNER}/bench-dna-results"
15
 
16
  # If you setup a cache later, just change HF_HOME
17
+ CACHE_PATH = os.getenv("HF_HOME", ".")
18
 
19
  # Local caches
20
  EVAL_REQUESTS_PATH = os.path.join(CACHE_PATH, "eval-queue")
src/leaderboard/read_evals.py CHANGED
@@ -8,28 +8,28 @@ import dateutil
8
  import numpy as np
9
 
10
  from src.display.formatting import make_clickable_model
11
- from src.display.utils import AutoEvalColumn, ModelType, Tasks, Precision, WeightType
12
  from src.submission.check_validity import is_model_on_hub
13
 
14
 
15
  @dataclass
16
  class EvalResult:
17
- """Represents one full evaluation. Built from a combination of the result and request file for a given run.
18
- """
19
- eval_name: str # org_model_precision (uid)
20
- full_model: str # org/model (path on hub)
21
  org: str
22
  model: str
23
- revision: str # commit hash, "" if main
24
  results: dict
25
  precision: Precision = Precision.Unknown
26
- model_type: ModelType = ModelType.Unknown # Pretrained, fine tuned, ...
27
- weight_type: WeightType = WeightType.Original # Original or Adapter
28
  architecture: str = "Unknown"
29
  license: str = "?"
30
  likes: int = 0
31
  num_params: int = 0
32
- date: str = "" # submission date of request file
33
  still_on_hub: bool = False
34
 
35
  @classmethod
@@ -86,9 +86,9 @@ class EvalResult:
86
  model=model,
87
  results=results,
88
  precision=precision,
89
- revision= config.get("model_sha", ""),
90
  still_on_hub=still_on_hub,
91
- architecture=architecture
92
  )
93
 
94
  def update_with_request_file(self, requests_path):
@@ -105,7 +105,9 @@ class EvalResult:
105
  self.num_params = request.get("params", 0)
106
  self.date = request.get("submitted_time", "")
107
  except Exception:
108
- print(f"Could not find request file for {self.org}/{self.model} with precision {self.precision.value.name}")
 
 
109
 
110
  def to_dict(self):
111
  """Converts the Eval Result to a dict compatible with our dataframe display"""
@@ -146,10 +148,7 @@ def get_request_file_for_model(requests_path, model_name, precision):
146
  for tmp_request_file in request_files:
147
  with open(tmp_request_file, "r") as f:
148
  req_content = json.load(f)
149
- if (
150
- req_content["status"] in ["FINISHED"]
151
- and req_content["precision"] == precision.split(".")[-1]
152
- ):
153
  request_file = tmp_request_file
154
  return request_file
155
 
@@ -188,7 +187,7 @@ def get_raw_eval_results(results_path: str, requests_path: str) -> list[EvalResu
188
  results = []
189
  for v in eval_results.values():
190
  try:
191
- v.to_dict() # we test if the dict version is complete
192
  results.append(v)
193
  except KeyError: # not all eval values present
194
  continue
 
8
  import numpy as np
9
 
10
  from src.display.formatting import make_clickable_model
11
+ from src.display.utils import AutoEvalColumn, ModelType, Precision, Tasks, WeightType
12
  from src.submission.check_validity import is_model_on_hub
13
 
14
 
15
  @dataclass
16
  class EvalResult:
17
+ """Represents one full evaluation. Built from a combination of the result and request file for a given run."""
18
+
19
+ eval_name: str # org_model_precision (uid)
20
+ full_model: str # org/model (path on hub)
21
  org: str
22
  model: str
23
+ revision: str # commit hash, "" if main
24
  results: dict
25
  precision: Precision = Precision.Unknown
26
+ model_type: ModelType = ModelType.Unknown # Pretrained, fine tuned, ...
27
+ weight_type: WeightType = WeightType.Original # Original or Adapter
28
  architecture: str = "Unknown"
29
  license: str = "?"
30
  likes: int = 0
31
  num_params: int = 0
32
+ date: str = "" # submission date of request file
33
  still_on_hub: bool = False
34
 
35
  @classmethod
 
86
  model=model,
87
  results=results,
88
  precision=precision,
89
+ revision=config.get("model_sha", ""),
90
  still_on_hub=still_on_hub,
91
+ architecture=architecture,
92
  )
93
 
94
  def update_with_request_file(self, requests_path):
 
105
  self.num_params = request.get("params", 0)
106
  self.date = request.get("submitted_time", "")
107
  except Exception:
108
+ print(
109
+ f"Could not find request file for {self.org}/{self.model} with precision {self.precision.value.name}"
110
+ )
111
 
112
  def to_dict(self):
113
  """Converts the Eval Result to a dict compatible with our dataframe display"""
 
148
  for tmp_request_file in request_files:
149
  with open(tmp_request_file, "r") as f:
150
  req_content = json.load(f)
151
+ if req_content["status"] in ["FINISHED"] and req_content["precision"] == precision.split(".")[-1]:
 
 
 
152
  request_file = tmp_request_file
153
  return request_file
154
 
 
187
  results = []
188
  for v in eval_results.values():
189
  try:
190
+ v.to_dict() # we test if the dict version is complete
191
  results.append(v)
192
  except KeyError: # not all eval values present
193
  continue
src/submission/check_validity.py CHANGED
@@ -10,6 +10,7 @@ from huggingface_hub.hf_api import ModelInfo
10
  from transformers import AutoConfig
11
  from transformers.models.auto.tokenization_auto import AutoTokenizer
12
 
 
13
  def check_model_card(repo_id: str) -> tuple[bool, str]:
14
  """Checks if the model card and license exist and have been filled"""
15
  try:
@@ -31,28 +32,35 @@ def check_model_card(repo_id: str) -> tuple[bool, str]:
31
 
32
  return True, ""
33
 
34
- def is_model_on_hub(model_name: str, revision: str, token: str = None, trust_remote_code=False, test_tokenizer=False) -> tuple[bool, str]:
 
 
 
35
  """Checks if the model model_name is on the hub, and whether it (and its tokenizer) can be loaded with AutoClasses."""
36
  try:
37
- config = AutoConfig.from_pretrained(model_name, revision=revision, trust_remote_code=trust_remote_code, token=token)
 
 
38
  if test_tokenizer:
39
  try:
40
- tk = AutoTokenizer.from_pretrained(model_name, revision=revision, trust_remote_code=trust_remote_code, token=token)
 
 
41
  except ValueError as e:
 
 
42
  return (
43
  False,
44
- f"uses a tokenizer which is not in a transformers release: {e}",
45
- None
46
  )
47
- except Exception as e:
48
- return (False, "'s tokenizer cannot be loaded. Is your tokenizer class in a stable transformers release, and correctly configured?", None)
49
  return True, None, config
50
 
51
  except ValueError:
52
  return (
53
  False,
54
  "needs to be launched with `trust_remote_code=True`. For safety reason, we do not allow these models to be automatically submitted to the leaderboard.",
55
- None
56
  )
57
 
58
  except Exception as e:
@@ -70,10 +78,12 @@ def get_model_size(model_info: ModelInfo, precision: str):
70
  model_size = size_factor * model_size
71
  return model_size
72
 
 
73
  def get_model_arch(model_info: ModelInfo):
74
  """Gets the model architecture from the configuration"""
75
  return model_info.config.get("architectures", "Unknown")
76
 
 
77
  def already_submitted_models(requested_models_dir: str) -> set[str]:
78
  """Gather a list of already submitted models to avoid duplicates"""
79
  depth = 1
 
10
  from transformers import AutoConfig
11
  from transformers.models.auto.tokenization_auto import AutoTokenizer
12
 
13
+
14
  def check_model_card(repo_id: str) -> tuple[bool, str]:
15
  """Checks if the model card and license exist and have been filled"""
16
  try:
 
32
 
33
  return True, ""
34
 
35
+
36
+ def is_model_on_hub(
37
+ model_name: str, revision: str, token: str = None, trust_remote_code=False, test_tokenizer=False
38
+ ) -> tuple[bool, str]:
39
  """Checks if the model model_name is on the hub, and whether it (and its tokenizer) can be loaded with AutoClasses."""
40
  try:
41
+ config = AutoConfig.from_pretrained(
42
+ model_name, revision=revision, trust_remote_code=trust_remote_code, token=token
43
+ )
44
  if test_tokenizer:
45
  try:
46
+ tk = AutoTokenizer.from_pretrained(
47
+ model_name, revision=revision, trust_remote_code=trust_remote_code, token=token
48
+ )
49
  except ValueError as e:
50
+ return (False, f"uses a tokenizer which is not in a transformers release: {e}", None)
51
+ except Exception as e:
52
  return (
53
  False,
54
+ "'s tokenizer cannot be loaded. Is your tokenizer class in a stable transformers release, and correctly configured?",
55
+ None,
56
  )
 
 
57
  return True, None, config
58
 
59
  except ValueError:
60
  return (
61
  False,
62
  "needs to be launched with `trust_remote_code=True`. For safety reason, we do not allow these models to be automatically submitted to the leaderboard.",
63
+ None,
64
  )
65
 
66
  except Exception as e:
 
78
  model_size = size_factor * model_size
79
  return model_size
80
 
81
+
82
  def get_model_arch(model_info: ModelInfo):
83
  """Gets the model architecture from the configuration"""
84
  return model_info.config.get("architectures", "Unknown")
85
 
86
+
87
  def already_submitted_models(requested_models_dir: str) -> set[str]:
88
  """Gather a list of already submitted models to avoid duplicates"""
89
  depth = 1
src/submission/submit.py CHANGED
@@ -3,17 +3,13 @@ import os
3
  from datetime import datetime, timezone
4
 
5
  from src.display.formatting import styled_error, styled_message, styled_warning
6
- from src.envs import API, EVAL_REQUESTS_PATH, TOKEN, QUEUE_REPO
7
- from src.submission.check_validity import (
8
- already_submitted_models,
9
- check_model_card,
10
- get_model_size,
11
- is_model_on_hub,
12
- )
13
 
14
  REQUESTED_MODELS = None
15
  USERS_TO_SUBMISSION_DATES = None
16
 
 
17
  def add_new_eval(
18
  model: str,
19
  base_model: str,
@@ -45,7 +41,9 @@ def add_new_eval(
45
 
46
  # Is the model on the hub?
47
  if weight_type in ["Delta", "Adapter"]:
48
- base_model_on_hub, error, _ = is_model_on_hub(model_name=base_model, revision=revision, token=TOKEN, test_tokenizer=True)
 
 
49
  if not base_model_on_hub:
50
  return styled_error(f'Base model "{base_model}" {error}')
51
 
 
3
  from datetime import datetime, timezone
4
 
5
  from src.display.formatting import styled_error, styled_message, styled_warning
6
+ from src.envs import API, EVAL_REQUESTS_PATH, QUEUE_REPO, TOKEN
7
+ from src.submission.check_validity import already_submitted_models, check_model_card, get_model_size, is_model_on_hub
 
 
 
 
 
8
 
9
  REQUESTED_MODELS = None
10
  USERS_TO_SUBMISSION_DATES = None
11
 
12
+
13
  def add_new_eval(
14
  model: str,
15
  base_model: str,
 
41
 
42
  # Is the model on the hub?
43
  if weight_type in ["Delta", "Adapter"]:
44
+ base_model_on_hub, error, _ = is_model_on_hub(
45
+ model_name=base_model, revision=revision, token=TOKEN, test_tokenizer=True
46
+ )
47
  if not base_model_on_hub:
48
  return styled_error(f'Base model "{base_model}" {error}')
49