DedeProGames commited on
Commit
80dbbce
·
verified ·
1 Parent(s): 5e01700

Only 22 models (incl. BackKiyo-10M); 100M+ models play each other freely

Browse files
Files changed (1) hide show
  1. app.py +9 -57
app.py CHANGED
@@ -30,28 +30,22 @@ RESULTS_REPO = os.environ.get("RESULTS_REPO", "DedeProGames/lm-tetris-arena-resu
30
  MAX_MODELS = int(os.environ.get("MAX_MODELS", 4))
31
  # The random ranked pick only groups models within this size gap, so big models don't farm Elo from tiny ones
32
  MAX_PARAM_GAP = int(os.environ.get("MAX_PARAM_GAP", 20_000_000))
 
 
33
  STORE = ResultsStore(RESULTS_REPO, _TOKEN, min_params=MIN_PARAMS)
34
 
35
  SUGGESTED = [
36
- '56m/Dumb-1.2-RC1',
37
- 'allura-org/Rambley-150M-RealBase',
38
- 'altslate/JugnuLM-110M-R2plus',
39
- 'appvoid/void.0',
40
- 'AtomixLabs/AtomixS2-5M-v1.0',
41
  'AxiomicLabs/GPT-S-1.4M',
42
  'AxiomicLabs/GPT-S2-5M',
43
  'AxiomicLabs/GPT-X2.5-135M',
44
  'BananaMind/BananaMind-2-Medium',
45
  'BananaMind/BananaMind-2-Micro',
46
  'BananaMind/BananaMind-2-Mini',
47
- 'BananaMind/BananaMind-2-MoE',
48
  'BananaMind/BananaMind-2-Nano',
49
  'BananaMind/BananaMind-2-Pro',
50
  'BananaMind/BananaMind-2.1-Pico-Preview',
51
  'BananaMind/BananaMind-2.1-Unified',
52
- 'bench-labs/cagliostro-v3',
53
- 'CNWPlayer/VegaLM1-42M-Base',
54
- 'CodeSoft/sorbet-v2-25m',
55
  'DedeProGames/DynamicMind-Mini',
56
  'DedeProGames/DynamicMind-MoE',
57
  'DedeProGames/Kiyo-230M-Preview',
@@ -59,58 +53,14 @@ SUGGESTED = [
59
  'DedeProGames/LowOnMind-1M',
60
  'DedeProGames/LowOnMind-300k',
61
  'DedeProGames/LowOnMind-5M',
62
- 'DedeProGames/LowOnMind-8M',
63
- 'DedeProGames/NanoDex-1M',
64
- 'EleutherAI/pythia-160m',
65
- 'EleutherAI/pythia-70m',
66
- 'finnianx/Gros-Michel-90m-Base-v2',
67
- 'FlameF0X/TinyMoE-100m-2x8-retrained',
68
- 'fromziro/ZeroS-Qana-5M',
69
- 'fromziro/ZeroS-v0.1-150M',
70
- 'FWKV/Myosotis-1-base',
71
- 'GODELEV/Rose-1.5-Medium',
72
- 'GODELEV/Rose-Mini',
73
- 'Harley-ml/Dillionv2-1.3M',
74
- 'HuggingFaceTB/SmolLM2-135M',
75
- 'IvmeLabs/Ivme-Conversate-v3-Base',
76
- 'jhu-clsp/ettin-decoder-150m',
77
- 'jhu-clsp/ettin-decoder-17m',
78
- 'jhu-clsp/ettin-decoder-32m',
79
- 'jhu-clsp/ettin-decoder-68m',
80
- 'joelhenwang/OdinNext-138M-Base',
81
- 'LH-Tech-AI/Spark-5M-Base-v4',
82
- 'MaliosDark/Nexus-Erebus-135M',
83
- 'MaliosDark/Nexus-Erebus-3M',
84
- 'MaliosDark/Nexus-Erebus-50M',
85
- 'MihaiPopa-1/CinnabarLM-1.4M-Base',
86
- 'MinimaLabs/KeyLM-75M',
87
- 'MinimaLabs/min-spark-1.1',
88
  'openai-community/gpt2',
89
- 'opencerebral/Boris-1.3-125M',
90
- 'opencerebral/Boris-1.3-75M',
91
- 'opencerebral/littlerock-1M',
92
- 'qikp/kite-7-15m-base',
93
- 'roneneldan/TinyStories-33M',
94
- 'SlayerLab/pollock-mini-lm-125m',
95
- 'StentorLabs/Stentor3-20M',
96
- 'StentorLabs/Stentor3-50M',
97
  'SupraLabs/Supra2-100M-Base',
98
  'SupraLabs/Supra2-Medium-Base',
99
- 'SupraLabs/SupraGDN-5M',
100
  'SupraLabs/SupraNeo-4M',
101
- 'SurjoLabs/Ember-2',
102
- 'SurjoLabs/Flare',
103
- 'SurjoLabs/Surjo-50m',
104
- 'sz14/cRia-LM-75M',
105
- 'TobiasLogic/Museko-125M',
106
- 'UniversalComputingResearch/Atom2.7m',
107
- 'User01110/CMA-8M',
108
- 'veyra-ai/Veyra2-Blueberry-10M-Base',
109
- 'veyra-ai/Veyra2-Blueberry-5M-Base',
110
  ]
111
- DEFAULT_MODELS = ["DedeProGames/Kiyo-65M", "BananaMind/BananaMind-2-Medium", "SurjoLabs/Surjo-50m", "StentorLabs/Stentor3-50M"]
112
  # Ranked: the arena picks the players at random from the suggested models (nobody chooses who plays)
113
- POOL = Pool(SUGGESTED, MAX_PARAM_GAP, MAX_MODELS)
114
  RANKED = RankedRunner(POOL, STORE)
115
  PROTOCOL_CHOICES = [("Guided: the rules are in the prompt", "guided"), ("Blind: no rules, only pre-training knowledge", "blind")]
116
  BASELINE_CHOICES = [(label, key) for key, label in BASELINES.items()]
@@ -336,7 +286,8 @@ Because the value is a difference, a model's general bias towards "good" or "bad
336
  - **Match** tab (friendly): pick any 2+ players (up to {MAX_MODELS} language models of any size, plus optional baselines)
337
  and the seed. Nothing is recorded.
338
  - **Ranked** tab: press Play and the arena picks up to {MAX_MODELS} models at random from its pool of {len(POOL.ids)} models,
339
- all within {fmt_params(MAX_PARAM_GAP)} parameters of each other, with a random seed. Nobody chooses who plays, so Elo can't be farmed.
 
340
  Models with fewer ranked games are more likely to be picked, so every model gets played. The match runs on the server and counts even if
341
  you close the page; only one ranked match runs at a time, and pressing Play while one is running lets you watch it.
342
  - Elo: K=32, multiplayer (every pair of players counts as a game, scaled by 1/(N−1)).
@@ -358,7 +309,8 @@ Results and every match (seed, commit SHA of each model, scores) are published i
358
  """
359
 
360
  RANKED_INTRO = f"""<p class="ah-note">Press <b>Play</b> and the arena picks up to {MAX_MODELS} language models <b>at random</b>
361
- from its pool of {len(POOL.ids)} models, all within ±{short_params(MAX_PARAM_GAP)} parameters of each other, with a random seed.
 
362
  The result updates the public Elo leaderboard. The match runs on the server: it finishes and counts even if you close the page.
363
  Only one ranked match runs at a time; if one is already running, you watch it.</p>"""
364
 
 
30
  MAX_MODELS = int(os.environ.get("MAX_MODELS", 4))
31
  # The random ranked pick only groups models within this size gap, so big models don't farm Elo from tiny ones
32
  MAX_PARAM_GAP = int(os.environ.get("MAX_PARAM_GAP", 20_000_000))
33
+ # ...except big models: when every model has at least this many parameters, any of them can play each other
34
+ LARGE_FROM = int(os.environ.get("LARGE_FROM", 100_000_000))
35
  STORE = ResultsStore(RESULTS_REPO, _TOKEN, min_params=MIN_PARAMS)
36
 
37
  SUGGESTED = [
 
 
 
 
 
38
  'AxiomicLabs/GPT-S-1.4M',
39
  'AxiomicLabs/GPT-S2-5M',
40
  'AxiomicLabs/GPT-X2.5-135M',
41
  'BananaMind/BananaMind-2-Medium',
42
  'BananaMind/BananaMind-2-Micro',
43
  'BananaMind/BananaMind-2-Mini',
 
44
  'BananaMind/BananaMind-2-Nano',
45
  'BananaMind/BananaMind-2-Pro',
46
  'BananaMind/BananaMind-2.1-Pico-Preview',
47
  'BananaMind/BananaMind-2.1-Unified',
48
+ 'DedeBckp/BackKiyo-10M',
 
 
49
  'DedeProGames/DynamicMind-Mini',
50
  'DedeProGames/DynamicMind-MoE',
51
  'DedeProGames/Kiyo-230M-Preview',
 
53
  'DedeProGames/LowOnMind-1M',
54
  'DedeProGames/LowOnMind-300k',
55
  'DedeProGames/LowOnMind-5M',
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
56
  'openai-community/gpt2',
 
 
 
 
 
 
 
 
57
  'SupraLabs/Supra2-100M-Base',
58
  'SupraLabs/Supra2-Medium-Base',
 
59
  'SupraLabs/SupraNeo-4M',
 
 
 
 
 
 
 
 
 
60
  ]
61
+ DEFAULT_MODELS = ["DedeProGames/Kiyo-65M", "BananaMind/BananaMind-2-Medium", "DedeProGames/DynamicMind-MoE", "SupraLabs/Supra2-Medium-Base"]
62
  # Ranked: the arena picks the players at random from the suggested models (nobody chooses who plays)
63
+ POOL = Pool(SUGGESTED, MAX_PARAM_GAP, MAX_MODELS, LARGE_FROM)
64
  RANKED = RankedRunner(POOL, STORE)
65
  PROTOCOL_CHOICES = [("Guided: the rules are in the prompt", "guided"), ("Blind: no rules, only pre-training knowledge", "blind")]
66
  BASELINE_CHOICES = [(label, key) for key, label in BASELINES.items()]
 
286
  - **Match** tab (friendly): pick any 2+ players (up to {MAX_MODELS} language models of any size, plus optional baselines)
287
  and the seed. Nothing is recorded.
288
  - **Ranked** tab: press Play and the arena picks up to {MAX_MODELS} models at random from its pool of {len(POOL.ids)} models,
289
+ all within {fmt_params(MAX_PARAM_GAP)} parameters of each other (models with {short_params(LARGE_FROM)}+ parameters can all play each other),
290
+ with a random seed. Nobody chooses who plays, so Elo can't be farmed.
291
  Models with fewer ranked games are more likely to be picked, so every model gets played. The match runs on the server and counts even if
292
  you close the page; only one ranked match runs at a time, and pressing Play while one is running lets you watch it.
293
  - Elo: K=32, multiplayer (every pair of players counts as a game, scaled by 1/(N−1)).
 
309
  """
310
 
311
  RANKED_INTRO = f"""<p class="ah-note">Press <b>Play</b> and the arena picks up to {MAX_MODELS} language models <b>at random</b>
312
+ from its pool of {len(POOL.ids)} models, all within ±{short_params(MAX_PARAM_GAP)} parameters of each other
313
+ (models with {short_params(LARGE_FROM)}+ parameters can all play each other), with a random seed.
314
  The result updates the public Elo leaderboard. The match runs on the server: it finishes and counts even if you close the page.
315
  Only one ranked match runs at a time; if one is already running, you watch it.</p>"""
316