Spaces:
Running
Running
Only 22 models (incl. BackKiyo-10M); 100M+ models play each other freely
Browse files
app.py
CHANGED
|
@@ -30,28 +30,22 @@ RESULTS_REPO = os.environ.get("RESULTS_REPO", "DedeProGames/lm-tetris-arena-resu
|
|
| 30 |
MAX_MODELS = int(os.environ.get("MAX_MODELS", 4))
|
| 31 |
# The random ranked pick only groups models within this size gap, so big models don't farm Elo from tiny ones
|
| 32 |
MAX_PARAM_GAP = int(os.environ.get("MAX_PARAM_GAP", 20_000_000))
|
|
|
|
|
|
|
| 33 |
STORE = ResultsStore(RESULTS_REPO, _TOKEN, min_params=MIN_PARAMS)
|
| 34 |
|
| 35 |
SUGGESTED = [
|
| 36 |
-
'56m/Dumb-1.2-RC1',
|
| 37 |
-
'allura-org/Rambley-150M-RealBase',
|
| 38 |
-
'altslate/JugnuLM-110M-R2plus',
|
| 39 |
-
'appvoid/void.0',
|
| 40 |
-
'AtomixLabs/AtomixS2-5M-v1.0',
|
| 41 |
'AxiomicLabs/GPT-S-1.4M',
|
| 42 |
'AxiomicLabs/GPT-S2-5M',
|
| 43 |
'AxiomicLabs/GPT-X2.5-135M',
|
| 44 |
'BananaMind/BananaMind-2-Medium',
|
| 45 |
'BananaMind/BananaMind-2-Micro',
|
| 46 |
'BananaMind/BananaMind-2-Mini',
|
| 47 |
-
'BananaMind/BananaMind-2-MoE',
|
| 48 |
'BananaMind/BananaMind-2-Nano',
|
| 49 |
'BananaMind/BananaMind-2-Pro',
|
| 50 |
'BananaMind/BananaMind-2.1-Pico-Preview',
|
| 51 |
'BananaMind/BananaMind-2.1-Unified',
|
| 52 |
-
'
|
| 53 |
-
'CNWPlayer/VegaLM1-42M-Base',
|
| 54 |
-
'CodeSoft/sorbet-v2-25m',
|
| 55 |
'DedeProGames/DynamicMind-Mini',
|
| 56 |
'DedeProGames/DynamicMind-MoE',
|
| 57 |
'DedeProGames/Kiyo-230M-Preview',
|
|
@@ -59,58 +53,14 @@ SUGGESTED = [
|
|
| 59 |
'DedeProGames/LowOnMind-1M',
|
| 60 |
'DedeProGames/LowOnMind-300k',
|
| 61 |
'DedeProGames/LowOnMind-5M',
|
| 62 |
-
'DedeProGames/LowOnMind-8M',
|
| 63 |
-
'DedeProGames/NanoDex-1M',
|
| 64 |
-
'EleutherAI/pythia-160m',
|
| 65 |
-
'EleutherAI/pythia-70m',
|
| 66 |
-
'finnianx/Gros-Michel-90m-Base-v2',
|
| 67 |
-
'FlameF0X/TinyMoE-100m-2x8-retrained',
|
| 68 |
-
'fromziro/ZeroS-Qana-5M',
|
| 69 |
-
'fromziro/ZeroS-v0.1-150M',
|
| 70 |
-
'FWKV/Myosotis-1-base',
|
| 71 |
-
'GODELEV/Rose-1.5-Medium',
|
| 72 |
-
'GODELEV/Rose-Mini',
|
| 73 |
-
'Harley-ml/Dillionv2-1.3M',
|
| 74 |
-
'HuggingFaceTB/SmolLM2-135M',
|
| 75 |
-
'IvmeLabs/Ivme-Conversate-v3-Base',
|
| 76 |
-
'jhu-clsp/ettin-decoder-150m',
|
| 77 |
-
'jhu-clsp/ettin-decoder-17m',
|
| 78 |
-
'jhu-clsp/ettin-decoder-32m',
|
| 79 |
-
'jhu-clsp/ettin-decoder-68m',
|
| 80 |
-
'joelhenwang/OdinNext-138M-Base',
|
| 81 |
-
'LH-Tech-AI/Spark-5M-Base-v4',
|
| 82 |
-
'MaliosDark/Nexus-Erebus-135M',
|
| 83 |
-
'MaliosDark/Nexus-Erebus-3M',
|
| 84 |
-
'MaliosDark/Nexus-Erebus-50M',
|
| 85 |
-
'MihaiPopa-1/CinnabarLM-1.4M-Base',
|
| 86 |
-
'MinimaLabs/KeyLM-75M',
|
| 87 |
-
'MinimaLabs/min-spark-1.1',
|
| 88 |
'openai-community/gpt2',
|
| 89 |
-
'opencerebral/Boris-1.3-125M',
|
| 90 |
-
'opencerebral/Boris-1.3-75M',
|
| 91 |
-
'opencerebral/littlerock-1M',
|
| 92 |
-
'qikp/kite-7-15m-base',
|
| 93 |
-
'roneneldan/TinyStories-33M',
|
| 94 |
-
'SlayerLab/pollock-mini-lm-125m',
|
| 95 |
-
'StentorLabs/Stentor3-20M',
|
| 96 |
-
'StentorLabs/Stentor3-50M',
|
| 97 |
'SupraLabs/Supra2-100M-Base',
|
| 98 |
'SupraLabs/Supra2-Medium-Base',
|
| 99 |
-
'SupraLabs/SupraGDN-5M',
|
| 100 |
'SupraLabs/SupraNeo-4M',
|
| 101 |
-
'SurjoLabs/Ember-2',
|
| 102 |
-
'SurjoLabs/Flare',
|
| 103 |
-
'SurjoLabs/Surjo-50m',
|
| 104 |
-
'sz14/cRia-LM-75M',
|
| 105 |
-
'TobiasLogic/Museko-125M',
|
| 106 |
-
'UniversalComputingResearch/Atom2.7m',
|
| 107 |
-
'User01110/CMA-8M',
|
| 108 |
-
'veyra-ai/Veyra2-Blueberry-10M-Base',
|
| 109 |
-
'veyra-ai/Veyra2-Blueberry-5M-Base',
|
| 110 |
]
|
| 111 |
-
DEFAULT_MODELS = ["DedeProGames/Kiyo-65M", "BananaMind/BananaMind-2-Medium", "
|
| 112 |
# Ranked: the arena picks the players at random from the suggested models (nobody chooses who plays)
|
| 113 |
-
POOL = Pool(SUGGESTED, MAX_PARAM_GAP, MAX_MODELS)
|
| 114 |
RANKED = RankedRunner(POOL, STORE)
|
| 115 |
PROTOCOL_CHOICES = [("Guided: the rules are in the prompt", "guided"), ("Blind: no rules, only pre-training knowledge", "blind")]
|
| 116 |
BASELINE_CHOICES = [(label, key) for key, label in BASELINES.items()]
|
|
@@ -336,7 +286,8 @@ Because the value is a difference, a model's general bias towards "good" or "bad
|
|
| 336 |
- **Match** tab (friendly): pick any 2+ players (up to {MAX_MODELS} language models of any size, plus optional baselines)
|
| 337 |
and the seed. Nothing is recorded.
|
| 338 |
- **Ranked** tab: press Play and the arena picks up to {MAX_MODELS} models at random from its pool of {len(POOL.ids)} models,
|
| 339 |
-
all within {fmt_params(MAX_PARAM_GAP)} parameters of each other
|
|
|
|
| 340 |
Models with fewer ranked games are more likely to be picked, so every model gets played. The match runs on the server and counts even if
|
| 341 |
you close the page; only one ranked match runs at a time, and pressing Play while one is running lets you watch it.
|
| 342 |
- Elo: K=32, multiplayer (every pair of players counts as a game, scaled by 1/(N−1)).
|
|
@@ -358,7 +309,8 @@ Results and every match (seed, commit SHA of each model, scores) are published i
|
|
| 358 |
"""
|
| 359 |
|
| 360 |
RANKED_INTRO = f"""<p class="ah-note">Press <b>Play</b> and the arena picks up to {MAX_MODELS} language models <b>at random</b>
|
| 361 |
-
from its pool of {len(POOL.ids)} models, all within ±{short_params(MAX_PARAM_GAP)} parameters of each other
|
|
|
|
| 362 |
The result updates the public Elo leaderboard. The match runs on the server: it finishes and counts even if you close the page.
|
| 363 |
Only one ranked match runs at a time; if one is already running, you watch it.</p>"""
|
| 364 |
|
|
|
|
| 30 |
MAX_MODELS = int(os.environ.get("MAX_MODELS", 4))
|
| 31 |
# The random ranked pick only groups models within this size gap, so big models don't farm Elo from tiny ones
|
| 32 |
MAX_PARAM_GAP = int(os.environ.get("MAX_PARAM_GAP", 20_000_000))
|
| 33 |
+
# ...except big models: when every model has at least this many parameters, any of them can play each other
|
| 34 |
+
LARGE_FROM = int(os.environ.get("LARGE_FROM", 100_000_000))
|
| 35 |
STORE = ResultsStore(RESULTS_REPO, _TOKEN, min_params=MIN_PARAMS)
|
| 36 |
|
| 37 |
SUGGESTED = [
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 38 |
'AxiomicLabs/GPT-S-1.4M',
|
| 39 |
'AxiomicLabs/GPT-S2-5M',
|
| 40 |
'AxiomicLabs/GPT-X2.5-135M',
|
| 41 |
'BananaMind/BananaMind-2-Medium',
|
| 42 |
'BananaMind/BananaMind-2-Micro',
|
| 43 |
'BananaMind/BananaMind-2-Mini',
|
|
|
|
| 44 |
'BananaMind/BananaMind-2-Nano',
|
| 45 |
'BananaMind/BananaMind-2-Pro',
|
| 46 |
'BananaMind/BananaMind-2.1-Pico-Preview',
|
| 47 |
'BananaMind/BananaMind-2.1-Unified',
|
| 48 |
+
'DedeBckp/BackKiyo-10M',
|
|
|
|
|
|
|
| 49 |
'DedeProGames/DynamicMind-Mini',
|
| 50 |
'DedeProGames/DynamicMind-MoE',
|
| 51 |
'DedeProGames/Kiyo-230M-Preview',
|
|
|
|
| 53 |
'DedeProGames/LowOnMind-1M',
|
| 54 |
'DedeProGames/LowOnMind-300k',
|
| 55 |
'DedeProGames/LowOnMind-5M',
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
'openai-community/gpt2',
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 57 |
'SupraLabs/Supra2-100M-Base',
|
| 58 |
'SupraLabs/Supra2-Medium-Base',
|
|
|
|
| 59 |
'SupraLabs/SupraNeo-4M',
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 60 |
]
|
| 61 |
+
DEFAULT_MODELS = ["DedeProGames/Kiyo-65M", "BananaMind/BananaMind-2-Medium", "DedeProGames/DynamicMind-MoE", "SupraLabs/Supra2-Medium-Base"]
|
| 62 |
# Ranked: the arena picks the players at random from the suggested models (nobody chooses who plays)
|
| 63 |
+
POOL = Pool(SUGGESTED, MAX_PARAM_GAP, MAX_MODELS, LARGE_FROM)
|
| 64 |
RANKED = RankedRunner(POOL, STORE)
|
| 65 |
PROTOCOL_CHOICES = [("Guided: the rules are in the prompt", "guided"), ("Blind: no rules, only pre-training knowledge", "blind")]
|
| 66 |
BASELINE_CHOICES = [(label, key) for key, label in BASELINES.items()]
|
|
|
|
| 286 |
- **Match** tab (friendly): pick any 2+ players (up to {MAX_MODELS} language models of any size, plus optional baselines)
|
| 287 |
and the seed. Nothing is recorded.
|
| 288 |
- **Ranked** tab: press Play and the arena picks up to {MAX_MODELS} models at random from its pool of {len(POOL.ids)} models,
|
| 289 |
+
all within {fmt_params(MAX_PARAM_GAP)} parameters of each other (models with {short_params(LARGE_FROM)}+ parameters can all play each other),
|
| 290 |
+
with a random seed. Nobody chooses who plays, so Elo can't be farmed.
|
| 291 |
Models with fewer ranked games are more likely to be picked, so every model gets played. The match runs on the server and counts even if
|
| 292 |
you close the page; only one ranked match runs at a time, and pressing Play while one is running lets you watch it.
|
| 293 |
- Elo: K=32, multiplayer (every pair of players counts as a game, scaled by 1/(N−1)).
|
|
|
|
| 309 |
"""
|
| 310 |
|
| 311 |
RANKED_INTRO = f"""<p class="ah-note">Press <b>Play</b> and the arena picks up to {MAX_MODELS} language models <b>at random</b>
|
| 312 |
+
from its pool of {len(POOL.ids)} models, all within ±{short_params(MAX_PARAM_GAP)} parameters of each other
|
| 313 |
+
(models with {short_params(LARGE_FROM)}+ parameters can all play each other), with a random seed.
|
| 314 |
The result updates the public Elo leaderboard. The match runs on the server: it finishes and counts even if you close the page.
|
| 315 |
Only one ranked match runs at a time; if one is already running, you watch it.</p>"""
|
| 316 |
|