Spaces:
Running
Running
File size: 21,066 Bytes
57a3d2c 348e4ed 8b961c3 348e4ed 8b961c3 62eb2bc 348e4ed 3e2ab3d 348e4ed 3e2ab3d 57a52a8 80dbbce 62eb2bc 348e4ed 3c65707 80dbbce 3c65707 348e4ed 80dbbce 3e2ab3d 80dbbce 3e2ab3d 348e4ed 8b961c3 3e2ab3d 8b961c3 348e4ed 3e2ab3d 348e4ed 3e2ab3d 348e4ed 3e2ab3d 91db64f 348e4ed 3e2ab3d 348e4ed 91db64f 348e4ed 3e2ab3d b5084d2 8b961c3 348e4ed 8b961c3 348e4ed 8b961c3 57a3d2c 62eb2bc 8b961c3 348e4ed 3e2ab3d 80dbbce 3e2ab3d 348e4ed 3e2ab3d 57a52a8 348e4ed 62eb2bc 348e4ed 3e2ab3d 80dbbce 3e2ab3d 57a3d2c 8b961c3 348e4ed 8b961c3 348e4ed 57a52a8 348e4ed 3e2ab3d 348e4ed 8b961c3 348e4ed 3e2ab3d 348e4ed 8b961c3 348e4ed 3e2ab3d 348e4ed b5084d2 3e2ab3d 8b961c3 348e4ed 8b961c3 348e4ed 8b961c3 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 | """SLM Tetris Arena — decoder-only LMs play Tetris zero-shot and earn Elo."""
import os
# Keep the write token out of the environment before any model code runs:
# only the results store receives it (custom model code runs in this process).
_TOKEN = os.environ.pop("HF_TOKEN", None) or os.environ.pop("HUGGING_FACE_HUB_TOKEN", None)
os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1")
import html
import inspect
import random
import time
import gradio as gr
import torch
from arena import MAX_PIECES, SEASON, ResultsStore, choose, rank_games
from leaderboard import LB_CSS, leaderboard_html
from leaderboard import fmt_params as short_params
from players import (BASELINES, MAX_PARAMS, MIN_PARAMS, ORACLE_ID, PROMPTS, RANDOM_ID, ModelRejected, OracleReaderPlayer,
RandomPlayer, fmt_params, load_player, precheck)
from ranked import Pool, RankedRunner
from render import CSS, arena_html, empty_html, results_html
from tetris import TetrisGame
# cpu-basic Spaces have 2 vCPUs; os.cpu_count() reports the host, which oversubscribes threads
torch.set_num_threads(int(os.environ.get("TORCH_THREADS", 2)))
RESULTS_REPO = os.environ.get("RESULTS_REPO", "DedeProGames/lm-tetris-arena-results")
MAX_MODELS = int(os.environ.get("MAX_MODELS", 4))
# The random ranked pick only groups models within this size gap, so big models don't farm Elo from tiny ones
MAX_PARAM_GAP = int(os.environ.get("MAX_PARAM_GAP", 20_000_000))
# ...except big models: when every model has at least this many parameters, any of them can play each other
LARGE_FROM = int(os.environ.get("LARGE_FROM", 100_000_000))
STORE = ResultsStore(RESULTS_REPO, _TOKEN, min_params=MIN_PARAMS)
SUGGESTED = [
'AxiomicLabs/GPT-S-1.4M',
'AxiomicLabs/GPT-S2-5M',
'AxiomicLabs/GPT-X2.5-135M',
'BananaMind/BananaMind-2-Medium',
'BananaMind/BananaMind-2-Micro',
'BananaMind/BananaMind-2-Mini',
'BananaMind/BananaMind-2-Nano',
'BananaMind/BananaMind-2-Pro',
'BananaMind/BananaMind-2.1-Pico-Preview',
'BananaMind/BananaMind-2.1-Unified',
'DedeBckp/BackKiyo-10M',
'DedeProGames/DynamicMind-Mini',
'DedeProGames/DynamicMind-MoE',
'DedeProGames/Kiyo-230M-Preview',
'DedeProGames/Kiyo-65M',
'DedeProGames/LowOnMind-1M',
'DedeProGames/LowOnMind-300k',
'DedeProGames/LowOnMind-5M',
'openai-community/gpt2',
'SupraLabs/Supra2-100M-Base',
'SupraLabs/Supra2-Medium-Base',
'SupraLabs/SupraNeo-4M',
]
DEFAULT_MODELS = ["DedeProGames/Kiyo-65M", "BananaMind/BananaMind-2-Medium", "DedeProGames/DynamicMind-MoE", "SupraLabs/Supra2-Medium-Base"]
# Ranked: the arena picks the players at random from the suggested models (nobody chooses who plays)
POOL = Pool(SUGGESTED, MAX_PARAM_GAP, MAX_MODELS, LARGE_FROM)
RANKED = RankedRunner(POOL, STORE)
PROTOCOL_CHOICES = [("Guided: the rules are in the prompt", "guided"), ("Blind: no rules, only pre-training knowledge", "blind")]
BASELINE_CHOICES = [(label, key) for key, label in BASELINES.items()]
# ---- Look & feel: same palette and type as the BananaMind SLM Leaderboard (dark + light) ----
_D = dict(bg="#0b0e0d", surface="#111613", surface2="#191e1b", text="#f0f1ec", muted="#929b93", line="#29312c", accent="#facc15")
_L = dict(bg="#f4f5f1", surface="#ffffff", surface2="#edf0e9", text="#17221b", muted="#626e64", line="#d7ded5", accent="#8c6500")
_CHECK = ("url(\"data:image/svg+xml,%3csvg viewBox='0 0 16 16' fill='%231b200e' xmlns='http://www.w3.org/2000/svg'%3e"
"%3cpath d='M12.207 4.793a1 1 0 010 1.414l-5 5a1 1 0 01-1.414 0l-2-2a1 1 0 011.414-1.414L6.5 9.086l4.293-4.293a1 1 0 011.414 0z'/%3e%3c/svg%3e\")")
_THEME_KEYS = set(inspect.signature(gr.themes.Base.set).parameters)
def _both(**pairs):
"""name=(light, dark) -> theme kwargs for both modes (skips variables this Gradio version lacks)."""
out = {}
for k, (light, dark) in pairs.items():
if k in _THEME_KEYS:
out[k] = light
if k + "_dark" in _THEME_KEYS:
out[k + "_dark"] = dark
return out
THEME = gr.themes.Base(
primary_hue=gr.themes.colors.yellow, secondary_hue=gr.themes.colors.yellow, neutral_hue=gr.themes.colors.stone,
font=[gr.themes.GoogleFont("DM Sans"), "Arial", "sans-serif"], font_mono=["ui-monospace", "SFMono-Regular", "monospace"],
).set(
block_border_width="1px", block_radius="13px", block_label_border_width="0px", block_title_text_weight="500",
input_radius="8px", checkbox_check=_CHECK,
**_both(
body_background_fill=(_L["bg"], _D["bg"]), body_text_color=(_L["text"], _D["text"]),
body_text_color_subdued=(_L["muted"], _D["muted"]),
background_fill_primary=(_L["surface"], _D["surface"]), background_fill_secondary=(_L["surface2"], _D["surface2"]),
block_background_fill=(_L["surface"], _D["surface"]), block_border_color=(_L["line"], _D["line"]),
block_shadow=("none", "none"), block_label_background_fill=("transparent", "transparent"),
block_label_text_color=(_L["muted"], _D["muted"]), block_label_shadow=("none", "none"),
block_title_background_fill=("transparent", "transparent"), block_title_text_color=(_L["muted"], _D["muted"]),
block_info_text_color=(_L["muted"], _D["muted"]),
border_color_primary=(_L["line"], _D["line"]), border_color_accent=("#facc15", "#facc15"),
color_accent=("#facc15", "#facc15"), color_accent_soft=("#facc1526", "#facc1526"),
input_background_fill=(_L["surface2"], _D["surface2"]), input_border_color=(_L["line"], _D["line"]),
input_border_color_focus=("#facc15", "#facc15"),
button_primary_background_fill=("#facc15", "#facc15"), button_primary_background_fill_hover=("#fde047", "#fde047"),
button_primary_text_color=("#1b200e", "#1b200e"), button_primary_border_color=("#facc15", "#facc15"),
button_secondary_background_fill=(_L["surface2"], _D["surface2"]),
button_secondary_background_fill_hover=(_L["line"], _D["line"]),
button_secondary_text_color=(_L["text"], _D["text"]), button_secondary_border_color=(_L["line"], _D["line"]),
checkbox_background_color_selected=("#facc15", "#facc15"), checkbox_border_color_selected=("#facc15", "#facc15"),
checkbox_label_background_fill=(_L["surface2"], _D["surface2"]),
checkbox_label_background_fill_selected=(_L["surface2"], _D["surface2"]),
checkbox_label_border_color=(_L["line"], _D["line"]), checkbox_label_border_color_selected=("#facc15", "#facc15"),
checkbox_label_text_color_selected=(_L["text"], _D["text"]),
slider_color=("#facc15", "#facc15"), loader_color=("#facc15", "#facc15"),
link_text_color=(_L["accent"], _D["accent"]), link_text_color_hover=(_L["accent"], _D["accent"]),
panel_background_fill=(_L["surface"], _D["surface"]), panel_border_color=(_L["line"], _D["line"]),
table_border_color=(_L["line"], _D["line"]), code_background_fill=(_L["surface2"], _D["surface2"]),
),
)
APP_CSS = """
@import url('https://fonts.googleapis.com/css2?family=DM+Sans:wght@400;500;600;700&family=Space+Grotesk:wght@400;500;600;700&display=swap');
.gradio-container{font-family:'DM Sans',Arial,sans-serif!important}
.gradio-container h1,.gradio-container h2,.gradio-container h3{font-family:'Space Grotesk',Arial,sans-serif!important;letter-spacing:-.4px}
.ah-heading{display:flex;justify-content:space-between;align-items:center;gap:20px;padding:10px 0 6px}
.ah-eyebrow{font:11px/1.5 monospace!important;letter-spacing:1.9px;color:var(--body-text-color-subdued)!important;margin:0 0 10px!important}
.ah-heading h1{font:500 clamp(32px,4vw,48px)/1.2 'Space Grotesk',Arial,sans-serif!important;letter-spacing:-2px!important;margin:0!important;
color:var(--body-text-color)!important}
.ah-accent{color:#8c6500}.dark .ah-accent{color:#facc15}
.ah-intro{margin:10px 0 0!important;font-size:15px!important;color:var(--body-text-color-subdued)!important;max-width:860px}
.ah-version{font:11px monospace;letter-spacing:1px;color:var(--body-text-color-subdued);display:flex;align-items:center;gap:10px;white-space:nowrap}
.ah-dot{height:6px;width:6px;background:#95c79a;border-radius:50%}
.ah-note{margin:0 0 4px!important;font-size:14px!important;line-height:1.55;color:var(--body-text-color-subdued)!important;max-width:900px}
.ah-note b{color:var(--body-text-color)}
.ah-warn{margin-top:14px;padding:10px 14px;border:1px solid #facc1566;border-radius:8px;font-size:13px;color:var(--body-text-color)}
button[role=tab]{font-size:14px!important;color:var(--body-text-color-subdued)!important}
button[role=tab][aria-selected=true]{color:var(--body-text-color)!important;border-color:#facc15!important}
@media(max-width:550px){.ah-version{display:none}}
"""
def _status(text, kind="info"):
icon = {"info": "⏳", "ok": "✅", "err": "⛔", "warn": "⚠️"}[kind]
return f"{icon} {text}"
def run_match(model_ids, baselines, protocol, seed, delay):
"""Friendly match: any models, any seed, nothing is recorded (Elo only changes in the Ranked tab)."""
model_ids = [m.strip() for m in (model_ids or []) if m and m.strip()]
model_ids = list(dict.fromkeys(model_ids))
baselines = baselines or []
protocol = protocol or "guided"
if not model_ids:
yield _status("Pick at least one language model.", "err"), empty_html(), ""
return
if len(model_ids) > MAX_MODELS:
yield _status(f"At most {MAX_MODELS} language models per match on this CPU.", "err"), empty_html(), ""
return
if len(model_ids) + len(baselines) < 2:
yield _status("A match needs at least 2 players: add another model or a baseline.", "err"), empty_html(), ""
return
players = []
try:
metas = []
for m in model_ids:
yield _status(f"Checking `{m}`…"), empty_html("Checking models…"), ""
meta = precheck(m)
if any(x["id"] == meta["id"] for x in metas):
continue # same repo typed twice with different casing
metas.append(meta)
model_ids = [x["id"] for x in metas]
for i, (m, meta) in enumerate(zip(model_ids, metas), 1):
yield _status(f"Loading `{m}` on CPU ({i}/{len(model_ids)})… first load downloads the weights."), empty_html("Loading models…"), ""
players.append(load_player(m, meta))
except ModelRejected as e:
yield _status(str(e), "err"), empty_html("Match cancelled."), ""
return
if RANDOM_ID in baselines:
players.append(RandomPlayer())
if ORACLE_ID in baselines:
players.append(OracleReaderPlayer())
seed = int(seed) if seed else random.SystemRandom().randrange(1, 10**9)
games = [TetrisGame(seed) for _ in players]
mode = "friendly"
yield _status(f"Seed {seed} · {protocol} · {mode}. Scoring the first moves…"), arena_html(games, players), ""
last = time.time()
try:
while True:
active = [(g, p) for g, p in zip(games, players) if g.alive and g.pieces < MAX_PIECES]
if not active:
break
for g, p in active:
choose(g, p, protocol, seed)
elapsed = time.time() - last
if elapsed < delay:
time.sleep(delay - elapsed)
last = time.time()
n = max(g.pieces for g in games)
alive = sum(g.alive for g in games)
yield _status(f"Seed {seed} · {protocol} · {mode} · piece {n}/{MAX_PIECES} · {alive} still playing"), arena_html(games, players), ""
except ModelRejected as e:
yield _status(str(e), "err"), arena_html(games, players), ""
return
except Exception as e:
yield _status(f"A model crashed during play: {type(e).__name__}: {str(e)[:200]}", "err"), arena_html(games, players), ""
return
ranks = rank_games(games)
order = sorted(range(len(players)), key=lambda i: ranks[i])
elos = None
note = "Friendly match: Elo not changed (only matches in the Ranked tab count)."
note += f" Ranking: score, then lines, then pieces survived. ✓ = still alive at the {MAX_PIECES}-piece cap."
results = results_html(order, ranks, players, games, elos, note)
yield _status(f"Match finished · seed {seed} · {protocol} · {mode}.", "ok"), arena_html(games, players, ranks, elos), results
def ranked_play(protocol):
"""Start a ranked match (models picked at random) or watch the one already running."""
match, started = RANKED.start_or_join(protocol or "guided")
joined = "" if started else f" · you joined the ranked match already in progress ({match.protocol})"
seen = -1
while True:
version, status, boards, results, done = match.snapshot()
if version != seen:
seen = version
yield status + joined, boards, results
if done:
return
time.sleep(0.1)
def stop_status(current):
# only claim a stop when a match was actually running
if (current or "").startswith("⏳"):
return _status("Match stopped. Nothing was recorded.", "warn")
return current
def leaderboard_view(protocol):
protocol = protocol or "guided"
entries = [e for e in STORE.rows(protocol) if e["model"] not in BASELINES]
return leaderboard_html(entries, protocol, MAX_PARAM_GAP, short_params(MAX_PARAM_GAP))
def refresh_leaderboard(protocol):
STORE.reload()
return leaderboard_view(protocol)
INTRO = f"""
<div class="ah-heading"><div><p class="ah-eyebrow">SMALL MODELS. ZERO-SHOT TETRIS.</p>
<h1>SLM Tetris Arena<span class="ah-accent">.</span></h1>
<p class="ah-intro">Decoder-only language models ({short_params(MIN_PARAMS)}–{short_params(MAX_PARAMS)} parameters, custom architectures welcome) play Tetris
zero-shot: no fine-tuning, no game data, only what they learned from pre-training on text. Every player gets the same
piece sequence, and ranked matches update a public Elo leaderboard.</p></div>
<span class="ah-version">SEASON {SEASON} <span class="ah-dot"></span></span></div>
"""
HOW = f"""
### How a model plays
For every new piece the game lists all legal placements (rotation × column, hard drop), simulates each one and
describes the outcome in plain English. The model never sees the grid; it judges the descriptions:
```
{PROMPTS['guided'].format(desc='drops the piece into the lowest part of the board, clears one line, creates no new holes, keeps the stack low and leaves the surface flat')}
```
The model's value for a placement is **log P(" good move") − log P(" bad move")** after that prompt. The placement with the
highest value is played; exact ties are broken by a seeded coin that is identical for every player.
Because the value is a difference, a model's general bias towards "good" or "bad" cancels out.
### Protocols (separate leaderboards)
- **Guided**: the first line states the goal ("clear lines, avoid holes, keep the stack low"). Tests reading comprehension.
- **Blind**: `{PROMPTS['blind'].splitlines()[0]}` No rules; the model must already know what is good in Tetris.
### Rules of a match
- Same 7-bag piece sequence for everyone. The game ends at top-out or after {MAX_PIECES} pieces.
- Placement = score (100/300/500/800 for 1/2/3/4 lines), then lines, then pieces survived.
- **Match** tab (friendly): pick any 2+ players (up to {MAX_MODELS} language models of any size, plus optional baselines)
and the seed. Nothing is recorded.
- **Ranked** tab: press Play and the arena picks up to {MAX_MODELS} models at random from its pool of {len(POOL.ids)} models,
all within {fmt_params(MAX_PARAM_GAP)} parameters of each other (models with {short_params(LARGE_FROM)}+ parameters can all play each other),
with a random seed. Nobody chooses who plays, so Elo can't be farmed.
Models with fewer ranked games are more likely to be picked, so every model gets played. The match runs on the server and counts even if
you close the page; only one ranked match runs at a time, and pressing Play while one is running lets you watch it.
- Elo: K=32, multiplayer (every pair of players counts as a game, scaled by 1/(N−1)).
### Baselines
- **🎲 Random**: every placement ties, so it plays uniformly at random. This is the floor a model should beat.
- **📏 Oracle reader**: reads the same descriptions and ranks them with fixed common sense (holes > lines > height > surface > landing).
This is roughly the ceiling for a perfect reader of the text.
Baselines can join friendly matches for comparison. They never play ranked and never change anyone's Elo.
### Model requirements
Public, not gated, loads with `AutoModelForCausalLM` + `AutoTokenizer` (PyTorch or safetensors weights), between {fmt_params(MIN_PARAMS)} and {fmt_params(MAX_PARAMS)} parameters.
Models with custom code (`auto_map`) load with `trust_remote_code=True`. That code runs on this Space's CPU, so only
submit repos you trust. Prompts are in English (the language most pre-training corpora such as FineWeb-edu use).
Results and every match (seed, commit SHA of each model, scores) are published in
[`{RESULTS_REPO}`](https://huggingface.co/datasets/{RESULTS_REPO}).
"""
RANKED_INTRO = f"""<p class="ah-note">Press <b>Play</b> and the arena picks up to {MAX_MODELS} language models <b>at random</b>
from its pool of {len(POOL.ids)} models, all within ±{short_params(MAX_PARAM_GAP)} parameters of each other
(models with {short_params(LARGE_FROM)}+ parameters can all play each other), with a random seed.
The result updates the public Elo leaderboard. The match runs on the server: it finishes and counts even if you close the page.
Only one ranked match runs at a time; if one is already running, you watch it.</p>"""
with gr.Blocks(title="SLM Tetris Arena") as demo:
gr.HTML(INTRO, padding=False)
if not STORE.persistent:
gr.HTML('<div class="ah-warn">⚠️ Results are not being saved: add an <code>HF_TOKEN</code> secret with write access '
'to the results dataset.</div>', padding=False)
with gr.Tabs():
with gr.Tab("⚔️ Match"):
with gr.Row():
with gr.Column(scale=3):
models = gr.Dropdown(
choices=SUGGESTED, value=DEFAULT_MODELS, multiselect=True, allow_custom_value=True,
max_choices=MAX_MODELS, label=f"Language models (1–{MAX_MODELS})",
info="Pick from the list or type any Hub repo id (owner/name) and press Enter.",
)
baselines = gr.CheckboxGroup(BASELINE_CHOICES, value=[RANDOM_ID], label="Baselines (optional, never rated)")
with gr.Column(scale=2):
protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol")
seed = gr.Number(value=42, precision=0, label="Seed (0 = random)",
info="Friendly match: any models, nothing is recorded. Elo only changes in the Ranked tab.")
delay = gr.Slider(0, 0.5, value=0.12, step=0.02, label="Seconds per piece (viewing speed)")
with gr.Row():
start = gr.Button("▶ Start match", variant="primary")
stop = gr.Button("■ Stop", variant="secondary")
status = gr.Markdown(_status("Ready.", "ok"))
boards = gr.HTML(empty_html())
results = gr.HTML()
with gr.Tab("🏅 Ranked"):
gr.HTML(RANKED_INTRO, padding=False)
with gr.Row(equal_height=True):
r_protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol", scale=3)
r_play = gr.Button("▶ Play ranked match", variant="primary", scale=1)
r_status = gr.Markdown(_status("Ready. Press Play: the arena picks the models.", "ok"))
r_boards = gr.HTML(empty_html("Press <b>Play ranked match</b>. The arena picks the models at random."))
r_results = gr.HTML()
with gr.Tab("🏆 Leaderboard"):
lb_protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol")
lb = gr.HTML(leaderboard_view("guided"), padding=False)
lb_refresh = gr.Button("↻ Refresh", variant="secondary")
with gr.Tab("📖 How it works"):
gr.Markdown(HOW)
match_event = start.click(
run_match, [models, baselines, protocol, seed, delay], [status, boards, results], concurrency_limit=1,
)
stop.click(stop_status, status, status, cancels=[match_event])
# viewers only watch; the ranked match itself runs in one background thread
r_play.click(ranked_play, r_protocol, [r_status, r_boards, r_results], concurrency_limit=16,
concurrency_id="ranked").then(leaderboard_view, lb_protocol, lb)
lb_protocol.change(leaderboard_view, lb_protocol, lb)
lb_refresh.click(refresh_leaderboard, lb_protocol, lb)
demo.load(leaderboard_view, lb_protocol, lb)
demo.queue(max_size=32)
if __name__ == "__main__":
demo.launch(css=APP_CSS + CSS + LB_CSS, theme=THEME, ssr_mode=False)
|