DedeProGames's picture
Remove 10 suggested models that currently fail to load
93d33bb verified
Raw History Blame
21.7 kB
"""LM Tetris Arena — decoder-only LMs play Tetris zero-shot and earn Elo."""
import os
# Keep the write token out of the environment before any model code runs:
# only the results store receives it (custom model code runs in this process).
_TOKEN = os.environ.pop("HF_TOKEN", None) or os.environ.pop("HUGGING_FACE_HUB_TOKEN", None)
os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1")
import html
import inspect
import random
import time
import gradio as gr
import torch
from arena import MAX_PIECES, SEASON, ResultsStore, choose, rank_games
from leaderboard import LB_CSS, leaderboard_html
from leaderboard import fmt_params as short_params
from players import (BASELINES, MAX_PARAMS, ORACLE_ID, PROMPTS, RANDOM_ID, ModelRejected, OracleReaderPlayer,
RandomPlayer, fmt_params, load_player, precheck)
from render import CSS, arena_html, empty_html, results_html
from tetris import TetrisGame
# cpu-basic Spaces have 2 vCPUs; os.cpu_count() reports the host, which oversubscribes threads
torch.set_num_threads(int(os.environ.get("TORCH_THREADS", 2)))
RESULTS_REPO = os.environ.get("RESULTS_REPO", "DedeProGames/lm-tetris-arena-results")
MAX_MODELS = int(os.environ.get("MAX_MODELS", 4))
# Ranked matches only between models of similar size, so big models can't farm Elo from tiny ones
MAX_PARAM_GAP = int(os.environ.get("MAX_PARAM_GAP", 20_000_000))
STORE = ResultsStore(RESULTS_REPO, _TOKEN)
SUGGESTED = [
'56m/Dumb-1.2-RC1',
'allura-org/Rambley-150M-RealBase',
'altslate/JugnuLM-110M-R2plus',
'appvoid/void.0',
'AtomixLabs/AtomixS2-5M-v1.0',
'AxiomicLabs/GPT-S-1.4M',
'AxiomicLabs/GPT-S2-5M',
'AxiomicLabs/GPT-X2.5-135M',
'BananaMind/BananaMind-2-Medium',
'BananaMind/BananaMind-2-Micro',
'BananaMind/BananaMind-2-Mini',
'BananaMind/BananaMind-2-MoE',
'BananaMind/BananaMind-2-Nano',
'BananaMind/BananaMind-2-Pro',
'BananaMind/BananaMind-2.1-Pico-Preview',
'BananaMind/BananaMind-2.1-Unified',
'bench-labs/cagliostro-v3',
'CNWPlayer/VegaLM1-42M-Base',
'CodeSoft/sorbet-v2-25m',
'DedeProGames/DynamicMind-Mini',
'DedeProGames/DynamicMind-MoE',
'DedeProGames/Kiyo-230M-Preview',
'DedeProGames/Kiyo-65M',
'DedeProGames/LowOnMind-1M',
'DedeProGames/LowOnMind-300k',
'DedeProGames/LowOnMind-5M',
'DedeProGames/LowOnMind-8M',
'DedeProGames/NanoDex-1M',
'EleutherAI/pythia-160m',
'EleutherAI/pythia-70m',
'finnianx/Gros-Michel-90m-Base-v2',
'FlameF0X/TinyMoE-100m-2x8-retrained',
'fromziro/ZeroS-Qana-5M',
'fromziro/ZeroS-v0.1-150M',
'FWKV/Myosotis-1-base',
'GODELEV/Rose-1.5-Medium',
'GODELEV/Rose-Mini',
'Harley-ml/Dillionv2-1.3M',
'HuggingFaceTB/SmolLM2-135M',
'IvmeLabs/Ivme-Conversate-v3-Base',
'jhu-clsp/ettin-decoder-150m',
'jhu-clsp/ettin-decoder-17m',
'jhu-clsp/ettin-decoder-32m',
'jhu-clsp/ettin-decoder-68m',
'joelhenwang/OdinNext-138M-Base',
'LH-Tech-AI/Spark-5M-Base-v4',
'MaliosDark/Nexus-Erebus-135M',
'MaliosDark/Nexus-Erebus-3M',
'MaliosDark/Nexus-Erebus-50M',
'MihaiPopa-1/CinnabarLM-1.4M-Base',
'MinimaLabs/KeyLM-75M',
'MinimaLabs/min-spark-1.1',
'openai-community/gpt2',
'opencerebral/Boris-1.3-125M',
'opencerebral/Boris-1.3-75M',
'opencerebral/littlerock-1M',
'qikp/kite-7-15m-base',
'roneneldan/TinyStories-33M',
'SlayerLab/pollock-mini-lm-125m',
'StentorLabs/Stentor3-20M',
'StentorLabs/Stentor3-50M',
'SupraLabs/Supra2-100M-Base',
'SupraLabs/Supra2-Medium-Base',
'SupraLabs/SupraGDN-5M',
'SupraLabs/SupraNeo-4M',
'SurjoLabs/Ember-2',
'SurjoLabs/Flare',
'SurjoLabs/Surjo-50m',
'sz14/cRia-LM-75M',
'TobiasLogic/Museko-125M',
'UniversalComputingResearch/Atom2.7m',
'User01110/CMA-8M',
'veyra-ai/Veyra2-Blueberry-10M-Base',
'veyra-ai/Veyra2-Blueberry-5M-Base',
]
DEFAULT_MODELS = ["DedeProGames/Kiyo-65M", "BananaMind/BananaMind-2-Medium", "SurjoLabs/Surjo-50m", "StentorLabs/Stentor3-50M"]
PROTOCOL_CHOICES = [("Guided: the rules are in the prompt", "guided"), ("Blind: no rules, only pre-training knowledge", "blind")]
BASELINE_CHOICES = [(label, key) for key, label in BASELINES.items()]
# ---- Look & feel: same palette and type as the BananaMind SLM Leaderboard (dark + light) ----
_D = dict(bg="#0b0e0d", surface="#111613", surface2="#191e1b", text="#f0f1ec", muted="#929b93", line="#29312c", accent="#facc15")
_L = dict(bg="#f4f5f1", surface="#ffffff", surface2="#edf0e9", text="#17221b", muted="#626e64", line="#d7ded5", accent="#8c6500")
_CHECK = ("url(\"data:image/svg+xml,%3csvg viewBox='0 0 16 16' fill='%231b200e' xmlns='http://www.w3.org/2000/svg'%3e"
"%3cpath d='M12.207 4.793a1 1 0 010 1.414l-5 5a1 1 0 01-1.414 0l-2-2a1 1 0 011.414-1.414L6.5 9.086l4.293-4.293a1 1 0 011.414 0z'/%3e%3c/svg%3e\")")
_THEME_KEYS = set(inspect.signature(gr.themes.Base.set).parameters)
def _both(**pairs):
"""name=(light, dark) -> theme kwargs for both modes (skips variables this Gradio version lacks)."""
out = {}
for k, (light, dark) in pairs.items():
if k in _THEME_KEYS:
out[k] = light
if k + "_dark" in _THEME_KEYS:
out[k + "_dark"] = dark
return out
THEME = gr.themes.Base(
primary_hue=gr.themes.colors.yellow, secondary_hue=gr.themes.colors.yellow, neutral_hue=gr.themes.colors.stone,
font=[gr.themes.GoogleFont("DM Sans"), "Arial", "sans-serif"], font_mono=["ui-monospace", "SFMono-Regular", "monospace"],
).set(
block_border_width="1px", block_radius="13px", block_label_border_width="0px", block_title_text_weight="500",
input_radius="8px", checkbox_check=_CHECK,
**_both(
body_background_fill=(_L["bg"], _D["bg"]), body_text_color=(_L["text"], _D["text"]),
body_text_color_subdued=(_L["muted"], _D["muted"]),
background_fill_primary=(_L["surface"], _D["surface"]), background_fill_secondary=(_L["surface2"], _D["surface2"]),
block_background_fill=(_L["surface"], _D["surface"]), block_border_color=(_L["line"], _D["line"]),
block_shadow=("none", "none"), block_label_background_fill=("transparent", "transparent"),
block_label_text_color=(_L["muted"], _D["muted"]), block_label_shadow=("none", "none"),
block_title_background_fill=("transparent", "transparent"), block_title_text_color=(_L["muted"], _D["muted"]),
block_info_text_color=(_L["muted"], _D["muted"]),
border_color_primary=(_L["line"], _D["line"]), border_color_accent=("#facc15", "#facc15"),
color_accent=("#facc15", "#facc15"), color_accent_soft=("#facc1526", "#facc1526"),
input_background_fill=(_L["surface2"], _D["surface2"]), input_border_color=(_L["line"], _D["line"]),
input_border_color_focus=("#facc15", "#facc15"),
button_primary_background_fill=("#facc15", "#facc15"), button_primary_background_fill_hover=("#fde047", "#fde047"),
button_primary_text_color=("#1b200e", "#1b200e"), button_primary_border_color=("#facc15", "#facc15"),
button_secondary_background_fill=(_L["surface2"], _D["surface2"]),
button_secondary_background_fill_hover=(_L["line"], _D["line"]),
button_secondary_text_color=(_L["text"], _D["text"]), button_secondary_border_color=(_L["line"], _D["line"]),
checkbox_background_color_selected=("#facc15", "#facc15"), checkbox_border_color_selected=("#facc15", "#facc15"),
checkbox_label_background_fill=(_L["surface2"], _D["surface2"]),
checkbox_label_background_fill_selected=(_L["surface2"], _D["surface2"]),
checkbox_label_border_color=(_L["line"], _D["line"]), checkbox_label_border_color_selected=("#facc15", "#facc15"),
checkbox_label_text_color_selected=(_L["text"], _D["text"]),
slider_color=("#facc15", "#facc15"), loader_color=("#facc15", "#facc15"),
link_text_color=(_L["accent"], _D["accent"]), link_text_color_hover=(_L["accent"], _D["accent"]),
panel_background_fill=(_L["surface"], _D["surface"]), panel_border_color=(_L["line"], _D["line"]),
table_border_color=(_L["line"], _D["line"]), code_background_fill=(_L["surface2"], _D["surface2"]),
),
)
APP_CSS = """
@import url('https://fonts.googleapis.com/css2?family=DM+Sans:wght@400;500;600;700&family=Space+Grotesk:wght@400;500;600;700&display=swap');
.gradio-container{font-family:'DM Sans',Arial,sans-serif!important}
.gradio-container h1,.gradio-container h2,.gradio-container h3{font-family:'Space Grotesk',Arial,sans-serif!important;letter-spacing:-.4px}
.ah-heading{display:flex;justify-content:space-between;align-items:center;gap:20px;padding:10px 0 6px}
.ah-eyebrow{font:11px/1.5 monospace!important;letter-spacing:1.9px;color:var(--body-text-color-subdued)!important;margin:0 0 10px!important}
.ah-heading h1{font:500 clamp(32px,4vw,48px)/1.2 'Space Grotesk',Arial,sans-serif!important;letter-spacing:-2px!important;margin:0!important;
color:var(--body-text-color)!important}
.ah-accent{color:#8c6500}.dark .ah-accent{color:#facc15}
.ah-intro{margin:10px 0 0!important;font-size:15px!important;color:var(--body-text-color-subdued)!important;max-width:860px}
.ah-version{font:11px monospace;letter-spacing:1px;color:var(--body-text-color-subdued);display:flex;align-items:center;gap:10px;white-space:nowrap}
.ah-dot{height:6px;width:6px;background:#95c79a;border-radius:50%}
.ah-warn{margin-top:14px;padding:10px 14px;border:1px solid #facc1566;border-radius:8px;font-size:13px;color:var(--body-text-color)}
button[role=tab]{font-size:14px!important;color:var(--body-text-color-subdued)!important}
button[role=tab][aria-selected=true]{color:var(--body-text-color)!important;border-color:#facc15!important}
@media(max-width:550px){.ah-version{display:none}}
"""
def _status(text, kind="info"):
icon = {"info": "⏳", "ok": "✅", "err": "⛔", "warn": "⚠️"}[kind]
return f"{icon} {text}"
def run_match(model_ids, baselines, protocol, ranked, seed, delay):
model_ids = [m.strip() for m in (model_ids or []) if m and m.strip()]
model_ids = list(dict.fromkeys(model_ids))
baselines = baselines or []
protocol = protocol or "guided"
if not model_ids:
yield _status("Pick at least one language model.", "err"), empty_html(), ""
return
if len(model_ids) > MAX_MODELS:
yield _status(f"At most {MAX_MODELS} language models per match on this CPU.", "err"), empty_html(), ""
return
if len(model_ids) + len(baselines) < 2:
yield _status("A match needs at least 2 players: add another model or a baseline.", "err"), empty_html(), ""
return
players = []
try:
metas = []
for m in model_ids:
yield _status(f"Checking `{m}`…"), empty_html("Checking models…"), ""
meta = precheck(m)
if any(x["id"] == meta["id"] for x in metas):
continue # same repo typed twice with different casing
metas.append(meta)
model_ids = [x["id"] for x in metas]
for i, (m, meta) in enumerate(zip(model_ids, metas), 1):
yield _status(f"Loading `{m}` on CPU ({i}/{len(model_ids)})… first load downloads the weights."), empty_html("Loading models…"), ""
players.append(load_player(m, meta))
except ModelRejected as e:
yield _status(str(e), "err"), empty_html("Match cancelled."), ""
return
# Ranked matches need 2+ language models of similar size; otherwise the match is blocked.
if ranked:
sizes = [p.n_params for p in players]
if len(players) < 2:
yield _status("Ranked match blocked: it needs at least 2 language models (baselines are never rated). "
"Add a model or uncheck Ranked.", "err"), empty_html("Match blocked."), ""
return
if max(sizes) - min(sizes) > MAX_PARAM_GAP:
listing = ", ".join(f"{p.model_id} ({fmt_params(p.n_params)})" for p in players)
yield _status(f"Ranked match blocked: models must be within ±{fmt_params(MAX_PARAM_GAP)} parameters of each other. "
f"This match spans {fmt_params(min(sizes))} to {fmt_params(max(sizes))}: {listing}. "
"Pick models of similar size or uncheck Ranked.", "err"), empty_html("Match blocked."), ""
return
if RANDOM_ID in baselines:
players.append(RandomPlayer())
if ORACLE_ID in baselines:
players.append(OracleReaderPlayer())
if ranked:
seed = random.SystemRandom().randrange(1, 10**9)
else:
seed = int(seed or 0)
games = [TetrisGame(seed) for _ in players]
mode = "ranked" if ranked else "unranked"
yield _status(f"Seed {seed} · {protocol} · {mode}. Scoring the first moves…"), arena_html(games, players), ""
last = time.time()
try:
while True:
active = [(g, p) for g, p in zip(games, players) if g.alive and g.pieces < MAX_PIECES]
if not active:
break
for g, p in active:
choose(g, p, protocol, seed)
elapsed = time.time() - last
if elapsed < delay:
time.sleep(delay - elapsed)
last = time.time()
n = max(g.pieces for g in games)
alive = sum(g.alive for g in games)
yield _status(f"Seed {seed} · {protocol} · {mode} · piece {n}/{MAX_PIECES} · {alive} still playing"), arena_html(games, players), ""
except ModelRejected as e:
yield _status(str(e), "err"), arena_html(games, players), ""
return
except Exception as e:
yield _status(f"A model crashed during play: {type(e).__name__}: {str(e)[:200]}", "err"), arena_html(games, players), ""
return
ranks = rank_games(games)
order = sorted(range(len(players)), key=lambda i: ranks[i])
elos = None
note = ""
if ranked:
rated = [i for i, p in enumerate(players) if p.model_id not in BASELINES] # baselines are never rated
match = STORE.record(protocol, seed, [players[i] for i in rated], [games[i] for i in rated])
elos = [None] * len(players)
for k, i in enumerate(rated):
elos[i] = (match["players"][k]["elo_before"], match["players"][k]["elo_after"])
if STORE.persistent and not STORE.save_error:
note = f'Elo updated and saved to the public leaderboard (<a href="https://huggingface.co/datasets/{RESULTS_REPO}" target="_blank">{RESULTS_REPO}</a>).'
elif STORE.persistent:
note = f"⚠️ Elo updated in memory, but {html.escape(STORE.save_error)}."
else:
note = "⚠️ Elo updated in memory only: the Space has no <code>HF_TOKEN</code> secret, so results are not saved."
else:
note = "Unranked match: Elo not changed."
note += f" Ranking: score, then lines, then pieces survived. ✓ = still alive at the {MAX_PIECES}-piece cap."
results = results_html(order, ranks, players, games, elos, note)
yield _status(f"Match finished · seed {seed} · {protocol} · {mode}.", "ok"), arena_html(games, players, ranks, elos), results
def stop_status(current):
# only claim a stop when a match was actually running
if (current or "").startswith("⏳"):
return _status("Match stopped. Nothing was recorded.", "warn")
return current
def leaderboard_view(protocol):
protocol = protocol or "guided"
entries = [e for e in STORE.rows(protocol) if e["model"] not in BASELINES]
return leaderboard_html(entries, protocol, MAX_PARAM_GAP, short_params(MAX_PARAM_GAP))
def refresh_leaderboard(protocol):
STORE.reload()
return leaderboard_view(protocol)
INTRO = f"""
<div class="ah-heading"><div><p class="ah-eyebrow">SMALL MODELS. ZERO-SHOT TETRIS.</p>
<h1>LM Tetris Arena<span class="ah-accent">.</span></h1>
<p class="ah-intro">Decoder-only language models (≤ {short_params(MAX_PARAMS)} parameters, custom architectures welcome) play Tetris
zero-shot: no fine-tuning, no game data, only what they learned from pre-training on text. Every player gets the same
piece sequence, and ranked matches update a public Elo leaderboard.</p></div>
<span class="ah-version">SEASON {SEASON} <span class="ah-dot"></span></span></div>
"""
HOW = f"""
### How a model plays
For every new piece the game lists all legal placements (rotation × column, hard drop), simulates each one and
describes the outcome in plain English. The model never sees the grid; it judges the descriptions:
```
{PROMPTS['guided'].format(desc='drops the piece into the lowest part of the board, clears one line, creates no new holes, keeps the stack low and leaves the surface flat')}
```
The model's value for a placement is **log P(" good move") − log P(" bad move")** after that prompt. The placement with the
highest value is played; exact ties are broken by a seeded coin that is identical for every player.
Because the value is a difference, a model's general bias towards "good" or "bad" cancels out.
### Protocols (separate leaderboards)
- **Guided**: the first line states the goal ("clear lines, avoid holes, keep the stack low"). Tests reading comprehension.
- **Blind**: `{PROMPTS['blind'].splitlines()[0]}` No rules; the model must already know what is good in Tetris.
### Rules of a match
- 2+ players, up to {MAX_MODELS} language models per match, plus optional baselines.
- Same 7-bag piece sequence for everyone. The game ends at top-out or after {MAX_PIECES} pieces.
- Placement = score (100/300/500/800 for 1/2/3/4 lines), then lines, then pieces survived.
- **Ranked** matches use a random seed and update Elo (K=32, multiplayer: every pair of players counts as a game, scaled by 1/(N−1)).
They need 2+ language models whose sizes are within {fmt_params(MAX_PARAM_GAP)} parameters of each other, so a big model can't farm Elo from tiny ones.
If the sizes are further apart, the ranked match is blocked (uncheck Ranked to play it unranked).
Unranked matches let you pick the seed and change nothing.
### Baselines
- **🎲 Random**: every placement ties, so it plays uniformly at random. This is the floor a model should beat.
- **📏 Oracle reader**: reads the same descriptions and ranks them with fixed common sense (holes > lines > height > surface > landing).
This is roughly the ceiling for a perfect reader of the text.
Baselines can join any match for comparison, but they are never rated and never change anyone's Elo.
### Model requirements
Public, not gated, loads with `AutoModelForCausalLM` + `AutoTokenizer` (PyTorch or safetensors weights), ≤ {fmt_params(MAX_PARAMS)} parameters.
Models with custom code (`auto_map`) load with `trust_remote_code=True`. That code runs on this Space's CPU, so only
submit repos you trust. Prompts are in English (the language most pre-training corpora such as FineWeb-edu use).
Results and every match (seed, commit SHA of each model, scores) are published in
[`{RESULTS_REPO}`](https://huggingface.co/datasets/{RESULTS_REPO}).
"""
with gr.Blocks(title="LM Tetris Arena") as demo:
gr.HTML(INTRO, padding=False)
if not STORE.persistent:
gr.HTML('<div class="ah-warn">⚠️ Results are not being saved: add an <code>HF_TOKEN</code> secret with write access '
'to the results dataset.</div>', padding=False)
with gr.Tabs():
with gr.Tab("⚔️ Match"):
with gr.Row():
with gr.Column(scale=3):
models = gr.Dropdown(
choices=SUGGESTED, value=DEFAULT_MODELS, multiselect=True, allow_custom_value=True,
max_choices=MAX_MODELS, label=f"Language models (1–{MAX_MODELS})",
info="Pick from the list or type any Hub repo id (owner/name) and press Enter.",
)
baselines = gr.CheckboxGroup(BASELINE_CHOICES, value=[RANDOM_ID], label="Baselines (optional, never rated)")
with gr.Column(scale=2):
protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol")
with gr.Row():
ranked = gr.Checkbox(value=True, label=f"Ranked (random seed, updates Elo; models within ±{short_params(MAX_PARAM_GAP)})")
seed = gr.Number(value=42, precision=0, label="Seed (unranked only)")
delay = gr.Slider(0, 0.5, value=0.12, step=0.02, label="Seconds per piece (viewing speed)")
with gr.Row():
start = gr.Button("▶ Start match", variant="primary")
stop = gr.Button("■ Stop", variant="secondary")
status = gr.Markdown(_status("Ready.", "ok"))
boards = gr.HTML(empty_html())
results = gr.HTML()
with gr.Tab("🏆 Leaderboard"):
lb_protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol")
lb = gr.HTML(leaderboard_view("guided"), padding=False)
lb_refresh = gr.Button("↻ Refresh", variant="secondary")
with gr.Tab("📖 How it works"):
gr.Markdown(HOW)
match_event = start.click(
run_match, [models, baselines, protocol, ranked, seed, delay], [status, boards, results], concurrency_limit=1,
)
stop.click(stop_status, status, status, cancels=[match_event])
match_event.then(leaderboard_view, lb_protocol, lb)
lb_protocol.change(leaderboard_view, lb_protocol, lb)
lb_refresh.click(refresh_leaderboard, lb_protocol, lb)
demo.load(leaderboard_view, lb_protocol, lb)
demo.queue(max_size=32)
if __name__ == "__main__":
demo.launch(css=APP_CSS + CSS + LB_CSS, theme=THEME, ssr_mode=False)