"""LM Tetris Arena — decoder-only LMs play Tetris zero-shot and earn Elo.""" import os # Keep the write token out of the environment before any model code runs: # only the results store receives it (custom model code runs in this process). _TOKEN = os.environ.pop("HF_TOKEN", None) or os.environ.pop("HUGGING_FACE_HUB_TOKEN", None) os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1") import html import random import time import gradio as gr import pandas as pd import torch from arena import MAX_PIECES, ResultsStore, choose, rank_games from players import (BASELINES, MAX_PARAMS, ORACLE_ID, PROMPTS, RANDOM_ID, ModelRejected, OracleReaderPlayer, RandomPlayer, fmt_params, load_player, precheck) from render import CSS, arena_html, empty_html, results_html from tetris import TetrisGame # cpu-basic Spaces have 2 vCPUs; os.cpu_count() reports the host, which oversubscribes threads torch.set_num_threads(int(os.environ.get("TORCH_THREADS", 2))) RESULTS_REPO = os.environ.get("RESULTS_REPO", "DedeProGames/lm-tetris-arena-results") MAX_MODELS = int(os.environ.get("MAX_MODELS", 4)) STORE = ResultsStore(RESULTS_REPO, _TOKEN) SUGGESTED = [ '56m/Dumb-1.2-RC1', 'allura-org/Rambley-150M-RealBase', 'altslate/JugnuLM-110M-R2plus', 'appvoid/void.0', 'AtomixLabs/AtomixS2-5M-v1.0', 'AxiomicLabs/GPT-S-1.4M', 'AxiomicLabs/GPT-S2-5M', 'AxiomicLabs/GPT-X2.5-135M', 'BananaMind/BananaMind-2-Medium', 'BananaMind/BananaMind-2-Micro', 'BananaMind/BananaMind-2-Mini', 'BananaMind/BananaMind-2-MoE', 'BananaMind/BananaMind-2-Nano', 'BananaMind/BananaMind-2-Pro', 'BananaMind/BananaMind-2.1-Pico-Preview', 'BananaMind/BananaMind-2.1-Unified', 'bench-labs/cagliostro-v3', 'CNWPlayer/VegaLM1-42M-Base', 'CodeSoft/sorbet-v2-25m', 'DALabCommunity/Haidass1.5-143M', 'DedeProGames/DynamicMind-Mini', 'DedeProGames/DynamicMind-MoE', 'DedeProGames/Kiyo-230M-Preview', 'DedeProGames/Kiyo-65M', 'DedeProGames/LowOnMind-1M', 'DedeProGames/LowOnMind-300k', 'DedeProGames/LowOnMind-5M', 'DedeProGames/LowOnMind-8M', 'DedeProGames/NanoDex-1M', 'Dream-W/ObsidianSmall-Base', 'egafni/pico-llama', 'EleutherAI/pythia-160m', 'EleutherAI/pythia-70m', 'finnianx/Gros-Michel-90m-Base-v2', 'FlameF0X/TinyMoE-100m-2x8-retrained', 'fromziro/ZeroS-Linear-50M', 'fromziro/ZeroS-Pico-v1.1', 'fromziro/ZeroS-Qana-5M', 'fromziro/ZeroS-v0.1-150M', 'FWKV/Myosotis-1-base', 'GODELEV/Rose-1.5-Medium', 'GODELEV/Rose-Mini', 'Harley-ml/Dillionv2-1.3M', 'HuggingFaceTB/SmolLM2-135M', 'IvmeLabs/Ivme-Conversate-v3-Base', 'jhu-clsp/ettin-decoder-150m', 'jhu-clsp/ettin-decoder-17m', 'jhu-clsp/ettin-decoder-32m', 'jhu-clsp/ettin-decoder-68m', 'joelhenwang/OdinNext-138M-Base', 'LH-Tech-AI/Spark-5M-Base-v4', 'MaliosDark/Nexus-Erebus-135M', 'MaliosDark/Nexus-Erebus-3M', 'MaliosDark/Nexus-Erebus-50M', 'MihaiPopa-1/CinnabarLM-1.4M-Base', 'MinimaLabs/KeyLM-75M', 'MinimaLabs/min-spark-1.1', 'Nikity/lille-130m-base', 'openai-community/gpt2', 'opencerebral/Boris-1.3-125M', 'opencerebral/Boris-1.3-75M', 'opencerebral/littlerock-1M', 'qikp/kite-7-15m-base', 'Quazim0t0/Escarda-86M-Base', 'roneneldan/TinyStories-33M', 'SlayerLab/pollock-mini-lm-125m', 'solintellegence/Sol-Lite-Base', 'specklabs/Speck2-140M', 'StentorLabs/Stentor3-20M', 'StentorLabs/Stentor3-50M', 'SupraLabs/Supra2-100M-Base', 'SupraLabs/Supra2-Medium-Base', 'SupraLabs/SupraGDN-5M', 'SupraLabs/SupraNeo-4M', 'SurjoLabs/Ember-2', 'SurjoLabs/Flare', 'SurjoLabs/Surjo-50m', 'sz14/cRia-LM-75M', 'TobiasLogic/Museko-125M', 'UniversalComputingResearch/Atom2.7m', 'User01110/CMA-8M', 'veyra-ai/Veyra2-Blueberry-10M-Base', 'veyra-ai/Veyra2-Blueberry-5M-Base', 'WhirlwindAI/MetaNova-1-60M', ] DEFAULT_MODELS = ["DedeProGames/Kiyo-65M", "BananaMind/BananaMind-2-Medium", "SupraLabs/Supra2-Medium-Base", "AxiomicLabs/GPT-X2.5-135M"] PROTOCOL_CHOICES = [("Guided: the rules are in the prompt", "guided"), ("Blind: no rules, only pre-training knowledge", "blind")] BASELINE_CHOICES = [(label, key) for key, label in BASELINES.items()] def _status(text, kind="info"): icon = {"info": "⏳", "ok": "✅", "err": "⛔", "warn": "⚠️"}[kind] return f"{icon} {text}" def run_match(model_ids, baselines, protocol, ranked, seed, delay): model_ids = [m.strip() for m in (model_ids or []) if m and m.strip()] model_ids = list(dict.fromkeys(model_ids)) baselines = baselines or [] protocol = protocol or "guided" if not model_ids: yield _status("Pick at least one language model.", "err"), empty_html(), "" return if len(model_ids) > MAX_MODELS: yield _status(f"At most {MAX_MODELS} language models per match on this CPU.", "err"), empty_html(), "" return if len(model_ids) + len(baselines) < 2: yield _status("A match needs at least 2 players: add another model or a baseline.", "err"), empty_html(), "" return players = [] try: metas = [] for m in model_ids: yield _status(f"Checking `{m}`…"), empty_html("Checking models…"), "" meta = precheck(m) if any(x["id"] == meta["id"] for x in metas): continue # same repo typed twice with different casing metas.append(meta) model_ids = [x["id"] for x in metas] for i, (m, meta) in enumerate(zip(model_ids, metas), 1): yield _status(f"Loading `{m}` on CPU ({i}/{len(model_ids)})… first load downloads the weights."), empty_html("Loading models…"), "" players.append(load_player(m, meta)) except ModelRejected as e: yield _status(str(e), "err"), empty_html("Match cancelled."), "" return if RANDOM_ID in baselines: players.append(RandomPlayer()) if ORACLE_ID in baselines: players.append(OracleReaderPlayer()) if ranked: seed = random.SystemRandom().randrange(1, 10**9) else: seed = int(seed or 0) games = [TetrisGame(seed) for _ in players] mode = "ranked" if ranked else "unranked" yield _status(f"Seed {seed} · {protocol} · {mode}. Scoring the first moves…"), arena_html(games, players), "" last = time.time() try: while True: active = [(g, p) for g, p in zip(games, players) if g.alive and g.pieces < MAX_PIECES] if not active: break for g, p in active: choose(g, p, protocol, seed) elapsed = time.time() - last if elapsed < delay: time.sleep(delay - elapsed) last = time.time() n = max(g.pieces for g in games) alive = sum(g.alive for g in games) yield _status(f"Seed {seed} · {protocol} · {mode} · piece {n}/{MAX_PIECES} · {alive} still playing"), arena_html(games, players), "" except ModelRejected as e: yield _status(str(e), "err"), arena_html(games, players), "" return except Exception as e: yield _status(f"A model crashed during play: {type(e).__name__}: {str(e)[:200]}", "err"), arena_html(games, players), "" return ranks = rank_games(games) order = sorted(range(len(players)), key=lambda i: ranks[i]) elos = None note = "" if ranked: match = STORE.record(protocol, seed, players, games) elos = [(pp["elo_before"], pp["elo_after"]) for pp in match["players"]] if STORE.persistent and not STORE.save_error: note = f'Elo updated and saved to the public leaderboard ({RESULTS_REPO}).' elif STORE.persistent: note = f"⚠️ Elo updated in memory, but {html.escape(STORE.save_error)}." else: note = "⚠️ Elo updated in memory only: the Space has no HF_TOKEN secret, so results are not saved." else: note = "Unranked match: Elo not changed." note += f" Ranking: score, then lines, then pieces survived. ✓ = still alive at the {MAX_PIECES}-piece cap." results = results_html(order, ranks, players, games, elos, note) yield _status(f"Match finished · seed {seed} · {protocol}.", "ok"), arena_html(games, players, ranks, elos), results def stop_status(current): # only claim a stop when a match was actually running if (current or "").startswith("⏳"): return _status("Match stopped. Nothing was recorded.", "warn") return current def leaderboard_df(protocol): rows = [] for i, e in enumerate(STORE.rows(protocol or "guided"), 1): mid = e["model"] name = BASELINES.get(mid) or f"[{mid}](https://huggingface.co/{mid})" g = max(1, e["games"]) rows.append([ i, name, round(e["elo"]), e["games"], e["wins"], round(e["total_pieces"] / g, 1), round(e["total_lines"] / g, 1), e["best_score"], "–" if mid in BASELINES else fmt_params(e.get("params")), ]) cols = ["#", "Model", "Elo", "Games", "1st places", "Avg pieces", "Avg lines", "Best score", "Params"] return pd.DataFrame(rows, columns=cols) def refresh_leaderboard(protocol): STORE.reload() return leaderboard_df(protocol) INTRO = f""" # 🧱 LM Tetris Arena Small **decoder-only language models** (≤ {fmt_params(MAX_PARAMS)} parameters, custom architectures welcome) play Tetris **zero-shot**: no fine-tuning, no game data, only what they learned from pre-training on text. All players get the **same piece sequence** (same seed). Ranked matches update a public **Elo** leaderboard. """ HOW = f""" ### How a model plays For every new piece the game lists all legal placements (rotation × column, hard drop), simulates each one and describes the outcome in plain English. The model never sees the grid; it judges the descriptions: ``` {PROMPTS['guided'].format(desc='drops the piece into the lowest part of the board, clears one line, creates no new holes, keeps the stack low and leaves the surface flat')} ``` The model's value for a placement is **log P(" good move") − log P(" bad move")** after that prompt. The placement with the highest value is played; exact ties are broken by a seeded coin that is identical for every player. Because the value is a difference, a model's general bias towards "good" or "bad" cancels out. ### Protocols (separate leaderboards) - **Guided**: the first line states the goal ("clear lines, avoid holes, keep the stack low"). Tests reading comprehension. - **Blind**: `{PROMPTS['blind'].splitlines()[0]}` No rules; the model must already know what is good in Tetris. ### Rules of a match - 2+ players, up to {MAX_MODELS} language models per match, plus optional baselines. - Same 7-bag piece sequence for everyone. The game ends at top-out or after {MAX_PIECES} pieces. - Placement = score (100/300/500/800 for 1/2/3/4 lines), then lines, then pieces survived. - **Ranked** matches use a random seed and update Elo (K=32, multiplayer: every pair of players counts as a game, scaled by 1/(N−1)). Unranked matches let you pick the seed and change nothing. ### Baselines - **🎲 Random**: every placement ties, so it plays uniformly at random. This is the floor a model should beat. - **📏 Oracle reader**: reads the same descriptions and ranks them with fixed common sense (holes > lines > height > surface > landing). This is roughly the ceiling for a perfect reader of the text. ### Model requirements Public, not gated, loads with `AutoModelForCausalLM` + `AutoTokenizer` (PyTorch or safetensors weights), ≤ {fmt_params(MAX_PARAMS)} parameters. Models with custom code (`auto_map`) load with `trust_remote_code=True`. That code runs on this Space's CPU, so only submit repos you trust. Prompts are in English (the language most pre-training corpora such as FineWeb-edu use). Results and every match (seed, commit SHA of each model, scores) are published in [`{RESULTS_REPO}`](https://huggingface.co/datasets/{RESULTS_REPO}). """ with gr.Blocks(title="LM Tetris Arena") as demo: gr.Markdown(INTRO) if not STORE.persistent: gr.Markdown("⚠️ **Results are not being saved**: add an `HF_TOKEN` secret with write access to the results dataset.") with gr.Tabs(): with gr.Tab("⚔️ Match"): with gr.Row(): with gr.Column(scale=3): models = gr.Dropdown( choices=SUGGESTED, value=DEFAULT_MODELS, multiselect=True, allow_custom_value=True, max_choices=MAX_MODELS, label=f"Language models (1–{MAX_MODELS})", info="Pick from the list or type any Hub repo id (owner/name) and press Enter.", ) baselines = gr.CheckboxGroup(BASELINE_CHOICES, value=[RANDOM_ID], label="Baselines (optional, also rated)") with gr.Column(scale=2): protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol") with gr.Row(): ranked = gr.Checkbox(value=True, label="Ranked (random seed, updates Elo)") seed = gr.Number(value=42, precision=0, label="Seed (unranked only)") delay = gr.Slider(0, 0.5, value=0.12, step=0.02, label="Seconds per piece (viewing speed)") with gr.Row(): start = gr.Button("▶ Start match", variant="primary") stop = gr.Button("■ Stop", variant="stop") status = gr.Markdown(_status("Ready.", "ok")) boards = gr.HTML(empty_html()) results = gr.HTML() with gr.Tab("🏆 Leaderboard"): lb_protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol") lb = gr.Dataframe( value=leaderboard_df("guided"), interactive=False, wrap=True, datatype=["number", "markdown", "number", "number", "number", "number", "number", "number", "str"], ) lb_refresh = gr.Button("↻ Refresh") with gr.Tab("📖 How it works"): gr.Markdown(HOW) match_event = start.click( run_match, [models, baselines, protocol, ranked, seed, delay], [status, boards, results], concurrency_limit=1, ) stop.click(stop_status, status, status, cancels=[match_event]) match_event.then(leaderboard_df, lb_protocol, lb) lb_protocol.change(leaderboard_df, lb_protocol, lb) lb_refresh.click(refresh_leaderboard, lb_protocol, lb) demo.load(leaderboard_df, lb_protocol, lb) demo.queue(max_size=32) if __name__ == "__main__": demo.launch(css=CSS, theme=gr.themes.Soft(primary_hue="violet"), ssr_mode=False)