Spaces:
Running
Running
Download app.py from DedeProGames/SLM-Tetris-Arena: direct link, hf CLI and curl.
- Browser
- Download file 14.9 kB
-
https://huggingface.co/spaces/DedeProGames/SLM-Tetris-Arena/resolve/d8b324c8e178dbdcba3eec40e1426b631436efa8/app.py
- Command line
-
hf download hf://spaces/DedeProGames/SLM-Tetris-Arena@d8b324c8e178dbdcba3eec40e1426b631436efa8/app.py
-
curl -L -o app.py https://huggingface.co/spaces/DedeProGames/SLM-Tetris-Arena/resolve/d8b324c8e178dbdcba3eec40e1426b631436efa8/app.py
14.9 kB
| """LM Tetris Arena — decoder-only LMs play Tetris zero-shot and earn Elo.""" | |
| import os | |
| # Keep the write token out of the environment before any model code runs: | |
| # only the results store receives it (custom model code runs in this process). | |
| _TOKEN = os.environ.pop("HF_TOKEN", None) or os.environ.pop("HUGGING_FACE_HUB_TOKEN", None) | |
| os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1") | |
| import html | |
| import random | |
| import time | |
| import gradio as gr | |
| import pandas as pd | |
| import torch | |
| from arena import MAX_PIECES, ResultsStore, choose, rank_games | |
| from players import (BASELINES, MAX_PARAMS, ORACLE_ID, PROMPTS, RANDOM_ID, ModelRejected, OracleReaderPlayer, | |
| RandomPlayer, fmt_params, load_player, precheck) | |
| from render import CSS, arena_html, empty_html, results_html | |
| from tetris import TetrisGame | |
| # cpu-basic Spaces have 2 vCPUs; os.cpu_count() reports the host, which oversubscribes threads | |
| torch.set_num_threads(int(os.environ.get("TORCH_THREADS", 2))) | |
| RESULTS_REPO = os.environ.get("RESULTS_REPO", "DedeProGames/lm-tetris-arena-results") | |
| MAX_MODELS = int(os.environ.get("MAX_MODELS", 4)) | |
| STORE = ResultsStore(RESULTS_REPO, _TOKEN) | |
| SUGGESTED = [ | |
| '56m/Dumb-1.2-RC1', | |
| 'allura-org/Rambley-150M-RealBase', | |
| 'altslate/JugnuLM-110M-R2plus', | |
| 'appvoid/void.0', | |
| 'AtomixLabs/AtomixS2-5M-v1.0', | |
| 'AxiomicLabs/GPT-S-1.4M', | |
| 'AxiomicLabs/GPT-S2-5M', | |
| 'AxiomicLabs/GPT-X2.5-135M', | |
| 'BananaMind/BananaMind-2-Medium', | |
| 'BananaMind/BananaMind-2-Micro', | |
| 'BananaMind/BananaMind-2-Mini', | |
| 'BananaMind/BananaMind-2-MoE', | |
| 'BananaMind/BananaMind-2-Nano', | |
| 'BananaMind/BananaMind-2-Pro', | |
| 'BananaMind/BananaMind-2.1-Pico-Preview', | |
| 'BananaMind/BananaMind-2.1-Unified', | |
| 'bench-labs/cagliostro-v3', | |
| 'CNWPlayer/VegaLM1-42M-Base', | |
| 'CodeSoft/sorbet-v2-25m', | |
| 'DALabCommunity/Haidass1.5-143M', | |
| 'DedeProGames/DynamicMind-Mini', | |
| 'DedeProGames/DynamicMind-MoE', | |
| 'DedeProGames/Kiyo-230M-Preview', | |
| 'DedeProGames/Kiyo-65M', | |
| 'DedeProGames/LowOnMind-1M', | |
| 'DedeProGames/LowOnMind-300k', | |
| 'DedeProGames/LowOnMind-5M', | |
| 'DedeProGames/LowOnMind-8M', | |
| 'DedeProGames/NanoDex-1M', | |
| 'Dream-W/ObsidianSmall-Base', | |
| 'egafni/pico-llama', | |
| 'EleutherAI/pythia-160m', | |
| 'EleutherAI/pythia-70m', | |
| 'finnianx/Gros-Michel-90m-Base-v2', | |
| 'FlameF0X/TinyMoE-100m-2x8-retrained', | |
| 'fromziro/ZeroS-Linear-50M', | |
| 'fromziro/ZeroS-Pico-v1.1', | |
| 'fromziro/ZeroS-Qana-5M', | |
| 'fromziro/ZeroS-v0.1-150M', | |
| 'FWKV/Myosotis-1-base', | |
| 'GODELEV/Rose-1.5-Medium', | |
| 'GODELEV/Rose-Mini', | |
| 'Harley-ml/Dillionv2-1.3M', | |
| 'HuggingFaceTB/SmolLM2-135M', | |
| 'IvmeLabs/Ivme-Conversate-v3-Base', | |
| 'jhu-clsp/ettin-decoder-150m', | |
| 'jhu-clsp/ettin-decoder-17m', | |
| 'jhu-clsp/ettin-decoder-32m', | |
| 'jhu-clsp/ettin-decoder-68m', | |
| 'joelhenwang/OdinNext-138M-Base', | |
| 'LH-Tech-AI/Spark-5M-Base-v4', | |
| 'MaliosDark/Nexus-Erebus-135M', | |
| 'MaliosDark/Nexus-Erebus-3M', | |
| 'MaliosDark/Nexus-Erebus-50M', | |
| 'MihaiPopa-1/CinnabarLM-1.4M-Base', | |
| 'MinimaLabs/KeyLM-75M', | |
| 'MinimaLabs/min-spark-1.1', | |
| 'Nikity/lille-130m-base', | |
| 'openai-community/gpt2', | |
| 'opencerebral/Boris-1.3-125M', | |
| 'opencerebral/Boris-1.3-75M', | |
| 'opencerebral/littlerock-1M', | |
| 'qikp/kite-7-15m-base', | |
| 'Quazim0t0/Escarda-86M-Base', | |
| 'roneneldan/TinyStories-33M', | |
| 'SlayerLab/pollock-mini-lm-125m', | |
| 'solintellegence/Sol-Lite-Base', | |
| 'specklabs/Speck2-140M', | |
| 'StentorLabs/Stentor3-20M', | |
| 'StentorLabs/Stentor3-50M', | |
| 'SupraLabs/Supra2-100M-Base', | |
| 'SupraLabs/Supra2-Medium-Base', | |
| 'SupraLabs/SupraGDN-5M', | |
| 'SupraLabs/SupraNeo-4M', | |
| 'SurjoLabs/Ember-2', | |
| 'SurjoLabs/Flare', | |
| 'SurjoLabs/Surjo-50m', | |
| 'sz14/cRia-LM-75M', | |
| 'TobiasLogic/Museko-125M', | |
| 'UniversalComputingResearch/Atom2.7m', | |
| 'User01110/CMA-8M', | |
| 'veyra-ai/Veyra2-Blueberry-10M-Base', | |
| 'veyra-ai/Veyra2-Blueberry-5M-Base', | |
| 'WhirlwindAI/MetaNova-1-60M', | |
| ] | |
| DEFAULT_MODELS = ["DedeProGames/Kiyo-65M", "BananaMind/BananaMind-2-Medium", "SupraLabs/Supra2-Medium-Base", "AxiomicLabs/GPT-X2.5-135M"] | |
| PROTOCOL_CHOICES = [("Guided: the rules are in the prompt", "guided"), ("Blind: no rules, only pre-training knowledge", "blind")] | |
| BASELINE_CHOICES = [(label, key) for key, label in BASELINES.items()] | |
| def _status(text, kind="info"): | |
| icon = {"info": "⏳", "ok": "✅", "err": "⛔", "warn": "⚠️"}[kind] | |
| return f"{icon} {text}" | |
| def run_match(model_ids, baselines, protocol, ranked, seed, delay): | |
| model_ids = [m.strip() for m in (model_ids or []) if m and m.strip()] | |
| model_ids = list(dict.fromkeys(model_ids)) | |
| baselines = baselines or [] | |
| protocol = protocol or "guided" | |
| if not model_ids: | |
| yield _status("Pick at least one language model.", "err"), empty_html(), "" | |
| return | |
| if len(model_ids) > MAX_MODELS: | |
| yield _status(f"At most {MAX_MODELS} language models per match on this CPU.", "err"), empty_html(), "" | |
| return | |
| if len(model_ids) + len(baselines) < 2: | |
| yield _status("A match needs at least 2 players: add another model or a baseline.", "err"), empty_html(), "" | |
| return | |
| players = [] | |
| try: | |
| metas = [] | |
| for m in model_ids: | |
| yield _status(f"Checking `{m}`…"), empty_html("Checking models…"), "" | |
| meta = precheck(m) | |
| if any(x["id"] == meta["id"] for x in metas): | |
| continue # same repo typed twice with different casing | |
| metas.append(meta) | |
| model_ids = [x["id"] for x in metas] | |
| for i, (m, meta) in enumerate(zip(model_ids, metas), 1): | |
| yield _status(f"Loading `{m}` on CPU ({i}/{len(model_ids)})… first load downloads the weights."), empty_html("Loading models…"), "" | |
| players.append(load_player(m, meta)) | |
| except ModelRejected as e: | |
| yield _status(str(e), "err"), empty_html("Match cancelled."), "" | |
| return | |
| if RANDOM_ID in baselines: | |
| players.append(RandomPlayer()) | |
| if ORACLE_ID in baselines: | |
| players.append(OracleReaderPlayer()) | |
| if ranked: | |
| seed = random.SystemRandom().randrange(1, 10**9) | |
| else: | |
| seed = int(seed or 0) | |
| games = [TetrisGame(seed) for _ in players] | |
| mode = "ranked" if ranked else "unranked" | |
| yield _status(f"Seed {seed} · {protocol} · {mode}. Scoring the first moves…"), arena_html(games, players), "" | |
| last = time.time() | |
| try: | |
| while True: | |
| active = [(g, p) for g, p in zip(games, players) if g.alive and g.pieces < MAX_PIECES] | |
| if not active: | |
| break | |
| for g, p in active: | |
| choose(g, p, protocol, seed) | |
| elapsed = time.time() - last | |
| if elapsed < delay: | |
| time.sleep(delay - elapsed) | |
| last = time.time() | |
| n = max(g.pieces for g in games) | |
| alive = sum(g.alive for g in games) | |
| yield _status(f"Seed {seed} · {protocol} · {mode} · piece {n}/{MAX_PIECES} · {alive} still playing"), arena_html(games, players), "" | |
| except ModelRejected as e: | |
| yield _status(str(e), "err"), arena_html(games, players), "" | |
| return | |
| except Exception as e: | |
| yield _status(f"A model crashed during play: {type(e).__name__}: {str(e)[:200]}", "err"), arena_html(games, players), "" | |
| return | |
| ranks = rank_games(games) | |
| order = sorted(range(len(players)), key=lambda i: ranks[i]) | |
| elos = None | |
| note = "" | |
| if ranked: | |
| match = STORE.record(protocol, seed, players, games) | |
| elos = [(pp["elo_before"], pp["elo_after"]) for pp in match["players"]] | |
| if STORE.persistent and not STORE.save_error: | |
| note = f'Elo updated and saved to the public leaderboard (<a href="https://huggingface.co/datasets/{RESULTS_REPO}" target="_blank">{RESULTS_REPO}</a>).' | |
| elif STORE.persistent: | |
| note = f"⚠️ Elo updated in memory, but {html.escape(STORE.save_error)}." | |
| else: | |
| note = "⚠️ Elo updated in memory only: the Space has no <code>HF_TOKEN</code> secret, so results are not saved." | |
| else: | |
| note = "Unranked match: Elo not changed." | |
| note += f" Ranking: score, then lines, then pieces survived. ✓ = still alive at the {MAX_PIECES}-piece cap." | |
| results = results_html(order, ranks, players, games, elos, note) | |
| yield _status(f"Match finished · seed {seed} · {protocol}.", "ok"), arena_html(games, players, ranks, elos), results | |
| def stop_status(current): | |
| # only claim a stop when a match was actually running | |
| if (current or "").startswith("⏳"): | |
| return _status("Match stopped. Nothing was recorded.", "warn") | |
| return current | |
| def leaderboard_df(protocol): | |
| rows = [] | |
| for i, e in enumerate(STORE.rows(protocol or "guided"), 1): | |
| mid = e["model"] | |
| name = BASELINES.get(mid) or f"[{mid}](https://huggingface.co/{mid})" | |
| g = max(1, e["games"]) | |
| rows.append([ | |
| i, name, round(e["elo"]), e["games"], e["wins"], | |
| round(e["total_pieces"] / g, 1), round(e["total_lines"] / g, 1), e["best_score"], | |
| "–" if mid in BASELINES else fmt_params(e.get("params")), | |
| ]) | |
| cols = ["#", "Model", "Elo", "Games", "1st places", "Avg pieces", "Avg lines", "Best score", "Params"] | |
| return pd.DataFrame(rows, columns=cols) | |
| def refresh_leaderboard(protocol): | |
| STORE.reload() | |
| return leaderboard_df(protocol) | |
| INTRO = f""" | |
| # 🧱 LM Tetris Arena | |
| Small **decoder-only language models** (≤ {fmt_params(MAX_PARAMS)} parameters, custom architectures welcome) play Tetris | |
| **zero-shot**: no fine-tuning, no game data, only what they learned from pre-training on text. | |
| All players get the **same piece sequence** (same seed). Ranked matches update a public **Elo** leaderboard. | |
| """ | |
| HOW = f""" | |
| ### How a model plays | |
| For every new piece the game lists all legal placements (rotation × column, hard drop), simulates each one and | |
| describes the outcome in plain English. The model never sees the grid; it judges the descriptions: | |
| ``` | |
| {PROMPTS['guided'].format(desc='drops the piece into the lowest part of the board, clears one line, creates no new holes, keeps the stack low and leaves the surface flat')} | |
| ``` | |
| The model's value for a placement is **log P(" good move") − log P(" bad move")** after that prompt. The placement with the | |
| highest value is played; exact ties are broken by a seeded coin that is identical for every player. | |
| Because the value is a difference, a model's general bias towards "good" or "bad" cancels out. | |
| ### Protocols (separate leaderboards) | |
| - **Guided**: the first line states the goal ("clear lines, avoid holes, keep the stack low"). Tests reading comprehension. | |
| - **Blind**: `{PROMPTS['blind'].splitlines()[0]}` No rules; the model must already know what is good in Tetris. | |
| ### Rules of a match | |
| - 2+ players, up to {MAX_MODELS} language models per match, plus optional baselines. | |
| - Same 7-bag piece sequence for everyone. The game ends at top-out or after {MAX_PIECES} pieces. | |
| - Placement = score (100/300/500/800 for 1/2/3/4 lines), then lines, then pieces survived. | |
| - **Ranked** matches use a random seed and update Elo (K=32, multiplayer: every pair of players counts as a game, scaled by 1/(N−1)). | |
| Unranked matches let you pick the seed and change nothing. | |
| ### Baselines | |
| - **🎲 Random**: every placement ties, so it plays uniformly at random. This is the floor a model should beat. | |
| - **📏 Oracle reader**: reads the same descriptions and ranks them with fixed common sense (holes > lines > height > surface > landing). | |
| This is roughly the ceiling for a perfect reader of the text. | |
| ### Model requirements | |
| Public, not gated, loads with `AutoModelForCausalLM` + `AutoTokenizer` (PyTorch or safetensors weights), ≤ {fmt_params(MAX_PARAMS)} parameters. | |
| Models with custom code (`auto_map`) load with `trust_remote_code=True`. That code runs on this Space's CPU, so only | |
| submit repos you trust. Prompts are in English (the language most pre-training corpora such as FineWeb-edu use). | |
| Results and every match (seed, commit SHA of each model, scores) are published in | |
| [`{RESULTS_REPO}`](https://huggingface.co/datasets/{RESULTS_REPO}). | |
| """ | |
| with gr.Blocks(title="LM Tetris Arena") as demo: | |
| gr.Markdown(INTRO) | |
| if not STORE.persistent: | |
| gr.Markdown("⚠️ **Results are not being saved**: add an `HF_TOKEN` secret with write access to the results dataset.") | |
| with gr.Tabs(): | |
| with gr.Tab("⚔️ Match"): | |
| with gr.Row(): | |
| with gr.Column(scale=3): | |
| models = gr.Dropdown( | |
| choices=SUGGESTED, value=DEFAULT_MODELS, multiselect=True, allow_custom_value=True, | |
| max_choices=MAX_MODELS, label=f"Language models (1–{MAX_MODELS})", | |
| info="Pick from the list or type any Hub repo id (owner/name) and press Enter.", | |
| ) | |
| baselines = gr.CheckboxGroup(BASELINE_CHOICES, value=[RANDOM_ID], label="Baselines (optional, also rated)") | |
| with gr.Column(scale=2): | |
| protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol") | |
| with gr.Row(): | |
| ranked = gr.Checkbox(value=True, label="Ranked (random seed, updates Elo)") | |
| seed = gr.Number(value=42, precision=0, label="Seed (unranked only)") | |
| delay = gr.Slider(0, 0.5, value=0.12, step=0.02, label="Seconds per piece (viewing speed)") | |
| with gr.Row(): | |
| start = gr.Button("▶ Start match", variant="primary") | |
| stop = gr.Button("■ Stop", variant="stop") | |
| status = gr.Markdown(_status("Ready.", "ok")) | |
| boards = gr.HTML(empty_html()) | |
| results = gr.HTML() | |
| with gr.Tab("🏆 Leaderboard"): | |
| lb_protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol") | |
| lb = gr.Dataframe( | |
| value=leaderboard_df("guided"), interactive=False, wrap=True, | |
| datatype=["number", "markdown", "number", "number", "number", "number", "number", "number", "str"], | |
| ) | |
| lb_refresh = gr.Button("↻ Refresh") | |
| with gr.Tab("📖 How it works"): | |
| gr.Markdown(HOW) | |
| match_event = start.click( | |
| run_match, [models, baselines, protocol, ranked, seed, delay], [status, boards, results], concurrency_limit=1, | |
| ) | |
| stop.click(stop_status, status, status, cancels=[match_event]) | |
| match_event.then(leaderboard_df, lb_protocol, lb) | |
| lb_protocol.change(leaderboard_df, lb_protocol, lb) | |
| lb_refresh.click(refresh_leaderboard, lb_protocol, lb) | |
| demo.load(leaderboard_df, lb_protocol, lb) | |
| demo.queue(max_size=32) | |
| if __name__ == "__main__": | |
| demo.launch(css=CSS, theme=gr.themes.Soft(primary_hue="violet"), ssr_mode=False) | |