File size: 21,769 Bytes
e3a31c4
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
84e622e
e3a31c4
 
 
 
 
 
84e622e
 
 
 
e3a31c4
84e622e
e3a31c4
 
3760590
e3a31c4
 
b2850a9
e3a31c4
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
"""SLM Tetris Arena — decoder-only LMs play Tetris zero-shot and earn Elo."""
import os

# Keep the write token out of the environment before any model code runs:
# only the results store receives it (custom model code runs in this process).
_TOKEN = os.environ.pop("HF_TOKEN", None) or os.environ.pop("HUGGING_FACE_HUB_TOKEN", None)
os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1")

import html
import inspect
import random
import time

import gradio as gr
import torch

from arena import MAX_PIECES, SEASON, ResultsStore, choose, rank_games
from leaderboard import LB_CSS, leaderboard_html
from leaderboard import fmt_params as short_params
from players import (BASELINES, MAX_PARAMS, MIN_PARAMS, ORACLE_ID, PROMPTS, RANDOM_ID, ModelRejected, OracleReaderPlayer,
                     RandomPlayer, fmt_params, load_player, precheck)
from ranked import Pool, RankedRunner
from render import CSS, arena_html, empty_html, results_html
from tetris import TetrisGame

# cpu-basic Spaces have 2 vCPUs; os.cpu_count() reports the host, which oversubscribes threads
torch.set_num_threads(int(os.environ.get("TORCH_THREADS", 2)))

RESULTS_REPO = os.environ.get("RESULTS_REPO", "DedeProGames/lm-tetris-arena-results")
MAX_MODELS = int(os.environ.get("MAX_MODELS", 4))
# Optional size limit for the random ranked pick (0 = none). Without it every model can meet every other, so all
# ratings sit on one comparable scale; Elo already weighs each win by the opponent's rating, so beating a much
# weaker model earns almost nothing once ratings have settled.
MAX_PARAM_GAP = int(os.environ.get("MAX_PARAM_GAP", 0))
# ...with a gap, models this size or bigger can still all play each other
LARGE_FROM = int(os.environ.get("LARGE_FROM", 100_000_000))
if MAX_PARAM_GAP:
    SIZE_RULE = (f"all within ±{short_params(MAX_PARAM_GAP)} parameters of each other "
                 f"(models with {short_params(LARGE_FROM)}+ parameters can all play each other)")
else:
    SIZE_RULE = "of any size (Elo weighs every win by the opponent's rating, so all models share one scale)"

SUGGESTED = [
    'AxiomicLabs/GPT-S-1.4M',
    'AxiomicLabs/GPT-S2-5M',
    'AxiomicLabs/GPT-X2.5-135M',
    'BananaMind/BananaMind-2-Medium',
    'BananaMind/BananaMind-2-Nano',
    'BananaMind/BananaMind-2-Pro',
    'BananaMind/BananaMind-2.1-Pico-Preview',
    'DedeBckp/BackKiyo-10M',
    'DedeProGames/DynamicMind-Mini',
    'DedeProGames/DynamicMind-MoE',
    'DedeProGames/GPT-U-20M',
    'DedeProGames/Kiyo-230M-Preview',
    'DedeProGames/Kiyo-65M',
    'DedeProGames/Overaddicted-500K',
    'DedeProGames/Wisp-15M',
    'DedeProGames/Wisp-5M',
    'fromziro/Er-Large-30M',
    'fromziro/Er-Medium-12.5M',
    'fromziro/Er-Tiny-1.3M',
    'GODELEV/Rose-Medium',
    'GODELEV/Rose-Mini',
    'GODELEV/Rose-Pro',
    'HuggingFaceTB/SmolLM-135M',
    'HuggingFaceTB/SmolLM2-135M',
    'Novi-AI/Novi-Micro-Base',
    'openai-community/gpt2',
    'opencerebral/Boris-1.3-75M',
    'SLM-Archive/LowOnMind-5M',
    'SupraLabs/Supra2-100M-Base',
    'SupraLabs/Supra2-Medium-Base',
    'SupraLabs/SupraNeo-4M',
    'veyra-ai/Veyra2-Apricot-50M-Base',
    'veyra-ai/Veyra2-Blueberry-5M-Base',
    'veyra-ai/Veyra2-Mango-30M-Base',
]
DEFAULT_MODELS = ["DedeProGames/Kiyo-65M", "BananaMind/BananaMind-2-Medium", "HuggingFaceTB/SmolLM2-135M", "SupraLabs/Supra2-Medium-Base"]
# The leaderboard only keeps pool models: removing a model from SUGGESTED also removes it from the leaderboard.
STORE = ResultsStore(RESULTS_REPO, _TOKEN, min_params=MIN_PARAMS, allowed=SUGGESTED)
# Ranked: the arena picks the players at random from the suggested models (nobody chooses who plays)
POOL = Pool(SUGGESTED, MAX_PARAM_GAP, MAX_MODELS, LARGE_FROM)
RANKED = RankedRunner(POOL, STORE)
PROTOCOL_CHOICES = [("Guided: the rules are in the prompt", "guided"), ("Blind: no rules, only pre-training knowledge", "blind")]
BASELINE_CHOICES = [(label, key) for key, label in BASELINES.items()]

# ---- Look & feel: BananaMind SLM Leaderboard palette and type (dark + light), light-blue accent ----
_D = dict(bg="#0b0e0d", surface="#111613", surface2="#191e1b", text="#f0f1ec", muted="#929b93", line="#29312c", accent="#4d9fff")
_L = dict(bg="#f4f5f1", surface="#ffffff", surface2="#edf0e9", text="#17221b", muted="#626e64", line="#d7ded5", accent="#1b64c4")
_CHECK = ("url(\"data:image/svg+xml,%3csvg viewBox='0 0 16 16' fill='%2307182e' xmlns='http://www.w3.org/2000/svg'%3e"
          "%3cpath d='M12.207 4.793a1 1 0 010 1.414l-5 5a1 1 0 01-1.414 0l-2-2a1 1 0 011.414-1.414L6.5 9.086l4.293-4.293a1 1 0 011.414 0z'/%3e%3c/svg%3e\")")


_THEME_KEYS = set(inspect.signature(gr.themes.Base.set).parameters)


def _both(**pairs):
    """name=(light, dark) -> theme kwargs for both modes (skips variables this Gradio version lacks)."""
    out = {}
    for k, (light, dark) in pairs.items():
        if k in _THEME_KEYS:
            out[k] = light
        if k + "_dark" in _THEME_KEYS:
            out[k + "_dark"] = dark
    return out


THEME = gr.themes.Base(
    primary_hue=gr.themes.colors.blue, secondary_hue=gr.themes.colors.blue, neutral_hue=gr.themes.colors.stone,
    font=[gr.themes.GoogleFont("DM Sans"), "Arial", "sans-serif"], font_mono=["ui-monospace", "SFMono-Regular", "monospace"],
).set(
    block_border_width="1px", block_radius="13px", block_label_border_width="0px", block_title_text_weight="500",
    input_radius="8px", checkbox_check=_CHECK,
    **_both(
        body_background_fill=(_L["bg"], _D["bg"]), body_text_color=(_L["text"], _D["text"]),
        body_text_color_subdued=(_L["muted"], _D["muted"]),
        background_fill_primary=(_L["surface"], _D["surface"]), background_fill_secondary=(_L["surface2"], _D["surface2"]),
        block_background_fill=(_L["surface"], _D["surface"]), block_border_color=(_L["line"], _D["line"]),
        block_shadow=("none", "none"), block_label_background_fill=("transparent", "transparent"),
        block_label_text_color=(_L["muted"], _D["muted"]), block_label_shadow=("none", "none"),
        block_title_background_fill=("transparent", "transparent"), block_title_text_color=(_L["muted"], _D["muted"]),
        block_info_text_color=(_L["muted"], _D["muted"]),
        border_color_primary=(_L["line"], _D["line"]), border_color_accent=("#4d9fff", "#4d9fff"),
        color_accent=("#4d9fff", "#4d9fff"), color_accent_soft=("#4d9fff26", "#4d9fff26"),
        input_background_fill=(_L["surface2"], _D["surface2"]), input_border_color=(_L["line"], _D["line"]),
        input_border_color_focus=("#4d9fff", "#4d9fff"),
        button_primary_background_fill=("#4d9fff", "#4d9fff"), button_primary_background_fill_hover=("#74b4ff", "#74b4ff"),
        button_primary_text_color=("#07182e", "#07182e"), button_primary_border_color=("#4d9fff", "#4d9fff"),
        button_secondary_background_fill=(_L["surface2"], _D["surface2"]),
        button_secondary_background_fill_hover=(_L["line"], _D["line"]),
        button_secondary_text_color=(_L["text"], _D["text"]), button_secondary_border_color=(_L["line"], _D["line"]),
        checkbox_background_color_selected=("#4d9fff", "#4d9fff"), checkbox_border_color_selected=("#4d9fff", "#4d9fff"),
        checkbox_label_background_fill=(_L["surface2"], _D["surface2"]),
        checkbox_label_background_fill_selected=(_L["surface2"], _D["surface2"]),
        checkbox_label_border_color=(_L["line"], _D["line"]), checkbox_label_border_color_selected=("#4d9fff", "#4d9fff"),
        checkbox_label_text_color_selected=(_L["text"], _D["text"]),
        slider_color=("#4d9fff", "#4d9fff"), loader_color=("#4d9fff", "#4d9fff"),
        link_text_color=(_L["accent"], _D["accent"]), link_text_color_hover=(_L["accent"], _D["accent"]),
        panel_background_fill=(_L["surface"], _D["surface"]), panel_border_color=(_L["line"], _D["line"]),
        table_border_color=(_L["line"], _D["line"]), code_background_fill=(_L["surface2"], _D["surface2"]),
    ),
)

APP_CSS = """
@import url('https://fonts.googleapis.com/css2?family=DM+Sans:wght@400;500;600;700&family=Space+Grotesk:wght@400;500;600;700&family=Press+Start+2P&display=swap');
.gradio-container{font-family:'DM Sans',Arial,sans-serif!important}
.gradio-container h1,.gradio-container h2,.gradio-container h3{font-family:'Space Grotesk',Arial,sans-serif!important;letter-spacing:-.4px}
.ah-heading{display:flex;justify-content:space-between;align-items:center;gap:20px;padding:10px 0 6px}
.ah-eyebrow{font:11px/1.5 monospace!important;letter-spacing:1.9px;color:var(--body-text-color-subdued)!important;margin:0 0 10px!important}
.ah-heading h1{font:500 clamp(32px,4vw,48px)/1.2 'Space Grotesk',Arial,sans-serif!important;letter-spacing:-2px!important;margin:0!important;
  color:var(--body-text-color)!important}
.ah-accent{color:#1b64c4}.dark .ah-accent{color:#4d9fff}
.ah-intro{margin:10px 0 0!important;font-size:15px!important;color:var(--body-text-color-subdued)!important;max-width:860px}
.ah-version{font:11px monospace;letter-spacing:1px;color:var(--body-text-color-subdued);display:flex;align-items:center;gap:10px;white-space:nowrap}
.ah-dot{height:6px;width:6px;background:#95c79a;border-radius:50%}
.ah-note{margin:0 0 4px!important;font-size:14px!important;line-height:1.55;color:var(--body-text-color-subdued)!important;max-width:900px}
.ah-note b{color:var(--body-text-color)}
.ah-warn{margin-top:14px;padding:10px 14px;border:1px solid #4d9fff66;border-radius:8px;font-size:13px;color:var(--body-text-color)}
button[role=tab]{font-size:14px!important;color:var(--body-text-color-subdued)!important}
button[role=tab][aria-selected=true]{color:var(--body-text-color)!important;border-color:#4d9fff!important}
@media(max-width:550px){.ah-version{display:none}}
"""


def _status(text, kind="info"):
    icon = {"info": "⏳", "ok": "✅", "err": "⛔", "warn": "⚠️"}[kind]
    return f"{icon} {text}"


def run_match(model_ids, baselines, protocol, seed, delay):
    """Friendly match: any models, any seed, nothing is recorded (Elo only changes in the Ranked tab)."""
    model_ids = [m.strip() for m in (model_ids or []) if m and m.strip()]
    model_ids = list(dict.fromkeys(model_ids))
    baselines = baselines or []
    protocol = protocol or "guided"
    if not model_ids:
        yield _status("Pick at least one language model.", "err"), empty_html(), ""
        return
    if len(model_ids) > MAX_MODELS:
        yield _status(f"At most {MAX_MODELS} language models per match on this CPU.", "err"), empty_html(), ""
        return
    if len(model_ids) + len(baselines) < 2:
        yield _status("A match needs at least 2 players: add another model or a baseline.", "err"), empty_html(), ""
        return

    players = []
    try:
        metas = []
        for m in model_ids:
            yield _status(f"Checking `{m}`…"), empty_html("Checking models…"), ""
            meta = precheck(m)
            if any(x["id"] == meta["id"] for x in metas):
                continue  # same repo typed twice with different casing
            metas.append(meta)
        model_ids = [x["id"] for x in metas]
        for i, (m, meta) in enumerate(zip(model_ids, metas), 1):
            yield _status(f"Loading `{m}` on CPU ({i}/{len(model_ids)})… first load downloads the weights."), empty_html("Loading models…"), ""
            players.append(load_player(m, meta))
    except ModelRejected as e:
        yield _status(str(e), "err"), empty_html("Match cancelled."), ""
        return
    if RANDOM_ID in baselines:
        players.append(RandomPlayer())
    if ORACLE_ID in baselines:
        players.append(OracleReaderPlayer())

    seed = int(seed) if seed else random.SystemRandom().randrange(1, 10**9)
    games = [TetrisGame(seed) for _ in players]
    mode = "friendly"
    yield _status(f"Seed {seed} · {protocol} · {mode}. Scoring the first moves…"), arena_html(games, players), ""

    last = time.time()
    try:
        while True:
            active = [(g, p) for g, p in zip(games, players) if g.alive and g.pieces < MAX_PIECES]
            if not active:
                break
            for g, p in active:
                choose(g, p, protocol, seed)
            elapsed = time.time() - last
            if elapsed < delay:
                time.sleep(delay - elapsed)
            last = time.time()
            n = max(g.pieces for g in games)
            alive = sum(g.alive for g in games)
            yield _status(f"Seed {seed} · {protocol} · {mode} · piece {n}/{MAX_PIECES} · {alive} still playing"), arena_html(games, players), ""
    except ModelRejected as e:
        yield _status(str(e), "err"), arena_html(games, players), ""
        return
    except Exception as e:
        yield _status(f"A model crashed during play: {type(e).__name__}: {str(e)[:200]}", "err"), arena_html(games, players), ""
        return

    ranks = rank_games(games)
    order = sorted(range(len(players)), key=lambda i: ranks[i])
    elos = None
    note = "Friendly match: Elo not changed (only matches in the Ranked tab count)."
    note += f" Ranking: score, then lines, then pieces survived. ✓ = still alive at the {MAX_PIECES}-piece cap."
    results = results_html(order, ranks, players, games, elos, note)
    yield _status(f"Match finished · seed {seed} · {protocol} · {mode}.", "ok"), arena_html(games, players, ranks, elos), results


def ranked_play(protocol):
    """Start a ranked match (models picked at random) or watch the one already running."""
    match, started = RANKED.start_or_join(protocol or "guided")
    joined = "" if started else f" · you joined the ranked match already in progress ({match.protocol})"
    seen = -1
    while True:
        version, status, boards, results, done = match.snapshot()
        if version != seen:
            seen = version
            yield status + joined, boards, results
        if done:
            return
        time.sleep(0.1)


def stop_status(current):
    # only claim a stop when a match was actually running
    if (current or "").startswith("⏳"):
        return _status("Match stopped. Nothing was recorded.", "warn")
    return current


def leaderboard_view(protocol):
    protocol = protocol or "guided"
    entries = [e for e in STORE.rows(protocol) if e["model"] not in BASELINES]
    return leaderboard_html(entries, protocol, MAX_PARAM_GAP, short_params(MAX_PARAM_GAP))


def refresh_leaderboard(protocol):
    STORE.reload()
    return leaderboard_view(protocol)


INTRO = f"""
<div class="ah-heading"><div><p class="ah-eyebrow">SMALL MODELS. ZERO-SHOT TETRIS.</p>
<h1>SLM Tetris Arena<span class="ah-accent">.</span></h1>
<p class="ah-intro">Decoder-only language models ({short_params(MIN_PARAMS)}–{short_params(MAX_PARAMS)} parameters, custom architectures welcome) play Tetris
zero-shot: no fine-tuning, no game data, only what they learned from pre-training on text. Every player gets the same
piece sequence, and ranked matches update a public Elo leaderboard.</p></div>
<span class="ah-version">SEASON {SEASON} <span class="ah-dot"></span></span></div>
"""

HOW = f"""
### How a model plays
For every new piece the game lists all legal placements (rotation × column, hard drop), simulates each one and
describes the outcome in plain English. The model never sees the grid; it judges the descriptions:

```
{PROMPTS['guided'].format(desc='drops the piece into the lowest part of the board, clears one line, creates no new holes, keeps the stack low and leaves the surface flat')}
```

The model's value for a placement is **log P(" good move") − log P(" bad move")** after that prompt. The placement with the
highest value is played; exact ties are broken by a seeded coin that is identical for every player.
Because the value is a difference, a model's general bias towards "good" or "bad" cancels out.

### Protocols (separate leaderboards)
- **Guided**: the first line states the goal ("clear lines, avoid holes, keep the stack low"). Tests reading comprehension.
- **Blind**: `{PROMPTS['blind'].splitlines()[0]}` No rules; the model must already know what is good in Tetris.

### Rules of a match
- Same 7-bag piece sequence for everyone. The game ends at top-out or after {MAX_PIECES} pieces.
- Placement = score (100/300/500/800 for 1/2/3/4 lines), then lines, then pieces survived.
- **Match** tab (friendly): pick any 2+ players (up to {MAX_MODELS} language models of any size, plus optional baselines)
  and the seed. Nothing is recorded.
- **Ranked** tab: press Play and the arena picks up to {MAX_MODELS} models at random from its pool of {len(POOL.ids)} models,
  {SIZE_RULE}, with a random seed. Nobody chooses who plays, so Elo can't be farmed.
  Models with fewer ranked games are more likely to be picked, so every model gets played. The match runs on the server and counts even if
  you close the page; only one ranked match runs at a time, and pressing Play while one is running lets you watch it.
- Elo: K=32, multiplayer (every pair of players counts as a game, scaled by 1/(N−1)).

### Baselines
- **🎲 Random**: every placement ties, so it plays uniformly at random. This is the floor a model should beat.
- **📏 Oracle reader**: reads the same descriptions and ranks them with fixed common sense (holes > lines > height > surface > landing).
  This is roughly the ceiling for a perfect reader of the text.

Baselines can join friendly matches for comparison. They never play ranked and never change anyone's Elo.

### Model requirements
Public, not gated, loads with `AutoModelForCausalLM` + `AutoTokenizer` (PyTorch or safetensors weights), between {fmt_params(MIN_PARAMS)} and {fmt_params(MAX_PARAMS)} parameters.
Models with custom code (`auto_map`) load with `trust_remote_code=True`. That code runs on this Space's CPU, so only
submit repos you trust. Prompts are in English (the language most pre-training corpora such as FineWeb-edu use).

Results and every match (seed, commit SHA of each model, scores) are published in
[`{RESULTS_REPO}`](https://huggingface.co/datasets/{RESULTS_REPO}).
"""

RANKED_INTRO = f"""<p class="ah-note">Press <b>Play</b> and the arena picks up to {MAX_MODELS} language models <b>at random</b>
from its pool of {len(POOL.ids)} models, {SIZE_RULE}, with a random seed.
The result updates the public Elo leaderboard. The match runs on the server: it finishes and counts even if you close the page.
Only one ranked match runs at a time; if one is already running, you watch it.</p>"""

with gr.Blocks(title="SLM Tetris Arena") as demo:
    gr.HTML(INTRO, padding=False)
    if not STORE.persistent:
        gr.HTML('<div class="ah-warn">⚠️ Results are not being saved: add an <code>HF_TOKEN</code> secret with write access '
                'to the results dataset.</div>', padding=False)
    with gr.Tabs():
        with gr.Tab("⚔️ Match"):
            with gr.Row():
                with gr.Column(scale=3):
                    models = gr.Dropdown(
                        choices=SUGGESTED, value=DEFAULT_MODELS, multiselect=True, allow_custom_value=True,
                        max_choices=MAX_MODELS, label=f"Language models (1–{MAX_MODELS})",
                        info="Pick from the list or type any Hub repo id (owner/name) and press Enter.",
                    )
                    baselines = gr.CheckboxGroup(BASELINE_CHOICES, value=[RANDOM_ID], label="Baselines (optional, never rated)")
                with gr.Column(scale=2):
                    protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol")
                    seed = gr.Number(value=42, precision=0, label="Seed (0 = random)",
                                     info="Friendly match: any models, nothing is recorded. Elo only changes in the Ranked tab.")
                    delay = gr.Slider(0, 0.5, value=0.12, step=0.02, label="Seconds per piece (viewing speed)")
            with gr.Row():
                start = gr.Button("▶ Start match", variant="primary")
                stop = gr.Button("■ Stop", variant="secondary")
            status = gr.Markdown(_status("Ready.", "ok"))
            boards = gr.HTML(empty_html())
            results = gr.HTML()
        with gr.Tab("🏅 Ranked"):
            gr.HTML(RANKED_INTRO, padding=False)
            with gr.Row(equal_height=True):
                r_protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol", scale=3)
                r_play = gr.Button("▶ Play ranked match", variant="primary", scale=1)
            r_status = gr.Markdown(_status("Ready. Press Play: the arena picks the models.", "ok"))
            r_boards = gr.HTML(empty_html("Press <b>Play ranked match</b>. The arena picks the models at random."))
            r_results = gr.HTML()
        with gr.Tab("🏆 Leaderboard"):
            lb_protocol = gr.Radio(PROTOCOL_CHOICES, value="guided", label="Protocol")
            lb = gr.HTML(leaderboard_view("guided"), padding=False)
            lb_refresh = gr.Button("↻ Refresh", variant="secondary")
        with gr.Tab("📖 How it works"):
            gr.Markdown(HOW)

    match_event = start.click(
        run_match, [models, baselines, protocol, seed, delay], [status, boards, results], concurrency_limit=1,
    )
    stop.click(stop_status, status, status, cancels=[match_event])
    # viewers only watch; the ranked match itself runs in one background thread
    r_play.click(ranked_play, r_protocol, [r_status, r_boards, r_results], concurrency_limit=16,
                 concurrency_id="ranked").then(leaderboard_view, lb_protocol, lb)
    lb_protocol.change(leaderboard_view, lb_protocol, lb)
    lb_refresh.click(refresh_leaderboard, lb_protocol, lb)
    demo.load(leaderboard_view, lb_protocol, lb)

demo.queue(max_size=32)

if __name__ == "__main__":
    demo.launch(css=APP_CSS + CSS + LB_CSS, theme=THEME, ssr_mode=False)