Spaces:
Running
Running
Season 6 pool: param counts for Rose-Pro, Rose-Medium, DynamicMind-MoE, Er-Tiny/Medium/Large
523b56d verified Download ranked.py from DedeProGames/SLM-Tetris-Arena: direct link, hf CLI and curl.
- Browser
- Download file 12.1 kB
-
https://huggingface.co/spaces/DedeProGames/SLM-Tetris-Arena/resolve/main/ranked.py
- Command line
-
hf download hf://spaces/DedeProGames/SLM-Tetris-Arena/ranked.py
-
curl -L -o ranked.py https://huggingface.co/spaces/DedeProGames/SLM-Tetris-Arena/resolve/main/ranked.py
12.1 kB
| """Ranked play: the arena picks the models at random, runs the match on the server and records Elo. | |
| Players can't choose who plays ranked, so nobody can farm Elo by pairing a model with weak opponents. | |
| A ranked match runs in a background thread: it finishes and counts even if every viewer leaves, so a | |
| match can't be aborted when it's going badly. Only one ranked match runs at a time; others watch it. | |
| """ | |
| from __future__ import annotations | |
| import html | |
| import random | |
| import threading | |
| import time | |
| import uuid | |
| from arena import MAX_PIECES, choose, rank_games | |
| from players import MAX_PARAMS, MIN_PARAMS, ModelRejected, load_player, precheck | |
| from render import arena_html, empty_html, results_html | |
| from tetris import TetrisGame | |
| PIECE_DELAY = 0.12 # seconds per piece, so viewers can follow the match | |
| LOAD_ATTEMPTS = 4 | |
| # Exact parameter counts (tied weights counted once) of the pool models, so the random pick | |
| # can respect the size gap without downloading anything. Models missing here are estimated | |
| # from their Hub metadata at startup, and every count is corrected after the model loads. | |
| KNOWN_PARAMS = { | |
| 'DedeProGames/Overaddicted-500K': 492_192, | |
| 'fromziro/Er-Tiny-1.3M': 1_332_744, | |
| 'AxiomicLabs/GPT-S-1.4M': 1_426_176, | |
| 'BananaMind/BananaMind-2.1-Pico-Preview': 1_480_516, | |
| 'SupraLabs/SupraNeo-4M': 4_070_240, | |
| 'SLM-Archive/LowOnMind-5M': 4_920_384, | |
| 'veyra-ai/Veyra2-Blueberry-5M-Base': 4_984_192, | |
| 'DedeProGames/Wisp-5M': 5_115_456, | |
| 'AxiomicLabs/GPT-S2-5M': 5_384_258, | |
| 'Novi-AI/Novi-Micro-Base': 5_656_240, | |
| 'DedeProGames/DynamicMind-Mini': 8_884_992, | |
| 'BananaMind/BananaMind-2-Nano': 9_968_128, | |
| 'DedeBckp/BackKiyo-10M': 9_976_832, | |
| 'fromziro/Er-Medium-12.5M': 12_497_520, | |
| 'DedeProGames/Wisp-15M': 15_531_840, | |
| 'DedeProGames/GPT-U-20M': 20_453_760, | |
| 'SupraLabs/Supra2-Medium-Base': 25_371_008, | |
| 'DedeProGames/DynamicMind-MoE': 30_150_912, | |
| 'veyra-ai/Veyra2-Mango-30M-Base': 30_683_520, | |
| 'fromziro/Er-Large-30M': 31_944_632, | |
| 'GODELEV/Rose-Mini': 49_443_074, | |
| 'veyra-ai/Veyra2-Apricot-50M-Base': 49_303_040, | |
| 'BananaMind/BananaMind-2-Medium': 49_559_552, | |
| 'DedeProGames/Kiyo-65M': 64_994_816, | |
| 'opencerebral/Boris-1.3-75M': 77_431_680, | |
| 'GODELEV/Rose-Medium': 97_820_162, | |
| 'SupraLabs/Supra2-100M-Base': 100_684_032, | |
| 'openai-community/gpt2': 124_439_808, | |
| 'HuggingFaceTB/SmolLM-135M': 134_515_008, | |
| 'HuggingFaceTB/SmolLM2-135M': 134_515_008, | |
| 'AxiomicLabs/GPT-X2.5-135M': 135_032_256, | |
| 'BananaMind/BananaMind-2-Pro': 138_971_520, | |
| 'GODELEV/Rose-Pro': 151_274_114, | |
| 'DedeProGames/Kiyo-230M-Preview': 229_688_064, | |
| } | |
| class Pool: | |
| """Models eligible for ranked play, with their parameter counts.""" | |
| def __init__(self, model_ids, gap: int, max_players: int, large_from: int | None = None): | |
| self.ids = list(dict.fromkeys(model_ids)) | |
| self.gap = gap | |
| self.max_players = max_players | |
| self.large_from = large_from # models this size or bigger can all play each other | |
| self.lock = threading.Lock() | |
| self.sizes = {m: KNOWN_PARAMS[m] for m in self.ids if m in KNOWN_PARAMS} | |
| self.broken: dict[str, str] = {} # models that failed this session -> reason | |
| missing = [m for m in self.ids if m not in self.sizes] | |
| if missing: | |
| threading.Thread(target=self._estimate, args=(missing,), daemon=True).start() | |
| def _estimate(self, ids): | |
| for m in ids: | |
| try: | |
| est = precheck(m)["est_params"] | |
| except Exception as e: | |
| self.mark_broken(m, str(e)) | |
| continue | |
| with self.lock: | |
| self.sizes.setdefault(m, est) | |
| def set_exact(self, model_id, n_params): | |
| with self.lock: | |
| self.sizes[model_id] = n_params | |
| def mark_broken(self, model_id, reason): | |
| with self.lock: | |
| self.broken[model_id] = reason[:300] | |
| def eligible(self) -> dict: | |
| with self.lock: | |
| return {m: p for m, p in self.sizes.items() | |
| if m not in self.broken and MIN_PARAMS <= p <= MAX_PARAMS} | |
| def fits(self, sizes) -> bool: | |
| """A ranked group is fair when all sizes fit within `gap`, or when every model is large. | |
| With no gap (0/None) any models can meet: Elo already weighs each win by the opponent's rating.""" | |
| if not self.gap: | |
| return True | |
| if self.large_from is not None and min(sizes) >= self.large_from: | |
| return True | |
| return max(sizes) - min(sizes) <= self.gap | |
| def pick(self, games_played: dict, rng) -> list: | |
| """Random group of up to `max_players` models that `fits` (any sizes when there is no gap; otherwise all sizes | |
| in one `gap`-wide window, or, for a large anchor, any mix of large models). | |
| Models with fewer ranked games are more likely to be drawn, so every model gets played.""" | |
| sizes = self.eligible() | |
| ids = sorted(sizes) | |
| anchors = [m for m in ids if any(o != m and self.fits([sizes[o], sizes[m]]) for o in ids)] | |
| if not anchors: | |
| return [] | |
| a = rng.choices(anchors, [1.0 / (1 + games_played.get(m, 0)) for m in anchors])[0] | |
| if not self.gap: | |
| windows = [[m for m in ids if m != a]] # no size limit: anyone can be drawn | |
| else: | |
| windows = self._windows(a, sizes, ids) | |
| others = list(rng.choice(windows)) | |
| group = [a] | |
| while others and len(group) < self.max_players: # weighted draw without replacement | |
| o = rng.choices(others, [1.0 / (1 + games_played.get(m, 0)) for m in others])[0] | |
| others.remove(o) | |
| group.append(o) | |
| return group | |
| def _windows(self, a, sizes, ids): | |
| """Candidate opponent sets for anchor `a`: every `gap`-wide size window containing it (+ all large models).""" | |
| windows = [] | |
| for s in sorted({sizes[m] for m in ids if sizes[a] - self.gap <= sizes[m] <= sizes[a]}): | |
| members = [m for m in ids if m != a and s <= sizes[m] <= s + self.gap] | |
| if members: | |
| windows.append(members) | |
| if self.large_from is not None and sizes[a] >= self.large_from: | |
| members = [m for m in ids if m != a and sizes[m] >= self.large_from] | |
| if members: | |
| windows.append(members) | |
| return windows | |
| class RankedMatch: | |
| """State of one ranked match, shared by the worker thread and every viewer.""" | |
| def __init__(self, protocol: str): | |
| self.id = uuid.uuid4().hex[:8] | |
| self.protocol = protocol | |
| self.lock = threading.Lock() | |
| self.version = 0 | |
| self.status = f"⏳ Ranked · {protocol} · picking models…" | |
| self.boards = empty_html("Picking models at random…") | |
| self.results = "" | |
| self.done = False | |
| def update(self, status=None, boards=None, results=None, done=None): | |
| with self.lock: | |
| if status is not None: | |
| self.status = status | |
| if boards is not None: | |
| self.boards = boards | |
| if results is not None: | |
| self.results = results | |
| if done is not None: | |
| self.done = done | |
| self.version += 1 | |
| def snapshot(self): | |
| with self.lock: | |
| return self.version, self.status, self.boards, self.results, self.done | |
| class RankedRunner: | |
| def __init__(self, pool: Pool, store): | |
| self.pool = pool | |
| self.store = store | |
| self.lock = threading.Lock() | |
| self.current: RankedMatch | None = None | |
| def start_or_join(self, protocol: str): | |
| """Returns (match, started). Joins the running match instead of starting a second one.""" | |
| with self.lock: | |
| if self.current is not None and not self.current.done: | |
| return self.current, False | |
| match = RankedMatch(protocol) | |
| self.current = match | |
| threading.Thread(target=self._run, args=(match,), daemon=True, name=f"ranked-{match.id}").start() | |
| return match, True | |
| def _run(self, m: RankedMatch): | |
| try: | |
| self._play(m) | |
| except Exception as e: | |
| m.update(status=f"⛔ Ranked match cancelled: {str(e)[:300]} Nothing was recorded.", done=True) | |
| def _load_group(self, m: RankedMatch, rng): | |
| played = {e["model"]: e["games"] for e in self.store.rows(m.protocol)} | |
| skipped = [] | |
| for _ in range(LOAD_ATTEMPTS): | |
| ids = self.pool.pick(played, rng) | |
| if len(ids) < 2: | |
| raise RuntimeError("the pool has no two models of similar size.") | |
| players = [] | |
| for i, model_id in enumerate(ids, 1): | |
| m.update(status=f"⏳ Ranked · {m.protocol} · loading `{model_id}` ({i}/{len(ids)})… first load downloads the weights.", | |
| boards=empty_html(f"Picked at random: {html.escape(', '.join(ids))}<br>Loading {i}/{len(ids)}…")) | |
| try: | |
| meta = precheck(model_id) | |
| player = load_player(meta["id"], meta) | |
| except ModelRejected as e: | |
| self.pool.mark_broken(model_id, str(e)) | |
| skipped.append(model_id) | |
| continue | |
| self.pool.set_exact(model_id, player.n_params) | |
| players.append(player) | |
| if len(players) >= 2 and self.pool.fits([p.n_params for p in players]): | |
| return players, skipped | |
| raise RuntimeError("could not load two models of similar size.") | |
| def _play(self, m: RankedMatch): | |
| rng = random.SystemRandom() | |
| players, skipped = self._load_group(m, rng) | |
| seed = rng.randrange(1, 10**9) | |
| games = [TetrisGame(seed) for _ in players] | |
| head = f"Ranked · seed {seed} · {m.protocol}" | |
| m.update(status=f"⏳ {head} · scoring the first moves…", boards=arena_html(games, players)) | |
| last = time.time() | |
| while True: | |
| active = [(g, p) for g, p in zip(games, players) if g.alive and g.pieces < MAX_PIECES] | |
| if not active: | |
| break | |
| for g, p in active: | |
| try: | |
| choose(g, p, m.protocol, seed) | |
| except Exception as e: | |
| self.pool.mark_broken(p.model_id, str(e)) | |
| raise RuntimeError(f"`{p.model_id}` crashed during play ({type(e).__name__}: {str(e)[:150]}).") | |
| elapsed = time.time() - last | |
| if elapsed < PIECE_DELAY: | |
| time.sleep(PIECE_DELAY - elapsed) | |
| last = time.time() | |
| n = max(g.pieces for g in games) | |
| alive = sum(g.alive for g in games) | |
| m.update(status=f"⏳ {head} · piece {n}/{MAX_PIECES} · {alive} still playing", boards=arena_html(games, players)) | |
| ranks = rank_games(games) | |
| order = sorted(range(len(players)), key=lambda i: ranks[i]) | |
| record = self.store.record(m.protocol, seed, players, games) | |
| elos = [(p["elo_before"], p["elo_after"]) for p in record["players"]] | |
| note = self._note() | |
| if skipped: | |
| note += f" Skipped (failed to load): {html.escape(', '.join(skipped))}." | |
| m.update(status=f"✅ Ranked match finished · seed {seed} · {m.protocol}.", | |
| boards=arena_html(games, players, ranks, elos), | |
| results=results_html(order, ranks, players, games, elos, note), done=True) | |
| def _note(self): | |
| s = self.store | |
| if s.persistent and not s.save_error: | |
| note = (f'Elo updated and saved to the public leaderboard ' | |
| f'(<a href="https://huggingface.co/datasets/{s.repo_id}" target="_blank">{s.repo_id}</a>).') | |
| elif s.persistent: | |
| note = f"⚠️ Elo updated in memory, but {html.escape(s.save_error)}." | |
| else: | |
| note = "⚠️ Elo updated in memory only: the Space has no <code>HF_TOKEN</code> secret, so results are not saved." | |
| return note + f" Ranking: score, then lines, then pieces survived. ✓ = still alive at the {MAX_PIECES}-piece cap." | |