"""Leaderboard tab: Elo bar chart, table and play-style scatter (server-rendered HTML/SVG).
Visual language follows the BananaMind SLM Leaderboard: org-coloured bars with logos and a rainbow glow for
size-class leaders. The second chart shows play style (survival vs line clears), which carries no size bias.
"""
from __future__ import annotations
import hashlib
import html
import math
MIN_GAMES_GLOW = 2 # a leader needs more than one ranked game, so a single lucky win doesn't glow
# Size classes are relative, not absolute. Two models are neighbours when the bigger one is at most `size_ratio`
# times the smaller one. The ratio is 1.25 (-20%/+25%) up to 30M and widens smoothly (log scale) to 1.5
# (-33%/+50%) from 100M on, so e.g. 50M vs 65M and 100M vs 135M share a class while tiny models stay fine-grained.
RATIO_SMALL, RATIO_LARGE = 1.25, 1.5
RATIO_FROM, RATIO_TO = 30e6, 100e6
def size_ratio(p: float) -> float:
t = min(1.0, max(0.0, math.log10(p / RATIO_FROM) / math.log10(RATIO_TO / RATIO_FROM)))
return RATIO_SMALL + (RATIO_LARGE - RATIO_SMALL) * t
def neighbours(a: float, b: float) -> bool:
"""Symmetric: the allowed ratio is taken at the pair's geometric-mean size."""
return max(a, b) / min(a, b) <= size_ratio(math.sqrt(a * b))
_CDN = "https://cdn-avatars.huggingface.co/v1/production/uploads/"
# owner on the Hub -> (display name, logo, "r, g, b", border colour). Same identities as the BananaMind leaderboard.
_ORGS = [
("BananaMind", "BananaMind", "69ae829a8408eeb0d7dd5491/0-2aeVpWWaWlufYtgkWtK.png", "250, 204, 21", "#facc15"),
("DALabCommunity", "DALabCommunity", "https://www.gravatar.com/avatar/887b2ff821a8d5f70752fd50b05e137", "217, 70, 239", "#d946ef"),
("SupraLabs", "SupraLabs", "697f2832c2c5e4daa93cece7/IQMtz5gg-vLFP7Gn75POT.png", "139, 92, 246", "#a78bfa"),
("openai-community", "OpenAI", "5dd96eb166059660ed1ee413/9NY4jfufqo1uyv8oNXQju.png", "16, 163, 127", "#10a37f"),
("GODELEV", "GODELEV", "67f03a82cb606619f36f9a51/ZkQyscQj9IdQMySbgLcb5.jpeg", "79, 70, 229", "#6366f1"),
("AxiomicLabs", "Axiomic Labs", "67b413df70aa5c739bda9e7a/pGOq2X7y_iLw1VklgfDFl.png", "194, 182, 255", "#c2b6ff"),
("HuggingFaceTB", "Hugging Face", "651e96991b97c9f33d26bde6/e4VK7uW5sTeCYupD0s_ob.png", "255, 157, 0", "#ff9d00"),
("veyra-ai", "veyra-ai", "6857f2cfae68b377f17aff8c/0Tl87LYtzyBEvumEe_QJ1.png", "212, 86, 114", "#d45672"),
("Eclipse-Senpai", "Eclipse-Senpai", "noauth/3Rm4xf1hvlObxbBxyvC6i.png", "6, 182, 212", "#06b6d4"),
("User01110", "User01110", "https://huggingface.co/avatars/93dace33d3ce104114776b02f3646c3b.svg", "168, 85, 247", "#a855f7"),
("AtomixLabs", "AtomixLabs", "64b433c3faa3181a5e98c87c/j2-Xd02dqerocdu-SWqJh.png", "190, 242, 100", "#bef264"),
("ThingAI", "ThingAI", "69e70c6a759e88fab12bde9f/A2pR_uu7ErE7Tbe2UGDnH.png", "180, 83, 9", "#b45309"),
("joelhenwang", "joelhenwang", "https://huggingface.co/avatars/94de3a736fac914944f1b57609e3819a.svg", "229, 231, 235", "#e5e7eb"),
("MultivexAI", "MultivexAI", "64b433c3faa3181a5e98c87c/ZRirYgVxVdxNCeCV_aoHT.png", "0, 240, 255", "#00f0ff"),
("finnianx", "finnianx", "6325c1d65cf955bfbbde74b6/9-sRu_OMmqSAAyeiSv7SO.jpeg", "45, 212, 191", "#2dd4bf"),
("EleutherAI", "EleutherAI", "1614054059123-603481bb60e3dd96631c9095.png", "239, 68, 68", "#ef4444"),
("fromziro", "FromZero", "68657cd96e07b797a219b593/qITdWZiMpLE8Kop68m9OZ.png", "210, 180, 140", "#d2b48c"),
("Harley-ml", "Harley ML", "68657cd96e07b797a219b593/nV8Apsw0hNHBrrHyf3kB7.jpeg", "153, 27, 27", "#991b1b"),
("UniversalComputingResearch", "UCR", "67fc2fb8b34e5f8a2dea939b/99X2TS_XeKtlJSjvVouUE.png", "59, 130, 246", "#3b82f6"),
("BananaMind-Model-Previewers", "BananaMind Model Previewers", "69ae829a8408eeb0d7dd5491/GBfhEbHUsGV3ps4YPLLnA.png", "251, 191, 36", "#fbbf24"),
("appvoid", "appvoid", "62a813dedbb9e28866a91b27/2fknEF_u6StSjp3uUF144.png", "244, 114, 182", "#f472b6"),
("DedeProGames", "DedeProGames", "685ea8ff7b4139b6845ce395/Im--QSnbrnAhHPPhpX8L0.png", "77, 159, 255", "#4d9fff"),
("opencerebral", "OpenCerebral", "689a3f0eec8a724449b85179/Rd2B98EVdHw99gOajD-aV.png", "14, 165, 233", "#0ea5e9"),
("NILKNARFGonzo", "NILKNARFGonzo", "noauth/N_mm9c94sF1JR76d1zNLA.png", "148, 163, 184", "#94a3b8"),
("allura-org", "allura-org", "634262af8d8089ebaefd410e/6zT9gVQI_9HKiW-6T6uXS.jpeg", "251, 113, 133", "#fb7185"),
("FlameF0X", "FlameF0X", "6615494716917dfdc645c44e/GGzgDi_WTW1Ci4CaDJd8I.jpeg", "249, 115, 22", "#f97316"),
("CNWPlayer", "CNWPlayer", "694742331f2408791d8e1472/qAkFOi18U_Yzv9UVnp7Wd.png", "34, 211, 238", "#22d3ee"),
("CodeSoft", "CodeSoft", "645aad59c4acfcf664022df5/BwD8ZMbxrK6h3CzxpwNfA.jpeg", "80, 162, 255", "#50a2ff"),
("DedeBckp", "DedeBckp", "noauth/sEn3rwht_EbEa83Ug8slJ.png", "116, 180, 255", "#74b4ff"),
("bananamind-research-community", "BananaMind Research Community", "69ae829a8408eeb0d7dd5491/POU3vsQeIkN2Lim-Lv0wR.png", "253, 224, 71", "#fde047"),
]
ORGS = {
owner.lower(): {"name": name, "logo": logo if logo.startswith("https://") else _CDN + logo,
"fill": f"rgba({rgb}, 0.70)", "border": border}
for owner, name, logo, rgb, border in _ORGS
}
def org_of(model_id: str) -> dict:
"""Known orgs keep their colours; unknown owners get a stable hue and an initial."""
owner = model_id.split("/")[0]
if owner.lower() in ORGS:
return ORGS[owner.lower()]
hue = int(hashlib.md5(owner.lower().encode()).hexdigest()[:6], 16) % 360
return {"name": owner, "logo": None, "fill": f"hsla({hue}, 70%, 58%, 0.70)", "border": f"hsl({hue}, 70%, 60%)"}
def _esc(s) -> str:
return html.escape(str(s), quote=True)
def fmt_params(p) -> str:
if not p:
return "?"
if p >= 1e9:
return f"{p / 1e9:.2f}".rstrip("0").rstrip(".") + "B"
if p >= 1e6:
return f"{p / 1e6:.1f}".rstrip("0").rstrip(".") + "M"
return f"{p / 1e3:.0f}K"
def expected_score(elo: float) -> float:
"""Bar height: expected score against a 1000-rated model (0-100). 1000 -> 50."""
return 100.0 / (1.0 + 10 ** ((1000.0 - elo) / 400.0))
def _logo(org, cls="lb-logo") -> str:
if org["logo"]:
return f''
return f'{_esc(org["name"][:1].upper())}'
def leaders(entries, gap: int = 0) -> set:
"""Size-class leaders: the best Elo among rated models of similar relative size.
A model glows when (1) it has at least MIN_GAMES_GLOW ranked games, (2) its Elo is above the 1000 start,
(3) at least one other rated model is a size neighbour (see `neighbours`), and
(4) none of those neighbours has a higher Elo. Ties glow together. `gap` is kept for API compatibility."""
out = set()
for e in entries:
p = e.get("params")
if not p or e.get("games", 0) < MIN_GAMES_GLOW or e["elo"] <= 1000:
continue
peers = [o for o in entries if o.get("params") and neighbours(p, o["params"])]
if len(peers) < 2:
continue # no neighbour to compare with
if all(o["elo"] <= e["elo"] for o in peers):
out.add(e["model"])
return out
# ----------------------------------------------------------------------------
# Bar chart + table
# ----------------------------------------------------------------------------
_TICKS = [1400, 1200, 1000, 800, 600]
def _bars(entries, glow):
ticks = "".join(f'{t}' for t in _TICKS)
grid = "".join(f'' for t in _TICKS)
bars = []
for e in entries:
org = org_of(e["model"])
h = expected_score(e["elo"])
name = e["model"].split("/", 1)[-1]
lead = e["model"] in glow
label = (f'{_esc(e["model"])} · Elo {e["elo"]:.0f} · {e["games"]} ranked games'
f'{" · size-class leader" if lead else ""}')
bars.append(
f''
f'{e["elo"]:.0f}'
f'{_logo(org)}{_esc(name)}'
f'{fmt_params(e.get("params"))} params{e["games"]} games'
)
return (f'
{ticks}
'
f'
{grid}
{"".join(bars)}
')
def _table(entries, glow):
rows = []
for i, e in enumerate(entries, 1):
org = org_of(e["model"])
g = max(1, e["games"])
dot = '' if e["model"] in glow else ""
rows.append(
f'
'
# ----------------------------------------------------------------------------
# Play style scatter: survival vs line clears (no size axis, so no size bias)
# ----------------------------------------------------------------------------
def _pearson(a, b):
if len(a) < 3:
return None
ma, mb = sum(a) / len(a), sum(b) / len(b)
sa = math.sqrt(sum((x - ma) ** 2 for x in a))
sb = math.sqrt(sum((y - mb) ** 2 for y in b))
if not sa or not sb:
return None
return sum((x - ma) * (y - mb) for x, y in zip(a, b)) / (sa * sb)
def _median(v):
a = sorted(v)
return (a[(len(a) - 1) // 2] + a[len(a) // 2]) / 2
def _style_scatter(entries):
"""x = average pieces survived per ranked game, y = average lines cleared (sqrt scale), dot size = Elo."""
pts = [dict(e, _pcs=e["total_pieces"] / e["games"], _lines=e["total_lines"] / e["games"])
for e in entries if e.get("games")]
if not pts:
return '
No rated models yet.
', ""
left, top, width, height = 66, 24, 950, 350
pcs = [e["_pcs"] for e in pts]
lines = [e["_lines"] for e in pts]
elos = [e["elo"] for e in pts]
xspan = max(pcs) - min(pcs)
xstep = 5 if xspan <= 40 else 10 if xspan <= 90 else 25 if xspan <= 250 else 50
xmin = max(0, math.floor((min(pcs) - xstep / 2) / xstep) * xstep)
xmax = max(xmin + 2 * xstep, math.ceil((max(pcs) + xstep / 2) / xstep) * xstep)
ymax = max(1.0, max(lines) * 1.15)
lo, hi = min(elos), max(elos)
def x(v):
return left + (v - xmin) / (xmax - xmin) * width
def y(v):
return top + height - math.sqrt(max(0.0, v) / ymax) * height
def radius(elo):
return 4.5 + (7.0 * (elo - lo) / (hi - lo) if hi > lo else 3.5)
axes = []
v = xmin
while v <= xmax + 1e-9:
axes.append(f''
f'{v:g}')
v += xstep
last = 1e9
for t in (0, 0.1, 0.25, 0.5, 1, 2, 3, 4, 6, 8, 10, 15, 20, 30, 50):
if t > ymax:
break
py = y(t)
if last - py < 22:
continue
last = py
axes.append(f''
f'{t:g}')
mid_x, mid_y = x(_median(pcs)), y(_median(lines))
# labels: best Elo first, so the strongest models keep their names when space is tight
boxes, labels = [], {}
for e in sorted(pts, key=lambda m: -m["elo"]):
px, py, r = x(e["_pcs"]), y(e["_lines"]), radius(e["elo"])
name = e["model"].split("/", 1)[-1]
lw = len(name) * 6.3
for dy in (-(r + 5), r + 13, -(r + 21), r + 29):
lx = px - lw - r - 5 if px + lw + r + 9 > left + width else px + r + 5
box = (lx, py + dy - 10, lw, 14)
if box[0] < left or box[1] < top or box[1] + box[3] > top + height:
continue
if any(box[0] < b[0] + b[2] + 5 and box[0] + box[2] + 5 > b[0] and box[1] < b[1] + b[3] + 3 and box[1] + box[3] + 3 > b[1]
for b in boxes):
continue
boxes.append(box)
labels[e["model"]] = f'{_esc(name)}'
break
points = []
for e in sorted(pts, key=lambda m: -radius(m["elo"])): # big dots first, small ones stay visible on top
org = org_of(e["model"])
px, py, r = x(e["_pcs"]), y(e["_lines"]), radius(e["elo"])
title = (f'{e["model"]} · Elo {e["elo"]:.0f} · {e["_pcs"]:.1f} pieces and {e["_lines"]:.2f} lines per game'
f' · {e["games"]} games')
points.append(
f''
f'{_esc(title)}'
f''
)
svg = (
f''
)
r_pcs, r_lines = _pearson(elos, pcs), _pearson(elos, lines)
corr = (f" Correlation with Elo: survival r = {r_pcs:+.2f}, line clears r = {r_lines:+.2f}."
if r_pcs is not None and r_lines is not None else "")
return svg, corr
def leaderboard_html(entries, protocol: str, gap: int, gap_label: str) -> str:
entries = sorted(entries, key=lambda e: -e["elo"])
proto = "Guided" if protocol == "guided" else "Blind"
glow = leaders(entries, gap)
if entries:
views = (
''
''
'
Model Elo
'
f'
{proto} protocol · ranked matches only · higher is better
'
'
'
'
'
f'
{_bars(entries, glow)}
'
f'
{_table(entries, glow)}
'
)
else:
views = (f'
Model Elo
{proto} protocol · ranked matches only
'
'
No ranked matches yet this season. Play a ranked match to put models on the board.
')
orgs = []
for e in entries:
o = org_of(e["model"])
if o["name"] not in [n for n, _ in orgs]:
orgs.append((o["name"], o["border"]))
legend = "".join(f'{_esc(n)}' for n, c in orgs)
style_svg, style_corr = _style_scatter(entries)
return (
'
'
f'{views}'
f''
'
Play Style
'
'
How each model plays in ranked games · further right survives longer, higher clears more lines
'
'
Survives long and clears lines'
'Bigger dot = higher Elo
'
f'
{legend}
'
f'
{style_svg}
'
f''
'
Averages over each model\'s ranked games in this protocol. Shading starts at the median of both axes.'
f'{style_corr}