Jean Ogier du Terrail commited on
Commit
7ee716d
·
1 Parent(s): d653fbd

Leaderboard: 95% CIs from the report's per-sample paired bootstrap (whiskers on bars, scatters and overview; CIs and paired deltas in tooltips)

Browse files
README.md CHANGED
@@ -103,6 +103,11 @@ causal-variant cohort and the tie-aware recall, so the Space and the article can
103
  diverge. The generator prints its checks (the S_bal of the four sizes, the cohort size
104
  and the headline recalls) so a mismatch with the manuscript is visible at build time.
105
 
 
 
 
 
 
106
  `data/abi5-locus.json` is a separate extract of the original Botanic1-XL 4,096 bp
107
  scoring output for ABI5: 5,948 SNPs in a 100 kbp locus. Scores are the absolute
108
  values of the source `score` column. The validated SNP at TAIR10 Chr2:15,205,931
 
103
  diverge. The generator prints its checks (the S_bal of the four sizes, the cohort size
104
  and the headline recalls) so a mismatch with the manuscript is visible at build time.
105
 
106
+ `data/leaderboard.json` also carries 95% confidence intervals (`s_bal_ci`, `families_ci`,
107
+ `metrics[].ci`, `vs_plantcad2l`) from the report's per-sample paired bootstrap (20,000
108
+ replicates, `figures/data/fig3_rerun/rerun_ci.json` in the report repository), added by
109
+ `scripts/add_bootstrap_ci.py` and centred on the reported scores as in the report.
110
+
111
  `data/abi5-locus.json` is a separate extract of the original Botanic1-XL 4,096 bp
112
  scoring output for ABI5: 5,948 SNPs in a 100 kbp locus. Scores are the absolute
113
  values of the source `score` column. The validated SNP at TAIR10 Chr2:15,205,931
assets/figure1.js CHANGED
@@ -22,7 +22,7 @@
22
  const M = lb.models, F = lb.families;
23
  const colour = (m) => m.ours ? css('--ours') : m.id.startsWith('Botanic0') ? css('--prior') : css('--other');
24
  const famTable = (m) => '<table>' + F.map(f => `<tr><td>${f.name}</td><td class="n">${fmt(m.families[f.key])}</td></tr>`).join('') + '</table>';
25
- const modelTip = (m) => `<b>${esc(m.name)}</b> · ${esc(m.org)}<br>${m.params_str} parameters · ${m.objective}, ${m.domain} pre-training${m.bp_equivalent ? ' · ' + esc(m.bp_equivalent) + ' bp seen' : ''}<br>downstream task performance <b>${fmt(m.s_bal, 4)}</b>`;
26
 
27
  // ------------------------------------------------------------------ intro card
28
  const intro = document.getElementById('intro-chart');
@@ -40,6 +40,7 @@
40
  const left = new Set(['Evo2-40b', 'CARBON-8B', 'botanic1-S']);
41
  const dy = { 'botanic1-S': -20, 'botanic1-XL': -12, 'PlantCAD2-L': 18, 'PlantCaduceus': 14, GPN: -10 };
42
  const pt = svg.selectAll('g.pt').data(rows).join('g').attr('class', 'pt').attr('transform', d => `translate(${x(Math.min(Math.max(d.params, 5e7), 5e10))},${y(Math.max(d.s_bal, 0.55))})`);
 
43
  pt.append('circle').attr('r', d => d.ours ? 14 : 11).attr('fill', '#fff').attr('stroke', d => d.ours ? css('--ours') : css('--rule-2')).attr('stroke-width', d => d.ours ? 2.5 : 1);
44
  pt.append('image').attr('href', d => 'assets/logos/' + d.logo).attr('x', d => d.ours ? -9 : -7).attr('y', d => d.ours ? -9 : -7).attr('width', d => d.ours ? 18 : 14).attr('height', d => d.ours ? 18 : 14).attr('preserveAspectRatio', 'xMidYMid meet');
45
  pt.filter(d => labelled.has(d.id)).append('text').attr('class', 'lbl-2').attr('x', d => left.has(d.id) ? -15 : (d.ours ? 18 : 15)).attr('text-anchor', d => left.has(d.id) ? 'end' : 'start').attr('y', d => 4 + (dy[d.id] || 0)).text(d => d.name);
@@ -193,7 +194,7 @@
193
  }
194
 
195
  // ------------------------------------------------------------------ (d) ranking
196
- const pd = panel('d', 'Model ranking', 'downstream task performance · hover a bar · click for the full leaderboard');
197
  {
198
  const rows = M.filter(m => m.in_fig1).sort((a, b) => b.s_bal - a.s_bal);
199
  const W = 1100, H = 330, L = 44, R = 12, T = 56, B = 92;
@@ -207,6 +208,7 @@
207
  const bw = x.bandwidth();
208
  g.append('rect').attr('class', 'bar').attr('x', d => x(d.id)).attr('width', bw).attr('y', y(0.5)).attr('height', 0).attr('rx', 2)
209
  .attr('fill', d => colour(d)).attr('stroke', d => d.ours ? 'none' : css('--rule-2')).attr('stroke-width', 1);
 
210
  const chip = g.append('g').attr('transform', d => `translate(${x(d.id) + bw / 2},${y(d.s_bal) - 16})`);
211
  chip.append('circle').attr('class', 'chip').attr('r', 11);
212
  chip.append('image').attr('href', d => 'assets/logos/' + d.logo).attr('x', -7).attr('y', -7).attr('width', 14).attr('height', 14).attr('preserveAspectRatio', 'xMidYMid meet');
 
22
  const M = lb.models, F = lb.families;
23
  const colour = (m) => m.ours ? css('--ours') : m.id.startsWith('Botanic0') ? css('--prior') : css('--other');
24
  const famTable = (m) => '<table>' + F.map(f => `<tr><td>${f.name}</td><td class="n">${fmt(m.families[f.key])}</td></tr>`).join('') + '</table>';
25
+ const modelTip = (m) => `<b>${esc(m.name)}</b> · ${esc(m.org)}<br>${m.params_str} parameters · ${m.objective}, ${m.domain} pre-training${m.bp_equivalent ? ' · ' + esc(m.bp_equivalent) + ' bp seen' : ''}<br>downstream task performance <b>${fmt(m.s_bal, 4)}</b>${m.s_bal_ci ? ` <span class="ci">95% CI [${fmt(m.s_bal_ci[0], 4)}, ${fmt(m.s_bal_ci[1], 4)}]</span>` : ''}`;
26
 
27
  // ------------------------------------------------------------------ intro card
28
  const intro = document.getElementById('intro-chart');
 
40
  const left = new Set(['Evo2-40b', 'CARBON-8B', 'botanic1-S']);
41
  const dy = { 'botanic1-S': -20, 'botanic1-XL': -12, 'PlantCAD2-L': 18, 'PlantCaduceus': 14, GPN: -10 };
42
  const pt = svg.selectAll('g.pt').data(rows).join('g').attr('class', 'pt').attr('transform', d => `translate(${x(Math.min(Math.max(d.params, 5e7), 5e10))},${y(Math.max(d.s_bal, 0.55))})`);
43
+ pt.filter(d => d.s_bal_ci).append('line').attr('class', 'whisker').attr('x1', 0).attr('x2', 0).attr('y1', d => y(Math.max(d.s_bal_ci[0], 0.55)) - y(Math.max(d.s_bal, 0.55))).attr('y2', d => y(Math.max(d.s_bal_ci[1], 0.55)) - y(Math.max(d.s_bal, 0.55)));
44
  pt.append('circle').attr('r', d => d.ours ? 14 : 11).attr('fill', '#fff').attr('stroke', d => d.ours ? css('--ours') : css('--rule-2')).attr('stroke-width', d => d.ours ? 2.5 : 1);
45
  pt.append('image').attr('href', d => 'assets/logos/' + d.logo).attr('x', d => d.ours ? -9 : -7).attr('y', d => d.ours ? -9 : -7).attr('width', d => d.ours ? 18 : 14).attr('height', d => d.ours ? 18 : 14).attr('preserveAspectRatio', 'xMidYMid meet');
46
  pt.filter(d => labelled.has(d.id)).append('text').attr('class', 'lbl-2').attr('x', d => left.has(d.id) ? -15 : (d.ours ? 18 : 15)).attr('text-anchor', d => left.has(d.id) ? 'end' : 'start').attr('y', d => 4 + (dy[d.id] || 0)).text(d => d.name);
 
194
  }
195
 
196
  // ------------------------------------------------------------------ (d) ranking
197
+ const pd = panel('d', 'Model ranking', 'downstream task performance · whiskers are 95% CIs · hover a bar · click for the full leaderboard');
198
  {
199
  const rows = M.filter(m => m.in_fig1).sort((a, b) => b.s_bal - a.s_bal);
200
  const W = 1100, H = 330, L = 44, R = 12, T = 56, B = 92;
 
208
  const bw = x.bandwidth();
209
  g.append('rect').attr('class', 'bar').attr('x', d => x(d.id)).attr('width', bw).attr('y', y(0.5)).attr('height', 0).attr('rx', 2)
210
  .attr('fill', d => colour(d)).attr('stroke', d => d.ours ? 'none' : css('--rule-2')).attr('stroke-width', 1);
211
+ g.filter(d => d.s_bal_ci).append('line').attr('class', 'whisker').attr('x1', d => x(d.id) + bw / 2).attr('x2', d => x(d.id) + bw / 2).attr('y1', d => y(d.s_bal_ci[0])).attr('y2', d => y(d.s_bal_ci[1]));
212
  const chip = g.append('g').attr('transform', d => `translate(${x(d.id) + bw / 2},${y(d.s_bal) - 16})`);
213
  chip.append('circle').attr('class', 'chip').attr('r', 11);
214
  chip.append('image').attr('href', d => 'assets/logos/' + d.logo).attr('x', -7).attr('y', -7).attr('width', 14).attr('height', 14).attr('preserveAspectRatio', 'xMidYMid meet');
assets/overview.js CHANGED
@@ -73,7 +73,7 @@
73
  }
74
 
75
  function benchmark(host, data) {
76
- const f = frame(host, '<span>Selected references</span><span class="plot-unit">S<sub>bal</sub><sup>test</sup> · axis starts at 0.5</span>');
77
  const references = new Set(['PlantCAD2-L', 'GPN', 'CARBON-8B', 'NTv3-pre', 'Evo2-7b', 'AgroNT']);
78
  let selected = 'botanic1-XL';
79
  responsive(host, W => {
@@ -94,10 +94,11 @@
94
  marks.append('rect').attr('class', 'benchmark-bar').attr('x', L).attr('y', -6).attr('width', d => x(d.s_bal) - L).attr('height', 12).attr('rx', 2)
95
  .attr('fill', d => d.ours ? 'var(--lime)' : 'var(--plot-reference)')
96
  .attr('stroke', d => d.ours ? 'var(--accent)' : 'var(--rule-2)').attr('stroke-width', .6);
97
- marks.append('text').attr('class', 'plot-value').attr('x', d => x(d.s_bal) + 5).attr('y', 4).text(d => d.s_bal.toFixed(4));
 
98
  const select = inspect(marks, d => `${d.name}, ${d.params_str} parameters, aggregate test score ${d.s_bal.toFixed(4)}`, d => {
99
  selected = d.id;
100
- f.reading.innerHTML = `<b>${esc(d.name)}</b> · ${esc(d.params_str)} parameters · score ${d.s_bal.toFixed(4)}`;
101
  });
102
  select(rows.find(d => d.id === selected) || rows[0]);
103
  });
 
73
  }
74
 
75
  function benchmark(host, data) {
76
+ const f = frame(host, '<span>Selected references</span><span class="plot-unit">S<sub>bal</sub><sup>test</sup> · axis starts at 0.5 · whiskers 95% CI</span>');
77
  const references = new Set(['PlantCAD2-L', 'GPN', 'CARBON-8B', 'NTv3-pre', 'Evo2-7b', 'AgroNT']);
78
  let selected = 'botanic1-XL';
79
  responsive(host, W => {
 
94
  marks.append('rect').attr('class', 'benchmark-bar').attr('x', L).attr('y', -6).attr('width', d => x(d.s_bal) - L).attr('height', 12).attr('rx', 2)
95
  .attr('fill', d => d.ours ? 'var(--lime)' : 'var(--plot-reference)')
96
  .attr('stroke', d => d.ours ? 'var(--accent)' : 'var(--rule-2)').attr('stroke-width', .6);
97
+ marks.filter(d => d.s_bal_ci).append('line').attr('class', 'benchmark-ci').attr('x1', d => x(d.s_bal_ci[0])).attr('x2', d => x(d.s_bal_ci[1])).attr('y1', 0).attr('y2', 0);
98
+ marks.append('text').attr('class', 'plot-value').attr('x', d => x(d.s_bal_ci ? d.s_bal_ci[1] : d.s_bal) + 5).attr('y', 4).text(d => d.s_bal.toFixed(4));
99
  const select = inspect(marks, d => `${d.name}, ${d.params_str} parameters, aggregate test score ${d.s_bal.toFixed(4)}`, d => {
100
  selected = d.id;
101
+ f.reading.innerHTML = `<b>${esc(d.name)}</b> · ${esc(d.params_str)} parameters · score ${d.s_bal.toFixed(4)}${d.s_bal_ci ? ` · 95% CI [${d.s_bal_ci[0].toFixed(4)}, ${d.s_bal_ci[1].toFixed(4)}]` : ''}`;
102
  });
103
  select(rows.find(d => d.id === selected) || rows[0]);
104
  });
assets/site.css CHANGED
@@ -330,6 +330,11 @@ input[type="search"] { min-width: 250px; }
330
  .scatter .pt.is-family .ptlabel { fill: var(--ink); }
331
  .scatter .pt { outline: none; }
332
  .scatter .point-dot { pointer-events: none; }
 
 
 
 
 
333
  .scatter .point-halo { fill: none; stroke: var(--ink); stroke-width: 1.5; opacity: 0; pointer-events: none; }
334
  .scatter .pt:hover .point-halo, .scatter .pt:focus-visible .point-halo { opacity: 1; }
335
  .chart .ann { fill: var(--ink-3); font-size: 11.5px; }
 
330
  .scatter .pt.is-family .ptlabel { fill: var(--ink); }
331
  .scatter .pt { outline: none; }
332
  .scatter .point-dot { pointer-events: none; }
333
+ .chart .whisker line { stroke: var(--ink); stroke-width: 1.4; pointer-events: none; }
334
+ .scatter .errbar { stroke-width: 1.6; opacity: .9; pointer-events: none; }
335
+ .scatter .pt.is-muted .errbar { opacity: .25; }
336
+ .tip .ci, .tip td.ci { color: var(--ink-3); font-size: .92em; }
337
+ .fig1 .whisker, .benchmark-ci { stroke: var(--ink); stroke-width: 1.4; pointer-events: none; }
338
  .scatter .point-halo { fill: none; stroke: var(--ink); stroke-width: 1.5; opacity: 0; pointer-events: none; }
339
  .scatter .pt:hover .point-halo, .scatter .pt:focus-visible .point-halo { opacity: 1; }
340
  .chart .ann { fill: var(--ink-3); font-size: 11.5px; }
data/leaderboard.json CHANGED
The diff for this file is too large to render. See raw diff
 
index.html CHANGED
@@ -51,7 +51,7 @@
51
  <h2>318M to 3.2B parameters</h2>
52
  <p>Botanic1-S scores 0.758 on the aggregate benchmark across nine task families. PlantCAD2-L scores 0.756, Carbon-8B 0.727, and Evo&nbsp;2-7B 0.695.</p>
53
  <figure class="overview-plot" id="overview-benchmark" aria-label="Model benchmark comparison" hidden></figure>
54
- <p class="fact-note">Frozen models · aggregate test score (S<sub>bal</sub><sup>test</sup>)</p>
55
  <a class="text-link" href="leaderboard.html">Scores by task and species <span aria-hidden="true">↗</span></a>
56
  </article>
57
  <article class="fact">
 
51
  <h2>318M to 3.2B parameters</h2>
52
  <p>Botanic1-S scores 0.758 on the aggregate benchmark across nine task families. PlantCAD2-L scores 0.756, Carbon-8B 0.727, and Evo&nbsp;2-7B 0.695.</p>
53
  <figure class="overview-plot" id="overview-benchmark" aria-label="Model benchmark comparison" hidden></figure>
54
+ <p class="fact-note">Frozen models · aggregate test score (S<sub>bal</sub><sup>test</sup>) · whiskers are 95% CIs</p>
55
  <a class="text-link" href="leaderboard.html">Scores by task and species <span aria-hidden="true">↗</span></a>
56
  </article>
57
  <article class="fact">
leaderboard.html CHANGED
@@ -43,7 +43,7 @@
43
  </div>
44
 
45
  <div class="chart" id="bars"></div>
46
- <p class="chartnote">Hover a bar for the nine family means. Botanic0-L is the previous generation, shown in olive. Parameter counts are trainable parameters as declared by each model's configuration and code; PlantBiMoE has 116M trainable parameters and 64M active per token.</p>
47
 
48
  <h2 id="families">The nine task families</h2>
49
  <p>Zooming in per task shows that Botanic1 leads overall without dominating every capability: the flagship Botanic1-XL tops the benchmark and two task families, genomic-region classification (0.527 versus 0.525 for PlantCAD2-L) and causal-variant discovery (recall AUC 0.752), Botanic1-L leads three more (conservation, translation initiation and PRO-seq) and Botanic1-M leads splicing (donor and acceptor mean 0.984 versus 0.978 for NTv3-650M-post, which uses test annotations during post-training). The autoregressive Evo 2 leads variant-effect prediction, where its 7B model reaches an LLR AUROC of 0.724 against 0.721 for Botanic1-XL (its 40B model reaches 0.719); PlantCAD2-L leads on chromatin accessibility (0.483 versus 0.481 for Botanic1-L); and GPN leads translation termination (0.888 versus 0.887). Botanic1-XL is the only model that stays competitive across all nine families.</p>
@@ -55,7 +55,7 @@
55
  <button data-v="metrics" aria-pressed="false">22 metrics</button>
56
  </div>
57
  </div>
58
- <span class="small">Click a column header to sort by it. Shading is relative within each column.</span>
59
  </div>
60
  <div class="tablewrap"><table class="heat" id="heat"></table></div>
61
 
@@ -65,7 +65,7 @@
65
  <div class="chart scatter" id="scatter-params"></div>
66
  <div class="chart scatter" id="scatter-tokens"></div>
67
  </div>
68
- <p class="chartnote">Tokens are converted to base pairs using approximations for 6-mer models. Models without a documented token count are omitted from the right panel.</p>
69
 
70
  <h2 id="uncertainty">Are leaderboard differences significant?</h2>
71
  <p>We assess leaderboard differences with two paired bootstrap tests, each using 20,000 replicates: one resampling the nine task families, and one resampling test examples within each benchmark cell. Together, they test whether observed margins are robust to variation across task families and to sampling noise in the benchmark. The technical report (v2) reports both tests: the table below reproduces its Supplementary Table on <span class="math">S<sub>bal</sub></span> uncertainty, and the per-sample intervals are drawn as error bars in Figure 1d and as paired differences to PlantCAD2-L in Figure 1e.</p>
@@ -156,10 +156,18 @@
156
  if (m.families[key] != null) return m.families[key];
157
  const mt = m.metrics.find(x => x.label === key); return mt ? mt.value : null;
158
  }
 
 
 
 
 
 
 
 
159
  function sorted(rows) { return rows.slice().sort((a, b) => (value(b, state.sort) ?? -1) - (value(a, state.sort) ?? -1)); }
160
 
161
  function famTable(m) {
162
- return '<table>' + F.map(f => `<tr><td>${f.name}</td><td class="n">${fmt(m.families[f.key])}</td></tr>`).join('') + '</table>';
163
  }
164
 
165
  // ---- bars ----
@@ -193,7 +201,12 @@
193
  s += `<g data-id="${esc(m.id)}" class="row">`;
194
  s += `<text class="rowlabel" x="${left - 14}" y="${y + rowH / 2 + 4}" text-anchor="end">${esc(m.name)}<tspan class="rowsub"> ${sub}</tspan></text>`;
195
  s += `<rect x="${left}" y="${y + 6}" width="${Math.max(0, end - left)}" height="${rowH - 12}" rx="0" fill="${colour(m)}"/>`;
196
- s += `<text class="value" x="${end + 8}" y="${y + rowH / 2 + 4}">${Number.isFinite(v) ? fmt(v) : 'n/a'}</text>`;
 
 
 
 
 
197
  s += `<rect class="hit" x="0" y="${y}" width="${W}" height="${rowH}"/>`;
198
  s += '</g>';
199
  });
@@ -202,7 +215,7 @@
202
  el.innerHTML = s;
203
  el.querySelectorAll('g.row').forEach(g => {
204
  const m = M.find(mm => mm.id === g.dataset.id);
205
- g.addEventListener('mousemove', (ev) => showTip(`<b>${esc(m.name)}</b> · ${esc(m.org)}<br>${m.params_str} parameters · ${m.objective}, ${m.domain} pre-training${m.bp_equivalent ? ' · ' + esc(m.bp_equivalent) + ' bp seen' : ''}<br>S<sub>bal</sub><sup>test</sup> = <b>${fmt(m.s_bal, 4)}</b>` + famTable(m), ev));
206
  g.addEventListener('mouseleave', hideTip);
207
  });
208
  }
@@ -222,8 +235,8 @@
222
  h += '</tr></thead><tbody>';
223
  rows.forEach(m => {
224
  h += `<tr class="${m.ours ? 'ours' : ''}"><td class="name"><span class="swatch" style="background:${colour(m)}"></span>${esc(m.name)}<span class="sub">${m.params_str}</span></td>`;
225
- h += `<td class="cell" ${shade(m.s_bal, sbr)}>${fmt(m.s_bal)}</td>`;
226
- cols.forEach(c => { const v = value(m, c.key); h += `<td class="cell" ${shade(v, colVals[c.key])}>${fmt(v)}</td>`; });
227
  h += '</tr>';
228
  });
229
  h += '</tbody>';
@@ -330,7 +343,7 @@
330
  s += '</g>';
331
  pts.forEach((pt, i) => {
332
  const m = pt.m;
333
- s += `<g data-id="${esc(m.id)}" data-family="${esc(scatterFamily(m))}" class="pt" tabindex="0" role="img" aria-label="${esc(m.name)}, ${esc(m.params_str)} parameters, aggregate score ${fmt(m.s_bal, 4)}"><circle class="point-dot" cx="${pt.cx}" cy="${pt.cy}" r="${pt.r}" fill="${scatterColour(m)}" stroke="var(--paper)" stroke-width="2"/><circle class="point-halo" cx="${pt.cx}" cy="${pt.cy}" r="${pt.r + 4}"/>`;
334
  const l = labels.find(x => x.i === i);
335
  if (l) {
336
  const anchor = l.dx > 0 ? 'start' : l.dx < 0 ? 'end' : 'middle';
@@ -343,7 +356,7 @@
343
  const el = document.getElementById(id); el.innerHTML = s;
344
  el.querySelectorAll('g.pt').forEach(g => {
345
  const m = M.find(mm => mm.id === g.dataset.id);
346
- const showPointTip = ev => showTip(`<b>${esc(m.name)}</b> · ${esc(m.org)}<br>${m.params_str} parameters${m.bp_equivalent ? ' · ' + esc(m.bp_equivalent) + ' bp seen' : ''}<br>S<sub>bal</sub><sup>test</sup> = <b>${fmt(m.s_bal, 4)}</b>`, ev);
347
  g.addEventListener('mouseenter', ev => { hoveredPoint = g; highlightScatterFamily(); showPointTip(ev); });
348
  g.addEventListener('mousemove', showPointTip);
349
  g.addEventListener('mouseleave', () => { hoveredPoint = null; highlightScatterFamily(); hideTip(); });
 
43
  </div>
44
 
45
  <div class="chart" id="bars"></div>
46
+ <p class="chartnote">Hover a bar for the nine family means. Whiskers are 95% confidence intervals from the per-sample paired bootstrap of the technical report (20,000 replicates), centred on the reported score. Botanic0-L is the previous generation, shown in olive. Parameter counts are trainable parameters as declared by each model's configuration and code; PlantBiMoE has 116M trainable parameters and 64M active per token.</p>
47
 
48
  <h2 id="families">The nine task families</h2>
49
  <p>Zooming in per task shows that Botanic1 leads overall without dominating every capability: the flagship Botanic1-XL tops the benchmark and two task families, genomic-region classification (0.527 versus 0.525 for PlantCAD2-L) and causal-variant discovery (recall AUC 0.752), Botanic1-L leads three more (conservation, translation initiation and PRO-seq) and Botanic1-M leads splicing (donor and acceptor mean 0.984 versus 0.978 for NTv3-650M-post, which uses test annotations during post-training). The autoregressive Evo 2 leads variant-effect prediction, where its 7B model reaches an LLR AUROC of 0.724 against 0.721 for Botanic1-XL (its 40B model reaches 0.719); PlantCAD2-L leads on chromatin accessibility (0.483 versus 0.481 for Botanic1-L); and GPN leads translation termination (0.888 versus 0.887). Botanic1-XL is the only model that stays competitive across all nine families.</p>
 
55
  <button data-v="metrics" aria-pressed="false">22 metrics</button>
56
  </div>
57
  </div>
58
+ <span class="small">Click a column header to sort by it. Shading is relative within each column. Hover a cell for its 95% confidence interval.</span>
59
  </div>
60
  <div class="tablewrap"><table class="heat" id="heat"></table></div>
61
 
 
65
  <div class="chart scatter" id="scatter-params"></div>
66
  <div class="chart scatter" id="scatter-tokens"></div>
67
  </div>
68
+ <p class="chartnote">Vertical bars are 95% confidence intervals from the per-sample bootstrap. Tokens are converted to base pairs using approximations for 6-mer models. Models without a documented token count are omitted from the right panel.</p>
69
 
70
  <h2 id="uncertainty">Are leaderboard differences significant?</h2>
71
  <p>We assess leaderboard differences with two paired bootstrap tests, each using 20,000 replicates: one resampling the nine task families, and one resampling test examples within each benchmark cell. Together, they test whether observed margins are robust to variation across task families and to sampling noise in the benchmark. The technical report (v2) reports both tests: the table below reproduces its Supplementary Table on <span class="math">S<sub>bal</sub></span> uncertainty, and the per-sample intervals are drawn as error bars in Figure 1d and as paired differences to PlantCAD2-L in Figure 1e.</p>
 
156
  if (m.families[key] != null) return m.families[key];
157
  const mt = m.metrics.find(x => x.label === key); return mt ? mt.value : null;
158
  }
159
+ function ci(m, key) {
160
+ if (key === 's_bal') return m.s_bal_ci || null;
161
+ if (m.families_ci && m.families_ci[key]) return m.families_ci[key];
162
+ const mt = m.metrics.find(x => x.label === key); return mt && mt.ci ? mt.ci : null;
163
+ }
164
+ const fmtCI = (iv, d = 3) => iv ? `[${fmt(iv[0], d)}, ${fmt(iv[1], d)}]` : '';
165
+ const fmtDelta = (p) => p ? `${p.delta >= 0 ? '+' : '−'}${Math.abs(p.delta).toFixed(4)} [${p.lo >= 0 ? '+' : '−'}${Math.abs(p.lo).toFixed(4)}, ${p.hi >= 0 ? '+' : '−'}${Math.abs(p.hi).toFixed(4)}]${p.significant ? '<sup>*</sup>' : ' (ns)'}` : '';
166
+ const sbalTip = (m) => `S<sub>bal</sub><sup>test</sup> = <b>${fmt(m.s_bal, 4)}</b>${m.s_bal_ci ? ` <span class="ci">95% CI ${fmtCI(m.s_bal_ci, 4)}</span>` : ''}${m.vs_plantcad2l ? `<br>Δ to PlantCAD2-L ${fmtDelta(m.vs_plantcad2l)}` : ''}`;
167
  function sorted(rows) { return rows.slice().sort((a, b) => (value(b, state.sort) ?? -1) - (value(a, state.sort) ?? -1)); }
168
 
169
  function famTable(m) {
170
+ return '<table>' + F.map(f => `<tr><td>${f.name}</td><td class="n">${fmt(m.families[f.key])}</td><td class="n ci">${fmtCI(m.families_ci && m.families_ci[f.key])}</td></tr>`).join('') + '</table>';
171
  }
172
 
173
  // ---- bars ----
 
201
  s += `<g data-id="${esc(m.id)}" class="row">`;
202
  s += `<text class="rowlabel" x="${left - 14}" y="${y + rowH / 2 + 4}" text-anchor="end">${esc(m.name)}<tspan class="rowsub"> ${sub}</tspan></text>`;
203
  s += `<rect x="${left}" y="${y + 6}" width="${Math.max(0, end - left)}" height="${rowH - 12}" rx="0" fill="${colour(m)}"/>`;
204
+ const iv = Number.isFinite(v) ? ci(m, key) : null;
205
+ if (iv) {
206
+ const lo = x(Math.max(iv[0], x0)), hi = x(Math.min(iv[1], x1)), cy = y + rowH / 2;
207
+ s += `<g class="whisker"><line x1="${lo}" x2="${hi}" y1="${cy}" y2="${cy}"/><line x1="${lo}" x2="${lo}" y1="${cy - 4}" y2="${cy + 4}"/><line x1="${hi}" x2="${hi}" y1="${cy - 4}" y2="${cy + 4}"/></g>`;
208
+ }
209
+ s += `<text class="value" x="${(iv ? x(Math.min(iv[1], x1)) : end) + 8}" y="${y + rowH / 2 + 4}">${Number.isFinite(v) ? fmt(v) : 'n/a'}</text>`;
210
  s += `<rect class="hit" x="0" y="${y}" width="${W}" height="${rowH}"/>`;
211
  s += '</g>';
212
  });
 
215
  el.innerHTML = s;
216
  el.querySelectorAll('g.row').forEach(g => {
217
  const m = M.find(mm => mm.id === g.dataset.id);
218
+ g.addEventListener('mousemove', (ev) => showTip(`<b>${esc(m.name)}</b> · ${esc(m.org)}<br>${m.params_str} parameters · ${m.objective}, ${m.domain} pre-training${m.bp_equivalent ? ' · ' + esc(m.bp_equivalent) + ' bp seen' : ''}<br>${sbalTip(m)}` + famTable(m), ev));
219
  g.addEventListener('mouseleave', hideTip);
220
  });
221
  }
 
235
  h += '</tr></thead><tbody>';
236
  rows.forEach(m => {
237
  h += `<tr class="${m.ours ? 'ours' : ''}"><td class="name"><span class="swatch" style="background:${colour(m)}"></span>${esc(m.name)}<span class="sub">${m.params_str}</span></td>`;
238
+ h += `<td class="cell" ${shade(m.s_bal, sbr)} title="${fmt(m.s_bal, 4)}${m.s_bal_ci ? ' · 95% CI ' + fmtCI(m.s_bal_ci, 4) : ''}">${fmt(m.s_bal)}</td>`;
239
+ cols.forEach(c => { const v = value(m, c.key); const iv = ci(m, c.key); h += `<td class="cell" ${shade(v, colVals[c.key])} title="${fmt(v, 4)}${iv ? ' · 95% CI ' + fmtCI(iv, 4) : ''}">${fmt(v)}</td>`; });
240
  h += '</tr>';
241
  });
242
  h += '</tbody>';
 
343
  s += '</g>';
344
  pts.forEach((pt, i) => {
345
  const m = pt.m;
346
+ s += `<g data-id="${esc(m.id)}" data-family="${esc(scatterFamily(m))}" class="pt" tabindex="0" role="img" aria-label="${esc(m.name)}, ${esc(m.params_str)} parameters, aggregate score ${fmt(m.s_bal, 4)}">${m.s_bal_ci ? `<line class="errbar" x1="${pt.cx}" x2="${pt.cx}" y1="${y(Math.min(Math.max(m.s_bal_ci[0], y0), y1))}" y2="${y(Math.min(Math.max(m.s_bal_ci[1], y0), y1))}" stroke="${scatterColour(m)}"/>` : ''}<circle class="point-dot" cx="${pt.cx}" cy="${pt.cy}" r="${pt.r}" fill="${scatterColour(m)}" stroke="var(--paper)" stroke-width="2"/><circle class="point-halo" cx="${pt.cx}" cy="${pt.cy}" r="${pt.r + 4}"/>`;
347
  const l = labels.find(x => x.i === i);
348
  if (l) {
349
  const anchor = l.dx > 0 ? 'start' : l.dx < 0 ? 'end' : 'middle';
 
356
  const el = document.getElementById(id); el.innerHTML = s;
357
  el.querySelectorAll('g.pt').forEach(g => {
358
  const m = M.find(mm => mm.id === g.dataset.id);
359
+ const showPointTip = ev => showTip(`<b>${esc(m.name)}</b> · ${esc(m.org)}<br>${m.params_str} parameters${m.bp_equivalent ? ' · ' + esc(m.bp_equivalent) + ' bp seen' : ''}<br>${sbalTip(m)}`, ev);
360
  g.addEventListener('mouseenter', ev => { hoveredPoint = g; highlightScatterFamily(); showPointTip(ev); });
361
  g.addEventListener('mousemove', showPointTip);
362
  g.addEventListener('mouseleave', () => { hoveredPoint = null; highlightScatterFamily(); hideTip(); });
scripts/add_bootstrap_ci.py ADDED
@@ -0,0 +1,108 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Add per-sample bootstrap confidence intervals to data/leaderboard.json.
3
+
4
+ Source: figures/data/fig3_rerun/rerun_ci.json in the technical report repository
5
+ (dotomics/BOTANIC1-technical-report, branch overleaf, commit 6e420d15), the
6
+ per-example paired bootstrap of the leaderboard cohort (20,000 replicates, draws
7
+ shared across models within each cell, 95% percentile intervals). The same file
8
+ feeds every interval in the v2 report (sec. S_bal uncertainty).
9
+
10
+ Convention (figures/scripts/fig3_rerun_ci.py in that repository): the point
11
+ estimates stay the reported scores; each interval is the rerun's percentile
12
+ interval translated so that it sits about the reported value, i.e. the offsets
13
+ (lo - mean, hi - mean) are added to the reported point.
14
+
15
+ Fields added per model: s_bal_ci [lo, hi]; families_ci {family: [lo, hi]};
16
+ metrics[i].ci [lo, hi]; vs_plantcad2l {delta, lo, hi, significant} (paired
17
+ example-level delta to PlantCAD2-L, centred on the reported difference).
18
+
19
+ Run: python3 scripts/add_bootstrap_ci.py path/to/rerun_ci.json
20
+ """
21
+ import json
22
+ import sys
23
+ from pathlib import Path
24
+
25
+ ROOT = Path(__file__).resolve().parents[1]
26
+ LB = ROOT / "data" / "leaderboard.json"
27
+
28
+ FAMILY = {"chromatin": "chromatin_access", "grc": "genomic_region_classification_v2",
29
+ "conservation": "plantcad_conservation", "splicing": "splicing", "tis": "plantcad_tis",
30
+ "tts": "plantcad_tts", "proseq": "pro_seq", "llr": "llr", "causal": "gwas"}
31
+ D = "downstream_tasks."
32
+ CELL = {
33
+ "GWAS · recall_auc": "gwas_eval_benchmark/recall_auc",
34
+ "Chromatin · arabidopsis": D + "chromatin_access/chromatin_access.arabidopis_thaliana/eval_aucpr_macro",
35
+ "Chromatin · brachypodium": D + "chromatin_access/chromatin_access.brachypodium_distachyon/eval_aucpr_macro",
36
+ "Chromatin · maize": D + "chromatin_access/chromatin_access.zea_mays/eval_aucpr_macro",
37
+ "Chromatin · rice MH63": D + "chromatin_access/chromatin_access.oryza_sativa_MH63_RS2/eval_aucpr_macro",
38
+ "Chromatin · rice ZS97": D + "chromatin_access/chromatin_access.oryza_sativa_ZS97_RS2/eval_aucpr_macro",
39
+ "Chromatin · setaria": D + "chromatin_access/chromatin_access.setaria_italica/eval_aucpr_macro",
40
+ "Chromatin · sorghum": D + "chromatin_access/chromatin_access.sorghum_bicolor/eval_aucpr_macro",
41
+ "Conservation · maize": D + "plantcad_conservation/plantcad_conservation.maize/eval_aucpr",
42
+ "Conservation · sorghum": D + "plantcad_conservation/plantcad_conservation.sorghum/eval_aucpr",
43
+ "GRC2 · arabidopsis": D + "genomic_region_classification_v2/genomic_region_classification_v2.arabidopsis_thaliana/eval_balanced_accuracy",
44
+ "GRC2 · maize": D + "genomic_region_classification_v2/genomic_region_classification_v2.zea_mays/eval_balanced_accuracy",
45
+ "GRC2 · rice": D + "genomic_region_classification_v2/genomic_region_classification_v2.oryza_sativa/eval_balanced_accuracy",
46
+ "GRC2 · tomato": D + "genomic_region_classification_v2/genomic_region_classification_v2.solanum_lycopersicum/eval_balanced_accuracy",
47
+ "LLR · arabidopsis": "llr_eval/arabidopsis_thaliana/auroc",
48
+ "LLR · maize": "llr_eval/zea_mays/auroc",
49
+ "LLR · tomato": "llr_eval/solanum_lycopersicum/auroc",
50
+ "PRO-seq · cassava": D + "pro_seq/pro_seq.m_esculenta/eval_aucpr",
51
+ "Splice acceptor · arabidopsis": D + "splicing/splicing.arabidopsis_thaliana_acceptor/eval_stratified_aucpr_corrected",
52
+ "Splice donor · arabidopsis": D + "splicing/splicing.arabidopsis_thaliana_donor/eval_stratified_aucpr_corrected",
53
+ "TIS · arabidopsis": D + "plantcad_tis/plantcad_tis.arabidopsis/eval_aucpr",
54
+ "TTS · arabidopsis": D + "plantcad_tts/plantcad_tts.arabidopsis/eval_aucpr",
55
+ }
56
+
57
+
58
+ def norm(model_id: str) -> str:
59
+ if model_id.startswith("botanic1-"):
60
+ return "Botanic1-" + model_id.split("-", 1)[1]
61
+ if model_id.startswith("CARBON-"):
62
+ return "Carbon-" + model_id.split("-", 1)[1]
63
+ return model_id
64
+
65
+
66
+ def about(entry, published, key="mean"):
67
+ shift = published - entry[key]
68
+ return [round(entry["lo"] + shift, 4), round(entry["hi"] + shift, 4)]
69
+
70
+
71
+ def main(rerun_path: str) -> None:
72
+ rerun = json.loads(Path(rerun_path).read_text())
73
+ lb = json.loads(LB.read_text())
74
+ ref = "PlantCAD2-L"
75
+ ref_sbal = next(m["s_bal"] for m in lb["models"] if m["id"] == ref)
76
+ n_ci = 0
77
+ for m in lb["models"]:
78
+ r = rerun["models"].get(norm(m["id"]))
79
+ if r is None:
80
+ print(f" no rerun entry for {m['id']}")
81
+ continue
82
+ m["s_bal_ci"] = about(r["sbal"], m["s_bal"])
83
+ m["families_ci"] = {k: about(r["families"][FAMILY[k]], v) for k, v in m["families"].items() if FAMILY[k] in r["families"]}
84
+ for mt in m["metrics"]:
85
+ c = r["cells"].get(CELL[mt["label"]])
86
+ if c is not None:
87
+ mt["ci"] = about(c, mt["value"])
88
+ if m["id"] != ref:
89
+ p = rerun["pairs"].get(f"{norm(m['id'])} vs {ref}")
90
+ if p is not None:
91
+ d = round(m["s_bal"] - ref_sbal, 4)
92
+ lo, hi = about(p, d, key="delta")
93
+ m["vs_plantcad2l"] = {"delta": d, "lo": lo, "hi": hi, "significant": bool(lo > 0 or hi < 0)}
94
+ n_ci += 1
95
+ # sanity: rerun point within 0.02 of the reported score
96
+ gap = abs(r["sbal"]["mean"] - m["s_bal"])
97
+ flag = " <-- check" if gap > 0.02 else ""
98
+ print(f"{m['id']:14s} s_bal={m['s_bal']:.4f} rerun={r['sbal']['mean']:.4f} ci={m['s_bal_ci']}{flag}")
99
+ lb["ci_note"] = ("95% confidence intervals from the per-sample paired bootstrap of the technical report "
100
+ "(20,000 replicates, draws shared across models within each cell), centred on the reported score.")
101
+ lb["ci_meta"] = {"replicates": rerun["meta"]["replicates"], "level": 95,
102
+ "source": "BOTANIC1-technical-report figures/data/fig3_rerun/rerun_ci.json @ 6e420d15"}
103
+ LB.write_text(json.dumps(lb, indent=1, ensure_ascii=False) + "\n")
104
+ print(f"wrote {LB} with intervals for {n_ci}/{len(lb['models'])} models")
105
+
106
+
107
+ if __name__ == "__main__":
108
+ main(sys.argv[1])