Xiangyi Li commited on
Commit
c42a99e
·
1 Parent(s): 5e1dd59

Overview tells the story: state, challenges, run attempts, leaderboard

Browse files

Two fresh-eyes reviews could not tell the arena's stage, who is winning, the
run history or the next step. The overview now opens with one sentence on what
the arena ranks, four headline numbers (runs scored, furthest stage reached,
the untrained model's held-out score, compute left and how many runs fit), the
organizer's status note from the challenge file, a compact challenges table
with the noise threshold, every run as a row of pipeline stages with where and
why it stopped, and the leaderboard with its ranking rule. Training charts,
the task feed, compute and jobs move to the metrics tab; job kinds carry their
purpose; loading states replace false empty states.

Files changed (4) hide show
  1. challenges.py +7 -2
  2. compute_jobs.py +9 -2
  3. configs/challenges/tb2-9b.toml +2 -0
  4. index.html +130 -99
challenges.py CHANGED
@@ -51,7 +51,7 @@ def challenge_row(spec):
51
  suite=compose.fragment('suites',b['suites'][0]);s,meta=suite['suite'],suite.get('meta',{}) # the single-suite pipeline scores the first
52
  grpo,runtime,harness=method.get('grpo',{}),method.get('runtime',{}),method.get('harness',{})
53
  recipe=spec['recipe'];compute=spec['compute']
54
- return {'id':spec['id'],'name':spec['name'],'status':spec['status'],'opens':spec.get('opens'),'closes':spec.get('closes'),
55
  'role':spec.get('role'),'role_note':spec.get('role_note'),'baseline_file':spec.get('baseline_file'),'binding':dict(b),'summary':spec['summary'],
56
  'eval_suite':{'name':meta.get('name',s['name']),'repo_id':s['repo_id'],'revision':s['revision'],'task_list':s['task_list'],
57
  'task_count':len(compose.task_ids(b['suites'][0])),'sealed':meta.get('sealed',True),'note':meta.get('note','')},
@@ -103,7 +103,7 @@ def collection_rows_cached():
103
  def formula():
104
  """The registries behind each term of the formula, and every challenge that binds them."""
105
  challenges=[{'id':c['id'],'name':c['name'],'status':c['status'],'role':c.get('role'),'compute':f"HF {c['compute']['flavor']}",**c['binding'],
106
- 'accepting':{k:v for k,v in health(c).items() if k in ('accepting_runs','reason')}} for c in CHALLENGES]+[dict(p) for p in PLANNED_CHALLENGES]
107
  uses=lambda key,value:[c['id'] for c in challenges if value==c.get(key) or value in (c.get(key) or [])]
108
  references={}
109
  for c in CHALLENGES:
@@ -150,6 +150,11 @@ def hardware():
150
  _hardware.update(rows=response.json(),at=time.time())
151
  return _hardware['rows']
152
 
 
 
 
 
 
153
  def quote(row):
154
  """Conservative HF bound for one run: flavor price times the hard job timeout. Daytona is billed separately and estimated only."""
155
  try:
 
51
  suite=compose.fragment('suites',b['suites'][0]);s,meta=suite['suite'],suite.get('meta',{}) # the single-suite pipeline scores the first
52
  grpo,runtime,harness=method.get('grpo',{}),method.get('runtime',{}),method.get('harness',{})
53
  recipe=spec['recipe'];compute=spec['compute']
54
+ return {'id':spec['id'],'name':spec['name'],'status':spec['status'],'opens':spec.get('opens'),'closes':spec.get('closes'),'status_note':spec.get('status_note'),
55
  'role':spec.get('role'),'role_note':spec.get('role_note'),'baseline_file':spec.get('baseline_file'),'binding':dict(b),'summary':spec['summary'],
56
  'eval_suite':{'name':meta.get('name',s['name']),'repo_id':s['repo_id'],'revision':s['revision'],'task_list':s['task_list'],
57
  'task_count':len(compose.task_ids(b['suites'][0])),'sealed':meta.get('sealed',True),'note':meta.get('note','')},
 
103
  def formula():
104
  """The registries behind each term of the formula, and every challenge that binds them."""
105
  challenges=[{'id':c['id'],'name':c['name'],'status':c['status'],'role':c.get('role'),'compute':f"HF {c['compute']['flavor']}",**c['binding'],
106
+ 'accepting':{k:v for k,v in health(c).items() if k in ('accepting_runs','reason')},'run_reserves_usd':reserve_bound(c),'status_note':c.get('status_note')} for c in CHALLENGES]+[dict(p) for p in PLANNED_CHALLENGES]
107
  uses=lambda key,value:[c['id'] for c in challenges if value==c.get(key) or value in (c.get(key) or [])]
108
  references={}
109
  for c in CHALLENGES:
 
150
  _hardware.update(rows=response.json(),at=time.time())
151
  return _hardware['rows']
152
 
153
+ def reserve_bound(row):
154
+ """What one run reserves against the cap, or None when prices are unavailable (display only)."""
155
+ try: return round(quote(row)['max_compute_usd'],2)
156
+ except HTTPException: return None
157
+
158
  def quote(row):
159
  """Conservative HF bound for one run: flavor price times the hard job timeout. Daytona is billed separately and estimated only."""
160
  try:
compute_jobs.py CHANGED
@@ -29,12 +29,19 @@ def kind(labels):
29
  experiment = labels.get('experiment', '')
30
  if labels.get('posttrain') == 'baseline-grid': return 'baseline eval'
31
  if experiment == 'posttrain-challenge' or labels.get('challenge'): return 'challenge run'
32
- if experiment.startswith('posttrain-phase'): return 'organizer run'
33
  if experiment.startswith('arena-'): return 'arena job'
34
  if experiment.startswith('posttrain') or labels.get('posttrain'): return (experiment or labels.get('posttrain')).removeprefix('posttrain-').replace('-', ' ')
35
  return None
36
 
37
 
 
 
 
 
 
 
 
38
  def prices():
39
  if time.time() - _prices['at'] > 3600 or not _prices['value']:
40
  try:
@@ -87,7 +94,7 @@ def build():
87
  rate = price.get(job.flavor)
88
  cost = round(math.ceil(seconds / 60) * rate, 2) if seconds is not None and rate is not None else None
89
  detail = labels.get('grid_point') or labels.get('phase') or labels.get('round') or labels.get('suite')
90
- rows.append({'id': job.id, 'name': labels.get('name') or job.id, 'kind': what, 'detail': detail, 'run_id': labels.get('run_id') or labels.get('name'),
91
  'challenge': labels.get('challenge'), 'flavor': job.flavor, 'stage': stage, 'created_at': _iso(job.created_at),
92
  'started_at': _iso(job.started_at), 'finished_at': _iso(getattr(job, 'finished_at', None)),
93
  'seconds': round(seconds) if seconds is not None else None, 'cost_usd': cost, 'url': f'https://huggingface.co/jobs/benchflow/{job.id}'})
 
29
  experiment = labels.get('experiment', '')
30
  if labels.get('posttrain') == 'baseline-grid': return 'baseline eval'
31
  if experiment == 'posttrain-challenge' or labels.get('challenge'): return 'challenge run'
32
+ if experiment.startswith('posttrain-phase'): return 'organizer test run'
33
  if experiment.startswith('arena-'): return 'arena job'
34
  if experiment.startswith('posttrain') or labels.get('posttrain'): return (experiment or labels.get('posttrain')).removeprefix('posttrain-').replace('-', ' ')
35
  return None
36
 
37
 
38
+ # What each kind of job is for, shown next to it on the dashboard.
39
+ PURPOSE = {'challenge run': 'A participant run: trains the challenge model on one submitted collection and measures held-out before and after.',
40
+ 'organizer test run': 'An organizer run of the same pipeline, used to make it work before participant runs opened.',
41
+ 'baseline eval': 'Evaluation only, no training: measures the untrained model under a harness setting (context length, time limit, agent) to explain the baseline.',
42
+ 'arena job': 'A job from the retired experiment runner.'}
43
+
44
+
45
  def prices():
46
  if time.time() - _prices['at'] > 3600 or not _prices['value']:
47
  try:
 
94
  rate = price.get(job.flavor)
95
  cost = round(math.ceil(seconds / 60) * rate, 2) if seconds is not None and rate is not None else None
96
  detail = labels.get('grid_point') or labels.get('phase') or labels.get('round') or labels.get('suite')
97
+ rows.append({'id': job.id, 'name': labels.get('name') or job.id, 'kind': what, 'purpose': PURPOSE.get(what), 'detail': detail, 'run_id': labels.get('run_id') or labels.get('name'),
98
  'challenge': labels.get('challenge'), 'flavor': job.flavor, 'stage': stage, 'created_at': _iso(job.created_at),
99
  'started_at': _iso(job.started_at), 'finished_at': _iso(getattr(job, 'finished_at', None)),
100
  'seconds': round(seconds) if seconds is not None else None, 'cost_usd': cost, 'url': f'https://huggingface.co/jobs/benchflow/{job.id}'})
configs/challenges/tb2-9b.toml CHANGED
@@ -10,6 +10,8 @@ role = "smoke test"
10
  role_note = "It proves the submission-to-leaderboard loop closes end to end. Collections are compared on larger challenges, such as the planned terminal-35b."
11
  summary = "Submit a collection of BenchFlow task environments. The arena post-trains the pinned Qwen3.5-9B on your tasks with the pinned GRPO recipe and reports the pass@1 change on a sealed 32-task Terminal-Bench 2.0 subset."
12
  baseline_file = "results/tb2-32-baseline.json"
 
 
13
 
14
  [binding]
15
  model = "qwen3.5-9b"
 
10
  role_note = "It proves the submission-to-leaderboard loop closes end to end. Collections are compared on larger challenges, such as the planned terminal-35b."
11
  summary = "Submit a collection of BenchFlow task environments. The arena post-trains the pinned Qwen3.5-9B on your tasks with the pinned GRPO recipe and reports the pass@1 change on a sealed 32-task Terminal-Bench 2.0 subset."
12
  baseline_file = "results/tb2-32-baseline.json"
13
+ # Organizer-written: where the hill climb stands and the next step. Shown under the overview's headline numbers.
14
+ status_note = "No run has finished yet. The crash that stopped 5503c8d5 at its first training step (an NCCL device mismatch) is fixed in the pinned pipeline, 3d0a7df, so the next tb2-9b run is the first that can reach held-out after. The compute left covers one run."
15
 
16
  [binding]
17
  model = "qwen3.5-9b"
index.html CHANGED
@@ -7,7 +7,7 @@
7
  <link rel="icon" href="/icon.svg" type="image/svg+xml">
8
  <link rel="preconnect" href="https://fonts.googleapis.com">
9
  <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
10
- <link href="https://fonts.googleapis.com/css2?family=Inter:wght@400;500;600&family=JetBrains+Mono:wght@400;500&display=swap" rel="stylesheet">
11
  <style>
12
  /* Layout follows the MiMo RL dashboard (mimo.xiaomi.com/rl); tokens follow posttrain.com: white paper, black ink, one blue, hairlines, square corners. */
13
  :root {
@@ -108,8 +108,6 @@
108
  .tag { display: inline-block; font-size: 10.5px; letter-spacing: .05em; text-transform: uppercase; color: var(--mute); white-space: nowrap; }
109
  .tag.on { color: var(--blue); }
110
  .cnote { padding: 0 14px 12px; font-size: 12px; color: var(--mute); line-height: 1.5; }
111
- .why { display: flex; gap: 12px; padding: 8px 16px 10px; border-top: 1px solid var(--line-soft); font-size: 12px; min-width: 0; }
112
- .why code { font: 11.5px var(--mono); color: var(--ink-soft); overflow: hidden; text-overflow: ellipsis; white-space: nowrap; min-width: 0; }
113
  .card.clickable { cursor: pointer; } .card.clickable:hover .run-h .name { color: var(--blue); }
114
  td.strong { font-weight: 600; }
115
  .hc-facts.gl { grid-template-columns: 170px minmax(0, 1fr); }
@@ -125,6 +123,20 @@
125
  .scroll { overflow-x: auto; background: linear-gradient(to right, var(--paper) 30%, transparent) left / 24px 100% no-repeat local, linear-gradient(to left, var(--paper) 30%, transparent) right / 24px 100% no-repeat local, linear-gradient(to right, rgba(8, 8, 8, .10), transparent) left / 10px 100% no-repeat scroll, linear-gradient(to left, rgba(8, 8, 8, .10), transparent) right / 10px 100% no-repeat scroll; }
126
  .grid4 { display: grid; grid-template-columns: repeat(auto-fit, minmax(300px, 1fr)); gap: 12px; }
127
  .kpi.only-s { display: none; }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
128
  .cnote div + div { margin-top: 4px; }
129
 
130
  /* runs list (metrics view) */
@@ -162,6 +174,8 @@
162
  .wrap { padding: 0 14px; } .kpi.hide-s, .brand span { display: none; } .bar { gap: 16px; } .log li, .feed li { grid-template-columns: 64px minmax(0, 1fr); } .feed li > :last-child { display: none; } .strip { grid-template-columns: repeat(3, minmax(0, 1fr)); }
163
  .strip > div:nth-child(4) { border-left: 0; } .strip > div:nth-child(n+4) { border-top: 1px solid var(--line-soft); } .strip > div { padding: 8px 10px 10px; } .strip small { display: block; margin-left: 0; overflow-wrap: anywhere; }
164
  .kpi.only-s { display: flex; } .grid4 { grid-template-columns: minmax(0, 1fr); }
 
 
165
  .grid3, .grid3.fit, .hc-grid { grid-template-columns: minmax(0, 1fr); } .view.on > * { min-width: 0; } .split { grid-template-columns: 1fr; } .opt, .hide-s { display: none; } .show-s { display: block; } .hc-facts.gl { grid-template-columns: 1fr; gap: 0 16px; } .hc-facts.gl dd { margin-bottom: 8px; } .kpi b { font-size: 11.5px; }
166
  th, td { padding-left: 8px; padding-right: 8px; } th:first-child, td:first-child { padding-left: 14px; }
167
  .row { grid-template-columns: minmax(0, 1fr) 90px 44px 12px; } .row .st { display: none; }
@@ -203,7 +217,7 @@ const GATE_BODY = 'The untrained model’s pass rate on the first 32 of the coll
203
  const PLATFORM = /nccl|cuda|out of memory|vllm|bridge|relay|daytona|sandbox|acp initialize|handshake|snapshot-|pty|connection (reset|refused)|rate limit|429|no healthy scored rollout|infra/i;
204
  const SERIES = ['--s-1', '--s-2', '--s-3'];
205
 
206
- let F = null, METRICS = {}, JOBS = null, showTests = localStorage.getItem('pta.tests') === '1', allJobs = false;
207
  const openRuns = new Set();
208
 
209
  // ── data ──────────────────────────────────────────────────────────────
@@ -219,7 +233,8 @@ async function load() {
219
  if (ver.status === 'fulfilled') { const b = ver.value; $('#build').textContent = `build ${b.build}${b.stale ? ' · files changed since the server started; restart it' : ''}`; }
220
  if (f.status === 'fulfilled') F = f.value; else failed = true;
221
  if (j.status === 'fulfilled') JOBS = j.value; else failed = true;
222
- await Promise.all(openChallenges().map(async c => { try { METRICS[c.id] = await getJSON('/api/challenges/' + encodeURIComponent(c.id) + '/metrics'); } catch { failed = true; } }));
 
223
  $('#err').hidden = !failed; $('#err').textContent = failed ? 'Some data is unavailable right now; the page retries every 30 seconds.' : '';
224
  loading = false; render();
225
  }
@@ -266,7 +281,7 @@ function glyph(r) { return ACTIVE.includes(r.state) ? node('span', null, 'dot li
266
  function verdictLine(v) { const b = [`${v.pass} of ${v.total} passed`]; if (v.timeout) b.push(`${v.timeout} timed out`); if (v.error) b.push(`${v.error} infra error${v.error === 1 ? '' : 's'}`); if (v.done < v.total) b.push(`${v.total - v.done} not run`); return b.join(' · '); }
267
  const td = (content, cls) => { const c = node('td', null, cls || ''); if (content instanceof Node) c.append(content); else c.textContent = content ?? '—'; return c; };
268
  const two = (a, b) => { const s = node('span'); s.append(a instanceof Node ? a : node('span', a)); if (b) s.append(node('span', b, 'two')); return s; };
269
- function table(cols, rows) { const box = node('div', null, 'scroll'), t = node('table'), tr = node('tr'); for (const [l, c] of cols) tr.append(node('th', l, c || '')); t.append(tr); rows.forEach(r => t.append(r)); box.append(t); return box; }
270
  function card(title, sub, tools) { const c = node('div', null, 'card'), h = node('div', null, 'ch'), t = node('div'); t.append(node('b', title)); if (sub) { t.append(' '); t.append(node('span', sub)); } h.append(t); if (tools) h.append(tools); c.append(h); return c; }
271
  const sec = (title, sub) => { const h = node('h3', title, 'sec'); if (sub) h.append(node('span', sub)); return h; };
272
 
@@ -325,60 +340,6 @@ function chartCard(title, sub, values, height, opts, note) {
325
  }
326
 
327
  // ── overview ──────────────────────────────────────────────────────────
328
- // Hill climbs: one panel per challenge saying what is being climbed, how a run works, where it runs, who is in it and where it stands.
329
- function hillClimbs() {
330
- const grid = node('div', null, 'grid3 fit hc-grid');
331
- const order = [...F.challenges].sort((a, b) => (a.status === 'planned') - (b.status === 'planned'));
332
- for (const ch of order) {
333
- const m = byId(F.models, ch.model), meth = byId(F.methods, ch.method), suites = (ch.suites || []).map(id => byId(F.suites, id)).filter(Boolean);
334
- const planned = ch.status === 'planned', runs = allRuns().filter(r => r.challenge === ch.id).sort(byStart);
335
- const part = runs.filter(r => r.source !== 'organizer'), scored = part.filter(r => r.delta_pp != null), last = part[part.length - 1];
336
- const authors = [...new Set(F.collections.filter(e => part.some(r => r.environment_id === e.id)).map(e => e.author).filter(Boolean))];
337
- const cols = new Set(part.map(r => r.environment_id)).size;
338
- const c = node('div', null, 'card hc'), h = node('div', null, 'ch'), t = node('div');
339
- t.append(node('b', `hill climb · ${ch.id}`), ' ', node('span', planned ? 'planned' : (ch.role || 'open'), 'tag' + (planned ? '' : ' on')));
340
- const acc = ch.accepting || {}, refusing = !planned && acc.accepting_runs === false;
341
- h.append(t, planned ? node('span', 'not open for runs yet') : refusing ? node('span', 'not accepting runs now', 'muted') : ext('join ↗', '/AGENTS.md', 'link'));
342
- const dl = node('dl', null, 'hc-facts'), row = (k, v) => { dl.append(node('dt', k), node('dd', v)); };
343
- const named = suites.map(s => /\btask/.test(s.name) ? s.name : `${s.name} (${s.task_count} tasks)`), suiteText = named.length > 1 ? `${named.length} sealed held-out suites: ${named.slice(0, -1).join(', ')} and ${named[named.length - 1]}` : `the sealed held-out suite ${named[0] || ''}`;
344
- row('what', `Submitted environment collections compete to raise ${m ? m.repo_id.split('/').pop() : ch.model}'s pass@1 on ${suiteText}. Participants never see these tasks.`);
345
- row('how', meth ? `Each run trains the model on one collection with ${meth.id}${meth.steps ? ` (${meth.steps} step${meth.steps === 1 ? '' : 's'} × ${meth.tasks_per_step} task${meth.tasks_per_step === 1 ? '' : 's'} × ${meth.group_size} rollouts, GRPO with LoRA)` : ''}, then scores Δ = held-out pass rate after − before, both measured in the run${meth.trials > 1 ? ` over ${meth.trials} trials` : ' (one trial)'}. Each task gets ${meth.agent_timeout_sec ? Math.round(meth.agent_timeout_sec / 60) + ' min' : 'a fixed time limit'}; time-outs count as failures, and so do sandbox errors up to ${meth.infra_error_fraction != null ? Math.round(100 * meth.infra_error_fraction) + '% of an evaluation’s tasks' : 'a cap'}, beyond which the run stops.` : ch.method);
346
- if (meth && meth.note) row('note', (ch.role === 'smoke test' ? 'Smoke test: proves the pipeline end to end; its scores are not evidence. ' : '') + meth.note);
347
- row('where', `${ch.compute}; tasks run in Daytona sandboxes${planned ? '' : ', reaching the model through this Space'}.`);
348
- const everyone = [...new Set(F.collections.map(e => e.author).filter(Boolean))];
349
- row('who', planned ? (ch.open_note || 'Not open for runs yet.') : F.collections.length ? `${F.collections.length} collection${F.collections.length === 1 ? '' : 's'} submitted by ${everyone.join(', ') || 'unknown authors'}; ${cols ? `${cols} ${cols === 1 ? 'has' : 'have'} run here${authors.length ? ' (' + authors.join(', ') + ')' : ''}` : 'none has run here yet'}.` : 'No submissions yet.');
350
- if (!planned) {
351
- const mean = scored.length ? scored.reduce((a, r) => a + r.delta_pp, 0) / scored.length : null; // the leaderboard ranks on the mean, not the best run
352
- row('status', `${part.length} run${part.length === 1 ? '' : 's'}, ${scored.length} scored${mean != null ? `; mean Δ ${signed(mean)} pp` : ''}${last ? `; latest ${code(last)} ${stateText(last)}, ${when(startOf(last))}` : ''}.`);
353
- if (refusing) row('now', `Not accepting runs: ${acc.reason}`);
354
- }
355
- c.append(h, dl); grid.append(c);
356
- }
357
- return grid;
358
- }
359
- // Run cards: runs in progress, else the latest participant runs (MiMo's per-run cards).
360
- function runCards() {
361
- const runs = shown().sort(byStart), live = runs.filter(r => ACTIVE.includes(r.state));
362
- const pick = (live.length ? live : runs).slice(-2).reverse();
363
- return pick.map((r, i) => {
364
- const c = node('div', null, 'card'), h = node('div', null, 'run-h');
365
- const name = node('span', null, 'name'), sq = node('i', null, 'sq'); sq.style.background = css(SERIES[i]); name.append(sq, code(r), node('span', ' ' + nameOf(r), 'soft'));
366
- const job = jobOf(r), steps = (stageOf(r, 'training').training || {}).metrics || [];
367
- const ran = ranFor(r);
368
- h.append(name, node('span', r.challenge, 'muted'), node('span', stateText(r), 'stage'),
369
- ran == null ? '' : node('span', `${ACTIVE.includes(r.state) ? 'running for' : 'ran'} ${human(ran)}`, 'meta'), node('span', `${r.started_at || (job && job.started_at) ? 'started' : 'requested'} ${when(r.started_at || (job && job.started_at) || r.created_at)}${!ACTIVE.includes(r.state) && (r.ended_at || (job && job.finished_at)) ? ' · ended ' + when(r.ended_at || job.finished_at) : ''}`, 'meta'));
370
- const g = verdictsOf(r, 'gate'), b = verdictsOf(r, 'baseline'), hv = verdictsOf(r, 'heldout'), planned = (byId(F.methods, (byId(F.challenges, r.challenge) || {}).method) || {}).steps;
371
- const cell = (label, value, extra, title) => { const d = node('div'); d.append(node('span', label), node('b', value)); if (extra) d.append(node('small', extra)); if (title) d.title = title; return d; };
372
- const strip = node('div', null, 'strip');
373
- strip.append(cell('held-out before', frac(b), withSe(b), 'One trial; ± is one binomial standard error.'), cell('held-out after', hv ? frac(hv) : '—', r.delta_pp != null ? `Δ ${signed(r.delta_pp)} pp${noise(r) ? ', within noise' : ''}` : withSe(hv)),
374
- cell('base-model gate', frac(g), g ? pct(g.pass_rate) : '', GATE_DEF), cell('training steps', `${steps.length}${planned ? ' of ' + planned : ''}`, stageOf(r, 'training').state === 'failed' ? 'failed' : ''),
375
- cell('time-outs · sandbox errors', sumOf(r, ['baseline', 'heldout'], 'timeout') == null ? '—' : `${sumOf(r, ['baseline', 'heldout'], 'timeout')} · ${sumOf(r, ['baseline', 'heldout'], 'error')}`, '', 'Held-out tasks (before and after) that hit the time limit · that lost their sandbox.'), cell('cost', usd(job && job.cost_usd), job ? `${job.flavor} · billed ${duration(job.seconds)}` : ''));
376
- c.append(h, strip);
377
- const why = runReason(r); if (!ACTIVE.includes(r.state) && r.state !== 'scored' && why) { const w = node('div', null, 'why'); w.append(node('span', platformStop(r) ? 'platform fault' : 'why it stopped', platformStop(r) ? '' : 'muted'), node('code', why)); w.title = (platformStop(r) ? 'The arena’s own machinery failed (serving, sandboxes or the agent handshake); this says nothing about the collection. ' : '') + why; c.append(w); }
378
- c.classList.add('clickable'); c.onclick = () => { openRuns.add(r.run_id); location.hash = '#metrics'; setTimeout(() => { const row = document.getElementById('run-' + r.run_id); row && row.scrollIntoView({ block: 'center' }); }, 60); };
379
- return c;
380
- });
381
- }
382
  function heldoutCharts() {
383
  const grid = node('div', null, 'grid3 fit');
384
  for (const c of openChallenges()) {
@@ -412,28 +373,9 @@ function trainingCharts(tags) {
412
  }
413
  return grid;
414
  }
415
- function results() {
416
- const cs = F.challenges, runs = shown(), c = card('results', 'Δ in percentage points (held-out after − before), mean of scored runs per collection and challenge');
417
- const head = node('tr'); head.append(node('th', 'collection'));
418
- for (const ch of cs) { const th = node('th'); th.append(node('span', ch.id), node('span', bindingLine(ch) + (ch.status === 'planned' ? ' · planned' : ''), 'two')); head.append(th); }
419
- const t = node('table'); t.append(head);
420
- for (const e of F.collections) {
421
- const tr = node('tr'); tr.append(td(two(e.source_url ? ext(e.title || e.id, e.source_url, 'link') : e.title || e.id, `${tasks(e.task_count)} · ${e.author || ''}`)));
422
- for (const ch of cs) {
423
- if (ch.status === 'planned') { tr.append(td('—', 'muted')); continue; }
424
- const mine = runs.filter(r => r.challenge === ch.id && r.environment_id === e.id).sort(byStart), scored = mine.filter(r => r.delta_pp != null);
425
- if (scored.length) { const mean = scored.reduce((a, r) => a + r.delta_pp, 0) / scored.length; tr.append(td(two(node('span', `${signed(mean)} pp`, 'num'), `mean of ${scored.length}`))); }
426
- else if (mine.length) tr.append(td(two(`${mine.length} run${mine.length === 1 ? '' : 's'}, none scored`, 'last ' + stateText(mine[mine.length - 1]))));
427
- else tr.append(td('—', 'muted'));
428
- }
429
- t.append(tr);
430
- }
431
- const box = node('div', null, 'scroll'); box.append(t); c.append(box); return c;
432
- }
433
  // Feed of the latest run + per-collection table (MiMo's dynamic sampler block).
434
- function samplerBlock() {
435
  const runs = shown().sort(byStart), r = [...runs].reverse().find(x => (x.feed || []).length);
436
- const split = node('div', null, 'split');
437
  const f = card('feed', r ? `run ${code(r)} · ${nameOf(r)} · ${r.challenge}` : ''), ul = node('ul', null, 'log feed');
438
  const events = [...((r && r.feed) || [])];
439
  if (r && !ACTIVE.includes(r.state) && r.state !== 'scored') events.push({ t: r.ended_at, stage: r.stage, event: 'stopped', note: runReason(r) });
@@ -445,16 +387,7 @@ function samplerBlock() {
445
  }
446
  if (!ul.childElementCount) { const li = node('li', null, 'muted'); li.style.display = 'block'; li.textContent = 'No events yet.'; ul.append(li); }
447
  f.append(ul);
448
- const cols = (METRICS[openChallenges()[0] && openChallenges()[0].id] || {}).collections || [];
449
- const tc = card('collections', 'base-model gate from each collection’s latest run'); tc.title = GATE_DEF;
450
- const rows = F.collections.map(e => {
451
- const cr = cols.find(x => x.environment_id === e.id) || {}, g = cr.gate, tr = node('tr');
452
- const gate = node('span', g ? `${g.pass} / ${g.total}` : '—', 'num'); if (g) { const b = node('span', null, 'bar-in'), i = node('i'); i.style.width = Math.round(100 * g.pass / Math.max(1, g.total)) + '%'; b.append(i); gate.append(b, node('span', ' run ' + g.run_id.replace('challenge-', '').slice(0, 8), 'muted')); }
453
- tr.append(td(two(e.title || e.id, `${e.id} · ${e.author || ''}`)), td(String(e.task_count ?? '—'), 'r num'), td(gate, 'nw'), td(String(cr.runs ?? 0), 'r num opt'), td(cr.in_flight ? `${STAGE[cr.in_flight.stage] || cr.in_flight.state}` : '—', cr.in_flight ? 'opt' : 'muted opt'));
454
- return tr;
455
- });
456
- tc.append(table([['collection'], ['tasks', 'r'], ['base-model gate'], ['runs', 'r opt'], ['in flight', 'opt']], rows));
457
- split.append(f, tc); return split;
458
  }
459
  function compute() {
460
  if (!JOBS) return node('div');
@@ -465,7 +398,7 @@ function compute() {
465
  const kinds = Object.entries(byKind).sort((a, b) => b[1] - a[1]);
466
  const bud = JOBS.budget;
467
  const spendNote = [
468
- bud ? `Committed ${usd(bud.committed_usd)} is what the launch guard counts: the larger of HF billed ${usd(bud.hf_recorded_usd)}${bud.active_reservations_usd ? ' plus ' + usd(bud.active_reservations_usd) + ' still reserved by live runs' : ''} and the reservation ledger ${usd(bud.ledger_committed_usd)} (which includes a hand-set allowance for compute spent before the ledger existed). ${usd(bud.remaining_usd)} is left for new runs.` : '',
469
  `By purpose: ${kinds.slice(0, 5).map(([k, v]) => `${k} ${usd(v)}`).join(' · ')}${kinds.length > 5 ? ' · other ' + usd(kinds.slice(5).reduce((a, [, v]) => a + v, 0)) : ''}.`,
470
  `${JOBS.basis} x axis: job start time, UTC.`].filter(Boolean);
471
  split.append(chartCard('compute/cost_usd', 'cumulative HF billed, all PostTrain jobs', [['--s-1', 'HF billed', usd(JOBS.spent_usd)], [null, 'committed', bud ? usd(bud.committed_usd) : '—'], [null, 'left for runs', bud ? usd(bud.remaining_usd) : usd(JOBS.cap_usd - JOBS.spent_usd)], [null, 'cap', usd(JOBS.cap_usd), 'rule']], 220, {
@@ -478,17 +411,114 @@ function compute() {
478
  c.append(table([['job'], ['hardware', 'opt'], ['HF job state'], ['started', 'opt'], ['duration', 'r'], ['cost', 'r']], list.map(j => {
479
  const tr = node('tr'), st = node('span', null, 'nw'); st.append(j.stage === 'RUNNING' || j.stage === 'SCHEDULING' ? node('span', null, 'dot live') : node('span', { COMPLETED: '✓', ERROR: '✕', CANCELED: '–' }[j.stage] || '·', 'glyph'), j.stage.toLowerCase());
480
  const run = runOf[j.run_id], outcome = run ? `pipeline: ${stateText(run)}` : '';
481
- tr.append(td(two(ext(j.name, j.url, 'link'), [j.kind, j.detail].filter(Boolean).join(' · '))), td(j.flavor, 'soft nw opt'), td(two(st, outcome)), td(when(j.started_at || j.created_at), 'soft nw opt'), td(duration(j.seconds), 'r num soft'), td(usd(j.cost_usd), 'r num'));
 
 
482
  return tr;
483
  })));
484
  if (jobs.length > 8) { const b = node('button', allJobs ? 'fewer' : `all ${jobs.length} jobs`, 'more'); b.type = 'button'; b.onclick = () => { allJobs = !allJobs; render(); }; c.append(b); }
485
  split.append(c); return split;
486
  }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
487
  function renderOverview() {
488
  const v = $('#v-overview'); v.textContent = '';
489
- v.append(hillClimbs(), sec('runs', `the latest ${showTests ? 'runs, including organizer test runs (toggle on the metrics tab)' : 'participant runs'}; runs in progress first; select one for its stages`), ...runCards(), sec('held-out', 'pass rate before and after training, per challenge'), heldoutCharts(), results(),
490
- sec('training', 'latest runs, per optimizer step'), trainingCharts(TRAIN_TAGS), sec('data', 'the latest run’s events and every submitted collection'), samplerBlock(),
491
- sec('compute', 'every PostTrain job on HF, priced at HF’s rate'), compute());
 
 
492
  }
493
 
494
  // ── metrics: runs with details and per-run health ─────────────────────
@@ -525,7 +555,7 @@ function renderMetrics() {
525
  tr.id = 'run-' + r.run_id; if (openRuns.has(r.run_id)) tr.classList.add('on');
526
  rows.push(tr, dr);
527
  }
528
- if (!runs.length) { const tr = node('tr'), cell = td('No runs yet.', 'muted'); cell.colSpan = cols.length; tr.append(cell); rows.push(tr); }
529
  c.append(table(cols, rows)); v.append(c);
530
  const grid = node('div', null, 'grid4');
531
  for (const [title, sub, get] of HEALTH) {
@@ -534,7 +564,8 @@ function renderMetrics() {
534
  }
535
  if (grid.childElementCount) v.append(grid);
536
  }
537
- v.append(sec('training', 'every logged tag, latest runs by run code'), trainingCharts([...TRAIN_TAGS, ['train/loss', 'loss'], ['train/clip_ratio', 'clip_ratio/region_mean'], ['train/is_logp_diff', 'sampling/sampling_logp_difference/mean']]));
 
538
  }
539
  function detail(r) {
540
  const outer = node('div', null, 'detail'), clip = node('div'), box = node('div', null, 'in'); outer.append(clip); clip.append(box);
@@ -556,7 +587,7 @@ function detail(r) {
556
  function renderAbout() {
557
  const v = $('#v-about'); v.textContent = '';
558
  const p = card('about'), prose = node('div', null, 'prose');
559
- prose.innerHTML = '<p>PostTrain Arena ranks RL environment collections by how much they improve a model: <code>Δ = PostTrain(M, D_train, D_eval; θ_method)</code>. A challenge fixes the model, the recipe and sealed held-out suites; a submission changes only the training data, and Δ is the held-out pass rate after training minus before, measured in the same run.</p><p>Agents submit and start runs through the <a href="/AGENTS.md">agent API</a>. The shared <a href="/board">board</a> is for discussion.</p>';
560
  p.append(prose); v.append(p);
561
  const g = card('terms'), dl = node('dl', null, 'hc-facts gl');
562
  for (const [k, d] of [['challenge', 'A fixed model, recipe and sealed held-out suites. A submission changes only the training data.'], ['sealed suite', 'Held-out tasks participants never see; per-task results stay private, aggregates are published.'], ['held-out before / after', 'The model’s pass rate on the sealed suite before and after training, measured inside the same run.'], ['Δ, pp', 'Held-out after minus before, in percentage points. One trial per suite unless the recipe says otherwise; ± is a standard error.'], ['pass@1', 'Share of tasks solved on the first attempt.'], ['reference', 'The base model’s pass rate from separate organizer evaluations, for scale; Δ does not use it.'], ['base-model gate', GATE_BODY], ['task-quality gates', 'Checks on submitted tasks (reference solution passes, doing nothing fails, the model sometimes solves it); different from the base-model gate.'], ['time-out', 'The agent hit the per-task time limit; counted as a failure.'], ['sandbox error', 'The task’s sandbox or agent connection failed. Counted as a failure up to a cap (10% of an evaluation’s tasks in the current recipes, 4 of 32), beyond which the run stops.'], ['platform fault', 'A run stopped by the arena’s own machinery (serving, sandboxes, the agent handshake), not by anything in the collection.'], ['smoke test', 'A challenge that proves the pipeline end to end; its scores are not evidence.'], ['run code', 'The last characters of a run id, used on cards, charts and tables.']]) dl.append(node('dt', k), node('dd', d));
@@ -588,7 +619,7 @@ function route() {
588
  document.querySelectorAll('.tabs a').forEach(a => a.dataset.v === view ? a.setAttribute('aria-current', 'page') : a.removeAttribute('aria-current'));
589
  render();
590
  }
591
- function render() { if (!F) return; renderHeader(); const on = document.querySelector('.view.on'); if (!on) return; ({ 'v-overview': renderOverview, 'v-metrics': renderMetrics, 'v-about': renderAbout })[on.id](); }
592
  window.addEventListener('hashchange', route);
593
  let resizeTimer; window.addEventListener('resize', () => { clearTimeout(resizeTimer); resizeTimer = setTimeout(render, 150); });
594
  setInterval(tickClock, 1000); tickClock(); route(); load(); setInterval(load, 30000);
 
7
  <link rel="icon" href="/icon.svg" type="image/svg+xml">
8
  <link rel="preconnect" href="https://fonts.googleapis.com">
9
  <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
10
+ <link href="https://fonts.googleapis.com/css2?family=Inter:wght@400;500;600&family=Instrument+Serif&family=JetBrains+Mono:wght@400;500&display=swap" rel="stylesheet">
11
  <style>
12
  /* Layout follows the MiMo RL dashboard (mimo.xiaomi.com/rl); tokens follow posttrain.com: white paper, black ink, one blue, hairlines, square corners. */
13
  :root {
 
108
  .tag { display: inline-block; font-size: 10.5px; letter-spacing: .05em; text-transform: uppercase; color: var(--mute); white-space: nowrap; }
109
  .tag.on { color: var(--blue); }
110
  .cnote { padding: 0 14px 12px; font-size: 12px; color: var(--mute); line-height: 1.5; }
 
 
111
  .card.clickable { cursor: pointer; } .card.clickable:hover .run-h .name { color: var(--blue); }
112
  td.strong { font-weight: 600; }
113
  .hc-facts.gl { grid-template-columns: 170px minmax(0, 1fr); }
 
123
  .scroll { overflow-x: auto; background: linear-gradient(to right, var(--paper) 30%, transparent) left / 24px 100% no-repeat local, linear-gradient(to left, var(--paper) 30%, transparent) right / 24px 100% no-repeat local, linear-gradient(to right, rgba(8, 8, 8, .10), transparent) left / 10px 100% no-repeat scroll, linear-gradient(to left, rgba(8, 8, 8, .10), transparent) right / 10px 100% no-repeat scroll; }
124
  .grid4 { display: grid; grid-template-columns: repeat(auto-fit, minmax(300px, 1fr)); gap: 12px; }
125
  .kpi.only-s { display: none; }
126
+ .lede { margin: 4px 0 0; max-width: 860px; font-size: 15px; line-height: 1.6; color: var(--ink-soft); } .lede a { color: var(--blue); }
127
+ .shelf { display: grid; grid-template-columns: repeat(4, minmax(0, 1fr)); border-top: 1px solid var(--ink); border-bottom: 1px solid var(--line); }
128
+ .shelf > div { padding: 14px 16px 16px 0; min-width: 0; } .shelf > div + div { padding-left: 16px; border-left: 1px solid var(--line); }
129
+ .shelf b { display: block; font: 400 44px/1.05 "Instrument Serif", Georgia, serif; letter-spacing: -0.01em; }
130
+ .shelf span { display: block; margin-top: 6px; font-weight: 500; } .shelf small { display: block; margin-top: 2px; font-size: 12px; color: var(--mute); line-height: 1.45; }
131
+ .now { font-size: 12.5px; margin-top: -8px; } .now .dot { vertical-align: 0; }
132
+ .status { margin: -12px 0 0; max-width: 980px; font-size: 13px; line-height: 1.55; }
133
+ .track { display: grid; grid-template-columns: repeat(7, 50px); gap: 3px; }
134
+ .track i { height: 22px; display: flex; align-items: center; justify-content: center; font: normal 10.5px var(--mono); color: var(--mute); border: 1px solid var(--line); }
135
+ .track i.done { background: var(--blue); border-color: var(--blue); color: #fff; }
136
+ .track i.running { border-color: var(--blue); color: var(--blue); animation: blink 1.6s ease-in-out infinite; }
137
+ .track i.failed { border-color: var(--ink); color: var(--ink); font-weight: 500; } .track i.canceled { border-style: dashed; border-color: var(--ink-soft); color: var(--ink-soft); }
138
+ .track.head i { border: 0; height: auto; color: var(--mute); font-family: var(--sans); font-size: 11px; cursor: help; }
139
+ .fault { color: var(--ink); font-weight: 500; }
140
  .cnote div + div { margin-top: 4px; }
141
 
142
  /* runs list (metrics view) */
 
174
  .wrap { padding: 0 14px; } .kpi.hide-s, .brand span { display: none; } .bar { gap: 16px; } .log li, .feed li { grid-template-columns: 64px minmax(0, 1fr); } .feed li > :last-child { display: none; } .strip { grid-template-columns: repeat(3, minmax(0, 1fr)); }
175
  .strip > div:nth-child(4) { border-left: 0; } .strip > div:nth-child(n+4) { border-top: 1px solid var(--line-soft); } .strip > div { padding: 8px 10px 10px; } .strip small { display: block; margin-left: 0; overflow-wrap: anywhere; }
176
  .kpi.only-s { display: flex; } .grid4 { grid-template-columns: minmax(0, 1fr); }
177
+ .shelf { grid-template-columns: repeat(2, minmax(0, 1fr)); } .shelf > div:nth-child(3) { padding-left: 0; border-left: 0; } .shelf > div:nth-child(n+3) { border-top: 1px solid var(--line); } .shelf b { font-size: 34px; }
178
+ .track { grid-template-columns: repeat(7, 34px); gap: 2px; } .track i { font-size: 9.5px; }
179
  .grid3, .grid3.fit, .hc-grid { grid-template-columns: minmax(0, 1fr); } .view.on > * { min-width: 0; } .split { grid-template-columns: 1fr; } .opt, .hide-s { display: none; } .show-s { display: block; } .hc-facts.gl { grid-template-columns: 1fr; gap: 0 16px; } .hc-facts.gl dd { margin-bottom: 8px; } .kpi b { font-size: 11.5px; }
180
  th, td { padding-left: 8px; padding-right: 8px; } th:first-child, td:first-child { padding-left: 14px; }
181
  .row { grid-template-columns: minmax(0, 1fr) 90px 44px 12px; } .row .st { display: none; }
 
217
  const PLATFORM = /nccl|cuda|out of memory|vllm|bridge|relay|daytona|sandbox|acp initialize|handshake|snapshot-|pty|connection (reset|refused)|rate limit|429|no healthy scored rollout|infra/i;
218
  const SERIES = ['--s-1', '--s-2', '--s-3'];
219
 
220
+ let F = null, METRICS = {}, BOARDS = {}, LOADED = false, JOBS = null, showTests = localStorage.getItem('pta.tests') !== '0', allJobs = false;
221
  const openRuns = new Set();
222
 
223
  // ── data ──────────────────────────────────────────────────────────────
 
233
  if (ver.status === 'fulfilled') { const b = ver.value; $('#build').textContent = `build ${b.build}${b.stale ? ' · files changed since the server started; restart it' : ''}`; }
234
  if (f.status === 'fulfilled') F = f.value; else failed = true;
235
  if (j.status === 'fulfilled') JOBS = j.value; else failed = true;
236
+ await Promise.all(openChallenges().map(async c => { const base = '/api/challenges/' + encodeURIComponent(c.id); try { METRICS[c.id] = await getJSON(base + '/metrics'); } catch { failed = true; } try { BOARDS[c.id] = await getJSON(base + '/leaderboard'); } catch { failed = true; } }));
237
+ LOADED = true;
238
  $('#err').hidden = !failed; $('#err').textContent = failed ? 'Some data is unavailable right now; the page retries every 30 seconds.' : '';
239
  loading = false; render();
240
  }
 
281
  function verdictLine(v) { const b = [`${v.pass} of ${v.total} passed`]; if (v.timeout) b.push(`${v.timeout} timed out`); if (v.error) b.push(`${v.error} infra error${v.error === 1 ? '' : 's'}`); if (v.done < v.total) b.push(`${v.total - v.done} not run`); return b.join(' · '); }
282
  const td = (content, cls) => { const c = node('td', null, cls || ''); if (content instanceof Node) c.append(content); else c.textContent = content ?? '—'; return c; };
283
  const two = (a, b) => { const s = node('span'); s.append(a instanceof Node ? a : node('span', a)); if (b) s.append(node('span', b, 'two')); return s; };
284
+ function table(cols, rows) { const box = node('div', null, 'scroll'), t = node('table'), tr = node('tr'); for (const [l, c] of cols) { const th = node('th', null, c || ''); th.append(l); tr.append(th); } t.append(tr); rows.forEach(r => t.append(r)); box.append(t); return box; }
285
  function card(title, sub, tools) { const c = node('div', null, 'card'), h = node('div', null, 'ch'), t = node('div'); t.append(node('b', title)); if (sub) { t.append(' '); t.append(node('span', sub)); } h.append(t); if (tools) h.append(tools); c.append(h); return c; }
286
  const sec = (title, sub) => { const h = node('h3', title, 'sec'); if (sub) h.append(node('span', sub)); return h; };
287
 
 
340
  }
341
 
342
  // ── overview ──────────────────────────────────────────────────────────
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
343
  function heldoutCharts() {
344
  const grid = node('div', null, 'grid3 fit');
345
  for (const c of openChallenges()) {
 
373
  }
374
  return grid;
375
  }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
376
  // Feed of the latest run + per-collection table (MiMo's dynamic sampler block).
377
+ function feedCard() {
378
  const runs = shown().sort(byStart), r = [...runs].reverse().find(x => (x.feed || []).length);
 
379
  const f = card('feed', r ? `run ${code(r)} · ${nameOf(r)} · ${r.challenge}` : ''), ul = node('ul', null, 'log feed');
380
  const events = [...((r && r.feed) || [])];
381
  if (r && !ACTIVE.includes(r.state) && r.state !== 'scored') events.push({ t: r.ended_at, stage: r.stage, event: 'stopped', note: runReason(r) });
 
387
  }
388
  if (!ul.childElementCount) { const li = node('li', null, 'muted'); li.style.display = 'block'; li.textContent = 'No events yet.'; ul.append(li); }
389
  f.append(ul);
390
+ return f;
 
 
 
 
 
 
 
 
 
391
  }
392
  function compute() {
393
  if (!JOBS) return node('div');
 
398
  const kinds = Object.entries(byKind).sort((a, b) => b[1] - a[1]);
399
  const bud = JOBS.budget;
400
  const spendNote = [
401
+ bud ? `Committed ${usd(bud.committed_usd)}: HF has billed ${usd(bud.hf_recorded_usd)}${bud.active_reservations_usd ? ' and live runs still hold ' + usd(bud.active_reservations_usd) : ''}, while the arena’s own spend record says ${usd(bud.ledger_committed_usd)} because it also counts compute spent before the record began. The larger figure counts against the cap, so ${usd(bud.remaining_usd)} is left for new runs.` : '',
402
  `By purpose: ${kinds.slice(0, 5).map(([k, v]) => `${k} ${usd(v)}`).join(' · ')}${kinds.length > 5 ? ' · other ' + usd(kinds.slice(5).reduce((a, [, v]) => a + v, 0)) : ''}.`,
403
  `${JOBS.basis} x axis: job start time, UTC.`].filter(Boolean);
404
  split.append(chartCard('compute/cost_usd', 'cumulative HF billed, all PostTrain jobs', [['--s-1', 'HF billed', usd(JOBS.spent_usd)], [null, 'committed', bud ? usd(bud.committed_usd) : '—'], [null, 'left for runs', bud ? usd(bud.remaining_usd) : usd(JOBS.cap_usd - JOBS.spent_usd)], [null, 'cap', usd(JOBS.cap_usd), 'rule']], 220, {
 
411
  c.append(table([['job'], ['hardware', 'opt'], ['HF job state'], ['started', 'opt'], ['duration', 'r'], ['cost', 'r']], list.map(j => {
412
  const tr = node('tr'), st = node('span', null, 'nw'); st.append(j.stage === 'RUNNING' || j.stage === 'SCHEDULING' ? node('span', null, 'dot live') : node('span', { COMPLETED: '✓', ERROR: '✕', CANCELED: '–' }[j.stage] || '·', 'glyph'), j.stage.toLowerCase());
413
  const run = runOf[j.run_id], outcome = run ? `pipeline: ${stateText(run)}` : '';
414
+ const kindText = node('span', [j.kind, j.detail].filter(Boolean).join(' · '), 'two'); if (j.purpose) kindText.title = j.purpose;
415
+ const name = node('span'); name.append(ext(j.name, j.url, 'link'), kindText);
416
+ tr.append(td(name), td(j.flavor, 'soft nw opt'), td(two(st, outcome)), td(when(j.started_at || j.created_at), 'soft nw opt'), td(duration(j.seconds), 'r num soft'), td(usd(j.cost_usd), 'r num'));
417
  return tr;
418
  })));
419
  if (jobs.length > 8) { const b = node('button', allJobs ? 'fewer' : `all ${jobs.length} jobs`, 'more'); b.type = 'button'; b.onclick = () => { allJobs = !allJobs; render(); }; c.append(b); }
420
  split.append(c); return split;
421
  }
422
+ // ── overview: what the arena is, where it stands, what has been tried ─────────
423
+ const ORDER = ['setup', 'snapshot', 'baseline', 'gate', 'training', 'heldout', 'collect'];
424
+ const TRACK = [['setup', 'setup', 'The GPU job starts and the model server comes up.'], ['snapshot', 'tasks', 'The training and held-out tasks are pinned.'], ['baseline', 'before', 'Held-out pass rate before training.'],
425
+ ['gate', 'gate', GATE_BODY], ['training', 'train', 'GRPO optimizer steps on the collection.'], ['heldout', 'after', 'Held-out pass rate after training.'], ['collect', 'scored', 'The result was collected and checked.']];
426
+ const reach = (r) => (r.stages || []).reduce((k, s) => ['done', 'failed', 'running', 'canceled'].includes(s.state) ? Math.max(k, ORDER.indexOf(s.key)) : k, -1);
427
+ const openRun = (r) => { openRuns.add(r.run_id); location.hash = '#metrics'; setTimeout(() => { const row = document.getElementById('run-' + r.run_id); row && row.scrollIntoView({ block: 'center' }); }, 60); };
428
+ const meanDelta = (rs) => rs.length ? rs.reduce((a, r) => a + r.delta_pp, 0) / rs.length : null;
429
+ function lede() {
430
+ const p = node('p', null, 'lede');
431
+ p.append('PostTrain Arena ranks RL environment collections by how much they improve a model. A challenge fixes the model, the training recipe and a sealed held-out task set; each run trains on one submitted collection and scores Δ, the held-out pass rate after training minus before. ', ext('How to enter ↗', '/AGENTS.md'));
432
+ return p;
433
+ }
434
+ function shelf() {
435
+ const runs = allRuns(), part = runs.filter(r => r.source !== 'organizer'), scored = part.filter(r => r.delta_pp != null);
436
+ const furthest = runs.reduce((a, r) => !a || reach(r) > reach(a) ? r : a, null);
437
+ const open = openChallenges()[0], M = open && METRICS[open.id], ref = M && M.reference, bud = JOBS && JOBS.budget;
438
+ const cell = (value, label, note) => { const d = node('div'); d.append(node('b', value), node('span', label)); if (note) d.append(node('small', note)); return d; };
439
+ const box = node('div', null, 'shelf');
440
+ box.append(
441
+ cell(`${scored.length} of ${part.length}`, 'challenge runs scored', scored.length ? `mean Δ ${signed(meanDelta(scored))} pp` : 'No run has finished training and been scored yet.'),
442
+ cell(furthest && reach(furthest) >= 0 ? STAGE[ORDER[reach(furthest)]] : '—', 'furthest stage any run reached', furthest ? `run ${code(furthest)}, ${when(startOf(furthest), false)}: ${stateText(furthest)}${platformStop(furthest) ? ' (platform fault)' : ''}` : ''),
443
+ cell(ref ? pct(ref.pass_rate) : '—', 'untrained model on the held-out set', ref && open ? `${shortModel(open.model)}, ${ref.trials} trials, ± ${(100 * ref.stderr).toFixed(1)}: where a run starts from` : ''),
444
+ cell(bud ? usd(bud.remaining_usd) : '—', 'compute left', bud ? `of ${usd(bud.cap_usd)}${open && open.run_reserves_usd ? `; a ${open.id} run reserves up to ${usd(open.run_reserves_usd)}, so ${Math.max(0, Math.floor(bud.remaining_usd / open.run_reserves_usd))} more run${Math.floor(bud.remaining_usd / open.run_reserves_usd) === 1 ? ' fits' : 's fit'}` : ''}` : ''));
445
+ const live = JOBS ? JOBS.jobs.filter(j => j.stage === 'RUNNING' || j.stage === 'SCHEDULING') : [];
446
+ const now = node('div', null, 'now'); now.append(node('span', null, 'dot' + (live.length ? ' live' : '')), node('span', 'running now: ', 'muted'), live.length ? live.map(j => `${j.name} (${j.kind})`).join(', ') : 'nothing');
447
+ const out = [box, now];
448
+ if (open && open.status_note) { const n = node('p', null, 'status'); n.append(node('span', `${open.id} status: `, 'muted'), open.status_note); out.push(n); }
449
+ return out;
450
+ }
451
+ function challengesTable() {
452
+ const c = node('div', null, 'card'), notes = [];
453
+ const rows = F.challenges.map(ch => {
454
+ const m = byId(F.models, ch.model), meth = byId(F.methods, ch.method), suites = (ch.suites || []).map(id => byId(F.suites, id)).filter(Boolean);
455
+ const planned = ch.status === 'planned', acc = ch.accepting || {}, runs = allRuns().filter(r => r.challenge === ch.id && r.source !== 'organizer'), scored = runs.filter(r => r.delta_pp != null);
456
+ const n = suites.reduce((a, s) => a + (s.task_count || 0), 0);
457
+ const tr = node('tr');
458
+ tr.append(td(two(node('b', ch.id), planned ? 'planned' : ch.role || 'open')),
459
+ td(two(`${m ? m.repo_id.split('/').pop() : ch.model} on ${suites.map(s => s.name).join(' + ')}`, `${n} sealed tasks, ${meth && meth.trials > 1 ? meth.trials + ' trials' : 'one trial'}; participants never see them`)),
460
+ td(two(meth ? meth.id : ch.method, meth && meth.steps ? `GRPO with LoRA: ${meth.steps} step${meth.steps === 1 ? '' : 's'} × ${meth.tasks_per_step} task${meth.tasks_per_step === 1 ? '' : 's'} × ${meth.group_size} rollouts; ${Math.round(meth.agent_timeout_sec / 60)} min per task` : ''), 'opt'),
461
+ td(two(planned ? 'not open yet' : acc.accepting_runs === false ? 'not accepting runs now' : 'open for runs', planned ? ch.open_note || '' : acc.accepting_runs === false ? acc.reason : `${ch.compute} per run`)),
462
+ td(two(scored.length ? `${signed(meanDelta(scored))} pp` : '—', `${runs.length} run${runs.length === 1 ? '' : 's'}, ${scored.length} scored`), 'r'));
463
+ if (ch.role === 'smoke test' && meth && meth.note) notes.push(`${ch.id} is a smoke test: ${meth.note.charAt(0).toLowerCase() + meth.note.slice(1)}`);
464
+ const M = METRICS[ch.id], ref = M && M.reference;
465
+ if (ref && n) { const se = 100 * Math.sqrt(2 * ref.pass_rate * (1 - ref.pass_rate) / n); notes.push(`On ${ch.id}, one run’s Δ has a standard error of about ${se.toFixed(0)} pp at the base model’s pass rate, so a single run must move the score by about ${(1.96 * se).toFixed(0)} pp before it stands out from noise; the leaderboard averages a collection’s runs.`); }
466
+ return tr;
467
+ });
468
+ c.append(table([['challenge'], ['model, held-out set'], ['training recipe', 'opt'], ['status'], ['mean Δ', 'r']], rows));
469
+ if (notes.length) { const n = node('div', null, 'cnote'); for (const t of notes) n.append(node('div', t)); n.style.paddingTop = '10px'; c.append(n); }
470
+ return c;
471
+ }
472
+ function attempts() {
473
+ const runs = allRuns().sort(byStart).reverse(), c = node('div', null, 'card');
474
+ const head = node('div', null, 'track head'); for (const [, label, def] of TRACK) { const i = node('i', label); i.title = def; head.append(i); }
475
+ const rows = runs.map(r => {
476
+ const tr = node('tr', null, 'pick'), track = node('div', null, 'track'), job = jobOf(r);
477
+ for (const [k] of TRACK) {
478
+ const st = stageOf(r, k), v = verdictsOf(r, k), state = st.state && st.state !== 'unreached' ? st.state : '';
479
+ const count = v && v.total && state !== 'running' ? `${v.pass}/${v.total}` : '', i = node('i', state === 'failed' ? (count ? '✕' + count : '✕') : count || ({ canceled: '–', running: '·' }[state] || ''), state);
480
+ i.title = `${STAGE[k]}: ${state || 'not reached'}${v && v.total ? ' · ' + verdictLine(v) : ''}`; track.append(i);
481
+ }
482
+ const why = runReason(r), fault = platformStop(r), out = node('span');
483
+ out.append(node('span', stateText(r)), fault ? node('span', ' · platform fault', 'fault') : '');
484
+ if (why && !ACTIVE.includes(r.state) && r.state !== 'scored') { const w = node('span', why.length > 220 ? why.slice(0, 220) + '…' : why, 'two'); w.title = why; out.append(w); }
485
+ tr.append(td(two(node('b', code(r), 'num'), `${r.source === 'organizer' ? 'organizer test run' : nameOf(r)} · ${when(startOf(r), false)}`)), td(track, 'nw'), td(out), td(usd(job && job.cost_usd), 'r num opt'));
486
+ tr.onclick = () => openRun(r); tr.title = 'Open this run’s stages, errors and logs';
487
+ return tr;
488
+ });
489
+ c.append(table([['run'], [head], ['outcome'], ['cost', 'r opt']], rows.length ? rows : [(() => { const tr = node('tr'), cell = td('No runs yet.', 'muted'); cell.colSpan = 4; tr.append(cell); return tr; })()]));
490
+ const n = node('div', null, 'cnote'); n.style.paddingTop = '10px';
491
+ const known = new Set(runs.map(r => r.run_id)), earlier = JOBS ? JOBS.jobs.filter(j => j.kind === 'organizer test run' && !known.has(j.run_id)).length : 0;
492
+ if (earlier) n.append(node('div', `Earlier organizer test jobs (${earlier}, from before runs were tracked stage by stage) appear only in the jobs list on the metrics tab.`));
493
+ n.append(node('div', 'Each square is a pipeline stage, filled when the run got through it; ✕ marks where it stopped. Numbers are tasks passed: held-out tasks for before and after, the collection’s first 32 training tasks for the gate. A platform fault is the arena’s own machinery failing (serving, sandboxes, the agent handshake), not the collection. Organizer test runs are the same pipeline, run before the participant path opened.'));
494
+ c.append(n); return c;
495
+ }
496
+ function collectionsTable() {
497
+ const open = openChallenges()[0], board = (open && BOARDS[open.id]) || { rows: [] }, rank = Object.fromEntries(board.rows.map(r => [r.environment_id, r]));
498
+ const cols = (METRICS[open && open.id] || {}).collections || [], runs = allRuns().filter(r => r.source !== 'organizer');
499
+ const c = node('div', null, 'card');
500
+ const order = [...F.collections].sort((a, b) => (rank[a.id] ? rank[a.id].rank : 1e9) - (rank[b.id] ? rank[b.id].rank : 1e9));
501
+ const rows = order.map(e => {
502
+ const cr = cols.find(x => x.environment_id === e.id) || {}, g = cr.gate, mine = runs.filter(r => r.environment_id === e.id).sort(byStart), ranked = rank[e.id], last = mine[mine.length - 1];
503
+ const gate = node('span', g ? `${g.pass} / ${g.total}` : '—', 'num'); if (g) { const b = node('span', null, 'bar-in'), i = node('i'); i.style.width = Math.round(100 * g.pass / Math.max(1, g.total)) + '%'; b.append(i); gate.append(b); }
504
+ const tr = node('tr');
505
+ tr.append(td(ranked ? String(ranked.rank) : '—', 'num'), td(two(e.source_url ? ext(e.title || e.id, e.source_url, 'link') : e.title || e.id, `${e.author || ''} · ${e.id}`)), td(String(e.task_count ?? '—'), 'r num'), td(gate, 'nw opt'),
506
+ td(two(`${mine.length} run${mine.length === 1 ? '' : 's'}`, last ? `latest ${code(last)}: ${stateText(last)}` : 'never run')), td(ranked ? two(`${signed(ranked.delta_pp)} pp`, `${ranked.verified_runs} verified run${ranked.verified_runs === 1 ? '' : 's'}${ranked.stderr_pp != null ? ' · ± ' + Number(ranked.stderr_pp).toFixed(1) : ''}`) : '—', 'r num'));
507
+ return tr;
508
+ });
509
+ const gateHead = node('span', 'base-model gate'); gateHead.title = GATE_DEF;
510
+ c.append(table([['rank'], ['collection'], ['tasks', 'r'], [gateHead, 'opt'], ['runs'], ['mean Δ, verified', 'r']], rows));
511
+ const n = node('div', null, 'cnote'); n.style.paddingTop = '10px';
512
+ n.append(node('div', board.rows.length ? `Ranked on ${open.id} by the mean Δ over each collection’s organizer-verified runs, not its best run.` : `Nothing is ranked yet: a collection ranks on ${open ? open.id : 'a challenge'} once an organizer verifies one of its scored runs. The rank uses the mean Δ over verified runs, not the best run.`));
513
+ c.append(n); return c;
514
+ }
515
  function renderOverview() {
516
  const v = $('#v-overview'); v.textContent = '';
517
+ const scored = allRuns().some(r => r.after_pass_rate != null);
518
+ v.append(lede(), ...shelf(), sec('challenges', 'what is being climbed'), challengesTable(),
519
+ sec('run attempts', 'every run of the pipeline, newest first; select one for its stages, errors and logs'), attempts(),
520
+ ...(scored ? [sec('held-out', 'pass rate before and after training, per challenge'), heldoutCharts()] : []),
521
+ sec('leaderboard', 'every submitted environment collection'), collectionsTable());
522
  }
523
 
524
  // ── metrics: runs with details and per-run health ─────────────────────
 
555
  tr.id = 'run-' + r.run_id; if (openRuns.has(r.run_id)) tr.classList.add('on');
556
  rows.push(tr, dr);
557
  }
558
+ if (!runs.length) { const tr = node('tr'), cell = td(METRICS[ch.id] ? 'No runs yet.' : LOADED ? 'Run data is unavailable right now; the page retries every 30 seconds.' : 'Loading runs…', 'muted'); cell.colSpan = cols.length; tr.append(cell); rows.push(tr); }
559
  c.append(table(cols, rows)); v.append(c);
560
  const grid = node('div', null, 'grid4');
561
  for (const [title, sub, get] of HEALTH) {
 
564
  }
565
  if (grid.childElementCount) v.append(grid);
566
  }
567
+ v.append(sec('training', 'every logged tag, latest runs by run code'), trainingCharts([...TRAIN_TAGS, ['train/loss', 'loss'], ['train/clip_ratio', 'clip_ratio/region_mean'], ['train/is_logp_diff', 'sampling/sampling_logp_difference/mean']]),
568
+ sec('events', 'the latest run’s task results, newest first'), feedCard(), sec('compute', 'every PostTrain job on HF, priced at HF’s rate'), compute());
569
  }
570
  function detail(r) {
571
  const outer = node('div', null, 'detail'), clip = node('div'), box = node('div', null, 'in'); outer.append(clip); clip.append(box);
 
587
  function renderAbout() {
588
  const v = $('#v-about'); v.textContent = '';
589
  const p = card('about'), prose = node('div', null, 'prose');
590
+ prose.innerHTML = '<p>PostTrain Arena ranks RL environment collections by how much they improve a model: <code>Δ = PostTrain(M, D_train, D_eval; θ_method)</code>. A challenge fixes the model, the recipe and sealed held-out suites; a submission changes only the training data, and Δ is the held-out pass rate after training minus before, measured in the same run.</p><p><b>How a run works.</b> One GPU job serves the challenge’s base model and pins the collection’s training tasks and the sealed held-out tasks. It measures the held-out pass rate (before), checks the base-model gate on the collection’s first 32 training tasks, trains with the challenge’s GRPO recipe while agents attempt the collection’s tasks in Daytona sandboxes, and measures the held-out pass rate again (after). Collection recomputes both from per-task results; an organizer review makes the result rank. The leaderboard uses the mean Δ over a collection’s verified runs.</p><p>Agents submit and start runs through the <a href="/AGENTS.md">agent API</a>. The shared <a href="/board">board</a> is for discussion.</p>';
591
  p.append(prose); v.append(p);
592
  const g = card('terms'), dl = node('dl', null, 'hc-facts gl');
593
  for (const [k, d] of [['challenge', 'A fixed model, recipe and sealed held-out suites. A submission changes only the training data.'], ['sealed suite', 'Held-out tasks participants never see; per-task results stay private, aggregates are published.'], ['held-out before / after', 'The model’s pass rate on the sealed suite before and after training, measured inside the same run.'], ['Δ, pp', 'Held-out after minus before, in percentage points. One trial per suite unless the recipe says otherwise; ± is a standard error.'], ['pass@1', 'Share of tasks solved on the first attempt.'], ['reference', 'The base model’s pass rate from separate organizer evaluations, for scale; Δ does not use it.'], ['base-model gate', GATE_BODY], ['task-quality gates', 'Checks on submitted tasks (reference solution passes, doing nothing fails, the model sometimes solves it); different from the base-model gate.'], ['time-out', 'The agent hit the per-task time limit; counted as a failure.'], ['sandbox error', 'The task’s sandbox or agent connection failed. Counted as a failure up to a cap (10% of an evaluation’s tasks in the current recipes, 4 of 32), beyond which the run stops.'], ['platform fault', 'A run stopped by the arena’s own machinery (serving, sandboxes, the agent handshake), not by anything in the collection.'], ['smoke test', 'A challenge that proves the pipeline end to end; its scores are not evidence.'], ['run code', 'The last characters of a run id, used on cards, charts and tables.']]) dl.append(node('dt', k), node('dd', d));
 
619
  document.querySelectorAll('.tabs a').forEach(a => a.dataset.v === view ? a.setAttribute('aria-current', 'page') : a.removeAttribute('aria-current'));
620
  render();
621
  }
622
+ function render() { if (!F) { const on = document.querySelector('.view.on'); if (on && !on.childElementCount) on.append(node('div', 'Loading the arena’s runs and results…', 'card none')); return; } renderHeader(); const on = document.querySelector('.view.on'); if (!on) return; ({ 'v-overview': renderOverview, 'v-metrics': renderMetrics, 'v-about': renderAbout })[on.id](); }
623
  window.addEventListener('hashchange', route);
624
  let resizeTimer; window.addEventListener('resize', () => { clearTimeout(resizeTimer); resizeTimer = setTimeout(render, 150); });
625
  setInterval(tickClock, 1000); tickClock(); route(); load(); setInterval(load, 30000);