Spaces:
Running
Running
Xiangyi Li commited on
Commit ·
c42a99e
1
Parent(s): 5e1dd59
Overview tells the story: state, challenges, run attempts, leaderboard
Browse filesTwo fresh-eyes reviews could not tell the arena's stage, who is winning, the
run history or the next step. The overview now opens with one sentence on what
the arena ranks, four headline numbers (runs scored, furthest stage reached,
the untrained model's held-out score, compute left and how many runs fit), the
organizer's status note from the challenge file, a compact challenges table
with the noise threshold, every run as a row of pipeline stages with where and
why it stopped, and the leaderboard with its ranking rule. Training charts,
the task feed, compute and jobs move to the metrics tab; job kinds carry their
purpose; loading states replace false empty states.
- challenges.py +7 -2
- compute_jobs.py +9 -2
- configs/challenges/tb2-9b.toml +2 -0
- index.html +130 -99
challenges.py
CHANGED
|
@@ -51,7 +51,7 @@ def challenge_row(spec):
|
|
| 51 |
suite=compose.fragment('suites',b['suites'][0]);s,meta=suite['suite'],suite.get('meta',{}) # the single-suite pipeline scores the first
|
| 52 |
grpo,runtime,harness=method.get('grpo',{}),method.get('runtime',{}),method.get('harness',{})
|
| 53 |
recipe=spec['recipe'];compute=spec['compute']
|
| 54 |
-
return {'id':spec['id'],'name':spec['name'],'status':spec['status'],'opens':spec.get('opens'),'closes':spec.get('closes'),
|
| 55 |
'role':spec.get('role'),'role_note':spec.get('role_note'),'baseline_file':spec.get('baseline_file'),'binding':dict(b),'summary':spec['summary'],
|
| 56 |
'eval_suite':{'name':meta.get('name',s['name']),'repo_id':s['repo_id'],'revision':s['revision'],'task_list':s['task_list'],
|
| 57 |
'task_count':len(compose.task_ids(b['suites'][0])),'sealed':meta.get('sealed',True),'note':meta.get('note','')},
|
|
@@ -103,7 +103,7 @@ def collection_rows_cached():
|
|
| 103 |
def formula():
|
| 104 |
"""The registries behind each term of the formula, and every challenge that binds them."""
|
| 105 |
challenges=[{'id':c['id'],'name':c['name'],'status':c['status'],'role':c.get('role'),'compute':f"HF {c['compute']['flavor']}",**c['binding'],
|
| 106 |
-
'accepting':{k:v for k,v in health(c).items() if k in ('accepting_runs','reason')}} for c in CHALLENGES]+[dict(p) for p in PLANNED_CHALLENGES]
|
| 107 |
uses=lambda key,value:[c['id'] for c in challenges if value==c.get(key) or value in (c.get(key) or [])]
|
| 108 |
references={}
|
| 109 |
for c in CHALLENGES:
|
|
@@ -150,6 +150,11 @@ def hardware():
|
|
| 150 |
_hardware.update(rows=response.json(),at=time.time())
|
| 151 |
return _hardware['rows']
|
| 152 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 153 |
def quote(row):
|
| 154 |
"""Conservative HF bound for one run: flavor price times the hard job timeout. Daytona is billed separately and estimated only."""
|
| 155 |
try:
|
|
|
|
| 51 |
suite=compose.fragment('suites',b['suites'][0]);s,meta=suite['suite'],suite.get('meta',{}) # the single-suite pipeline scores the first
|
| 52 |
grpo,runtime,harness=method.get('grpo',{}),method.get('runtime',{}),method.get('harness',{})
|
| 53 |
recipe=spec['recipe'];compute=spec['compute']
|
| 54 |
+
return {'id':spec['id'],'name':spec['name'],'status':spec['status'],'opens':spec.get('opens'),'closes':spec.get('closes'),'status_note':spec.get('status_note'),
|
| 55 |
'role':spec.get('role'),'role_note':spec.get('role_note'),'baseline_file':spec.get('baseline_file'),'binding':dict(b),'summary':spec['summary'],
|
| 56 |
'eval_suite':{'name':meta.get('name',s['name']),'repo_id':s['repo_id'],'revision':s['revision'],'task_list':s['task_list'],
|
| 57 |
'task_count':len(compose.task_ids(b['suites'][0])),'sealed':meta.get('sealed',True),'note':meta.get('note','')},
|
|
|
|
| 103 |
def formula():
|
| 104 |
"""The registries behind each term of the formula, and every challenge that binds them."""
|
| 105 |
challenges=[{'id':c['id'],'name':c['name'],'status':c['status'],'role':c.get('role'),'compute':f"HF {c['compute']['flavor']}",**c['binding'],
|
| 106 |
+
'accepting':{k:v for k,v in health(c).items() if k in ('accepting_runs','reason')},'run_reserves_usd':reserve_bound(c),'status_note':c.get('status_note')} for c in CHALLENGES]+[dict(p) for p in PLANNED_CHALLENGES]
|
| 107 |
uses=lambda key,value:[c['id'] for c in challenges if value==c.get(key) or value in (c.get(key) or [])]
|
| 108 |
references={}
|
| 109 |
for c in CHALLENGES:
|
|
|
|
| 150 |
_hardware.update(rows=response.json(),at=time.time())
|
| 151 |
return _hardware['rows']
|
| 152 |
|
| 153 |
+
def reserve_bound(row):
|
| 154 |
+
"""What one run reserves against the cap, or None when prices are unavailable (display only)."""
|
| 155 |
+
try: return round(quote(row)['max_compute_usd'],2)
|
| 156 |
+
except HTTPException: return None
|
| 157 |
+
|
| 158 |
def quote(row):
|
| 159 |
"""Conservative HF bound for one run: flavor price times the hard job timeout. Daytona is billed separately and estimated only."""
|
| 160 |
try:
|
compute_jobs.py
CHANGED
|
@@ -29,12 +29,19 @@ def kind(labels):
|
|
| 29 |
experiment = labels.get('experiment', '')
|
| 30 |
if labels.get('posttrain') == 'baseline-grid': return 'baseline eval'
|
| 31 |
if experiment == 'posttrain-challenge' or labels.get('challenge'): return 'challenge run'
|
| 32 |
-
if experiment.startswith('posttrain-phase'): return 'organizer run'
|
| 33 |
if experiment.startswith('arena-'): return 'arena job'
|
| 34 |
if experiment.startswith('posttrain') or labels.get('posttrain'): return (experiment or labels.get('posttrain')).removeprefix('posttrain-').replace('-', ' ')
|
| 35 |
return None
|
| 36 |
|
| 37 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 38 |
def prices():
|
| 39 |
if time.time() - _prices['at'] > 3600 or not _prices['value']:
|
| 40 |
try:
|
|
@@ -87,7 +94,7 @@ def build():
|
|
| 87 |
rate = price.get(job.flavor)
|
| 88 |
cost = round(math.ceil(seconds / 60) * rate, 2) if seconds is not None and rate is not None else None
|
| 89 |
detail = labels.get('grid_point') or labels.get('phase') or labels.get('round') or labels.get('suite')
|
| 90 |
-
rows.append({'id': job.id, 'name': labels.get('name') or job.id, 'kind': what, 'detail': detail, 'run_id': labels.get('run_id') or labels.get('name'),
|
| 91 |
'challenge': labels.get('challenge'), 'flavor': job.flavor, 'stage': stage, 'created_at': _iso(job.created_at),
|
| 92 |
'started_at': _iso(job.started_at), 'finished_at': _iso(getattr(job, 'finished_at', None)),
|
| 93 |
'seconds': round(seconds) if seconds is not None else None, 'cost_usd': cost, 'url': f'https://huggingface.co/jobs/benchflow/{job.id}'})
|
|
|
|
| 29 |
experiment = labels.get('experiment', '')
|
| 30 |
if labels.get('posttrain') == 'baseline-grid': return 'baseline eval'
|
| 31 |
if experiment == 'posttrain-challenge' or labels.get('challenge'): return 'challenge run'
|
| 32 |
+
if experiment.startswith('posttrain-phase'): return 'organizer test run'
|
| 33 |
if experiment.startswith('arena-'): return 'arena job'
|
| 34 |
if experiment.startswith('posttrain') or labels.get('posttrain'): return (experiment or labels.get('posttrain')).removeprefix('posttrain-').replace('-', ' ')
|
| 35 |
return None
|
| 36 |
|
| 37 |
|
| 38 |
+
# What each kind of job is for, shown next to it on the dashboard.
|
| 39 |
+
PURPOSE = {'challenge run': 'A participant run: trains the challenge model on one submitted collection and measures held-out before and after.',
|
| 40 |
+
'organizer test run': 'An organizer run of the same pipeline, used to make it work before participant runs opened.',
|
| 41 |
+
'baseline eval': 'Evaluation only, no training: measures the untrained model under a harness setting (context length, time limit, agent) to explain the baseline.',
|
| 42 |
+
'arena job': 'A job from the retired experiment runner.'}
|
| 43 |
+
|
| 44 |
+
|
| 45 |
def prices():
|
| 46 |
if time.time() - _prices['at'] > 3600 or not _prices['value']:
|
| 47 |
try:
|
|
|
|
| 94 |
rate = price.get(job.flavor)
|
| 95 |
cost = round(math.ceil(seconds / 60) * rate, 2) if seconds is not None and rate is not None else None
|
| 96 |
detail = labels.get('grid_point') or labels.get('phase') or labels.get('round') or labels.get('suite')
|
| 97 |
+
rows.append({'id': job.id, 'name': labels.get('name') or job.id, 'kind': what, 'purpose': PURPOSE.get(what), 'detail': detail, 'run_id': labels.get('run_id') or labels.get('name'),
|
| 98 |
'challenge': labels.get('challenge'), 'flavor': job.flavor, 'stage': stage, 'created_at': _iso(job.created_at),
|
| 99 |
'started_at': _iso(job.started_at), 'finished_at': _iso(getattr(job, 'finished_at', None)),
|
| 100 |
'seconds': round(seconds) if seconds is not None else None, 'cost_usd': cost, 'url': f'https://huggingface.co/jobs/benchflow/{job.id}'})
|
configs/challenges/tb2-9b.toml
CHANGED
|
@@ -10,6 +10,8 @@ role = "smoke test"
|
|
| 10 |
role_note = "It proves the submission-to-leaderboard loop closes end to end. Collections are compared on larger challenges, such as the planned terminal-35b."
|
| 11 |
summary = "Submit a collection of BenchFlow task environments. The arena post-trains the pinned Qwen3.5-9B on your tasks with the pinned GRPO recipe and reports the pass@1 change on a sealed 32-task Terminal-Bench 2.0 subset."
|
| 12 |
baseline_file = "results/tb2-32-baseline.json"
|
|
|
|
|
|
|
| 13 |
|
| 14 |
[binding]
|
| 15 |
model = "qwen3.5-9b"
|
|
|
|
| 10 |
role_note = "It proves the submission-to-leaderboard loop closes end to end. Collections are compared on larger challenges, such as the planned terminal-35b."
|
| 11 |
summary = "Submit a collection of BenchFlow task environments. The arena post-trains the pinned Qwen3.5-9B on your tasks with the pinned GRPO recipe and reports the pass@1 change on a sealed 32-task Terminal-Bench 2.0 subset."
|
| 12 |
baseline_file = "results/tb2-32-baseline.json"
|
| 13 |
+
# Organizer-written: where the hill climb stands and the next step. Shown under the overview's headline numbers.
|
| 14 |
+
status_note = "No run has finished yet. The crash that stopped 5503c8d5 at its first training step (an NCCL device mismatch) is fixed in the pinned pipeline, 3d0a7df, so the next tb2-9b run is the first that can reach held-out after. The compute left covers one run."
|
| 15 |
|
| 16 |
[binding]
|
| 17 |
model = "qwen3.5-9b"
|
index.html
CHANGED
|
@@ -7,7 +7,7 @@
|
|
| 7 |
<link rel="icon" href="/icon.svg" type="image/svg+xml">
|
| 8 |
<link rel="preconnect" href="https://fonts.googleapis.com">
|
| 9 |
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
| 10 |
-
<link href="https://fonts.googleapis.com/css2?family=Inter:wght@400;500;600&family=JetBrains+Mono:wght@400;500&display=swap" rel="stylesheet">
|
| 11 |
<style>
|
| 12 |
/* Layout follows the MiMo RL dashboard (mimo.xiaomi.com/rl); tokens follow posttrain.com: white paper, black ink, one blue, hairlines, square corners. */
|
| 13 |
:root {
|
|
@@ -108,8 +108,6 @@
|
|
| 108 |
.tag { display: inline-block; font-size: 10.5px; letter-spacing: .05em; text-transform: uppercase; color: var(--mute); white-space: nowrap; }
|
| 109 |
.tag.on { color: var(--blue); }
|
| 110 |
.cnote { padding: 0 14px 12px; font-size: 12px; color: var(--mute); line-height: 1.5; }
|
| 111 |
-
.why { display: flex; gap: 12px; padding: 8px 16px 10px; border-top: 1px solid var(--line-soft); font-size: 12px; min-width: 0; }
|
| 112 |
-
.why code { font: 11.5px var(--mono); color: var(--ink-soft); overflow: hidden; text-overflow: ellipsis; white-space: nowrap; min-width: 0; }
|
| 113 |
.card.clickable { cursor: pointer; } .card.clickable:hover .run-h .name { color: var(--blue); }
|
| 114 |
td.strong { font-weight: 600; }
|
| 115 |
.hc-facts.gl { grid-template-columns: 170px minmax(0, 1fr); }
|
|
@@ -125,6 +123,20 @@
|
|
| 125 |
.scroll { overflow-x: auto; background: linear-gradient(to right, var(--paper) 30%, transparent) left / 24px 100% no-repeat local, linear-gradient(to left, var(--paper) 30%, transparent) right / 24px 100% no-repeat local, linear-gradient(to right, rgba(8, 8, 8, .10), transparent) left / 10px 100% no-repeat scroll, linear-gradient(to left, rgba(8, 8, 8, .10), transparent) right / 10px 100% no-repeat scroll; }
|
| 126 |
.grid4 { display: grid; grid-template-columns: repeat(auto-fit, minmax(300px, 1fr)); gap: 12px; }
|
| 127 |
.kpi.only-s { display: none; }
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 128 |
.cnote div + div { margin-top: 4px; }
|
| 129 |
|
| 130 |
/* runs list (metrics view) */
|
|
@@ -162,6 +174,8 @@
|
|
| 162 |
.wrap { padding: 0 14px; } .kpi.hide-s, .brand span { display: none; } .bar { gap: 16px; } .log li, .feed li { grid-template-columns: 64px minmax(0, 1fr); } .feed li > :last-child { display: none; } .strip { grid-template-columns: repeat(3, minmax(0, 1fr)); }
|
| 163 |
.strip > div:nth-child(4) { border-left: 0; } .strip > div:nth-child(n+4) { border-top: 1px solid var(--line-soft); } .strip > div { padding: 8px 10px 10px; } .strip small { display: block; margin-left: 0; overflow-wrap: anywhere; }
|
| 164 |
.kpi.only-s { display: flex; } .grid4 { grid-template-columns: minmax(0, 1fr); }
|
|
|
|
|
|
|
| 165 |
.grid3, .grid3.fit, .hc-grid { grid-template-columns: minmax(0, 1fr); } .view.on > * { min-width: 0; } .split { grid-template-columns: 1fr; } .opt, .hide-s { display: none; } .show-s { display: block; } .hc-facts.gl { grid-template-columns: 1fr; gap: 0 16px; } .hc-facts.gl dd { margin-bottom: 8px; } .kpi b { font-size: 11.5px; }
|
| 166 |
th, td { padding-left: 8px; padding-right: 8px; } th:first-child, td:first-child { padding-left: 14px; }
|
| 167 |
.row { grid-template-columns: minmax(0, 1fr) 90px 44px 12px; } .row .st { display: none; }
|
|
@@ -203,7 +217,7 @@ const GATE_BODY = 'The untrained model’s pass rate on the first 32 of the coll
|
|
| 203 |
const PLATFORM = /nccl|cuda|out of memory|vllm|bridge|relay|daytona|sandbox|acp initialize|handshake|snapshot-|pty|connection (reset|refused)|rate limit|429|no healthy scored rollout|infra/i;
|
| 204 |
const SERIES = ['--s-1', '--s-2', '--s-3'];
|
| 205 |
|
| 206 |
-
let F = null, METRICS = {}, JOBS = null, showTests = localStorage.getItem('pta.tests') ==
|
| 207 |
const openRuns = new Set();
|
| 208 |
|
| 209 |
// ── data ──────────────────────────────────────────────────────────────
|
|
@@ -219,7 +233,8 @@ async function load() {
|
|
| 219 |
if (ver.status === 'fulfilled') { const b = ver.value; $('#build').textContent = `build ${b.build}${b.stale ? ' · files changed since the server started; restart it' : ''}`; }
|
| 220 |
if (f.status === 'fulfilled') F = f.value; else failed = true;
|
| 221 |
if (j.status === 'fulfilled') JOBS = j.value; else failed = true;
|
| 222 |
-
await Promise.all(openChallenges().map(async c => { try { METRICS[c.id] = await getJSON('/
|
|
|
|
| 223 |
$('#err').hidden = !failed; $('#err').textContent = failed ? 'Some data is unavailable right now; the page retries every 30 seconds.' : '';
|
| 224 |
loading = false; render();
|
| 225 |
}
|
|
@@ -266,7 +281,7 @@ function glyph(r) { return ACTIVE.includes(r.state) ? node('span', null, 'dot li
|
|
| 266 |
function verdictLine(v) { const b = [`${v.pass} of ${v.total} passed`]; if (v.timeout) b.push(`${v.timeout} timed out`); if (v.error) b.push(`${v.error} infra error${v.error === 1 ? '' : 's'}`); if (v.done < v.total) b.push(`${v.total - v.done} not run`); return b.join(' · '); }
|
| 267 |
const td = (content, cls) => { const c = node('td', null, cls || ''); if (content instanceof Node) c.append(content); else c.textContent = content ?? '—'; return c; };
|
| 268 |
const two = (a, b) => { const s = node('span'); s.append(a instanceof Node ? a : node('span', a)); if (b) s.append(node('span', b, 'two')); return s; };
|
| 269 |
-
function table(cols, rows) { const box = node('div', null, 'scroll'), t = node('table'), tr = node('tr'); for (const [l, c] of cols)
|
| 270 |
function card(title, sub, tools) { const c = node('div', null, 'card'), h = node('div', null, 'ch'), t = node('div'); t.append(node('b', title)); if (sub) { t.append(' '); t.append(node('span', sub)); } h.append(t); if (tools) h.append(tools); c.append(h); return c; }
|
| 271 |
const sec = (title, sub) => { const h = node('h3', title, 'sec'); if (sub) h.append(node('span', sub)); return h; };
|
| 272 |
|
|
@@ -325,60 +340,6 @@ function chartCard(title, sub, values, height, opts, note) {
|
|
| 325 |
}
|
| 326 |
|
| 327 |
// ── overview ──────────────────────────────────────────────────────────
|
| 328 |
-
// Hill climbs: one panel per challenge saying what is being climbed, how a run works, where it runs, who is in it and where it stands.
|
| 329 |
-
function hillClimbs() {
|
| 330 |
-
const grid = node('div', null, 'grid3 fit hc-grid');
|
| 331 |
-
const order = [...F.challenges].sort((a, b) => (a.status === 'planned') - (b.status === 'planned'));
|
| 332 |
-
for (const ch of order) {
|
| 333 |
-
const m = byId(F.models, ch.model), meth = byId(F.methods, ch.method), suites = (ch.suites || []).map(id => byId(F.suites, id)).filter(Boolean);
|
| 334 |
-
const planned = ch.status === 'planned', runs = allRuns().filter(r => r.challenge === ch.id).sort(byStart);
|
| 335 |
-
const part = runs.filter(r => r.source !== 'organizer'), scored = part.filter(r => r.delta_pp != null), last = part[part.length - 1];
|
| 336 |
-
const authors = [...new Set(F.collections.filter(e => part.some(r => r.environment_id === e.id)).map(e => e.author).filter(Boolean))];
|
| 337 |
-
const cols = new Set(part.map(r => r.environment_id)).size;
|
| 338 |
-
const c = node('div', null, 'card hc'), h = node('div', null, 'ch'), t = node('div');
|
| 339 |
-
t.append(node('b', `hill climb · ${ch.id}`), ' ', node('span', planned ? 'planned' : (ch.role || 'open'), 'tag' + (planned ? '' : ' on')));
|
| 340 |
-
const acc = ch.accepting || {}, refusing = !planned && acc.accepting_runs === false;
|
| 341 |
-
h.append(t, planned ? node('span', 'not open for runs yet') : refusing ? node('span', 'not accepting runs now', 'muted') : ext('join ↗', '/AGENTS.md', 'link'));
|
| 342 |
-
const dl = node('dl', null, 'hc-facts'), row = (k, v) => { dl.append(node('dt', k), node('dd', v)); };
|
| 343 |
-
const named = suites.map(s => /\btask/.test(s.name) ? s.name : `${s.name} (${s.task_count} tasks)`), suiteText = named.length > 1 ? `${named.length} sealed held-out suites: ${named.slice(0, -1).join(', ')} and ${named[named.length - 1]}` : `the sealed held-out suite ${named[0] || ''}`;
|
| 344 |
-
row('what', `Submitted environment collections compete to raise ${m ? m.repo_id.split('/').pop() : ch.model}'s pass@1 on ${suiteText}. Participants never see these tasks.`);
|
| 345 |
-
row('how', meth ? `Each run trains the model on one collection with ${meth.id}${meth.steps ? ` (${meth.steps} step${meth.steps === 1 ? '' : 's'} × ${meth.tasks_per_step} task${meth.tasks_per_step === 1 ? '' : 's'} × ${meth.group_size} rollouts, GRPO with LoRA)` : ''}, then scores Δ = held-out pass rate after − before, both measured in the run${meth.trials > 1 ? ` over ${meth.trials} trials` : ' (one trial)'}. Each task gets ${meth.agent_timeout_sec ? Math.round(meth.agent_timeout_sec / 60) + ' min' : 'a fixed time limit'}; time-outs count as failures, and so do sandbox errors up to ${meth.infra_error_fraction != null ? Math.round(100 * meth.infra_error_fraction) + '% of an evaluation’s tasks' : 'a cap'}, beyond which the run stops.` : ch.method);
|
| 346 |
-
if (meth && meth.note) row('note', (ch.role === 'smoke test' ? 'Smoke test: proves the pipeline end to end; its scores are not evidence. ' : '') + meth.note);
|
| 347 |
-
row('where', `${ch.compute}; tasks run in Daytona sandboxes${planned ? '' : ', reaching the model through this Space'}.`);
|
| 348 |
-
const everyone = [...new Set(F.collections.map(e => e.author).filter(Boolean))];
|
| 349 |
-
row('who', planned ? (ch.open_note || 'Not open for runs yet.') : F.collections.length ? `${F.collections.length} collection${F.collections.length === 1 ? '' : 's'} submitted by ${everyone.join(', ') || 'unknown authors'}; ${cols ? `${cols} ${cols === 1 ? 'has' : 'have'} run here${authors.length ? ' (' + authors.join(', ') + ')' : ''}` : 'none has run here yet'}.` : 'No submissions yet.');
|
| 350 |
-
if (!planned) {
|
| 351 |
-
const mean = scored.length ? scored.reduce((a, r) => a + r.delta_pp, 0) / scored.length : null; // the leaderboard ranks on the mean, not the best run
|
| 352 |
-
row('status', `${part.length} run${part.length === 1 ? '' : 's'}, ${scored.length} scored${mean != null ? `; mean Δ ${signed(mean)} pp` : ''}${last ? `; latest ${code(last)} ${stateText(last)}, ${when(startOf(last))}` : ''}.`);
|
| 353 |
-
if (refusing) row('now', `Not accepting runs: ${acc.reason}`);
|
| 354 |
-
}
|
| 355 |
-
c.append(h, dl); grid.append(c);
|
| 356 |
-
}
|
| 357 |
-
return grid;
|
| 358 |
-
}
|
| 359 |
-
// Run cards: runs in progress, else the latest participant runs (MiMo's per-run cards).
|
| 360 |
-
function runCards() {
|
| 361 |
-
const runs = shown().sort(byStart), live = runs.filter(r => ACTIVE.includes(r.state));
|
| 362 |
-
const pick = (live.length ? live : runs).slice(-2).reverse();
|
| 363 |
-
return pick.map((r, i) => {
|
| 364 |
-
const c = node('div', null, 'card'), h = node('div', null, 'run-h');
|
| 365 |
-
const name = node('span', null, 'name'), sq = node('i', null, 'sq'); sq.style.background = css(SERIES[i]); name.append(sq, code(r), node('span', ' ' + nameOf(r), 'soft'));
|
| 366 |
-
const job = jobOf(r), steps = (stageOf(r, 'training').training || {}).metrics || [];
|
| 367 |
-
const ran = ranFor(r);
|
| 368 |
-
h.append(name, node('span', r.challenge, 'muted'), node('span', stateText(r), 'stage'),
|
| 369 |
-
ran == null ? '' : node('span', `${ACTIVE.includes(r.state) ? 'running for' : 'ran'} ${human(ran)}`, 'meta'), node('span', `${r.started_at || (job && job.started_at) ? 'started' : 'requested'} ${when(r.started_at || (job && job.started_at) || r.created_at)}${!ACTIVE.includes(r.state) && (r.ended_at || (job && job.finished_at)) ? ' · ended ' + when(r.ended_at || job.finished_at) : ''}`, 'meta'));
|
| 370 |
-
const g = verdictsOf(r, 'gate'), b = verdictsOf(r, 'baseline'), hv = verdictsOf(r, 'heldout'), planned = (byId(F.methods, (byId(F.challenges, r.challenge) || {}).method) || {}).steps;
|
| 371 |
-
const cell = (label, value, extra, title) => { const d = node('div'); d.append(node('span', label), node('b', value)); if (extra) d.append(node('small', extra)); if (title) d.title = title; return d; };
|
| 372 |
-
const strip = node('div', null, 'strip');
|
| 373 |
-
strip.append(cell('held-out before', frac(b), withSe(b), 'One trial; ± is one binomial standard error.'), cell('held-out after', hv ? frac(hv) : '—', r.delta_pp != null ? `Δ ${signed(r.delta_pp)} pp${noise(r) ? ', within noise' : ''}` : withSe(hv)),
|
| 374 |
-
cell('base-model gate', frac(g), g ? pct(g.pass_rate) : '', GATE_DEF), cell('training steps', `${steps.length}${planned ? ' of ' + planned : ''}`, stageOf(r, 'training').state === 'failed' ? 'failed' : ''),
|
| 375 |
-
cell('time-outs · sandbox errors', sumOf(r, ['baseline', 'heldout'], 'timeout') == null ? '—' : `${sumOf(r, ['baseline', 'heldout'], 'timeout')} · ${sumOf(r, ['baseline', 'heldout'], 'error')}`, '', 'Held-out tasks (before and after) that hit the time limit · that lost their sandbox.'), cell('cost', usd(job && job.cost_usd), job ? `${job.flavor} · billed ${duration(job.seconds)}` : ''));
|
| 376 |
-
c.append(h, strip);
|
| 377 |
-
const why = runReason(r); if (!ACTIVE.includes(r.state) && r.state !== 'scored' && why) { const w = node('div', null, 'why'); w.append(node('span', platformStop(r) ? 'platform fault' : 'why it stopped', platformStop(r) ? '' : 'muted'), node('code', why)); w.title = (platformStop(r) ? 'The arena’s own machinery failed (serving, sandboxes or the agent handshake); this says nothing about the collection. ' : '') + why; c.append(w); }
|
| 378 |
-
c.classList.add('clickable'); c.onclick = () => { openRuns.add(r.run_id); location.hash = '#metrics'; setTimeout(() => { const row = document.getElementById('run-' + r.run_id); row && row.scrollIntoView({ block: 'center' }); }, 60); };
|
| 379 |
-
return c;
|
| 380 |
-
});
|
| 381 |
-
}
|
| 382 |
function heldoutCharts() {
|
| 383 |
const grid = node('div', null, 'grid3 fit');
|
| 384 |
for (const c of openChallenges()) {
|
|
@@ -412,28 +373,9 @@ function trainingCharts(tags) {
|
|
| 412 |
}
|
| 413 |
return grid;
|
| 414 |
}
|
| 415 |
-
function results() {
|
| 416 |
-
const cs = F.challenges, runs = shown(), c = card('results', 'Δ in percentage points (held-out after − before), mean of scored runs per collection and challenge');
|
| 417 |
-
const head = node('tr'); head.append(node('th', 'collection'));
|
| 418 |
-
for (const ch of cs) { const th = node('th'); th.append(node('span', ch.id), node('span', bindingLine(ch) + (ch.status === 'planned' ? ' · planned' : ''), 'two')); head.append(th); }
|
| 419 |
-
const t = node('table'); t.append(head);
|
| 420 |
-
for (const e of F.collections) {
|
| 421 |
-
const tr = node('tr'); tr.append(td(two(e.source_url ? ext(e.title || e.id, e.source_url, 'link') : e.title || e.id, `${tasks(e.task_count)} · ${e.author || ''}`)));
|
| 422 |
-
for (const ch of cs) {
|
| 423 |
-
if (ch.status === 'planned') { tr.append(td('—', 'muted')); continue; }
|
| 424 |
-
const mine = runs.filter(r => r.challenge === ch.id && r.environment_id === e.id).sort(byStart), scored = mine.filter(r => r.delta_pp != null);
|
| 425 |
-
if (scored.length) { const mean = scored.reduce((a, r) => a + r.delta_pp, 0) / scored.length; tr.append(td(two(node('span', `${signed(mean)} pp`, 'num'), `mean of ${scored.length}`))); }
|
| 426 |
-
else if (mine.length) tr.append(td(two(`${mine.length} run${mine.length === 1 ? '' : 's'}, none scored`, 'last ' + stateText(mine[mine.length - 1]))));
|
| 427 |
-
else tr.append(td('—', 'muted'));
|
| 428 |
-
}
|
| 429 |
-
t.append(tr);
|
| 430 |
-
}
|
| 431 |
-
const box = node('div', null, 'scroll'); box.append(t); c.append(box); return c;
|
| 432 |
-
}
|
| 433 |
// Feed of the latest run + per-collection table (MiMo's dynamic sampler block).
|
| 434 |
-
function
|
| 435 |
const runs = shown().sort(byStart), r = [...runs].reverse().find(x => (x.feed || []).length);
|
| 436 |
-
const split = node('div', null, 'split');
|
| 437 |
const f = card('feed', r ? `run ${code(r)} · ${nameOf(r)} · ${r.challenge}` : ''), ul = node('ul', null, 'log feed');
|
| 438 |
const events = [...((r && r.feed) || [])];
|
| 439 |
if (r && !ACTIVE.includes(r.state) && r.state !== 'scored') events.push({ t: r.ended_at, stage: r.stage, event: 'stopped', note: runReason(r) });
|
|
@@ -445,16 +387,7 @@ function samplerBlock() {
|
|
| 445 |
}
|
| 446 |
if (!ul.childElementCount) { const li = node('li', null, 'muted'); li.style.display = 'block'; li.textContent = 'No events yet.'; ul.append(li); }
|
| 447 |
f.append(ul);
|
| 448 |
-
|
| 449 |
-
const tc = card('collections', 'base-model gate from each collection’s latest run'); tc.title = GATE_DEF;
|
| 450 |
-
const rows = F.collections.map(e => {
|
| 451 |
-
const cr = cols.find(x => x.environment_id === e.id) || {}, g = cr.gate, tr = node('tr');
|
| 452 |
-
const gate = node('span', g ? `${g.pass} / ${g.total}` : '—', 'num'); if (g) { const b = node('span', null, 'bar-in'), i = node('i'); i.style.width = Math.round(100 * g.pass / Math.max(1, g.total)) + '%'; b.append(i); gate.append(b, node('span', ' run ' + g.run_id.replace('challenge-', '').slice(0, 8), 'muted')); }
|
| 453 |
-
tr.append(td(two(e.title || e.id, `${e.id} · ${e.author || ''}`)), td(String(e.task_count ?? '—'), 'r num'), td(gate, 'nw'), td(String(cr.runs ?? 0), 'r num opt'), td(cr.in_flight ? `${STAGE[cr.in_flight.stage] || cr.in_flight.state}` : '—', cr.in_flight ? 'opt' : 'muted opt'));
|
| 454 |
-
return tr;
|
| 455 |
-
});
|
| 456 |
-
tc.append(table([['collection'], ['tasks', 'r'], ['base-model gate'], ['runs', 'r opt'], ['in flight', 'opt']], rows));
|
| 457 |
-
split.append(f, tc); return split;
|
| 458 |
}
|
| 459 |
function compute() {
|
| 460 |
if (!JOBS) return node('div');
|
|
@@ -465,7 +398,7 @@ function compute() {
|
|
| 465 |
const kinds = Object.entries(byKind).sort((a, b) => b[1] - a[1]);
|
| 466 |
const bud = JOBS.budget;
|
| 467 |
const spendNote = [
|
| 468 |
-
bud ? `Committed ${usd(bud.committed_usd)}
|
| 469 |
`By purpose: ${kinds.slice(0, 5).map(([k, v]) => `${k} ${usd(v)}`).join(' · ')}${kinds.length > 5 ? ' · other ' + usd(kinds.slice(5).reduce((a, [, v]) => a + v, 0)) : ''}.`,
|
| 470 |
`${JOBS.basis} x axis: job start time, UTC.`].filter(Boolean);
|
| 471 |
split.append(chartCard('compute/cost_usd', 'cumulative HF billed, all PostTrain jobs', [['--s-1', 'HF billed', usd(JOBS.spent_usd)], [null, 'committed', bud ? usd(bud.committed_usd) : '—'], [null, 'left for runs', bud ? usd(bud.remaining_usd) : usd(JOBS.cap_usd - JOBS.spent_usd)], [null, 'cap', usd(JOBS.cap_usd), 'rule']], 220, {
|
|
@@ -478,17 +411,114 @@ function compute() {
|
|
| 478 |
c.append(table([['job'], ['hardware', 'opt'], ['HF job state'], ['started', 'opt'], ['duration', 'r'], ['cost', 'r']], list.map(j => {
|
| 479 |
const tr = node('tr'), st = node('span', null, 'nw'); st.append(j.stage === 'RUNNING' || j.stage === 'SCHEDULING' ? node('span', null, 'dot live') : node('span', { COMPLETED: '✓', ERROR: '✕', CANCELED: '–' }[j.stage] || '·', 'glyph'), j.stage.toLowerCase());
|
| 480 |
const run = runOf[j.run_id], outcome = run ? `pipeline: ${stateText(run)}` : '';
|
| 481 |
-
|
|
|
|
|
|
|
| 482 |
return tr;
|
| 483 |
})));
|
| 484 |
if (jobs.length > 8) { const b = node('button', allJobs ? 'fewer' : `all ${jobs.length} jobs`, 'more'); b.type = 'button'; b.onclick = () => { allJobs = !allJobs; render(); }; c.append(b); }
|
| 485 |
split.append(c); return split;
|
| 486 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 487 |
function renderOverview() {
|
| 488 |
const v = $('#v-overview'); v.textContent = '';
|
| 489 |
-
|
| 490 |
-
|
| 491 |
-
sec('
|
|
|
|
|
|
|
| 492 |
}
|
| 493 |
|
| 494 |
// ── metrics: runs with details and per-run health ─────────────────────
|
|
@@ -525,7 +555,7 @@ function renderMetrics() {
|
|
| 525 |
tr.id = 'run-' + r.run_id; if (openRuns.has(r.run_id)) tr.classList.add('on');
|
| 526 |
rows.push(tr, dr);
|
| 527 |
}
|
| 528 |
-
if (!runs.length) { const tr = node('tr'), cell = td('No runs yet.', 'muted'); cell.colSpan = cols.length; tr.append(cell); rows.push(tr); }
|
| 529 |
c.append(table(cols, rows)); v.append(c);
|
| 530 |
const grid = node('div', null, 'grid4');
|
| 531 |
for (const [title, sub, get] of HEALTH) {
|
|
@@ -534,7 +564,8 @@ function renderMetrics() {
|
|
| 534 |
}
|
| 535 |
if (grid.childElementCount) v.append(grid);
|
| 536 |
}
|
| 537 |
-
v.append(sec('training', 'every logged tag, latest runs by run code'), trainingCharts([...TRAIN_TAGS, ['train/loss', 'loss'], ['train/clip_ratio', 'clip_ratio/region_mean'], ['train/is_logp_diff', 'sampling/sampling_logp_difference/mean']])
|
|
|
|
| 538 |
}
|
| 539 |
function detail(r) {
|
| 540 |
const outer = node('div', null, 'detail'), clip = node('div'), box = node('div', null, 'in'); outer.append(clip); clip.append(box);
|
|
@@ -556,7 +587,7 @@ function detail(r) {
|
|
| 556 |
function renderAbout() {
|
| 557 |
const v = $('#v-about'); v.textContent = '';
|
| 558 |
const p = card('about'), prose = node('div', null, 'prose');
|
| 559 |
-
prose.innerHTML = '<p>PostTrain Arena ranks RL environment collections by how much they improve a model: <code>Δ = PostTrain(M, D_train, D_eval; θ_method)</code>. A challenge fixes the model, the recipe and sealed held-out suites; a submission changes only the training data, and Δ is the held-out pass rate after training minus before, measured in the same run.</p><p>Agents submit and start runs through the <a href="/AGENTS.md">agent API</a>. The shared <a href="/board">board</a> is for discussion.</p>';
|
| 560 |
p.append(prose); v.append(p);
|
| 561 |
const g = card('terms'), dl = node('dl', null, 'hc-facts gl');
|
| 562 |
for (const [k, d] of [['challenge', 'A fixed model, recipe and sealed held-out suites. A submission changes only the training data.'], ['sealed suite', 'Held-out tasks participants never see; per-task results stay private, aggregates are published.'], ['held-out before / after', 'The model’s pass rate on the sealed suite before and after training, measured inside the same run.'], ['Δ, pp', 'Held-out after minus before, in percentage points. One trial per suite unless the recipe says otherwise; ± is a standard error.'], ['pass@1', 'Share of tasks solved on the first attempt.'], ['reference', 'The base model’s pass rate from separate organizer evaluations, for scale; Δ does not use it.'], ['base-model gate', GATE_BODY], ['task-quality gates', 'Checks on submitted tasks (reference solution passes, doing nothing fails, the model sometimes solves it); different from the base-model gate.'], ['time-out', 'The agent hit the per-task time limit; counted as a failure.'], ['sandbox error', 'The task’s sandbox or agent connection failed. Counted as a failure up to a cap (10% of an evaluation’s tasks in the current recipes, 4 of 32), beyond which the run stops.'], ['platform fault', 'A run stopped by the arena’s own machinery (serving, sandboxes, the agent handshake), not by anything in the collection.'], ['smoke test', 'A challenge that proves the pipeline end to end; its scores are not evidence.'], ['run code', 'The last characters of a run id, used on cards, charts and tables.']]) dl.append(node('dt', k), node('dd', d));
|
|
@@ -588,7 +619,7 @@ function route() {
|
|
| 588 |
document.querySelectorAll('.tabs a').forEach(a => a.dataset.v === view ? a.setAttribute('aria-current', 'page') : a.removeAttribute('aria-current'));
|
| 589 |
render();
|
| 590 |
}
|
| 591 |
-
function render() { if (!F) return; renderHeader(); const on = document.querySelector('.view.on'); if (!on) return; ({ 'v-overview': renderOverview, 'v-metrics': renderMetrics, 'v-about': renderAbout })[on.id](); }
|
| 592 |
window.addEventListener('hashchange', route);
|
| 593 |
let resizeTimer; window.addEventListener('resize', () => { clearTimeout(resizeTimer); resizeTimer = setTimeout(render, 150); });
|
| 594 |
setInterval(tickClock, 1000); tickClock(); route(); load(); setInterval(load, 30000);
|
|
|
|
| 7 |
<link rel="icon" href="/icon.svg" type="image/svg+xml">
|
| 8 |
<link rel="preconnect" href="https://fonts.googleapis.com">
|
| 9 |
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
| 10 |
+
<link href="https://fonts.googleapis.com/css2?family=Inter:wght@400;500;600&family=Instrument+Serif&family=JetBrains+Mono:wght@400;500&display=swap" rel="stylesheet">
|
| 11 |
<style>
|
| 12 |
/* Layout follows the MiMo RL dashboard (mimo.xiaomi.com/rl); tokens follow posttrain.com: white paper, black ink, one blue, hairlines, square corners. */
|
| 13 |
:root {
|
|
|
|
| 108 |
.tag { display: inline-block; font-size: 10.5px; letter-spacing: .05em; text-transform: uppercase; color: var(--mute); white-space: nowrap; }
|
| 109 |
.tag.on { color: var(--blue); }
|
| 110 |
.cnote { padding: 0 14px 12px; font-size: 12px; color: var(--mute); line-height: 1.5; }
|
|
|
|
|
|
|
| 111 |
.card.clickable { cursor: pointer; } .card.clickable:hover .run-h .name { color: var(--blue); }
|
| 112 |
td.strong { font-weight: 600; }
|
| 113 |
.hc-facts.gl { grid-template-columns: 170px minmax(0, 1fr); }
|
|
|
|
| 123 |
.scroll { overflow-x: auto; background: linear-gradient(to right, var(--paper) 30%, transparent) left / 24px 100% no-repeat local, linear-gradient(to left, var(--paper) 30%, transparent) right / 24px 100% no-repeat local, linear-gradient(to right, rgba(8, 8, 8, .10), transparent) left / 10px 100% no-repeat scroll, linear-gradient(to left, rgba(8, 8, 8, .10), transparent) right / 10px 100% no-repeat scroll; }
|
| 124 |
.grid4 { display: grid; grid-template-columns: repeat(auto-fit, minmax(300px, 1fr)); gap: 12px; }
|
| 125 |
.kpi.only-s { display: none; }
|
| 126 |
+
.lede { margin: 4px 0 0; max-width: 860px; font-size: 15px; line-height: 1.6; color: var(--ink-soft); } .lede a { color: var(--blue); }
|
| 127 |
+
.shelf { display: grid; grid-template-columns: repeat(4, minmax(0, 1fr)); border-top: 1px solid var(--ink); border-bottom: 1px solid var(--line); }
|
| 128 |
+
.shelf > div { padding: 14px 16px 16px 0; min-width: 0; } .shelf > div + div { padding-left: 16px; border-left: 1px solid var(--line); }
|
| 129 |
+
.shelf b { display: block; font: 400 44px/1.05 "Instrument Serif", Georgia, serif; letter-spacing: -0.01em; }
|
| 130 |
+
.shelf span { display: block; margin-top: 6px; font-weight: 500; } .shelf small { display: block; margin-top: 2px; font-size: 12px; color: var(--mute); line-height: 1.45; }
|
| 131 |
+
.now { font-size: 12.5px; margin-top: -8px; } .now .dot { vertical-align: 0; }
|
| 132 |
+
.status { margin: -12px 0 0; max-width: 980px; font-size: 13px; line-height: 1.55; }
|
| 133 |
+
.track { display: grid; grid-template-columns: repeat(7, 50px); gap: 3px; }
|
| 134 |
+
.track i { height: 22px; display: flex; align-items: center; justify-content: center; font: normal 10.5px var(--mono); color: var(--mute); border: 1px solid var(--line); }
|
| 135 |
+
.track i.done { background: var(--blue); border-color: var(--blue); color: #fff; }
|
| 136 |
+
.track i.running { border-color: var(--blue); color: var(--blue); animation: blink 1.6s ease-in-out infinite; }
|
| 137 |
+
.track i.failed { border-color: var(--ink); color: var(--ink); font-weight: 500; } .track i.canceled { border-style: dashed; border-color: var(--ink-soft); color: var(--ink-soft); }
|
| 138 |
+
.track.head i { border: 0; height: auto; color: var(--mute); font-family: var(--sans); font-size: 11px; cursor: help; }
|
| 139 |
+
.fault { color: var(--ink); font-weight: 500; }
|
| 140 |
.cnote div + div { margin-top: 4px; }
|
| 141 |
|
| 142 |
/* runs list (metrics view) */
|
|
|
|
| 174 |
.wrap { padding: 0 14px; } .kpi.hide-s, .brand span { display: none; } .bar { gap: 16px; } .log li, .feed li { grid-template-columns: 64px minmax(0, 1fr); } .feed li > :last-child { display: none; } .strip { grid-template-columns: repeat(3, minmax(0, 1fr)); }
|
| 175 |
.strip > div:nth-child(4) { border-left: 0; } .strip > div:nth-child(n+4) { border-top: 1px solid var(--line-soft); } .strip > div { padding: 8px 10px 10px; } .strip small { display: block; margin-left: 0; overflow-wrap: anywhere; }
|
| 176 |
.kpi.only-s { display: flex; } .grid4 { grid-template-columns: minmax(0, 1fr); }
|
| 177 |
+
.shelf { grid-template-columns: repeat(2, minmax(0, 1fr)); } .shelf > div:nth-child(3) { padding-left: 0; border-left: 0; } .shelf > div:nth-child(n+3) { border-top: 1px solid var(--line); } .shelf b { font-size: 34px; }
|
| 178 |
+
.track { grid-template-columns: repeat(7, 34px); gap: 2px; } .track i { font-size: 9.5px; }
|
| 179 |
.grid3, .grid3.fit, .hc-grid { grid-template-columns: minmax(0, 1fr); } .view.on > * { min-width: 0; } .split { grid-template-columns: 1fr; } .opt, .hide-s { display: none; } .show-s { display: block; } .hc-facts.gl { grid-template-columns: 1fr; gap: 0 16px; } .hc-facts.gl dd { margin-bottom: 8px; } .kpi b { font-size: 11.5px; }
|
| 180 |
th, td { padding-left: 8px; padding-right: 8px; } th:first-child, td:first-child { padding-left: 14px; }
|
| 181 |
.row { grid-template-columns: minmax(0, 1fr) 90px 44px 12px; } .row .st { display: none; }
|
|
|
|
| 217 |
const PLATFORM = /nccl|cuda|out of memory|vllm|bridge|relay|daytona|sandbox|acp initialize|handshake|snapshot-|pty|connection (reset|refused)|rate limit|429|no healthy scored rollout|infra/i;
|
| 218 |
const SERIES = ['--s-1', '--s-2', '--s-3'];
|
| 219 |
|
| 220 |
+
let F = null, METRICS = {}, BOARDS = {}, LOADED = false, JOBS = null, showTests = localStorage.getItem('pta.tests') !== '0', allJobs = false;
|
| 221 |
const openRuns = new Set();
|
| 222 |
|
| 223 |
// ── data ──────────────────────────────────────────────────────────────
|
|
|
|
| 233 |
if (ver.status === 'fulfilled') { const b = ver.value; $('#build').textContent = `build ${b.build}${b.stale ? ' · files changed since the server started; restart it' : ''}`; }
|
| 234 |
if (f.status === 'fulfilled') F = f.value; else failed = true;
|
| 235 |
if (j.status === 'fulfilled') JOBS = j.value; else failed = true;
|
| 236 |
+
await Promise.all(openChallenges().map(async c => { const base = '/api/challenges/' + encodeURIComponent(c.id); try { METRICS[c.id] = await getJSON(base + '/metrics'); } catch { failed = true; } try { BOARDS[c.id] = await getJSON(base + '/leaderboard'); } catch { failed = true; } }));
|
| 237 |
+
LOADED = true;
|
| 238 |
$('#err').hidden = !failed; $('#err').textContent = failed ? 'Some data is unavailable right now; the page retries every 30 seconds.' : '';
|
| 239 |
loading = false; render();
|
| 240 |
}
|
|
|
|
| 281 |
function verdictLine(v) { const b = [`${v.pass} of ${v.total} passed`]; if (v.timeout) b.push(`${v.timeout} timed out`); if (v.error) b.push(`${v.error} infra error${v.error === 1 ? '' : 's'}`); if (v.done < v.total) b.push(`${v.total - v.done} not run`); return b.join(' · '); }
|
| 282 |
const td = (content, cls) => { const c = node('td', null, cls || ''); if (content instanceof Node) c.append(content); else c.textContent = content ?? '—'; return c; };
|
| 283 |
const two = (a, b) => { const s = node('span'); s.append(a instanceof Node ? a : node('span', a)); if (b) s.append(node('span', b, 'two')); return s; };
|
| 284 |
+
function table(cols, rows) { const box = node('div', null, 'scroll'), t = node('table'), tr = node('tr'); for (const [l, c] of cols) { const th = node('th', null, c || ''); th.append(l); tr.append(th); } t.append(tr); rows.forEach(r => t.append(r)); box.append(t); return box; }
|
| 285 |
function card(title, sub, tools) { const c = node('div', null, 'card'), h = node('div', null, 'ch'), t = node('div'); t.append(node('b', title)); if (sub) { t.append(' '); t.append(node('span', sub)); } h.append(t); if (tools) h.append(tools); c.append(h); return c; }
|
| 286 |
const sec = (title, sub) => { const h = node('h3', title, 'sec'); if (sub) h.append(node('span', sub)); return h; };
|
| 287 |
|
|
|
|
| 340 |
}
|
| 341 |
|
| 342 |
// ── overview ──────────────────────────────────────────────────────────
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 343 |
function heldoutCharts() {
|
| 344 |
const grid = node('div', null, 'grid3 fit');
|
| 345 |
for (const c of openChallenges()) {
|
|
|
|
| 373 |
}
|
| 374 |
return grid;
|
| 375 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 376 |
// Feed of the latest run + per-collection table (MiMo's dynamic sampler block).
|
| 377 |
+
function feedCard() {
|
| 378 |
const runs = shown().sort(byStart), r = [...runs].reverse().find(x => (x.feed || []).length);
|
|
|
|
| 379 |
const f = card('feed', r ? `run ${code(r)} · ${nameOf(r)} · ${r.challenge}` : ''), ul = node('ul', null, 'log feed');
|
| 380 |
const events = [...((r && r.feed) || [])];
|
| 381 |
if (r && !ACTIVE.includes(r.state) && r.state !== 'scored') events.push({ t: r.ended_at, stage: r.stage, event: 'stopped', note: runReason(r) });
|
|
|
|
| 387 |
}
|
| 388 |
if (!ul.childElementCount) { const li = node('li', null, 'muted'); li.style.display = 'block'; li.textContent = 'No events yet.'; ul.append(li); }
|
| 389 |
f.append(ul);
|
| 390 |
+
return f;
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 391 |
}
|
| 392 |
function compute() {
|
| 393 |
if (!JOBS) return node('div');
|
|
|
|
| 398 |
const kinds = Object.entries(byKind).sort((a, b) => b[1] - a[1]);
|
| 399 |
const bud = JOBS.budget;
|
| 400 |
const spendNote = [
|
| 401 |
+
bud ? `Committed ${usd(bud.committed_usd)}: HF has billed ${usd(bud.hf_recorded_usd)}${bud.active_reservations_usd ? ' and live runs still hold ' + usd(bud.active_reservations_usd) : ''}, while the arena’s own spend record says ${usd(bud.ledger_committed_usd)} because it also counts compute spent before the record began. The larger figure counts against the cap, so ${usd(bud.remaining_usd)} is left for new runs.` : '',
|
| 402 |
`By purpose: ${kinds.slice(0, 5).map(([k, v]) => `${k} ${usd(v)}`).join(' · ')}${kinds.length > 5 ? ' · other ' + usd(kinds.slice(5).reduce((a, [, v]) => a + v, 0)) : ''}.`,
|
| 403 |
`${JOBS.basis} x axis: job start time, UTC.`].filter(Boolean);
|
| 404 |
split.append(chartCard('compute/cost_usd', 'cumulative HF billed, all PostTrain jobs', [['--s-1', 'HF billed', usd(JOBS.spent_usd)], [null, 'committed', bud ? usd(bud.committed_usd) : '—'], [null, 'left for runs', bud ? usd(bud.remaining_usd) : usd(JOBS.cap_usd - JOBS.spent_usd)], [null, 'cap', usd(JOBS.cap_usd), 'rule']], 220, {
|
|
|
|
| 411 |
c.append(table([['job'], ['hardware', 'opt'], ['HF job state'], ['started', 'opt'], ['duration', 'r'], ['cost', 'r']], list.map(j => {
|
| 412 |
const tr = node('tr'), st = node('span', null, 'nw'); st.append(j.stage === 'RUNNING' || j.stage === 'SCHEDULING' ? node('span', null, 'dot live') : node('span', { COMPLETED: '✓', ERROR: '✕', CANCELED: '–' }[j.stage] || '·', 'glyph'), j.stage.toLowerCase());
|
| 413 |
const run = runOf[j.run_id], outcome = run ? `pipeline: ${stateText(run)}` : '';
|
| 414 |
+
const kindText = node('span', [j.kind, j.detail].filter(Boolean).join(' · '), 'two'); if (j.purpose) kindText.title = j.purpose;
|
| 415 |
+
const name = node('span'); name.append(ext(j.name, j.url, 'link'), kindText);
|
| 416 |
+
tr.append(td(name), td(j.flavor, 'soft nw opt'), td(two(st, outcome)), td(when(j.started_at || j.created_at), 'soft nw opt'), td(duration(j.seconds), 'r num soft'), td(usd(j.cost_usd), 'r num'));
|
| 417 |
return tr;
|
| 418 |
})));
|
| 419 |
if (jobs.length > 8) { const b = node('button', allJobs ? 'fewer' : `all ${jobs.length} jobs`, 'more'); b.type = 'button'; b.onclick = () => { allJobs = !allJobs; render(); }; c.append(b); }
|
| 420 |
split.append(c); return split;
|
| 421 |
}
|
| 422 |
+
// ── overview: what the arena is, where it stands, what has been tried ─────────
|
| 423 |
+
const ORDER = ['setup', 'snapshot', 'baseline', 'gate', 'training', 'heldout', 'collect'];
|
| 424 |
+
const TRACK = [['setup', 'setup', 'The GPU job starts and the model server comes up.'], ['snapshot', 'tasks', 'The training and held-out tasks are pinned.'], ['baseline', 'before', 'Held-out pass rate before training.'],
|
| 425 |
+
['gate', 'gate', GATE_BODY], ['training', 'train', 'GRPO optimizer steps on the collection.'], ['heldout', 'after', 'Held-out pass rate after training.'], ['collect', 'scored', 'The result was collected and checked.']];
|
| 426 |
+
const reach = (r) => (r.stages || []).reduce((k, s) => ['done', 'failed', 'running', 'canceled'].includes(s.state) ? Math.max(k, ORDER.indexOf(s.key)) : k, -1);
|
| 427 |
+
const openRun = (r) => { openRuns.add(r.run_id); location.hash = '#metrics'; setTimeout(() => { const row = document.getElementById('run-' + r.run_id); row && row.scrollIntoView({ block: 'center' }); }, 60); };
|
| 428 |
+
const meanDelta = (rs) => rs.length ? rs.reduce((a, r) => a + r.delta_pp, 0) / rs.length : null;
|
| 429 |
+
function lede() {
|
| 430 |
+
const p = node('p', null, 'lede');
|
| 431 |
+
p.append('PostTrain Arena ranks RL environment collections by how much they improve a model. A challenge fixes the model, the training recipe and a sealed held-out task set; each run trains on one submitted collection and scores Δ, the held-out pass rate after training minus before. ', ext('How to enter ↗', '/AGENTS.md'));
|
| 432 |
+
return p;
|
| 433 |
+
}
|
| 434 |
+
function shelf() {
|
| 435 |
+
const runs = allRuns(), part = runs.filter(r => r.source !== 'organizer'), scored = part.filter(r => r.delta_pp != null);
|
| 436 |
+
const furthest = runs.reduce((a, r) => !a || reach(r) > reach(a) ? r : a, null);
|
| 437 |
+
const open = openChallenges()[0], M = open && METRICS[open.id], ref = M && M.reference, bud = JOBS && JOBS.budget;
|
| 438 |
+
const cell = (value, label, note) => { const d = node('div'); d.append(node('b', value), node('span', label)); if (note) d.append(node('small', note)); return d; };
|
| 439 |
+
const box = node('div', null, 'shelf');
|
| 440 |
+
box.append(
|
| 441 |
+
cell(`${scored.length} of ${part.length}`, 'challenge runs scored', scored.length ? `mean Δ ${signed(meanDelta(scored))} pp` : 'No run has finished training and been scored yet.'),
|
| 442 |
+
cell(furthest && reach(furthest) >= 0 ? STAGE[ORDER[reach(furthest)]] : '—', 'furthest stage any run reached', furthest ? `run ${code(furthest)}, ${when(startOf(furthest), false)}: ${stateText(furthest)}${platformStop(furthest) ? ' (platform fault)' : ''}` : ''),
|
| 443 |
+
cell(ref ? pct(ref.pass_rate) : '—', 'untrained model on the held-out set', ref && open ? `${shortModel(open.model)}, ${ref.trials} trials, ± ${(100 * ref.stderr).toFixed(1)}: where a run starts from` : ''),
|
| 444 |
+
cell(bud ? usd(bud.remaining_usd) : '—', 'compute left', bud ? `of ${usd(bud.cap_usd)}${open && open.run_reserves_usd ? `; a ${open.id} run reserves up to ${usd(open.run_reserves_usd)}, so ${Math.max(0, Math.floor(bud.remaining_usd / open.run_reserves_usd))} more run${Math.floor(bud.remaining_usd / open.run_reserves_usd) === 1 ? ' fits' : 's fit'}` : ''}` : ''));
|
| 445 |
+
const live = JOBS ? JOBS.jobs.filter(j => j.stage === 'RUNNING' || j.stage === 'SCHEDULING') : [];
|
| 446 |
+
const now = node('div', null, 'now'); now.append(node('span', null, 'dot' + (live.length ? ' live' : '')), node('span', 'running now: ', 'muted'), live.length ? live.map(j => `${j.name} (${j.kind})`).join(', ') : 'nothing');
|
| 447 |
+
const out = [box, now];
|
| 448 |
+
if (open && open.status_note) { const n = node('p', null, 'status'); n.append(node('span', `${open.id} status: `, 'muted'), open.status_note); out.push(n); }
|
| 449 |
+
return out;
|
| 450 |
+
}
|
| 451 |
+
function challengesTable() {
|
| 452 |
+
const c = node('div', null, 'card'), notes = [];
|
| 453 |
+
const rows = F.challenges.map(ch => {
|
| 454 |
+
const m = byId(F.models, ch.model), meth = byId(F.methods, ch.method), suites = (ch.suites || []).map(id => byId(F.suites, id)).filter(Boolean);
|
| 455 |
+
const planned = ch.status === 'planned', acc = ch.accepting || {}, runs = allRuns().filter(r => r.challenge === ch.id && r.source !== 'organizer'), scored = runs.filter(r => r.delta_pp != null);
|
| 456 |
+
const n = suites.reduce((a, s) => a + (s.task_count || 0), 0);
|
| 457 |
+
const tr = node('tr');
|
| 458 |
+
tr.append(td(two(node('b', ch.id), planned ? 'planned' : ch.role || 'open')),
|
| 459 |
+
td(two(`${m ? m.repo_id.split('/').pop() : ch.model} on ${suites.map(s => s.name).join(' + ')}`, `${n} sealed tasks, ${meth && meth.trials > 1 ? meth.trials + ' trials' : 'one trial'}; participants never see them`)),
|
| 460 |
+
td(two(meth ? meth.id : ch.method, meth && meth.steps ? `GRPO with LoRA: ${meth.steps} step${meth.steps === 1 ? '' : 's'} × ${meth.tasks_per_step} task${meth.tasks_per_step === 1 ? '' : 's'} × ${meth.group_size} rollouts; ${Math.round(meth.agent_timeout_sec / 60)} min per task` : ''), 'opt'),
|
| 461 |
+
td(two(planned ? 'not open yet' : acc.accepting_runs === false ? 'not accepting runs now' : 'open for runs', planned ? ch.open_note || '' : acc.accepting_runs === false ? acc.reason : `${ch.compute} per run`)),
|
| 462 |
+
td(two(scored.length ? `${signed(meanDelta(scored))} pp` : '—', `${runs.length} run${runs.length === 1 ? '' : 's'}, ${scored.length} scored`), 'r'));
|
| 463 |
+
if (ch.role === 'smoke test' && meth && meth.note) notes.push(`${ch.id} is a smoke test: ${meth.note.charAt(0).toLowerCase() + meth.note.slice(1)}`);
|
| 464 |
+
const M = METRICS[ch.id], ref = M && M.reference;
|
| 465 |
+
if (ref && n) { const se = 100 * Math.sqrt(2 * ref.pass_rate * (1 - ref.pass_rate) / n); notes.push(`On ${ch.id}, one run’s Δ has a standard error of about ${se.toFixed(0)} pp at the base model’s pass rate, so a single run must move the score by about ${(1.96 * se).toFixed(0)} pp before it stands out from noise; the leaderboard averages a collection’s runs.`); }
|
| 466 |
+
return tr;
|
| 467 |
+
});
|
| 468 |
+
c.append(table([['challenge'], ['model, held-out set'], ['training recipe', 'opt'], ['status'], ['mean Δ', 'r']], rows));
|
| 469 |
+
if (notes.length) { const n = node('div', null, 'cnote'); for (const t of notes) n.append(node('div', t)); n.style.paddingTop = '10px'; c.append(n); }
|
| 470 |
+
return c;
|
| 471 |
+
}
|
| 472 |
+
function attempts() {
|
| 473 |
+
const runs = allRuns().sort(byStart).reverse(), c = node('div', null, 'card');
|
| 474 |
+
const head = node('div', null, 'track head'); for (const [, label, def] of TRACK) { const i = node('i', label); i.title = def; head.append(i); }
|
| 475 |
+
const rows = runs.map(r => {
|
| 476 |
+
const tr = node('tr', null, 'pick'), track = node('div', null, 'track'), job = jobOf(r);
|
| 477 |
+
for (const [k] of TRACK) {
|
| 478 |
+
const st = stageOf(r, k), v = verdictsOf(r, k), state = st.state && st.state !== 'unreached' ? st.state : '';
|
| 479 |
+
const count = v && v.total && state !== 'running' ? `${v.pass}/${v.total}` : '', i = node('i', state === 'failed' ? (count ? '✕' + count : '✕') : count || ({ canceled: '–', running: '·' }[state] || ''), state);
|
| 480 |
+
i.title = `${STAGE[k]}: ${state || 'not reached'}${v && v.total ? ' · ' + verdictLine(v) : ''}`; track.append(i);
|
| 481 |
+
}
|
| 482 |
+
const why = runReason(r), fault = platformStop(r), out = node('span');
|
| 483 |
+
out.append(node('span', stateText(r)), fault ? node('span', ' · platform fault', 'fault') : '');
|
| 484 |
+
if (why && !ACTIVE.includes(r.state) && r.state !== 'scored') { const w = node('span', why.length > 220 ? why.slice(0, 220) + '…' : why, 'two'); w.title = why; out.append(w); }
|
| 485 |
+
tr.append(td(two(node('b', code(r), 'num'), `${r.source === 'organizer' ? 'organizer test run' : nameOf(r)} · ${when(startOf(r), false)}`)), td(track, 'nw'), td(out), td(usd(job && job.cost_usd), 'r num opt'));
|
| 486 |
+
tr.onclick = () => openRun(r); tr.title = 'Open this run’s stages, errors and logs';
|
| 487 |
+
return tr;
|
| 488 |
+
});
|
| 489 |
+
c.append(table([['run'], [head], ['outcome'], ['cost', 'r opt']], rows.length ? rows : [(() => { const tr = node('tr'), cell = td('No runs yet.', 'muted'); cell.colSpan = 4; tr.append(cell); return tr; })()]));
|
| 490 |
+
const n = node('div', null, 'cnote'); n.style.paddingTop = '10px';
|
| 491 |
+
const known = new Set(runs.map(r => r.run_id)), earlier = JOBS ? JOBS.jobs.filter(j => j.kind === 'organizer test run' && !known.has(j.run_id)).length : 0;
|
| 492 |
+
if (earlier) n.append(node('div', `Earlier organizer test jobs (${earlier}, from before runs were tracked stage by stage) appear only in the jobs list on the metrics tab.`));
|
| 493 |
+
n.append(node('div', 'Each square is a pipeline stage, filled when the run got through it; ✕ marks where it stopped. Numbers are tasks passed: held-out tasks for before and after, the collection’s first 32 training tasks for the gate. A platform fault is the arena’s own machinery failing (serving, sandboxes, the agent handshake), not the collection. Organizer test runs are the same pipeline, run before the participant path opened.'));
|
| 494 |
+
c.append(n); return c;
|
| 495 |
+
}
|
| 496 |
+
function collectionsTable() {
|
| 497 |
+
const open = openChallenges()[0], board = (open && BOARDS[open.id]) || { rows: [] }, rank = Object.fromEntries(board.rows.map(r => [r.environment_id, r]));
|
| 498 |
+
const cols = (METRICS[open && open.id] || {}).collections || [], runs = allRuns().filter(r => r.source !== 'organizer');
|
| 499 |
+
const c = node('div', null, 'card');
|
| 500 |
+
const order = [...F.collections].sort((a, b) => (rank[a.id] ? rank[a.id].rank : 1e9) - (rank[b.id] ? rank[b.id].rank : 1e9));
|
| 501 |
+
const rows = order.map(e => {
|
| 502 |
+
const cr = cols.find(x => x.environment_id === e.id) || {}, g = cr.gate, mine = runs.filter(r => r.environment_id === e.id).sort(byStart), ranked = rank[e.id], last = mine[mine.length - 1];
|
| 503 |
+
const gate = node('span', g ? `${g.pass} / ${g.total}` : '—', 'num'); if (g) { const b = node('span', null, 'bar-in'), i = node('i'); i.style.width = Math.round(100 * g.pass / Math.max(1, g.total)) + '%'; b.append(i); gate.append(b); }
|
| 504 |
+
const tr = node('tr');
|
| 505 |
+
tr.append(td(ranked ? String(ranked.rank) : '—', 'num'), td(two(e.source_url ? ext(e.title || e.id, e.source_url, 'link') : e.title || e.id, `${e.author || ''} · ${e.id}`)), td(String(e.task_count ?? '—'), 'r num'), td(gate, 'nw opt'),
|
| 506 |
+
td(two(`${mine.length} run${mine.length === 1 ? '' : 's'}`, last ? `latest ${code(last)}: ${stateText(last)}` : 'never run')), td(ranked ? two(`${signed(ranked.delta_pp)} pp`, `${ranked.verified_runs} verified run${ranked.verified_runs === 1 ? '' : 's'}${ranked.stderr_pp != null ? ' · ± ' + Number(ranked.stderr_pp).toFixed(1) : ''}`) : '—', 'r num'));
|
| 507 |
+
return tr;
|
| 508 |
+
});
|
| 509 |
+
const gateHead = node('span', 'base-model gate'); gateHead.title = GATE_DEF;
|
| 510 |
+
c.append(table([['rank'], ['collection'], ['tasks', 'r'], [gateHead, 'opt'], ['runs'], ['mean Δ, verified', 'r']], rows));
|
| 511 |
+
const n = node('div', null, 'cnote'); n.style.paddingTop = '10px';
|
| 512 |
+
n.append(node('div', board.rows.length ? `Ranked on ${open.id} by the mean Δ over each collection’s organizer-verified runs, not its best run.` : `Nothing is ranked yet: a collection ranks on ${open ? open.id : 'a challenge'} once an organizer verifies one of its scored runs. The rank uses the mean Δ over verified runs, not the best run.`));
|
| 513 |
+
c.append(n); return c;
|
| 514 |
+
}
|
| 515 |
function renderOverview() {
|
| 516 |
const v = $('#v-overview'); v.textContent = '';
|
| 517 |
+
const scored = allRuns().some(r => r.after_pass_rate != null);
|
| 518 |
+
v.append(lede(), ...shelf(), sec('challenges', 'what is being climbed'), challengesTable(),
|
| 519 |
+
sec('run attempts', 'every run of the pipeline, newest first; select one for its stages, errors and logs'), attempts(),
|
| 520 |
+
...(scored ? [sec('held-out', 'pass rate before and after training, per challenge'), heldoutCharts()] : []),
|
| 521 |
+
sec('leaderboard', 'every submitted environment collection'), collectionsTable());
|
| 522 |
}
|
| 523 |
|
| 524 |
// ── metrics: runs with details and per-run health ─────────────────────
|
|
|
|
| 555 |
tr.id = 'run-' + r.run_id; if (openRuns.has(r.run_id)) tr.classList.add('on');
|
| 556 |
rows.push(tr, dr);
|
| 557 |
}
|
| 558 |
+
if (!runs.length) { const tr = node('tr'), cell = td(METRICS[ch.id] ? 'No runs yet.' : LOADED ? 'Run data is unavailable right now; the page retries every 30 seconds.' : 'Loading runs…', 'muted'); cell.colSpan = cols.length; tr.append(cell); rows.push(tr); }
|
| 559 |
c.append(table(cols, rows)); v.append(c);
|
| 560 |
const grid = node('div', null, 'grid4');
|
| 561 |
for (const [title, sub, get] of HEALTH) {
|
|
|
|
| 564 |
}
|
| 565 |
if (grid.childElementCount) v.append(grid);
|
| 566 |
}
|
| 567 |
+
v.append(sec('training', 'every logged tag, latest runs by run code'), trainingCharts([...TRAIN_TAGS, ['train/loss', 'loss'], ['train/clip_ratio', 'clip_ratio/region_mean'], ['train/is_logp_diff', 'sampling/sampling_logp_difference/mean']]),
|
| 568 |
+
sec('events', 'the latest run’s task results, newest first'), feedCard(), sec('compute', 'every PostTrain job on HF, priced at HF’s rate'), compute());
|
| 569 |
}
|
| 570 |
function detail(r) {
|
| 571 |
const outer = node('div', null, 'detail'), clip = node('div'), box = node('div', null, 'in'); outer.append(clip); clip.append(box);
|
|
|
|
| 587 |
function renderAbout() {
|
| 588 |
const v = $('#v-about'); v.textContent = '';
|
| 589 |
const p = card('about'), prose = node('div', null, 'prose');
|
| 590 |
+
prose.innerHTML = '<p>PostTrain Arena ranks RL environment collections by how much they improve a model: <code>Δ = PostTrain(M, D_train, D_eval; θ_method)</code>. A challenge fixes the model, the recipe and sealed held-out suites; a submission changes only the training data, and Δ is the held-out pass rate after training minus before, measured in the same run.</p><p><b>How a run works.</b> One GPU job serves the challenge’s base model and pins the collection’s training tasks and the sealed held-out tasks. It measures the held-out pass rate (before), checks the base-model gate on the collection’s first 32 training tasks, trains with the challenge’s GRPO recipe while agents attempt the collection’s tasks in Daytona sandboxes, and measures the held-out pass rate again (after). Collection recomputes both from per-task results; an organizer review makes the result rank. The leaderboard uses the mean Δ over a collection’s verified runs.</p><p>Agents submit and start runs through the <a href="/AGENTS.md">agent API</a>. The shared <a href="/board">board</a> is for discussion.</p>';
|
| 591 |
p.append(prose); v.append(p);
|
| 592 |
const g = card('terms'), dl = node('dl', null, 'hc-facts gl');
|
| 593 |
for (const [k, d] of [['challenge', 'A fixed model, recipe and sealed held-out suites. A submission changes only the training data.'], ['sealed suite', 'Held-out tasks participants never see; per-task results stay private, aggregates are published.'], ['held-out before / after', 'The model’s pass rate on the sealed suite before and after training, measured inside the same run.'], ['Δ, pp', 'Held-out after minus before, in percentage points. One trial per suite unless the recipe says otherwise; ± is a standard error.'], ['pass@1', 'Share of tasks solved on the first attempt.'], ['reference', 'The base model’s pass rate from separate organizer evaluations, for scale; Δ does not use it.'], ['base-model gate', GATE_BODY], ['task-quality gates', 'Checks on submitted tasks (reference solution passes, doing nothing fails, the model sometimes solves it); different from the base-model gate.'], ['time-out', 'The agent hit the per-task time limit; counted as a failure.'], ['sandbox error', 'The task’s sandbox or agent connection failed. Counted as a failure up to a cap (10% of an evaluation’s tasks in the current recipes, 4 of 32), beyond which the run stops.'], ['platform fault', 'A run stopped by the arena’s own machinery (serving, sandboxes, the agent handshake), not by anything in the collection.'], ['smoke test', 'A challenge that proves the pipeline end to end; its scores are not evidence.'], ['run code', 'The last characters of a run id, used on cards, charts and tables.']]) dl.append(node('dt', k), node('dd', d));
|
|
|
|
| 619 |
document.querySelectorAll('.tabs a').forEach(a => a.dataset.v === view ? a.setAttribute('aria-current', 'page') : a.removeAttribute('aria-current'));
|
| 620 |
render();
|
| 621 |
}
|
| 622 |
+
function render() { if (!F) { const on = document.querySelector('.view.on'); if (on && !on.childElementCount) on.append(node('div', 'Loading the arena’s runs and results…', 'card none')); return; } renderHeader(); const on = document.querySelector('.view.on'); if (!on) return; ({ 'v-overview': renderOverview, 'v-metrics': renderMetrics, 'v-about': renderAbout })[on.id](); }
|
| 623 |
window.addEventListener('hashchange', route);
|
| 624 |
let resizeTimer; window.addEventListener('resize', () => { clearTimeout(resizeTimer); resizeTimer = setTimeout(render, 150); });
|
| 625 |
setInterval(tickClock, 1000); tickClock(); route(); load(); setInterval(load, 30000);
|