posttrain-arena / board_api.py
Xiangyi Li
Benchmarks: rank and plot one benchmark at a time, per domain where it has domains
d077a80
Raw History Blame Contribute Delete
14.7 kB
"""Read API for the submissions app (index.html, at /arena): where each challenge stands, from the same SQLite databases
as app_api.py (?source=live|mock on every route). app_api.py mounts it under /api/app/board with one include line. The
Agent Collabs board at / (board.html) reads /challenges with source=live for its challenge strip.
The app reads everything else through the /api/app routes; these add what a challenge list and a challenge page need
in one call: runs by state, the collections ranked and the top of the leaderboard, whether a run can start now and why
not (an organizer's pause is named as one), the runs and queue in flight, and the shared compute budget.
app_api is imported inside the routes because it includes this router when it loads.
"""
import json, math
from fastapi import APIRouter, HTTPException, Query
router = APIRouter(prefix='/board')
Source = Query('live', pattern='^(live|mock)$', description='live (the default): what people submitted and the arena ran; '
'mock: a simulated competition under the same rules, with invented teams, runs and results')
STATES = ('running', 'queued', 'scored', 'failed', 'canceled') # challenges.run_view; anything else is 'unknown'
PAUSED = 'Runs are paused by the organizers: ' # how challenges.open_check words a refusal during a pause
def paused(source, challenge_id, accepting):
"""The organizers' pause (runs_paused in the challenge file), or None. Live data follows the challenge files this
process loaded. The mock world runs without the real arena's pause (mock_world.patched clears it), so for mock data
only a refusal the run checks worded as a pause counts."""
if source == 'live':
import challenges
row = next((c for c in challenges.CHALLENGES if c['id'] == challenge_id), None)
if row is not None: return row.get('runs_paused') or None
reason = (accepting or {}).get('reason') or ''
return reason[len(PAUSED):] if reason.startswith(PAUSED) else None
def fits(remaining, reserve):
"""How many more runs the budget covers: each run reserves its worst case until it ends."""
if remaining is None or not reserve: return None
return max(0, math.floor(remaining / reserve + 1e-9))
def review(run):
"""Where a scored run is in review: valid (counts), pending (collected, waiting for an organizer), invalid, or uncollected."""
return run['verification'] if run['verification'] in ('valid', 'pending', 'invalid') else 'uncollected'
def standing(con, source):
"""{'budget', 'as_of', 'challenges': [...]}: every challenge, in the configs' order, with where it stands."""
import app_api as a
m = a.meta(con); budget = m.get('budget') or {}
runs = a.rows(con, 'SELECT r.id, r.label, r.challenge_id, r.collection_id, r.state, r.stage, r.verification, r.created_at, r.started_at, r.author, c.title, c.team '
'FROM runs r LEFT JOIN collections c ON c.id = r.collection_id ORDER BY COALESCE(r.started_at, r.created_at)')
board = a.rows(con, 'SELECT b.challenge_id, b.collection_id, b.rank, b.delta_pp, b.stderr_pp, b.verified_runs, c.title, c.team '
'FROM board b LEFT JOIN collections c ON c.id = b.collection_id ORDER BY b.rank, b.delta_pp DESC')
queue = a.queue_of(con, a.now_of(con)); out = []
for c in a.rows(con, 'SELECT id, name, status, role, status_note, open_note, accepting, run_reserves_usd, rules FROM challenges ORDER BY sort'):
cid, accepting = c['id'], json.loads(c['accepting']) if c['accepting'] else None
reserve = c['run_reserves_usd'] if c['run_reserves_usd'] is not None else (json.loads(c['rules']) if c['rules'] else {}).get('reserve_usd')
mine, ranked = [r for r in runs if r['challenge_id'] == cid], [b for b in board if b['challenge_id'] == cid]
n = lambda test: sum(1 for r in mine if test(r))
scored = lambda verdict: n(lambda r: r['state'] == 'scored' and review(r) == verdict)
stats = {'runs': len(mine), 'running': n(lambda r: r['state'] == 'running'), 'queued': n(lambda r: r['state'] == 'queued'),
'scored': n(lambda r: r['state'] == 'scored'), 'verified': scored('valid'), 'in_review': scored('pending'), 'rejected': scored('invalid'),
'uncollected': scored('uncollected'), 'failed': n(lambda r: r['state'] == 'failed'), 'canceled': n(lambda r: r['state'] == 'canceled'),
'unknown': n(lambda r: r['state'] not in STATES), # the HF job's status could not be read
'collections': len({r['collection_id'] for r in mine if r['collection_id']}), 'organizer_runs': n(lambda r: not r['collection_id']),
'ranked': len(ranked), 'last_run_at': max((r['started_at'] or r['created_at'] for r in mine if r['started_at'] or r['created_at']), default=None)}
out.append({'id': cid, 'name': c['name'], 'status': c['status'], 'role': c['role'], 'status_note': c['status_note'], 'open_note': c['open_note'],
'accepting_runs': bool(accepting and accepting.get('accepting_runs')), 'reason': (accepting or {}).get('reason'),
'runs_paused': paused(source, cid, accepting), 'reserve_usd': reserve, 'runs_that_fit': fits(budget.get('remaining_usd'), reserve),
'stats': stats, 'top': ranked[:3],
'active': [{k: r[k] for k in ('id', 'label', 'state', 'stage', 'collection_id', 'title', 'team', 'author', 'started_at', 'created_at')}
for r in mine if r['state'] in ('running', 'queued')],
'queue': [{k: q.get(k) for k in ('collection_id', 'title', 'team', 'requested_at', 'position', 'waited_s', 'blocked')} for q in queue if q['challenge_id'] == cid]})
return {'budget': budget, 'as_of': m.get('as_of'), 'challenges': out}
@router.get('/challenges')
def board_challenges(source: str = Source):
"""Every challenge with where it stands, and the shared compute budget."""
import app_api as a
with a.db(source) as con: return standing(con, source)
@router.get('/challenges/{challenge_id}')
def board_challenge(challenge_id: str, source: str = Source):
"""One challenge with where it stands, and the shared compute budget."""
import app_api as a
with a.db(source) as con: s = standing(con, source)
row = next((c for c in s['challenges'] if c['id'] == challenge_id), None)
if not row: raise HTTPException(404, 'No such challenge in this data source.')
return {'budget': s['budget'], 'as_of': s['as_of'], 'challenge': row}
# ── Benchmarks: one leaderboard and one line plot per benchmark (held-out suite), and per domain where it has domains ──
def benchmark_list(con, source):
"""Every benchmark with its domains, the challenges that score on it and where the scoring challenge stands, in
GET /api/benchmarks' order: the default first, then those an open challenge scores, then the rest."""
import app_api as a
chs = [parse_suites(c) for c in a.rows(con, 'SELECT id, name, status, suites FROM challenges ORDER BY sort')]
stand = {c['id']: c for c in standing(con, source)['challenges']}
out = []
for s in a.rows(con, 'SELECT * FROM suites ORDER BY sort'):
mine = [{'id': c['id'], 'name': c['name'], 'status': c['status'], 'scores': 'alone' if len(c['suites']) == 1 else 'with ' + ', '.join(x for x in c['suites'] if x != s['id']),
'stats': (stand.get(c['id']) or {}).get('stats'), 'runs_paused': (stand.get(c['id']) or {}).get('runs_paused'),
'accepting_runs': (stand.get(c['id']) or {}).get('accepting_runs'), 'reason': (stand.get(c['id']) or {}).get('reason')}
for c in chs if s['id'] in c['suites']]
out.append({'id': s['id'], 'name': s['name'] or s['id'], 'status': s['status'], 'task_count': s['task_count'], 'sealed': bool(s['sealed']), 'default': bool(s['is_default']),
'note': s['note'] or '', 'domains': json.loads(s['domains']) if s['domains'] else [], 'suite_name': s['suite_name'] or s['id'],
'challenges': mine, 'scored': any(c['status'] == 'open' for c in mine)})
out.sort(key=lambda r: (not r['default'], not r['scored'], r['id']))
return out
def parse_suites(c):
c = dict(c); c['suites'] = json.loads(c['suites']) if c['suites'] else []
return c
def ranked(runs):
"""Leaderboard rows from verified measurements (oldest first), the way challenges.leaderboard ranks: mean Δ over every
verified run of a collection, the pooled run-to-run spread, ties sharing a rank."""
import challenges
by = {}
for r in runs:
if r['verification'] == 'valid': by.setdefault(r['collection_id'], []).append(r)
pooled = challenges.pooled_run_variance(by.values()); rows = []
for cid, mine in by.items():
m = challenges.mean_delta(mine, pooled)
rows.append({'collection_id': cid, 'title': mine[-1]['title'], 'team': mine[-1]['team'], 'control': mine[-1]['control'], 'rejected_runs': sum(1 for r in runs if r['collection_id'] == cid and r['verification'] == 'invalid'),
**{k: m[k] for k in ('delta_pp', 'stderr_pp', 'verified_runs')}, 'run_ids': m['run_ids'], 'last_at': mine[-1]['ended_at']})
rows.sort(key=lambda r: (-r['delta_pp'], r['last_at'] or ''))
rank, previous = 0, None
for i, r in enumerate(rows):
if r['delta_pp'] != previous: rank = i + 1
r['rank'] = rank; previous = r['delta_pp']
return rows, (round(pooled ** 0.5, 2) if pooled else None)
def benchmark_view(con, source, benchmark_id, challenge_id=None, domain=None):
"""One benchmark's leaderboard and line plot for one challenge that scores on it (the open one by default), on the
whole benchmark or on one of its domains. Every number is a scored run's change on this benchmark (or domain) alone:
the run's own Δ when the challenge scores this benchmark alone, else its per-suite (or per-domain) score; a run with
no such score is left out, never estimated."""
import app_api as a
rows = benchmark_list(con, source)
b = next((r for r in rows if r['id'] == benchmark_id), None)
if not b: raise HTTPException(404, 'No such benchmark in this data source.')
if domain is not None and domain not in [d['name'] for d in b['domains']]:
raise HTTPException(404, f"Benchmark {benchmark_id} has no domain {domain!r}." + (f" Its domains: {', '.join(d['name'] for d in b['domains'])}." if b['domains'] else ' It lists no domains.'))
if challenge_id is not None and challenge_id not in [c['id'] for c in b['challenges']]:
raise HTTPException(404, f'Challenge {challenge_id} does not score benchmark {benchmark_id}.')
c = next((x for x in b['challenges'] if x['id'] == challenge_id), None) or next((x for x in b['challenges'] if x['status'] == 'open'), None) or (b['challenges'] or [None])[0]
view = {'benchmark': b, 'benchmarks': [{k: r[k] for k in ('id', 'name', 'default', 'scored', 'task_count')} for r in rows], 'challenge': c, 'domain': domain,
'rows': [], 'runs': [], 'per_run_sd_pp': None, 'pending_count': 0, 'reference': None, 'domain_scores': []}
if not c or c['status'] != 'open': return view
only = c['scores'] == 'alone'
base = ('SELECT r.id, r.label, r.collection_id, r.ended_at, r.verification, c.title, c.team, c.control, {cols} FROM runs r JOIN collections c ON c.id = r.collection_id {join} '
"WHERE r.challenge_id = ? AND r.state = 'scored' {where} ORDER BY r.ended_at")
if domain is not None:
q = base.format(cols='d.delta_pp, d.stderr_pp', join='JOIN run_domains d ON d.run_id = r.id', where='AND d.domain = ? AND d.suite IN (?, ?) AND d.delta_pp IS NOT NULL')
runs = a.rows(con, q, c['id'], domain, b['id'], b['suite_name'])
elif only:
runs = a.rows(con, base.format(cols='r.delta_pp, r.stderr_pp', join='', where='AND r.delta_pp IS NOT NULL'), c['id'])
else:
q = base.format(cols='s.delta_pp, s.stderr_pp', join='JOIN run_suites s ON s.run_id = r.id', where='AND s.name IN (?, ?) AND s.delta_pp IS NOT NULL')
runs = a.rows(con, q, c['id'], b['id'], b['suite_name'])
view['runs'] = [r for r in runs if r['ended_at']]
view['pending_count'] = sum(1 for r in runs if r['verification'] == 'pending')
if only and domain is None: # the challenge's own leaderboard, exactly as ranked (challenges.leaderboard)
board = [a.parse(x, 'run_ids') for x in a.rows(con, 'SELECT b.collection_id, b.rank, b.control, b.delta_pp, b.stderr_pp, b.verified_runs, b.rejected_runs, b.run_ids, c.title, c.team '
'FROM board b JOIN collections c ON c.id = b.collection_id WHERE b.challenge_id = ? ORDER BY b.rank, b.delta_pp DESC', c['id'])]
m = a.parse(a.one(con, 'SELECT * FROM board_meta WHERE challenge_id = ?', c['id']) or {}, 'reference')
view.update(rows=board, per_run_sd_pp=m.get('per_run_sd_pp'), pending_count=m.get('pending_count') or view['pending_count'], reference=m.get('reference'))
else:
view['rows'], view['per_run_sd_pp'] = ranked(runs)
counts = {d['domain']: d for d in a.rows(con, "SELECT d.domain, COUNT(*) scored, SUM(r.verification = 'valid') verified FROM run_domains d JOIN runs r ON r.id = d.run_id "
"WHERE r.challenge_id = ? AND d.suite IN (?, ?) AND d.delta_pp IS NOT NULL GROUP BY d.domain", c['id'], b['id'], b['suite_name'])}
view['domain_scores'] = [{'name': d['name'], 'task_count': d['task_count'], 'scored_runs': (counts.get(d['name']) or {}).get('scored', 0),
'verified_runs': (counts.get(d['name']) or {}).get('verified', 0) or 0} for d in b['domains']]
return view
@router.get('/benchmarks')
def board_benchmarks(source: str = Source):
"""Every benchmark with its domains and the challenges that score on it; default is the one the board opens on."""
import app_api as a
with a.db(source) as con: rows = benchmark_list(con, source)
return {'default': next((r['id'] for r in rows if r['default']), rows[0]['id'] if rows else None), 'benchmarks': rows}
@router.get('/benchmarks/{benchmark_id}')
def board_benchmark(benchmark_id: str, source: str = Source, challenge: str | None = None, domain: str | None = None):
"""One benchmark's leaderboard and line plot (every scored run's change on it), optionally for one challenge that
scores on it and one of its domains."""
import app_api as a
with a.db(source) as con: return benchmark_view(con, source, benchmark_id, challenge, domain)