Spaces:
Running
Running
Xiangyi Li
Benchmarks: rank and plot one benchmark at a time, per domain where it has domains
d077a80 Download board_api.py from benchflow/posttrain-arena: direct link, hf CLI and curl.
- Browser
- Download file 14.7 kB
-
https://huggingface.co/spaces/benchflow/posttrain-arena/resolve/main/board_api.py
- Command line
-
hf download hf://spaces/benchflow/posttrain-arena/board_api.py
-
curl -L -o board_api.py https://huggingface.co/spaces/benchflow/posttrain-arena/resolve/main/board_api.py
14.7 kB
| """Read API for the submissions app (index.html, at /arena): where each challenge stands, from the same SQLite databases | |
| as app_api.py (?source=live|mock on every route). app_api.py mounts it under /api/app/board with one include line. The | |
| Agent Collabs board at / (board.html) reads /challenges with source=live for its challenge strip. | |
| The app reads everything else through the /api/app routes; these add what a challenge list and a challenge page need | |
| in one call: runs by state, the collections ranked and the top of the leaderboard, whether a run can start now and why | |
| not (an organizer's pause is named as one), the runs and queue in flight, and the shared compute budget. | |
| app_api is imported inside the routes because it includes this router when it loads. | |
| """ | |
| import json, math | |
| from fastapi import APIRouter, HTTPException, Query | |
| router = APIRouter(prefix='/board') | |
| Source = Query('live', pattern='^(live|mock)$', description='live (the default): what people submitted and the arena ran; ' | |
| 'mock: a simulated competition under the same rules, with invented teams, runs and results') | |
| STATES = ('running', 'queued', 'scored', 'failed', 'canceled') # challenges.run_view; anything else is 'unknown' | |
| PAUSED = 'Runs are paused by the organizers: ' # how challenges.open_check words a refusal during a pause | |
| def paused(source, challenge_id, accepting): | |
| """The organizers' pause (runs_paused in the challenge file), or None. Live data follows the challenge files this | |
| process loaded. The mock world runs without the real arena's pause (mock_world.patched clears it), so for mock data | |
| only a refusal the run checks worded as a pause counts.""" | |
| if source == 'live': | |
| import challenges | |
| row = next((c for c in challenges.CHALLENGES if c['id'] == challenge_id), None) | |
| if row is not None: return row.get('runs_paused') or None | |
| reason = (accepting or {}).get('reason') or '' | |
| return reason[len(PAUSED):] if reason.startswith(PAUSED) else None | |
| def fits(remaining, reserve): | |
| """How many more runs the budget covers: each run reserves its worst case until it ends.""" | |
| if remaining is None or not reserve: return None | |
| return max(0, math.floor(remaining / reserve + 1e-9)) | |
| def review(run): | |
| """Where a scored run is in review: valid (counts), pending (collected, waiting for an organizer), invalid, or uncollected.""" | |
| return run['verification'] if run['verification'] in ('valid', 'pending', 'invalid') else 'uncollected' | |
| def standing(con, source): | |
| """{'budget', 'as_of', 'challenges': [...]}: every challenge, in the configs' order, with where it stands.""" | |
| import app_api as a | |
| m = a.meta(con); budget = m.get('budget') or {} | |
| runs = a.rows(con, 'SELECT r.id, r.label, r.challenge_id, r.collection_id, r.state, r.stage, r.verification, r.created_at, r.started_at, r.author, c.title, c.team ' | |
| 'FROM runs r LEFT JOIN collections c ON c.id = r.collection_id ORDER BY COALESCE(r.started_at, r.created_at)') | |
| board = a.rows(con, 'SELECT b.challenge_id, b.collection_id, b.rank, b.delta_pp, b.stderr_pp, b.verified_runs, c.title, c.team ' | |
| 'FROM board b LEFT JOIN collections c ON c.id = b.collection_id ORDER BY b.rank, b.delta_pp DESC') | |
| queue = a.queue_of(con, a.now_of(con)); out = [] | |
| for c in a.rows(con, 'SELECT id, name, status, role, status_note, open_note, accepting, run_reserves_usd, rules FROM challenges ORDER BY sort'): | |
| cid, accepting = c['id'], json.loads(c['accepting']) if c['accepting'] else None | |
| reserve = c['run_reserves_usd'] if c['run_reserves_usd'] is not None else (json.loads(c['rules']) if c['rules'] else {}).get('reserve_usd') | |
| mine, ranked = [r for r in runs if r['challenge_id'] == cid], [b for b in board if b['challenge_id'] == cid] | |
| n = lambda test: sum(1 for r in mine if test(r)) | |
| scored = lambda verdict: n(lambda r: r['state'] == 'scored' and review(r) == verdict) | |
| stats = {'runs': len(mine), 'running': n(lambda r: r['state'] == 'running'), 'queued': n(lambda r: r['state'] == 'queued'), | |
| 'scored': n(lambda r: r['state'] == 'scored'), 'verified': scored('valid'), 'in_review': scored('pending'), 'rejected': scored('invalid'), | |
| 'uncollected': scored('uncollected'), 'failed': n(lambda r: r['state'] == 'failed'), 'canceled': n(lambda r: r['state'] == 'canceled'), | |
| 'unknown': n(lambda r: r['state'] not in STATES), # the HF job's status could not be read | |
| 'collections': len({r['collection_id'] for r in mine if r['collection_id']}), 'organizer_runs': n(lambda r: not r['collection_id']), | |
| 'ranked': len(ranked), 'last_run_at': max((r['started_at'] or r['created_at'] for r in mine if r['started_at'] or r['created_at']), default=None)} | |
| out.append({'id': cid, 'name': c['name'], 'status': c['status'], 'role': c['role'], 'status_note': c['status_note'], 'open_note': c['open_note'], | |
| 'accepting_runs': bool(accepting and accepting.get('accepting_runs')), 'reason': (accepting or {}).get('reason'), | |
| 'runs_paused': paused(source, cid, accepting), 'reserve_usd': reserve, 'runs_that_fit': fits(budget.get('remaining_usd'), reserve), | |
| 'stats': stats, 'top': ranked[:3], | |
| 'active': [{k: r[k] for k in ('id', 'label', 'state', 'stage', 'collection_id', 'title', 'team', 'author', 'started_at', 'created_at')} | |
| for r in mine if r['state'] in ('running', 'queued')], | |
| 'queue': [{k: q.get(k) for k in ('collection_id', 'title', 'team', 'requested_at', 'position', 'waited_s', 'blocked')} for q in queue if q['challenge_id'] == cid]}) | |
| return {'budget': budget, 'as_of': m.get('as_of'), 'challenges': out} | |
| def board_challenges(source: str = Source): | |
| """Every challenge with where it stands, and the shared compute budget.""" | |
| import app_api as a | |
| with a.db(source) as con: return standing(con, source) | |
| def board_challenge(challenge_id: str, source: str = Source): | |
| """One challenge with where it stands, and the shared compute budget.""" | |
| import app_api as a | |
| with a.db(source) as con: s = standing(con, source) | |
| row = next((c for c in s['challenges'] if c['id'] == challenge_id), None) | |
| if not row: raise HTTPException(404, 'No such challenge in this data source.') | |
| return {'budget': s['budget'], 'as_of': s['as_of'], 'challenge': row} | |
| # ── Benchmarks: one leaderboard and one line plot per benchmark (held-out suite), and per domain where it has domains ── | |
| def benchmark_list(con, source): | |
| """Every benchmark with its domains, the challenges that score on it and where the scoring challenge stands, in | |
| GET /api/benchmarks' order: the default first, then those an open challenge scores, then the rest.""" | |
| import app_api as a | |
| chs = [parse_suites(c) for c in a.rows(con, 'SELECT id, name, status, suites FROM challenges ORDER BY sort')] | |
| stand = {c['id']: c for c in standing(con, source)['challenges']} | |
| out = [] | |
| for s in a.rows(con, 'SELECT * FROM suites ORDER BY sort'): | |
| mine = [{'id': c['id'], 'name': c['name'], 'status': c['status'], 'scores': 'alone' if len(c['suites']) == 1 else 'with ' + ', '.join(x for x in c['suites'] if x != s['id']), | |
| 'stats': (stand.get(c['id']) or {}).get('stats'), 'runs_paused': (stand.get(c['id']) or {}).get('runs_paused'), | |
| 'accepting_runs': (stand.get(c['id']) or {}).get('accepting_runs'), 'reason': (stand.get(c['id']) or {}).get('reason')} | |
| for c in chs if s['id'] in c['suites']] | |
| out.append({'id': s['id'], 'name': s['name'] or s['id'], 'status': s['status'], 'task_count': s['task_count'], 'sealed': bool(s['sealed']), 'default': bool(s['is_default']), | |
| 'note': s['note'] or '', 'domains': json.loads(s['domains']) if s['domains'] else [], 'suite_name': s['suite_name'] or s['id'], | |
| 'challenges': mine, 'scored': any(c['status'] == 'open' for c in mine)}) | |
| out.sort(key=lambda r: (not r['default'], not r['scored'], r['id'])) | |
| return out | |
| def parse_suites(c): | |
| c = dict(c); c['suites'] = json.loads(c['suites']) if c['suites'] else [] | |
| return c | |
| def ranked(runs): | |
| """Leaderboard rows from verified measurements (oldest first), the way challenges.leaderboard ranks: mean Δ over every | |
| verified run of a collection, the pooled run-to-run spread, ties sharing a rank.""" | |
| import challenges | |
| by = {} | |
| for r in runs: | |
| if r['verification'] == 'valid': by.setdefault(r['collection_id'], []).append(r) | |
| pooled = challenges.pooled_run_variance(by.values()); rows = [] | |
| for cid, mine in by.items(): | |
| m = challenges.mean_delta(mine, pooled) | |
| rows.append({'collection_id': cid, 'title': mine[-1]['title'], 'team': mine[-1]['team'], 'control': mine[-1]['control'], 'rejected_runs': sum(1 for r in runs if r['collection_id'] == cid and r['verification'] == 'invalid'), | |
| **{k: m[k] for k in ('delta_pp', 'stderr_pp', 'verified_runs')}, 'run_ids': m['run_ids'], 'last_at': mine[-1]['ended_at']}) | |
| rows.sort(key=lambda r: (-r['delta_pp'], r['last_at'] or '')) | |
| rank, previous = 0, None | |
| for i, r in enumerate(rows): | |
| if r['delta_pp'] != previous: rank = i + 1 | |
| r['rank'] = rank; previous = r['delta_pp'] | |
| return rows, (round(pooled ** 0.5, 2) if pooled else None) | |
| def benchmark_view(con, source, benchmark_id, challenge_id=None, domain=None): | |
| """One benchmark's leaderboard and line plot for one challenge that scores on it (the open one by default), on the | |
| whole benchmark or on one of its domains. Every number is a scored run's change on this benchmark (or domain) alone: | |
| the run's own Δ when the challenge scores this benchmark alone, else its per-suite (or per-domain) score; a run with | |
| no such score is left out, never estimated.""" | |
| import app_api as a | |
| rows = benchmark_list(con, source) | |
| b = next((r for r in rows if r['id'] == benchmark_id), None) | |
| if not b: raise HTTPException(404, 'No such benchmark in this data source.') | |
| if domain is not None and domain not in [d['name'] for d in b['domains']]: | |
| raise HTTPException(404, f"Benchmark {benchmark_id} has no domain {domain!r}." + (f" Its domains: {', '.join(d['name'] for d in b['domains'])}." if b['domains'] else ' It lists no domains.')) | |
| if challenge_id is not None and challenge_id not in [c['id'] for c in b['challenges']]: | |
| raise HTTPException(404, f'Challenge {challenge_id} does not score benchmark {benchmark_id}.') | |
| c = next((x for x in b['challenges'] if x['id'] == challenge_id), None) or next((x for x in b['challenges'] if x['status'] == 'open'), None) or (b['challenges'] or [None])[0] | |
| view = {'benchmark': b, 'benchmarks': [{k: r[k] for k in ('id', 'name', 'default', 'scored', 'task_count')} for r in rows], 'challenge': c, 'domain': domain, | |
| 'rows': [], 'runs': [], 'per_run_sd_pp': None, 'pending_count': 0, 'reference': None, 'domain_scores': []} | |
| if not c or c['status'] != 'open': return view | |
| only = c['scores'] == 'alone' | |
| base = ('SELECT r.id, r.label, r.collection_id, r.ended_at, r.verification, c.title, c.team, c.control, {cols} FROM runs r JOIN collections c ON c.id = r.collection_id {join} ' | |
| "WHERE r.challenge_id = ? AND r.state = 'scored' {where} ORDER BY r.ended_at") | |
| if domain is not None: | |
| q = base.format(cols='d.delta_pp, d.stderr_pp', join='JOIN run_domains d ON d.run_id = r.id', where='AND d.domain = ? AND d.suite IN (?, ?) AND d.delta_pp IS NOT NULL') | |
| runs = a.rows(con, q, c['id'], domain, b['id'], b['suite_name']) | |
| elif only: | |
| runs = a.rows(con, base.format(cols='r.delta_pp, r.stderr_pp', join='', where='AND r.delta_pp IS NOT NULL'), c['id']) | |
| else: | |
| q = base.format(cols='s.delta_pp, s.stderr_pp', join='JOIN run_suites s ON s.run_id = r.id', where='AND s.name IN (?, ?) AND s.delta_pp IS NOT NULL') | |
| runs = a.rows(con, q, c['id'], b['id'], b['suite_name']) | |
| view['runs'] = [r for r in runs if r['ended_at']] | |
| view['pending_count'] = sum(1 for r in runs if r['verification'] == 'pending') | |
| if only and domain is None: # the challenge's own leaderboard, exactly as ranked (challenges.leaderboard) | |
| board = [a.parse(x, 'run_ids') for x in a.rows(con, 'SELECT b.collection_id, b.rank, b.control, b.delta_pp, b.stderr_pp, b.verified_runs, b.rejected_runs, b.run_ids, c.title, c.team ' | |
| 'FROM board b JOIN collections c ON c.id = b.collection_id WHERE b.challenge_id = ? ORDER BY b.rank, b.delta_pp DESC', c['id'])] | |
| m = a.parse(a.one(con, 'SELECT * FROM board_meta WHERE challenge_id = ?', c['id']) or {}, 'reference') | |
| view.update(rows=board, per_run_sd_pp=m.get('per_run_sd_pp'), pending_count=m.get('pending_count') or view['pending_count'], reference=m.get('reference')) | |
| else: | |
| view['rows'], view['per_run_sd_pp'] = ranked(runs) | |
| counts = {d['domain']: d for d in a.rows(con, "SELECT d.domain, COUNT(*) scored, SUM(r.verification = 'valid') verified FROM run_domains d JOIN runs r ON r.id = d.run_id " | |
| "WHERE r.challenge_id = ? AND d.suite IN (?, ?) AND d.delta_pp IS NOT NULL GROUP BY d.domain", c['id'], b['id'], b['suite_name'])} | |
| view['domain_scores'] = [{'name': d['name'], 'task_count': d['task_count'], 'scored_runs': (counts.get(d['name']) or {}).get('scored', 0), | |
| 'verified_runs': (counts.get(d['name']) or {}).get('verified', 0) or 0} for d in b['domains']] | |
| return view | |
| def board_benchmarks(source: str = Source): | |
| """Every benchmark with its domains and the challenges that score on it; default is the one the board opens on.""" | |
| import app_api as a | |
| with a.db(source) as con: rows = benchmark_list(con, source) | |
| return {'default': next((r['id'] for r in rows if r['default']), rows[0]['id'] if rows else None), 'benchmarks': rows} | |
| def board_benchmark(benchmark_id: str, source: str = Source, challenge: str | None = None, domain: str | None = None): | |
| """One benchmark's leaderboard and line plot (every scored run's change on it), optionally for one challenge that | |
| scores on it and one of its domains.""" | |
| import app_api as a | |
| with a.db(source) as con: return benchmark_view(con, source, benchmark_id, challenge, domain) | |