Spaces:
Running
Running
Xiangyi Li commited on
Commit ·
515a2d6
1
Parent(s): f50b2d2
Close the gaps: every task per collection, filters in the URL, filterable jobs; cut repeated information.
Browse filesTasks: the mock lists every task with outcomes that match each collection's gate counts exactly; live lists every task of each pinned revision from its static-gate report (codes in plain words, excluded / needs controls / passed). /api/app/tasks filters and pages on the server. Every list keeps its search, filters and sort in the URL. Compute lists only jobs that belong to no run, with search and status filters. The submission page drops its summary boxes and timeline for one status line; the run page drops details its header already shows; the Runs page folds its panels into one line of filter links.
- app_api.py +32 -7
- challenges.py +13 -4
- index.html +0 -0
- mock_data.py +18 -34
- store.py +34 -2
- test_compose.py +19 -0
app_api.py
CHANGED
|
@@ -159,8 +159,6 @@ def submission(collection_id: str, source: str = Source):
|
|
| 159 |
if r['state'] in ('running', 'queued'): r['progress'] = progress(r, r['stages'], r['steps_done'], r['steps_total'], med.get(r['challenge_id'], {}), now)
|
| 160 |
latest = next((r for r in reversed(runs) if steps.get(r['id'], 0) > 2), None)
|
| 161 |
curve = [dict(step=s['step'], t=s['t'], **json.loads(s['metrics'])) for s in rows(con, 'SELECT * FROM run_steps WHERE run_id = ? ORDER BY step', latest['id'])] if latest else []
|
| 162 |
-
learnt_run = next((r for r in reversed(runs) if one(con, 'SELECT 1 x FROM run_task_stats WHERE run_id = ? LIMIT 1', r['id'])), None)
|
| 163 |
-
learnt = rows(con, 'SELECT * FROM run_task_stats WHERE run_id = ?', learnt_run['id']) if learnt_run else []
|
| 164 |
board = [parse(b, 'suites', 'run_ids') for b in rows(con, 'SELECT * FROM board WHERE collection_id = ?', collection_id)]
|
| 165 |
timeline = [{'t': c['created_at'], 'what': 'submitted', 'detail': f"{c['task_count']} tasks"}]
|
| 166 |
for r in runs:
|
|
@@ -172,10 +170,9 @@ def submission(collection_id: str, source: str = Source):
|
|
| 172 |
mine_q = [q for q in queue_of(con, now) if q['collection_id'] == collection_id]
|
| 173 |
for q in mine_q: timeline.append({'t': q['requested_at'], 'what': 'run queued', 'detail': f"{q['challenge_id']}, number {q['position']} in line"})
|
| 174 |
timeline.sort(key=lambda x: (x['t'] or '', x.get('order', 0)))
|
| 175 |
-
return {**c, '
|
| 176 |
'queued': [q['challenge_id'] for q in mine_q], 'queue': mine_q,
|
| 177 |
-
'curve': {'run_id': latest['id'], 'label': latest['label'], 'steps': curve} if latest else None,
|
| 178 |
-
'learnt': {'run_id': learnt_run['id'], 'label': learnt_run['label'], 'tasks': learnt} if learnt_run else None, 'timeline': timeline}
|
| 179 |
|
| 180 |
|
| 181 |
@router.get('/runs')
|
|
@@ -209,10 +206,38 @@ def run_detail(run_id: str, source: str = Source):
|
|
| 209 |
'job': one(con, 'SELECT * FROM jobs WHERE run_id = ?', run_id), 'method': method}
|
| 210 |
|
| 211 |
|
|
|
|
|
|
|
|
|
|
| 212 |
@router.get('/tasks')
|
| 213 |
-
def tasks_list(source: str = Source
|
|
|
|
|
|
|
|
|
|
|
|
|
| 214 |
with db(source) as con:
|
| 215 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 216 |
|
| 217 |
|
| 218 |
@router.get('/leaderboard')
|
|
|
|
| 159 |
if r['state'] in ('running', 'queued'): r['progress'] = progress(r, r['stages'], r['steps_done'], r['steps_total'], med.get(r['challenge_id'], {}), now)
|
| 160 |
latest = next((r for r in reversed(runs) if steps.get(r['id'], 0) > 2), None)
|
| 161 |
curve = [dict(step=s['step'], t=s['t'], **json.loads(s['metrics'])) for s in rows(con, 'SELECT * FROM run_steps WHERE run_id = ? ORDER BY step', latest['id'])] if latest else []
|
|
|
|
|
|
|
| 162 |
board = [parse(b, 'suites', 'run_ids') for b in rows(con, 'SELECT * FROM board WHERE collection_id = ?', collection_id)]
|
| 163 |
timeline = [{'t': c['created_at'], 'what': 'submitted', 'detail': f"{c['task_count']} tasks"}]
|
| 164 |
for r in runs:
|
|
|
|
| 170 |
mine_q = [q for q in queue_of(con, now) if q['collection_id'] == collection_id]
|
| 171 |
for q in mine_q: timeline.append({'t': q['requested_at'], 'what': 'run queued', 'detail': f"{q['challenge_id']}, number {q['position']} in line"})
|
| 172 |
timeline.sort(key=lambda x: (x['t'] or '', x.get('order', 0)))
|
| 173 |
+
return {**c, 'runs': runs, 'board': board,
|
| 174 |
'queued': [q['challenge_id'] for q in mine_q], 'queue': mine_q,
|
| 175 |
+
'curve': {'run_id': latest['id'], 'label': latest['label'], 'steps': curve} if latest else None, 'timeline': timeline}
|
|
|
|
| 176 |
|
| 177 |
|
| 178 |
@router.get('/runs')
|
|
|
|
| 206 |
'job': one(con, 'SELECT * FROM jobs WHERE run_id = ?', run_id), 'method': method}
|
| 207 |
|
| 208 |
|
| 209 |
+
OUTCOME_ORDER = "CASE t.outcome WHEN 'trained' THEN 0 WHEN 'eligible' THEN 1 WHEN 'needs controls' THEN 2 WHEN 'out of band' THEN 3 WHEN 'failed controls' THEN 4 WHEN 'excluded' THEN 5 ELSE 6 END"
|
| 210 |
+
|
| 211 |
+
|
| 212 |
@router.get('/tasks')
|
| 213 |
+
def tasks_list(source: str = Source, q: str | None = None, outcome: str | None = None, cats: str | None = None, collection: str | None = None,
|
| 214 |
+
limit: int = Query(100, ge=1, le=1000), offset: int = Query(0, ge=0)):
|
| 215 |
+
"""Every task, filtered on the server: search, gate outcome, categories (comma-separated) and submission. Counts per outcome
|
| 216 |
+
ignore the outcome filter and counts per category ignore the category filter, so filter chips show what each choice would give.
|
| 217 |
+
For one submission, each trained task carries its pass rate early and late in the submission's latest run that recorded it."""
|
| 218 |
with db(source) as con:
|
| 219 |
+
base, args = ['1 = 1'], []
|
| 220 |
+
if collection: base.append('t.collection_id = ?'); args.append(collection)
|
| 221 |
+
if q: base.append('(t.name LIKE ? OR c.title LIKE ? OR c.team LIKE ?)'); args += [f'%{q}%'] * 3
|
| 222 |
+
cat_list = [x for x in (cats or '').split(',') if x]
|
| 223 |
+
by_cat = (f"t.category IN ({','.join('?' * len(cat_list))})", cat_list) if cat_list else ('1 = 1', [])
|
| 224 |
+
by_out = ('t.outcome = ?', [outcome]) if outcome else ('1 = 1', [])
|
| 225 |
+
frm = 'FROM tasks t JOIN collections c ON c.id = t.collection_id WHERE ' + ' AND '.join(base)
|
| 226 |
+
counts_out = {r['outcome']: r['n'] for r in rows(con, f'SELECT t.outcome, COUNT(*) n {frm} AND {by_cat[0]} GROUP BY t.outcome', *args, *by_cat[1])}
|
| 227 |
+
counts_cat = {r['category']: r['n'] for r in rows(con, f'SELECT t.category, COUNT(*) n {frm} AND {by_out[0]} GROUP BY t.category', *args, *by_out[1])}
|
| 228 |
+
where = f'{frm} AND {by_cat[0]} AND {by_out[0]}'; wargs = [*args, *by_cat[1], *by_out[1]]
|
| 229 |
+
total = one(con, f'SELECT COUNT(*) n {where}', *wargs)['n']
|
| 230 |
+
out = rows(con, f'SELECT t.*, c.title AS collection_title, c.team {where} ORDER BY {OUTCOME_ORDER}, c.title, t.name LIMIT ? OFFSET ?', *wargs, limit, offset)
|
| 231 |
+
learnt = None
|
| 232 |
+
if collection:
|
| 233 |
+
run = one(con, 'SELECT r.id, r.label FROM runs r WHERE r.collection_id = ? AND EXISTS (SELECT 1 FROM run_task_stats s WHERE s.run_id = r.id) '
|
| 234 |
+
'ORDER BY COALESCE(r.started_at, r.created_at) DESC LIMIT 1', collection)
|
| 235 |
+
if run:
|
| 236 |
+
stats = {x['name']: x for x in rows(con, 'SELECT name, start, finish FROM run_task_stats WHERE run_id = ?', run['id'])}
|
| 237 |
+
for t in out:
|
| 238 |
+
if t['name'] in stats: t['start'], t['finish'] = stats[t['name']]['start'], stats[t['name']]['finish']
|
| 239 |
+
learnt = run
|
| 240 |
+
return {'total': total, 'rows': out, 'by_outcome': counts_out, 'by_category': counts_cat, 'learnt_run': learnt}
|
| 241 |
|
| 242 |
|
| 243 |
@router.get('/leaderboard')
|
challenges.py
CHANGED
|
@@ -423,12 +423,21 @@ def eligibility(record):
|
|
| 423 |
stored static report when it was computed under this policy, otherwise re-checked from the pinned source once per
|
| 424 |
process (read-only, like the gate plan)."""
|
| 425 |
static=(record.get('quality_gates') or {}).get('static') or {}
|
| 426 |
-
if not (static
|
| 427 |
-
key=(record['id'],record['revision'],gates.VERSION)
|
| 428 |
-
if key not in _eligibility: _eligibility[key]=gates.compact(pinned_static(record,gates.REQUIRE_ORACLE))
|
| 429 |
-
static=_eligibility[key]
|
| 430 |
return static['summary']['eligible'],static['summary']['tasks'],static['excluded'],static.get('needs_controls',[])
|
| 431 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 432 |
def mirror_check(row,source):
|
| 433 |
"""Whether the pinned submission can be mirrored for a run, from file listings only: nothing is copied."""
|
| 434 |
client=hub();target=mirror_path(source['id'],source['revision'])
|
|
|
|
| 423 |
stored static report when it was computed under this policy, otherwise re-checked from the pinned source once per
|
| 424 |
process (read-only, like the gate plan)."""
|
| 425 |
static=(record.get('quality_gates') or {}).get('static') or {}
|
| 426 |
+
if not static_current(static): static=static_report(record)
|
|
|
|
|
|
|
|
|
|
| 427 |
return static['summary']['eligible'],static['summary']['tasks'],static['excluded'],static.get('needs_controls',[])
|
| 428 |
|
| 429 |
+
def static_current(static):
|
| 430 |
+
return static.get('version')==gates.VERSION and (static.get('policy') or {}).get('require_oracle')==gates.REQUIRE_ORACLE and 'excluded' in static
|
| 431 |
+
|
| 432 |
+
def static_report(record):
|
| 433 |
+
"""The compact static-gate report of a record under the current policy: the stored one when it is current and lists
|
| 434 |
+
every task, otherwise recomputed from the pinned source once per process (read-only)."""
|
| 435 |
+
static=(record.get('quality_gates') or {}).get('static') or {}
|
| 436 |
+
if static_current(static) and 'task_identity' in static: return static
|
| 437 |
+
key=(record['id'],record['revision'],gates.VERSION)
|
| 438 |
+
if key not in _eligibility: _eligibility[key]=gates.compact(pinned_static(record,gates.REQUIRE_ORACLE))
|
| 439 |
+
return _eligibility[key]
|
| 440 |
+
|
| 441 |
def mirror_check(row,source):
|
| 442 |
"""Whether the pinned submission can be mirrored for a run, from file listings only: nothing is copied."""
|
| 443 |
client=hub();target=mirror_path(source['id'],source['revision'])
|
index.html
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
mock_data.py
CHANGED
|
@@ -89,41 +89,25 @@ DESCRIBE = {
|
|
| 89 |
for _c, _extra in {'system-administration': ['Move a service to a new user without downtime', 'Find what keeps filling /tmp at 3 am', 'Repair a broken PAM stack from a rescue shell', 'Pin a kernel module that breaks on upgrade', 'Migrate crontabs to systemd timers'], 'debugging': ['Find the off-by-one in a pagination bug', 'Explain a memory spike after a config reload', 'Fix a deadlock between two worker pools', 'Find why a test passes alone but fails in the suite', 'Track down a timezone bug in nightly reports'], 'security': ['Find the SSRF in a small web service', 'Remove secrets committed to a repo’s history', 'Lock down a world-writable cron directory', 'Patch a path traversal in a file server', 'Find the weak JWT signing configuration'], 'software-engineering': ['Replace a hand-rolled argument parser', 'Add retries with backoff to a flaky client', 'Remove a circular import without behaviour change', 'Make a CLI’s JSON output stable across versions', 'Upgrade a library across a breaking API change'], 'data-processing': ['Parse a fixed-width mainframe export', 'Convert timestamps from five time zones to UTC', 'Fill gaps in a sensor series without inventing peaks', 'Split one giant XML file into valid records', 'Match customer records across two spellings'], 'file-operations': ['Recover files from a half-written zip', 'Rename 10,000 files from a CSV mapping', 'Find which files changed between two backups', 'Fix permissions on a copied home directory', 'Deduplicate a directory with hard links'], 'scientific-computing': ['Fix a unit mix-up in a simulation input', 'Make a Monte Carlo run reproducible', 'Port a MATLAB script to NumPy exactly', 'Interpolate missing values in a climate grid', 'Speed up a pairwise distance matrix'], 'data-science': ['Rebuild a dashboard number from raw events', 'Find duplicated users inflating a metric', 'Fit a model that respects a time split', 'Correct a biased sample with known weights', 'Find the join that fans out revenue'], 'machine-learning': ['Find the label leak in a feature pipeline', 'Match a paper’s reported accuracy', 'Fix a tokenizer mismatch at inference', 'Make evaluation deterministic across runs', 'Recover from a bad learning-rate schedule'], 'model-training': ['Fix a checkpoint that will not load on CPU', 'Find why loss plateaus after warmup', 'Shard a dataset for four workers evenly', 'Fix mixed precision that overflows', 'Resume a run with the right data order'], 'mathematics': ['Verify a closed form against brute force', 'Find integer solutions to a small system', 'Fix a rounding bug in a financial formula', 'Compute exact probabilities for a dice game', 'Prove two SQL queries return the same rows'], 'optimization': ['Remove an N+1 query from a report', 'Cache a slow API without stale reads', 'Cut a Docker image from 2 GB to 200 MB', 'Make a hot loop allocation-free', 'Reduce a cold start from 8 s to 2 s'], 'games': ['Solve a nonogram from its clue file', 'Finish a maze with limited moves', 'Win tic-tac-toe variants against a script', 'Complete a text puzzle box in 30 moves', 'Find the winning line in a Go problem'], 'tool-use': ['Batch-resize images with ImageMagick', 'Extract tables from a PDF with CLI tools', 'Sync two directories over rsync safely', 'Query a SQLite file for a weekly report', 'Script tmux to lay out a dev session']}.items(): TASKS[_c] += _extra
|
| 90 |
|
| 91 |
SETTINGS = ['on Debian 12', 'on Alpine', 'in a Python 3.12 repo', 'in a Rust workspace', 'with only busybox', 'on a read-only root', 'in a Go monorepo',
|
| 92 |
-
'on Ubuntu 24.04', 'inside a container', 'on a 1 GB VM', 'with no network', 'in a Node 22 project']
|
| 93 |
_setting = iter(range(10 ** 6))
|
| 94 |
_used = set()
|
| 95 |
_uses = {} # how often each title has been sampled, so collections draw the least-used titles first
|
| 96 |
|
| 97 |
|
| 98 |
-
def
|
| 99 |
-
"""
|
| 100 |
-
|
| 101 |
-
|
| 102 |
-
|
| 103 |
-
|
| 104 |
-
|
| 105 |
-
|
| 106 |
-
|
| 107 |
-
|
| 108 |
-
|
| 109 |
-
|
| 110 |
-
|
| 111 |
-
shares = [('excluded', excluded), ('failed controls', controls_failed), ('out of band', out_band), ('trained', in_band)]
|
| 112 |
-
plan = [st for st, n in shares for _ in range(round(k * n / max(1, total)))]
|
| 113 |
-
plan = (plan + ['trained'] * k)[:len(names)]; trng.shuffle(plan)
|
| 114 |
-
out = []
|
| 115 |
-
base = next(_setting)
|
| 116 |
-
for (title, c), st in zip(names, plan):
|
| 117 |
-
if title in _used: # a title another collection already has gets its own setting
|
| 118 |
-
for shift in range(len(SETTINGS)):
|
| 119 |
-
full = f'{title}, {SETTINGS[(base + shift) % len(SETTINGS)]}'
|
| 120 |
-
if full not in _used: break
|
| 121 |
-
title = full
|
| 122 |
-
_used.add(title)
|
| 123 |
-
if st == 'excluded': out.append({'name': title, 'category': c, 'status': st, 'passes': None, 'why': trng.choice(['reference solution readable inside the sandbox', 'answer file baked into the image', 'verifier only checks that a file exists'])})
|
| 124 |
-
elif st == 'failed controls': out.append({'name': title, 'category': c, 'status': st, 'passes': None, 'why': trng.choice(['doing nothing passes the verifier', 'the reference solution fails 3 of 8 reruns'])})
|
| 125 |
-
elif st == 'out of band': kb = 0 if trng.random() < band[0] / max(1, out_band) else BAND; out.append({'name': title, 'category': c, 'status': st, 'passes': kb, 'why': 'never solved: no learning signal' if kb == 0 else 'always solved: no learning signal'})
|
| 126 |
-
else: out.append({'name': title, 'category': c, 'status': 'trained', 'passes': trng.choices(range(1, BAND), weights=[band[j] + 0.01 for j in range(1, BAND)])[0], 'why': None})
|
| 127 |
return out
|
| 128 |
|
| 129 |
|
|
@@ -155,7 +139,7 @@ def collections():
|
|
| 155 |
'source_url': None, 'categories': counts,
|
| 156 |
'funnel': {'submitted': n, 'valid': n, 'static_eligible': valid, 'controls_passed': controls, 'in_band': trained, 'trained': trained},
|
| 157 |
'band': band, 'origin': {'original': round(n * rng.uniform(0.3, 0.8)), 'adapted': 0, 'generated': 0},
|
| 158 |
-
'description': DESCRIBE[theme], '
|
| 159 |
'_effect': round(effect, 2)})
|
| 160 |
o = rows[-1]['origin']; rest = n - o['original']; o['adapted'] = round(rest * rng.uniform(0.3, 0.9)); o['generated'] = rest - o['adapted']
|
| 161 |
return rows
|
|
@@ -252,7 +236,7 @@ def run_record(i, env, start, fate, now):
|
|
| 252 |
order = ['setup', 'snapshot', 'baseline', 'gate', 'training', 'heldout', 'collect']; stages.sort(key=lambda s: order.index(s['key']))
|
| 253 |
if metrics: next(s for s in stages if s['key'] == 'training')['training'] = {'metrics': metrics}
|
| 254 |
task_stats = [] # per trained sample task: base passes (of 8), pass rate over the first and the last 4 steps (recipe v2 records this per task)
|
| 255 |
-
for tk in env.get('
|
| 256 |
if tk['status'] != 'trained' or not metrics: continue
|
| 257 |
first = clamp(tk['passes'] / BAND + trng.gauss(0, 0.06), 0, 1)
|
| 258 |
last = clamp(first + (gain * 1.6 + trng.gauss(0, 0.12)) * min(1, len(metrics) / 32), 0, 1)
|
|
@@ -305,7 +289,7 @@ def schedule(envs, control_envs=()):
|
|
| 305 |
age = (NOW - datetime.fromisoformat(r['ended_at'].replace('Z', '+00:00'))).total_seconds() / 3600
|
| 306 |
r['verification'] = 'valid' if r.get('_control') else 'invalid' if rng.random() < 0.04 else 'valid' if age > rng.uniform(12, 36) else 'pending'
|
| 307 |
env = next((e for e in envs if e['id'] == r['environment_id']), None) or next(e for e in control_envs if e['id'] == r['environment_id'])
|
| 308 |
-
bad = next((t['name'] for t in env.get('
|
| 309 |
r['verification_note'] = {'invalid': f'Rejected: the reference solution of “{bad}” was readable inside its sandbox, so the model could copy it (found in review). The task is excluded from future runs; this run does not count.',
|
| 310 |
'valid': 'Organizer control; evidence reviewed.' if r.get('_control') else 'Evidence reviewed: held-out scores match the per-task results.', 'pending': None}[r['verification']]
|
| 311 |
for r in runs: r.pop('_control', None)
|
|
@@ -412,7 +396,7 @@ def controls():
|
|
| 412 |
env = {'id': f'env-control-{key}', 'title': title, 'repo_id': f'benchflow/pta-control-{key}', 'revision': '0' * 40, 'task_count': 120, 'author': 'organizers', 'team': 'organizers',
|
| 413 |
'created_at': iso(OPEN - timedelta(days=5)), 'status': 'Validated', 'source_url': None, 'categories': mix, 'control': key, 'description': text,
|
| 414 |
'funnel': {'submitted': 120, 'valid': 120, 'static_eligible': 120, 'controls_passed': 108, 'in_band': 60, 'trained': 60}, 'band': band,
|
| 415 |
-
'origin': {'original': 120, 'adapted': 0, 'generated': 0}, '
|
| 416 |
out_envs.append(env)
|
| 417 |
return out_envs
|
| 418 |
|
|
|
|
| 89 |
for _c, _extra in {'system-administration': ['Move a service to a new user without downtime', 'Find what keeps filling /tmp at 3 am', 'Repair a broken PAM stack from a rescue shell', 'Pin a kernel module that breaks on upgrade', 'Migrate crontabs to systemd timers'], 'debugging': ['Find the off-by-one in a pagination bug', 'Explain a memory spike after a config reload', 'Fix a deadlock between two worker pools', 'Find why a test passes alone but fails in the suite', 'Track down a timezone bug in nightly reports'], 'security': ['Find the SSRF in a small web service', 'Remove secrets committed to a repo’s history', 'Lock down a world-writable cron directory', 'Patch a path traversal in a file server', 'Find the weak JWT signing configuration'], 'software-engineering': ['Replace a hand-rolled argument parser', 'Add retries with backoff to a flaky client', 'Remove a circular import without behaviour change', 'Make a CLI’s JSON output stable across versions', 'Upgrade a library across a breaking API change'], 'data-processing': ['Parse a fixed-width mainframe export', 'Convert timestamps from five time zones to UTC', 'Fill gaps in a sensor series without inventing peaks', 'Split one giant XML file into valid records', 'Match customer records across two spellings'], 'file-operations': ['Recover files from a half-written zip', 'Rename 10,000 files from a CSV mapping', 'Find which files changed between two backups', 'Fix permissions on a copied home directory', 'Deduplicate a directory with hard links'], 'scientific-computing': ['Fix a unit mix-up in a simulation input', 'Make a Monte Carlo run reproducible', 'Port a MATLAB script to NumPy exactly', 'Interpolate missing values in a climate grid', 'Speed up a pairwise distance matrix'], 'data-science': ['Rebuild a dashboard number from raw events', 'Find duplicated users inflating a metric', 'Fit a model that respects a time split', 'Correct a biased sample with known weights', 'Find the join that fans out revenue'], 'machine-learning': ['Find the label leak in a feature pipeline', 'Match a paper’s reported accuracy', 'Fix a tokenizer mismatch at inference', 'Make evaluation deterministic across runs', 'Recover from a bad learning-rate schedule'], 'model-training': ['Fix a checkpoint that will not load on CPU', 'Find why loss plateaus after warmup', 'Shard a dataset for four workers evenly', 'Fix mixed precision that overflows', 'Resume a run with the right data order'], 'mathematics': ['Verify a closed form against brute force', 'Find integer solutions to a small system', 'Fix a rounding bug in a financial formula', 'Compute exact probabilities for a dice game', 'Prove two SQL queries return the same rows'], 'optimization': ['Remove an N+1 query from a report', 'Cache a slow API without stale reads', 'Cut a Docker image from 2 GB to 200 MB', 'Make a hot loop allocation-free', 'Reduce a cold start from 8 s to 2 s'], 'games': ['Solve a nonogram from its clue file', 'Finish a maze with limited moves', 'Win tic-tac-toe variants against a script', 'Complete a text puzzle box in 30 moves', 'Find the winning line in a Go problem'], 'tool-use': ['Batch-resize images with ImageMagick', 'Extract tables from a PDF with CLI tools', 'Sync two directories over rsync safely', 'Query a SQLite file for a weekly report', 'Script tmux to lay out a dev session']}.items(): TASKS[_c] += _extra
|
| 90 |
|
| 91 |
SETTINGS = ['on Debian 12', 'on Alpine', 'in a Python 3.12 repo', 'in a Rust workspace', 'with only busybox', 'on a read-only root', 'in a Go monorepo',
|
| 92 |
+
'on Ubuntu 24.04', 'inside a container', 'on a 1 GB VM', 'with no network', 'in a Node 22 project', 'on Fedora 40', 'in a Java 21 service', 'under strict SELinux', 'in a PHP 8 app']
|
| 93 |
_setting = iter(range(10 ** 6))
|
| 94 |
_used = set()
|
| 95 |
_uses = {} # how often each title has been sampled, so collections draw the least-used titles first
|
| 96 |
|
| 97 |
|
| 98 |
+
def all_tasks(counts, band, excluded, controls_failed):
|
| 99 |
+
"""Every task of a collection: categories in its mix, and outcomes exactly as its funnel counts them."""
|
| 100 |
+
cats = [c for c, k in counts.items() for _ in range(k)]; trng.shuffle(cats)
|
| 101 |
+
plan = ['excluded'] * excluded + ['failed controls'] * controls_failed + [k for k in range(BAND + 1) for _ in range(band[k])]
|
| 102 |
+
plan += ['excluded'] * (len(cats) - len(plan)); trng.shuffle(plan)
|
| 103 |
+
pools, out = {}, []
|
| 104 |
+
for i, (c, st) in enumerate(zip(cats, plan)):
|
| 105 |
+
pool = pools.setdefault(c, trng.sample([f'{t}, {where}' for t in TASKS[c] for where in SETTINGS], len(TASKS[c]) * len(SETTINGS)))
|
| 106 |
+
name = pool.pop() if pool else f'{trng.choice(TASKS[c])}, variant {i}'
|
| 107 |
+
if st == 'excluded': out.append({'name': name, 'category': c, 'status': st, 'passes': None, 'why': trng.choice(['the reference solution is copied into the sandbox', 'answer-like files are in the sandbox', 'the verifier only checks that files exist', 'the prompt overlaps a sealed held-out task'])})
|
| 108 |
+
elif st == 'failed controls': out.append({'name': name, 'category': c, 'status': st, 'passes': None, 'why': trng.choice(['doing nothing passes the verifier', 'the reference solution fails 3 of 8 reruns'])})
|
| 109 |
+
elif st in (0, BAND): out.append({'name': name, 'category': c, 'status': 'out of band', 'passes': st, 'why': 'never solved: no learning signal' if st == 0 else 'always solved: no learning signal'})
|
| 110 |
+
else: out.append({'name': name, 'category': c, 'status': 'trained', 'passes': st, 'why': None})
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 111 |
return out
|
| 112 |
|
| 113 |
|
|
|
|
| 139 |
'source_url': None, 'categories': counts,
|
| 140 |
'funnel': {'submitted': n, 'valid': n, 'static_eligible': valid, 'controls_passed': controls, 'in_band': trained, 'trained': trained},
|
| 141 |
'band': band, 'origin': {'original': round(n * rng.uniform(0.3, 0.8)), 'adapted': 0, 'generated': 0},
|
| 142 |
+
'description': DESCRIBE[theme], 'tasks': all_tasks(counts, band, excluded, valid - controls),
|
| 143 |
'_effect': round(effect, 2)})
|
| 144 |
o = rows[-1]['origin']; rest = n - o['original']; o['adapted'] = round(rest * rng.uniform(0.3, 0.9)); o['generated'] = rest - o['adapted']
|
| 145 |
return rows
|
|
|
|
| 236 |
order = ['setup', 'snapshot', 'baseline', 'gate', 'training', 'heldout', 'collect']; stages.sort(key=lambda s: order.index(s['key']))
|
| 237 |
if metrics: next(s for s in stages if s['key'] == 'training')['training'] = {'metrics': metrics}
|
| 238 |
task_stats = [] # per trained sample task: base passes (of 8), pass rate over the first and the last 4 steps (recipe v2 records this per task)
|
| 239 |
+
for tk in env.get('tasks', []):
|
| 240 |
if tk['status'] != 'trained' or not metrics: continue
|
| 241 |
first = clamp(tk['passes'] / BAND + trng.gauss(0, 0.06), 0, 1)
|
| 242 |
last = clamp(first + (gain * 1.6 + trng.gauss(0, 0.12)) * min(1, len(metrics) / 32), 0, 1)
|
|
|
|
| 289 |
age = (NOW - datetime.fromisoformat(r['ended_at'].replace('Z', '+00:00'))).total_seconds() / 3600
|
| 290 |
r['verification'] = 'valid' if r.get('_control') else 'invalid' if rng.random() < 0.04 else 'valid' if age > rng.uniform(12, 36) else 'pending'
|
| 291 |
env = next((e for e in envs if e['id'] == r['environment_id']), None) or next(e for e in control_envs if e['id'] == r['environment_id'])
|
| 292 |
+
bad = next((t['name'] for t in env.get('tasks', []) if t['status'] == 'trained'), 'one trained task')
|
| 293 |
r['verification_note'] = {'invalid': f'Rejected: the reference solution of “{bad}” was readable inside its sandbox, so the model could copy it (found in review). The task is excluded from future runs; this run does not count.',
|
| 294 |
'valid': 'Organizer control; evidence reviewed.' if r.get('_control') else 'Evidence reviewed: held-out scores match the per-task results.', 'pending': None}[r['verification']]
|
| 295 |
for r in runs: r.pop('_control', None)
|
|
|
|
| 396 |
env = {'id': f'env-control-{key}', 'title': title, 'repo_id': f'benchflow/pta-control-{key}', 'revision': '0' * 40, 'task_count': 120, 'author': 'organizers', 'team': 'organizers',
|
| 397 |
'created_at': iso(OPEN - timedelta(days=5)), 'status': 'Validated', 'source_url': None, 'categories': mix, 'control': key, 'description': text,
|
| 398 |
'funnel': {'submitted': 120, 'valid': 120, 'static_eligible': 120, 'controls_passed': 108, 'in_band': 60, 'trained': 60}, 'band': band,
|
| 399 |
+
'origin': {'original': 120, 'adapted': 0, 'generated': 0}, 'tasks': all_tasks(mix, band, 0, 12), '_effect': effect}
|
| 400 |
out_envs.append(env)
|
| 401 |
return out_envs
|
| 402 |
|
store.py
CHANGED
|
@@ -43,7 +43,7 @@ CREATE TABLE board_meta(challenge_id TEXT PRIMARY KEY, per_run_sd_pp REAL, pendi
|
|
| 43 |
CREATE TABLE jobs(id TEXT PRIMARY KEY, name TEXT, kind TEXT, purpose TEXT, detail TEXT, run_id TEXT, challenge_id TEXT, flavor TEXT, provider TEXT,
|
| 44 |
stage TEXT, created_at TEXT, started_at TEXT, finished_at TEXT, seconds REAL, cost_usd REAL, url TEXT);
|
| 45 |
CREATE TABLE known_issues(match TEXT, cause TEXT, fixed TEXT, blame TEXT);
|
| 46 |
-
CREATE INDEX runs_collection ON runs(collection_id); CREATE INDEX runs_challenge ON runs(challenge_id);
|
| 47 |
CREATE INDEX stages_run ON run_stages(run_id); CREATE INDEX steps_run ON run_steps(run_id); CREATE INDEX tasks_collection ON tasks(collection_id);
|
| 48 |
"""
|
| 49 |
|
|
@@ -71,7 +71,7 @@ def load(path, payload, source, meta=None):
|
|
| 71 |
for e in f.get('collections', []):
|
| 72 |
ins('collections', (e['id'], e.get('title'), e.get('team') or e.get('author'), e.get('created_at'), e.get('description'), e.get('source_url'), e.get('task_count'),
|
| 73 |
e.get('control'), j(e.get('categories')), j(e.get('funnel')), j(e.get('band')), j(e.get('origin')), e.get('status'), e.get('repo_id'), e.get('revision')))
|
| 74 |
-
for t in e.get('tasks_preview') or []: ins('tasks', (e['id'], t['name'], t.get('category'), t.get('status'), t.get('passes'), t.get('why')))
|
| 75 |
for cid, m in metrics.items():
|
| 76 |
for q in m.get('queued') or []: ins('queue', (cid, q.get('environment_id'), q.get('requested_at')))
|
| 77 |
for r in m.get('runs') or []:
|
|
@@ -99,10 +99,42 @@ def load(path, payload, source, meta=None):
|
|
| 99 |
return path
|
| 100 |
|
| 101 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 102 |
def live_payload():
|
| 103 |
"""The live arena in the loader's payload shape, from the Space's own read paths (HF datasets and HF Jobs)."""
|
| 104 |
import challenges, compute_jobs
|
| 105 |
formula = challenges.formula(); metrics, boards = {}, {}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 106 |
for c in challenges.CHALLENGES:
|
| 107 |
try: metrics[c['id']] = challenges.metrics(c['id'])
|
| 108 |
except Exception: metrics[c['id']] = {'runs': []}
|
|
|
|
| 43 |
CREATE TABLE jobs(id TEXT PRIMARY KEY, name TEXT, kind TEXT, purpose TEXT, detail TEXT, run_id TEXT, challenge_id TEXT, flavor TEXT, provider TEXT,
|
| 44 |
stage TEXT, created_at TEXT, started_at TEXT, finished_at TEXT, seconds REAL, cost_usd REAL, url TEXT);
|
| 45 |
CREATE TABLE known_issues(match TEXT, cause TEXT, fixed TEXT, blame TEXT);
|
| 46 |
+
CREATE INDEX tasks_outcome ON tasks(outcome); CREATE INDEX runs_collection ON runs(collection_id); CREATE INDEX runs_challenge ON runs(challenge_id);
|
| 47 |
CREATE INDEX stages_run ON run_stages(run_id); CREATE INDEX steps_run ON run_steps(run_id); CREATE INDEX tasks_collection ON tasks(collection_id);
|
| 48 |
"""
|
| 49 |
|
|
|
|
| 71 |
for e in f.get('collections', []):
|
| 72 |
ins('collections', (e['id'], e.get('title'), e.get('team') or e.get('author'), e.get('created_at'), e.get('description'), e.get('source_url'), e.get('task_count'),
|
| 73 |
e.get('control'), j(e.get('categories')), j(e.get('funnel')), j(e.get('band')), j(e.get('origin')), e.get('status'), e.get('repo_id'), e.get('revision')))
|
| 74 |
+
for t in e.get('tasks') or e.get('tasks_preview') or []: ins('tasks', (e['id'], t['name'], t.get('category'), t.get('status'), t.get('passes'), t.get('why')))
|
| 75 |
for cid, m in metrics.items():
|
| 76 |
for q in m.get('queued') or []: ins('queue', (cid, q.get('environment_id'), q.get('requested_at')))
|
| 77 |
for r in m.get('runs') or []:
|
|
|
|
| 99 |
return path
|
| 100 |
|
| 101 |
|
| 102 |
+
# Static-gate finding codes (validation_gates.py) in plain words, for the task lists.
|
| 103 |
+
CODE_TEXT = {'S-NO-ORACLE': 'no reference solution, so the oracle control cannot run', 'S-ORACLE-STUB': 'the reference solution does no work',
|
| 104 |
+
'L-ORACLE-NETWORK': 'the reference solution downloads from the network', 'L-ORACLE-IN-IMAGE': 'the reference solution is copied into the sandbox',
|
| 105 |
+
'L-VERIFIER-IN-IMAGE': 'verifier tests or expected outputs are copied into the sandbox', 'L-TEST-FILE-IN-IMAGE': 'test files in the sandbox may reveal expected outputs',
|
| 106 |
+
'L-ANSWER-FILE': 'answer-like files are in the sandbox', 'L-BUILD-CACHE': 'build caches or version history are in the sandbox', 'L-REMOTE-ADD': 'the image adds remote content',
|
| 107 |
+
'H-NO-ASSERTIONS': 'the verifier has no assertions', 'H-EXISTENCE-ONLY': 'the verifier only checks that files exist', 'L-GRADER-DATA-IN-IMAGE': 'grading data is readable in the sandbox',
|
| 108 |
+
'H-UNCONDITIONAL-REWARD': 'the verifier always gives reward 1', 'D-NAME-COLLISION': 'same name as a sealed held-out task', 'D-NGRAM-OVERLAP': 'the prompt overlaps a sealed held-out task'}
|
| 109 |
+
|
| 110 |
+
|
| 111 |
+
def static_tasks(report):
|
| 112 |
+
"""Every task of a collection with its static-gate outcome: excluded, needs controls (no working reference solution), or eligible."""
|
| 113 |
+
out = []
|
| 114 |
+
for name, ident in sorted((report.get('task_identity') or {}).items()):
|
| 115 |
+
excluded = (report.get('excluded') or {}).get(name)
|
| 116 |
+
codes = excluded or (report.get('tasks') or {}).get(name) or []
|
| 117 |
+
outcome = 'excluded' if excluded else 'needs controls' if name in (report.get('needs_controls') or []) else 'eligible'
|
| 118 |
+
why = '; '.join(dict.fromkeys(CODE_TEXT.get(c, c) for c in codes)) or None
|
| 119 |
+
out.append({'name': name, 'category': ident.get('category') or 'other', 'status': outcome, 'why': ('flagged: ' + why) if why and outcome == 'eligible' else why})
|
| 120 |
+
return out
|
| 121 |
+
|
| 122 |
+
|
| 123 |
def live_payload():
|
| 124 |
"""The live arena in the loader's payload shape, from the Space's own read paths (HF datasets and HF Jobs)."""
|
| 125 |
import challenges, compute_jobs
|
| 126 |
formula = challenges.formula(); metrics, boards = {}, {}
|
| 127 |
+
records = {r.get('id'): r for r in challenges.collection_rows_cached()}
|
| 128 |
+
for c in formula.get('collections') or []: # every task with its static-gate outcome, from the pinned revision (cached per revision)
|
| 129 |
+
try:
|
| 130 |
+
report = challenges.static_report(records[c['id']]); c['tasks'] = static_tasks(report)
|
| 131 |
+
if 'controls_passed' not in (c.get('funnel') or {}): # static checks only: split out the tasks that still need their controls
|
| 132 |
+
c['funnel'] = {'submitted': report['summary']['tasks'], 'static_eligible': report['summary']['eligible'], 'needs_controls': len(report.get('needs_controls') or [])}
|
| 133 |
+
if not c.get('categories'):
|
| 134 |
+
cats = {}
|
| 135 |
+
for t in c['tasks']: cats[t['category']] = cats.get(t['category'], 0) + 1
|
| 136 |
+
c['categories'] = cats
|
| 137 |
+
except Exception: pass
|
| 138 |
for c in challenges.CHALLENGES:
|
| 139 |
try: metrics[c['id']] = challenges.metrics(c['id'])
|
| 140 |
except Exception: metrics[c['id']] = {'runs': []}
|
test_compose.py
CHANGED
|
@@ -263,6 +263,25 @@ class AppDataTest(unittest.TestCase):
|
|
| 263 |
compute = client.get('/api/app/compute?source=mock').json()
|
| 264 |
self.assertEqual([q['position'] for q in compute['queue'] if q['challenge_id'] == 'terminal-35b'], list(range(1, len(data['metrics']['terminal-35b']['queued']) + 1)))
|
| 265 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 266 |
def test_live_database_is_loaded_from_the_live_payload(self):
|
| 267 |
import store
|
| 268 |
payload = {'formula': {'challenges': [{'id': 'tb2-9b', 'name': 'x', 'status': 'open'}], 'collections': [{'id': 'env-1', 'title': 'Pack', 'author': 'ada', 'task_count': 3}]},
|
|
|
|
| 263 |
compute = client.get('/api/app/compute?source=mock').json()
|
| 264 |
self.assertEqual([q['position'] for q in compute['queue'] if q['challenge_id'] == 'terminal-35b'], list(range(1, len(data['metrics']['terminal-35b']['queued']) + 1)))
|
| 265 |
|
| 266 |
+
def test_every_task_is_listed_and_filtered_on_the_server(self):
|
| 267 |
+
from fastapi import FastAPI
|
| 268 |
+
from fastapi.testclient import TestClient
|
| 269 |
+
import app_api, store
|
| 270 |
+
client = TestClient((lambda a: (a.include_router(app_api.router), a)[1])(FastAPI()))
|
| 271 |
+
subs = client.get('/api/app/submissions?source=mock').json(); one = next(s for s in subs if not s['control'])
|
| 272 |
+
d = client.get(f"/api/app/tasks?source=mock&collection={one['id']}&limit=1000").json()
|
| 273 |
+
self.assertEqual(d['total'], one['task_count']); self.assertEqual(d['by_outcome'].get('trained'), one['funnel']['in_band'])
|
| 274 |
+
self.assertEqual(d['by_outcome'].get('excluded'), one['funnel']['submitted'] - one['funnel']['static_eligible'])
|
| 275 |
+
band = client.get(f"/api/app/tasks?source=mock&collection={one['id']}&outcome=trained").json()
|
| 276 |
+
self.assertTrue(band['rows'] and all(t['outcome'] == 'trained' for t in band['rows'])); self.assertEqual(band['by_outcome'], d['by_outcome'])
|
| 277 |
+
cat = next(iter(d['by_category'])); only = client.get(f"/api/app/tasks?source=mock&collection={one['id']}&cats={cat}").json()
|
| 278 |
+
self.assertEqual(only['total'], d['by_category'][cat])
|
| 279 |
+
# live: a static-gate report becomes one row per task, with codes in plain words
|
| 280 |
+
report = {'task_identity': {'a': {'category': 'debugging'}, 'b': {}, 'c': {'category': 'security'}}, 'excluded': {'b': ['L-ORACLE-IN-IMAGE']}, 'needs_controls': ['c'], 'tasks': {'a': ['H-EXISTENCE-ONLY']}}
|
| 281 |
+
rows = {t['name']: t for t in store.static_tasks(report)}
|
| 282 |
+
self.assertEqual((rows['a']['status'], rows['b']['status'], rows['c']['status']), ('eligible', 'excluded', 'needs controls'))
|
| 283 |
+
self.assertEqual(rows['b']['why'], 'the reference solution is copied into the sandbox'); self.assertTrue(rows['a']['why'].startswith('flagged: '))
|
| 284 |
+
|
| 285 |
def test_live_database_is_loaded_from_the_live_payload(self):
|
| 286 |
import store
|
| 287 |
payload = {'formula': {'challenges': [{'id': 'tb2-9b', 'name': 'x', 'status': 'open'}], 'collections': [{'id': 'env-1', 'title': 'Pack', 'author': 'ada', 'task_count': 3}]},
|