Xiangyi Li commited on
Commit
515a2d6
·
1 Parent(s): f50b2d2

Close the gaps: every task per collection, filters in the URL, filterable jobs; cut repeated information.

Browse files

Tasks: the mock lists every task with outcomes that match each collection's gate counts exactly; live lists every task of each pinned revision from its static-gate report (codes in plain words, excluded / needs controls / passed). /api/app/tasks filters and pages on the server. Every list keeps its search, filters and sort in the URL. Compute lists only jobs that belong to no run, with search and status filters. The submission page drops its summary boxes and timeline for one status line; the run page drops details its header already shows; the Runs page folds its panels into one line of filter links.

Files changed (6) hide show
  1. app_api.py +32 -7
  2. challenges.py +13 -4
  3. index.html +0 -0
  4. mock_data.py +18 -34
  5. store.py +34 -2
  6. test_compose.py +19 -0
app_api.py CHANGED
@@ -159,8 +159,6 @@ def submission(collection_id: str, source: str = Source):
159
  if r['state'] in ('running', 'queued'): r['progress'] = progress(r, r['stages'], r['steps_done'], r['steps_total'], med.get(r['challenge_id'], {}), now)
160
  latest = next((r for r in reversed(runs) if steps.get(r['id'], 0) > 2), None)
161
  curve = [dict(step=s['step'], t=s['t'], **json.loads(s['metrics'])) for s in rows(con, 'SELECT * FROM run_steps WHERE run_id = ? ORDER BY step', latest['id'])] if latest else []
162
- learnt_run = next((r for r in reversed(runs) if one(con, 'SELECT 1 x FROM run_task_stats WHERE run_id = ? LIMIT 1', r['id'])), None)
163
- learnt = rows(con, 'SELECT * FROM run_task_stats WHERE run_id = ?', learnt_run['id']) if learnt_run else []
164
  board = [parse(b, 'suites', 'run_ids') for b in rows(con, 'SELECT * FROM board WHERE collection_id = ?', collection_id)]
165
  timeline = [{'t': c['created_at'], 'what': 'submitted', 'detail': f"{c['task_count']} tasks"}]
166
  for r in runs:
@@ -172,10 +170,9 @@ def submission(collection_id: str, source: str = Source):
172
  mine_q = [q for q in queue_of(con, now) if q['collection_id'] == collection_id]
173
  for q in mine_q: timeline.append({'t': q['requested_at'], 'what': 'run queued', 'detail': f"{q['challenge_id']}, number {q['position']} in line"})
174
  timeline.sort(key=lambda x: (x['t'] or '', x.get('order', 0)))
175
- return {**c, 'tasks': rows(con, 'SELECT name, category, outcome, passes, why FROM tasks WHERE collection_id = ?', collection_id), 'runs': runs, 'board': board,
176
  'queued': [q['challenge_id'] for q in mine_q], 'queue': mine_q,
177
- 'curve': {'run_id': latest['id'], 'label': latest['label'], 'steps': curve} if latest else None,
178
- 'learnt': {'run_id': learnt_run['id'], 'label': learnt_run['label'], 'tasks': learnt} if learnt_run else None, 'timeline': timeline}
179
 
180
 
181
  @router.get('/runs')
@@ -209,10 +206,38 @@ def run_detail(run_id: str, source: str = Source):
209
  'job': one(con, 'SELECT * FROM jobs WHERE run_id = ?', run_id), 'method': method}
210
 
211
 
 
 
 
212
  @router.get('/tasks')
213
- def tasks_list(source: str = Source):
 
 
 
 
214
  with db(source) as con:
215
- return rows(con, 'SELECT t.*, c.title AS collection_title, c.team FROM tasks t JOIN collections c ON c.id = t.collection_id ORDER BY c.title, t.name')
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
216
 
217
 
218
  @router.get('/leaderboard')
 
159
  if r['state'] in ('running', 'queued'): r['progress'] = progress(r, r['stages'], r['steps_done'], r['steps_total'], med.get(r['challenge_id'], {}), now)
160
  latest = next((r for r in reversed(runs) if steps.get(r['id'], 0) > 2), None)
161
  curve = [dict(step=s['step'], t=s['t'], **json.loads(s['metrics'])) for s in rows(con, 'SELECT * FROM run_steps WHERE run_id = ? ORDER BY step', latest['id'])] if latest else []
 
 
162
  board = [parse(b, 'suites', 'run_ids') for b in rows(con, 'SELECT * FROM board WHERE collection_id = ?', collection_id)]
163
  timeline = [{'t': c['created_at'], 'what': 'submitted', 'detail': f"{c['task_count']} tasks"}]
164
  for r in runs:
 
170
  mine_q = [q for q in queue_of(con, now) if q['collection_id'] == collection_id]
171
  for q in mine_q: timeline.append({'t': q['requested_at'], 'what': 'run queued', 'detail': f"{q['challenge_id']}, number {q['position']} in line"})
172
  timeline.sort(key=lambda x: (x['t'] or '', x.get('order', 0)))
173
+ return {**c, 'runs': runs, 'board': board,
174
  'queued': [q['challenge_id'] for q in mine_q], 'queue': mine_q,
175
+ 'curve': {'run_id': latest['id'], 'label': latest['label'], 'steps': curve} if latest else None, 'timeline': timeline}
 
176
 
177
 
178
  @router.get('/runs')
 
206
  'job': one(con, 'SELECT * FROM jobs WHERE run_id = ?', run_id), 'method': method}
207
 
208
 
209
+ OUTCOME_ORDER = "CASE t.outcome WHEN 'trained' THEN 0 WHEN 'eligible' THEN 1 WHEN 'needs controls' THEN 2 WHEN 'out of band' THEN 3 WHEN 'failed controls' THEN 4 WHEN 'excluded' THEN 5 ELSE 6 END"
210
+
211
+
212
  @router.get('/tasks')
213
+ def tasks_list(source: str = Source, q: str | None = None, outcome: str | None = None, cats: str | None = None, collection: str | None = None,
214
+ limit: int = Query(100, ge=1, le=1000), offset: int = Query(0, ge=0)):
215
+ """Every task, filtered on the server: search, gate outcome, categories (comma-separated) and submission. Counts per outcome
216
+ ignore the outcome filter and counts per category ignore the category filter, so filter chips show what each choice would give.
217
+ For one submission, each trained task carries its pass rate early and late in the submission's latest run that recorded it."""
218
  with db(source) as con:
219
+ base, args = ['1 = 1'], []
220
+ if collection: base.append('t.collection_id = ?'); args.append(collection)
221
+ if q: base.append('(t.name LIKE ? OR c.title LIKE ? OR c.team LIKE ?)'); args += [f'%{q}%'] * 3
222
+ cat_list = [x for x in (cats or '').split(',') if x]
223
+ by_cat = (f"t.category IN ({','.join('?' * len(cat_list))})", cat_list) if cat_list else ('1 = 1', [])
224
+ by_out = ('t.outcome = ?', [outcome]) if outcome else ('1 = 1', [])
225
+ frm = 'FROM tasks t JOIN collections c ON c.id = t.collection_id WHERE ' + ' AND '.join(base)
226
+ counts_out = {r['outcome']: r['n'] for r in rows(con, f'SELECT t.outcome, COUNT(*) n {frm} AND {by_cat[0]} GROUP BY t.outcome', *args, *by_cat[1])}
227
+ counts_cat = {r['category']: r['n'] for r in rows(con, f'SELECT t.category, COUNT(*) n {frm} AND {by_out[0]} GROUP BY t.category', *args, *by_out[1])}
228
+ where = f'{frm} AND {by_cat[0]} AND {by_out[0]}'; wargs = [*args, *by_cat[1], *by_out[1]]
229
+ total = one(con, f'SELECT COUNT(*) n {where}', *wargs)['n']
230
+ out = rows(con, f'SELECT t.*, c.title AS collection_title, c.team {where} ORDER BY {OUTCOME_ORDER}, c.title, t.name LIMIT ? OFFSET ?', *wargs, limit, offset)
231
+ learnt = None
232
+ if collection:
233
+ run = one(con, 'SELECT r.id, r.label FROM runs r WHERE r.collection_id = ? AND EXISTS (SELECT 1 FROM run_task_stats s WHERE s.run_id = r.id) '
234
+ 'ORDER BY COALESCE(r.started_at, r.created_at) DESC LIMIT 1', collection)
235
+ if run:
236
+ stats = {x['name']: x for x in rows(con, 'SELECT name, start, finish FROM run_task_stats WHERE run_id = ?', run['id'])}
237
+ for t in out:
238
+ if t['name'] in stats: t['start'], t['finish'] = stats[t['name']]['start'], stats[t['name']]['finish']
239
+ learnt = run
240
+ return {'total': total, 'rows': out, 'by_outcome': counts_out, 'by_category': counts_cat, 'learnt_run': learnt}
241
 
242
 
243
  @router.get('/leaderboard')
challenges.py CHANGED
@@ -423,12 +423,21 @@ def eligibility(record):
423
  stored static report when it was computed under this policy, otherwise re-checked from the pinned source once per
424
  process (read-only, like the gate plan)."""
425
  static=(record.get('quality_gates') or {}).get('static') or {}
426
- if not (static.get('version')==gates.VERSION and (static.get('policy') or {}).get('require_oracle')==gates.REQUIRE_ORACLE and 'excluded' in static):
427
- key=(record['id'],record['revision'],gates.VERSION)
428
- if key not in _eligibility: _eligibility[key]=gates.compact(pinned_static(record,gates.REQUIRE_ORACLE))
429
- static=_eligibility[key]
430
  return static['summary']['eligible'],static['summary']['tasks'],static['excluded'],static.get('needs_controls',[])
431
 
 
 
 
 
 
 
 
 
 
 
 
 
432
  def mirror_check(row,source):
433
  """Whether the pinned submission can be mirrored for a run, from file listings only: nothing is copied."""
434
  client=hub();target=mirror_path(source['id'],source['revision'])
 
423
  stored static report when it was computed under this policy, otherwise re-checked from the pinned source once per
424
  process (read-only, like the gate plan)."""
425
  static=(record.get('quality_gates') or {}).get('static') or {}
426
+ if not static_current(static): static=static_report(record)
 
 
 
427
  return static['summary']['eligible'],static['summary']['tasks'],static['excluded'],static.get('needs_controls',[])
428
 
429
+ def static_current(static):
430
+ return static.get('version')==gates.VERSION and (static.get('policy') or {}).get('require_oracle')==gates.REQUIRE_ORACLE and 'excluded' in static
431
+
432
+ def static_report(record):
433
+ """The compact static-gate report of a record under the current policy: the stored one when it is current and lists
434
+ every task, otherwise recomputed from the pinned source once per process (read-only)."""
435
+ static=(record.get('quality_gates') or {}).get('static') or {}
436
+ if static_current(static) and 'task_identity' in static: return static
437
+ key=(record['id'],record['revision'],gates.VERSION)
438
+ if key not in _eligibility: _eligibility[key]=gates.compact(pinned_static(record,gates.REQUIRE_ORACLE))
439
+ return _eligibility[key]
440
+
441
  def mirror_check(row,source):
442
  """Whether the pinned submission can be mirrored for a run, from file listings only: nothing is copied."""
443
  client=hub();target=mirror_path(source['id'],source['revision'])
index.html CHANGED
The diff for this file is too large to render. See raw diff
 
mock_data.py CHANGED
@@ -89,41 +89,25 @@ DESCRIBE = {
89
  for _c, _extra in {'system-administration': ['Move a service to a new user without downtime', 'Find what keeps filling /tmp at 3 am', 'Repair a broken PAM stack from a rescue shell', 'Pin a kernel module that breaks on upgrade', 'Migrate crontabs to systemd timers'], 'debugging': ['Find the off-by-one in a pagination bug', 'Explain a memory spike after a config reload', 'Fix a deadlock between two worker pools', 'Find why a test passes alone but fails in the suite', 'Track down a timezone bug in nightly reports'], 'security': ['Find the SSRF in a small web service', 'Remove secrets committed to a repo’s history', 'Lock down a world-writable cron directory', 'Patch a path traversal in a file server', 'Find the weak JWT signing configuration'], 'software-engineering': ['Replace a hand-rolled argument parser', 'Add retries with backoff to a flaky client', 'Remove a circular import without behaviour change', 'Make a CLI’s JSON output stable across versions', 'Upgrade a library across a breaking API change'], 'data-processing': ['Parse a fixed-width mainframe export', 'Convert timestamps from five time zones to UTC', 'Fill gaps in a sensor series without inventing peaks', 'Split one giant XML file into valid records', 'Match customer records across two spellings'], 'file-operations': ['Recover files from a half-written zip', 'Rename 10,000 files from a CSV mapping', 'Find which files changed between two backups', 'Fix permissions on a copied home directory', 'Deduplicate a directory with hard links'], 'scientific-computing': ['Fix a unit mix-up in a simulation input', 'Make a Monte Carlo run reproducible', 'Port a MATLAB script to NumPy exactly', 'Interpolate missing values in a climate grid', 'Speed up a pairwise distance matrix'], 'data-science': ['Rebuild a dashboard number from raw events', 'Find duplicated users inflating a metric', 'Fit a model that respects a time split', 'Correct a biased sample with known weights', 'Find the join that fans out revenue'], 'machine-learning': ['Find the label leak in a feature pipeline', 'Match a paper’s reported accuracy', 'Fix a tokenizer mismatch at inference', 'Make evaluation deterministic across runs', 'Recover from a bad learning-rate schedule'], 'model-training': ['Fix a checkpoint that will not load on CPU', 'Find why loss plateaus after warmup', 'Shard a dataset for four workers evenly', 'Fix mixed precision that overflows', 'Resume a run with the right data order'], 'mathematics': ['Verify a closed form against brute force', 'Find integer solutions to a small system', 'Fix a rounding bug in a financial formula', 'Compute exact probabilities for a dice game', 'Prove two SQL queries return the same rows'], 'optimization': ['Remove an N+1 query from a report', 'Cache a slow API without stale reads', 'Cut a Docker image from 2 GB to 200 MB', 'Make a hot loop allocation-free', 'Reduce a cold start from 8 s to 2 s'], 'games': ['Solve a nonogram from its clue file', 'Finish a maze with limited moves', 'Win tic-tac-toe variants against a script', 'Complete a text puzzle box in 30 moves', 'Find the winning line in a Go problem'], 'tool-use': ['Batch-resize images with ImageMagick', 'Extract tables from a PDF with CLI tools', 'Sync two directories over rsync safely', 'Query a SQLite file for a weekly report', 'Script tmux to lay out a dev session']}.items(): TASKS[_c] += _extra
90
 
91
  SETTINGS = ['on Debian 12', 'on Alpine', 'in a Python 3.12 repo', 'in a Rust workspace', 'with only busybox', 'on a read-only root', 'in a Go monorepo',
92
- 'on Ubuntu 24.04', 'inside a container', 'on a 1 GB VM', 'with no network', 'in a Node 22 project']
93
  _setting = iter(range(10 ** 6))
94
  _used = set()
95
  _uses = {} # how often each title has been sampled, so collections draw the least-used titles first
96
 
97
 
98
- def sample_tasks(counts, band, excluded, controls_failed, k=10):
99
- """k example tasks in the collection's category mix (no repeats), with outcomes in the same proportions as its funnel."""
100
- total = sum(counts.values()); cats = sorted(counts, key=lambda c: -counts[c]); names = []
101
- quota = {c: max(1, round(k * counts[c] / total)) for c in cats}
102
- fresh = lambda c: sorted(TASKS[c], key=lambda t: (_uses.get(t, 0), trng.random())) # least-used titles first
103
- for c in cats:
104
- for t in fresh(c)[:quota[c]]:
105
- if len(names) < k: names.append((t, c))
106
- for c in cats: # top up from the largest categories when quotas round short
107
- for t in fresh(c):
108
- if len(names) < k and (t, c) not in names: names.append((t, c))
109
- for t, _ in names: _uses[t] = _uses.get(t, 0) + 1
110
- in_band = sum(band[1:BAND]); out_band = band[0] + band[BAND]
111
- shares = [('excluded', excluded), ('failed controls', controls_failed), ('out of band', out_band), ('trained', in_band)]
112
- plan = [st for st, n in shares for _ in range(round(k * n / max(1, total)))]
113
- plan = (plan + ['trained'] * k)[:len(names)]; trng.shuffle(plan)
114
- out = []
115
- base = next(_setting)
116
- for (title, c), st in zip(names, plan):
117
- if title in _used: # a title another collection already has gets its own setting
118
- for shift in range(len(SETTINGS)):
119
- full = f'{title}, {SETTINGS[(base + shift) % len(SETTINGS)]}'
120
- if full not in _used: break
121
- title = full
122
- _used.add(title)
123
- if st == 'excluded': out.append({'name': title, 'category': c, 'status': st, 'passes': None, 'why': trng.choice(['reference solution readable inside the sandbox', 'answer file baked into the image', 'verifier only checks that a file exists'])})
124
- elif st == 'failed controls': out.append({'name': title, 'category': c, 'status': st, 'passes': None, 'why': trng.choice(['doing nothing passes the verifier', 'the reference solution fails 3 of 8 reruns'])})
125
- elif st == 'out of band': kb = 0 if trng.random() < band[0] / max(1, out_band) else BAND; out.append({'name': title, 'category': c, 'status': st, 'passes': kb, 'why': 'never solved: no learning signal' if kb == 0 else 'always solved: no learning signal'})
126
- else: out.append({'name': title, 'category': c, 'status': 'trained', 'passes': trng.choices(range(1, BAND), weights=[band[j] + 0.01 for j in range(1, BAND)])[0], 'why': None})
127
  return out
128
 
129
 
@@ -155,7 +139,7 @@ def collections():
155
  'source_url': None, 'categories': counts,
156
  'funnel': {'submitted': n, 'valid': n, 'static_eligible': valid, 'controls_passed': controls, 'in_band': trained, 'trained': trained},
157
  'band': band, 'origin': {'original': round(n * rng.uniform(0.3, 0.8)), 'adapted': 0, 'generated': 0},
158
- 'description': DESCRIBE[theme], 'tasks_preview': sample_tasks(counts, band, excluded, valid - controls),
159
  '_effect': round(effect, 2)})
160
  o = rows[-1]['origin']; rest = n - o['original']; o['adapted'] = round(rest * rng.uniform(0.3, 0.9)); o['generated'] = rest - o['adapted']
161
  return rows
@@ -252,7 +236,7 @@ def run_record(i, env, start, fate, now):
252
  order = ['setup', 'snapshot', 'baseline', 'gate', 'training', 'heldout', 'collect']; stages.sort(key=lambda s: order.index(s['key']))
253
  if metrics: next(s for s in stages if s['key'] == 'training')['training'] = {'metrics': metrics}
254
  task_stats = [] # per trained sample task: base passes (of 8), pass rate over the first and the last 4 steps (recipe v2 records this per task)
255
- for tk in env.get('tasks_preview', []):
256
  if tk['status'] != 'trained' or not metrics: continue
257
  first = clamp(tk['passes'] / BAND + trng.gauss(0, 0.06), 0, 1)
258
  last = clamp(first + (gain * 1.6 + trng.gauss(0, 0.12)) * min(1, len(metrics) / 32), 0, 1)
@@ -305,7 +289,7 @@ def schedule(envs, control_envs=()):
305
  age = (NOW - datetime.fromisoformat(r['ended_at'].replace('Z', '+00:00'))).total_seconds() / 3600
306
  r['verification'] = 'valid' if r.get('_control') else 'invalid' if rng.random() < 0.04 else 'valid' if age > rng.uniform(12, 36) else 'pending'
307
  env = next((e for e in envs if e['id'] == r['environment_id']), None) or next(e for e in control_envs if e['id'] == r['environment_id'])
308
- bad = next((t['name'] for t in env.get('tasks_preview', []) if t['status'] == 'trained'), 'one trained task')
309
  r['verification_note'] = {'invalid': f'Rejected: the reference solution of “{bad}” was readable inside its sandbox, so the model could copy it (found in review). The task is excluded from future runs; this run does not count.',
310
  'valid': 'Organizer control; evidence reviewed.' if r.get('_control') else 'Evidence reviewed: held-out scores match the per-task results.', 'pending': None}[r['verification']]
311
  for r in runs: r.pop('_control', None)
@@ -412,7 +396,7 @@ def controls():
412
  env = {'id': f'env-control-{key}', 'title': title, 'repo_id': f'benchflow/pta-control-{key}', 'revision': '0' * 40, 'task_count': 120, 'author': 'organizers', 'team': 'organizers',
413
  'created_at': iso(OPEN - timedelta(days=5)), 'status': 'Validated', 'source_url': None, 'categories': mix, 'control': key, 'description': text,
414
  'funnel': {'submitted': 120, 'valid': 120, 'static_eligible': 120, 'controls_passed': 108, 'in_band': 60, 'trained': 60}, 'band': band,
415
- 'origin': {'original': 120, 'adapted': 0, 'generated': 0}, 'tasks_preview': [], '_effect': effect}
416
  out_envs.append(env)
417
  return out_envs
418
 
 
89
  for _c, _extra in {'system-administration': ['Move a service to a new user without downtime', 'Find what keeps filling /tmp at 3 am', 'Repair a broken PAM stack from a rescue shell', 'Pin a kernel module that breaks on upgrade', 'Migrate crontabs to systemd timers'], 'debugging': ['Find the off-by-one in a pagination bug', 'Explain a memory spike after a config reload', 'Fix a deadlock between two worker pools', 'Find why a test passes alone but fails in the suite', 'Track down a timezone bug in nightly reports'], 'security': ['Find the SSRF in a small web service', 'Remove secrets committed to a repo’s history', 'Lock down a world-writable cron directory', 'Patch a path traversal in a file server', 'Find the weak JWT signing configuration'], 'software-engineering': ['Replace a hand-rolled argument parser', 'Add retries with backoff to a flaky client', 'Remove a circular import without behaviour change', 'Make a CLI’s JSON output stable across versions', 'Upgrade a library across a breaking API change'], 'data-processing': ['Parse a fixed-width mainframe export', 'Convert timestamps from five time zones to UTC', 'Fill gaps in a sensor series without inventing peaks', 'Split one giant XML file into valid records', 'Match customer records across two spellings'], 'file-operations': ['Recover files from a half-written zip', 'Rename 10,000 files from a CSV mapping', 'Find which files changed between two backups', 'Fix permissions on a copied home directory', 'Deduplicate a directory with hard links'], 'scientific-computing': ['Fix a unit mix-up in a simulation input', 'Make a Monte Carlo run reproducible', 'Port a MATLAB script to NumPy exactly', 'Interpolate missing values in a climate grid', 'Speed up a pairwise distance matrix'], 'data-science': ['Rebuild a dashboard number from raw events', 'Find duplicated users inflating a metric', 'Fit a model that respects a time split', 'Correct a biased sample with known weights', 'Find the join that fans out revenue'], 'machine-learning': ['Find the label leak in a feature pipeline', 'Match a paper’s reported accuracy', 'Fix a tokenizer mismatch at inference', 'Make evaluation deterministic across runs', 'Recover from a bad learning-rate schedule'], 'model-training': ['Fix a checkpoint that will not load on CPU', 'Find why loss plateaus after warmup', 'Shard a dataset for four workers evenly', 'Fix mixed precision that overflows', 'Resume a run with the right data order'], 'mathematics': ['Verify a closed form against brute force', 'Find integer solutions to a small system', 'Fix a rounding bug in a financial formula', 'Compute exact probabilities for a dice game', 'Prove two SQL queries return the same rows'], 'optimization': ['Remove an N+1 query from a report', 'Cache a slow API without stale reads', 'Cut a Docker image from 2 GB to 200 MB', 'Make a hot loop allocation-free', 'Reduce a cold start from 8 s to 2 s'], 'games': ['Solve a nonogram from its clue file', 'Finish a maze with limited moves', 'Win tic-tac-toe variants against a script', 'Complete a text puzzle box in 30 moves', 'Find the winning line in a Go problem'], 'tool-use': ['Batch-resize images with ImageMagick', 'Extract tables from a PDF with CLI tools', 'Sync two directories over rsync safely', 'Query a SQLite file for a weekly report', 'Script tmux to lay out a dev session']}.items(): TASKS[_c] += _extra
90
 
91
  SETTINGS = ['on Debian 12', 'on Alpine', 'in a Python 3.12 repo', 'in a Rust workspace', 'with only busybox', 'on a read-only root', 'in a Go monorepo',
92
+ 'on Ubuntu 24.04', 'inside a container', 'on a 1 GB VM', 'with no network', 'in a Node 22 project', 'on Fedora 40', 'in a Java 21 service', 'under strict SELinux', 'in a PHP 8 app']
93
  _setting = iter(range(10 ** 6))
94
  _used = set()
95
  _uses = {} # how often each title has been sampled, so collections draw the least-used titles first
96
 
97
 
98
+ def all_tasks(counts, band, excluded, controls_failed):
99
+ """Every task of a collection: categories in its mix, and outcomes exactly as its funnel counts them."""
100
+ cats = [c for c, k in counts.items() for _ in range(k)]; trng.shuffle(cats)
101
+ plan = ['excluded'] * excluded + ['failed controls'] * controls_failed + [k for k in range(BAND + 1) for _ in range(band[k])]
102
+ plan += ['excluded'] * (len(cats) - len(plan)); trng.shuffle(plan)
103
+ pools, out = {}, []
104
+ for i, (c, st) in enumerate(zip(cats, plan)):
105
+ pool = pools.setdefault(c, trng.sample([f'{t}, {where}' for t in TASKS[c] for where in SETTINGS], len(TASKS[c]) * len(SETTINGS)))
106
+ name = pool.pop() if pool else f'{trng.choice(TASKS[c])}, variant {i}'
107
+ if st == 'excluded': out.append({'name': name, 'category': c, 'status': st, 'passes': None, 'why': trng.choice(['the reference solution is copied into the sandbox', 'answer-like files are in the sandbox', 'the verifier only checks that files exist', 'the prompt overlaps a sealed held-out task'])})
108
+ elif st == 'failed controls': out.append({'name': name, 'category': c, 'status': st, 'passes': None, 'why': trng.choice(['doing nothing passes the verifier', 'the reference solution fails 3 of 8 reruns'])})
109
+ elif st in (0, BAND): out.append({'name': name, 'category': c, 'status': 'out of band', 'passes': st, 'why': 'never solved: no learning signal' if st == 0 else 'always solved: no learning signal'})
110
+ else: out.append({'name': name, 'category': c, 'status': 'trained', 'passes': st, 'why': None})
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
111
  return out
112
 
113
 
 
139
  'source_url': None, 'categories': counts,
140
  'funnel': {'submitted': n, 'valid': n, 'static_eligible': valid, 'controls_passed': controls, 'in_band': trained, 'trained': trained},
141
  'band': band, 'origin': {'original': round(n * rng.uniform(0.3, 0.8)), 'adapted': 0, 'generated': 0},
142
+ 'description': DESCRIBE[theme], 'tasks': all_tasks(counts, band, excluded, valid - controls),
143
  '_effect': round(effect, 2)})
144
  o = rows[-1]['origin']; rest = n - o['original']; o['adapted'] = round(rest * rng.uniform(0.3, 0.9)); o['generated'] = rest - o['adapted']
145
  return rows
 
236
  order = ['setup', 'snapshot', 'baseline', 'gate', 'training', 'heldout', 'collect']; stages.sort(key=lambda s: order.index(s['key']))
237
  if metrics: next(s for s in stages if s['key'] == 'training')['training'] = {'metrics': metrics}
238
  task_stats = [] # per trained sample task: base passes (of 8), pass rate over the first and the last 4 steps (recipe v2 records this per task)
239
+ for tk in env.get('tasks', []):
240
  if tk['status'] != 'trained' or not metrics: continue
241
  first = clamp(tk['passes'] / BAND + trng.gauss(0, 0.06), 0, 1)
242
  last = clamp(first + (gain * 1.6 + trng.gauss(0, 0.12)) * min(1, len(metrics) / 32), 0, 1)
 
289
  age = (NOW - datetime.fromisoformat(r['ended_at'].replace('Z', '+00:00'))).total_seconds() / 3600
290
  r['verification'] = 'valid' if r.get('_control') else 'invalid' if rng.random() < 0.04 else 'valid' if age > rng.uniform(12, 36) else 'pending'
291
  env = next((e for e in envs if e['id'] == r['environment_id']), None) or next(e for e in control_envs if e['id'] == r['environment_id'])
292
+ bad = next((t['name'] for t in env.get('tasks', []) if t['status'] == 'trained'), 'one trained task')
293
  r['verification_note'] = {'invalid': f'Rejected: the reference solution of “{bad}” was readable inside its sandbox, so the model could copy it (found in review). The task is excluded from future runs; this run does not count.',
294
  'valid': 'Organizer control; evidence reviewed.' if r.get('_control') else 'Evidence reviewed: held-out scores match the per-task results.', 'pending': None}[r['verification']]
295
  for r in runs: r.pop('_control', None)
 
396
  env = {'id': f'env-control-{key}', 'title': title, 'repo_id': f'benchflow/pta-control-{key}', 'revision': '0' * 40, 'task_count': 120, 'author': 'organizers', 'team': 'organizers',
397
  'created_at': iso(OPEN - timedelta(days=5)), 'status': 'Validated', 'source_url': None, 'categories': mix, 'control': key, 'description': text,
398
  'funnel': {'submitted': 120, 'valid': 120, 'static_eligible': 120, 'controls_passed': 108, 'in_band': 60, 'trained': 60}, 'band': band,
399
+ 'origin': {'original': 120, 'adapted': 0, 'generated': 0}, 'tasks': all_tasks(mix, band, 0, 12), '_effect': effect}
400
  out_envs.append(env)
401
  return out_envs
402
 
store.py CHANGED
@@ -43,7 +43,7 @@ CREATE TABLE board_meta(challenge_id TEXT PRIMARY KEY, per_run_sd_pp REAL, pendi
43
  CREATE TABLE jobs(id TEXT PRIMARY KEY, name TEXT, kind TEXT, purpose TEXT, detail TEXT, run_id TEXT, challenge_id TEXT, flavor TEXT, provider TEXT,
44
  stage TEXT, created_at TEXT, started_at TEXT, finished_at TEXT, seconds REAL, cost_usd REAL, url TEXT);
45
  CREATE TABLE known_issues(match TEXT, cause TEXT, fixed TEXT, blame TEXT);
46
- CREATE INDEX runs_collection ON runs(collection_id); CREATE INDEX runs_challenge ON runs(challenge_id);
47
  CREATE INDEX stages_run ON run_stages(run_id); CREATE INDEX steps_run ON run_steps(run_id); CREATE INDEX tasks_collection ON tasks(collection_id);
48
  """
49
 
@@ -71,7 +71,7 @@ def load(path, payload, source, meta=None):
71
  for e in f.get('collections', []):
72
  ins('collections', (e['id'], e.get('title'), e.get('team') or e.get('author'), e.get('created_at'), e.get('description'), e.get('source_url'), e.get('task_count'),
73
  e.get('control'), j(e.get('categories')), j(e.get('funnel')), j(e.get('band')), j(e.get('origin')), e.get('status'), e.get('repo_id'), e.get('revision')))
74
- for t in e.get('tasks_preview') or []: ins('tasks', (e['id'], t['name'], t.get('category'), t.get('status'), t.get('passes'), t.get('why')))
75
  for cid, m in metrics.items():
76
  for q in m.get('queued') or []: ins('queue', (cid, q.get('environment_id'), q.get('requested_at')))
77
  for r in m.get('runs') or []:
@@ -99,10 +99,42 @@ def load(path, payload, source, meta=None):
99
  return path
100
 
101
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
102
  def live_payload():
103
  """The live arena in the loader's payload shape, from the Space's own read paths (HF datasets and HF Jobs)."""
104
  import challenges, compute_jobs
105
  formula = challenges.formula(); metrics, boards = {}, {}
 
 
 
 
 
 
 
 
 
 
 
106
  for c in challenges.CHALLENGES:
107
  try: metrics[c['id']] = challenges.metrics(c['id'])
108
  except Exception: metrics[c['id']] = {'runs': []}
 
43
  CREATE TABLE jobs(id TEXT PRIMARY KEY, name TEXT, kind TEXT, purpose TEXT, detail TEXT, run_id TEXT, challenge_id TEXT, flavor TEXT, provider TEXT,
44
  stage TEXT, created_at TEXT, started_at TEXT, finished_at TEXT, seconds REAL, cost_usd REAL, url TEXT);
45
  CREATE TABLE known_issues(match TEXT, cause TEXT, fixed TEXT, blame TEXT);
46
+ CREATE INDEX tasks_outcome ON tasks(outcome); CREATE INDEX runs_collection ON runs(collection_id); CREATE INDEX runs_challenge ON runs(challenge_id);
47
  CREATE INDEX stages_run ON run_stages(run_id); CREATE INDEX steps_run ON run_steps(run_id); CREATE INDEX tasks_collection ON tasks(collection_id);
48
  """
49
 
 
71
  for e in f.get('collections', []):
72
  ins('collections', (e['id'], e.get('title'), e.get('team') or e.get('author'), e.get('created_at'), e.get('description'), e.get('source_url'), e.get('task_count'),
73
  e.get('control'), j(e.get('categories')), j(e.get('funnel')), j(e.get('band')), j(e.get('origin')), e.get('status'), e.get('repo_id'), e.get('revision')))
74
+ for t in e.get('tasks') or e.get('tasks_preview') or []: ins('tasks', (e['id'], t['name'], t.get('category'), t.get('status'), t.get('passes'), t.get('why')))
75
  for cid, m in metrics.items():
76
  for q in m.get('queued') or []: ins('queue', (cid, q.get('environment_id'), q.get('requested_at')))
77
  for r in m.get('runs') or []:
 
99
  return path
100
 
101
 
102
+ # Static-gate finding codes (validation_gates.py) in plain words, for the task lists.
103
+ CODE_TEXT = {'S-NO-ORACLE': 'no reference solution, so the oracle control cannot run', 'S-ORACLE-STUB': 'the reference solution does no work',
104
+ 'L-ORACLE-NETWORK': 'the reference solution downloads from the network', 'L-ORACLE-IN-IMAGE': 'the reference solution is copied into the sandbox',
105
+ 'L-VERIFIER-IN-IMAGE': 'verifier tests or expected outputs are copied into the sandbox', 'L-TEST-FILE-IN-IMAGE': 'test files in the sandbox may reveal expected outputs',
106
+ 'L-ANSWER-FILE': 'answer-like files are in the sandbox', 'L-BUILD-CACHE': 'build caches or version history are in the sandbox', 'L-REMOTE-ADD': 'the image adds remote content',
107
+ 'H-NO-ASSERTIONS': 'the verifier has no assertions', 'H-EXISTENCE-ONLY': 'the verifier only checks that files exist', 'L-GRADER-DATA-IN-IMAGE': 'grading data is readable in the sandbox',
108
+ 'H-UNCONDITIONAL-REWARD': 'the verifier always gives reward 1', 'D-NAME-COLLISION': 'same name as a sealed held-out task', 'D-NGRAM-OVERLAP': 'the prompt overlaps a sealed held-out task'}
109
+
110
+
111
+ def static_tasks(report):
112
+ """Every task of a collection with its static-gate outcome: excluded, needs controls (no working reference solution), or eligible."""
113
+ out = []
114
+ for name, ident in sorted((report.get('task_identity') or {}).items()):
115
+ excluded = (report.get('excluded') or {}).get(name)
116
+ codes = excluded or (report.get('tasks') or {}).get(name) or []
117
+ outcome = 'excluded' if excluded else 'needs controls' if name in (report.get('needs_controls') or []) else 'eligible'
118
+ why = '; '.join(dict.fromkeys(CODE_TEXT.get(c, c) for c in codes)) or None
119
+ out.append({'name': name, 'category': ident.get('category') or 'other', 'status': outcome, 'why': ('flagged: ' + why) if why and outcome == 'eligible' else why})
120
+ return out
121
+
122
+
123
  def live_payload():
124
  """The live arena in the loader's payload shape, from the Space's own read paths (HF datasets and HF Jobs)."""
125
  import challenges, compute_jobs
126
  formula = challenges.formula(); metrics, boards = {}, {}
127
+ records = {r.get('id'): r for r in challenges.collection_rows_cached()}
128
+ for c in formula.get('collections') or []: # every task with its static-gate outcome, from the pinned revision (cached per revision)
129
+ try:
130
+ report = challenges.static_report(records[c['id']]); c['tasks'] = static_tasks(report)
131
+ if 'controls_passed' not in (c.get('funnel') or {}): # static checks only: split out the tasks that still need their controls
132
+ c['funnel'] = {'submitted': report['summary']['tasks'], 'static_eligible': report['summary']['eligible'], 'needs_controls': len(report.get('needs_controls') or [])}
133
+ if not c.get('categories'):
134
+ cats = {}
135
+ for t in c['tasks']: cats[t['category']] = cats.get(t['category'], 0) + 1
136
+ c['categories'] = cats
137
+ except Exception: pass
138
  for c in challenges.CHALLENGES:
139
  try: metrics[c['id']] = challenges.metrics(c['id'])
140
  except Exception: metrics[c['id']] = {'runs': []}
test_compose.py CHANGED
@@ -263,6 +263,25 @@ class AppDataTest(unittest.TestCase):
263
  compute = client.get('/api/app/compute?source=mock').json()
264
  self.assertEqual([q['position'] for q in compute['queue'] if q['challenge_id'] == 'terminal-35b'], list(range(1, len(data['metrics']['terminal-35b']['queued']) + 1)))
265
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
266
  def test_live_database_is_loaded_from_the_live_payload(self):
267
  import store
268
  payload = {'formula': {'challenges': [{'id': 'tb2-9b', 'name': 'x', 'status': 'open'}], 'collections': [{'id': 'env-1', 'title': 'Pack', 'author': 'ada', 'task_count': 3}]},
 
263
  compute = client.get('/api/app/compute?source=mock').json()
264
  self.assertEqual([q['position'] for q in compute['queue'] if q['challenge_id'] == 'terminal-35b'], list(range(1, len(data['metrics']['terminal-35b']['queued']) + 1)))
265
 
266
+ def test_every_task_is_listed_and_filtered_on_the_server(self):
267
+ from fastapi import FastAPI
268
+ from fastapi.testclient import TestClient
269
+ import app_api, store
270
+ client = TestClient((lambda a: (a.include_router(app_api.router), a)[1])(FastAPI()))
271
+ subs = client.get('/api/app/submissions?source=mock').json(); one = next(s for s in subs if not s['control'])
272
+ d = client.get(f"/api/app/tasks?source=mock&collection={one['id']}&limit=1000").json()
273
+ self.assertEqual(d['total'], one['task_count']); self.assertEqual(d['by_outcome'].get('trained'), one['funnel']['in_band'])
274
+ self.assertEqual(d['by_outcome'].get('excluded'), one['funnel']['submitted'] - one['funnel']['static_eligible'])
275
+ band = client.get(f"/api/app/tasks?source=mock&collection={one['id']}&outcome=trained").json()
276
+ self.assertTrue(band['rows'] and all(t['outcome'] == 'trained' for t in band['rows'])); self.assertEqual(band['by_outcome'], d['by_outcome'])
277
+ cat = next(iter(d['by_category'])); only = client.get(f"/api/app/tasks?source=mock&collection={one['id']}&cats={cat}").json()
278
+ self.assertEqual(only['total'], d['by_category'][cat])
279
+ # live: a static-gate report becomes one row per task, with codes in plain words
280
+ report = {'task_identity': {'a': {'category': 'debugging'}, 'b': {}, 'c': {'category': 'security'}}, 'excluded': {'b': ['L-ORACLE-IN-IMAGE']}, 'needs_controls': ['c'], 'tasks': {'a': ['H-EXISTENCE-ONLY']}}
281
+ rows = {t['name']: t for t in store.static_tasks(report)}
282
+ self.assertEqual((rows['a']['status'], rows['b']['status'], rows['c']['status']), ('eligible', 'excluded', 'needs controls'))
283
+ self.assertEqual(rows['b']['why'], 'the reference solution is copied into the sandbox'); self.assertTrue(rows['a']['why'].startswith('flagged: '))
284
+
285
  def test_live_database_is_loaded_from_the_live_payload(self):
286
  import store
287
  payload = {'formula': {'challenges': [{'id': 'tb2-9b', 'name': 'x', 'status': 'open'}], 'collections': [{'id': 'env-1', 'title': 'Pack', 'author': 'ada', 'task_count': 3}]},