Xiangyi Li commited on
Commit
55ebf58
·
1 Parent(s): fe751d2

Dogfood visibility: organizer pipeline runs and base-model row on the dashboard, board posts on run events, BenchFlow-native task packages accepted, challenge runs serve on four GPUs

Browse files
__pycache__/challenges.cpython-312.pyc CHANGED
Binary files a/__pycache__/challenges.cpython-312.pyc and b/__pycache__/challenges.cpython-312.pyc differ
 
__pycache__/check_task.cpython-312.pyc CHANGED
Binary files a/__pycache__/check_task.cpython-312.pyc and b/__pycache__/check_task.cpython-312.pyc differ
 
__pycache__/collab.cpython-312.pyc CHANGED
Binary files a/__pycache__/collab.cpython-312.pyc and b/__pycache__/collab.cpython-312.pyc differ
 
__pycache__/test_challenges.cpython-312-pytest-7.4.4.pyc CHANGED
Binary files a/__pycache__/test_challenges.cpython-312-pytest-7.4.4.pyc and b/__pycache__/test_challenges.cpython-312-pytest-7.4.4.pyc differ
 
__pycache__/test_check_task.cpython-312-pytest-7.4.4.pyc ADDED
Binary file (3.47 kB). View file
 
challenges.py CHANGED
@@ -36,7 +36,7 @@ CHALLENGES=[{
36
  'recipe':{'id':'grpo-v1','method':'GRPO (TRL) with LoRA r32/alpha64 on the policy, OpenCode rollouts in Daytona sandboxes',
37
  'template':'fixture/challenge-tb2-9b.toml','learning_rate':1e-6,'num_generations':8,'generation_batch_size':8,'max_steps':2,'max_completion_length':32768,
38
  'rollout_attempts':2,'require_reward_variance':True,'gate_task_count':32,'harness':{'agent':'opencode','skill_mode':'no-skill','concurrency':8,'agent_timeout_sec':900,'agent_idle_timeout_sec':900},
39
- 'sandbox':'daytona','sft':False,'teacher':False,
40
  'note':'v1 caps training at 2 optimizer steps (64 OpenCode rollouts each) so baseline, gate, training and held-out evaluation fit one 8 h job at the measured rollout throughput (~20 model calls/min through one GPU). It proves the loop end to end; it is not enough training to expect a held-out change. A longer recipe needs cross-job checkpoint resume or more serving GPUs.'},
41
  'compute':{'provider':'huggingface','flavor':pipeline_jobs.FLAVOR,'timeout_seconds':8*3600,'daytona_minutes_estimate':900,'runs_per_submission_per_day':1,'concurrent_runs':1},
42
  'ingress':'Sandboxes reach the policy through this Space (/relay/<run>/v1) with a per-run key; the job connects outbound. No tunnel.',
@@ -208,11 +208,13 @@ def start_run(challenge_id:str,value:RunRequest,request:Request):
208
  if not created: return public(record)
209
  try:
210
  job=pipeline_jobs.launch(run_name=record['run_id'],config=config_path.replace('bundle/',''),bundle_rev=bundle_revision,timeout_seconds=allocation['timeout_seconds'],
211
- labels={'experiment':'posttrain-challenge','challenge':challenge_id,'run_id':record['run_id']},space_origin=auth.origin(),relay_key=secrets.token_urlsafe(48))
212
  except Exception:
213
  release(record['run_id'],'Job submission failed; reservation released.')
214
  raise HTTPException(502,'Job submission failed. The reservation was released; retry with the same request ID.') from None
215
- return public(jobs.record_job(record,job))
 
 
216
 
217
  def release(run_id,note):
218
  client=jobs.api()
@@ -306,7 +308,9 @@ def collect(challenge_id:str,run_id:str,request:Request):
306
  old=next((r for r in rows if r['run_id']==run_id),None)
307
  if old: return old
308
  rows.append(result);return result
309
- return env.replace_file(RESULTS,change,[])
 
 
310
 
311
  @router.post('/{challenge_id}/runs/{run_id}/review')
312
  def review(challenge_id:str,run_id:str,value:Review,request:Request):
@@ -319,7 +323,9 @@ def review(challenge_id:str,run_id:str,value:Review,request:Request):
319
  if all(result.get(k)==v for k,v in update.items()): return result
320
  raise HTTPException(409,'Review is immutable.')
321
  result.update(**update,reviewed_at=now());return result
322
- return env.replace_file(RESULTS,change,[])
 
 
323
 
324
  def submission_rows(challenge_id):
325
  """Every validated submission with its latest run for this challenge (state, spend bound, milestones when running)."""
@@ -338,6 +344,19 @@ def submission_rows(challenge_id):
338
  rows.append(row)
339
  return rows
340
 
 
 
 
 
 
 
 
 
 
 
 
 
 
341
  @router.get('/{challenge_id}/dashboard')
342
  def dashboard(challenge_id:str):
343
  """One call for the dashboard: challenge card, verified leaderboard, submissions with run state, pending results."""
@@ -352,7 +371,7 @@ def dashboard(challenge_id:str):
352
  card['per_run_allocation']=allocation;card['baseline']=board['baseline']
353
  card['participant_flow']=['python arena_cli.py environments submit --file environment.json','python arena_cli.py run --challenge '+row['id']+' --id ENVIRONMENT_ID --file run.json --execute','python arena_cli.py result collect --challenge '+row['id']+' --run-id RUN_ID']
354
  counts={'submissions':len(subs),'runs':sum(s['run_count'] for s in subs),'running':sum(1 for s in subs if s['state'] in ('running','queued')),'ranked':len(board['rows']),'pending_review':board['pending_count']}
355
- return {'challenge':card,'challenges':[{'id':c['id'],'name':c['name'],'status':c['status']} for c in CHALLENGES],'leaderboard':board['rows'],'explanation':board['explanation'],'submissions':subs,'counts':counts,'generated_at':now()}
356
 
357
  @router.get('/{challenge_id}/leaderboard')
358
  def leaderboard(challenge_id:str):
 
36
  'recipe':{'id':'grpo-v1','method':'GRPO (TRL) with LoRA r32/alpha64 on the policy, OpenCode rollouts in Daytona sandboxes',
37
  'template':'fixture/challenge-tb2-9b.toml','learning_rate':1e-6,'num_generations':8,'generation_batch_size':8,'max_steps':2,'max_completion_length':32768,
38
  'rollout_attempts':2,'require_reward_variance':True,'gate_task_count':32,'harness':{'agent':'opencode','skill_mode':'no-skill','concurrency':8,'agent_timeout_sec':900,'agent_idle_timeout_sec':900},
39
+ 'sandbox':'daytona','sft':False,'teacher':False,'serving':{'tensor_parallel':4,'gpus':'4-7','note':'policy server on four A100s; one GPU could not keep 88 long-context tasks inside an 8 h job'},
40
  'note':'v1 caps training at 2 optimizer steps (64 OpenCode rollouts each) so baseline, gate, training and held-out evaluation fit one 8 h job at the measured rollout throughput (~20 model calls/min through one GPU). It proves the loop end to end; it is not enough training to expect a held-out change. A longer recipe needs cross-job checkpoint resume or more serving GPUs.'},
41
  'compute':{'provider':'huggingface','flavor':pipeline_jobs.FLAVOR,'timeout_seconds':8*3600,'daytona_minutes_estimate':900,'runs_per_submission_per_day':1,'concurrent_runs':1},
42
  'ingress':'Sandboxes reach the policy through this Space (/relay/<run>/v1) with a per-run key; the job connects outbound. No tunnel.',
 
208
  if not created: return public(record)
209
  try:
210
  job=pipeline_jobs.launch(run_name=record['run_id'],config=config_path.replace('bundle/',''),bundle_rev=bundle_revision,timeout_seconds=allocation['timeout_seconds'],
211
+ labels={'experiment':'posttrain-challenge','challenge':challenge_id,'run_id':record['run_id']},space_origin=auth.origin(),relay_key=secrets.token_urlsafe(48),vllm_tp=row['recipe']['serving']['tensor_parallel'],vllm_gpus='4,5,6,7')
212
  except Exception:
213
  release(record['run_id'],'Job submission failed; reservation released.')
214
  raise HTTPException(502,'Job submission failed. The reservation was released; retry with the same request ID.') from None
215
+ receipt=public(jobs.record_job(record,job))
216
+ collab.system_post(f"Run {receipt['run_id']} started on challenge {challenge_id} for submission {source['id']} ({source.get('title','')}, {len(mirrored['tasks'])} tasks) by {user['name']}. HF job: {receipt.get('job_url')}")
217
+ return receipt
218
 
219
  def release(run_id,note):
220
  client=jobs.api()
 
308
  old=next((r for r in rows if r['run_id']==run_id),None)
309
  if old: return old
310
  rows.append(result);return result
311
+ stored=env.replace_file(RESULTS,change,[])
312
+ collab.system_post(f"Evidence collected for run {run_id} on {challenge_id}: baseline {100*baseline['pass_rate']:.1f}% -> after {100*final['pass_rate']:.1f}% (delta {100*delta:+.1f} pp, GRPO ran: {grpo}). Awaiting organizer review.")
313
+ return stored
314
 
315
  @router.post('/{challenge_id}/runs/{run_id}/review')
316
  def review(challenge_id:str,run_id:str,value:Review,request:Request):
 
323
  if all(result.get(k)==v for k,v in update.items()): return result
324
  raise HTTPException(409,'Review is immutable.')
325
  result.update(**update,reviewed_at=now());return result
326
+ reviewed=env.replace_file(RESULTS,change,[])
327
+ collab.system_post(f"Run {run_id} on {challenge_id} reviewed by {reviewer['name']}: {'valid, now ranked' if value.accepted else 'invalid, not ranked'}. {value.note[:300]}")
328
+ return reviewed
329
 
330
  def submission_rows(challenge_id):
331
  """Every validated submission with its latest run for this challenge (state, spend bound, milestones when running)."""
 
344
  rows.append(row)
345
  return rows
346
 
347
+ def organizer_runs(limit=10):
348
+ """Organizer pipeline runs (phase-2 loop completion) from results/phase2-runs.json, with live job stage for open ones."""
349
+ try: rows=json.loads(Path(hf_hub_download(RUNS,'results/phase2-runs.json',repo_type='dataset',token=os.environ.get('HF_TOKEN'),force_download=True)).read_text())
350
+ except Exception: return []
351
+ out=[]
352
+ for r in rows[-limit:][::-1]:
353
+ item={k:r.get(k) for k in ('run_name','job_id','job_url','status','created_at','purpose','note')}
354
+ if item['status'] not in jobs.TERMINAL:
355
+ try: item['status']=str(jobs.api().inspect_job(job_id=item['job_id'],namespace='benchflow').status.stage)
356
+ except Exception: pass
357
+ out.append(item)
358
+ return out
359
+
360
  @router.get('/{challenge_id}/dashboard')
361
  def dashboard(challenge_id:str):
362
  """One call for the dashboard: challenge card, verified leaderboard, submissions with run state, pending results."""
 
371
  card['per_run_allocation']=allocation;card['baseline']=board['baseline']
372
  card['participant_flow']=['python arena_cli.py environments submit --file environment.json','python arena_cli.py run --challenge '+row['id']+' --id ENVIRONMENT_ID --file run.json --execute','python arena_cli.py result collect --challenge '+row['id']+' --run-id RUN_ID']
373
  counts={'submissions':len(subs),'runs':sum(s['run_count'] for s in subs),'running':sum(1 for s in subs if s['state'] in ('running','queued')),'ranked':len(board['rows']),'pending_review':board['pending_count']}
374
+ return {'challenge':card,'challenges':[{'id':c['id'],'name':c['name'],'status':c['status']} for c in CHALLENGES],'leaderboard':board['rows'],'explanation':board['explanation'],'submissions':subs,'counts':counts,'organizer_runs':organizer_runs(),'generated_at':now()}
375
 
376
  @router.get('/{challenge_id}/leaderboard')
377
  def leaderboard(challenge_id:str):
check_task.py CHANGED
@@ -33,6 +33,7 @@ REQUIRED_FRONTMATTER = (
33
  "verifier",
34
  "environment",
35
  )
 
36
  REQUIRED_METADATA = (
37
  "author_name",
38
  "author_email",
@@ -74,6 +75,27 @@ def parse_metadata_keys(block: str) -> set[str]:
74
  return keys
75
 
76
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
77
  def check_task(task_dir: Path) -> list[str]:
78
  issues: list[str] = []
79
 
@@ -90,6 +112,18 @@ def check_task(task_dir: Path) -> list[str]:
90
  frontmatter = m.group(1)
91
  body = text[m.end():]
92
  top_keys = parse_yaml_keys(frontmatter)
 
 
 
 
 
 
 
 
 
 
 
 
93
  for required in REQUIRED_FRONTMATTER:
94
  if required not in top_keys:
95
  issues.append(f"task.md frontmatter missing: {required}")
@@ -100,23 +134,7 @@ def check_task(task_dir: Path) -> list[str]:
100
  if "## prompt" not in body:
101
  issues.append("task.md body must contain a '## prompt' section")
102
 
103
- # environment/ ----------------------------------------------------------
104
- dockerfile = task_dir / "environment" / "Dockerfile"
105
- if not dockerfile.exists():
106
- issues.append("Missing required file: environment/Dockerfile")
107
- else:
108
- # Skip comment lines and blanks; the first executable instruction
109
- # must be FROM.
110
- first_instr = next(
111
- (
112
- line
113
- for line in dockerfile.read_text(encoding="utf-8").splitlines()
114
- if line.strip() and not line.lstrip().startswith("#")
115
- ),
116
- "",
117
- )
118
- if not first_instr.upper().startswith("FROM "):
119
- issues.append("environment/Dockerfile first instruction must be FROM")
120
 
121
  # verifier/ -------------------------------------------------------------
122
  for f in ("verifier/test.sh", "verifier/test_outputs.py", "verifier/verifier.md"):
 
33
  "verifier",
34
  "environment",
35
  )
36
+ BENCHFLOW_FRONTMATTER = ("task", "metadata", "agent", "verifier", "sandbox")
37
  REQUIRED_METADATA = (
38
  "author_name",
39
  "author_email",
 
75
  return keys
76
 
77
 
78
+ def _check_environment(task_dir: Path) -> list[str]:
79
+ issues: list[str] = []
80
+ dockerfile = task_dir / "environment" / "Dockerfile"
81
+ if not dockerfile.exists():
82
+ issues.append("Missing required file: environment/Dockerfile")
83
+ else:
84
+ # Skip comment lines and blanks; the first executable instruction
85
+ # must be FROM.
86
+ first_instr = next(
87
+ (
88
+ line
89
+ for line in dockerfile.read_text(encoding="utf-8").splitlines()
90
+ if line.strip() and not line.lstrip().startswith("#")
91
+ ),
92
+ "",
93
+ )
94
+ if not first_instr.upper().startswith("FROM "):
95
+ issues.append("environment/Dockerfile first instruction must be FROM")
96
+ return issues
97
+
98
+
99
  def check_task(task_dir: Path) -> list[str]:
100
  issues: list[str] = []
101
 
 
112
  frontmatter = m.group(1)
113
  body = text[m.end():]
114
  top_keys = parse_yaml_keys(frontmatter)
115
+ if "schema_version" in top_keys and "task" in top_keys:
116
+ # BenchFlow-native task (schema 1.1): `bench tasks check` is the authority; the arena only
117
+ # requires the executable pieces the pipeline snapshots and runs.
118
+ for required in BENCHFLOW_FRONTMATTER:
119
+ if required not in top_keys:
120
+ issues.append(f"task.md frontmatter missing: {required}")
121
+ if "## prompt" not in body:
122
+ issues.append("task.md body must contain a '## prompt' section")
123
+ issues.extend(_check_environment(task_dir))
124
+ if not (task_dir / "verifier" / "test.sh").exists():
125
+ issues.append("Missing required file: verifier/test.sh")
126
+ return issues
127
  for required in REQUIRED_FRONTMATTER:
128
  if required not in top_keys:
129
  issues.append(f"task.md frontmatter missing: {required}")
 
134
  if "## prompt" not in body:
135
  issues.append("task.md body must contain a '## prompt' section")
136
 
137
+ issues.extend(_check_environment(task_dir))
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
138
 
139
  # verifier/ -------------------------------------------------------------
140
  for f in ("verifier/test.sh", "verifier/test_outputs.py", "verifier/verifier.md"):
collab.py CHANGED
@@ -122,6 +122,27 @@ def post_message(value: Message, request: Request):
122
  result = env.replace_file(MESSAGES, change, [])
123
  return {'item':message_item(result),'mentions_delivered':[],'auto_subscribed':False}
124
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
125
  class ModelRef(Strict):
126
  repo_id: str = Field(pattern=env.SAFE_REPO, max_length=150)
127
  revision: str = Field(pattern=r'^[0-9a-f]{40}$')
 
122
  result = env.replace_file(MESSAGES, change, [])
123
  return {'item':message_item(result),'mentions_delivered':[],'auto_subscribed':False}
124
 
125
+ SYSTEM_AGENT='arena-system'
126
+
127
+ def system_post(body, refs=None):
128
+ """Board post from the arena itself on run lifecycle events. Never raises: the main flow must not depend on the board."""
129
+ try:
130
+ def ensure_agent(rows):
131
+ if not any(a['agent_id']==SYSTEM_AGENT for a in rows):
132
+ rows.append({'agent_id':SYSTEM_AGENT,'description':'Arena system notices: run started, evidence collected, result reviewed.','model':'','harness':'posttrain-arena','owner':'benchflow','created_at':now()})
133
+ return rows
134
+ env.replace_file(AGENTS, ensure_agent, [])
135
+ payload={'body':body[:6000],'refs':list(refs or []),'broadcast':False}
136
+ row={**payload,'request_id':digest([SYSTEM_AGENT,body,int(time.time()//60)]),'owner':'benchflow','agent_id':SYSTEM_AGENT,'created_at':now(),
137
+ 'filename':datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')+'_'+SYSTEM_AGENT+'_'+uuid.uuid4().hex[:6]+'.md','type':'agent','payload_hash':digest(payload)}
138
+ def change(rows):
139
+ if any(m.get('request_id')==row['request_id'] for m in rows): return rows
140
+ rows.append(row); return rows
141
+ env.replace_file(MESSAGES, change, [])
142
+ return row
143
+ except Exception:
144
+ return None
145
+
146
  class ModelRef(Strict):
147
  repo_id: str = Field(pattern=env.SAFE_REPO, max_length=150)
148
  revision: str = Field(pattern=r'^[0-9a-f]{40}$')
index.html CHANGED
@@ -1249,6 +1249,14 @@
1249
  </table>
1250
  </div>
1251
 
 
 
 
 
 
 
 
 
1252
  <details class="legacy-block" id="legacyBlock">
1253
  <summary class="section-title">Practice experiments (seen-task, legacy)<span class="hint">single-task oracle-overfit runs; not held-out</span></summary>
1254
  <div style="margin-bottom:16px"><label for="experimentGroup">Comparison group</label> <select class="btn" id="experimentGroup" aria-describedby="groupDescription"><option value="">Loading groups…</option></select><p class="subtitle" id="groupDescription">Results are ranked only within the same model, evaluation and protocol.</p></div>
@@ -4375,7 +4383,13 @@ function renderChallengeBoard(page) {
4375
  const rows = page.leaderboard || [];
4376
  document.getElementById('clbStatus').textContent = rows.length ? `${rows.length} ranked` : (page.counts.pending_review ? `${page.counts.pending_review} awaiting review` : 'no verified results yet');
4377
  document.getElementById('clbExplanation').textContent = page.explanation || '';
4378
- if (!rows.length) { body.innerHTML = `<tr><td colspan="9" class="subtitle">No organizer-verified result yet. Submissions and their runs are listed below.</td></tr>`; return; }
 
 
 
 
 
 
4379
  for (const r of rows) {
4380
  const d = Number(r.delta_pp); const cls = d > 0 ? 'delta-pos' : '';
4381
  body.insertAdjacentHTML('beforeend', `<tr>
@@ -4389,6 +4403,18 @@ function renderChallengeBoard(page) {
4389
  <td><span class="state-badge published">${escapeHtml(r.verification)}</span></td></tr>`);
4390
  }
4391
  }
 
 
 
 
 
 
 
 
 
 
 
 
4392
  function renderSubmissions(page) {
4393
  const body = document.getElementById('subsBody'); body.innerHTML = '';
4394
  const subs = page.submissions || [];
@@ -4446,7 +4472,7 @@ async function refreshChallenge() {
4446
  const r = await fetchWithTimeout('/api/challenges/' + encodeURIComponent(selectedChallenge) + '/dashboard');
4447
  if (!r.ok) throw new Error('HTTP ' + r.status);
4448
  const page = await r.json();
4449
- renderChallengeCard(page); renderChallengeBoard(page); renderChallengeChart(page.leaderboard || []); renderSubmissions(page);
4450
  } catch (err) {
4451
  document.getElementById('clbStatus').textContent = 'unavailable';
4452
  document.getElementById('subsStatus').textContent = 'unavailable';
 
1249
  </table>
1250
  </div>
1251
 
1252
+ <div class="section-title">Organizer pipeline runs<span class="hint" id="orgRunsStatus">phase-2 loop completion on the same pipeline</span></div>
1253
+ <div style="overflow-x:auto">
1254
+ <table class="lb-table" style="min-width:700px">
1255
+ <thead><tr><th style="width:260px">Run</th><th style="width:110px">Status</th><th>Note</th><th style="width:70px">Job</th></tr></thead>
1256
+ <tbody id="orgRunsBody"></tbody>
1257
+ </table>
1258
+ </div>
1259
+
1260
  <details class="legacy-block" id="legacyBlock">
1261
  <summary class="section-title">Practice experiments (seen-task, legacy)<span class="hint">single-task oracle-overfit runs; not held-out</span></summary>
1262
  <div style="margin-bottom:16px"><label for="experimentGroup">Comparison group</label> <select class="btn" id="experimentGroup" aria-describedby="groupDescription"><option value="">Loading groups…</option></select><p class="subtitle" id="groupDescription">Results are ranked only within the same model, evaluation and protocol.</p></div>
 
4383
  const rows = page.leaderboard || [];
4384
  document.getElementById('clbStatus').textContent = rows.length ? `${rows.length} ranked` : (page.counts.pending_review ? `${page.counts.pending_review} awaiting review` : 'no verified results yet');
4385
  document.getElementById('clbExplanation').textContent = page.explanation || '';
4386
+ const base = page.challenge && page.challenge.baseline;
4387
+ if (base && base.pass_rate != null) {
4388
+ body.insertAdjacentHTML('beforeend', `<tr class="lb-invalid-sep"><td>—</td><td><strong>${escapeHtml(page.challenge.base_model.repo_id)}</strong><div class="subtitle">base model, no post-training · organizer-measured over ${base.trials} trial${base.trials === 1 ? '' : 's'}</div></td>
4389
+ <td class="num">${pct(base.pass_rate)}</td><td class="num">—</td><td class="num">0.0 ± ${base.stderr != null ? (100 * base.stderr).toFixed(1) : '—'}</td><td>none</td><td>—</td>
4390
+ <td>${base.source ? `<a href="${escapeHtml(base.source)}" target="_blank" rel="noopener noreferrer">aggregate</a>` : '—'}</td><td><span class="state-badge">reference</span></td></tr>`);
4391
+ }
4392
+ if (!rows.length) { body.insertAdjacentHTML('beforeend', `<tr><td colspan="9" class="subtitle">No organizer-verified submission result yet. Submissions and their runs are listed below.</td></tr>`); return; }
4393
  for (const r of rows) {
4394
  const d = Number(r.delta_pp); const cls = d > 0 ? 'delta-pos' : '';
4395
  body.insertAdjacentHTML('beforeend', `<tr>
 
4403
  <td><span class="state-badge published">${escapeHtml(r.verification)}</span></td></tr>`);
4404
  }
4405
  }
4406
+ function renderOrganizerRuns(page) {
4407
+ const body = document.getElementById('orgRunsBody'); if (!body) return;
4408
+ const rows = page.organizer_runs || []; body.innerHTML = '';
4409
+ document.getElementById('orgRunsStatus').textContent = rows.length ? `${rows.length} most recent · ${rows.filter(r => r.status === 'RUNNING').length} running` : 'none recorded';
4410
+ if (!rows.length) { body.innerHTML = `<tr><td colspan="4" class="subtitle">No organizer runs recorded.</td></tr>`; return; }
4411
+ for (const r of rows) {
4412
+ body.insertAdjacentHTML('beforeend', `<tr><td><strong>${escapeHtml(r.run_name || '')}</strong><div class="subtitle">${escapeHtml(String(r.created_at || '').slice(0, 16))}</div></td>
4413
+ <td><span class="state-badge ${r.status === 'RUNNING' ? 'running' : (r.status === 'COMPLETED' ? 'published' : 'failed')}">${escapeHtml(String(r.status || '').toLowerCase())}</span></td>
4414
+ <td class="subtitle">${escapeHtml(r.note || r.purpose || '')}</td>
4415
+ <td>${r.job_url ? `<a href="${escapeHtml(r.job_url)}" target="_blank" rel="noopener noreferrer">job</a>` : '—'}</td></tr>`);
4416
+ }
4417
+ }
4418
  function renderSubmissions(page) {
4419
  const body = document.getElementById('subsBody'); body.innerHTML = '';
4420
  const subs = page.submissions || [];
 
4472
  const r = await fetchWithTimeout('/api/challenges/' + encodeURIComponent(selectedChallenge) + '/dashboard');
4473
  if (!r.ok) throw new Error('HTTP ' + r.status);
4474
  const page = await r.json();
4475
+ renderChallengeCard(page); renderChallengeBoard(page); renderChallengeChart(page.leaderboard || []); renderSubmissions(page); renderOrganizerRuns(page);
4476
  } catch (err) {
4477
  document.getElementById('clbStatus').textContent = 'unavailable';
4478
  document.getElementById('subsStatus').textContent = 'unavailable';
test_challenges.py CHANGED
@@ -17,7 +17,7 @@ class ChallengeTests(unittest.TestCase):
17
  self.ledger={'prior_allowance_usd':50,'runs':[]};self.launches=[];self.stage='RUNNING';self.registry={challenges.RESULTS:[]}
18
  self.source={'id':'env-abc123abc123','challenge_id':'skillsbench','repo_type':'github','repo_id':'org/pack','revision':REV,'entry_path':'sub','title':'Pack','author':'owner','status':'Validated','source_url':'https://github.com/org/pack/tree/'+REV+'/sub'}
19
  self.manifest={'environment_id':self.source['id'],'revision':REV,'tasks':['alpha','beta'],'file_count':12,'bytes':1000,'mirror_revision':MIRROR}
20
- self.summary=None;self.scores={}
21
  api=SimpleNamespace(token='isolated',repo_info=lambda *a,**k:SimpleNamespace(sha=HEAD),inspect_job=lambda **k:SimpleNamespace(status=SimpleNamespace(stage=self.stage)),fetch_job_logs=lambda **k:['vllm up','bridge up','tunnel reachable'])
22
  patches=[patch.dict(os.environ,{'HF_TOKEN':'isolated','DAYTONA_API_KEY':'isolated'}),patch.object(jobs,'api',return_value=api),
23
  patch.object(jobs,'read',side_effect=lambda head=None:copy.deepcopy(self.ledger)),patch.object(jobs,'write',side_effect=self.write_ledger),
@@ -26,6 +26,7 @@ class ChallengeTests(unittest.TestCase):
26
  patch.object(challenges,'mirror',side_effect=lambda row:copy.deepcopy(self.manifest)),patch.object(challenges,'write_bundle',side_effect=lambda row,run_id,m:(BUNDLE,f'bundle/challenge-runs/{run_id}/config.toml')),
27
  patch.object(challenges,'hub',return_value=api),patch.object(challenges,'report',side_effect=lambda run_id:copy.deepcopy(self.summary)),patch.object(challenges,'stage_scores',side_effect=lambda run_id,stage,head:copy.deepcopy(self.scores[stage])),
28
  patch.object(challenges,'baseline_reference',return_value={'pass_rate':0.017,'stderr':0.006,'trials':2}),patch.object(pipeline_jobs,'launch',side_effect=self.launch_job),
 
29
  patch.object(socket,'create_connection',side_effect=AssertionError('Unexpected network in isolated challenge test'))]
30
  for p in patches:p.start();self.addCleanup(p.stop)
31
  app=FastAPI();app.include_router(challenges.router);self.client=TestClient(app,raise_server_exceptions=False);auth._sessions.clear()
@@ -73,6 +74,7 @@ class ChallengeTests(unittest.TestCase):
73
  self.assertEqual(len(self.launches),1);launch=self.launches[0]
74
  self.assertEqual(launch['run_name'],record['run_id']);self.assertEqual(launch['config'],'challenge-runs/'+record['run_id']+'/config.toml');self.assertEqual(launch['bundle_rev'],BUNDLE);self.assertEqual(launch['timeout_seconds'],8*3600)
75
  self.assertTrue(launch['space_origin'].startswith('https://'));self.assertGreaterEqual(len(launch['relay_key']),48);self.assertNotIn('relay_key',json.dumps(self.ledger))
 
76
  ledger=self.ledger['runs'][0];self.assertEqual(ledger['kind'],'challenge-run');self.assertEqual(ledger['job_id'],'1'*24)
77
  self.assertEqual(self.start(),record);self.assertEqual(len(self.launches),1)
78
  self.post('/api/challenges/tb2-9b/runs',{'request_id':'stable-run-request','environment_id':'env-other'},expected=409)
@@ -140,5 +142,6 @@ class ChallengeTests(unittest.TestCase):
140
  self.post('/api/challenges/tb2-9b/runs/'+record['run_id']+'/review',{'accepted':True,'note':'Evidence reviewed: job log, score.json and per-task results agree.'},user=EDITOR)
141
  page=self.client.get('/api/challenges/tb2-9b/dashboard').json()
142
  self.assertEqual(page['submissions'][0]['state'],'published');self.assertEqual(page['leaderboard'][0]['rank'],1);self.assertEqual(page['counts']['ranked'],1)
 
143
 
144
  if __name__=='__main__':unittest.main()
 
17
  self.ledger={'prior_allowance_usd':50,'runs':[]};self.launches=[];self.stage='RUNNING';self.registry={challenges.RESULTS:[]}
18
  self.source={'id':'env-abc123abc123','challenge_id':'skillsbench','repo_type':'github','repo_id':'org/pack','revision':REV,'entry_path':'sub','title':'Pack','author':'owner','status':'Validated','source_url':'https://github.com/org/pack/tree/'+REV+'/sub'}
19
  self.manifest={'environment_id':self.source['id'],'revision':REV,'tasks':['alpha','beta'],'file_count':12,'bytes':1000,'mirror_revision':MIRROR}
20
+ self.summary=None;self.scores={};self.posts=[]
21
  api=SimpleNamespace(token='isolated',repo_info=lambda *a,**k:SimpleNamespace(sha=HEAD),inspect_job=lambda **k:SimpleNamespace(status=SimpleNamespace(stage=self.stage)),fetch_job_logs=lambda **k:['vllm up','bridge up','tunnel reachable'])
22
  patches=[patch.dict(os.environ,{'HF_TOKEN':'isolated','DAYTONA_API_KEY':'isolated'}),patch.object(jobs,'api',return_value=api),
23
  patch.object(jobs,'read',side_effect=lambda head=None:copy.deepcopy(self.ledger)),patch.object(jobs,'write',side_effect=self.write_ledger),
 
26
  patch.object(challenges,'mirror',side_effect=lambda row:copy.deepcopy(self.manifest)),patch.object(challenges,'write_bundle',side_effect=lambda row,run_id,m:(BUNDLE,f'bundle/challenge-runs/{run_id}/config.toml')),
27
  patch.object(challenges,'hub',return_value=api),patch.object(challenges,'report',side_effect=lambda run_id:copy.deepcopy(self.summary)),patch.object(challenges,'stage_scores',side_effect=lambda run_id,stage,head:copy.deepcopy(self.scores[stage])),
28
  patch.object(challenges,'baseline_reference',return_value={'pass_rate':0.017,'stderr':0.006,'trials':2}),patch.object(pipeline_jobs,'launch',side_effect=self.launch_job),
29
+ patch.object(challenges,'organizer_runs',return_value=[{'run_name':'phase2-r21','job_id':'2'*24,'status':'RUNNING','note':'in flight'}]),patch.object(challenges.collab,'system_post',side_effect=lambda body,refs=None:self.posts.append(body)),
30
  patch.object(socket,'create_connection',side_effect=AssertionError('Unexpected network in isolated challenge test'))]
31
  for p in patches:p.start();self.addCleanup(p.stop)
32
  app=FastAPI();app.include_router(challenges.router);self.client=TestClient(app,raise_server_exceptions=False);auth._sessions.clear()
 
74
  self.assertEqual(len(self.launches),1);launch=self.launches[0]
75
  self.assertEqual(launch['run_name'],record['run_id']);self.assertEqual(launch['config'],'challenge-runs/'+record['run_id']+'/config.toml');self.assertEqual(launch['bundle_rev'],BUNDLE);self.assertEqual(launch['timeout_seconds'],8*3600)
76
  self.assertTrue(launch['space_origin'].startswith('https://'));self.assertGreaterEqual(len(launch['relay_key']),48);self.assertNotIn('relay_key',json.dumps(self.ledger))
77
+ self.assertEqual((launch['vllm_tp'],launch['vllm_gpus']),(4,'4,5,6,7'));self.assertEqual(len(self.posts),1);self.assertIn('started on challenge tb2-9b',self.posts[0])
78
  ledger=self.ledger['runs'][0];self.assertEqual(ledger['kind'],'challenge-run');self.assertEqual(ledger['job_id'],'1'*24)
79
  self.assertEqual(self.start(),record);self.assertEqual(len(self.launches),1)
80
  self.post('/api/challenges/tb2-9b/runs',{'request_id':'stable-run-request','environment_id':'env-other'},expected=409)
 
142
  self.post('/api/challenges/tb2-9b/runs/'+record['run_id']+'/review',{'accepted':True,'note':'Evidence reviewed: job log, score.json and per-task results agree.'},user=EDITOR)
143
  page=self.client.get('/api/challenges/tb2-9b/dashboard').json()
144
  self.assertEqual(page['submissions'][0]['state'],'published');self.assertEqual(page['leaderboard'][0]['rank'],1);self.assertEqual(page['counts']['ranked'],1)
145
+ self.assertEqual(page['organizer_runs'][0]['run_name'],'phase2-r21');self.assertEqual(len(self.posts),3);self.assertIn('Evidence collected',self.posts[1]);self.assertIn('reviewed by editor',self.posts[2])
146
 
147
  if __name__=='__main__':unittest.main()
test_check_task.py ADDED
@@ -0,0 +1,45 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """The structural checker accepts BenchFlow-native task packages (schema 1.1) as well as the starter-kit layout."""
2
+ import tempfile, unittest
3
+ from pathlib import Path
4
+ from check_task import check_task
5
+
6
+ BENCHFLOW_TASK = """---
7
+ schema_version: '1.1'
8
+ task:
9
+ name: tmax/task_x
10
+ metadata:
11
+ source: legacy
12
+ agent:
13
+ timeout_sec: 600.0
14
+ verifier:
15
+ timeout_sec: 120.0
16
+ sandbox:
17
+ cpus: 1
18
+ ---
19
+
20
+ ## prompt
21
+
22
+ Do the thing.
23
+ """
24
+
25
+ class CheckTaskTests(unittest.TestCase):
26
+ def make(self, task_md, files):
27
+ d = Path(tempfile.mkdtemp()) / 'task'; d.mkdir()
28
+ (d / 'task.md').write_text(task_md)
29
+ for rel, body in files.items():
30
+ (d / rel).parent.mkdir(parents=True, exist_ok=True); (d / rel).write_text(body)
31
+ return d
32
+ def test_benchflow_native_task_needs_only_dockerfile_and_test_sh(self):
33
+ d = self.make(BENCHFLOW_TASK, {'environment/Dockerfile': 'FROM python:3.12\n', 'verifier/test.sh': '#!/bin/bash\n'})
34
+ self.assertEqual(check_task(d), [])
35
+ d2 = self.make(BENCHFLOW_TASK, {'environment/Dockerfile': 'FROM python:3.12\n'})
36
+ self.assertIn('Missing required file: verifier/test.sh', check_task(d2))
37
+ d3 = self.make(BENCHFLOW_TASK.replace('sandbox:\n cpus: 1\n', ''), {'environment/Dockerfile': 'FROM x\n', 'verifier/test.sh': ''})
38
+ self.assertIn('task.md frontmatter missing: sandbox', check_task(d3))
39
+ def test_starter_kit_layout_still_requires_oracle_and_rubrics(self):
40
+ kit = """---\nversion: \"1.0\"\nmetadata:\n author_name: a\n author_email: a@b\n category: c\nagent:\n timeout_sec: 1\nverifier:\n timeout_sec: 1\nenvironment:\n cpus: 1\n---\n\n## prompt\n"""
41
+ d = self.make(kit, {'environment/Dockerfile': 'FROM x\n', 'verifier/test.sh': '', 'verifier/test_outputs.py': '', 'verifier/verifier.md': ''})
42
+ issues = check_task(d)
43
+ self.assertIn('Missing required directory: verifier/rubrics/', issues); self.assertIn('Missing required file: oracle/solve.sh', issues)
44
+
45
+ if __name__ == '__main__': unittest.main()