Spaces:
Running
Running
Xiangyi Li commited on
Commit ·
55ebf58
1
Parent(s): fe751d2
Dogfood visibility: organizer pipeline runs and base-model row on the dashboard, board posts on run events, BenchFlow-native task packages accepted, challenge runs serve on four GPUs
Browse files- __pycache__/challenges.cpython-312.pyc +0 -0
- __pycache__/check_task.cpython-312.pyc +0 -0
- __pycache__/collab.cpython-312.pyc +0 -0
- __pycache__/test_challenges.cpython-312-pytest-7.4.4.pyc +0 -0
- __pycache__/test_check_task.cpython-312-pytest-7.4.4.pyc +0 -0
- challenges.py +25 -6
- check_task.py +35 -17
- collab.py +21 -0
- index.html +28 -2
- test_challenges.py +4 -1
- test_check_task.py +45 -0
__pycache__/challenges.cpython-312.pyc
CHANGED
|
Binary files a/__pycache__/challenges.cpython-312.pyc and b/__pycache__/challenges.cpython-312.pyc differ
|
|
|
__pycache__/check_task.cpython-312.pyc
CHANGED
|
Binary files a/__pycache__/check_task.cpython-312.pyc and b/__pycache__/check_task.cpython-312.pyc differ
|
|
|
__pycache__/collab.cpython-312.pyc
CHANGED
|
Binary files a/__pycache__/collab.cpython-312.pyc and b/__pycache__/collab.cpython-312.pyc differ
|
|
|
__pycache__/test_challenges.cpython-312-pytest-7.4.4.pyc
CHANGED
|
Binary files a/__pycache__/test_challenges.cpython-312-pytest-7.4.4.pyc and b/__pycache__/test_challenges.cpython-312-pytest-7.4.4.pyc differ
|
|
|
__pycache__/test_check_task.cpython-312-pytest-7.4.4.pyc
ADDED
|
Binary file (3.47 kB). View file
|
|
|
challenges.py
CHANGED
|
@@ -36,7 +36,7 @@ CHALLENGES=[{
|
|
| 36 |
'recipe':{'id':'grpo-v1','method':'GRPO (TRL) with LoRA r32/alpha64 on the policy, OpenCode rollouts in Daytona sandboxes',
|
| 37 |
'template':'fixture/challenge-tb2-9b.toml','learning_rate':1e-6,'num_generations':8,'generation_batch_size':8,'max_steps':2,'max_completion_length':32768,
|
| 38 |
'rollout_attempts':2,'require_reward_variance':True,'gate_task_count':32,'harness':{'agent':'opencode','skill_mode':'no-skill','concurrency':8,'agent_timeout_sec':900,'agent_idle_timeout_sec':900},
|
| 39 |
-
'sandbox':'daytona','sft':False,'teacher':False,
|
| 40 |
'note':'v1 caps training at 2 optimizer steps (64 OpenCode rollouts each) so baseline, gate, training and held-out evaluation fit one 8 h job at the measured rollout throughput (~20 model calls/min through one GPU). It proves the loop end to end; it is not enough training to expect a held-out change. A longer recipe needs cross-job checkpoint resume or more serving GPUs.'},
|
| 41 |
'compute':{'provider':'huggingface','flavor':pipeline_jobs.FLAVOR,'timeout_seconds':8*3600,'daytona_minutes_estimate':900,'runs_per_submission_per_day':1,'concurrent_runs':1},
|
| 42 |
'ingress':'Sandboxes reach the policy through this Space (/relay/<run>/v1) with a per-run key; the job connects outbound. No tunnel.',
|
|
@@ -208,11 +208,13 @@ def start_run(challenge_id:str,value:RunRequest,request:Request):
|
|
| 208 |
if not created: return public(record)
|
| 209 |
try:
|
| 210 |
job=pipeline_jobs.launch(run_name=record['run_id'],config=config_path.replace('bundle/',''),bundle_rev=bundle_revision,timeout_seconds=allocation['timeout_seconds'],
|
| 211 |
-
labels={'experiment':'posttrain-challenge','challenge':challenge_id,'run_id':record['run_id']},space_origin=auth.origin(),relay_key=secrets.token_urlsafe(48))
|
| 212 |
except Exception:
|
| 213 |
release(record['run_id'],'Job submission failed; reservation released.')
|
| 214 |
raise HTTPException(502,'Job submission failed. The reservation was released; retry with the same request ID.') from None
|
| 215 |
-
|
|
|
|
|
|
|
| 216 |
|
| 217 |
def release(run_id,note):
|
| 218 |
client=jobs.api()
|
|
@@ -306,7 +308,9 @@ def collect(challenge_id:str,run_id:str,request:Request):
|
|
| 306 |
old=next((r for r in rows if r['run_id']==run_id),None)
|
| 307 |
if old: return old
|
| 308 |
rows.append(result);return result
|
| 309 |
-
|
|
|
|
|
|
|
| 310 |
|
| 311 |
@router.post('/{challenge_id}/runs/{run_id}/review')
|
| 312 |
def review(challenge_id:str,run_id:str,value:Review,request:Request):
|
|
@@ -319,7 +323,9 @@ def review(challenge_id:str,run_id:str,value:Review,request:Request):
|
|
| 319 |
if all(result.get(k)==v for k,v in update.items()): return result
|
| 320 |
raise HTTPException(409,'Review is immutable.')
|
| 321 |
result.update(**update,reviewed_at=now());return result
|
| 322 |
-
|
|
|
|
|
|
|
| 323 |
|
| 324 |
def submission_rows(challenge_id):
|
| 325 |
"""Every validated submission with its latest run for this challenge (state, spend bound, milestones when running)."""
|
|
@@ -338,6 +344,19 @@ def submission_rows(challenge_id):
|
|
| 338 |
rows.append(row)
|
| 339 |
return rows
|
| 340 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 341 |
@router.get('/{challenge_id}/dashboard')
|
| 342 |
def dashboard(challenge_id:str):
|
| 343 |
"""One call for the dashboard: challenge card, verified leaderboard, submissions with run state, pending results."""
|
|
@@ -352,7 +371,7 @@ def dashboard(challenge_id:str):
|
|
| 352 |
card['per_run_allocation']=allocation;card['baseline']=board['baseline']
|
| 353 |
card['participant_flow']=['python arena_cli.py environments submit --file environment.json','python arena_cli.py run --challenge '+row['id']+' --id ENVIRONMENT_ID --file run.json --execute','python arena_cli.py result collect --challenge '+row['id']+' --run-id RUN_ID']
|
| 354 |
counts={'submissions':len(subs),'runs':sum(s['run_count'] for s in subs),'running':sum(1 for s in subs if s['state'] in ('running','queued')),'ranked':len(board['rows']),'pending_review':board['pending_count']}
|
| 355 |
-
return {'challenge':card,'challenges':[{'id':c['id'],'name':c['name'],'status':c['status']} for c in CHALLENGES],'leaderboard':board['rows'],'explanation':board['explanation'],'submissions':subs,'counts':counts,'generated_at':now()}
|
| 356 |
|
| 357 |
@router.get('/{challenge_id}/leaderboard')
|
| 358 |
def leaderboard(challenge_id:str):
|
|
|
|
| 36 |
'recipe':{'id':'grpo-v1','method':'GRPO (TRL) with LoRA r32/alpha64 on the policy, OpenCode rollouts in Daytona sandboxes',
|
| 37 |
'template':'fixture/challenge-tb2-9b.toml','learning_rate':1e-6,'num_generations':8,'generation_batch_size':8,'max_steps':2,'max_completion_length':32768,
|
| 38 |
'rollout_attempts':2,'require_reward_variance':True,'gate_task_count':32,'harness':{'agent':'opencode','skill_mode':'no-skill','concurrency':8,'agent_timeout_sec':900,'agent_idle_timeout_sec':900},
|
| 39 |
+
'sandbox':'daytona','sft':False,'teacher':False,'serving':{'tensor_parallel':4,'gpus':'4-7','note':'policy server on four A100s; one GPU could not keep 88 long-context tasks inside an 8 h job'},
|
| 40 |
'note':'v1 caps training at 2 optimizer steps (64 OpenCode rollouts each) so baseline, gate, training and held-out evaluation fit one 8 h job at the measured rollout throughput (~20 model calls/min through one GPU). It proves the loop end to end; it is not enough training to expect a held-out change. A longer recipe needs cross-job checkpoint resume or more serving GPUs.'},
|
| 41 |
'compute':{'provider':'huggingface','flavor':pipeline_jobs.FLAVOR,'timeout_seconds':8*3600,'daytona_minutes_estimate':900,'runs_per_submission_per_day':1,'concurrent_runs':1},
|
| 42 |
'ingress':'Sandboxes reach the policy through this Space (/relay/<run>/v1) with a per-run key; the job connects outbound. No tunnel.',
|
|
|
|
| 208 |
if not created: return public(record)
|
| 209 |
try:
|
| 210 |
job=pipeline_jobs.launch(run_name=record['run_id'],config=config_path.replace('bundle/',''),bundle_rev=bundle_revision,timeout_seconds=allocation['timeout_seconds'],
|
| 211 |
+
labels={'experiment':'posttrain-challenge','challenge':challenge_id,'run_id':record['run_id']},space_origin=auth.origin(),relay_key=secrets.token_urlsafe(48),vllm_tp=row['recipe']['serving']['tensor_parallel'],vllm_gpus='4,5,6,7')
|
| 212 |
except Exception:
|
| 213 |
release(record['run_id'],'Job submission failed; reservation released.')
|
| 214 |
raise HTTPException(502,'Job submission failed. The reservation was released; retry with the same request ID.') from None
|
| 215 |
+
receipt=public(jobs.record_job(record,job))
|
| 216 |
+
collab.system_post(f"Run {receipt['run_id']} started on challenge {challenge_id} for submission {source['id']} ({source.get('title','')}, {len(mirrored['tasks'])} tasks) by {user['name']}. HF job: {receipt.get('job_url')}")
|
| 217 |
+
return receipt
|
| 218 |
|
| 219 |
def release(run_id,note):
|
| 220 |
client=jobs.api()
|
|
|
|
| 308 |
old=next((r for r in rows if r['run_id']==run_id),None)
|
| 309 |
if old: return old
|
| 310 |
rows.append(result);return result
|
| 311 |
+
stored=env.replace_file(RESULTS,change,[])
|
| 312 |
+
collab.system_post(f"Evidence collected for run {run_id} on {challenge_id}: baseline {100*baseline['pass_rate']:.1f}% -> after {100*final['pass_rate']:.1f}% (delta {100*delta:+.1f} pp, GRPO ran: {grpo}). Awaiting organizer review.")
|
| 313 |
+
return stored
|
| 314 |
|
| 315 |
@router.post('/{challenge_id}/runs/{run_id}/review')
|
| 316 |
def review(challenge_id:str,run_id:str,value:Review,request:Request):
|
|
|
|
| 323 |
if all(result.get(k)==v for k,v in update.items()): return result
|
| 324 |
raise HTTPException(409,'Review is immutable.')
|
| 325 |
result.update(**update,reviewed_at=now());return result
|
| 326 |
+
reviewed=env.replace_file(RESULTS,change,[])
|
| 327 |
+
collab.system_post(f"Run {run_id} on {challenge_id} reviewed by {reviewer['name']}: {'valid, now ranked' if value.accepted else 'invalid, not ranked'}. {value.note[:300]}")
|
| 328 |
+
return reviewed
|
| 329 |
|
| 330 |
def submission_rows(challenge_id):
|
| 331 |
"""Every validated submission with its latest run for this challenge (state, spend bound, milestones when running)."""
|
|
|
|
| 344 |
rows.append(row)
|
| 345 |
return rows
|
| 346 |
|
| 347 |
+
def organizer_runs(limit=10):
|
| 348 |
+
"""Organizer pipeline runs (phase-2 loop completion) from results/phase2-runs.json, with live job stage for open ones."""
|
| 349 |
+
try: rows=json.loads(Path(hf_hub_download(RUNS,'results/phase2-runs.json',repo_type='dataset',token=os.environ.get('HF_TOKEN'),force_download=True)).read_text())
|
| 350 |
+
except Exception: return []
|
| 351 |
+
out=[]
|
| 352 |
+
for r in rows[-limit:][::-1]:
|
| 353 |
+
item={k:r.get(k) for k in ('run_name','job_id','job_url','status','created_at','purpose','note')}
|
| 354 |
+
if item['status'] not in jobs.TERMINAL:
|
| 355 |
+
try: item['status']=str(jobs.api().inspect_job(job_id=item['job_id'],namespace='benchflow').status.stage)
|
| 356 |
+
except Exception: pass
|
| 357 |
+
out.append(item)
|
| 358 |
+
return out
|
| 359 |
+
|
| 360 |
@router.get('/{challenge_id}/dashboard')
|
| 361 |
def dashboard(challenge_id:str):
|
| 362 |
"""One call for the dashboard: challenge card, verified leaderboard, submissions with run state, pending results."""
|
|
|
|
| 371 |
card['per_run_allocation']=allocation;card['baseline']=board['baseline']
|
| 372 |
card['participant_flow']=['python arena_cli.py environments submit --file environment.json','python arena_cli.py run --challenge '+row['id']+' --id ENVIRONMENT_ID --file run.json --execute','python arena_cli.py result collect --challenge '+row['id']+' --run-id RUN_ID']
|
| 373 |
counts={'submissions':len(subs),'runs':sum(s['run_count'] for s in subs),'running':sum(1 for s in subs if s['state'] in ('running','queued')),'ranked':len(board['rows']),'pending_review':board['pending_count']}
|
| 374 |
+
return {'challenge':card,'challenges':[{'id':c['id'],'name':c['name'],'status':c['status']} for c in CHALLENGES],'leaderboard':board['rows'],'explanation':board['explanation'],'submissions':subs,'counts':counts,'organizer_runs':organizer_runs(),'generated_at':now()}
|
| 375 |
|
| 376 |
@router.get('/{challenge_id}/leaderboard')
|
| 377 |
def leaderboard(challenge_id:str):
|
check_task.py
CHANGED
|
@@ -33,6 +33,7 @@ REQUIRED_FRONTMATTER = (
|
|
| 33 |
"verifier",
|
| 34 |
"environment",
|
| 35 |
)
|
|
|
|
| 36 |
REQUIRED_METADATA = (
|
| 37 |
"author_name",
|
| 38 |
"author_email",
|
|
@@ -74,6 +75,27 @@ def parse_metadata_keys(block: str) -> set[str]:
|
|
| 74 |
return keys
|
| 75 |
|
| 76 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 77 |
def check_task(task_dir: Path) -> list[str]:
|
| 78 |
issues: list[str] = []
|
| 79 |
|
|
@@ -90,6 +112,18 @@ def check_task(task_dir: Path) -> list[str]:
|
|
| 90 |
frontmatter = m.group(1)
|
| 91 |
body = text[m.end():]
|
| 92 |
top_keys = parse_yaml_keys(frontmatter)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 93 |
for required in REQUIRED_FRONTMATTER:
|
| 94 |
if required not in top_keys:
|
| 95 |
issues.append(f"task.md frontmatter missing: {required}")
|
|
@@ -100,23 +134,7 @@ def check_task(task_dir: Path) -> list[str]:
|
|
| 100 |
if "## prompt" not in body:
|
| 101 |
issues.append("task.md body must contain a '## prompt' section")
|
| 102 |
|
| 103 |
-
|
| 104 |
-
dockerfile = task_dir / "environment" / "Dockerfile"
|
| 105 |
-
if not dockerfile.exists():
|
| 106 |
-
issues.append("Missing required file: environment/Dockerfile")
|
| 107 |
-
else:
|
| 108 |
-
# Skip comment lines and blanks; the first executable instruction
|
| 109 |
-
# must be FROM.
|
| 110 |
-
first_instr = next(
|
| 111 |
-
(
|
| 112 |
-
line
|
| 113 |
-
for line in dockerfile.read_text(encoding="utf-8").splitlines()
|
| 114 |
-
if line.strip() and not line.lstrip().startswith("#")
|
| 115 |
-
),
|
| 116 |
-
"",
|
| 117 |
-
)
|
| 118 |
-
if not first_instr.upper().startswith("FROM "):
|
| 119 |
-
issues.append("environment/Dockerfile first instruction must be FROM")
|
| 120 |
|
| 121 |
# verifier/ -------------------------------------------------------------
|
| 122 |
for f in ("verifier/test.sh", "verifier/test_outputs.py", "verifier/verifier.md"):
|
|
|
|
| 33 |
"verifier",
|
| 34 |
"environment",
|
| 35 |
)
|
| 36 |
+
BENCHFLOW_FRONTMATTER = ("task", "metadata", "agent", "verifier", "sandbox")
|
| 37 |
REQUIRED_METADATA = (
|
| 38 |
"author_name",
|
| 39 |
"author_email",
|
|
|
|
| 75 |
return keys
|
| 76 |
|
| 77 |
|
| 78 |
+
def _check_environment(task_dir: Path) -> list[str]:
|
| 79 |
+
issues: list[str] = []
|
| 80 |
+
dockerfile = task_dir / "environment" / "Dockerfile"
|
| 81 |
+
if not dockerfile.exists():
|
| 82 |
+
issues.append("Missing required file: environment/Dockerfile")
|
| 83 |
+
else:
|
| 84 |
+
# Skip comment lines and blanks; the first executable instruction
|
| 85 |
+
# must be FROM.
|
| 86 |
+
first_instr = next(
|
| 87 |
+
(
|
| 88 |
+
line
|
| 89 |
+
for line in dockerfile.read_text(encoding="utf-8").splitlines()
|
| 90 |
+
if line.strip() and not line.lstrip().startswith("#")
|
| 91 |
+
),
|
| 92 |
+
"",
|
| 93 |
+
)
|
| 94 |
+
if not first_instr.upper().startswith("FROM "):
|
| 95 |
+
issues.append("environment/Dockerfile first instruction must be FROM")
|
| 96 |
+
return issues
|
| 97 |
+
|
| 98 |
+
|
| 99 |
def check_task(task_dir: Path) -> list[str]:
|
| 100 |
issues: list[str] = []
|
| 101 |
|
|
|
|
| 112 |
frontmatter = m.group(1)
|
| 113 |
body = text[m.end():]
|
| 114 |
top_keys = parse_yaml_keys(frontmatter)
|
| 115 |
+
if "schema_version" in top_keys and "task" in top_keys:
|
| 116 |
+
# BenchFlow-native task (schema 1.1): `bench tasks check` is the authority; the arena only
|
| 117 |
+
# requires the executable pieces the pipeline snapshots and runs.
|
| 118 |
+
for required in BENCHFLOW_FRONTMATTER:
|
| 119 |
+
if required not in top_keys:
|
| 120 |
+
issues.append(f"task.md frontmatter missing: {required}")
|
| 121 |
+
if "## prompt" not in body:
|
| 122 |
+
issues.append("task.md body must contain a '## prompt' section")
|
| 123 |
+
issues.extend(_check_environment(task_dir))
|
| 124 |
+
if not (task_dir / "verifier" / "test.sh").exists():
|
| 125 |
+
issues.append("Missing required file: verifier/test.sh")
|
| 126 |
+
return issues
|
| 127 |
for required in REQUIRED_FRONTMATTER:
|
| 128 |
if required not in top_keys:
|
| 129 |
issues.append(f"task.md frontmatter missing: {required}")
|
|
|
|
| 134 |
if "## prompt" not in body:
|
| 135 |
issues.append("task.md body must contain a '## prompt' section")
|
| 136 |
|
| 137 |
+
issues.extend(_check_environment(task_dir))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 138 |
|
| 139 |
# verifier/ -------------------------------------------------------------
|
| 140 |
for f in ("verifier/test.sh", "verifier/test_outputs.py", "verifier/verifier.md"):
|
collab.py
CHANGED
|
@@ -122,6 +122,27 @@ def post_message(value: Message, request: Request):
|
|
| 122 |
result = env.replace_file(MESSAGES, change, [])
|
| 123 |
return {'item':message_item(result),'mentions_delivered':[],'auto_subscribed':False}
|
| 124 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 125 |
class ModelRef(Strict):
|
| 126 |
repo_id: str = Field(pattern=env.SAFE_REPO, max_length=150)
|
| 127 |
revision: str = Field(pattern=r'^[0-9a-f]{40}$')
|
|
|
|
| 122 |
result = env.replace_file(MESSAGES, change, [])
|
| 123 |
return {'item':message_item(result),'mentions_delivered':[],'auto_subscribed':False}
|
| 124 |
|
| 125 |
+
SYSTEM_AGENT='arena-system'
|
| 126 |
+
|
| 127 |
+
def system_post(body, refs=None):
|
| 128 |
+
"""Board post from the arena itself on run lifecycle events. Never raises: the main flow must not depend on the board."""
|
| 129 |
+
try:
|
| 130 |
+
def ensure_agent(rows):
|
| 131 |
+
if not any(a['agent_id']==SYSTEM_AGENT for a in rows):
|
| 132 |
+
rows.append({'agent_id':SYSTEM_AGENT,'description':'Arena system notices: run started, evidence collected, result reviewed.','model':'','harness':'posttrain-arena','owner':'benchflow','created_at':now()})
|
| 133 |
+
return rows
|
| 134 |
+
env.replace_file(AGENTS, ensure_agent, [])
|
| 135 |
+
payload={'body':body[:6000],'refs':list(refs or []),'broadcast':False}
|
| 136 |
+
row={**payload,'request_id':digest([SYSTEM_AGENT,body,int(time.time()//60)]),'owner':'benchflow','agent_id':SYSTEM_AGENT,'created_at':now(),
|
| 137 |
+
'filename':datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')+'_'+SYSTEM_AGENT+'_'+uuid.uuid4().hex[:6]+'.md','type':'agent','payload_hash':digest(payload)}
|
| 138 |
+
def change(rows):
|
| 139 |
+
if any(m.get('request_id')==row['request_id'] for m in rows): return rows
|
| 140 |
+
rows.append(row); return rows
|
| 141 |
+
env.replace_file(MESSAGES, change, [])
|
| 142 |
+
return row
|
| 143 |
+
except Exception:
|
| 144 |
+
return None
|
| 145 |
+
|
| 146 |
class ModelRef(Strict):
|
| 147 |
repo_id: str = Field(pattern=env.SAFE_REPO, max_length=150)
|
| 148 |
revision: str = Field(pattern=r'^[0-9a-f]{40}$')
|
index.html
CHANGED
|
@@ -1249,6 +1249,14 @@
|
|
| 1249 |
</table>
|
| 1250 |
</div>
|
| 1251 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1252 |
<details class="legacy-block" id="legacyBlock">
|
| 1253 |
<summary class="section-title">Practice experiments (seen-task, legacy)<span class="hint">single-task oracle-overfit runs; not held-out</span></summary>
|
| 1254 |
<div style="margin-bottom:16px"><label for="experimentGroup">Comparison group</label> <select class="btn" id="experimentGroup" aria-describedby="groupDescription"><option value="">Loading groups…</option></select><p class="subtitle" id="groupDescription">Results are ranked only within the same model, evaluation and protocol.</p></div>
|
|
@@ -4375,7 +4383,13 @@ function renderChallengeBoard(page) {
|
|
| 4375 |
const rows = page.leaderboard || [];
|
| 4376 |
document.getElementById('clbStatus').textContent = rows.length ? `${rows.length} ranked` : (page.counts.pending_review ? `${page.counts.pending_review} awaiting review` : 'no verified results yet');
|
| 4377 |
document.getElementById('clbExplanation').textContent = page.explanation || '';
|
| 4378 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 4379 |
for (const r of rows) {
|
| 4380 |
const d = Number(r.delta_pp); const cls = d > 0 ? 'delta-pos' : '';
|
| 4381 |
body.insertAdjacentHTML('beforeend', `<tr>
|
|
@@ -4389,6 +4403,18 @@ function renderChallengeBoard(page) {
|
|
| 4389 |
<td><span class="state-badge published">${escapeHtml(r.verification)}</span></td></tr>`);
|
| 4390 |
}
|
| 4391 |
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 4392 |
function renderSubmissions(page) {
|
| 4393 |
const body = document.getElementById('subsBody'); body.innerHTML = '';
|
| 4394 |
const subs = page.submissions || [];
|
|
@@ -4446,7 +4472,7 @@ async function refreshChallenge() {
|
|
| 4446 |
const r = await fetchWithTimeout('/api/challenges/' + encodeURIComponent(selectedChallenge) + '/dashboard');
|
| 4447 |
if (!r.ok) throw new Error('HTTP ' + r.status);
|
| 4448 |
const page = await r.json();
|
| 4449 |
-
renderChallengeCard(page); renderChallengeBoard(page); renderChallengeChart(page.leaderboard || []); renderSubmissions(page);
|
| 4450 |
} catch (err) {
|
| 4451 |
document.getElementById('clbStatus').textContent = 'unavailable';
|
| 4452 |
document.getElementById('subsStatus').textContent = 'unavailable';
|
|
|
|
| 1249 |
</table>
|
| 1250 |
</div>
|
| 1251 |
|
| 1252 |
+
<div class="section-title">Organizer pipeline runs<span class="hint" id="orgRunsStatus">phase-2 loop completion on the same pipeline</span></div>
|
| 1253 |
+
<div style="overflow-x:auto">
|
| 1254 |
+
<table class="lb-table" style="min-width:700px">
|
| 1255 |
+
<thead><tr><th style="width:260px">Run</th><th style="width:110px">Status</th><th>Note</th><th style="width:70px">Job</th></tr></thead>
|
| 1256 |
+
<tbody id="orgRunsBody"></tbody>
|
| 1257 |
+
</table>
|
| 1258 |
+
</div>
|
| 1259 |
+
|
| 1260 |
<details class="legacy-block" id="legacyBlock">
|
| 1261 |
<summary class="section-title">Practice experiments (seen-task, legacy)<span class="hint">single-task oracle-overfit runs; not held-out</span></summary>
|
| 1262 |
<div style="margin-bottom:16px"><label for="experimentGroup">Comparison group</label> <select class="btn" id="experimentGroup" aria-describedby="groupDescription"><option value="">Loading groups…</option></select><p class="subtitle" id="groupDescription">Results are ranked only within the same model, evaluation and protocol.</p></div>
|
|
|
|
| 4383 |
const rows = page.leaderboard || [];
|
| 4384 |
document.getElementById('clbStatus').textContent = rows.length ? `${rows.length} ranked` : (page.counts.pending_review ? `${page.counts.pending_review} awaiting review` : 'no verified results yet');
|
| 4385 |
document.getElementById('clbExplanation').textContent = page.explanation || '';
|
| 4386 |
+
const base = page.challenge && page.challenge.baseline;
|
| 4387 |
+
if (base && base.pass_rate != null) {
|
| 4388 |
+
body.insertAdjacentHTML('beforeend', `<tr class="lb-invalid-sep"><td>—</td><td><strong>${escapeHtml(page.challenge.base_model.repo_id)}</strong><div class="subtitle">base model, no post-training · organizer-measured over ${base.trials} trial${base.trials === 1 ? '' : 's'}</div></td>
|
| 4389 |
+
<td class="num">${pct(base.pass_rate)}</td><td class="num">—</td><td class="num">0.0 ± ${base.stderr != null ? (100 * base.stderr).toFixed(1) : '—'}</td><td>none</td><td>—</td>
|
| 4390 |
+
<td>${base.source ? `<a href="${escapeHtml(base.source)}" target="_blank" rel="noopener noreferrer">aggregate</a>` : '—'}</td><td><span class="state-badge">reference</span></td></tr>`);
|
| 4391 |
+
}
|
| 4392 |
+
if (!rows.length) { body.insertAdjacentHTML('beforeend', `<tr><td colspan="9" class="subtitle">No organizer-verified submission result yet. Submissions and their runs are listed below.</td></tr>`); return; }
|
| 4393 |
for (const r of rows) {
|
| 4394 |
const d = Number(r.delta_pp); const cls = d > 0 ? 'delta-pos' : '';
|
| 4395 |
body.insertAdjacentHTML('beforeend', `<tr>
|
|
|
|
| 4403 |
<td><span class="state-badge published">${escapeHtml(r.verification)}</span></td></tr>`);
|
| 4404 |
}
|
| 4405 |
}
|
| 4406 |
+
function renderOrganizerRuns(page) {
|
| 4407 |
+
const body = document.getElementById('orgRunsBody'); if (!body) return;
|
| 4408 |
+
const rows = page.organizer_runs || []; body.innerHTML = '';
|
| 4409 |
+
document.getElementById('orgRunsStatus').textContent = rows.length ? `${rows.length} most recent · ${rows.filter(r => r.status === 'RUNNING').length} running` : 'none recorded';
|
| 4410 |
+
if (!rows.length) { body.innerHTML = `<tr><td colspan="4" class="subtitle">No organizer runs recorded.</td></tr>`; return; }
|
| 4411 |
+
for (const r of rows) {
|
| 4412 |
+
body.insertAdjacentHTML('beforeend', `<tr><td><strong>${escapeHtml(r.run_name || '')}</strong><div class="subtitle">${escapeHtml(String(r.created_at || '').slice(0, 16))}</div></td>
|
| 4413 |
+
<td><span class="state-badge ${r.status === 'RUNNING' ? 'running' : (r.status === 'COMPLETED' ? 'published' : 'failed')}">${escapeHtml(String(r.status || '').toLowerCase())}</span></td>
|
| 4414 |
+
<td class="subtitle">${escapeHtml(r.note || r.purpose || '')}</td>
|
| 4415 |
+
<td>${r.job_url ? `<a href="${escapeHtml(r.job_url)}" target="_blank" rel="noopener noreferrer">job</a>` : '—'}</td></tr>`);
|
| 4416 |
+
}
|
| 4417 |
+
}
|
| 4418 |
function renderSubmissions(page) {
|
| 4419 |
const body = document.getElementById('subsBody'); body.innerHTML = '';
|
| 4420 |
const subs = page.submissions || [];
|
|
|
|
| 4472 |
const r = await fetchWithTimeout('/api/challenges/' + encodeURIComponent(selectedChallenge) + '/dashboard');
|
| 4473 |
if (!r.ok) throw new Error('HTTP ' + r.status);
|
| 4474 |
const page = await r.json();
|
| 4475 |
+
renderChallengeCard(page); renderChallengeBoard(page); renderChallengeChart(page.leaderboard || []); renderSubmissions(page); renderOrganizerRuns(page);
|
| 4476 |
} catch (err) {
|
| 4477 |
document.getElementById('clbStatus').textContent = 'unavailable';
|
| 4478 |
document.getElementById('subsStatus').textContent = 'unavailable';
|
test_challenges.py
CHANGED
|
@@ -17,7 +17,7 @@ class ChallengeTests(unittest.TestCase):
|
|
| 17 |
self.ledger={'prior_allowance_usd':50,'runs':[]};self.launches=[];self.stage='RUNNING';self.registry={challenges.RESULTS:[]}
|
| 18 |
self.source={'id':'env-abc123abc123','challenge_id':'skillsbench','repo_type':'github','repo_id':'org/pack','revision':REV,'entry_path':'sub','title':'Pack','author':'owner','status':'Validated','source_url':'https://github.com/org/pack/tree/'+REV+'/sub'}
|
| 19 |
self.manifest={'environment_id':self.source['id'],'revision':REV,'tasks':['alpha','beta'],'file_count':12,'bytes':1000,'mirror_revision':MIRROR}
|
| 20 |
-
self.summary=None;self.scores={}
|
| 21 |
api=SimpleNamespace(token='isolated',repo_info=lambda *a,**k:SimpleNamespace(sha=HEAD),inspect_job=lambda **k:SimpleNamespace(status=SimpleNamespace(stage=self.stage)),fetch_job_logs=lambda **k:['vllm up','bridge up','tunnel reachable'])
|
| 22 |
patches=[patch.dict(os.environ,{'HF_TOKEN':'isolated','DAYTONA_API_KEY':'isolated'}),patch.object(jobs,'api',return_value=api),
|
| 23 |
patch.object(jobs,'read',side_effect=lambda head=None:copy.deepcopy(self.ledger)),patch.object(jobs,'write',side_effect=self.write_ledger),
|
|
@@ -26,6 +26,7 @@ class ChallengeTests(unittest.TestCase):
|
|
| 26 |
patch.object(challenges,'mirror',side_effect=lambda row:copy.deepcopy(self.manifest)),patch.object(challenges,'write_bundle',side_effect=lambda row,run_id,m:(BUNDLE,f'bundle/challenge-runs/{run_id}/config.toml')),
|
| 27 |
patch.object(challenges,'hub',return_value=api),patch.object(challenges,'report',side_effect=lambda run_id:copy.deepcopy(self.summary)),patch.object(challenges,'stage_scores',side_effect=lambda run_id,stage,head:copy.deepcopy(self.scores[stage])),
|
| 28 |
patch.object(challenges,'baseline_reference',return_value={'pass_rate':0.017,'stderr':0.006,'trials':2}),patch.object(pipeline_jobs,'launch',side_effect=self.launch_job),
|
|
|
|
| 29 |
patch.object(socket,'create_connection',side_effect=AssertionError('Unexpected network in isolated challenge test'))]
|
| 30 |
for p in patches:p.start();self.addCleanup(p.stop)
|
| 31 |
app=FastAPI();app.include_router(challenges.router);self.client=TestClient(app,raise_server_exceptions=False);auth._sessions.clear()
|
|
@@ -73,6 +74,7 @@ class ChallengeTests(unittest.TestCase):
|
|
| 73 |
self.assertEqual(len(self.launches),1);launch=self.launches[0]
|
| 74 |
self.assertEqual(launch['run_name'],record['run_id']);self.assertEqual(launch['config'],'challenge-runs/'+record['run_id']+'/config.toml');self.assertEqual(launch['bundle_rev'],BUNDLE);self.assertEqual(launch['timeout_seconds'],8*3600)
|
| 75 |
self.assertTrue(launch['space_origin'].startswith('https://'));self.assertGreaterEqual(len(launch['relay_key']),48);self.assertNotIn('relay_key',json.dumps(self.ledger))
|
|
|
|
| 76 |
ledger=self.ledger['runs'][0];self.assertEqual(ledger['kind'],'challenge-run');self.assertEqual(ledger['job_id'],'1'*24)
|
| 77 |
self.assertEqual(self.start(),record);self.assertEqual(len(self.launches),1)
|
| 78 |
self.post('/api/challenges/tb2-9b/runs',{'request_id':'stable-run-request','environment_id':'env-other'},expected=409)
|
|
@@ -140,5 +142,6 @@ class ChallengeTests(unittest.TestCase):
|
|
| 140 |
self.post('/api/challenges/tb2-9b/runs/'+record['run_id']+'/review',{'accepted':True,'note':'Evidence reviewed: job log, score.json and per-task results agree.'},user=EDITOR)
|
| 141 |
page=self.client.get('/api/challenges/tb2-9b/dashboard').json()
|
| 142 |
self.assertEqual(page['submissions'][0]['state'],'published');self.assertEqual(page['leaderboard'][0]['rank'],1);self.assertEqual(page['counts']['ranked'],1)
|
|
|
|
| 143 |
|
| 144 |
if __name__=='__main__':unittest.main()
|
|
|
|
| 17 |
self.ledger={'prior_allowance_usd':50,'runs':[]};self.launches=[];self.stage='RUNNING';self.registry={challenges.RESULTS:[]}
|
| 18 |
self.source={'id':'env-abc123abc123','challenge_id':'skillsbench','repo_type':'github','repo_id':'org/pack','revision':REV,'entry_path':'sub','title':'Pack','author':'owner','status':'Validated','source_url':'https://github.com/org/pack/tree/'+REV+'/sub'}
|
| 19 |
self.manifest={'environment_id':self.source['id'],'revision':REV,'tasks':['alpha','beta'],'file_count':12,'bytes':1000,'mirror_revision':MIRROR}
|
| 20 |
+
self.summary=None;self.scores={};self.posts=[]
|
| 21 |
api=SimpleNamespace(token='isolated',repo_info=lambda *a,**k:SimpleNamespace(sha=HEAD),inspect_job=lambda **k:SimpleNamespace(status=SimpleNamespace(stage=self.stage)),fetch_job_logs=lambda **k:['vllm up','bridge up','tunnel reachable'])
|
| 22 |
patches=[patch.dict(os.environ,{'HF_TOKEN':'isolated','DAYTONA_API_KEY':'isolated'}),patch.object(jobs,'api',return_value=api),
|
| 23 |
patch.object(jobs,'read',side_effect=lambda head=None:copy.deepcopy(self.ledger)),patch.object(jobs,'write',side_effect=self.write_ledger),
|
|
|
|
| 26 |
patch.object(challenges,'mirror',side_effect=lambda row:copy.deepcopy(self.manifest)),patch.object(challenges,'write_bundle',side_effect=lambda row,run_id,m:(BUNDLE,f'bundle/challenge-runs/{run_id}/config.toml')),
|
| 27 |
patch.object(challenges,'hub',return_value=api),patch.object(challenges,'report',side_effect=lambda run_id:copy.deepcopy(self.summary)),patch.object(challenges,'stage_scores',side_effect=lambda run_id,stage,head:copy.deepcopy(self.scores[stage])),
|
| 28 |
patch.object(challenges,'baseline_reference',return_value={'pass_rate':0.017,'stderr':0.006,'trials':2}),patch.object(pipeline_jobs,'launch',side_effect=self.launch_job),
|
| 29 |
+
patch.object(challenges,'organizer_runs',return_value=[{'run_name':'phase2-r21','job_id':'2'*24,'status':'RUNNING','note':'in flight'}]),patch.object(challenges.collab,'system_post',side_effect=lambda body,refs=None:self.posts.append(body)),
|
| 30 |
patch.object(socket,'create_connection',side_effect=AssertionError('Unexpected network in isolated challenge test'))]
|
| 31 |
for p in patches:p.start();self.addCleanup(p.stop)
|
| 32 |
app=FastAPI();app.include_router(challenges.router);self.client=TestClient(app,raise_server_exceptions=False);auth._sessions.clear()
|
|
|
|
| 74 |
self.assertEqual(len(self.launches),1);launch=self.launches[0]
|
| 75 |
self.assertEqual(launch['run_name'],record['run_id']);self.assertEqual(launch['config'],'challenge-runs/'+record['run_id']+'/config.toml');self.assertEqual(launch['bundle_rev'],BUNDLE);self.assertEqual(launch['timeout_seconds'],8*3600)
|
| 76 |
self.assertTrue(launch['space_origin'].startswith('https://'));self.assertGreaterEqual(len(launch['relay_key']),48);self.assertNotIn('relay_key',json.dumps(self.ledger))
|
| 77 |
+
self.assertEqual((launch['vllm_tp'],launch['vllm_gpus']),(4,'4,5,6,7'));self.assertEqual(len(self.posts),1);self.assertIn('started on challenge tb2-9b',self.posts[0])
|
| 78 |
ledger=self.ledger['runs'][0];self.assertEqual(ledger['kind'],'challenge-run');self.assertEqual(ledger['job_id'],'1'*24)
|
| 79 |
self.assertEqual(self.start(),record);self.assertEqual(len(self.launches),1)
|
| 80 |
self.post('/api/challenges/tb2-9b/runs',{'request_id':'stable-run-request','environment_id':'env-other'},expected=409)
|
|
|
|
| 142 |
self.post('/api/challenges/tb2-9b/runs/'+record['run_id']+'/review',{'accepted':True,'note':'Evidence reviewed: job log, score.json and per-task results agree.'},user=EDITOR)
|
| 143 |
page=self.client.get('/api/challenges/tb2-9b/dashboard').json()
|
| 144 |
self.assertEqual(page['submissions'][0]['state'],'published');self.assertEqual(page['leaderboard'][0]['rank'],1);self.assertEqual(page['counts']['ranked'],1)
|
| 145 |
+
self.assertEqual(page['organizer_runs'][0]['run_name'],'phase2-r21');self.assertEqual(len(self.posts),3);self.assertIn('Evidence collected',self.posts[1]);self.assertIn('reviewed by editor',self.posts[2])
|
| 146 |
|
| 147 |
if __name__=='__main__':unittest.main()
|
test_check_task.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""The structural checker accepts BenchFlow-native task packages (schema 1.1) as well as the starter-kit layout."""
|
| 2 |
+
import tempfile, unittest
|
| 3 |
+
from pathlib import Path
|
| 4 |
+
from check_task import check_task
|
| 5 |
+
|
| 6 |
+
BENCHFLOW_TASK = """---
|
| 7 |
+
schema_version: '1.1'
|
| 8 |
+
task:
|
| 9 |
+
name: tmax/task_x
|
| 10 |
+
metadata:
|
| 11 |
+
source: legacy
|
| 12 |
+
agent:
|
| 13 |
+
timeout_sec: 600.0
|
| 14 |
+
verifier:
|
| 15 |
+
timeout_sec: 120.0
|
| 16 |
+
sandbox:
|
| 17 |
+
cpus: 1
|
| 18 |
+
---
|
| 19 |
+
|
| 20 |
+
## prompt
|
| 21 |
+
|
| 22 |
+
Do the thing.
|
| 23 |
+
"""
|
| 24 |
+
|
| 25 |
+
class CheckTaskTests(unittest.TestCase):
|
| 26 |
+
def make(self, task_md, files):
|
| 27 |
+
d = Path(tempfile.mkdtemp()) / 'task'; d.mkdir()
|
| 28 |
+
(d / 'task.md').write_text(task_md)
|
| 29 |
+
for rel, body in files.items():
|
| 30 |
+
(d / rel).parent.mkdir(parents=True, exist_ok=True); (d / rel).write_text(body)
|
| 31 |
+
return d
|
| 32 |
+
def test_benchflow_native_task_needs_only_dockerfile_and_test_sh(self):
|
| 33 |
+
d = self.make(BENCHFLOW_TASK, {'environment/Dockerfile': 'FROM python:3.12\n', 'verifier/test.sh': '#!/bin/bash\n'})
|
| 34 |
+
self.assertEqual(check_task(d), [])
|
| 35 |
+
d2 = self.make(BENCHFLOW_TASK, {'environment/Dockerfile': 'FROM python:3.12\n'})
|
| 36 |
+
self.assertIn('Missing required file: verifier/test.sh', check_task(d2))
|
| 37 |
+
d3 = self.make(BENCHFLOW_TASK.replace('sandbox:\n cpus: 1\n', ''), {'environment/Dockerfile': 'FROM x\n', 'verifier/test.sh': ''})
|
| 38 |
+
self.assertIn('task.md frontmatter missing: sandbox', check_task(d3))
|
| 39 |
+
def test_starter_kit_layout_still_requires_oracle_and_rubrics(self):
|
| 40 |
+
kit = """---\nversion: \"1.0\"\nmetadata:\n author_name: a\n author_email: a@b\n category: c\nagent:\n timeout_sec: 1\nverifier:\n timeout_sec: 1\nenvironment:\n cpus: 1\n---\n\n## prompt\n"""
|
| 41 |
+
d = self.make(kit, {'environment/Dockerfile': 'FROM x\n', 'verifier/test.sh': '', 'verifier/test_outputs.py': '', 'verifier/verifier.md': ''})
|
| 42 |
+
issues = check_task(d)
|
| 43 |
+
self.assertIn('Missing required directory: verifier/rubrics/', issues); self.assertIn('Missing required file: oracle/solve.sh', issues)
|
| 44 |
+
|
| 45 |
+
if __name__ == '__main__': unittest.main()
|