Spaces:
Running
Running
Xiangyi Li commited on
Commit ·
e533464
1
Parent(s): 1ac3e29
Check submissions against every sealed suite; count only failed time-outs
Browse filesDecontamination and name checks now cover every sealed suite in configs/suites (full TB2 and LHTB,
not only tb2-9b's 32 tasks), once per dataset revision. Recipe v2's per-suite stage markers
(baseline_eval.<suite>.tNN) map to their stage. A task that passes after the time limit is a pass,
not a time-out.
- challenges.py +2 -2
- pipeline_jobs.py +1 -2
- test_compose.py +18 -0
- validation_gates.py +17 -10
challenges.py
CHANGED
|
@@ -550,7 +550,7 @@ def parse_log(events):
|
|
| 550 |
stage=stages[current]
|
| 551 |
if 'VLLM_HEALTH_FAILED' in line: health_failures+=1;continue
|
| 552 |
if line.startswith('[posttrainarena] '):
|
| 553 |
-
name=line[17:].split(':',1)[0].strip();rollout=ROLLOUT.match(name);key='training' if rollout else STAGE_OF.get(name)
|
| 554 |
if key:
|
| 555 |
enter(key,t);pipeline_stage='train_grpo' if rollout else name
|
| 556 |
if rollout: train_step=int(rollout.group(1));stages[key]['rollouts']+=1;stages[key]['steps'].add(train_step)
|
|
@@ -566,7 +566,7 @@ def parse_log(events):
|
|
| 566 |
kind={'PASS':'pass','FAIL':'fail','ERR':'error'}[verdict.group(1)];stage[kind]+=1
|
| 567 |
note=(verdict.group(3) or '').rstrip('…').strip()
|
| 568 |
feed.append({'t':t,'stage':current,'event':kind,'task':verdict.group(2),'note':note[:120] or None,'step':train_step if current=='training' else None})
|
| 569 |
-
if 'wall-clock' in note or 'idle timeout' in note.lower(): stage['timeout']+=1
|
| 570 |
elif kind=='error': note=note[:90] or 'unspecified';stage['reasons'][note]=stage['reasons'].get(note,0)+1
|
| 571 |
if current=='training' and train_step is not None:
|
| 572 |
counts=step_verdicts.setdefault(train_step,{'pass':0,'fail':0,'error':0});counts[kind]+=1
|
|
|
|
| 550 |
stage=stages[current]
|
| 551 |
if 'VLLM_HEALTH_FAILED' in line: health_failures+=1;continue
|
| 552 |
if line.startswith('[posttrainarena] '):
|
| 553 |
+
name=line[17:].split(':',1)[0].strip();rollout=ROLLOUT.match(name);key='training' if rollout else STAGE_OF.get(name) or STAGE_OF.get(name.split('.',1)[0]) # recipe v2: baseline_eval.<suite>.tNN
|
| 554 |
if key:
|
| 555 |
enter(key,t);pipeline_stage='train_grpo' if rollout else name
|
| 556 |
if rollout: train_step=int(rollout.group(1));stages[key]['rollouts']+=1;stages[key]['steps'].add(train_step)
|
|
|
|
| 566 |
kind={'PASS':'pass','FAIL':'fail','ERR':'error'}[verdict.group(1)];stage[kind]+=1
|
| 567 |
note=(verdict.group(3) or '').rstrip('…').strip()
|
| 568 |
feed.append({'t':t,'stage':current,'event':kind,'task':verdict.group(2),'note':note[:120] or None,'step':train_step if current=='training' else None})
|
| 569 |
+
if kind!='pass' and ('wall-clock' in note or 'idle timeout' in note.lower()): stage['timeout']+=1 # a pass after the limit is a pass
|
| 570 |
elif kind=='error': note=note[:90] or 'unspecified';stage['reasons'][note]=stage['reasons'].get(note,0)+1
|
| 571 |
if current=='training' and train_step is not None:
|
| 572 |
counts=step_verdicts.setdefault(train_step,{'pass':0,'fail':0,'error':0});counts[kind]+=1
|
pipeline_jobs.py
CHANGED
|
@@ -4,8 +4,7 @@ The job mirrors pipelines/benchflow-task-posttrain/scripts/bootstrap_gpu.sh (tor
|
|
| 4 |
vLLM 0.23 cu129 wheel, pipeline[train]), serves the policy with the TRL vLLM server on one GPU,
|
| 5 |
reaches the model bridge on loopback (no inbound tunnel; rollout ingress goes through the Space relay), and runs
|
| 6 |
`posttrainarena-train run` on GPUs 0-3 with Daytona sandboxes. Results upload to RUNS.
|
| 7 |
-
|
| 8 |
-
when changing one of them.
|
| 9 |
"""
|
| 10 |
import os
|
| 11 |
from pathlib import Path
|
|
|
|
| 4 |
vLLM 0.23 cu129 wheel, pipeline[train]), serves the policy with the TRL vLLM server on one GPU,
|
| 5 |
reaches the model bridge on loopback (no inbound tunnel; rollout ingress goes through the Space relay), and runs
|
| 6 |
`posttrainarena-train run` on GPUs 0-3 with Daytona sandboxes. Results upload to RUNS.
|
| 7 |
+
The organizer's baseline grid (~/benchflow/pta-work/launchers/eval_grid.py) reuses this bootstrap; keep them aligned.
|
|
|
|
| 8 |
"""
|
| 9 |
import os
|
| 10 |
from pathlib import Path
|
test_compose.py
CHANGED
|
@@ -86,3 +86,21 @@ class ComposeTest(unittest.TestCase):
|
|
| 86 |
|
| 87 |
if __name__ == '__main__':
|
| 88 |
unittest.main()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 86 |
|
| 87 |
if __name__ == '__main__':
|
| 88 |
unittest.main()
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
class SealedCoverageTest(unittest.TestCase):
|
| 92 |
+
def test_every_sealed_suite_is_checked_once_per_dataset_revision(self):
|
| 93 |
+
import validation_gates as gates
|
| 94 |
+
with mock.patch.object(gates, '_sealed_instructions', return_value={}):
|
| 95 |
+
suites = {s.repo_id: s for s in gates.sealed_suites()}
|
| 96 |
+
tb2 = suites['benchflow/tb2-benchflow']
|
| 97 |
+
self.assertEqual(len(tb2.evaluated), 86) # tb2-32 and tb2 share one revision: the union
|
| 98 |
+
self.assertEqual(len(suites['benchflow/lhtb-nongame-benchflow'].evaluated), 38)
|
| 99 |
+
|
| 100 |
+
def test_recipe_v2_stage_names_and_late_passes(self):
|
| 101 |
+
import challenges
|
| 102 |
+
events = [('t1', '[posttrainarena] baseline_eval.lhtb.t02: bench eval run ...'), ('t2', '[PASS] task-a (tools=3) (Agent prompt exceeded wall-clock budget 900s)'),
|
| 103 |
+
('t3', '[FAIL] task-b (tools=2) (Agent prompt exceeded wall-clock budget 900s)')]
|
| 104 |
+
parsed = challenges.parse_log(events)
|
| 105 |
+
self.assertIn('baseline', parsed['stages'])
|
| 106 |
+
self.assertEqual((parsed['stages']['baseline']['pass'], parsed['stages']['baseline']['timeout']), (1, 1))
|
validation_gates.py
CHANGED
|
@@ -475,20 +475,27 @@ def _sealed_instructions(repo_id: str, revision: str):
|
|
| 475 |
|
| 476 |
|
| 477 |
def sealed_suites():
|
| 478 |
-
"""
|
| 479 |
-
|
| 480 |
-
|
| 481 |
-
|
| 482 |
-
|
| 483 |
-
|
| 484 |
-
|
|
|
|
|
|
|
| 485 |
continue
|
| 486 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 487 |
try:
|
| 488 |
-
instructions, error = _sealed_instructions(
|
| 489 |
except Exception as exc: # network, token, or dataset problem: degrade to a review note, never block
|
| 490 |
instructions, error = None, type(exc).__name__
|
| 491 |
-
suites.append(SealedSuite(
|
| 492 |
return suites
|
| 493 |
|
| 494 |
|
|
|
|
| 475 |
|
| 476 |
|
| 477 |
def sealed_suites():
|
| 478 |
+
"""Every sealed held-out suite registered in configs/suites, planned ones included, so a submission cannot copy a
|
| 479 |
+
task a current or future challenge evaluates. Suites on the same dataset revision are checked once, with the union
|
| 480 |
+
of their task lists. Instructions come from the private dataset with the Space token; if they cannot be read, name
|
| 481 |
+
checks still run against the task lists."""
|
| 482 |
+
import compose # lazy, like the challenge registry this replaced
|
| 483 |
+
by_source = {}
|
| 484 |
+
for suite_id in compose.ids('suites'):
|
| 485 |
+
data = compose.fragment('suites', suite_id)
|
| 486 |
+
if not data.get('meta', {}).get('sealed'):
|
| 487 |
continue
|
| 488 |
+
suite = data['suite']
|
| 489 |
+
entry = by_source.setdefault((suite['repo_id'], suite['revision']), {'ids': [], 'tasks': set()})
|
| 490 |
+
entry['ids'].append(suite_id)
|
| 491 |
+
entry['tasks'].update(compose.task_ids(suite_id))
|
| 492 |
+
suites = []
|
| 493 |
+
for (repo_id, revision), entry in by_source.items():
|
| 494 |
try:
|
| 495 |
+
instructions, error = _sealed_instructions(repo_id, revision), None
|
| 496 |
except Exception as exc: # network, token, or dataset problem: degrade to a review note, never block
|
| 497 |
instructions, error = None, type(exc).__name__
|
| 498 |
+
suites.append(SealedSuite(' + '.join(sorted(entry['ids'])), repo_id, revision, sorted(entry['tasks']), instructions, error))
|
| 499 |
return suites
|
| 500 |
|
| 501 |
|