Xiangyi Li commited on
Commit
e533464
·
1 Parent(s): 1ac3e29

Check submissions against every sealed suite; count only failed time-outs

Browse files

Decontamination and name checks now cover every sealed suite in configs/suites (full TB2 and LHTB,
not only tb2-9b's 32 tasks), once per dataset revision. Recipe v2's per-suite stage markers
(baseline_eval.<suite>.tNN) map to their stage. A task that passes after the time limit is a pass,
not a time-out.

Files changed (4) hide show
  1. challenges.py +2 -2
  2. pipeline_jobs.py +1 -2
  3. test_compose.py +18 -0
  4. validation_gates.py +17 -10
challenges.py CHANGED
@@ -550,7 +550,7 @@ def parse_log(events):
550
  stage=stages[current]
551
  if 'VLLM_HEALTH_FAILED' in line: health_failures+=1;continue
552
  if line.startswith('[posttrainarena] '):
553
- name=line[17:].split(':',1)[0].strip();rollout=ROLLOUT.match(name);key='training' if rollout else STAGE_OF.get(name)
554
  if key:
555
  enter(key,t);pipeline_stage='train_grpo' if rollout else name
556
  if rollout: train_step=int(rollout.group(1));stages[key]['rollouts']+=1;stages[key]['steps'].add(train_step)
@@ -566,7 +566,7 @@ def parse_log(events):
566
  kind={'PASS':'pass','FAIL':'fail','ERR':'error'}[verdict.group(1)];stage[kind]+=1
567
  note=(verdict.group(3) or '').rstrip('…').strip()
568
  feed.append({'t':t,'stage':current,'event':kind,'task':verdict.group(2),'note':note[:120] or None,'step':train_step if current=='training' else None})
569
- if 'wall-clock' in note or 'idle timeout' in note.lower(): stage['timeout']+=1
570
  elif kind=='error': note=note[:90] or 'unspecified';stage['reasons'][note]=stage['reasons'].get(note,0)+1
571
  if current=='training' and train_step is not None:
572
  counts=step_verdicts.setdefault(train_step,{'pass':0,'fail':0,'error':0});counts[kind]+=1
 
550
  stage=stages[current]
551
  if 'VLLM_HEALTH_FAILED' in line: health_failures+=1;continue
552
  if line.startswith('[posttrainarena] '):
553
+ name=line[17:].split(':',1)[0].strip();rollout=ROLLOUT.match(name);key='training' if rollout else STAGE_OF.get(name) or STAGE_OF.get(name.split('.',1)[0]) # recipe v2: baseline_eval.<suite>.tNN
554
  if key:
555
  enter(key,t);pipeline_stage='train_grpo' if rollout else name
556
  if rollout: train_step=int(rollout.group(1));stages[key]['rollouts']+=1;stages[key]['steps'].add(train_step)
 
566
  kind={'PASS':'pass','FAIL':'fail','ERR':'error'}[verdict.group(1)];stage[kind]+=1
567
  note=(verdict.group(3) or '').rstrip('…').strip()
568
  feed.append({'t':t,'stage':current,'event':kind,'task':verdict.group(2),'note':note[:120] or None,'step':train_step if current=='training' else None})
569
+ if kind!='pass' and ('wall-clock' in note or 'idle timeout' in note.lower()): stage['timeout']+=1 # a pass after the limit is a pass
570
  elif kind=='error': note=note[:90] or 'unspecified';stage['reasons'][note]=stage['reasons'].get(note,0)+1
571
  if current=='training' and train_step is not None:
572
  counts=step_verdicts.setdefault(train_step,{'pass':0,'fail':0,'error':0});counts[kind]+=1
pipeline_jobs.py CHANGED
@@ -4,8 +4,7 @@ The job mirrors pipelines/benchflow-task-posttrain/scripts/bootstrap_gpu.sh (tor
4
  vLLM 0.23 cu129 wheel, pipeline[train]), serves the policy with the TRL vLLM server on one GPU,
5
  reaches the model bridge on loopback (no inbound tunnel; rollout ingress goes through the Space relay), and runs
6
  `posttrainarena-train run` on GPUs 0-3 with Daytona sandboxes. Results upload to RUNS.
7
- This is the launcher that completed the phase-2 loop; keep it byte-for-byte with /tmp/pta-live/phase2-job.py
8
- when changing one of them.
9
  """
10
  import os
11
  from pathlib import Path
 
4
  vLLM 0.23 cu129 wheel, pipeline[train]), serves the policy with the TRL vLLM server on one GPU,
5
  reaches the model bridge on loopback (no inbound tunnel; rollout ingress goes through the Space relay), and runs
6
  `posttrainarena-train run` on GPUs 0-3 with Daytona sandboxes. Results upload to RUNS.
7
+ The organizer's baseline grid (~/benchflow/pta-work/launchers/eval_grid.py) reuses this bootstrap; keep them aligned.
 
8
  """
9
  import os
10
  from pathlib import Path
test_compose.py CHANGED
@@ -86,3 +86,21 @@ class ComposeTest(unittest.TestCase):
86
 
87
  if __name__ == '__main__':
88
  unittest.main()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
86
 
87
  if __name__ == '__main__':
88
  unittest.main()
89
+
90
+
91
+ class SealedCoverageTest(unittest.TestCase):
92
+ def test_every_sealed_suite_is_checked_once_per_dataset_revision(self):
93
+ import validation_gates as gates
94
+ with mock.patch.object(gates, '_sealed_instructions', return_value={}):
95
+ suites = {s.repo_id: s for s in gates.sealed_suites()}
96
+ tb2 = suites['benchflow/tb2-benchflow']
97
+ self.assertEqual(len(tb2.evaluated), 86) # tb2-32 and tb2 share one revision: the union
98
+ self.assertEqual(len(suites['benchflow/lhtb-nongame-benchflow'].evaluated), 38)
99
+
100
+ def test_recipe_v2_stage_names_and_late_passes(self):
101
+ import challenges
102
+ events = [('t1', '[posttrainarena] baseline_eval.lhtb.t02: bench eval run ...'), ('t2', '[PASS] task-a (tools=3) (Agent prompt exceeded wall-clock budget 900s)'),
103
+ ('t3', '[FAIL] task-b (tools=2) (Agent prompt exceeded wall-clock budget 900s)')]
104
+ parsed = challenges.parse_log(events)
105
+ self.assertIn('baseline', parsed['stages'])
106
+ self.assertEqual((parsed['stages']['baseline']['pass'], parsed['stages']['baseline']['timeout']), (1, 1))
validation_gates.py CHANGED
@@ -475,20 +475,27 @@ def _sealed_instructions(repo_id: str, revision: str):
475
 
476
 
477
  def sealed_suites():
478
- """Sealed suites of the arena challenges. Instructions come from the private dataset at the pinned revision with
479
- the Space token; if they cannot be read, name checks still run against the evaluated task list."""
480
- import challenges # lazy: challenges imports environments, which imports this module
481
- suites = []
482
- for row in challenges.CHALLENGES:
483
- suite = row.get('eval_suite') or {}
484
- if not suite.get('sealed'):
 
 
485
  continue
486
- evaluated = challenges.suite_task_ids(row)
 
 
 
 
 
487
  try:
488
- instructions, error = _sealed_instructions(suite['repo_id'], suite['revision']), None
489
  except Exception as exc: # network, token, or dataset problem: degrade to a review note, never block
490
  instructions, error = None, type(exc).__name__
491
- suites.append(SealedSuite(row['id'], suite['repo_id'], suite['revision'], evaluated, instructions, error))
492
  return suites
493
 
494
 
 
475
 
476
 
477
  def sealed_suites():
478
+ """Every sealed held-out suite registered in configs/suites, planned ones included, so a submission cannot copy a
479
+ task a current or future challenge evaluates. Suites on the same dataset revision are checked once, with the union
480
+ of their task lists. Instructions come from the private dataset with the Space token; if they cannot be read, name
481
+ checks still run against the task lists."""
482
+ import compose # lazy, like the challenge registry this replaced
483
+ by_source = {}
484
+ for suite_id in compose.ids('suites'):
485
+ data = compose.fragment('suites', suite_id)
486
+ if not data.get('meta', {}).get('sealed'):
487
  continue
488
+ suite = data['suite']
489
+ entry = by_source.setdefault((suite['repo_id'], suite['revision']), {'ids': [], 'tasks': set()})
490
+ entry['ids'].append(suite_id)
491
+ entry['tasks'].update(compose.task_ids(suite_id))
492
+ suites = []
493
+ for (repo_id, revision), entry in by_source.items():
494
  try:
495
+ instructions, error = _sealed_instructions(repo_id, revision), None
496
  except Exception as exc: # network, token, or dataset problem: degrade to a review note, never block
497
  instructions, error = None, type(exc).__name__
498
+ suites.append(SealedSuite(' + '.join(sorted(entry['ids'])), repo_id, revision, sorted(entry['tasks']), instructions, error))
499
  return suites
500
 
501