"""The PostTrain commands on a submission's page (the submissions app at /arena): evaluate a model on a submitted collection's tasks, and hill-climb on them: a stronger model's verified attempts at some of the tasks become SFT data for a smaller model, which is then compared with its base on the tasks it didn't train on. They are the commands of PostTrain 0.1.11's guide, "Sandboxed agent environments" (GUIDE), filled in with the collection's repository, commit and folder: its hill climb (a teacher's verified attempts as per-turn rows without thinking, LoRA r=16 at lr 1e-4 on a constant schedule, two epochs as two warm-started runs, every eval of the student asked the same way on one deployment that serves the base and its LoRAs alike). The models are the ones the guide's recorded runs used. The page shows none of the guide's results. Two differences from the guide's text: the student's eval settings are written out in each command instead of the guide's S=(...) and "${S[@]}", so a command copied alone keeps them; and the SFT runs follow their job instead of detaching (-d), so the block runs in order. `posttrain env add` (0.1.9) reads task packages by what a run needs: a prompt, environment/Dockerfile and verifier/test.sh. A package without the parts only the Arena's spec asks for (oracle/solve.sh, verifier/test_outputs.py, verifier/verifier.md, rubrics) is added with a note, as BenchFlow's own packages (TMax) are; without oracle/solve.sh, no oracle run can show that the verifier passes a correct answer, so a task no model solves may be broken rather than hard. The arena's static checks report a missing oracle/solve.sh as S-NO-ORACLE, which the page says. A collection none of whose tasks has one is packaged the way BenchFlow's own tasks are, and those are written for root, as the Arena's own runs assumed (they work in /home/user, which the image's root owns): its evals run the agent as root (--set sandbox_user=root), as the guide's TMax eval does. It shows no result of these commands: none has run on a collection here. Round 10 (Sept 29, 2026): the eval starts with 4 tasks (--limit 4, as the guide does); the hill climb registers its held-out tasks as the project's `quick` suite right after adding them, so PostTrain's overlap check keeps their attempts out of the SFT rows (without a suite, rows from a held-out task passed the SFT dry run; with it, the dry run was refused); each command comes with the cost and time its dry run estimates (estimate()); and for tasks without reference solutions the comparison is called indicative, since a held-out task no model solves may be broken rather than hard. Round 11 (Sept 29, 2026; PostTrain 0.1.10, whose dry runs print the same costs): the held-out folder is added and set as the `quick` suite before the training folder is added, so the training folder's `env add` runs its held-out overlap check (added first, it said "not run: … no suite of the project holds out an environment"); the setup installs the data checks' libraries (`posttrain extras install data`), which the hill climb's `from-rollouts --base` needs to render the smaller model's chat template; and a first eval's time allows for an attempt that runs into its task's time limit: TMax's first eval (4 tasks, time limit 600 s) took 11 minutes, twice, where the page said 3 to 8. Round 12 (Sept 29, 2026; PostTrain 0.1.11): the static checks (gates-v3) call an answer-like file that holds what its verifier checks grading data, as env add does, which excludes TMax's task_000519 (its /tmp/expected_results.json); the page's counts follow env add's: grading data in the leaky tasks, answer-like files only in the others (a note, not a leak). On TMax's 59 kept tasks both say 3 and 15. Round 12, F12-01 (PostTrain 0.1.11): an eval of task packages leaves out the tasks env add marks leaky unless --include-leaky, and --limit takes the first of the others by name. The page names those tasks (`leaky`: the kept tasks with grading data in their sandbox, where the static checks and env add agree) and its estimates count only the tasks an eval scores. 0.1.11's dry runs of TMax's own commands on its 59 kept tasks (Sept 29, 2026; nothing launched): env add 3 leaky (task_000048_e745a1e9, task_000742_084d8048, task_000830_4a4c601e, all in the training split) and 15 notes; the first eval $0.09 ($0.03–$0.26), 4 of the 56 it scores; every task twice $2.55 ($0.90–$7.22), 56 tasks; the teacher $1.87 ($0.66–$5.28), 41 of the training split's 44; each held-out eval $10.62 ($5.96–$18.93), 15 tasks, 38–120 min of attempts. """ import math import re GUIDE = 'https://app.posttrain.com/docs/stages/agent-environments' QUICKSTART = 'https://app.posttrain.com/docs/quickstart-terminal' SPEC = 'https://posttrain.com/docs/spec' POSTTRAIN = '0.1.11' # the release these commands were checked with (round 12: its dry runs of them) TEACHER = 'accounts/fireworks/models/glm-5p3-flash' # the guide's serverless model: its verified attempts become the data STUDENT = 'accounts/fireworks/models/qwen3-4b' # the guide's student, which Fireworks serves on a deployment STUDENT_HF = 'Qwen/Qwen3-4B' # the student's chat template and tokenizer, for the dataset's checks EVERY = 4 # the 1st, 5th, 9th, ... package is held out of training; a collection of fewer packages has too few to split # The guide's settings for every eval of the student and its fine-tunes (its S="..."): four attempts per task, the model # asked without thinking (the format the rows train), a 30-turn cap against loops, and the sampling Qwen3 suggests for it. ASK = '--set k=4 --set max_turns=30 --set temperature=0.7 --set top_p=0.8 --set top_k=20 --set max_tokens=4096 --set reasoning_effort=none' # The guide's SFT: LoRA r=16, lr 1e-4 on a constant schedule, one epoch per run (the second run warm-starts from the first). SFT = '--set method=lora --set lora.r=16 --set epochs=1 --set batch_size=16 --set lr=1e-4 --set scheduler=constant --set max_seq_len=32768' # The guide's rows: one per model turn, without the teacher's thinking, leaving out attempts that loop or ramble. ROWS = '--per-turn --drop-reasoning --max-repeats 2 --max-turns 20 --max-turn-tokens 4000' ROOT = '--set sandbox_user=root' # for tasks written for root (BenchFlow's own packages, TMax's), as the Arena's runs gave the agent root FIRST = 4 # the first eval's tasks (--limit 4, one attempt each, as the guide's first TMax eval) # What `--dry-run` estimates for these commands: PostTrain's estimate of an agent eval on Fireworks (backends/fireworks.py, # estimate_agent_eval; the same in 0.1.9 to 0.1.11), so the page gives cost and time without a launch, for the attempts an # eval makes: from 0.1.11, at the tasks it scores (not the leaky ones). The dry runs of TMax dogfood 64's commands printed # these same figures with 0.1.11 on Sept 29, 2026: $0.09 ($0.03–$0.26) for the first eval, $2.55 ($0.90–$7.22) for every # task twice, $1.87 ($0.66–$5.28) for the teacher, $10.62 ($5.96–$18.93) for each held-out eval of the student. # An attempt: ~20K–300K prompt and 0.5K–8K completion tokens, and a Daytona sandbox (1 vCPU, 4 GiB, 10 GiB) of ~2.5–8 min. SANDBOX_HOUR = 0.0504 + 4 * 0.0162 + 5 * 0.000108 # Daytona's list prices, USD an hour: a vCPU, 4 GiB, disk beyond the free 5 GiB MINUTES = (2.5, 8) # an attempt's sandbox, low and high LIMIT = 10 # minutes an attempt runs when it runs into its task's time limit (TMax's: 600 s) START = 2 # minutes to start an eval's sandboxes and to grade its last attempts TOKENS = ((20_000, 500), (300_000, 8_000)) # an attempt's prompt and completion tokens, low and high TEACHER_PRICE = (0.15, 0.5) # glm-5p3-flash on Fireworks serverless, USD per million tokens in and out H100_HOUR, WARM, IDLE = 8.0, 10, 5 # the student's deployment: USD an hour, minutes to scale up, idle minutes AT_ONCE = 4 # attempts at once (the eval's concurrency) SFT_PRICE = 0.5 # Fireworks LoRA SFT of a model this size, USD per million training tokens def estimate(attempts, deployment=False): """{'usd', 'low', 'high', 'minutes': (low, high)}: the dry run's estimate for `attempts` attempts, on Fireworks serverless (the teacher) or on the student's H100 deployment, which bills from scaling up to its idle minutes. The time is the page's own: AT_ONCE attempts at a time, but never less than one attempt that runs into its task's time limit (the slowest attempt of a small eval sets its pace), plus START.""" box = [SANDBOX_HOUR * m / 60 for m in MINUTES] run = [attempts * m / AT_ONCE for m in MINUTES] wall = [max(MINUTES[0], run[0]) + START, max(LIMIT, run[1]) + START] if deployment: low = H100_HOUR * (run[0] + IDLE) / 60 + attempts * box[0] high = H100_HOUR * (WARM + run[1] + IDLE) / 60 + attempts * box[1] minutes = (wall[0], WARM + wall[1]) else: low, high = (attempts * ((p * TEACHER_PRICE[0] + c * TEACHER_PRICE[1]) / 1e6 + b) for (p, c), b in zip(TOKENS, box)) minutes = tuple(wall) return {'usd': round(math.sqrt(low * high), 2), 'low': round(low, 2), 'high': round(high, 2), 'minutes': [int(m + 0.5) for m in minutes]} # half up: 2.5 minutes shows as 3; a list, as JSON gives it def slug(text): return re.sub(r'[^a-z0-9]+', '-', str(text or '').lower()).strip('-') or 'collection' def repo_name(c): return str(c.get('repo_id') or '').rstrip('/').split('/')[-1] def env_name(c): """The environment's name in PostTrain: the collection's folder in its repository, else the repository's name.""" entry = str(c.get('entry_path') or '').strip('/') return slug(entry.split('/')[-1] if entry else repo_name(c)) def checkout(c): """Where the download goes: the repository's name and the commit's first 8 characters, so it can't land in an existing clone of the same repository (the starter kit clones benchflow-ai/posttrainarena as posttrainarena).""" return f"{slug(repo_name(c))}-{str(c.get('revision') or '')[:8]}" def folder(c): """The downloaded collection's folder: the one with submission.yaml and envs/.""" entry = str(c.get('entry_path') or '').strip('/') return checkout(c) + (f'/{entry}' if entry else '') def fetch(c): """The collection at the commit it was submitted at.""" rev, d = c.get('revision'), checkout(c) if c.get('repo_type') == 'github': return [f"git clone https://github.com/{c['repo_id']} {d}", f'git -C {d} checkout {rev}'] return [f"hf download {c['repo_id']} --repo-type dataset --revision {rev} --local-dir {d}"] def held_out(n): """How many of n packages the split holds out of training (every EVERY-th, from the first), or 0 when n is too few.""" return -(-n // EVERY) if n >= EVERY else 0 def commands(c, tasks=()): """Everything the page shows, as data: the commands per step and the facts its notes need. `tasks` are the collection's task rows with their static-check finding codes (store.static_tasks). None without a repository and commit to fetch.""" if not c.get('repo_id') or not c.get('revision'): return None name, src, n = env_name(c), folder(c), c.get('task_count') or len(tasks) # the tasks the static checks excluded (they leak their answer or overlap a sealed suite) are removed after the # download, so the eval, the split and its counts cover the rest excluded = sorted(t['name'] for t in tasks if t.get('outcome') == 'excluded' and t.get('name')) kept = n - len(excluded) # the kept tasks PostTrain's env add marks leaky: grading data the agent can read in its sandbox, the static checks' # L-GRADER-DATA-IN-IMAGE on a task they keep (gates-v3 and env add agree on every task). From 0.1.11 an eval leaves them # out unless --include-leaky, and --limit takes the first of the others by name (F12-01) kept_rows = sorted((x for x in tasks if x.get('outcome') != 'excluded'), key=lambda x: str(x.get('name') or '')) grading = lambda x: 'L-GRADER-DATA-IN-IMAGE' in (x.get('codes') or []) leaky = [x['name'] for x in kept_rows if grading(x) and x.get('name')] scored = kept - len(leaky) # the split holds out every EVERY-th package in the order the shell lists them (by name, from the first); the leaky ones # on each side are left out of the teacher's eval and of the held-out evals. With nothing to score on a side, no hill climb. held = held_out(kept) held_leaky = sum(1 for i, x in enumerate(kept_rows) if i % EVERY == 0 and x.get('name') in leaky) if held else 0 train_leaky = len(leaky) - held_leaky if held else 0 if held and (held_leaky == held or train_leaky == kept - held): held = 0 no_oracle = sum(1 for t in tasks if t.get('outcome') != 'excluded' and 'S-NO-ORACLE' in (t.get('codes') or [])) as_root = bool(kept) and no_oracle == kept # packaged the way BenchFlow's own tasks are: written for root root = f' {ROOT}' if as_root else '' # No command carries a `# comment`: pasted into zsh, which reads comments only with interactivecomments set (off by # default, and zsh is macOS's shell), the words after # would become the command's arguments. # the hill climb's from-rollouts --base renders the smaller model's chat template, with the data checks' libraries extras = ['posttrain extras install data'] if held else [] setup = {'install': ['npm install -g posttrain'] + extras + ["uv tool install --python 3.12 'benchflow[sandbox-daytona]'"] + ([] if c.get('repo_type') == 'github' else ['uv tool install hf']), 'server': ['posttrain server'], 'project': [f'mkdir -p {name} && cd {name}', 'posttrain login --url http://localhost:7880', f'posttrain init --org arena --name {name} --base {STUDENT_HF}', 'export FIREWORKS_API_KEY=... DAYTONA_API_KEY=...', 'posttrain compute add fw --kind fireworks', 'posttrain compute test fw', 'posttrain compute edit fw --set deploy_addons=true --set deploy_accelerator=NVIDIA_H100_80GB --set deploy_precision=BF16']} split = [f'mkdir -p {name}-train {name}-heldout', f'i=0; for t in {src}/envs/*/; do i=$((i+1)); if [ $((i % {EVERY})) = 1 ]; then cp -R "${{t%/}}" {name}-heldout/; ' f'else cp -R "${{t%/}}" {name}-train/; fi; done', f'posttrain env add {name}-heldout --name {name}-heldout', # the held-out tasks as the project's suite: PostTrain's overlap check then keeps their attempts out of the SFT rows, # and the training folder's env add, after it, checks that no training task repeats a held-out one (F11-15) f'posttrain suites set quick --bench env:{name}-heldout', f'posttrain env add {name}-train --name {name}-train'] climb = [f'posttrain eval {TEACHER} --bench env:{name}-train --on fw --set k=2 --set max_turns=30{root} --name teacher-{name} --yes', f'posttrain data from-rollouts teacher-{name} --name {name}-verified {ROWS} --base {STUDENT_HF}', f'posttrain eval {STUDENT} --bench env:{name}-heldout --on fw {ASK}{root} --yes', f'posttrain train sft --base {STUDENT} --data {name}-verified --on fw {SFT} --name sft-{name}-e1 --yes', f'posttrain train sft --base SFT_RUN --data {name}-verified --on fw {SFT} --name sft-{name}-e2 --yes', f'posttrain eval TUNED_RUN --bench env:{name}-heldout --on fw {ASK}{root} --yes', 'posttrain evals compare BASE_EVAL TUNED_EVAL'] remove = [f"rm -r {' '.join(f'{src}/envs/{x}' for x in excluded)}"] if excluded else [] first = min(FIRST, scored or kept) train = kept - held # the attempts each eval makes, as its dry run counts them: at the tasks it scores costs = {'first': estimate(first), 'eval': estimate(scored * 2)} if held: teacher, heldout = estimate((train - train_leaky) * 2), estimate((held - held_leaky) * 4, deployment=True) costs.update(teacher=teacher, heldout=heldout, sft_price=SFT_PRICE, total={k: round(teacher[k] + 2 * heldout[k], 2) for k in ('usd', 'low', 'high')}, total_minutes=[teacher['minutes'][0] + 2 * heldout['minutes'][0], teacher['minutes'][1] + 2 * heldout['minutes'][1]]) # what the static checks found in the kept tasks' sandboxes (F10-03), counted the way PostTrain's env add counts them (0.1.11): # grading data an attempt can read (env add marks the task leaky), and answer-like files in the other tasks, which hold # nothing their verifier checks (a note, not a leak; one that held it would be grading data, which excludes its task) readable = {'grading': sum(grading(x) for x in kept_rows), 'answers': sum('L-ANSWER-FILE' in (x.get('codes') or []) and not grading(x) for x in kept_rows)} # the collection's own notes may say it is for training only (TMax's: "training-only, not for held-out evaluation") training_only = bool(re.search(r'training[- ]only|not for held[- ]out', str(c.get('description') or ''), re.I)) return {'guide': GUIDE, 'compare': GUIDE + '#train-on-them-and-compare', 'quickstart': QUICKSTART, 'spec': SPEC, 'posttrain': POSTTRAIN, 'env': name, 'folder': src, 'teacher': TEACHER, 'student': STUDENT, 'tasks': n, 'excluded': len(excluded), 'kept': kept, 'no_oracle': no_oracle, 'root': as_root, 'setup': setup, 'add': fetch(c) + remove + [f'posttrain env add {src} --name {name}'], 'eval': [f'posttrain eval {TEACHER} --bench env:{name} --on fw --limit {first}{root} --yes', f'posttrain eval {TEACHER} --bench env:{name} --on fw --set k=2{root} --yes'], 'first': first, 'costs': costs, 'training_only': training_only, 'readable': readable, 'leaky': leaky, 'scored': scored, 'split': {'train': kept - held, 'heldout': held, 'train_leaky': train_leaky, 'heldout_leaky': held_leaky} if held else None, 'hillclimb': split + climb if held else []}