Spaces:
Running
Running
File size: 143,509 Bytes
7c71ad3 5d25978 589e433 3f790fd 7c71ad3 3f790fd 281dd89 bbd9760 77d8d06 2b7cca9 77d8d06 bbd9760 2b7cca9 77d8d06 bbd9760 310222a b0d404c 310222a 7c71ad3 310222a e64e5c7 310222a 3295f27 310222a 3295f27 310222a 3295f27 310222a d3ab0f1 310222a 3295f27 310222a e44be29 310222a 3295f27 310222a 3295f27 310222a 9772595 3295f27 1846e22 310222a 3295f27 310222a e64e5c7 3295f27 310222a c7c6db0 310222a 30b9d92 310222a 3295f27 310222a 3295f27 310222a 3295f27 310222a 1846e22 310222a 3295f27 310222a 3295f27 310222a 3295f27 310222a fd5a2c2 3f790fd 310222a fa99013 d3ab0f1 310222a 77d8d06 310222a 3a438f1 b0d404c 3a438f1 7d12fca 30b9d92 88195b9 3a438f1 88195b9 3a438f1 5dd94e9 fd8fa6b 7f0c678 fd8fa6b 7f0c678 30b9d92 5dd94e9 3a438f1 7c71ad3 30b9d92 18a6a08 fd8fa6b 742eec9 5d25978 3a438f1 5dd94e9 1d1adb0 3a438f1 15bbfe4 30b9d92 1d1adb0 7f0c678 3a438f1 30b9d92 3a438f1 3f790fd 3a438f1 ec24d42 77d8d06 ec24d42 30b9d92 77d8d06 ce59b36 1d1adb0 fd8fa6b 7f0c678 fd8fa6b 7f0c678 ce59b36 30b9d92 1d1adb0 30b9d92 77d8d06 30b9d92 3a438f1 63e0c56 fd8fa6b 30b9d92 fd8fa6b 4dca4c4 88195b9 310222a 3a438f1 7d12fca 310222a fa99013 5dd94e9 310222a 3a438f1 7d12fca 3a438f1 7d12fca 3a438f1 15bbfe4 7d12fca 30b9d92 63e0c56 30b9d92 3295f27 77d8d06 1846e22 1424942 30b9d92 1d1adb0 30b9d92 77d8d06 30b9d92 3295f27 30b9d92 1d1adb0 30b9d92 3a438f1 30b9d92 ce59b36 30b9d92 5b96664 77d8d06 30b9d92 43c863d 5bc2eea 1d1adb0 30b9d92 1d1adb0 30b9d92 1d1adb0 30b9d92 77d8d06 30b9d92 1b4bef2 48142c1 30b9d92 77d8d06 48142c1 1d1adb0 77d8d06 30b9d92 1d1adb0 30b9d92 77d8d06 30b9d92 77d8d06 1d1adb0 77d8d06 1d1adb0 30b9d92 1d1adb0 30b9d92 d06882d 30b9d92 7d12fca 30b9d92 88195b9 30b9d92 7d12fca 30b9d92 63e0c56 30b9d92 77d8d06 30b9d92 d06882d 30b9d92 88195b9 30b9d92 88195b9 30b9d92 ce4ad00 d06882d 30b9d92 88195b9 30b9d92 1d1adb0 30b9d92 ce59b36 30b9d92 ec24d42 30b9d92 1d1adb0 77d8d06 30b9d92 47a84b1 30b9d92 47a84b1 30b9d92 1d1adb0 30b9d92 ce59b36 30b9d92 47a84b1 d06882d 310222a d06882d 47a84b1 d06882d ce59b36 30b9d92 d06882d 47a84b1 d06882d 47a84b1 d06882d 77d8d06 30b9d92 77d8d06 d06882d 3295f27 30b9d92 d06882d 30b9d92 47a84b1 30b9d92 48142c1 30b9d92 48142c1 77d8d06 30b9d92 77d8d06 30b9d92 77d8d06 47a84b1 48142c1 d06882d 48142c1 77d8d06 48142c1 77d8d06 48142c1 1d1adb0 48142c1 515a2d6 30b9d92 63e0c56 30b9d92 5696a21 47a84b1 30b9d92 5696a21 47a84b1 30b9d92 63e0c56 4300e1c 63e0c56 4300e1c 63e0c56 4300e1c 63e0c56 30b9d92 63e0c56 30b9d92 63e0c56 c6e9811 30b9d92 63e0c56 c7c6db0 63e0c56 30b9d92 63e0c56 77d8d06 63e0c56 30b9d92 63e0c56 a136c53 e60e093 63e0c56 c7c6db0 63e0c56 c7c6db0 63e0c56 c7c6db0 63e0c56 4300e1c 63e0c56 c7c6db0 63e0c56 c7c6db0 63e0c56 c7c6db0 63e0c56 9af2a67 2146f9c c7c6db0 7be229e f718bf5 7be229e 63e0c56 c7c6db0 f718bf5 c7c6db0 6a7b4af 742eec9 6a7b4af 742eec9 638569e db2ed0d 638569e f718bf5 638569e 6a7b4af 63e0c56 6a7b4af 63e0c56 742eec9 c7c6db0 f718bf5 6a7b4af f718bf5 ff1df36 7be229e 059dba7 f718bf5 742eec9 63e0c56 30b9d92 63e0c56 30b9d92 63e0c56 7f0c678 63e0c56 7f0c678 30b9d92 e60e093 63e0c56 77d8d06 63e0c56 5696a21 30b9d92 63e0c56 30b9d92 47a84b1 5696a21 30b9d92 5696a21 30b9d92 5696a21 30b9d92 63e0c56 30b9d92 63e0c56 5696a21 30b9d92 5696a21 48142c1 5b96664 5696a21 48142c1 5b96664 48142c1 30b9d92 77d8d06 834a04e ecb54e8 834a04e 70cd5c4 834a04e c507d59 834a04e c507d59 5696a21 c507d59 63e0c56 48142c1 d002a77 48142c1 30b9d92 7d12fca 88195b9 30b9d92 5696a21 48142c1 5e82102 5b96664 48142c1 3a438f1 48142c1 1d1adb0 3a438f1 30b9d92 5696a21 48142c1 5b96664 48142c1 3a438f1 fd8fa6b 7f0c678 fd8fa6b 63e0c56 1d1adb0 5dd94e9 7f0c678 30b9d92 5dd94e9 3a438f1 30b9d92 fd8fa6b 30b9d92 63e0c56 4300e1c 63e0c56 30b9d92 fd8fa6b 63e0c56 30b9d92 fd8fa6b 30b9d92 7d12fca 88195b9 fd8fa6b 3455420 88195b9 7d12fca 1b4bef2 7f0c678 f76773c 3f790fd | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629 630 631 632 633 634 635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 675 676 677 678 679 680 681 682 683 684 685 686 687 688 689 690 691 692 693 694 695 696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712 713 714 715 716 717 718 719 720 721 722 723 724 725 726 727 728 729 730 731 732 733 734 735 736 737 738 739 740 741 742 743 744 745 746 747 748 749 750 751 752 753 754 755 756 757 758 759 760 761 762 763 764 765 766 767 768 769 770 771 772 773 774 775 776 777 778 779 780 781 782 783 784 785 786 787 788 789 790 791 792 793 794 795 796 797 798 799 800 801 802 803 804 805 806 807 808 809 810 811 812 813 814 815 816 817 818 819 820 821 822 823 824 825 826 827 828 829 830 831 832 833 834 835 836 837 838 839 840 841 842 843 844 845 846 847 848 849 850 851 852 853 854 855 856 857 858 859 860 861 862 863 864 865 866 867 868 869 870 871 872 873 874 875 876 877 878 879 880 881 882 883 884 885 886 887 888 889 890 891 892 893 894 895 896 897 898 899 900 901 902 903 904 905 906 907 908 909 910 911 912 913 914 915 916 917 918 919 920 921 922 923 924 925 926 927 928 929 930 931 932 933 934 935 936 937 938 939 940 941 942 943 944 945 946 947 948 949 950 951 952 953 954 955 956 957 958 959 960 961 962 963 964 965 966 967 968 969 970 971 972 973 974 975 976 977 978 979 980 981 982 983 984 985 986 987 988 989 990 991 992 993 994 995 996 997 998 999 1000 1001 1002 1003 1004 1005 1006 1007 1008 1009 1010 1011 1012 1013 1014 1015 1016 1017 1018 1019 1020 1021 1022 1023 1024 1025 1026 1027 1028 1029 1030 1031 1032 1033 1034 1035 1036 1037 1038 1039 1040 1041 1042 1043 1044 1045 1046 1047 1048 1049 1050 1051 1052 1053 1054 1055 1056 1057 1058 1059 1060 1061 1062 1063 1064 1065 1066 1067 | <!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>PostTrain Arena</title>
<link rel="icon" href="/icon.svg" type="image/svg+xml">
<!-- Link previews (Slack, X, Discord): static, because crawlers do not run the page's script. The card is og-card.jpg,
loaded from the Hub (huggingface.co answered every fetch, while this Space's proxy sometimes answers 502 with its own page). -->
<meta name="description" content="Submit a collection of RL task environments to PostTrain Arena. A challenge fixes the model, the training recipe and a held-out suite; a run scores a collection by the held-out pass rate after training minus before.">
<meta property="og:type" content="website">
<meta property="og:site_name" content="PostTrain Arena">
<meta property="og:url" content="https://benchflow-posttrain-arena.hf.space/arena">
<meta property="og:title" content="PostTrain Arena Β· Challenges and submissions">
<meta property="og:description" content="Submit a collection of RL task environments to PostTrain Arena. A challenge fixes the model, the training recipe and a held-out suite; a run scores a collection by the held-out pass rate after training minus before.">
<meta property="og:image" content="https://huggingface.co/spaces/benchflow/posttrain-arena/resolve/main/og-card.jpg">
<meta property="og:image:width" content="1200">
<meta property="og:image:height" content="630">
<meta property="og:image:alt" content="PostTrain Arena: the arena on Hugging Face. Submit RL environment collections; a run scores the held-out change.">
<meta name="twitter:card" content="summary_large_image">
<meta name="twitter:title" content="PostTrain Arena Β· Challenges and submissions">
<meta name="twitter:description" content="Submit a collection of RL task environments to PostTrain Arena. A challenge fixes the model, the training recipe and a held-out suite; a run scores a collection by the held-out pass rate after training minus before.">
<meta name="twitter:image" content="https://huggingface.co/spaces/benchflow/posttrain-arena/resolve/main/og-card.jpg">
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link href="https://fonts.googleapis.com/css2?family=JetBrains+Mono:wght@300;400;500;600&family=Inter:wght@300;400;500;600&display=swap" rel="stylesheet">
<style>
/* The board's design (board.html at /), so the app reads as the same site: its tokens, its two faces (Inter for text,
JetBrains Mono for titles, counts, labels and tables), square hairline panels, one deep blue for links and actions. */
* { box-sizing: border-box; }
:root {
--bg: #fafafa; --bg-soft: #f4f4f4; --bg-card: #ffffff; --border: #ddd; --border-soft: #eee;
--ink: #1a1a1a; --ink-2: #2a2a2a; --ink-3: #444; --muted: #555; --muted-2: #777; --muted-3: #888; --muted-4: #999; --muted-5: #aaa;
--accent: #0f3787; --accent-deep: #0a275f; --accent-soft: #dde6f5; --accent-hover-row: #c8d6ee;
/* states in the board's palette: its live dot (green), its paused pill (amber) and its error text (red) */
--ok: #15803d; --warn: #8a5a00; --warn-line: #d9b36c; --bad: #b91c1c;
--mono: "JetBrains Mono", ui-monospace, SFMono-Regular, Menlo, monospace;
}
body { margin: 0; font-family: "Inter", "Helvetica Neue", sans-serif; font-size: 12px; font-weight: 300; line-height: 1.6; color: var(--ink); background: var(--bg); padding: 24px 32px 64px; overflow-x: hidden; }
a { color: var(--accent); text-decoration: underline; text-decoration-color: rgba(15, 55, 135, .3); text-underline-offset: 2px; }
a:hover { color: var(--accent-deep); text-decoration-color: currentColor; }
b, strong { font-weight: 500; } p { margin: 6px 0; } ul, ol { margin: 6px 0; padding-left: 22px; } li { margin: 3px 0; }
code, .mono { font-family: var(--mono); font-size: 11px; } code { background: var(--bg-soft); padding: 0 4px; border-radius: 2px; color: var(--ink-3); }
.muted { color: var(--muted-2); } .small { font-size: 11px; } .nw { white-space: nowrap; } .detail { color: var(--muted); }
main p, main li, main dl, .lede { max-width: 980px; }
/* --- Header: the board's header row --- */
/* a grid, so the tagline runs under both columns as it does on the board, whose toolbar is shorter */
.header-row { display: grid; grid-template-columns: minmax(0, 1fr) auto; column-gap: 24px; align-items: start; margin-bottom: 16px; padding-bottom: 12px; border-bottom: 1px solid var(--border); }
.header-row > .subtitle { grid-column: 1 / -1; grid-row: 2; }
.header-row > .header-side { grid-column: 2; grid-row: 1 / span 2; } /* spans the tagline's row, so the tagline follows the counts line as on the board */
.title-block { min-width: 0; }
.title-row { display: flex; align-items: center; gap: 18px; flex-wrap: wrap; }
.brand { font-family: var(--mono); font-size: 24px; font-weight: 500; letter-spacing: 0.2px; color: var(--ink); line-height: 1.25; text-decoration: none; }
.brand:hover { color: var(--ink); text-decoration: none; }
.subtext { font-family: var(--mono); font-size: 11px; font-weight: 500; color: var(--muted-2); letter-spacing: 0.4px; margin-top: 10px; font-variant-numeric: tabular-nums; }
.subtext .sep { color: var(--muted-4); margin: 0 10px; font-weight: 400; } .subtext .n { color: var(--accent); font-weight: 600; }
.subtitle { font-size: 13px; font-weight: 300; line-height: 1.55; color: var(--muted); margin-top: 8px; max-width: 880px; }
.header-side { display: flex; flex-direction: column; align-items: flex-end; gap: 15px; }
.toolbar, .toolbar nav, .toolbar .right { display: flex; align-items: center; gap: 8px; flex-wrap: wrap; }
.toolbar .right::before { content: ''; width: 1px; height: 16px; background: var(--border); } /* the app's pages, then the board, sign-in and the data source */
.toolbar .right > span { font-family: var(--mono); font-size: 10px; letter-spacing: 0.5px; color: var(--muted-3); white-space: nowrap; }
.toolbar .right > span b { color: var(--ink); font-weight: 500; }
.toolbar .right > span.muted { display: inline-flex; align-items: center; min-height: 26px; line-height: 1.4; padding: 5px 11px; border: 1px solid var(--border); border-radius: 3px; background: #fff; color: var(--muted-4); cursor: default; }
/* the board's outlined mono button; the page you are on is filled, as the board's .btn.active */
.btn, .toolbar a, .toolbar button, button.plain { font-family: var(--mono); font-size: 10px; font-weight: 400; letter-spacing: 0.5px; padding: 5px 11px; border: 1px solid #ccc; border-radius: 3px;
background: #fff; color: var(--muted); cursor: pointer; transition: all 0.15s; text-decoration: none; display: inline-flex; align-items: center; line-height: 1.4; min-height: 26px; white-space: nowrap; }
.btn:hover, .toolbar a:hover, .toolbar button:hover, button.plain:hover { border-color: var(--muted-3); color: var(--ink); text-decoration: none; }
.btn.active, .toolbar nav a.on, button.plain.on { background: var(--ink); color: #fff; border-color: var(--ink); }
.btn-primary { font-family: var(--mono); font-size: 11px; font-weight: 500; letter-spacing: 0.8px; text-transform: uppercase; padding: 7px 14px; border: 1px solid var(--accent); background: var(--accent); color: #fff;
cursor: pointer; border-radius: 3px; transition: all 0.15s; display: inline-flex; align-items: center; text-decoration: none; flex: 0 0 auto; line-height: normal; }
.btn-primary:hover { background: var(--accent-deep); border-color: var(--accent-deep); color: #fff; text-decoration: none; }
.btn-primary:focus-visible { outline: 2px solid var(--accent-soft); outline-offset: 1px; }
.btn-primary .plus { font-weight: 700; font-size: 15px; margin-right: 6px; line-height: 1; display: inline-block; vertical-align: -1px; }
/* OpenEnv's mark under the toolbar, as on the board */
.openenv { display: inline-flex; align-items: center; gap: 7px; font-family: var(--mono); font-size: 11px; font-weight: 500; letter-spacing: 0.4px; color: var(--muted-2); text-decoration: none; line-height: 1.6; }
.openenv img { width: 16px; height: 16px; object-fit: contain; flex: none; }
.openenv span { border-bottom: 1px solid var(--border); }
.openenv:hover { color: var(--ink); } .openenv:hover span { border-bottom-color: var(--muted-3); }
/* at the board's width for it (1270 px), the toolbar goes under the tagline and the mark lines up under its first button */
@media (max-width: 1270px) { .header-row { grid-template-columns: minmax(0, 1fr); } .header-row > .header-side { grid-column: 1; grid-row: 3; align-items: flex-start; gap: 12px; margin-top: 24px; } }
/* --- buttons, fields and filters --- */
select, input, textarea, button { font: inherit; color: inherit; }
button { font-family: var(--mono); font-size: 11px; font-weight: 500; letter-spacing: 1px; text-transform: uppercase; padding: 5px 14px; min-height: 26px; border: 1px solid var(--accent); border-radius: 2px;
background: var(--accent); color: #fff; cursor: pointer; transition: background 0.15s, border-color 0.15s, color 0.15s; }
button:not(.plain):hover:not(:disabled) { background: var(--accent-deep); border-color: var(--accent-deep); } /* the outlined (.plain) buttons keep the board's outlined hover */
button:disabled { background: #fff; color: var(--muted-4); border-color: var(--border); cursor: not-allowed; }
button.plain { text-transform: none; }
button.plain:disabled { background: #fff; color: var(--muted-4); border-color: var(--border); }
select, input, textarea { font-family: var(--mono); font-size: 11px; font-weight: 400; border: 1px solid var(--border); border-radius: 2px; padding: 4px 9px; min-height: 26px; background: #fff; color: var(--ink); max-width: 100%; }
select { padding-right: 4px; cursor: pointer; } input::placeholder, textarea::placeholder { color: var(--muted-4); font-weight: 300; }
select:focus, input:focus, textarea:focus { outline: none; border-color: var(--accent); box-shadow: 0 0 0 3px rgba(15, 55, 135, 0.10); }
input[type=radio] { min-height: 0; accent-color: var(--accent); }
.filters { display: flex; flex-wrap: wrap; gap: 8px; align-items: center; margin: 10px 0; }
.filters input[type=search] { width: 320px; }
/* --- page parts --- */
h1 { font-family: var(--mono); font-size: 18px; font-weight: 500; letter-spacing: 0.2px; line-height: 1.3; margin: 8px 0 6px; color: var(--ink); }
/* section heads: the board's CHALLENGES and MESSAGES */
h2 { font-family: var(--mono); font-size: 11px; font-weight: 400; text-transform: uppercase; letter-spacing: 2px; color: var(--ink-3); margin: 28px 0 10px; padding-bottom: 6px; border-bottom: 1px solid var(--border); line-height: 1.6; }
h2[id] { scroll-margin-top: var(--head, 72px); }
#main > div > h1:first-child { font-size: 11px; font-weight: 400; text-transform: uppercase; letter-spacing: 2px; color: var(--ink-3); margin: 0 0 10px; padding-bottom: 6px; border-bottom: 1px solid var(--border); line-height: 1.6; }
h1, .titlerow h2 { text-wrap: balance; }
.lede { font-size: 13px; font-weight: 300; line-height: 1.55; color: var(--muted); }
/* loading, a page that could not load, and an empty table: the board's state panel (centered, mono, muted) */
#main > p:only-child { font-family: var(--mono); font-size: 11px; line-height: 1.7; color: var(--muted-2); text-align: center; padding: 32px 16px; margin: 0; max-width: none; background: #fff; border: 1px solid var(--border); }
td.muted[colspan] { font-family: var(--mono); font-size: 11px; color: var(--muted-3); text-align: center; padding: 24px 16px; }
.note { margin: -4px 0 16px; } .note div { border: 1px solid var(--border); border-left: 2px solid var(--warn-line); background: #fff; padding: 8px 12px; font-size: 12px; color: var(--ink); }
.note button.plain { float: right; margin-left: 12px; min-height: 22px; padding: 1px 9px; }
/* square hairline panels, as the board's challenge strip; a warning or a failure keeps a coloured spine */
.box { background: #fff; border: 1px solid var(--border); padding: 10px 14px; margin: 12px 0; }
.box.warn { border-left: 2px solid var(--warn-line); } .box.bad, .card-bad { border-left: 2px solid var(--bad); }
.box .org { border-top: 1px solid var(--border-soft); margin-top: 10px; padding-top: 9px; color: var(--muted); }
.card-bad { background: #fff; border: 1px solid var(--border); border-left: 2px solid var(--bad); padding: 12px 16px; margin: 14px 0; }
.card-bad h2, .box h2 { font-family: "Inter", sans-serif; font-size: 13px; font-weight: 500; text-transform: none; letter-spacing: 0; color: var(--ink); border: 0; padding: 0; margin: 0 0 4px; }
.state { font-weight: 500; } .s-running, .s-training, .s-queued { color: var(--accent); } .s-verified, .s-ok, .s-accepting { color: var(--ok); } .s-review { color: var(--warn); }
.s-failed, .s-rejected, .s-excluded, .s-closed { color: var(--bad); } .s-canceled, .s-none { color: var(--muted-3); }
/* tables: the board's leaderboard */
.scroll { overflow-x: auto; }
table { font-family: var(--mono); width: 100%; border-collapse: collapse; font-size: 11px; font-weight: 300; background: #fff; border: 1px solid var(--border); }
th, td { text-align: left; padding: 8px 12px; vertical-align: top; font-variant-numeric: tabular-nums; }
th { font-size: 10px; font-weight: 500; text-transform: uppercase; letter-spacing: 1px; color: var(--muted-2); border-bottom: 1px solid var(--border); background: var(--bg-soft); vertical-align: bottom; white-space: nowrap; }
td.r { white-space: nowrap; }
tr + tr td { border-top: 1px solid var(--border-soft); }
td.r, th.r { text-align: right; } td.zero { color: var(--muted-3); } td .mono { overflow-wrap: anywhere; }
td a { text-decoration: none; } td a:hover { text-decoration: underline; }
tr[data-href] { cursor: pointer; } tr[data-href]:hover td { background: #fafafa; } tr[data-href]:focus-visible td { background: var(--accent-soft); outline: none; }
.reason { display: block; font-family: "Inter", sans-serif; font-size: 11.5px; color: var(--muted-3); margin-top: 1px; }
td .small { font-family: "Inter", sans-serif; font-size: 11.5px; }
details summary { cursor: pointer; list-style: none; font-family: var(--mono); font-size: 11px; letter-spacing: 0.4px; color: var(--ink-3); }
details summary::-webkit-details-marker { display: none; } details summary::before { content: '\25B8 '; color: var(--muted-3); } details[open] > summary::before { content: '\25BE '; }
dl { display: grid; grid-template-columns: 220px 1fr; gap: 6px 18px; margin: 6px 0; }
dt { font-family: var(--mono); font-size: 10px; font-weight: 500; text-transform: uppercase; letter-spacing: 1px; color: var(--muted-2); padding-top: 2px; }
dd { margin: 0; overflow-wrap: anywhere; }
pre { font: 11px/1.6 var(--mono); background: var(--bg-soft); border: 1px solid var(--border); padding: 10px 14px; color: var(--ink-3); white-space: pre-wrap; overflow-wrap: anywhere; overflow-x: auto; margin: 6px 0; }
pre.cmd { white-space: pre; overflow-wrap: normal; } /* commands keep their lines: a wrap after a hyphen reads as a different flag; they scroll inside the block */
.cmdbox { margin: 6px 0 10px; } .cmdbox .cmdbar { display: flex; justify-content: flex-end; margin: 0 0 3px; } .cmdbox .copy { min-height: 22px; padding: 1px 9px; } .steps .cmdbox pre, .cmdbox pre { margin: 0; } /* Copy takes the whole text, however much of a line is scrolled out of view */
/* a challenge's frame: where you are, its state, its tabs */
.frame { margin: 0 0 20px; } .crumb { font-family: var(--mono); font-size: 11px; letter-spacing: 0.4px; color: var(--muted-3); margin: 0 0 4px; } .crumb a { color: var(--muted-2); text-decoration: none; } .crumb a:hover { color: var(--accent); text-decoration: underline; }
.status-line { display: flex; flex-wrap: wrap; align-items: center; font-family: var(--mono); font-size: 11px; font-weight: 400; letter-spacing: 0.4px; color: var(--muted-2); row-gap: 4px; }
.status-line > span + span::before { content: '|'; margin: 0 10px; color: var(--muted-4); font-weight: 400; }
.status-line a { text-decoration: none; }
/* a challenge's state is the board's pill: blue with a dot while it takes runs or runs one, amber when paused */
.state.pill { display: inline-block; font-size: 10px; font-weight: 400; letter-spacing: 0.5px; text-transform: uppercase; padding: 1px 8px; border-radius: 999px; border: 1px solid var(--border); color: var(--muted-2); line-height: 1.4; white-space: nowrap; }
.state.pill.s-ok, .state.pill.s-running { border-color: var(--accent); color: var(--accent); }
.state.pill.s-ok::before, .state.pill.s-running::before { content: '\25CF'; margin: 0 5px 0 0; color: inherit; }
.state.pill.s-review { border-color: var(--warn-line); color: var(--warn); }
.status-line > .pill::before { content: none; } /* its separators sit outside the pill */
.status-line > span:has(+ .pill)::after { content: '|'; margin: 0 10px; color: var(--muted-4); font-weight: 400; }
td .pill + .reason { margin-top: 4px; }
.tabs { display: flex; gap: 22px; border-bottom: 1px solid var(--border); margin-top: 14px; overflow-x: auto; }
.tabs a { font-family: var(--mono); font-size: 11px; font-weight: 400; text-transform: uppercase; letter-spacing: 1.5px; color: var(--muted-3); text-decoration: none; padding: 7px 0 8px; border-bottom: 2px solid transparent; margin-bottom: -1px; white-space: nowrap; }
.tabs a:hover { color: var(--ink); text-decoration: none; } .tabs a.on { color: var(--ink); border-bottom-color: var(--accent); font-weight: 500; }
.frame + p.small { font-family: var(--mono); font-size: 11px; letter-spacing: 0.4px; margin: 0 0 4px; } .frame + p.small a { color: var(--muted-2); text-decoration: none; } .frame + p.small a:hover { color: var(--accent); text-decoration: underline; }
.titlerow { display: flex; flex-wrap: wrap; gap: 4px 18px; align-items: baseline; justify-content: space-between; margin: 4px 0 2px; }
.titlerow h2 { font-size: 16px; font-weight: 500; text-transform: none; letter-spacing: 0.2px; color: var(--ink); border: 0; padding: 0; margin: 0; }
/* the rules at a glance, as the board's hover card: mono labels, a value, one line of detail */
.layout { display: grid; grid-template-columns: minmax(0, 1fr) 290px; gap: 32px; }
.facts { background: #fff; border: 1px solid var(--border); padding: 4px 16px; align-self: start; }
.facts > div { padding: 9px 0; border-bottom: 1px solid var(--border-soft); } .facts > div:last-of-type { border: 0; }
.facts .k { font-family: var(--mono); font-size: 10px; font-weight: 500; letter-spacing: 1px; text-transform: uppercase; color: var(--muted-2); }
.facts .v { font-family: var(--mono); font-size: 12px; font-weight: 500; margin-top: 1px; } .facts .d { font-size: 11.5px; color: var(--muted); }
/* numbered steps, as the board's Add your agent steps */
ol.steps { list-style: none; counter-reset: step; padding-left: 0; }
ol.steps > li { counter-increment: step; position: relative; padding-left: 42px; margin: 18px 0; max-width: none; }
ol.steps > li::before { content: counter(step); position: absolute; left: 0; top: 0; width: 24px; height: 24px; border-radius: 50%; background: var(--accent); color: #fff; font: 600 11px/24px var(--mono); text-align: center; }
ol.steps > li > b:first-child { display: block; font-family: var(--mono); font-size: 10px; font-weight: 500; letter-spacing: 1.2px; text-transform: uppercase; color: var(--ink-3); padding-top: 5px; margin-bottom: 6px; }
ol.steps pre { margin: 6px 0; }
.help { font-family: var(--mono); font-size: 10px; letter-spacing: 0.3px; color: var(--muted-3); margin: 5px 0 4px; }
.actions { display: flex; gap: 8px; justify-content: flex-end; align-items: center; flex-wrap: wrap; margin: 14px 0; max-width: 644px; }
/* the submit form: the board's create-channel fields */
.form { display: grid; grid-template-columns: 150px minmax(0, 480px); gap: 2px 14px; align-items: start; margin: 12px 0; }
.form label { font-family: var(--mono); font-size: 10px; font-weight: 500; letter-spacing: 1.2px; text-transform: uppercase; color: var(--muted-2); padding-top: 8px; }
.form .field { margin-bottom: 10px; } .form .fixed { padding-top: 6px; }
.form input, .form select, .form textarea { width: 100%; font-size: 12px; padding: 6px 10px; min-height: 32px; }
.form textarea { font-family: "Inter", sans-serif; font-weight: 300; line-height: 1.5; min-height: 72px; resize: vertical; }
/* the data-source dialog: the board's modal */
dialog { border: 1px solid var(--border); padding: 24px; max-width: 560px; width: calc(100% - 40px); color: var(--ink); }
dialog::backdrop { background: rgba(0, 0, 0, 0.4); }
dialog h2 { font-size: 13px; letter-spacing: 1.5px; margin: 0 0 14px; padding-bottom: 8px; }
dialog label { cursor: pointer; } dialog details { margin: 6px 0; font-size: 11px; }
/* the footer: the agents' links in the counts line's type */
footer { margin-top: 48px; font-family: var(--mono); font-size: 11px; letter-spacing: 0.4px; color: var(--muted-2); }
footer > div { border-top: 1px solid var(--border); padding-top: 12px; }
footer a { color: var(--muted); text-decoration: none; border-bottom: 1px solid var(--border); } footer a:hover { color: var(--ink); border-bottom-color: var(--muted-3); }
@media (max-width: 700px) { table.wide { min-width: 760px; } .sideways::before { content: 'Scroll sideways for more columns β'; display: block; position: sticky; left: 0; font-family: "Inter", sans-serif; font-size: 11px; color: var(--muted-3); margin: 0 0 4px; }
table.wide td:first-child { min-width: 150px; } .layout { grid-template-columns: 1fr; } dl, .form { grid-template-columns: 1fr; } dl { gap: 2px 0; } dd { margin-bottom: 6px; } .form label { padding-top: 0; } .hide-s { display: none; } .tabs { gap: 18px; } }
/* a phone: the board's phone sizes, and a compact header (the tagline is left to wider screens) */
@media (max-width: 640px) {
body { padding: 16px 14px 48px; }
.header-row > .header-side { margin-top: 12px; } .title-row { gap: 10px; } .brand { font-size: 18px; letter-spacing: 0.1px; }
.subtext { font-size: 10px; letter-spacing: 0.3px; margin-top: 8px; } .subtext .sep { margin: 0 6px; } .subtitle { display: none; }
.btn-primary { font-size: 10px; padding: 6px 10px; } .btn-primary .plus { font-size: 13px; margin-right: 4px; }
.toolbar { flex-direction: column; align-items: flex-start; gap: 8px; } .toolbar .right::before { content: none; }
.toolbar a, .toolbar button, .toolbar .right > span.muted { padding: 5px 8px; } .openenv { font-size: 10px; letter-spacing: 0.3px; }
.status-line { column-gap: 12px; } .status-line > span + span::before { content: none; margin: 0; } .status-line > span:has(+ .pill)::after { content: none; margin: 0; } /* two rules: a browser without :has() drops only the second */
.filters > input { flex: 1 1 100%; width: auto; } .filters > select { flex: 1 1 160px; }
}
</style>
</head>
<body>
<header>
<div class="header-row">
<div class="title-block">
<div class="title-row">
<a class="brand" href="#/">PostTrain Arena</a>
<a class="btn-primary" href="#/submit"><span class="plus" aria-hidden="true">+</span>Submit a collection</a>
</div>
<div class="subtext" id="counts" hidden></div>
</div>
<div class="header-side">
<div class="toolbar"><nav id="nav" aria-label="Arena"></nav><span class="right" id="who"></span></div>
<a class="openenv" href="https://github.com/huggingface/OpenEnv" target="_blank" rel="noopener" aria-labelledby="openenvWords">
<img alt="Hugging Face" width="16" height="16" src="data:image/svg+xml;base64,PHN2ZyB4bWxucz0iaHR0cDovL3d3dy53My5vcmcvMjAwMC9zdmciIHdpZHRoPSI5NSIgaGVpZ2h0PSI4OCIgZmlsbD0ibm9uZSI+Cgk8cGF0aCBmaWxsPSIjRkZEMjFFIiBkPSJNNDcuMjEgNzYuNWEzNC43NSAzNC43NSAwIDEgMCAwLTY5LjUgMzQuNzUgMzQuNzUgMCAwIDAgMCA2OS41WiIgLz4KCTxwYXRoCgkJZmlsbD0iI0ZGOUQwQiIKCQlkPSJNODEuOTYgNDEuNzVhMzQuNzUgMzQuNzUgMCAxIDAtNjkuNSAwIDM0Ljc1IDM0Ljc1IDAgMCAwIDY5LjUgMFptLTczLjUgMGEzOC43NSAzOC43NSAwIDEgMSA3Ny41IDAgMzguNzUgMzguNzUgMCAwIDEtNzcuNSAwWiIKCS8+Cgk8cGF0aAoJCWZpbGw9IiMzQTNCNDUiCgkJZD0iTTU4LjUgMzIuM2MxLjI4LjQ0IDEuNzggMy4wNiAzLjA3IDIuMzhhNSA1IDAgMSAwLTYuNzYtMi4wN2MuNjEgMS4xNSAyLjU1LS43MiAzLjctLjMyWk0zNC45NSAzMi4zYy0xLjI4LjQ0LTEuNzkgMy4wNi0zLjA3IDIuMzhhNSA1IDAgMSAxIDYuNzYtMi4wN2MtLjYxIDEuMTUtMi41Ni0uNzItMy43LS4zMloiCgkvPgoJPHBhdGgKCQlmaWxsPSIjRkYzMjNEIgoJCWQ9Ik00Ni45NiA1Ni4yOWM5LjgzIDAgMTMtOC43NiAxMy0xMy4yNiAwLTIuMzQtMS41Ny0xLjYtNC4wOS0uMzYtMi4zMyAxLjE1LTUuNDYgMi43NC04LjkgMi43NC03LjE5IDAtMTMtNi44OC0xMy0yLjM4czMuMTYgMTMuMjYgMTMgMTMuMjZaIgoJLz4KCTxwYXRoCgkJZmlsbD0iIzNBM0I0NSIKCQlmaWxsLXJ1bGU9ImV2ZW5vZGQiCgkJZD0iTTM5LjQzIDU0YTguNyA4LjcgMCAwIDEgNS4zLTQuNDljLjQtLjEyLjgxLjU3IDEuMjQgMS4yOC40LjY4LjgyIDEuMzcgMS4yNCAxLjM3LjQ1IDAgLjktLjY4IDEuMzMtMS4zNS40NS0uNy44OS0xLjM4IDEuMzItMS4yNWE4LjYxIDguNjEgMCAwIDEgNSA0LjE3YzMuNzMtMi45NCA1LjEtNy43NCA1LjEtMTAuNyAwLTIuMzQtMS41Ny0xLjYtNC4wOS0uMzZsLS4xNC4wN2MtMi4zMSAxLjE1LTUuMzkgMi42Ny04Ljc3IDIuNjdzLTYuNDUtMS41Mi04Ljc3LTIuNjdjLTIuNi0xLjI5LTQuMjMtMi4xLTQuMjMuMjkgMCAzLjA1IDEuNDYgOC4wNiA1LjQ3IDEwLjk3WiIKCQljbGlwLXJ1bGU9ImV2ZW5vZGQiCgkvPgoJPHBhdGgKCQlmaWxsPSIjRkY5RDBCIgoJCWQ9Ik03MC43MSAzN2EzLjI1IDMuMjUgMCAxIDAgMC02LjUgMy4yNSAzLjI1IDAgMCAwIDAgNi41Wk0yNC4yMSAzN2EzLjI1IDMuMjUgMCAxIDAgMC02LjUgMy4yNSAzLjI1IDAgMCAwIDAgNi41Wk0xNy41MiA0OGMtMS42MiAwLTMuMDYuNjYtNC4wNyAxLjg3YTUuOTcgNS45NyAwIDAgMC0xLjMzIDMuNzYgNy4xIDcuMSAwIDAgMC0xLjk0LS4zYy0xLjU1IDAtMi45NS41OS0zLjk0IDEuNjZhNS44IDUuOCAwIDAgMC0uOCA3IDUuMyA1LjMgMCAwIDAtMS43OSAyLjgyYy0uMjQuOS0uNDggMi44LjggNC43NGE1LjIyIDUuMjIgMCAwIDAtLjM3IDUuMDJjMS4wMiAyLjMyIDMuNTcgNC4xNCA4LjUyIDYuMSAzLjA3IDEuMjIgNS44OSAyIDUuOTEgMi4wMWE0NC4zMyA0NC4zMyAwIDAgMCAxMC45MyAxLjZjNS44NiAwIDEwLjA1LTEuOCAxMi40Ni01LjM0IDMuODgtNS42OSAzLjMzLTEwLjktMS43LTE1LjkyLTIuNzctMi43OC00LjYyLTYuODctNS03Ljc3LS43OC0yLjY2LTIuODQtNS42Mi02LjI1LTUuNjJhNS43IDUuNyAwIDAgMC00LjYgMi40NmMtMS0xLjI2LTEuOTgtMi4yNS0yLjg2LTIuODJBNy40IDcuNCAwIDAgMCAxNy41MiA0OFptMCA0Yy41MSAwIDEuMTQuMjIgMS44Mi42NSAyLjE0IDEuMzYgNi4yNSA4LjQzIDcuNzYgMTEuMTguNS45MiAxLjM3IDEuMzEgMi4xNCAxLjMxIDEuNTUgMCAyLjc1LTEuNTMuMTUtMy40OC0zLjkyLTIuOTMtMi41NS03LjcyLS42OC04LjAxLjA4LS4wMi4xNy0uMDIuMjQtLjAyIDEuNyAwIDIuNDUgMi45MyAyLjQ1IDIuOTNzMi4yIDUuNTIgNS45OCA5LjNjMy43NyAzLjc3IDMuOTcgNi44IDEuMjIgMTAuODMtMS44OCAyLjc1LTUuNDcgMy41OC05LjE2IDMuNTgtMy44MSAwLTcuNzMtLjktOS45Mi0xLjQ2LS4xMS0uMDMtMTMuNDUtMy44LTExLjc2LTcgLjI4LS41NC43NS0uNzYgMS4zNC0uNzYgMi4zOCAwIDYuNyAzLjU0IDguNTcgMy41NC40MSAwIC43LS4xNy44My0uNi43OS0yLjg1LTEyLjA2LTQuMDUtMTAuOTgtOC4xNy4yLS43My43MS0xLjAyIDEuNDQtMS4wMiAzLjE0IDAgMTAuMiA1LjUzIDExLjY4IDUuNTMuMTEgMCAuMi0uMDMuMjQtLjEuNzQtMS4yLjMzLTIuMDQtNC45LTUuMi01LjIxLTMuMTYtOC44OC01LjA2LTYuOC03LjMzLjI0LS4yNi41OC0uMzggMS0uMzggMy4xNyAwIDEwLjY2IDYuODIgMTAuNjYgNi44MnMyLjAyIDIuMSAzLjI1IDIuMWMuMjggMCAuNTItLjEuNjgtLjM4Ljg2LTEuNDYtOC4wNi04LjIyLTguNTYtMTEuMDEtLjM0LTEuOS4yNC0yLjg1IDEuMzEtMi44NVoiCgkvPgoJPHBhdGgKCQlmaWxsPSIjRkZEMjFFIgoJCWQ9Ik0zOC42IDc2LjY5YzIuNzUtNC4wNCAyLjU1LTcuMDctMS4yMi0xMC44NC0zLjc4LTMuNzctNS45OC05LjMtNS45OC05LjNzLS44Mi0zLjItMi42OS0yLjljLTEuODcuMy0zLjI0IDUuMDguNjggOC4wMSAzLjkxIDIuOTMtLjc4IDQuOTItMi4yOSAyLjE3LTEuNS0yLjc1LTUuNjItOS44Mi03Ljc2LTExLjE4LTIuMTMtMS4zNS0zLjYzLS42LTMuMTMgMi4yLjUgMi43OSA5LjQzIDkuNTUgOC41NiAxMS0uODcgMS40Ny0zLjkzLTEuNzEtMy45My0xLjcxcy05LjU3LTguNzEtMTEuNjYtNi40NGMtMi4wOCAyLjI3IDEuNTkgNC4xNyA2LjggNy4zMyA1LjIzIDMuMTYgNS42NCA0IDQuOSA1LjItLjc1IDEuMi0xMi4yOC04LjUzLTEzLjM2LTQuNC0xLjA4IDQuMTEgMTEuNzcgNS4zIDEwLjk4IDguMTUtLjggMi44NS05LjA2LTUuMzgtMTAuNzQtMi4xOC0xLjcgMy4yMSAxMS42NSA2Ljk4IDExLjc2IDcuMDEgNC4zIDEuMTIgMTUuMjUgMy40OSAxOS4wOC0yLjEyWiIKCS8+Cgk8cGF0aAoJCWZpbGw9IiNGRjlEMEIiCgkJZD0iTTc3LjQgNDhjMS42MiAwIDMuMDcuNjYgNC4wNyAxLjg3YTUuOTcgNS45NyAwIDAgMSAxLjMzIDMuNzYgNy4xIDcuMSAwIDAgMSAxLjk1LS4zYzEuNTUgMCAyLjk1LjU5IDMuOTQgMS42NmE1LjggNS44IDAgMCAxIC44IDcgNS4zIDUuMyAwIDAgMSAxLjc4IDIuODJjLjI0LjkuNDggMi44LS44IDQuNzRhNS4yMiA1LjIyIDAgMCAxIC4zNyA1LjAyYy0xLjAyIDIuMzItMy41NyA0LjE0LTguNTEgNi4xLTMuMDggMS4yMi01LjkgMi01LjkyIDIuMDFhNDQuMzMgNDQuMzMgMCAwIDEtMTAuOTMgMS42Yy01Ljg2IDAtMTAuMDUtMS44LTEyLjQ2LTUuMzQtMy44OC01LjY5LTMuMzMtMTAuOSAxLjctMTUuOTIgMi43OC0yLjc4IDQuNjMtNi44NyA1LjAxLTcuNzcuNzgtMi42NiAyLjgzLTUuNjIgNi4yNC01LjYyYTUuNyA1LjcgMCAwIDEgNC42IDIuNDZjMS0xLjI2IDEuOTgtMi4yNSAyLjg3LTIuODJBNy40IDcuNCAwIDAgMSA3Ny40IDQ4Wm0wIDRjLS41MSAwLTEuMTMuMjItMS44Mi42NS0yLjEzIDEuMzYtNi4yNSA4LjQzLTcuNzYgMTEuMThhMi40MyAyLjQzIDAgMCAxLTIuMTQgMS4zMWMtMS41NCAwLTIuNzUtMS41My0uMTQtMy40OCAzLjkxLTIuOTMgMi41NC03LjcyLjY3LTguMDFhMS41NCAxLjU0IDAgMCAwLS4yNC0uMDJjLTEuNyAwLTIuNDUgMi45My0yLjQ1IDIuOTNzLTIuMiA1LjUyLTUuOTcgOS4zYy0zLjc4IDMuNzctMy45OCA2LjgtMS4yMiAxMC44MyAxLjg3IDIuNzUgNS40NyAzLjU4IDkuMTUgMy41OCAzLjgyIDAgNy43My0uOSA5LjkzLTEuNDYuMS0uMDMgMTMuNDUtMy44IDExLjc2LTctLjI5LS41NC0uNzUtLjc2LTEuMzQtLjc2LTIuMzggMC02LjcxIDMuNTQtOC41NyAzLjU0LS40MiAwLS43MS0uMTctLjgzLS42LS44LTIuODUgMTIuMDUtNC4wNSAxMC45Ny04LjE3LS4xOS0uNzMtLjctMS4wMi0xLjQ0LTEuMDItMy4xNCAwLTEwLjIgNS41My0xMS42OCA1LjUzLS4xIDAtLjE5LS4wMy0uMjMtLjEtLjc0LTEuMi0uMzQtMi4wNCA0Ljg4LTUuMiA1LjIzLTMuMTYgOC45LTUuMDYgNi44LTcuMzMtLjIzLS4yNi0uNTctLjM4LS45OC0uMzgtMy4xOCAwLTEwLjY3IDYuODItMTAuNjcgNi44MnMtMi4wMiAyLjEtMy4yNCAyLjFhLjc0Ljc0IDAgMCAxLS42OC0uMzhjLS44Ny0xLjQ2IDguMDUtOC4yMiA4LjU1LTExLjAxLjM0LTEuOS0uMjQtMi44NS0xLjMxLTIuODVaIgoJLz4KCTxwYXRoCgkJZmlsbD0iI0ZGRDIxRSIKCQlkPSJNNTYuMzMgNzYuNjljLTIuNzUtNC4wNC0yLjU2LTcuMDcgMS4yMi0xMC44NCAzLjc3LTMuNzcgNS45Ny05LjMgNS45Ny05LjNzLjgyLTMuMiAyLjctMi45YzEuODYuMyAzLjIzIDUuMDgtLjY4IDguMDEtMy45MiAyLjkzLjc4IDQuOTIgMi4yOCAyLjE3IDEuNTEtMi43NSA1LjYzLTkuODIgNy43Ni0xMS4xOCAyLjEzLTEuMzUgMy42NC0uNiAzLjEzIDIuMi0uNSAyLjc5LTkuNDIgOS41NS04LjU1IDExIC44NiAxLjQ3IDMuOTItMS43MSAzLjkyLTEuNzFzOS41OC04LjcxIDExLjY2LTYuNDRjMi4wOCAyLjI3LTEuNTggNC4xNy02LjggNy4zMy01LjIzIDMuMTYtNS42MyA0LTQuOSA1LjIuNzUgMS4yIDEyLjI4LTguNTMgMTMuMzYtNC40IDEuMDggNC4xMS0xMS43NiA1LjMtMTAuOTcgOC4xNS44IDIuODUgOS4wNS01LjM4IDEwLjc0LTIuMTggMS42OSAzLjIxLTExLjY1IDYuOTgtMTEuNzYgNy4wMS00LjMxIDEuMTItMTUuMjYgMy40OS0xOS4wOC0yLjEyWiIKCS8+Cjwvc3ZnPgo=">
<span id="openenvWords">Supported by OpenEnv</span></a>
</div>
<div class="subtitle" id="tagline">Submit RL environment collections; a fixed recipe post-trains a fixed model on them and scores held-out tasks.</div>
</div>
</header>
<div class="note" id="note" hidden><div></div></div>
<main id="main">Loadingβ¦</main>
<footer><div><span id="foot"></span></div></footer>
<dialog id="settings"></dialog>
<script>
// ββ data ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
const P0 = new URLSearchParams(location.search);
// Every visitor starts on live data. The simulated competition (source=mock) is an explicit opt-in: ?source=mock, or the
// Data button, whose choice lasts for this browser tab only. Older builds kept the choice across visits in localStorage.
localStorage.removeItem('pta.source');
let SOURCE = ['live', 'mock'].includes(P0.get('source')) ? P0.get('source') : sessionStorage.getItem('pta.source') === 'mock' ? 'mock' : 'live';
let META = null, BOARD = null, ME = null, CH = localStorage.getItem('pta.challenge') || null; const CACHE = new Map();
// Hugging Face's proxy in front of the Space sometimes answers 502, 503 or 504 with its own HTML error page instead of the
// Space's answer (10-20% of requests did on Sept 28, 2026). Such an answer, or a dropped connection, is sent again after
// these delays (ms). Every request this app makes is safe to repeat: reads, a check (it stores nothing), a submission (the
// same repository, commit and folder map to one record), a launch (its request_id returns the run it started) and a
// collect (it returns the stored result). The Space's own answers, JSON even when they are errors, are never retried.
const RETRY_MS = [600, 1500, 3000];
const isProxyAnswer = (r) => [502, 503, 504].includes(r.status) && !(r.headers.get('content-type') || '').includes('json');
const BRIEFLY = 'The Space is briefly unavailable';
async function getJSON(url, opts) {
let r;
for (let i = 0; ; i++) {
try { r = await fetch(url, { credentials: 'same-origin', ...opts }); if (!isProxyAnswer(r) || i === RETRY_MS.length) break; }
catch (e) { if (i === RETRY_MS.length) throw Object.assign(new Error(`${BRIEFLY}: it could not be reached.`), { transient: true }); }
await new Promise(res => setTimeout(res, RETRY_MS[i]));
}
if (isProxyAnswer(r)) throw Object.assign(new Error(`${BRIEFLY}: Hugging Face answered HTTP ${r.status} instead of the Space.`), { status: r.status, transient: true });
let d = null; try { d = await r.json(); } catch {} if (!r.ok) throw Object.assign(new Error((d && (d.detail || d.message)) || `HTTP ${r.status}`), { status: r.status, body: d }); return d; }
async function api(path) { const url = `/api/app/${path}${path.includes('?') ? '&' : '?'}source=${SOURCE}`, hit = CACHE.get(url); if (hit && Date.now() - hit.at < 30000) return hit.data; const data = await getJSON(url); CACHE.set(url, { at: Date.now(), data }); return data; }
// Who is signed in. When the Space cannot be reached, that is unknown (ME.unchecked), not "sign-in is not available":
// ask again, backing off to a minute, and redraw the page once it answers, keeping whatever was typed into it.
let meRetry = null, meRetryMs = 5000;
async function me() {
clearTimeout(meRetry);
try { ME = await getJSON('/api/auth/me'); meRetryMs = 5000; }
catch (e) {
ME = { authenticated: false, oauth_enabled: null, unchecked: e.message };
meRetry = setTimeout(async () => {
await me(); if (ME.unchecked) return;
const at = location.hash, fields = () => [...$('#main').querySelectorAll('input, select, textarea')], typed = fields().map(x => x.value);
await route(); if (location.hash === at) fields().forEach((x, i) => { if (i < typed.length) x.value = typed[i]; });
}, meRetryMs);
meRetryMs = Math.min(meRetryMs * 2, 60000);
}
return ME;
}
const UNCHECKED = `${BRIEFLY}, so your sign-in could not be checked; retrying.`;
async function act(method, url, body) { // a real arena action: the signed-in session with its CSRF token
return getJSON(url, { method, headers: { 'Content-Type': 'application/json', ...(ME && ME.csrf_token ? { 'X-CSRF-Token': ME.csrf_token } : {}) }, body: body ? JSON.stringify(body) : undefined });
}
// The app's pages at their own address: /arena/submissions[/<id>] (/arena/collections⦠too) and
// /arena/challenges[/<id>[/<tab>]]. app_api.pages serves each with its own link-preview tags, which a #fragment can't
// have (crawlers never send it), and it opens the page #/submissions/<id> or #/challenges/<id>/<tab> opens. With a
// #fragment, the fragment decides the page; the app's links to a submission or a challenge, and Copy link, use addresses.
const PATH_ROUTE = /^\/arena\/((?:submissions|collections)(?:\/[^/?#]+)?|challenges(?:\/[^/?#]+){0,2})\/?$/;
const onPath = () => !location.hash && PATH_ROUTE.test(location.pathname);
const routePath = () => onPath() ? PATH_ROUTE.exec(location.pathname)[1].replace(/^collections/, 'submissions') : location.hash.replace(/^#\/?/, '').split('?')[0];
function qs() { if (onPath()) return new URLSearchParams(location.search); const h = location.hash, i = h.indexOf('?'); return new URLSearchParams(i < 0 ? '' : h.slice(i + 1)); }
function setQs(o) { const p = qs(); for (const [k, v] of Object.entries(o)) { if (v == null || v === '' || v === 'all') p.delete(k); else p.set(k, v); } const s = p.toString(); history.replaceState(null, '', onPath() ? location.pathname + (s ? '?' + s : '') : (location.hash.split('?')[0] || '#/') + (s ? '?' + s : '')); } // /arena itself has no fragment: #/ then
const enc = encodeURIComponent;
// ββ helpers βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
const $ = (s) => document.querySelector(s);
const E = (tag, attrs, ...kids) => { const e = document.createElement(tag); for (const [k, v] of Object.entries(attrs || {})) { if (v == null || v === false) continue; if (k === 'on') for (const [ev, f] of Object.entries(v)) e.addEventListener(ev, f); else if (k in e && typeof v !== 'string') e[k] = v; else e.setAttribute(k, v === true ? '' : v); } for (const k of kids.flat(Infinity)) if (k != null && k !== false) e.append(k instanceof Node ? k : String(k)); return e; };
const A = (text, href, cls) => E('a', { href, class: cls || null }, text);
// Sign-in comes back to the page it started on (auth.py's next: this app at /arena, with its #fragment); without next it
// would land on the board at /.
const signInHref = () => '/auth/login?next=' + encodeURIComponent(onPath() ? '/arena#/' + routePath() + location.search : location.pathname + location.hash);
// Framed on huggingface.co/spaces/... (the board's Submissions link keeps you in its iframe), the iframe's sandbox allows
// new tabs but not top-level navigation, and Hugging Face's sign-in page doesn't load inside a frame: sign-in then opens
// this Space's own address in a new tab, where the OAuth round trip is first-party (board.html does the same).
const FRAMED = window.self !== window.top;
const signIn = (text) => E('a', { href: signInHref(), target: FRAMED ? '_blank' : null, rel: FRAMED ? 'noopener' : null, on: { click: (e) => { e.currentTarget.href = signInHref(); } } }, text);
const MONTHS = ['Jan', 'Feb', 'Mar', 'Apr', 'May', 'Jun', 'Jul', 'Aug', 'Sep', 'Oct', 'Nov', 'Dec'];
const when = (iso) => { const d = new Date(iso); return !iso || isNaN(d) ? '' : `${MONTHS[d.getUTCMonth()]} ${d.getUTCDate()}, ${String(d.getUTCHours()).padStart(2, '0')}:${String(d.getUTCMinutes()).padStart(2, '0')} UTC`; };
const dur = (s) => s == null ? '' : s < 90 ? `${Math.round(s)} s` : s < 5400 ? `${Math.round(s / 60)} min` : `${(s / 3600).toFixed(1)} h`;
const nowMs = () => SOURCE === 'mock' && META && META.as_of ? Date.parse(META.as_of) : Date.now();
const r1 = (v) => (Math.round(Math.abs(v) * 10 + 1e-6) / 10).toFixed(1); // one decimal, halves up: 6.05 and 6.0515 both print 6.1
const signed = (v) => { if (v == null) return 'β'; const x = r1(v); return (x === '0.0' ? '' : v > 0 ? '+' : 'β') + x; };
const dse = (d, se) => E('span', { class: 'nw' }, `${signed(d)} Β± ${se == null ? 'β' : r1(se)}`);
const usd = (v) => v == null ? 'β' : '$' + Number(v).toFixed(2);
const plural = (n, w, ws) => `${n} ${n === 1 ? w : ws || w + 's'}`;
const first = (s) => String(s || '').split(/(?<=\.)\s/)[0];
const cap = (s) => s ? s[0].toUpperCase() + s.slice(1) : '';
const frac = (p, n) => p == null ? 'β' : n ? `${Math.round(p * n)}/${n}` : `${(100 * p).toFixed(1)}%`;
function table(cols, rows, empty = 'Nothing here yet.', head, cls) { const t = E('table', { class: cls || null }, ...(head || [E('tr', {}, cols.map(([l, c]) => E('th', { class: c || '' }, l)))])); if (!rows.length) t.append(E('tr', {}, E('td', { colspan: cols.length, class: 'muted' }, empty))); rows.forEach(r => t.append(r)); return E('div', { class: 'scroll' + (cls === 'wide' && rows.length ? ' sideways' : '') }, t); }
const row = (href, cells) => { const tr = E('tr', href ? { 'data-href': href, tabindex: '0', on: { click: (e) => { if (!e.target.closest('a,button')) go(href); }, keydown: (e) => { if (e.key === 'Enter') go(href); } } } : {}); cells.forEach(([v, c]) => tr.append(E('td', { class: c || '' }, v ?? 'β'))); return tr; };
const cell = (v, c) => [v, c];
const dl = (items) => E('dl', {}, ...items.filter(([, x]) => x != null && x !== '').flatMap(([k, x]) => [E('dt', {}, k), E('dd', {}, x)]));
const setTitle = (t) => { document.title = `${t} Β· PostTrain Arena`; };
function page(title, ...kids) { setTitle(title); return [E('h1', {}, title), ...kids]; }
// a run's state in the arena's words: what the protocol records (state, stage, verification) and nothing else
function runState(r) {
if (r.state === 'running' || r.state === 'queued') return ['running', `${r.state === 'queued' ? 'queued' : 'running'}${r.stage ? ': ' + r.stage : ''}`];
if (r.state === 'scored') return r.verification === 'valid' ? ['verified', 'scored, verified'] : r.verification === 'invalid' ? ['rejected', 'scored, rejected in review'] : r.verification === 'pending' ? ['review', 'scored, in review'] : ['review', 'scored, result not collected yet'];
if (r.state === 'failed') return ['failed', `failed${r.stage ? ' at ' + r.stage : ''}`];
return [r.state === 'canceled' ? 'canceled' : 'none', r.state || 'β'];
}
const stateEl = (r) => { const [k, t] = runState(r); return E('span', { class: 'state s-' + k }, t); };
const issueFor = (reason) => ((META && META.known_issues) || []).find(i => i.match && (reason || '').toLowerCase().includes(i.match.toLowerCase()));
const OUTCOME = { eligible: 'passed static checks', flagged: 'passed, flagged for review', 'needs controls': 'needs controls', excluded: 'excluded', trained: 'in band', 'out of band': 'out of band', 'failed controls': 'failed controls' };
const OUTCOME_NOTE = 'Passed: no finding. Flagged: a finding worth a look, such as a verifier that only checks that files exist; the task is still trained on. Needs controls: the task has no working reference solution; runs still train on it, and it is meant to count only once two checks pass (doing nothing must score 0, and the untrained model must solve it at least once in a few attempts), which organizers run by hand. Excluded: the task leaks the answer or overlaps the held-out suite, so runs never train on it.';
const outcomeClass = (o) => o === 'excluded' ? 'excluded' : o === 'needs controls' || o === 'flagged' ? 'review' : 'ok';
// ββ where links go ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
// A run's page is on this board (#/runs/<id>): its state, why it stopped, its score on the held-out suite and its review.
// The dashboard at /dashboard shows other runs (BenchFlow's Fireworks runs and public post-training runs), not the
// arena's, so nothing here links a run to it.
const subHref = (id) => '#/runs/' + enc(id);
const chHref = (id, tab) => '/arena/challenges/' + enc(id) + (tab ? '/' + tab : ''); // a challenge's own address (F11-10)
const colHref = (id) => '/arena/submissions/' + enc(id); // a submission's own address (#/submissions/<id> and #/collections/<id> open it too)
// A link inside the app: a #fragment changes the page as always; an address is pushed and routed without reloading.
function go(href) { if (href.startsWith('#')) { location.hash = href; return; } history.pushState(null, '', href); route(); }
document.addEventListener('click', (e) => {
const a = e.target.closest && e.target.closest('a[href^="/arena/"]');
if (!a || e.defaultPrevented || e.button !== 0 || e.metaKey || e.ctrlKey || e.shiftKey || e.altKey || a.target || !PATH_ROUTE.test(a.pathname)) return;
e.preventDefault(); go(a.getAttribute('href'));
});
const runLink = (r) => E('span', { class: 'nw' }, A(r.label, subHref(r.id)));
// ββ challenges ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
// A challenge's pages (overview, leaderboard, runs, rules) carry its id in the URL; the pages every challenge
// shares (starter kit, submit) act on the last open challenge you looked at, or the main one.
const CHS = () => (META && META.challenges) || [];
const chById = (id) => CHS().find(c => c.id === id);
const mainCh = () => CHS().find(c => c.role === 'main challenge') || CHS().find(c => c.status === 'open') || CHS()[0];
const ch = () => { const c = chById(CH); return c && c.status === 'open' ? c : mainCh(); };
const boardOf = (id) => ((BOARD && BOARD.challenges) || []).find(x => x.id === id);
const rulesOf = (c) => (c && c.rules) || {};
const suiteN = (c) => (rulesOf(c).eval_suite || {}).task_count || null;
// a challenge's compute in words: its GPUs and where they run (a provider other than HF Jobs, such as Nebius, may still be planned)
const PROVIDERS = { huggingface: 'Hugging Face Jobs', nebius: 'Nebius' };
const opensWord = (day) => day > new Date().toISOString().slice(0, 10) ? 'opens' : 'opened'; // a challenge listed before its first day
const gpus = (f, cp) => { const m = /^([a-z]+\d+)x(\d+)$/i.exec(f || ''), p = (cp && cp.provider) || 'huggingface', where = (PROVIDERS[p] || p) + (cp && cp.provider_status === 'planned' ? ' (planned)' : '');
return m ? (p === 'huggingface' ? `${m[2]} ${m[1].toUpperCase()} GPUs on ${where} (${f})` : `${m[2]} ${m[1].toUpperCase()} GPUs on ${where}`) : f ? `${where} ${f}` : 'the GPU job'; };
const modelName = (c) => ((rulesOf(c).base_model || {}).repo_id || (c.model_info || {}).repo_id || c.model || 'the model').split('/').pop();
function setCurrent(id) { CH = id; localStorage.setItem('pta.challenge', id); }
function chState(c, B) { // [class, words]: can a run start on this challenge now
if (c.status === 'planned') return ['none', 'planned'];
if (c.status !== 'open') return ['none', c.status || 'closed'];
if (!B) return ['none', 'open'];
if (B.runs_paused) return ['review', 'runs paused'];
if (B.accepting_runs) return ['ok', 'taking runs'];
if ((B.active || []).some(r => r.state === 'running')) return ['running', 'a run is in progress'];
return ['review', 'not taking runs now'];
}
// ββ shell βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
const NAV = [['submissions', 'Submissions'], ['challenges', 'Challenges'], ['tasks', 'Tasks'], ['starter', 'Starter kit']];
const navHref = (k) => ['submissions', 'challenges'].includes(k) ? '/arena/' + k : '#/' + k; // the two lists have addresses
function shell(on) {
$('#nav').replaceChildren(...NAV.map(([k, l]) => E('a', { href: navHref(k), class: on === k ? 'on' : null, 'aria-current': on === k ? 'page' : null }, l)));
const who = $('#who'); who.replaceChildren(E('a', { href: '/', class: 'ext', title: 'The Agent Collabs board: messages between participants, organizers and their agents' }, 'Board'));
if (SOURCE === 'live') who.append(ME && ME.authenticated ? E('span', {}, 'Signed in as ', E('b', {}, ME.user.name)) : ME && ME.oauth_enabled ? signIn('Sign in with Hugging Face')
: E('span', { class: 'muted small', title: ME && ME.unchecked ? UNCHECKED : null }, ME && ME.unchecked ? 'checking sign-inβ¦' : 'sign-in unavailable here'));
who.append(E('button', { class: 'plain', title: 'Choose live data or a simulated competition', on: { click: settings } }, `Data: ${SOURCE === 'mock' ? 'simulated' : 'live'}`)); // Submit a collection is the header's filled button
counts();
const n = $('#note'); n.hidden = SOURCE !== 'mock';
if (SOURCE === 'mock') n.firstChild.replaceChildren(E('b', {}, 'Simulated data, not the real arena. '), `A mock competition that follows this arenaβs real rules and is read by the same code as live data, as of ${when(META.as_of)}. Teams, runs and results are invented. `,
E('button', { class: 'plain', on: { click: () => setSource('live') } }, 'Show live data'));
$('#foot').replaceChildren('For agents: ', A('AGENTS.md', '/AGENTS.md'), ', ', A('arena_cli.py', '/arena_cli.py'), ' and the ', A('API reference', '/docs'), '.');
}
// the counts line under the title, as the board's: collections submitted, runs (and how many are going), collections ranked
function counts() {
const k = (META && META.counts) || {}, open = CHS().filter(c => c.status === 'open'), el = $('#counts');
const n = (v) => E('span', { class: 'n' }, String(v ?? 'β')), sep = () => E('span', { class: 'sep' }, '|');
const ranked = BOARD && open.every(c => boardOf(c.id)) ? open.reduce((a, c) => a + ((boardOf(c.id).stats || {}).ranked || 0), 0) : null; // unknown without the board's data
el.replaceChildren('submissions: ', n(k.submissions), sep(), 'runs: ', n(k.runs), ...(k.running ? [' (', n(k.running), ' running)'] : []), ...(ranked == null ? [] : [sep(), 'ranked: ', n(ranked)]));
el.hidden = false;
}
function settings() {
const d = $('#settings'); d.replaceChildren(E('h2', { style: 'margin-top:0' }, 'Data source'),
...[['live', 'Live data (the default)', 'What people have submitted and what the arena has run, rebuilt every two minutes from the arenaβs records.'], ['mock', 'Simulated competition', 'Mock data: a competition simulated under the same rules and read through the same code. Teams, runs and results are invented. It stays on in this browser tab until you switch back.']].map(([k, t, text]) =>
E('p', {}, E('label', {}, E('input', { type: 'radio', name: 'src', checked: SOURCE === k, on: { change: () => { d.close(); setSource(k); } } }), ' ', E('b', {}, t), ' β ', text))),
...(SOURCE === 'mock' && META && META.basis ? [E('details', {}, E('summary', {}, 'How the simulated data is made'), E('ul', {}, META.basis.map(b => E('li', { class: 'small' }, b))))] : []),
E('p', {}, E('button', { class: 'plain', on: { click: () => d.close() } }, 'Close')));
d.showModal();
}
function setSource(s) { SOURCE = s; sessionStorage.setItem('pta.source', s); const u = new URL(location.href); u.searchParams.delete('source'); history.replaceState(null, '', u); CACHE.clear(); META = null; BOARD = null; route(); }
// a challenge's frame: where you are, whether it takes runs, its tabs
const TABS = [['', 'Overview'], ['leaderboard', 'Leaderboard'], ['runs', 'Runs'], ['rules', 'Rules']];
function frame(c, tab) {
const B = boardOf(c.id), R = rulesOf(c), [k, words] = chState(c, B), role = R.role || c.role;
const line = [E('span', { class: 'mono' }, c.id), c.status === 'open' ? E('span', {}, 'open') : null, E('span', { class: 'state pill s-' + k }, words), role ? E('span', {}, role) : null,
R.opens ? E('span', {}, `${opensWord(R.opens)} ${R.opens}${R.closes ? ', closes ' + R.closes : ', no closing date yet'}`) : null].filter(Boolean);
return E('div', { class: 'frame' }, E('div', { class: 'crumb' }, A('Challenges', '#/challenges'), ' / ', c.id), E('h1', {}, (c.name || c.id).replace(/ Β· /g, '\u00a0Β· ')), E('div', { class: 'status-line' }, ...line),
c.status === 'open' ? E('nav', { class: 'tabs', 'aria-label': 'Challenge' }, ...TABS.map(([t, l]) => E('a', { href: chHref(c.id, t), class: tab === t ? 'on' : null, 'aria-current': tab === t ? 'page' : null }, l))) : E('div', { class: 'tabs' }));
}
const noise = (rows) => rows.filter(x => x.stderr_pp && Math.abs(x.delta_pp) > 2 * x.stderr_pp).length; // entries more than two standard errors from zero
const tasksPerStep = (c) => (c.method_info || {}).tasks_per_step;
// ββ Challenges: every challenge, whether it takes runs, and where it stands ββββββββ
async function challengesPage(v) {
v.append(...page('Challenges'), E('p', { class: 'lede' }, 'A challenge fixes the model, the training recipe and a held-out suite of test tasks, so the only thing that differs between its runs is the collection trained on. A run scores a collection by how much training on it changes the modelβs pass rate on the held-out tasks, which the run never trains on. Any submitted collection can run on any open challenge.'));
v.append(E('div', { style: 'height:8px' }), table([['Challenge'], ['State'], ['Model and recipe'], ['Held-out suite', 'hide-s'], ['Runs', 'r'], ['Leaderboard']], CHS().map(c => {
const B = boardOf(c.id), s = (B && B.stats) || {}, me = c.method_info || {}, su = c.suite_info || [], [k, words] = chState(c, B), top = ((B && B.top) || [])[0], R = rulesOf(c);
const live = ((B && B.active) || []).find(r => r.state === 'running');
const why = c.status !== 'open' ? c.open_note : !B ? '' : B.runs_paused ? cap(first(B.runs_paused)) : B.accepting_runs ? 'A run can start now.'
: live ? `Run ${live.label} of ${live.title || 'an organizer test'} is at ${live.stage || 'setup'}; the arena runs one at a time.` : first(B.reason);
const recipe = me.steps ? `${c.method}: ${plural(me.steps, 'step')}, ${me.group_size} attempts per task${me.tasks_per_step > 1 ? `, ${me.tasks_per_step} tasks per step` : ''}` : c.method;
const trials = me.trials || (R.metric || {}).trials_per_run;
return row(chHref(c.id), [cell(E('span', {}, A(c.name || c.id, chHref(c.id)), E('span', { class: 'reason' }, [c.id, R.role || c.role].filter(Boolean).join(' Β· ')))),
cell(E('span', {}, E('span', { class: 'state pill s-' + k }, words), why ? E('span', { class: 'reason' }, why) : '')),
cell(E('span', {}, modelName(c), E('span', { class: 'reason' }, recipe || ''))),
cell(E('span', {}, su.map(x => x.name).join('; ') || 'β', su.length ? E('span', { class: 'reason' }, `${su.map(x => x.task_count).join(' + ')} tasks${trials ? `, ${plural(trials, 'attempt')} per task` : ''}`) : ''), 'hide-s'),
cell(c.status === 'open' ? E('span', {}, String(s.runs || 0), E('span', { class: 'reason' }, s.runs ? (s.scored ? `${s.scored} scored` : 'none scored') : 'none yet')) : 'β', 'r'),
cell(c.status !== 'open' ? 'β' : s.ranked ? E('span', {}, `${plural(s.ranked, 'collection')} ranked`, top ? E('span', { class: 'reason' }, `top: ${top.title}, `, dse(top.delta_pp, top.stderr_pp), ' pp', noise(B.top || []) ? '' : ', within noise') : '') : E('span', { class: 'muted' }, 'none ranked yet'))]);
}), 'No challenge in this data source.', null, 'wide'));
}
// ββ Overview: what the challenge is, whether a run can start, how it works, where it stands; the rules at the side ββ
async function overview(v, c) {
const B = boardOf(c.id), R = rulesOf(c);
setTitle(c.name || c.id); v.append(frame(c, ''));
if (c.status !== 'open') { v.append(plannedPage(c)); return; }
const left = E('div', {}, R.summary ? E('p', {}, R.summary) : '', R.role ? E('p', { class: 'muted' }, `This challenge is a ${R.role}. ${R.role_note || ''}`) : '');
if (B) left.append(nowBox(c, B)); else left.append(E('p', { class: 'muted' }, 'Whether a run can start could not be read right now.'));
left.append(E('h2', {}, 'How it works'), howItWorks(c));
if (B) left.append(E('h2', {}, 'Where it stands'), ...stands(c, B), E('h2', {}, 'Compute budget'), ...budgetFacts(c, B));
v.append(E('div', { class: 'layout' }, left, facts(c)));
}
function nowBox(c, B) {
const box = E('div', { class: 'box ' + (B.accepting_runs ? '' : 'warn') });
box.append(E('p', {}, ...(B.runs_paused ? [E('b', {}, 'Runs are paused by the organizers. '), cap(B.runs_paused)]
: B.accepting_runs ? [E('b', {}, 'A run can start now. '), `The arena runs one at a time, and a run reserves ${usd(B.reserve_usd)} of the shared budget until it ends.`]
: [E('b', {}, 'No run can start right now. '), B.reason || ''])));
for (const r of B.active || []) box.append(E('p', {}, r.state === 'queued' ? 'Starting: ' : 'Running: ', A(`run ${r.label}`, subHref(r.id)), ` of ${r.title || 'an organizer test'}${r.team || r.author ? ` (${r.team || r.author})` : ''}, ${r.stage ? 'now at ' + r.stage : 'starting'}; started ${when(r.started_at || r.created_at)}.`));
if ((B.queue || []).length) box.append(E('p', {}, 'Waiting to start: ', ...B.queue.map((q, i) => [i ? ', ' : '', A(q.title || q.collection_id, colHref(q.collection_id)), q.team ? ` (${q.team})` : '']), '.'));
if (c.status_note) box.append(E('p', { class: 'small org' }, E('b', {}, 'From the organizers. '), c.status_note));
return box;
}
function howItWorks(c) {
const n = suiteN(c);
return E('ol', {},
E('li', {}, E('b', {}, 'Write tasks. '), 'Each task is a sandbox, a prompt and a verifier that checks the result. The ', A('starter kit', '#/starter'), ' has a template and eight examples to copy.'),
E('li', {}, E('b', {}, 'Submit the collection. '), 'The arena reads your repository at one commit and runs the static checks on every task; tasks that leak the answer or copy the held-out suite are left out.'),
E('li', {}, E('b', {}, 'Start a run. '), `The arena post-trains ${modelName(c)} on your tasks with the fixed recipe${tasksPerStep(c) === 1 ? ' (this recipe trains on one task, drawn from your collection with a fixed seed)' : ''}, then scores it on ${n ? n + ' ' : 'the '}held-out tasks it never trained on.`),
E('li', {}, E('b', {}, 'Your score is the change. '), 'Held-out pass rate after training minus before, measured inside the same run. An organizer reviews the evidence, and the ', A('leaderboard', chHref(c.id, 'leaderboard')), ' ranks collections by their mean change over verified runs.'));
}
function stands(c, B) {
const s = B.stats || {}, out = [];
if (!s.runs) out.push(E('p', {}, 'No run yet.'));
else {
const who = !s.collections ? `, all of them organizer test runs` : s.organizer_runs ? `: ${s.runs - s.organizer_runs} on ${plural(s.collections, 'collection')} and ${plural(s.organizer_runs, 'organizer test run')}` : ` on ${plural(s.collections, 'collection')}`;
const parts = [[s.verified, 'scored and verified'], [s.in_review, 'scored and in review'], [s.uncollected, 'scored, result not collected yet'], [s.rejected, 'rejected in review'], [s.running, 'running'], [s.queued, 'queued'], [s.failed, 'failed'], [s.canceled, 'canceled'], [s.unknown, 'whose state could not be read']].filter(([x]) => x).map(([x, w]) => `${x} ${w}`);
out.push(E('p', {}, `${plural(s.runs, 'run')} so far${who}. `, s.scored ? `${cap(parts.join(', '))}. ` : `None has been scored yet: ${parts.join(', ')}. `, A('Every run', chHref(c.id, 'runs')), '.'));
}
if (s.ranked) out.push(E('p', {}, `${plural(s.ranked, 'collection is', 'collections are')} on the leaderboard. `, A('The full leaderboard', chHref(c.id, 'leaderboard')), '.'), boardTable(B.top || [], null, 'No collection is ranked yet.'),
noise(B.top || []) ? '' : E('p', { class: 'small muted' }, 'None of these differs from zero by more than two standard errors, so this order is noise so far.'));
else out.push(E('p', { class: 'muted' }, 'No collection is on the leaderboard yet: a collection enters once an organizer verifies one of its scored runs.'));
return out;
}
function budgetFacts(c, B) {
const b = (BOARD && BOARD.budget) || {}, cp = rulesOf(c).compute || {};
if (b.cap_usd == null) return [E('p', { class: 'muted' }, 'The budget could not be read for this data source.')];
const d = (x) => E('span', { class: 'detail' }, x);
return [E('p', { class: 'muted' }, 'One budget pays for every challengeβs runs and the organizersβ own jobs. A run starts only if what is left covers its reservation.'),
dl([['Project cap', [usd(b.cap_usd), d(' β fixed; it does not reset')]],
['Committed', [usd(b.committed_usd), d(' β settled runs, the organizersβ other jobs and earlier spending, and the reservations of runs still going')]],
['Held for runs in progress', b.active_reservations_usd ? usd(b.active_reservations_usd) : null],
['Left', E('b', {}, usd(b.remaining_usd))],
['One run reserves', B.reserve_usd != null ? [usd(B.reserve_usd), d(` β the price of ${gpus(cp.flavor, cp)} for the whole ${cp.timeout_seconds ? cp.timeout_seconds / 3600 + ' h ' : ''}job timeout; held until the run ends, which is then charged its actual cost`)] : cp.provider && cp.provider !== 'huggingface' ? `none yet: runs on ${gpus(cp.flavor, cp)} are not connected, so no price is quoted` : 'unknown: the GPU price could not be read'],
['Runs that still fit', B.runs_that_fit != null ? String(B.runs_that_fit) : 'β']]),
b.basis ? E('p', { class: 'muted small' }, 'How committed spending is counted: ', b.basis[0].toLowerCase() + b.basis.slice(1)) : ''];
}
// the rules at a glance, as label / value / one line of detail (AIcrowd's fact strip)
function facts(c) {
const R = rulesOf(c), m = R.base_model || {}, rec = R.recipe || {}, s = R.eval_suite || {}, cp = R.compute || {};
const items = [['Model', modelName(c), 'fixed; every run starts from the same weights'],
['Training', `${(rec.method || 'GRPO').split(' ')[0]}, ${plural(rec.max_steps || 0, 'step')}`, `${rec.num_generations} attempts ${tasksPerStep(c) === 1 ? 'at one task drawn from your collection' : 'per task'}, the ${(rec.harness || {}).agent || 'agent'} agent, ${(rec.harness || {}).agent_timeout_sec} s each${(c.method_info || {}).status === 'planned' ? '; placeholder values until the organizers set the recipe' : ''}`],
['Held-out suite', `${s.task_count} tasks`, `${s.name || ''}; held out: runs never train on them; ${plural((R.metric || {}).trials_per_run || 1, 'attempt')} per task${s.sealed === false ? '; a public benchmark' : ', names private'}`],
['Score', 'Ξ pass rate, pp', 'after training minus before, same run'],
['Daily limit', `${cp.runs_per_submission_per_day || 1} run`, 'per collection per 24 h; failed and canceled runs do not count'],
['Compute', `${plural(cp.concurrent_runs || 1, 'run')} at a time`, `${gpus(cp.flavor, cp)}, up to ${(cp.timeout_seconds || 0) / 3600} h each`],
[R.opens && opensWord(R.opens) === 'opens' ? 'Opens' : 'Opened', R.opens || 'β', R.closes ? `closes ${R.closes}` : 'no closing date yet']];
return E('aside', { class: 'facts' }, ...items.map(([k, x, d]) => E('div', {}, E('div', { class: 'k' }, k), E('div', { class: 'v' }, x), d ? E('div', { class: 'd' }, d) : '')), E('p', { class: 'small' }, A('All rules', chHref(c.id, 'rules'))));
}
// a planned challenge: what its configs bind, and what it waits for
function plannedPage(c) {
const m = c.model_info || {}, me = c.method_info || {}, su = c.suite_info || [];
return E('div', {}, E('div', { class: 'box warn' }, E('p', {}, E('b', {}, cap(c.status || 'planned') + '. '), 'No run can start on it yet.', c.open_note ? ' ' + c.open_note : '')),
E('h2', {}, 'What it will run'), E('p', { class: 'muted' }, 'From the challengeβs config; these may change before it opens.'),
dl([['Model', m.repo_id ? `${m.repo_id}${m.params ? ` (${m.params})` : ''}` : c.model],
['Recipe', me.method ? `${c.method}: ${me.method}` : c.method],
['Training', me.steps ? `${plural(me.steps, 'optimizer step')}; ${me.group_size} attempts per task, ${plural(me.tasks_per_step || 1, 'task')} per step; learning rate ${me.learning_rate}` : null],
['Agent time limit', me.agent_timeout_sec ? `${me.agent_timeout_sec} s per task` : null],
['Held-out suites', su.length ? su.map(s => `${s.name} (${s.task_count} tasks)`).join('; ') : null],
['Held-out attempts', me.trials ? `${plural(me.trials, 'attempt')} per task per run` : null],
['Compute', c.compute],
['About the recipe', me.note]]));
}
// ββ Leaderboard βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
function boardTable(rows, withLatest, empty) {
return table([['Rank'], ['Collection'], ['Team', 'hide-s'], ['Verified runs', 'r'], ['Mean Ξ Β± SE, pp', 'r'], ...(withLatest ? [['Latest verified run', 'hide-s']] : [])], rows.map(x => row(colHref(x.collection_id), [
cell(x.rank ?? 'β'), cell(A(x.title || x.collection_id, colHref(x.collection_id))), cell(x.team, 'hide-s'), cell(x.verified_runs, 'r'), cell(dse(x.delta_pp, x.stderr_pp), 'r'),
...(withLatest ? [cell((x.run_ids || []).length ? runLink({ id: x.run_ids[x.run_ids.length - 1], label: String(x.run_ids[x.run_ids.length - 1]).replace('challenge-', '').slice(0, 8) }) : 'β', 'hide-s')] : [])])), empty);
}
async function leaderboard(v, c) {
setTitle(`Leaderboard Β· ${c.id}`); v.append(frame(c, 'leaderboard'));
if (c.status !== 'open') { v.append(E('p', { class: 'muted' }, `${c.id} is ${c.status}: it has no runs and no leaderboard yet.`)); return; }
const d = await api('leaderboard?challenge=' + enc(c.id)), m = d.meta || {}, rows = d.rows || [], R = rulesOf(c), ref = m.reference, s = (boardOf(c.id) || {}).stats || {};
if (R.role === 'smoke test') v.append(E('p', {}, E('b', {}, 'Smoke test. '), R.role_note || ''));
v.append(E('p', { class: 'muted' }, 'Collections are ranked by the mean change over every organizer-verified run, not their best run, so running more often does not help; ties share a rank. Each run measures its before-training score itself, on the same held-out tasks.'));
const n = suiteN(c), befores = (await api(`runs?challenge=${enc(c.id)}`)).filter(r => r.before != null).map(r => Math.round(r.before * (n || 1)));
const seen = befores.length && n ? ` Runs so far measured ${Math.min(...befores) === Math.max(...befores) ? Math.min(...befores) : `${Math.min(...befores)} to ${Math.max(...befores)}`} of ${n} before training.` : '';
if (ref && ref.pass_rate != null) v.append(E('p', { class: 'small' }, `For scale: before any training, ${modelName(c)} passes ${(100 * ref.pass_rate).toFixed(1)}% Β± ${(100 * (ref.stderr || 0)).toFixed(1)} of the held-out tasks in the organizersβ reference measurement (${plural(ref.trials || 1, 'trial')} with the arenaβs harness).${seen}`));
const clear = noise(rows);
if (rows.length) v.append(E('div', { class: 'box ' + (clear ? '' : 'warn') }, E('p', {}, clear ? `${plural(clear, 'entry', 'entries')} differ from zero by more than two standard errors.` : 'No entry differs from zero by more than two standard errors, so this order is noise so far.',
' ', m.per_run_sd_pp ? `Β± uses the run-to-run spread pooled over collections with repeat runs (about ${m.per_run_sd_pp} pp per run).` : 'No collection has a repeat verified run yet, so each Β± is that one runβs own standard error.')));
if (m.pending_count) v.append(E('p', {}, `${plural(m.pending_count, 'collected run')} awaiting an organizerβs review.`));
v.append(boardTable(rows, true, s.runs ? `No collection is ranked yet. ${plural(s.runs, 'run')} so far, ${s.scored ? `${s.scored} scored and ${s.verified} verified` : 'none scored'}; a collection enters once an organizer verifies one of its scored runs.` : 'No collection is ranked yet: no run has started.'));
v.append(E('p', { class: 'small muted' }, 'Ξ is the held-out pass rate after training minus before, in percentage points (pp); Β± is one standard error.'));
}
// ββ Runs: every run with its status and a one-line reason; your collections first ββββββββ
function reasonLine(r) {
const [k] = runState(r);
if (k === 'failed') { const hit = issueFor(r.reason); return hit ? first(hit.cause) : r.reason ? 'stopped on an error that is not among the known issues yet; its page shows the error' : ''; }
if (k === 'canceled') return 'canceled before it finished';
if (k === 'running') return r.state === 'queued' ? 'waiting for the running run to end' : r.started_at ? `for ${dur((nowMs() - Date.parse(r.started_at)) / 1000)} so far` : '';
if (r.state === 'scored') return r.verification === 'valid' ? 'counts on the leaderboard' : r.verification === 'pending' ? 'waiting for an organizerβs review' : r.verification === 'invalid' ? 'rejected in review' : 'scored; the result is not collected yet';
return '';
}
// who "you" are: the signed-in account on live data; on simulated (mock) data, one invented team
async function identity(c) {
const live = SOURCE === 'live', signedIn = live && ME && ME.authenticated;
if (live) return { you: signedIn ? ME.user.name : null, live, signedIn, enabled: signedIn && c.status === 'open',
why: !signedIn ? (ME && ME.oauth_enabled ? 'Sign in first.' : ME && ME.unchecked ? UNCHECKED : 'Sign-in is not available on this server.') : c.status !== 'open' ? `${c.id} is ${c.status} and accepts nothing yet.` : '' };
const all = await api(`runs?challenge=${enc(c.id)}`), subs = await api('submissions');
return { you: (all.find(r => r.state === 'scored') || all[0] || {}).author || (subs[0] || {}).team, live, signedIn: false, enabled: false, why: 'Simulated data: switch to live data to act.' };
}
async function submissions(v, c) {
setTitle(`Runs Β· ${c.id}`); v.append(frame(c, 'runs'));
if (c.status !== 'open') { v.append(E('p', { class: 'muted' }, `${c.id} is ${c.status}: it accepts no runs yet.`)); return; }
const list = await api(`runs?challenge=${enc(c.id)}`), subs = await api('submissions'), I = await identity(c), P = qs(), R = rulesOf(c), B = boardOf(c.id) || {};
v.append(E('p', { class: 'muted' }, `A run post-trains ${modelName(c)} on one collectionβs tasks with the fixed recipe, then scores it on the held-out suite. The arena runs one at a time; a collectionβs author starts its runs here.`));
// yours: your collections with today's allowance and the run actions, then your scored runs to collect
if (I.you) {
const mine = subs.filter(s => (s.team || s.author) === I.you), limit = (R.compute || {}).runs_per_submission_per_day || 1, dayAgo = nowMs() - 86400000, slot = E('div');
v.append(E('h2', {}, I.live ? 'Your collections' : `Your collections (as team ${I.you})`));
v.append(table([['Collection'], ['Eligible tasks', 'r'], ['Runs today', 'r'], ['']], mine.map(s => {
const fl = s.funnel || {}, counted = list.filter(r => r.collection_id === s.id && ['scored', 'running', 'queued'].includes(r.state) && Date.parse(r.created_at) > dayAgo).length;
const pre = E('button', { class: 'plain', disabled: !I.enabled, title: I.why }, 'Preflight'), start = E('button', { disabled: true, title: I.why || 'Preflight first.' }, 'Start a run'), requestId = `${s.id}-${Date.now()}`;
pre.onclick = async () => { slot.replaceChildren(E('p', {}, `Running every check for ${s.title}β¦`)); try { const r = await getJSON(`/api/challenges/${enc(c.id)}/runs/preflight?environment_id=${enc(s.id)}`); slot.replaceChildren(checksList(r)); start.disabled = !r.allowed; } catch (e) { slot.replaceChildren(E('div', { class: 'box bad' }, e.message)); } };
start.onclick = async () => { if (!confirm(`Start a run of β${s.title}β on ${c.id}? It reserves up to ${usd(B.reserve_usd ?? R.reserve_usd)} of the shared budget, takes several hours, and no other run can start until it ends.`)) return;
try { const r = await act('POST', `/api/challenges/${enc(c.id)}/runs`, { request_id: requestId, environment_id: s.id }); slot.replaceChildren(E('div', { class: 'box' }, E('p', {}, E('b', {}, `Run ${r.run_id} is ${r.state}. `), 'It appears below within two minutes.'))); CACHE.clear(); }
catch (e) { slot.replaceChildren(E('div', { class: 'box bad' }, e.message, e.transient ? ' Whether the run started is unknown. Start it again from this button in a minute: it sends the same request, which never starts a second run.' : '')); } };
return row(null, [cell(A(s.title, colHref(s.id))), cell(fl.static_eligible != null ? `${fl.static_eligible} of ${fl.submitted}` : s.task_count, 'r'), cell(`${counted} of ${limit}`, 'r'), cell(E('span', { class: 'nw' }, pre, ' ', start))]);
}), I.live ? 'You have not submitted a collection yet.' : 'This team has no collection.', null, 'wide'), slot);
v.append(E('p', { class: 'small muted' }, I.why ? I.why + ' ' : '', 'Eligible tasks are the ones the static checks did not exclude; a run draws its training task from them. Preflight runs every check the arena makes before it starts a run, and reserves nothing.'));
const collect = list.filter(r => r.author === I.you && r.state === 'scored' && (!r.verification || r.verification === 'uncollected'));
if (collect.length) v.append(E('p', {}, 'Scored and waiting for you to collect: ', ...collect.map(r => [E('button', { class: 'plain', disabled: !I.enabled, title: I.why, on: { click: async (e) => { try { const x = await act('POST', `/api/challenges/${enc(c.id)}/runs/${enc(r.id)}/collect`); e.target.replaceWith(E('span', {}, `${r.label}: Ξ ${signed(x.delta_pp)} pp, in review`)); CACHE.clear(); } catch (err) { e.target.replaceWith(E('span', { class: 's-failed' }, err.message)); } } } }, `Collect ${r.label}`), ' '])));
} else v.append(E('p', { class: 'muted' }, ME && ME.oauth_enabled ? [signIn('Sign in'), ' to see your collections and start runs.'] : ME && ME.unchecked ? UNCHECKED : 'Your collections appear here once you are signed in on the arenaβs Space.'));
if ((B.queue || []).length) v.append(E('h2', {}, 'Waiting to start'), table([['Collection'], ['Team'], ['In line', 'r'], ['Asked', 'hide-s']], B.queue.map(q => row(colHref(q.collection_id), [cell(A(q.title || q.collection_id, colHref(q.collection_id))), cell(q.team), cell(q.position, 'r'), cell(when(q.requested_at), 'hide-s nw')]))));
// everyone's runs
v.append(E('h2', {}, 'All runs'));
const input = E('input', { type: 'search', placeholder: 'Find a run, collection or team', value: P.get('q') || '', style: 'min-width:260px' }), sel = E('select', { 'aria-label': 'Status' }, ...[['', 'Every status'], ['verified', 'Counted'], ['review', 'Scored, not counted yet'], ['running', 'Running'], ['failed', 'Failed'], ['canceled', 'Canceled']].map(([k, l]) => E('option', { value: k, selected: (P.get('status') || '') === k }, l))), holder = E('div');
const draw = () => { const q = input.value.toLowerCase(), f = list.filter(r => (!q || [r.label, r.id, r.collection_title, r.team, r.author].join(' ').toLowerCase().includes(q)) && (!sel.value || runState(r)[0] === sel.value || (sel.value === 'review' && runState(r)[0] === 'rejected')));
holder.replaceChildren(table([['Run'], ['Collection'], ['Status'], ['Ξ Β± SE, pp', 'r'], ['Started', 'hide-s']], f.map(r => row(subHref(r.id), [cell(A(r.label, subHref(r.id))), cell(E('span', {}, r.collection_title || r.title || 'organizer test run', E('span', { class: 'reason' }, r.team || r.author || ''))), cell(E('span', {}, stateEl(r), E('span', { class: 'reason' }, reasonLine(r)))), cell(r.delta_pp != null ? dse(r.delta_pp, r.stderr_pp) : 'β', 'r'), cell(when(r.started_at || r.created_at), 'hide-s nw')])), list.length ? 'No run matches.' : 'No run yet.', null, 'wide')); };
input.oninput = () => { setQs({ q: input.value }); draw(); }; sel.onchange = () => { setQs({ status: sel.value }); draw(); };
v.append(E('div', { class: 'filters' }, input, sel), holder); draw();
}
function checkResult(r) {
const tasks = Object.entries((r.quality_gates || {}).tasks || {}), box = E('div', { class: 'box' });
box.append(E('p', {}, E('b', {}, `${plural(r.task_count || tasks.length, 'task')} at commit ${String(r.revision || '').slice(0, 12)}. `), (r.quality_gates || {}).summary ? `${r.quality_gates.summary.eligible} eligible (${r.quality_gates.summary.needs_controls} of them need controls), ${r.quality_gates.summary.blocked + r.quality_gates.summary.rejected} excluded.` : ''));
if ((r.errors || []).length) box.append(E('p', { class: 's-failed' }, 'Errors: '), E('ul', {}, r.errors.map(x => E('li', {}, x))));
if ((r.warnings || []).length) box.append(E('details', {}, E('summary', {}, plural(r.warnings.length, 'warning')), E('ul', {}, r.warnings.map(x => E('li', { class: 'small' }, x)))));
box.append(table([['Task'], ['Outcome'], ['Findings']], tasks.map(([name, t]) => {
const sev = new Set((t.findings || []).map(x => x.severity)), out = sev.has('block') || sev.has('reject') ? 'excluded' : t.needs_controls ? 'needs controls' : 'passed static checks';
return row(null, [cell(E('span', { class: 'mono' }, name)), cell(E('span', { class: 'state s-' + (out === 'excluded' ? 'excluded' : out === 'needs controls' ? 'review' : 'ok') }, out)), cell((t.findings || []).map(x => E('div', { class: 'small' }, `${x.code}: ${x.message}`)))]);
})));
return box;
}
function checksList(r) { return E('div', { class: 'box ' + (r.allowed ? '' : 'warn') }, E('p', {}, E('b', {}, r.allowed ? 'Every check passes; you can start a run.' : 'A check fails; the run would be refused.')), E('ul', {}, (r.checks || []).map(x => E('li', {}, E('span', { class: 'state s-' + (x.ok ? 'ok' : x.ok === false ? 'failed' : 'none') }, x.ok ? 'ok' : x.ok === false ? 'fails' : 'not checked'), ` ${x.name}: ${x.detail}`)))); }
// ββ one run: the competition's record of it βββββββββββββββββββββββββββββββββββββ
const STAGE_WHAT = { setup: 'start the GPU job and the model server', snapshot: 'copy the collectionβs tasks and the held-out suite into the run', baseline: 'the untrained model on the held-out suite',
gate: 'the untrained model on the training tasks, once each', training: 'GRPO on the training tasks', heldout: 'the trained model on the held-out suite', collect: 'the author collects the result; the arena recomputes the score from per-task results' };
const STAGE_STATE = { done: 'done', active: 'running', running: 'running', failed: 'failed', canceled: 'canceled', pending: 'not started', unreached: 'not reached', skipped: 'skipped' };
function stageResult(s) {
const T = s.training, v = (T && T.rollout_verdicts) || {}, tried = (v.pass || 0) + (v.fail || 0) + (v.error || 0);
if (T) return tried ? `${v.pass || 0} of ${plural(tried, 'attempt')} passed${v.error ? `, ${v.error} errored` : ''}, counting retries` : 'no attempt finished';
const extra = `${s.errors ? `, ${plural(s.errors, 'error')}` : ''}${s.timeouts ? `, ${s.timeouts} over the time limit` : ''}`;
if (s.total && s.done != null && s.done < s.total) return `${s.done} of ${s.total} done${s.state === 'active' || s.state === 'running' ? ' so far' : ''}: ${s.pass} passed${extra}`;
if (s.total) return `${s.pass} of ${s.total} passed${extra}`;
return '';
}
const partial = (s) => s && s.total && s.done != null && s.done < s.total; // an evaluation that stopped (or is still going) before every task was tried
function gateText(r) { const g = (r.stages || []).find(s => s.key === 'gate'); return !g || !g.total ? 'β' : partial(g) ? `${g.pass} of ${g.done} passed (${g.total} planned)` : `${g.pass} of ${g.total} passed`; }
async function runPage(v, id) {
const r = await api('runs/' + enc(id)), c = chById(r.challenge_id);
if (!c) throw new Error(`run ${id} belongs to ${r.challenge_id}, which this data source does not list.`);
setCurrent(c.id); setTitle(`Run ${r.label} Β· ${c.id}`); v.append(frame(c, 'runs'));
const [k] = runState(r), hit = issueFor(r.reason), n = suiteN(c), gate = (r.stages || []).find(s => s.key === 'gate'), rec = rulesOf(c).recipe || {};
v.append(E('p', { class: 'small' }, A('All runs', chHref(c.id, 'runs'))), E('div', { class: 'titlerow' }, E('h2', {}, `Run ${r.label}`)));
v.append(E('p', {}, stateEl(r), ' Β· ', r.collection_id ? A(r.collection_title || r.collection_id, colHref(r.collection_id)) : 'organizer test run', [r.team || r.author, `started ${when(r.started_at || r.created_at)}${r.ended_at ? ', ended ' + when(r.ended_at) : ''}`].filter(Boolean).map(x => ' Β· ' + x).join('')));
if (r.note) v.append(E('p', { class: 'muted' }, E('b', {}, 'Organizersβ note. '), r.note));
if (k === 'failed' || k === 'canceled') {
const stage = (r.stages || []).find(s => s.state === 'failed' || s.state === 'canceled'), at = stage ? stage.key : r.stage || 'the run', own = !!r.collection_id;
// a canceled run was stopped by someone; its last error line is not necessarily why, so it is shown without a cause
const cause = k === 'canceled' ? '' : hit ? hit.cause : r.reason ? 'Its cause is not among the organizersβ known issues yet.' : '';
const where = at === 'the run' ? 'It stopped' : `It ${k === 'canceled' ? 'was canceled' : 'stopped'} during the ${at} stage`;
v.append(E('div', { class: k === 'canceled' ? 'box' : 'card-bad' }, E('h2', {}, k === 'canceled' ? 'The run was canceled.' : 'The run couldnβt finish.'),
E('p', {}, `${where}. `, cause),
E('p', {}, own ? 'Nothing was scored, and it does not count toward the collectionβs daily run. ' : 'Nothing was scored. ',
k === 'failed' && hit && hit.blame === 'platform' ? (own ? 'The organizersβ known issues list this as a platform fault, not the collectionβs; the run can be repeated once it is fixed. ' : 'The organizersβ known issues list this as a platform fault. ')
: k === 'failed' && hit && hit.blame === 'collection' ? 'The organizersβ known issues list this as a problem in the collection. ' : '', k === 'failed' && hit && hit.fixed ? hit.fixed : ''),
r.reason ? [E('p', { class: 'small muted', style: 'margin-bottom:2px' }, k === 'canceled' ? 'The last error line in the jobβs log before it was canceled:' : 'The last error line in the jobβs log:'), E('pre', { style: 'margin-top:0' }, r.reason)] : '', E('p', { class: 'small muted' }, 'Why runs stop: ', A('known issues', chHref(c.id, 'rules') + '?at=known-issues'), '.')));
}
if (k === 'running') { const cur = (r.stages || []).find(s => s.state === 'active' || s.state === 'running');
if (cur) v.append(E('div', { class: 'box' }, E('p', {}, E('b', {}, `Now at ${cur.key}: `), `${STAGE_WHAT[cur.key] || ''}${cur.total ? `; ${cur.done ?? 0} of ${cur.total} tasks done` : ''}.`, cur.started_at ? ` This stage has run for ${dur((nowMs() - Date.parse(cur.started_at)) / 1000)}${r.started_at ? `, the whole run for ${dur((nowMs() - Date.parse(r.started_at)) / 1000)}` : ''}.` : ''))); }
const all = (r.train_task_count || 0) + (r.excluded_task_count || 0);
const one = tasksPerStep(c) === 1 ? '; this recipe trains on one task drawn from them' : '';
const trained = r.excluded_task_count ? `${r.train_task_count || 0} of the collectionβs ${all}; the static checks excluded ${r.excluded_task_count}${one}` : r.excluded_task_count === 0 ? `all ${r.train_task_count}; the static checks excluded none${one}`
: r.train_task_count != null ? `all ${r.train_task_count} of the collectionβs tasks: this run started before runs left excluded tasks out${one}` : null;
const stopped = (k === 'failed' || k === 'canceled') && r.before != null && r.after == null;
const held = [['Before training', r.before != null ? `${frac(r.before, n)} passed` : null], ['After training', r.after != null ? `${frac(r.after, n)} passed` : stopped ? 'not measured: the run stopped before it scored the trained model' : null],
['Ξ', r.delta_pp != null ? [dse(r.delta_pp, r.stderr_pp), ((rulesOf(c).eval_suite || {}).sealed === false ? ' pp' : ' pp (the sealed task names stay private)')] : null],
['Review', r.state === 'scored' ? [r.verification === 'valid' ? 'verified' : r.verification === 'invalid' ? 'rejected' : r.verification === 'pending' ? 'collected, awaiting an organizer' : 'not collected yet', r.verification_note ? ` β ${r.verification_note}` : ''] : null]];
if (held.some(([, x]) => x != null)) v.append(E('h2', {}, 'Score on the held-out suite'), dl(held));
if (gate && gate.total) v.append(E('h2', {}, 'Base-model gate'), E('p', {}, `Before training, the untrained model tried ${partial(gate) ? `${gate.done} of the ${gate.total} planned` : gate.total} training tasks once each and passed ${gate.pass}${partial(gate) && gate.state !== 'active' ? `; the gate ${gate.state === 'canceled' ? 'was canceled' : 'stopped'} before the rest` : ''}. `, E('span', { class: 'muted' }, rec.run_policy === 'always' ? 'Its score is only reported and never stops a run; the stage itself can still fail, for example when too many attempts lose their sandbox.' : 'Its score must pass for training to start.')));
v.append(E('h2', {}, 'Stages'), table([['Stage'], ['State'], ['Started', 'hide-s'], ['Took', 'r'], ['Result']], (r.stages || []).map(s => row(null, [cell(E('span', {}, s.key, E('span', { class: 'reason' }, STAGE_WHAT[s.key] || ''))), cell(E('span', { class: { failed: 's-failed', canceled: 's-canceled', active: 's-running', running: 's-running' }[s.state] || null }, s.key === 'collect' && s.state === 'pending' && r.state === 'scored' ? 'not collected yet' : STAGE_STATE[s.state] || s.state)), cell(when(s.started_at), 'hide-s nw'),
cell((s.state === 'active' || s.state === 'running') && s.started_at ? `${dur((nowMs() - Date.parse(s.started_at)) / 1000)} so far` : dur(s.duration_s), 'r nw'), cell(stageResult(s))]))));
v.append(E('h2', {}, 'Details'), dl([['Full run id', E('span', { class: 'mono' }, r.id)], ['Eligible tasks', trained],
['Cost', r.cost_usd != null ? usd(r.cost_usd) : r.job && r.job.cost_usd != null ? usd(r.job.cost_usd) : 'not settled yet'],
['HF job', r.job && r.job.url ? A(r.job.id || r.job.url, r.job.url, 'mono') : null]]));
const others = r.siblings || [];
if (others.length) v.append(E('h2', {}, `Other runs of ${r.collection_title || 'this collection'} on ${c.id}`), table([['Run'], ['Status'], ['Ξ Β± SE, pp', 'r'], ['Ended', 'hide-s']], [...others].reverse().map(x => row(subHref(x.id), [cell(runLink(x)), cell(E('span', {}, stateEl(x), E('span', { class: 'reason' }, reasonLine(x)))), cell(x.delta_pp != null ? signed(x.delta_pp) : 'β', 'r'), cell(when(x.ended_at) || 'β', 'hide-s nw')]))));
}
// ββ Rules: the plain summary, then every number the challenge runs under ββββββββββββ
async function rulesPage(v, c) {
setTitle(`Rules Β· ${c.id}`); v.append(frame(c, 'rules'));
if (c.status !== 'open') { v.append(plannedPage(c)); return; }
const R = rulesOf(c), m = R.base_model || {}, rec = R.recipe || {}, s = R.eval_suite || {}, cp = R.compute || {}, met = R.metric || {}, h = rec.harness || {}, b = (BOARD && BOARD.budget) || {}, B = boardOf(c.id) || {}, reserve = B.reserve_usd ?? R.reserve_usd;
v.append(E('p', { class: 'muted' }, 'Everything a run of this challenge is held to. The numbers come from the challengeβs config, the same file the arena runs.'));
v.append(E('h2', {}, 'In short'), E('ul', {},
E('li', {}, `Every run trains the same model, ${m.repo_id}, with the same recipe; only your tasks differ.`),
E('li', {}, `Your score is the held-out pass rate after training minus before, in percentage points, on ${s.task_count} held-out tasks the run never trains on. Both are measured inside the same run, ${plural(met.trials_per_run || 1, 'attempt')} per task.`),
E('li', {}, 'A collection is ranked by the mean change over all its organizer-verified runs, not its best run, so running more often does not help.'),
E('li', {}, `One run at a time in the whole arena, and ${plural(cp.runs_per_submission_per_day || 1, 'counted run')} per collection per 24 hours. Failed and canceled runs do not count.`),
E('li', {}, `Runs draw on one shared compute budget: ${b.cap_usd != null ? `${usd(b.remaining_usd)} of ${usd(b.cap_usd)} is left, ` : ''}${reserve != null ? `and each run reserves ${usd(reserve)} until it ends, when it is charged its actual cost` : 'and a run reserves its price once its compute provider is connected'}. The budget does not reset; when what is left cannot cover a reservation, no run can start.`),
E('li', {}, 'The static checks leave out of training any task that leaks the answer or overlaps the held-out suite. A task whose verifier looks weak, for example one that only checks that files exist, is flagged for review but still trained on.')));
v.append(E('h2', {}, 'In full'), dl([['Model', m.repo_id ? `${m.repo_id} at revision ${String(m.revision || '').slice(0, 12)}` : c.model], ['Recipe', rec.method ? `${rec.id}: ${rec.method}` : c.method],
['Training', rec.max_steps != null ? `${plural(rec.max_steps, 'optimizer step')}; each step trains on ${rec.num_generations} attempts at ${tasksPerStep(c) > 1 ? `each of ${tasksPerStep(c)}` : 'one'} of your tasks; learning rate ${rec.learning_rate}${(c.method_info || {}).status === 'planned' ? ' (placeholder values: the organizers have not set this recipe yet)' : ''}` : null],
['Which task', !rec.num_generations ? null : tasksPerStep(c) > 1 ? 'every eligible task (the ones the static checks did not exclude), drawn in a fixed-seed order that covers them all' : 'drawn with a fixed seed from your eligible tasks (the ones the static checks did not exclude), so every run of one commit trains on the same task'],
['Retries', rec.rollout_attempts ? `an attempt that fails to finish (for example a timeout) is retried ${rec.rollout_attempts === 2 ? 'once' : plural(rec.rollout_attempts - 1, 'time')}` : null],
['When every attempt scores the same', rec.require_reward_variance ? 'the run stops: GRPO learns from differences between attempts, so there is nothing to learn' : null],
['Base-model gate', rec.gate_task_count ? `before training, the untrained model tries up to ${rec.gate_task_count} of your tasks once each; its score is reported and ${rec.run_policy === 'always' ? 'never stops the run, though the stage itself can fail on infrastructure errors' : 'must pass for training to start'}` : null],
['Agent', h.agent ? `${h.agent}, ${h.concurrency} tasks at a time, ${h.agent_timeout_sec} s per task` : null],
['Held-out suite', s.name ? `${s.name}: ${s.task_count} tasks, ${plural(met.trials_per_run || 1, 'attempt')} per task per run${s.sealed === false ? ' (a public benchmark: the static checks block copies of its tasks)' : ''}` : null],
['Compute per run', cp.flavor ? `${gpus(cp.flavor, cp)}, ${cp.timeout_seconds / 3600} h job timeout; ${reserve != null ? `${usd(reserve)} reserved until the run ends` : 'no price is quoted until the provider is connected'}` : c.compute],
['Same for every run', cp.resources && cp.resources.gpus ? `${cp.resources.gpus} ${cp.resources.gpu_type || ''} GPUs, ${cp.timeout_seconds / 3600} h, up to ${cp.resources.sandbox_concurrency} sandboxes at once (each at most ${cp.resources.sandbox_max_vcpu} vCPU and ${cp.resources.sandbox_max_memory_gb} GB), ${plural(cp.resources.eval_trials || 1, 'evaluation trial')}` : null],
['Review', 'an organizer checks each collected result (per-task outcomes, the training update, train/eval isolation) before it counts'],
['Window', R.opens ? `${opensWord(R.opens)} ${R.opens}${R.closes ? ', closes ' + R.closes : ', no closing date yet'}` : null]]));
if (rec.note || s.note) v.append(E('h2', {}, 'Notes from the organizers'), rec.note ? E('p', {}, E('b', {}, 'Recipe. '), rec.note) : '', s.note ? E('p', {}, E('b', {}, 'Suite. '), s.note) : '');
const known = (META.known_issues || []);
if (known.length) v.append(E('h2', { id: 'known-issues' }, 'Known issues'), E('p', { class: 'muted small' }, 'Why runs have stopped, in the organizersβ words. A runβs page shows the matching entry. Platform faults are the arenaβs; the collection is not at fault and the run can be repeated.'),
E('ul', {}, known.map(i => E('li', {}, E('b', {}, i.blame === 'platform' ? 'Platform. ' : i.blame === 'collection' ? 'Collection. ' : ''), i.cause, i.fixed ? ` ${i.fixed}` : ''))));
}
// ββ Submissions: every submitted collection, what the checks found, what its runs did, and its verified result βββββ
function collState(s) { // [class, words, detail]: the collection's arena status (app_api.status_of) in the board's words
const runs = s.runs || [], last = runs[runs.length - 1], live = runs.find(r => r.state === 'running' || r.state === 'queued');
switch (s.status) {
case 'training': case 'running': return ['running', 'running', live ? `${live.challenge_id}, at ${live.stage || 'setup'}` : ''];
case 'queued': return ['running', 'queued', (s.queued || []).join(', ')];
case 'in review': return ['review', 'in review', 'a collected run waits for an organizer'];
case 'ranked': return ['verified', 'ranked', ''];
case 'failed': return ['failed', 'last run failed', last ? `on ${last.challenge_id}${last.stage ? ', at ' + last.stage : ''}` : ''];
case 'stopped': return !last ? ['none', 'stopped', ''] : last.state === 'canceled' ? ['canceled', 'last run canceled', `on ${last.challenge_id}`] : last.verification === 'invalid' ? ['rejected', 'last run rejected', 'in review'] : ['review', 'scored, not collected', `on ${last.challenge_id}`];
default: return ['none', 'not run yet', ''];
}
}
const unread = (s) => (s.task_count || 0) > 0 && !Object.values(s.sample_outcomes || {}).some(Boolean); // the static checks' per-task results did not reach this data source
const unreadNote = (list) => list.length ? E('p', { class: 'small muted' }, 'No per-task results in this data source for ', list.map((s, i) => [i ? '; ' : '', A(s.title, colHref(s.id)), `, ${plural(s.task_count, 'task')}`]), `. The static checksβ report for ${list.length === 1 ? 'its' : 'each oneβs'} pinned commit could not be read, so ${list.length === 1 ? 'its' : 'their'} tasks are not listed or counted here.`) : '';
const STATIC = [['eligible', 'Passed', 'passed'], ['flagged', 'Flagged', 'flagged'], ['needs controls', 'Needs controls', 'need controls'], ['excluded', 'Excluded', 'excluded']];
function staticLine(s) {
const o = s.sample_outcomes || {}, known = STATIC.reduce((a, [k]) => a + (o[k] || 0), 0);
if (!known) return E('p', { class: 'muted' }, 'No per-task results for this collection in this data source: the static checksβ report for its pinned commit could not be read, so its tasks are not listed.');
return E('p', {}, `${plural(s.task_count || known, 'task')}: `, STATIC.filter(([k]) => o[k]).map(([k, , words], i) => [i ? ', ' : '', E('span', { class: 'state s-' + outcomeClass(k) }, `${o[k]} ${o[k] === 1 && k === 'needs controls' ? 'needs controls' : words}`)]), '.');
}
const day = (iso) => { const d = new Date(iso); return !iso || isNaN(d) ? '' : `${MONTHS[d.getUTCMonth()]} ${d.getUTCDate()}`; };
const fits = (n) => n === 0 ? 'no other run fits' : n === 1 ? 'one more fits' : `${n} more fit`;
// A row's cells, from app_api's submissions fields. The checks: [class, words, detail]. Every stored collection passed the
// structure check when it was submitted (the registry marks it Validated); the static checks read every task's files.
function checksOf(s) {
const k = s.checks || {}, st = k.static, out = st ? (st.blocked || 0) + (st.rejected || 0) : 0;
const structure = k.structure === 'Validated' ? 'structure validated' : k.structure ? `structure: ${String(k.structure).toLowerCase()}` : 'structure not recorded';
if (!st) return ['none', 'static checks not read', structure];
return [out ? 'review' : 'ok', `${st.eligible} of ${plural(st.tasks, 'task')} eligible`,
[structure, out && `${out} excluded`, st.needs_controls && `${st.needs_controls} need${st.needs_controls === 1 ? 's' : ''} controls`, st.review && `${st.review} flagged`].filter(Boolean).join(' Β· ')];
}
function runsOf(s) { // [how many, the latest in plain words]
const r = s.latest_run, n = (s.run_counts || {}).runs || 0;
if (!r) return [plural(n, 'run'), 'none yet'];
const [k, words] = runState(r);
return [plural(n, 'run'), [E('span', { class: 's-' + k }, `latest ${words}`), `, ${day(r.started_at || r.created_at)}`]];
}
function resultOf(s) { // [its best verified change, or why it has none; one line of detail]
const b = s.best, n = s.run_counts || {};
if (b) return [E('span', { class: 'nw' }, dse(b.delta_pp, b.stderr_pp), ' pp'), `${b.challenge_id}, ${plural(b.verified_runs, 'verified run')}${b.rank ? `, rank ${b.rank}` : ''}`];
const waiting = [[n.in_review, 'in review'], [n.uncollected, 'not collected'], [n.rejected, 'rejected']].filter(([x]) => x).map(([x, w]) => `${x} ${w}`).join(', ');
if (n.scored && n.rejected === n.scored) return [E('span', { class: 'state s-rejected' }, 'rejected in review'), `${plural(n.scored, 'scored run')}, none verified`];
if (n.scored) return [E('span', { class: 'state s-review' }, 'not verified yet'), waiting];
return [E('span', { class: 'muted' }, 'not scored'), n.runs ? `none of its ${plural(n.runs, 'run')} scored` : 'no run yet'];
}
function sourceOf(s) { // the repository or dataset at the commit it was submitted at
if (!s.repo_id) return 'β';
const kind = { github: 'GitHub', dataset: 'HF dataset' }[s.repo_type] || '', rev = String(s.revision || '').slice(0, 12);
return E('span', {}, s.source_url ? A(s.repo_id, s.source_url, 'mono') : E('span', { class: 'mono' }, s.repo_id),
E('span', { class: 'reason' }, [kind, rev && `at ${rev}`, s.entry_path && `folder ${s.entry_path}`].filter(Boolean).join(' Β· ')));
}
// the arena as its data says, above the list: what was submitted and run, whether a run can start now, the budget left
function arenaNow(subs) {
const open = CHS().filter(c => c.status === 'open'), read = open.every(c => boardOf(c.id)), stats = open.map(c => (boardOf(c.id) || {}).stats || {}), sum = (k) => stats.reduce((a, x) => a + (x[k] || 0), 0);
const runs = read ? sum('runs') : ((META && META.counts) || {}).runs, scored = sum('scored'), org = sum('organizer_runs'), tasks = subs.reduce((a, s) => a + (s.task_count || 0), 0);
const box = E('div', { class: 'box' + (open.some(c => !(boardOf(c.id) || {}).accepting_runs) ? ' warn' : '') }, E('p', {}, E('b', {}, `${plural(subs.length, 'collection')} submitted, ${plural(tasks, 'task')}. `),
!runs ? 'No run yet.' : !read ? `${plural(runs, 'run')} so far; how many were scored could not be read right now.`
: `${plural(runs, 'run')} so far${org ? ` (${org} of them organizer test runs)` : ''}: ${scored ? `${scored} scored, ${sum('verified')} verified` : 'none scored'}.`));
for (const c of open) {
const B = boardOf(c.id);
const id = A(c.id, chHref(c.id));
box.append(E('p', {}, ...(!B ? ['Whether a run can start on ', id, ' could not be read right now.']
: B.runs_paused ? [E('b', {}, 'Runs on ', id, ' are paused by the organizers: '), first(B.runs_paused), ' Submitting and checking collections still work.']
: B.accepting_runs ? [E('b', {}, id, ' takes runs now. '), 'The arena runs one at a time.']
: [E('b', {}, 'No run can start on ', id, ' right now. '), first(B.reason || '')])));
}
const b = (BOARD && BOARD.budget) || {}, R = open.map(c => boardOf(c.id)).find(B => B && B.reserve_usd != null);
if (b.cap_usd != null) box.append(E('p', {}, `Compute budget: ${usd(b.remaining_usd)} of ${usd(b.cap_usd)} left`, R ? `; a run reserves ${usd(R.reserve_usd)}${R.runs_that_fit != null ? `, so ${fits(R.runs_that_fit)}` : ''}.` : '.'));
return box;
}
async function submissionsPage(v) {
const subs = await api('submissions'), P = qs(), c = mainCh(), I = c ? await identity(c) : { you: null, live: SOURCE === 'live' };
v.append(...page('Submissions'), E('p', { class: 'lede' }, 'Every collection submitted to the arena, pinned to the commit it was submitted at: what the checks found in its tasks, what its runs did, and its verified result.'),
E('p', { class: 'small' }, A('Submit a collection', '#/submit'), ' Β· ', A('Starter kit', '#/starter'), ' Β· ', A('Instructions for agents', '/AGENTS.md'), ' (AGENTS.md)'), arenaNow(subs));
const counts = {}; subs.forEach(s => { const w = collState(s)[1]; counts[w] = (counts[w] || 0) + 1; });
const input = E('input', { type: 'search', placeholder: 'Find a collection, author or repository', 'aria-label': 'Find a collection', value: P.get('q') || '', style: 'min-width:280px' }), holder = E('div');
const sel = E('select', { 'aria-label': 'State' }, E('option', { value: '' }, `Every state (${subs.length})`), ...Object.entries(counts).map(([w, n]) => E('option', { value: w, selected: P.get('state') === w }, `${w} (${n})`)));
const on = () => !!I.you && mine.getAttribute('aria-pressed') === 'true';
const mine = E('button', { class: 'plain', disabled: !I.you, 'aria-pressed': String(!!I.you && P.get('mine') === '1'),
title: I.you ? (I.live ? `Only the collections ${I.you} submitted` : `Simulated data: only team ${I.you}βs collections`) : I.why || 'Sign in first.' }, 'Yours'); // I.why: why not signed in (identity)
const draw = () => {
const q = input.value.toLowerCase(); mine.classList.toggle('on', on());
const f = subs.filter(s => (!q || [s.title, s.author, s.team, s.repo_id, s.entry_path, s.revision].join(' ').toLowerCase().includes(q)) && (!sel.value || collState(s)[1] === sel.value) && (!on() || (s.author || s.team) === I.you));
holder.replaceChildren(table([['Submission'], ['Tasks', 'r'], ['Source'], ['Checks'], ['Runs'], ['Verified change']], f.map(s => {
const [ck, cw, cd] = checksOf(s), [rn, rd] = runsOf(s), [bv, bd] = resultOf(s);
return row(colHref(s.id), [cell(E('span', {}, A(s.title, colHref(s.id)), E('span', { class: 'reason' }, [s.author || s.team, `submitted ${when(s.created_at)}`].filter(Boolean).join(' Β· ')))),
cell(s.task_count, 'r'), cell(sourceOf(s)), cell(E('span', {}, E('span', { class: 'state s-' + ck }, cw), E('span', { class: 'reason' }, cd))),
cell(E('span', {}, rn, E('span', { class: 'reason' }, rd))), cell(E('span', {}, bv, bd ? E('span', { class: 'reason' }, bd) : ''))]);
}), subs.length ? (on() ? `No collection by ${I.you} matches.` : 'No collection matches.') : 'No collection has been submitted yet.', null, 'wide'));
};
input.oninput = () => { setQs({ q: input.value }); draw(); }; sel.onchange = () => { setQs({ state: sel.value }); draw(); };
mine.onclick = () => { mine.setAttribute('aria-pressed', String(!on())); setQs({ mine: on() ? '1' : '' }); draw(); };
v.append(E('div', { class: 'filters' }, input, mine, sel), holder, unreadNote(subs.filter(unread)),
E('p', { class: 'small muted' }, 'Eligible tasks are the ones the static checks did not exclude; runs now train only on them. The verified change is the mean change in the held-out pass rate over a collectionβs organizer-verified runs on one challenge, in percentage points (pp) Β± one standard error; a collection that ranks on several challenges shows its best.'));
draw();
}
// ββ a submission's page: where it stands, its results, runs and checks, the PostTrain commands, its tasks βββββββββ
const headHeight = () => { const hd = $('header'); document.documentElement.style.setProperty('--head', `${getComputedStyle(hd).position === 'sticky' ? hd.offsetHeight + 12 : 12}px`); }; // h2[id]'s scroll margin: clear of the header while it stays on screen
addEventListener('resize', headHeight);
function whyStopped(r) { // why a run ended, in the organizers' words when they have them (configs/known_issues.toml)
const [k] = runState(r), hit = issueFor(r.reason);
if (k === 'running') return 'still running';
if (r.state === 'scored') return 'it finished and was scored';
if (k === 'canceled') return r.started_at ? 'canceled before it finished' : 'canceled before it started';
if (k !== 'failed') return 'β';
if (hit) return [first(hit.cause), hit.blame === 'platform' ? ' The organizers list it as a platform fault.' : hit.blame === 'collection' ? ' The organizers list it as a problem in the collection.' : ''];
return r.reason ? ['an error that is not among the known issues yet: ', E('span', { class: 'mono' }, r.reason.slice(0, 160))] : 'no reason was recorded';
}
function runsTable(runs) { // newest first: its state, the stage it reached, why it stopped, what it cost
return table([['Run'], ['Challenge'], ['State'], ['Stage reached'], ['Why it stopped'], ['Cost', 'r'], ['Started', 'hide-s']], [...runs].reverse().map(r => {
const [k] = runState(r);
return row(subHref(r.id), [cell(runLink(r)), cell(A(r.challenge_id, chHref(r.challenge_id))), cell(stateEl(r)),
cell(r.stage ? E('span', {}, r.stage, E('span', { class: 'reason' }, r.stage === 'training' && tasksPerStep(chById(r.challenge_id) || {}) === 1 ? 'GRPO on one task drawn from the collection' : STAGE_WHAT[r.stage] || '')) : E('span', { class: 'muted' }, 'none')),
cell(E('span', { class: 'small' }, whyStopped(r))), cell(r.cost_usd != null ? usd(r.cost_usd) : k === 'running' ? 'not settled yet' : 'β', 'r nw'),
cell(r.started_at ? when(r.started_at) : E('span', { class: 'muted' }, `never; asked ${when(r.created_at)}`), 'hide-s nw')]);
}), 'No run yet.', null, 'wide');
}
function standing(s, listed) { // one paragraph: its state, its result or why it has none, whether its challenge takes runs
const [k, w, why] = collState(listed), n = s.run_counts || {}, c = mainCh(), B = c && boardOf(c.id), ended = [[n.failed, 'failed'], [n.canceled, 'canceled'], [n.running, 'running']].filter(([x]) => x).map(([x, t]) => `${x} ${t}`).join(', ');
const parts = [E('span', { class: 'state s-' + k }, cap(w)), why ? ` (${why})` : '', '. '];
if (s.best) parts.push('Verified: ', dse(s.best.delta_pp, s.best.stderr_pp), ` pp on ${s.best.challenge_id}. `);
else if (n.scored) parts.push(n.rejected === n.scored ? `${plural(n.scored, 'run')} scored, and rejected in review. ` : `${plural(n.scored, 'run')} scored, none verified yet. `);
else if (n.runs) parts.push(`Not scored: none of its ${plural(n.runs, 'run')} reached a score${ended ? ` (${ended})` : ''}. `);
else parts.push('Nothing is scored until a run finishes. ');
if (B && B.runs_paused) parts.push(`Runs on ${c.id} are paused by the organizers.`);
return E('div', { class: 'box' }, E('p', {}, ...parts));
}
function controlsLine(k) { // the verdicts organizers attached per challenge (store.collection_checks), or what the controls are
const v = Object.entries(k.controls || {});
if (!v.length) return 'none recorded yet. An organizer runs them, not the arena: the taskβs image builds, its reference solution (if it has one) scores 1, doing nothing scores 0, and the untrained model solves it in 1 to 3 of 4 attempts (the defaults). Runs donβt wait for them.';
return v.map(([cid, x]) => { const n = x.summary || {};
return `${cid}: ${n.accepted ?? 0} of ${plural(n.tasks ?? 0, 'task')} accepted${n.rejected ? `, ${n.rejected} rejected` : ''}${n.inconclusive ? `, ${n.inconclusive} inconclusive` : ''}${x.attached_at ? ` (attached ${when(x.attached_at)}${x.attached_by ? ' by ' + x.attached_by : ''})` : ''}`; }).join('; ');
}
function checksDl(s) {
const k = s.checks || {}, st = k.static, cr = k.credit, out = st ? (st.blocked || 0) + (st.rejected || 0) : 0, missing = Object.entries((cr && cr.missing) || {});
return dl([['Structure', k.structure === 'Validated' ? 'validated when it was submitted: every task has the files a run needs (task.md with a prompt, environment/Dockerfile, verifier/test.sh β¦)' : k.structure || 'not recorded'],
['Static checks', st ? [E('b', {}, `${st.eligible} of ${plural(st.tasks, 'task')} eligible`), '; ', [out ? `${out} excluded: runs now leave them out of training` : 'none excluded', st.needs_controls && `${st.needs_controls} need${st.needs_controls === 1 ? 's' : ''} controls (no working reference solution)`,
st.review && `${st.review} flagged for review`].filter(Boolean).join('; '), '. They read every taskβs files at submission, without running anything.'] : 'the report for its pinned commit could not be read'],
['Controls', controlsLine(k)],
['Credit metadata', cr ? `${cr.complete} of ${plural(cr.tasks, 'task')} declare every credit field (author name and email, license, category, origin)${missing.length ? `; missing: ${missing.map(([f, x]) => `${f.replace(/_/g, ' ')} on ${x}`).join(', ')}` : ''}` : null]]);
}
function postTrain(p) { // the commands of PostTrain's guide for this collection (app_api: posttrain_path.py); instructions, no results
const pre = (lines) => { const text = lines.join('\n');
return E('div', { class: 'cmdbox' }, E('div', { class: 'cmdbar' }, E('button', { class: 'plain copy', title: 'Copy these commands', on: { click: async (e) => { const b = e.currentTarget;
try { await navigator.clipboard.writeText(text); b.textContent = 'Copied'; } catch { b.textContent = 'Select to copy'; } setTimeout(() => { b.textContent = 'Copy'; }, 2000); } } }, 'Copy')), E('pre', { class: 'cmd' }, text)); };
const step = (title, ...kids) => E('li', {}, E('b', {}, title), ...kids), sp = p.split, k = p.costs || {};
// what the dry runs estimate (posttrain_path.estimate: PostTrain 0.1.11's figures for these commands, at the tasks each eval scores; the time is the page's own)
const est = (e) => `about ${usd(e.usd)} (${usd(e.low)}β${usd(e.high)})`, hm = (m) => m < 90 ? `${m} min` : `${(m / 60).toFixed(1).replace(/\.0$/, '')} h`;
const span = ([lo, hi]) => hi < 90 ? `${lo} to ${hi} minutes` : lo >= 90 ? `${hm(lo).slice(0, -2)} to ${hm(hi).slice(0, -2)} hours` : `${hm(lo)} to ${hm(hi)}`;
const out = [E('h2', { id: 'posttrain' }, 'Improve a model on it with PostTrain'),
E('p', {}, 'PostTrain evaluates a model on these tasks (an agent attempts each one in the taskβs own sandbox, and the taskβs verifier grades it) and can hill-climb on them: a stronger modelβs verified attempts at some of the tasks train a smaller model, which is then compared with its base on the tasks it didnβt train on. These are the commands of the guideβs verified path, filled in for this collection with the guideβs models. None has run on this collection, so no result is shown. ',
A('The guide: Sandboxed agent environments', p.guide), '.')];
if (!p.kept) { out.push(E('div', { class: 'box warn' }, E('p', {}, 'The static checks excluded every one of its tasks, so there is nothing here to evaluate or train on.'))); return out; }
// the kept tasks PostTrain's env add marks leaky (posttrain_path.py): its evals leave them out unless --include-leaky
const lk = p.leaky || [], one = lk.length === 1, them = one ? 'it' : 'them';
const lkNames = lk.slice(0, 4).join(', ') + (lk.length > 4 ? ` and ${lk.length - 4} more` : '');
if (lk.length && !p.scored) { out.push(E('div', { class: 'box warn' }, E('p', {}, `PostTrainβs env add marks every one of its ${plural(p.kept, 'task')} leaky (grading data the agent can read in its sandbox), and its evals leave leaky tasks out, so there is nothing here to evaluate or train on.`))); return out; }
const none = p.no_oracle === p.kept, eligible = p.excluded ? `${p.kept} eligible tasks` : plural(p.kept, 'task');
// tasks without a reference solution: added all the same (since 0.1.9), with a note; packaged as BenchFlow's own, they run as root
if (p.no_oracle) out.push(E('div', { class: 'box' }, E('p', {}, E('b', {}, none ? `None of its ${eligible} has a reference solution (oracle/solve.sh)` : `${p.no_oracle} of its ${eligible} have no reference solution (oracle/solve.sh)`),
none ? ': they are packaged the way BenchFlowβs own tasks are, with only what a run needs. ' : '. ', 'PostTrainβs ', E('code', {}, 'posttrain env add'),
' adds them with a note on each: without a reference solution, no oracle run can show that the verifier passes a correct answer, so a task no model solves may be broken rather than hard.'),
p.root ? E('p', {}, 'Its evals below run the agent as root (', E('code', {}, '--set sandbox_user=root'), '): tasks packaged this way are written for root, as the Arenaβs own runs assumed. BenchFlow then keeps nothing away from the agent; the packages hold no reference solution to find.') : ''));
// what the static checks found in the kept tasks' sandboxes, counted as PostTrain's env add counts them (posttrain_path.py):
// grading data an attempt can read (env add marks those tasks leaky), and answer-like files in the others, which hold nothing
// their verifier checks (a note, not a leak); the Tasks table names them
const rd = p.readable || {}, of = ` of these ${plural(p.kept, 'task')}`;
const notes = (n, lead) => `${lead}${n === 1 ? ' has answer-like files that hold nothing its' : ' have answer-like files that hold nothing their'} verifier checks: a note, not a leak.`;
if (rd.grading) out.push(E('p', {}, E('b', {}, 'Some tasks can be passed without solving them. '),
`The static checks found grading data the agent can read in its sandbox in ${rd.grading}${of} (the Tasks table below names them). PostTrainβs env add marks ${rd.grading === 1 ? 'that task' : 'those tasks'} leaky, so data from-rollouts leaves attempts at ${rd.grading === 1 ? 'it' : 'them'} out of the training rows, and an eval leaves ${rd.grading === 1 ? 'it' : 'them'} out unless you add --include-leaky.`,
rd.answers ? ' ' + notes(rd.answers, plural(rd.answers, 'other task')) : ''));
else if (rd.answers) out.push(E('p', {}, notes(rd.answers, `${rd.answers}${of}`), ' The Tasks table below names them.'));
const hf = p.setup.install.includes('uv tool install hf'), data = p.setup.install.includes('posttrain extras install data');
out.push(E('details', {}, E('summary', {}, 'Before the first command: install PostTrain and BenchFlow, start a console, make a project'),
E('p', {}, `PostTrain ${p.posttrain} or later${data ? ' with its data checksβ libraries (transformers and jinja2, without PyTorch: the hill climbβs from-rollouts renders the smaller modelβs chat template with them)' : ''}, BenchFlow with its Daytona sandboxes (it runs the agent)${hf ? ', and hf, the Hugging Face command line' : ''}:`), pre(p.setup.install),
E('p', {}, 'A console, in a second terminal (it keeps running):'), pre(p.setup.server),
E('p', {}, 'Back in the first terminal: a project, your Fireworks and Daytona keys, and a Fireworks target that serves the smaller model and its fine-tunes alike, on one H100 deployment that scales to zero after 5 idle minutes:'), pre(p.setup.project),
E('p', { class: 'small muted' }, A('The quickstart', p.quickstart), ' explains the console and projects.')));
const steps = E('ol', { class: 'steps' },
step(p.excluded ? `Get its tasks at the submitted commit, remove the ${p.excluded} the static checks excluded, and add the other ${p.kept}. ` : 'Get its tasks at the submitted commit and add them. ', pre(p.add)),
step('Evaluate a model on its tasks. ', lk.length ? `PostTrainβs evals leave out the ${one ? 'task' : `${lk.length} tasks`} its env add marks leaky (${lkNames}) and score the other ${p.scored}; --include-leaky runs ${them} too, and the evalβs results mark ${one ? 'its' : 'their'} attempts. First ${p.first} of the ${p.scored}, one attempt each (--limit ${p.first}, as the guide does), then each of them twice (k=2).`
: `First ${plural(p.first, 'task')}, one attempt each (--limit ${p.first}, as the guide does), then every task twice (k=2).`, ' An eval reports a score with its standard error, the result of each task and every attemptβs transcript.', pre(p.eval),
k.first ? E('p', { class: 'small muted' }, `Their dry runs estimate the first at ${est(k.first)} and the second at ${est(k.eval)}. With 4 attempts at once, each taking 2.5 to 8 minutes, or up to its taskβs time limit (10 minutes in this estimate), plus about 2 minutes to start the sandboxes and grade, the first takes about ${span(k.first.minutes)} and the second about ${span(k.eval.minutes)}.`) : ''),
step('Hill-climb. ', ...(sp ? [`Hold out ${sp.heldout} of the ${p.kept} tasks (every fourth) as the projectβs quick suite, so PostTrainβs overlap check keeps attempts at them out of the training rows (the other tasks are added after the suite is set, so their env add checks that none repeats a held-out task); collect the stronger modelβs verified attempts at the other ${sp.train}${sp.train_leaky ? ` (its eval leaves out the ${sp.train_leaky} marked leaky)` : ''} as training rows, train the smaller model on them for two epochs (the second run continues the first), and compare it with its base on the held-out ${sp.heldout === 1 ? 'task' : 'tasks'}${sp.heldout_leaky ? ` (their evals leave out the ${sp.heldout_leaky} marked leaky)` : ''}, both asked the same way. `,
'Four values come from earlier commands: SFT_RUN and TUNED_RUN are the run ids the two train sft commands print (run_β¦), and BASE_EVAL and TUNED_EVAL the eval ids the two held-out evals print (eval_β¦).', pre(p.hillclimb),
p.no_oracle ? E('p', {}, E('b', {}, 'The comparison is indicative. '), `${p.no_oracle === p.kept ? 'None of these tasks has' : `${p.no_oracle} of these tasks have no`} a reference solution, so a held-out task no model solves may be broken rather than hard, and a change on it says little.`,
p.training_only ? ' Its notes call it training-only, not for held-out evaluation: holding tasks out here doesnβt evaluate the collection, it only checks whether the smaller model learned something it didnβt train on.' : '') : '',
k.teacher ? E('p', { class: 'small muted' }, `Their dry runs estimate the teacherβs attempts at ${est(k.teacher)} and ${span(k.teacher.minutes)}, and each held-out eval of the smaller model at ${est(k.heldout)} and ${span(k.heldout.minutes)} on its H100 deployment, which bills while it runs. The two SFT runs cost $${k.sft_price.toFixed(2)} per million training tokens each, which their dry runs give once the rows exist (the guideβs 560 rows, 7.7M tokens, cost $3.81 an epoch). In all, the evals come to ${est(k.total)} and ${span(k.total_minutes)}, plus the SFT.`) : '',
E('p', { class: 'small muted' }, 'The guideβs section ', A('Train on them and compare', p.compare), ' explains the settings and how to read the comparison.')]
: [p.kept >= 4 && lk.length ? 'A hill climb compares the trained model with its base on tasks it didnβt train on; holding out every fourth of these tasks would leave only tasks marked leaky, which evals leave out, on one side.'
: `A hill climb compares the trained model with its base on tasks it didnβt train on; with ${plural(p.kept, 'task')} to use, this collection has too few to hold some out.`])));
out.push(E('p', {}, 'Evals and training bill your own Fireworks and Daytona accounts. Add --dry-run to a command to see what it would run and its cost estimate, without running it.'), steps);
return out;
}
async function submissionPage(v, id) {
const [s, subs] = await Promise.all([api('submissions/' + enc(id)), api('submissions')]), listed = subs.find(x => x.id === id) || { ...s, status: 'not run' }, n = s.run_counts || {};
setTitle(s.title);
const rev = String(s.revision || '').slice(0, 12), src = s.repo_id ? `${s.repo_id}@${rev}` : null;
const link = E('button', { class: 'plain', title: 'Copy this pageβs address: its link previews as this submission', on: { click: async (e) => { const b = e.currentTarget;
try { await navigator.clipboard.writeText(location.origin + colHref(id)); b.textContent = 'Copied'; } catch { b.textContent = location.origin + colHref(id); } setTimeout(() => { b.textContent = 'Copy link'; }, 2000); } } }, 'Copy link');
v.append(E('div', { class: 'crumb' }, A('Submissions', '#/'), ' / ', s.title), E('h1', {}, s.title),
E('div', { class: 'status-line' }, ...[s.author || s.team, `submitted ${when(s.created_at)}`, plural(s.task_count || 0, 'task'), src ? (s.source_url ? A(src, s.source_url, 'mono') : E('span', { class: 'mono' }, src)) : null, link].filter(Boolean).map(x => E('span', {}, x))));
if (s.description) v.append(E('p', { class: 'lede' }, s.description));
const jump = (text, at) => E('a', { href: `${colHref(id)}?at=${at}`, on: { click: (e) => { const t = document.getElementById(at); if (!t) return; e.preventDefault(); history.replaceState(null, '', e.currentTarget.getAttribute('href')); headHeight(); t.scrollIntoView(); } } }, text); // in place, without loading the page again
headHeight();
v.append(standing(s, listed), E('p', { class: 'small' }, 'On this page: ', ...[['Results', 'results'], ['Runs', 'runs'], ['Checks', 'checks'], ['Improve a model on it with PostTrain', 'posttrain'], ['Tasks', 'tasks']].map(([t, at], i) => [i ? ' Β· ' : '', jump(t, at)])));
// results: verified means per challenge, then every scored run with its review
const scored = (s.runs || []).filter(r => r.state === 'scored').reverse();
v.append(E('h2', { id: 'results' }, 'Results'));
if ((s.board || []).length) v.append(table([['Challenge'], ['Rank', 'r'], ['Verified runs', 'r'], ['Mean Ξ Β± SE, pp', 'r']], s.board.map(b => row(chHref(b.challenge_id, 'leaderboard'), [cell(A(b.challenge_id, chHref(b.challenge_id, 'leaderboard'))), cell(b.rank ?? 'β', 'r'), cell(b.verified_runs, 'r'), cell(dse(b.delta_pp, b.stderr_pp), 'r')]))));
if (scored.length) v.append(E('div', { style: 'height:8px' }), table([['Run'], ['Challenge'], ['Ξ Β± SE, pp', 'r'], ['Review']], scored.map(r => row(subHref(r.id), [cell(runLink(r)), cell(A(r.challenge_id, chHref(r.challenge_id))), cell(dse(r.delta_pp, r.stderr_pp), 'r'),
cell(E('span', {}, stateEl(r), r.verification_note ? E('span', { class: 'reason' }, r.verification_note) : ''))]))));
if (!scored.length) v.append(E('p', {}, n.runs ? `None yet: none of its ${plural(n.runs, 'run')} reached a score.` : 'None yet: it has not run.', ' ',
E('span', { class: 'muted' }, 'A result is the change (Ξ) in the held-out pass rate from before training to after, measured inside one run, in percentage points Β± one standard error; it counts once an organizer verifies it.')));
else v.append(E('p', { class: 'small muted' }, 'Ξ is the held-out pass rate after training minus before, measured inside the same run, in percentage points (pp); Β± is one standard error. A result counts once an organizer verifies it.'));
const c = mainCh();
v.append(E('h2', { id: 'runs' }, 'Runs'), runsTable(s.runs || []),
E('p', { class: 'small muted' }, 'Its author starts a run from a challengeβs ', c ? A('Runs page', chHref(c.id, 'runs')) : 'Runs page', ' or with arena_cli.py; the arena runs one at a time. Cost: the runβs GPU job on Hugging Face, as the arenaβs ledger settled it when the run ended.'));
v.append(E('h2', { id: 'checks' }, 'Checks'), checksDl(s));
if (s.posttrain) v.append(...postTrain(s.posttrain));
v.append(E('h2', { id: 'tasks' }, 'Tasks'), staticLine({ ...listed, task_count: s.task_count }));
if (unread({ ...listed, task_count: s.task_count })) return;
const holder = E('div'); v.append(holder); await taskList(holder, { collection: s.id });
v.append(E('p', { class: 'small muted' }, 'Verifier and reference solution: what the static checks found in the taskβs verifier/ and oracle/solve.sh. ', OUTCOME_NOTE));
}
// ββ Tasks: every task, filtered by collection, outcome, category or name βββββββββββββ
async function tasks(v) {
const subs = await api('submissions');
v.append(...page('Tasks'), E('p', { class: 'lede' }, 'Every task of every collection, with what the static checks found. Pick a collection to browse only its tasks.'));
const h = E('div'); v.append(h); await taskList(h, { pick: subs, collection: qs().get('collection') || '' });
v.append(E('p', { class: 'small muted' }, 'Static checks: ', OUTCOME_NOTE));
}
async function taskList(host, opts) { // opts.collection fixes (or, with opts.pick, preselects) one collection
const P = qs(), st = { collection: opts.collection || '', outcome: P.get('outcome') || '', cat: P.get('cat') || '', q: P.get('q') || '', n: Number(P.get('n')) || 50 };
const input = E('input', { type: 'search', placeholder: opts.pick ? 'Find a task, collection or team' : 'Find a task', value: st.q, style: 'min-width:250px' }), out = E('select', { 'aria-label': 'Outcome' }), cats = E('select', { 'aria-label': 'Category' }), body = E('div'), more = E('p');
const cols = opts.pick ? E('select', { 'aria-label': 'Collection' }, E('option', { value: '' }, 'Every collection'), ...[...opts.pick].sort((a, b) => String(a.title).localeCompare(String(b.title))).map(s => E('option', { value: s.id, selected: st.collection === s.id }, `${s.title} β ${s.team || s.author}, ${plural(s.task_count || 0, 'task')}`))) : null;
let t; input.oninput = () => { clearTimeout(t); t = setTimeout(() => { st.q = input.value; st.n = 50; setQs({ q: st.q, n: '' }); load(); }, 200); };
out.onchange = () => { st.outcome = out.value; st.n = 50; setQs({ outcome: st.outcome, n: '' }); load(); };
cats.onchange = () => { st.cat = cats.value; st.n = 50; setQs({ cat: st.cat, n: '' }); load(); };
if (cols) cols.onchange = () => { st.collection = cols.value; st.n = 50; setQs({ collection: st.collection, n: '' }); load(); };
const note = E('div'); host.append(E('div', { class: 'filters' }, cols, input, out, cats), note, body, more);
async function load() {
const d = await api(`tasks?limit=${st.n}${st.collection ? '&collection=' + enc(st.collection) : ''}${st.q ? '&q=' + enc(st.q) : ''}${st.outcome ? '&outcome=' + enc(st.outcome) : ''}${st.cat ? '&cats=' + enc(st.cat) : ''}`);
const all = Object.values(d.by_outcome).reduce((a, x) => a + x, 0), allc = Object.values(d.by_category).reduce((a, x) => a + x, 0), one = !!st.collection;
const missing = (opts.pick || []).filter(unread), chosen = one && missing.find(x => x.id === st.collection);
note.replaceChildren(!one ? unreadNote(missing) : '');
out.replaceChildren(E('option', { value: '' }, `Every outcome (${all})`), ...Object.keys(OUTCOME).filter(k => d.by_outcome[k] || st.outcome === k).map(k => E('option', { value: k, selected: st.outcome === k }, `${OUTCOME[k]} (${d.by_outcome[k] || 0})`)));
cats.replaceChildren(E('option', { value: '' }, `Every category (${allc})`), ...Object.keys(d.by_category).sort().map(k => E('option', { value: k, selected: st.cat === k }, `${k.replace(/-/g, ' ')} (${d.by_category[k]})`)));
// Verifier and Reference solution: what the static checks found in verifier/ and oracle/solve.sh (app_api.task_findings);
// under the outcome, the reason for an exclusion, or any other finding. A row without codes (an older database) shows why only.
body.replaceChildren(table([['Task'], ...(one ? [] : [['Collection']]), ['Category', 'hide-s'], ['Verifier'], ['Reference solution'], ['Outcome']], d.rows.map(x => row(null, [cell(E('span', { class: 'mono' }, x.name)),
...(one ? [] : [cell(E('span', {}, A(x.collection_title, colHref(x.collection_id)), E('span', { class: 'reason' }, x.team || '')))]), cell((x.category || '').replace(/-/g, ' '), 'hide-s'),
cell(x.codes == null ? 'β' : x.verifier ? E('span', { class: 'small' }, x.verifier) : E('span', { class: 'small muted' }, 'no finding')),
cell(x.solution == null ? 'β' : E('span', { class: 'small' + (x.solution === 'present' ? '' : ' s-review') }, x.solution === 'present' ? 'yes' : x.solution)),
cell(E('span', {}, E('span', { class: 'state nw s-' + outcomeClass(x.outcome) }, OUTCOME[x.outcome] || x.outcome), E('span', { class: 'reason' }, x.outcome === 'excluded' || x.codes == null ? x.why || '' : x.other || '')))])),
chosen ? 'No per-task results for this collection in this data source: the static checksβ report for its pinned commit could not be read.' : st.q || st.outcome || st.cat ? 'No task matches.' : 'No per-task results from the static checks in this data source yet.', null, 'wide'));
more.replaceChildren(E('span', { class: 'small muted' }, `Showing ${Math.min(st.n, d.total)} of ${d.total}. `), ...(d.total > st.n ? [E('button', { class: 'plain', on: { click: () => { st.n += 100; setQs({ n: st.n }); load(); } } }, 'Show more')] : []));
}
return load();
}
// the organizers' pause, where a newcomer starts (starter kit, submit)
function pausedBox(c) { const B = c && boardOf(c.id); return B && B.runs_paused ? E('div', { class: 'box warn' }, E('p', {}, E('b', {}, `Runs on ${c.id} are paused by the organizers. `), cap(first(B.runs_paused)), ' Submitting and checking collections still work; ', A('the overview', chHref(c.id)), ' says more.')) : ''; }
// ββ Starter kit: numbered steps, mostly one command each βββββββββββββββββββββββββββββ
async function starter(v) {
const c = ch(), R = rulesOf(c), repo = 'https://github.com/benchflow-ai/posttrainarena', h = (R.recipe || {}).harness || {};
v.append(...page('Starter kit'), E('p', { class: 'lede' }, 'From nothing to a scored run. Most steps are one command; the browser works too (', A('Submit a collection', '#/submit'), `). The commands use ${c.id}, the open challenge${CHS().filter(x => x.status === 'open').length > 1 ? ' you looked at last' : ''}.`), pausedBox(c));
const step = (title, ...kids) => E('li', {}, E('b', {}, title), ...kids);
v.append(E('ol', { class: 'steps' },
step('Copy the starter kit. ', 'The public ', A('starting kit', repo + '/tree/main/starting-kit'), ' has a task template and eight worked examples in the format the arena reads; the same repository has the local check scripts.', E('pre', {}, `git clone ${repo}
mkdir -p my-collection/envs
cp -R posttrainarena/starting-kit/template my-collection/envs/my-task`)),
step('Write your tasks. ', 'One directory per task under envs/, and a submission.yaml with your own team name and contact email:', E('pre', {}, `my-collection/
submission.yaml team_name: β¦ contact_email: β¦ track: environments
envs/
my-task/
task.md frontmatter (author, license, category, origin, timeouts) and "## prompt"
environment/Dockerfile the agent's sandbox; never copy the solution or the tests in
verifier/test.sh runs the checks and writes the reward (1 or 0)
verifier/test_outputs.py the checks
verifier/verifier.md and at least one verifier/rubrics/*.md
oracle/solve.sh the reference solution; without one the task needs controls`),
E('p', { class: 'small muted' }, 'The ', A('spec', 'https://posttrain.com/docs/spec'), ' describes every file and field; the ', A('agent guide', '/AGENTS.md'), 'βs Task credit metadata section lists the 18 category values and the license and origin fields that credit you.'),
E('p', {}, E('b', {}, 'Write tasks the untrained model solves some of the time. '), `Training compares ${(R.recipe || {}).num_generations || 8} attempts at the same task and moves the model toward the better ones. If every attempt fails, or every attempt passes, there is nothing to learn and the run stops. For comparison, `, A('Base Labsβ RL study', 'https://labs.baseten.co/articles/when-does-distillation-help-reinforcement-learning'), ' kept a task family only when a single attempt succeeded 5% to 45% of the time and fewer than 5% of replies hit the length limit.'),
E('p', {}, E('b', {}, 'Keep each task short. '), `An attempt has ${h.agent_timeout_sec || 900} s, and under the current pipeline a long attempt that fills the modelβs context is cut off mid-reply (see the `, A('known issues', chHref(c.id, 'rules') + '?at=known-issues'), '). Tasks an agent finishes in a few dozen tool calls give the cleanest signal.')),
step('Check it locally. ', 'The structure check and the static gates need no token or Docker (the arenaβs copy of the gates also checks overlap with the held-out benchmark, at validation); both warn about a verifier that downloads tools or fetches data when it runs while the task turns the network off. The two replays need Docker: with its reference solution the task must score 1, and doing nothing (--skip-oracle) must score 0.', E('pre', {}, `python3 posttrainarena/scripts/check_task.py my-collection/envs
curl -fsSO ${location.origin}/validation_gates.py
python3 validation_gates.py static my-collection/envs
posttrainarena/scripts/run_local.sh my-collection/envs/my-task
posttrainarena/scripts/run_local.sh my-collection/envs/my-task --skip-oracle`)),
step('Upload the collection to Hugging Face. ', 'As a public dataset you own, which is what we recommend. A public GitHub repository works too, but the arena checks GitHub collections within one hourly GitHub limit that every participant shares, so a check can have to wait for the next hour. Create the dataset first, then a token that can write to that one dataset: the boardβs ', A('Add your agent', '/#add-your-agent'), ' says how (its step 1). Below, YOUR_NAME/arena-tasks stands for that dataset. The first line installs hf, Hugging Faceβs command line; hf auth login asks for the token once and saves it.', E('pre', {}, `uv tool install hf
hf auth login
hf upload YOUR_NAME/arena-tasks my-collection --repo-type dataset`)),
step('Check and submit it. ', 'The static checks report every task before anything is stored. arena_cli.py sends the token hf auth login saved, which only identifies you; never type a token into a command, where it stays in your shellβs history.', E('pre', {}, `curl -fsSO ${location.origin}/arena_cli.py
python3 arena_cli.py validate --file environment.json
python3 arena_cli.py submit --file environment.json`), E('p', { class: 'small muted' }, 'environment.json names the dataset (YOUR_NAME/arena-tasks), commit, folder and title (the API and the CLI call a collection an environment); the ', A('agent guide', '/AGENTS.md'), ' has the fields. submit prints the collectionβs id, which starts with env-.')),
step('Start a run. ', 'The first command is the preflight: every check the arena makes before it spends compute, with nothing reserved. The second starts the run.', E('pre', {}, `python3 arena_cli.py run --challenge ${c.id} --id ENVIRONMENT_ID
python3 arena_cli.py run --challenge ${c.id} --id ENVIRONMENT_ID --file run.json --execute`),
E('p', { class: 'small muted' }, 'ENVIRONMENT_ID is the id submit printed. run.json holds a request id you choose and keep, for example {"request_id": "my-run-001"}: retrying with the same file never starts a second run.')),
step('Collect the result. ', 'When the run is scored, collect it: the arena recomputes the score from the per-task results, an organizer reviews the evidence, and a verified run joins the leaderboard.', E('pre', {}, `python3 arena_cli.py result collect --challenge ${c.id} --run-id RUN_ID`))));
v.append(E('h2', {}, 'Examples'), E('p', {}, 'The starting kitβs ', A('eight examples', repo + '/tree/main/starting-kit/examples'), ' show every part of a task. Here, ', A('Submissions', '#/'), ' lists every submitted collection with its checks, runs and results, and ', A('Tasks', '#/tasks'), ' every submitted task with what the static checks found.'));
}
// ββ Submit a collection: AC2's form pattern (title, one purpose line, help under each field, the consequence before the button) ββ
async function submit(v) {
const c = ch(), I = await identity(c);
v.append(...page('Submit a collection'), E('p', { class: 'lede' }, 'The arena reads your repository at one commit, runs the static checks on every task, and stores the collection pinned to that commit. Submitting does not start a run.'));
if (!I.live) v.append(E('p', { class: 'small' }, 'On simulated data the form shows the steps, but checking and submitting are real actions, so they work only on live data.'));
else if (!I.signedIn) v.append(E('p', {}, ME && ME.oauth_enabled ? [signIn('Sign in with Hugging Face'), ' to check and submit. '] : ME && ME.unchecked ? UNCHECKED + ' ' : 'Sign-in is not available on this server. ', 'Agents can use the CLI (', A('starter kit', '#/starter'), ').'));
if (!c || c.status !== 'open') { v.append(E('div', { class: 'box warn' }, E('p', {}, 'No challenge is open, so nothing can be submitted yet. ', A('Challenges', '#/challenges'), ' says when each opens.'))); return; }
v.append(pausedBox(c));
const f = { repo_type: E('select', {}, E('option', { value: 'dataset' }, 'Hugging Face dataset'), E('option', { value: 'github' }, 'GitHub repository')), repo_id: E('input', { placeholder: 'owner/name' }), revision: E('input', { value: 'main' }), entry_path: E('input', { placeholder: 'empty for the repository root' }), title: E('input', {}), notes: E('textarea', {}) };
const help = { 'Where': 'Where the repository lives. A Hugging Face dataset is recommended: every participantβs GitHub checks share one hourly GitHub limit.', 'Repository': 'owner/name on the Hub or GitHub.', 'Commit or branch': 'A branch is pinned to its current commit when you submit; later pushes change nothing.', 'Task folder': 'The folder that holds submission.yaml and envs/; leave it empty when they are at the repository root.', 'Title': 'Shown on the leaderboard and on your runs.', 'Notes': 'What the tasks are and why they should help the model; organizers read this in review.' };
const fill = () => { f.repo_type.value = 'dataset'; f.repo_id.value = 'benchflow/posttrain-generic-dogfood-20260922'; f.revision.value = 'main'; f.entry_path.value = ''; f.title.value = 'Example: BenchFlowβs three-task test collection'; };
if (qs().get('example') === '1') fill();
const out = E('div'), checkBtn = E('button', { class: 'plain', disabled: !I.enabled, title: I.why }, 'Check'), submitBtn = E('button', { disabled: true, title: I.why || 'Check the collection first.' }, 'Submit');
const body = () => ({ challenge_id: c.id, repo_type: f.repo_type.value, repo_id: f.repo_id.value.trim(), revision: f.revision.value.trim() || 'main', entry_path: f.entry_path.value.trim(), title: f.title.value.trim(), notes: f.notes.value.trim() });
checkBtn.onclick = async () => { out.replaceChildren(E('p', {}, 'Checking⦠this reads the repository and can take a minute.')); submitBtn.disabled = true;
try { const r = await act('POST', '/api/v2/environments/validate', body()); out.replaceChildren(checkResult(r)); submitBtn.disabled = !I.enabled; } catch (e) { out.replaceChildren(E('div', { class: 'box bad' }, e.message)); } };
submitBtn.onclick = async () => { try { const r = await act('POST', '/api/v2/environments', body()); out.append(E('div', { class: 'box' }, E('p', {}, E('b', {}, r.existing ? 'Already submitted: ' : 'Submitted: '), A(r.title || r.id, colHref(r.id)), ` (${r.id}, ${plural(r.task_count, 'task')}, commit ${String(r.revision).slice(0, 12)}). Start a run from `, A(`${c.id}βs Runs page`, chHref(c.id, 'runs')), '; the collection appears there within two minutes.'), r.note ? E('p', { class: 'muted' }, r.note) : '')); CACHE.clear(); }
catch (e) { out.append(E('div', { class: 'box bad' }, e.message)); } };
v.append(E('div', { class: 'form' }, ...Object.entries({ 'Where': f.repo_type, 'Repository': f.repo_id, 'Commit or branch': f.revision, 'Task folder': f.entry_path, 'Title': f.title, 'Notes': f.notes }).flatMap(([k, x]) => [E('label', {}, k), E('div', { class: 'field' }, x, E('div', { class: 'help' }, help[k]))]),
E('label', {}, 'Challenge'), E('div', { class: 'field fixed' }, E('span', { class: 'mono' }, c.id), E('div', { class: 'help' }, 'Recorded with the collection, which can run on any open challenge.'))),
E('p', { class: 'small muted', style: 'max-width:642px' }, 'Check runs the static checks and shows every taskβs result; nothing is stored. Submit stores the collection at the checked commit. The format is in the ', A('starter kit', '#/starter'), '.'),
E('div', { class: 'actions' }, E('button', { class: 'plain', title: 'BenchFlowβs three-task test collection, to try Check', on: { click: fill } }, 'Fill in an example'), checkBtn, submitBtn), out);
}
// ββ router ββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
// #/ Submissions: every submitted collection (#/submissions, #/collections and /arena/submissions too)
// /arena/submissions/<id> a submission's page at its own address, which previews as the submission (PATH_ROUTE)
// #/submissions/<id> the same page (#/collections/<id> too; a run's id opens the run: links made before Sept 25)
// #/challenges every challenge (/arena/challenges too)
// /arena/challenges/<id>[/<tab>] a challenge: overview, leaderboard, runs or rules, previewing as the challenge; #/challenges/<id>[/<tab>] too
// #/runs/<run id> one run, in its challenge's frame
// #/leaderboard, #/runs, #/rules the current challenge's tab (old links)
// #/tasks #/starter #/submit
const TAB_PAGES = { '': overview, leaderboard, runs: submissions, submissions, rules: rulesPage }; // submissions: the tab's old name
async function route() {
if (location.hash && PATH_ROUTE.test(location.pathname)) history.replaceState(null, '', '/arena' + location.hash); // a #/β¦ link followed from a submission's address
const v = $('#main'), [p = '', a, b] = routePath().split('/').map(x => { try { return decodeURIComponent(x); } catch { return x; } });
v.replaceChildren(E('p', { class: 'muted' }, 'Loadingβ¦'));
try {
if (!META) { META = await api('meta'); await me(); }
try { BOARD = await api('board/challenges'); } catch { BOARD = null; } // the pages still render without it, with less on each challenge
if (!CH || !chById(CH)) CH = (mainCh() || {}).id;
const missing = (text) => Object.assign(new Error(text), { status: 404 }); // the not-found state, as for an unknown submission
const inCh = (id) => { const c = chById(id); if (!c) throw missing(`there is no challenge called β${id}β in this data source.`); if (c.status === 'open') setCurrent(c.id); return c; };
let fn, on = 'challenges';
if (p === '' || ((p === 'submissions' || p === 'collections') && !a)) { on = 'submissions'; fn = submissionsPage; }
else if (p === 'submissions' || p === 'collections') { // a collection's id; a run's id there is an older link to the run
const sub = p === 'collections' || a.startsWith('env-') || (await api('submissions')).some(s => s.id === a);
on = sub ? 'submissions' : 'challenges'; fn = sub ? (x) => submissionPage(x, a) : (x) => runPage(x, a);
}
else if (p === 'challenges' && !a) fn = challengesPage;
else if (p === 'challenges') { const c = inCh(a), f = TAB_PAGES[b || '']; if (!f) throw missing(`a challenge has no page called β${b}β.`); fn = (x) => f(x, c); }
else if (p === 'runs' && a) fn = (x) => runPage(x, a);
else if (['leaderboard', 'rules', 'runs'].includes(p)) { const c = ch(), f = TAB_PAGES[p]; fn = (x) => f(x, c); }
else { on = { play: 'submit' }[p] || p; fn = { tasks, starter, submit, play: submit }[p]; }
if (!fn) throw missing(`there is no page called β${p}β.`);
shell(on);
const box = E('div'); await fn(box); v.replaceChildren(box);
const at = qs().get('at'), target = at && document.getElementById(at); if (target) target.scrollIntoView(); else window.scrollTo(0, 0);
} catch (e) {
if (META) shell('');
const building = e.status === 503 && !e.transient; // right after a restart the live data takes a minute to build
if (e.transient) v.replaceChildren(E('p', {}, `${e.message} Retrying in 10 seconds.`));
else if (e.status === 404) { setTitle('Not found'); v.replaceChildren(E('p', {}, `${cap(e.message)} `, A('Back to Submissions', '#/'))); } // what the address names doesn't exist (the Space answered 404 for it too)
else v.replaceChildren(E('p', {}, `Could not load this page: ${e.message} `, building ? 'Retrying in 15 seconds.' : A('Back to Submissions', '#/'))); // #/ opens Submissions
if (building || e.transient) { const at = location.href; setTimeout(() => { if (location.href === at) route(); }, building ? 15000 : 10000); }
}
}
// A #fragment change routes on hashchange; going back or forward to a submission's address (no fragment) on popstate.
// Each skips what the other handles, since a traversal between the two kinds of entry fires both.
window.addEventListener('hashchange', () => { if (location.hash) route(); });
window.addEventListener('popstate', () => { if (!location.hash) route(); });
route();
</script>
</body>
</html>
|