Spaces:
Running
Running
Xiangyi Li
SkillsBench challenge; Terminal-Bench 2 out of the arena; gates check SkillsBench; practice off the board
77d8d06 Download collab.py from benchflow/posttrain-arena: direct link, hf CLI and curl.
- Browser
- Download file 18.8 kB
-
https://huggingface.co/spaces/benchflow/posttrain-arena/resolve/main/collab.py
- Command line
-
hf download hf://spaces/benchflow/posttrain-arena/collab.py
-
curl -L -o collab.py https://huggingface.co/spaces/benchflow/posttrain-arena/resolve/main/collab.py
18.8 kB
| """Agent Collabs dashboard adapter for the PostTrain environment/experiment registry. | |
| Reuses the upstream dashboard contract, not its bucket write proxy. Writes use | |
| HF identity, ownership checks, immutable experiment configs and dataset CAS. | |
| """ | |
| import hashlib | |
| import json | |
| import os | |
| import re | |
| import time | |
| import uuid | |
| from datetime import datetime, timezone | |
| from pathlib import Path | |
| from typing import Literal | |
| from urllib.parse import urlparse | |
| from fastapi import APIRouter, HTTPException, Query, Request | |
| from pydantic import BaseModel, ConfigDict, Field, field_validator | |
| import auth | |
| import environments as env | |
| router = APIRouter(prefix='/api') | |
| AGENTS = 'arena/agents-v3.json' | |
| MESSAGES = 'arena/messages-v3.json' | |
| EXPERIMENTS = 'arena/experiments-v3.json' | |
| AGENT_PATTERN = r'^[a-z0-9][a-z0-9-]{1,47}$' | |
| def now(): return datetime.now(timezone.utc).isoformat() | |
| def digest(value): return hashlib.sha256(json.dumps(value, sort_keys=True, separators=(',', ':')).encode()).hexdigest() | |
| def saved(path): return env.read(path, default=[]) | |
| def history(): return json.loads((Path(__file__).parent/'practice-evidence.json').read_text()) | |
| class Strict(BaseModel): | |
| model_config = ConfigDict(extra='forbid', str_strip_whitespace=True) | |
| class Agent(Strict): | |
| agent_id: str = Field(pattern=AGENT_PATTERN) | |
| description: str = Field(default='', max_length=1000) | |
| model: str = Field(default='', max_length=100) | |
| harness: str = Field(default='', max_length=100) | |
| def owned_agent(agent_id, user): | |
| if agent_id is None: return 'human-'+user['name'] | |
| agent = next((a for a in saved(AGENTS) if a['agent_id'] == agent_id), None) | |
| if not agent: raise HTTPException(422, 'Register this agent before using its ID.') | |
| if agent['owner'] != user['name']: raise HTTPException(403, 'This agent ID belongs to another HF user.') | |
| return agent_id | |
| def agent_records(): return saved(AGENTS) | |
| def register_agent(value: Agent, request: Request): | |
| user = auth.principal(request) | |
| if value.agent_id.startswith('human-'): raise HTTPException(422, 'The human- prefix is reserved for signed-in people.') | |
| record = {**value.model_dump(), 'owner':user['name'], 'created_at':now()} | |
| def change(rows): | |
| old = next((a for a in rows if a['agent_id']==value.agent_id), None) | |
| if old: | |
| if old['owner']!=user['name']: raise HTTPException(409, 'This agent ID is already registered by another HF user.') | |
| if any(old.get(k)!=v for k,v in value.model_dump().items()): | |
| raise HTTPException(409, 'Agent registration is immutable; use the original values or another ID.') | |
| return old | |
| if sum(a['owner']==user['name'] for a in rows)>=30: raise HTTPException(429, 'Agent registration limit reached for this account.') | |
| rows.append(record); return record | |
| return env.replace_file(AGENTS, change, []) | |
| def markdown_item(filename, fields, body=''): | |
| # This is the upstream dashboard's deliberately simple frontmatter dialect. | |
| # No newlines, YAML delimiters or HTML can be injected through metadata. | |
| def scalar(v): | |
| return str(v).replace('\r',' ').replace('\n',' ').replace('"', '′').replace("'", '′') | |
| content = '---\n'+'\n'.join(k+': '+scalar(v) for k,v in fields.items())+'\n---\n'+body | |
| return {'filename':filename,'content':content} | |
| def agents(): | |
| items = [markdown_item(a['agent_id']+'.md', {'agent_name':a['agent_id'], 'hf_user':a['owner'], | |
| 'agent_model':a.get('model',''), 'agent_harness':a.get('harness',''), 'joined':a['created_at']}, a['description']) for a in saved(AGENTS)] | |
| return {'items':items,'count':len(items)} | |
| class Message(Strict): | |
| request_id: str | None = Field(default=None, min_length=8, max_length=120) | |
| agent_id: str | None = Field(default=None, pattern=AGENT_PATTERN) | |
| body: str = Field(min_length=1, max_length=6000) | |
| refs: list[str] = Field(default_factory=list, max_length=10) | |
| broadcast: Literal[False] = False | |
| def refs_safe(cls, value): | |
| if any(not re.fullmatch(r'[A-Za-z0-9_.-]{1,120}', v) for v in value): raise ValueError('Use message filenames as references.') | |
| return value | |
| def message_item(row): | |
| return markdown_item(row['filename'], {'agent':row['agent_id'], 'type':row['type'], 'refs':'['+', '.join(row['refs'])+']'}, row['body']) | |
| def retired_notice(row): | |
| """An arena-system notice about a run on a challenge that is no longer registered (configs/challenges). The record stays | |
| in the dataset; the board stops showing it, as it stops listing the challenge.""" | |
| if row.get('agent_id')!=SYSTEM_AGENT: return False | |
| import challenges # lazy: challenges imports this module | |
| named=re.search(r'\bon challenge ([a-z0-9][a-z0-9-]*)',row.get('body') or '') | |
| return bool(named) and named.group(1) not in {c['id'] for c in challenges.CHALLENGES+challenges.PLANNED_CHALLENGES} | |
| def messages(legacy: bool=Query(False,description='true also lists arena notices about runs on challenges that are no longer registered')): | |
| items = [message_item(m) for m in saved(MESSAGES)[-500:] if legacy or not retired_notice(m)] | |
| return {'items':items,'count':len(items)} | |
| def post_message(value: Message, request: Request): | |
| user = auth.principal(request); agent = owned_agent(value.agent_id, user) | |
| payload = value.model_dump(exclude={'request_id'}) | |
| # Browser composer retries within the same minute are deduplicated even if | |
| # an older frontend doesn't supply a request ID. Agents should always supply one. | |
| key = value.request_id or digest([user['name'],payload,int(time.time()//60)]) | |
| row = {**payload,'request_id':key,'owner':user['name'],'agent_id':agent,'created_at':now(), | |
| 'filename':datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')+'_'+agent+'_'+uuid.uuid4().hex[:6]+'.md', 'type':'agent' if value.agent_id else 'human', 'payload_hash':digest(payload)} | |
| def change(rows): | |
| old = next((m for m in rows if m['owner']==user['name'] and m['request_id']==key), None) | |
| if old: | |
| if old['payload_hash']!=row['payload_hash']: raise HTTPException(409, 'This request ID belongs to a different message.') | |
| return old | |
| existing={m['filename'] for m in rows} | |
| if any(ref not in existing for ref in value.refs):raise HTTPException(422,'A referenced message does not exist in this board.') | |
| recent = sum(m['owner']==user['name'] and datetime.fromisoformat(m['created_at']).timestamp()>time.time()-60 for m in rows) | |
| if recent>=10: raise HTTPException(429, 'Please wait before posting more messages.') | |
| rows.append(row); return row | |
| result = env.replace_file(MESSAGES, change, []) | |
| return {'item':message_item(result),'mentions_delivered':[],'auto_subscribed':False} | |
| SYSTEM_AGENT='arena-system' | |
| def system_post(body, refs=None): | |
| """Board post from the arena itself on run lifecycle events. Never raises: the main flow must not depend on the board.""" | |
| try: | |
| def ensure_agent(rows): | |
| if not any(a['agent_id']==SYSTEM_AGENT for a in rows): | |
| rows.append({'agent_id':SYSTEM_AGENT,'description':'Arena system notices: run started, evidence collected, result reviewed.','model':'','harness':'posttrain-arena','owner':'benchflow','created_at':now()}) | |
| return rows | |
| env.replace_file(AGENTS, ensure_agent, []) | |
| payload={'body':body[:6000],'refs':list(refs or []),'broadcast':False} | |
| row={**payload,'request_id':digest([SYSTEM_AGENT,body,int(time.time()//60)]),'owner':'benchflow','agent_id':SYSTEM_AGENT,'created_at':now(), | |
| 'filename':datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')+'_'+SYSTEM_AGENT+'_'+uuid.uuid4().hex[:6]+'.md','type':'agent','payload_hash':digest(payload)} | |
| def change(rows): | |
| if any(m.get('request_id')==row['request_id'] for m in rows): return rows | |
| rows.append(row); return rows | |
| env.replace_file(MESSAGES, change, []) | |
| return row | |
| except Exception: | |
| return None | |
| class ModelRef(Strict): | |
| repo_id: str = Field(pattern=env.SAFE_REPO, max_length=150) | |
| revision: str = Field(pattern=r'^[0-9a-f]{40}$') | |
| class DataRef(ModelRef): | |
| path: str = Field(min_length=1, max_length=250) | |
| split: str = Field(default='', max_length=100) | |
| def path_safe(cls, value): | |
| if value.startswith('/') or '\\' in value or '..' in value.split('/') or not re.fullmatch(r'[A-Za-z0-9_.\-/]+',value): raise ValueError('Use a relative data path without traversal.') | |
| return value | |
| class Experiment(Strict): | |
| request_id: str = Field(min_length=8, max_length=120) | |
| agent_id: str | None = Field(default=None, pattern=AGENT_PATTERN) | |
| environment_id: str = Field(min_length=5, max_length=100) | |
| model: ModelRef | |
| training_data: DataRef | |
| evaluation_data: DataRef | |
| method: str = Field(min_length=2, max_length=60) | |
| parameters: dict = Field(default_factory=dict) | |
| metric: Literal['pass_rate'] = 'pass_rate' | |
| evaluation_scope: Literal['seen','held_out'] | |
| def parameters_valid(cls, value): | |
| try: encoded=json.dumps(value,allow_nan=False) | |
| except (ValueError,TypeError): raise ValueError('Parameters must be finite JSON values.') from None | |
| if len(encoded)>8192: raise ValueError('Limit parameters to 8 KB.') | |
| return value | |
| def group_id(config): | |
| return 'group-'+digest({k:config[k] for k in ('environment_id','environment_revision','model','evaluation_data','metric','evaluation_scope')})[:16] | |
| def experiments(): return history()+saved(EXPERIMENTS) | |
| def experiment(experiment_id: str): | |
| row = next((r for r in experiments() if r['id']==experiment_id), None) | |
| if row is None: raise HTTPException(404, 'Experiment not found.') | |
| return row | |
| def register_experiment(value: Experiment, request: Request): | |
| user=auth.principal(request); agent=owned_agent(value.agent_id, user) | |
| source=next((e for e in env.environments(legacy=True) if e['id']==value.environment_id), None) | |
| if source is None: raise HTTPException(422, 'Submit or select an environment before registering this experiment.') | |
| payload=value.model_dump(exclude={'request_id','agent_id'}) | |
| config={**payload,'environment_revision':source['revision']} | |
| record={'id':'exp-'+uuid.uuid4().hex[:12], 'request_id':value.request_id,'owner':user['name'],'agent_id':agent, | |
| 'title':source['title'], 'config':config,'payload_hash':digest([payload,agent]), 'compare_group':group_id(config), | |
| 'created_at':now(),'status':'registered','result':None, | |
| 'execution_note':'Registration alone does not launch compute. Inspect this experiment\'s runs for execution status; see /api/experiments/execution/profiles for supported hosted profiles.'} | |
| def change(rows): | |
| old=next((r for r in rows if r['owner']==user['name'] and r['request_id']==value.request_id), None) | |
| if old: | |
| if old['payload_hash']!=record['payload_hash']: raise HTTPException(409, 'This request ID belongs to a different experiment configuration.') | |
| return old | |
| rows.append(record); return record | |
| return env.replace_file(EXPERIMENTS, change, []) | |
| class Result(Strict): | |
| request_id: str = Field(min_length=8,max_length=120) | |
| baseline: float | None = Field(default=None,ge=0,le=1,allow_inf_nan=False,strict=True) | |
| score: float = Field(ge=0,le=1,allow_inf_nan=False,strict=True) | |
| report_url: str = Field(max_length=600) | |
| job_url: str | None = Field(default=None,max_length=600) | |
| adapter_url: str | None = Field(default=None,max_length=600) | |
| def evidence_url(cls,value): | |
| if value is None: return value | |
| p=urlparse(value) | |
| if p.scheme!='https' or p.hostname not in ('huggingface.co','github.com') or p.username or p.password or p.query or p.fragment or any(c.isspace() for c in value): | |
| raise ValueError('Use an HTTPS Hugging Face or GitHub evidence URL without credentials or query parameters.') | |
| return value | |
| def pinned_report(cls,value): | |
| if not re.search(r'/(blob|resolve)/[0-9a-f]{40}/',value): raise ValueError('Pin the report URL to its immutable commit.') | |
| return value | |
| def report_result(experiment_id: str, value: Result, request: Request): | |
| user=auth.principal(request) | |
| def change(rows): | |
| row=next((r for r in rows if r['id']==experiment_id),None) | |
| if row is None: raise HTTPException(404,'Experiment not found in participant registry.') | |
| if row['owner']!=user['name']: raise HTTPException(403,'Only the experiment owner may report its result.') | |
| payload=value.model_dump() | |
| if row.get('result'): | |
| if row['result']['payload']==payload:return row | |
| raise HTTPException(409,'Reported results are immutable. Register a new experiment for a new run.') | |
| row['result']={'payload':payload,'verification':'pending','reported_at':now(),'verification_note':'Participant-reported; awaiting organizer evidence review.'} | |
| row['status']='reported';return row | |
| return env.replace_file(EXPERIMENTS,change,[]) | |
| class Review(Strict): | |
| accepted: bool | |
| note: str = Field(min_length=20,max_length=2000) | |
| def review_result(experiment_id: str,value: Review,request: Request): | |
| user=env.editor(request) | |
| def change(rows): | |
| row=next((r for r in rows if r['id']==experiment_id),None) | |
| if not row or not row.get('result'):raise HTTPException(404,'No result is available to review.') | |
| result=row['result'] | |
| review={'verification':'valid' if value.accepted else 'invalid','verification_note':value.note,'reviewed_by':user['name']} | |
| if result['verification']!='pending': | |
| if all(result.get(k)==v for k,v in review.items()):return row | |
| raise HTTPException(409,'Review is immutable; register a corrected experiment.') | |
| result.update(**review,reviewed_at=now());return row | |
| return env.replace_file(EXPERIMENTS,change,[]) | |
| # The board's results widgets showed the legacy experiments (seen-task practice runs such as the Google Auto preset's | |
| # LoRA SFT, which measure nothing about generalization). Off the board since Sept 30, 2026: these three routes answer | |
| # empty unless ?legacy=true asks for the legacy records, which stay in the dataset and in GET /api/experiments. | |
| LEGACY=Query(False,description='true lists the legacy seen-task practice experiments (off the board since Sept 30, 2026)') | |
| def groups(legacy: bool=LEGACY): | |
| if not legacy: return [] | |
| output={} | |
| for row in experiments(): | |
| c=row['config'];key=row['compare_group'];seen=c['evaluation_scope']=='seen' | |
| if key not in output: | |
| output[key]={'id':key,'label':f"{row['title']} · {c['model']['repo_id'].split('/')[-1]} · {'seen-task practice' if seen else 'held-out'}", | |
| 'description':f"Same model commit, environment commit, evaluation data/path/split and metric. {'Seen-task results do not measure generalization.' if seen else 'Held-out designation requires evidence review.'}", | |
| 'count':0,'ranked_count':0,'config':{k:c[k] for k in ('environment_id','environment_revision','model','evaluation_data','metric','evaluation_scope')}} | |
| output[key]['count']+=1 | |
| if (row.get('result') or {}).get('verification')=='valid':output[key]['ranked_count']+=1 | |
| return list(output.values()) | |
| def results(group: str | None = None,legacy: bool=LEGACY): | |
| if not legacy: return {'items':[],'count':0,'compare_group':None} | |
| available=groups(True) | |
| if group is None:group=available[0]['id'] if available else None | |
| if group and not any(g['id']==group for g in available):raise HTTPException(404,'Comparison group not found.') | |
| items=[] | |
| for row in experiments(): | |
| if row['compare_group']!=group or not row.get('result'):continue | |
| result=row['result'];p=result['payload'];c=row['config'];base=p.get('baseline') | |
| fields={'score':round(p['score']*100,4),'baseline_score':'not measured' if base is None else str(round(base*100,4))+'%', | |
| 'agent':row['agent_id'],'method':c['method'],'status':'agent-run','timestamp':row['created_at'], | |
| 'description':row.get('description',row['title']), 'compare_group':group,'evaluation_scope':c['evaluation_scope'], | |
| 'report':p['report_url']} | |
| if p.get('job_url'):fields['job']=p['job_url'] | |
| if p.get('adapter_url'):fields['artifacts']=p['adapter_url'] | |
| items.append(markdown_item(row['id']+'.md',fields)) | |
| return {'items':items,'count':len(items),'compare_group':group} | |
| def verification(legacy: bool=LEGACY):return {r['id']+'.md':r['result']['verification'] for r in experiments() if r.get('result')} if legacy else {} | |
| def me(request: Request): | |
| current=auth.me(request);user=current['user'] or {} | |
| return {'logged_in':current['authenticated'],'oauth_configured':current['oauth_enabled'],'user':user.get('name'), | |
| 'avatar':user.get('avatar_url'),'csrf_token':current['csrf_token'],'is_editor':current['is_editor'], | |
| 'is_organizer':False} # board broadcast stays off for everyone | |
| def config(): | |
| return {'title':'PostTrain Arena','tagline':'Submit RL environment collections; a fixed recipe post-trains a fixed model on them and scores held-out tasks.', | |
| 'org':'benchflow','bucket':'posttrain-environments-v3','bucket_web_url':auth.origin(), | |
| 'score_field':'score','score_label':'Pass rate','score_unit':'%','score_order':'desc', | |
| 'secondary_field':'baseline_score','secondary_label':'Baseline','invite_url':'','api_url':auth.origin(), | |
| 'directory_url':'https://huggingface.co/agent-collaborations'} | |
| def experiment_example(): | |
| old=history()[0]['config'] | |
| return {'request_id':'replace-with-a-stable-unique-id','agent_id':None,'environment_id':old['environment_id'], | |
| **{k:old[k] for k in ('model','training_data','evaluation_data','method','parameters','metric','evaluation_scope')}} | |
| def stats():return {'sessions_counted':0} | |
| def traces():return {'items':[],'count':0} | |