Spaces:
Running
Running
Run pinned submitted environments on HF with evidence collection and reviewed result publication
Browse files- public_results.py +46 -2
- test_public_results.py +9 -0
public_results.py
CHANGED
|
@@ -1,5 +1,9 @@
|
|
| 1 |
"""Explicit organizer publication of reviewed, allowlisted result metadata."""
|
| 2 |
import json
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
from fastapi import APIRouter, HTTPException, Request
|
| 4 |
import collab
|
| 5 |
import environments as env
|
|
@@ -16,10 +20,50 @@ def sanitized(row):
|
|
| 16 |
'result':{'verification':'valid','reviewed_at':result.get('reviewed_at'),
|
| 17 |
'payload':{k:p[k] for k in ('baseline','score','report_url','job_url','adapter_url') if k in p}}}
|
| 18 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 19 |
def publish_feed():
|
| 20 |
rows=[r for r in collab.experiments() if r.get('public_result') and (r.get('result') or {}).get('verification')=='valid']
|
| 21 |
-
|
| 22 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 23 |
for _ in range(4):
|
| 24 |
head=client.repo_info(REPO,repo_type='dataset').sha
|
| 25 |
try:
|
|
|
|
| 1 |
"""Explicit organizer publication of reviewed, allowlisted result metadata."""
|
| 2 |
import json
|
| 3 |
+
import hashlib
|
| 4 |
+
import re
|
| 5 |
+
from pathlib import Path
|
| 6 |
+
from huggingface_hub import CommitOperationAdd, hf_hub_download
|
| 7 |
from fastapi import APIRouter, HTTPException, Request
|
| 8 |
import collab
|
| 9 |
import environments as env
|
|
|
|
| 20 |
'result':{'verification':'valid','reviewed_at':result.get('reviewed_at'),
|
| 21 |
'payload':{k:p[k] for k in ('baseline','score','report_url','job_url','adapter_url') if k in p}}}
|
| 22 |
|
| 23 |
+
def summary(row, report, raw_hash):
|
| 24 |
+
"""Public machine evidence without model text, prompts, credentials or private logs."""
|
| 25 |
+
def evaluation(value):
|
| 26 |
+
if not isinstance(value,dict):return None
|
| 27 |
+
safe={k:value[k] for k in ('status','passed_tests','total_tests','score','verifier_reward','sandbox_job_id','sandbox_terminated') if k in value}
|
| 28 |
+
safe['tests']=[{k:test[k] for k in ('name','status') if k in test} for test in value.get('tests',[]) if isinstance(test,dict)]
|
| 29 |
+
return safe
|
| 30 |
+
config=report.get('config',{})
|
| 31 |
+
return {'schema_version':1,'experiment_id':row['id'],'status':report.get('status'),
|
| 32 |
+
'scope':report.get('scope'),'verification':'valid','reviewed_at':row['result'].get('reviewed_at'),
|
| 33 |
+
'source_report':row['result']['payload']['report_url'],'source_report_sha256':raw_hash,
|
| 34 |
+
'job_url':row['result']['payload'].get('job_url'),
|
| 35 |
+
'config':{k:config[k] for k in ('run_id','experiment_id','environment_id','package_repo','package_revision','environment_path','model','model_revision','steps','rate','rank','seed','max_new_tokens','sandbox_image','sandbox_image_digest','sandbox_image_revision','profile') if k in config},
|
| 36 |
+
'decoding':report.get('decoding'),'source_file_sha256':report.get('source_file_sha256'),
|
| 37 |
+
'global_steps':report.get('global_steps'),'training_loss':report.get('training_loss'),
|
| 38 |
+
'adapter_reloaded':report.get('adapter_reloaded'),'sandbox_terminated':report.get('sandbox_terminated'),
|
| 39 |
+
'network_isolation':report.get('network_isolation'),'protocol_deviations':report.get('protocol_deviations'),
|
| 40 |
+
'controls':{k:evaluation(v) for k,v in report.get('controls',{}).items() if k in ('empty','oracle')},
|
| 41 |
+
'baseline':evaluation(report.get('baseline')),'final':evaluation(report.get('final'))}
|
| 42 |
+
|
| 43 |
def publish_feed():
|
| 44 |
rows=[r for r in collab.experiments() if r.get('public_result') and (r.get('result') or {}).get('verification')=='valid']
|
| 45 |
+
client=env.api(); public_rows=[sanitized(r) for r in rows]
|
| 46 |
+
operations=[]; summary_paths={}
|
| 47 |
+
for row in rows:
|
| 48 |
+
run_id=(row.get('result') or {}).get('execution_run_id')
|
| 49 |
+
if not run_id or not re.fullmatch(r'arena-[0-9a-f]{12}',run_id):continue
|
| 50 |
+
url=row['result']['payload']['report_url']
|
| 51 |
+
pattern=r'https://huggingface.co/datasets/'+re.escape(env.REPO)+r'/(?:blob|resolve)/([0-9a-f]{40})/(arena/results/'+re.escape(run_id)+r'\.json)'
|
| 52 |
+
match=re.fullmatch(pattern,url)
|
| 53 |
+
if not match:raise HTTPException(409,'Collected report does not match its reserved run.')
|
| 54 |
+
raw=Path(hf_hub_download(env.REPO,match[2],repo_type='dataset',revision=match[1],token=client.token)).read_bytes()
|
| 55 |
+
if len(raw)>2_000_000:raise HTTPException(409,'Report is too large for public evidence export.')
|
| 56 |
+
report=json.loads(raw)
|
| 57 |
+
if report.get('status')!='completed' or report.get('experiment_id')!=row['id']:raise HTTPException(409,'Cannot publish incomplete runner evidence.')
|
| 58 |
+
path='reports/'+run_id+'.json';summary_paths[row['id']]=path
|
| 59 |
+
operations.append(CommitOperationAdd(path_in_repo=path,path_or_fileobj=json.dumps(summary(row,report,hashlib.sha256(raw).hexdigest()),indent=2).encode()))
|
| 60 |
+
if operations:
|
| 61 |
+
head=client.repo_info(REPO,repo_type='dataset').sha
|
| 62 |
+
try: receipt=client.create_commit(repo_id=REPO,repo_type='dataset',parent_commit=head,operations=operations,commit_message='Publish sanitized reviewed execution evidence')
|
| 63 |
+
except Exception:raise HTTPException(503,'Evidence publication could not be confirmed. Retry the same publication request.') from None
|
| 64 |
+
for row in public_rows:
|
| 65 |
+
if row['id'] in summary_paths:row['result']['payload']['report_url']=f'https://huggingface.co/datasets/{REPO}/blob/{receipt.oid}/{summary_paths[row["id"]]}'
|
| 66 |
+
content={'schema_version':1,'generated_at':collab.now(),'experiments':public_rows}
|
| 67 |
for _ in range(4):
|
| 68 |
head=client.repo_info(REPO,repo_type='dataset').sha
|
| 69 |
try:
|
test_public_results.py
CHANGED
|
@@ -11,6 +11,15 @@ class Tests(unittest.TestCase):
|
|
| 11 |
value=public.sanitized(self.row)
|
| 12 |
self.assertNotIn('notes',value);self.assertNotIn('training_data',value['config']);self.assertNotIn('verification_note',value['result'])
|
| 13 |
self.assertIsNone(value['result']['payload']['baseline'])
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 14 |
def test_unreviewed_results_rejected(self):
|
| 15 |
for status in ('pending','invalid'):
|
| 16 |
self.row['result']['verification']=status
|
|
|
|
| 11 |
value=public.sanitized(self.row)
|
| 12 |
self.assertNotIn('notes',value);self.assertNotIn('training_data',value['config']);self.assertNotIn('verification_note',value['result'])
|
| 13 |
self.assertIsNone(value['result']['payload']['baseline'])
|
| 14 |
+
def test_machine_evidence_excludes_generated_text_and_logs(self):
|
| 15 |
+
report={'status':'completed','experiment_id':self.row['id'],'config':{'model':'org/model','HF_TOKEN':'never-publish'},'baseline':{'passed_tests':8,'total_tests':9,'generated_text':'private-output','verifier_log':'private-log','tests':[{'name':'test_one','status':'passed','message':'private-failure-details'}]},'controls':{'oracle':{'artifacts':{'private':'data'},'passed_tests':9,'total_tests':9}}}
|
| 16 |
+
value=public.summary(self.row,report,'a'*64)
|
| 17 |
+
import json
|
| 18 |
+
text=json.dumps(value)
|
| 19 |
+
for secret in ('never-publish','private-output','private-log','private-failure-details'):self.assertNotIn(secret,text)
|
| 20 |
+
self.assertNotIn('artifacts',value['controls']['oracle'])
|
| 21 |
+
self.assertEqual(value['baseline']['passed_tests'],8)
|
| 22 |
+
self.assertEqual(value['baseline']['tests'],[{'name':'test_one','status':'passed'}])
|
| 23 |
def test_unreviewed_results_rejected(self):
|
| 24 |
for status in ('pending','invalid'):
|
| 25 |
self.row['result']['verification']=status
|