Spaces:
Running
Running
Xiangyi Li
Retire per-task image builds and hosted profile runs (410); their Spaces were deleted
9fbaddb | """Hosted execution profiles for registered, immutable experiments. | |
| v1 is the reviewed shift-schedule JSON-artifact profile. v2 covers any submitted | |
| task package whose sandbox image has been built (images.py): the model writes | |
| the task's solve script, and the package's own oracle and verifier score it. | |
| """ | |
| import json | |
| import re | |
| from pathlib import Path | |
| from fastapi import APIRouter, HTTPException, Request | |
| from pydantic import BaseModel, ConfigDict, Field | |
| from huggingface_hub import hf_hub_download | |
| import arena_jobs as jobs | |
| import collab | |
| import environments as env | |
| import images | |
| router=APIRouter(prefix='/api/experiments') | |
| PACKAGE='benchflow/posttrain-agent-dogfood-20260921' | |
| REVISION='9f9440e50642f824098bf50791394a106b9c7b46' | |
| TASK='envs/shift-schedule-verify' | |
| IMAGE_DIGEST='sha256:626d7c098267d4ab027e859e4aa8351f4fce7a852f905aa6121b69267702c216' | |
| IMAGE_REVISION='d30b164b16319c50092d5f5de12f54876e640771' | |
| IMAGE='registry.hf.space/benchflow-posttrain-shift-schedule-eval-image@'+IMAGE_DIGEST | |
| RUNNER='arena/environment-v1/runner_environment.py' | |
| RUNNER_V2='arena/environment-v2/runner_environment.py' | |
| V1='shift-schedule-files-v1' | |
| STEPS=(20,50,100);RATES=(0.0001,0.0002);RANKS=(16,32) | |
| class RunRequest(BaseModel): | |
| model_config=ConfigDict(extra='forbid') | |
| request_id: str=Field(min_length=8,max_length=120) | |
| def v1_profile(): | |
| template={'request_id':'replace-with-a-stable-unique-id','agent_id':None,'environment_id':'YOUR_ENVIRONMENT_ID', | |
| 'model':{'repo_id':jobs.MODEL,'revision':jobs.REV}, | |
| 'training_data':{'repo_id':PACKAGE,'revision':REVISION,'path':TASK+'/oracle/solve.sh','split':'oracle-generated-seen'}, | |
| 'evaluation_data':{'repo_id':PACKAGE,'revision':REVISION,'path':TASK+'/verifier/test_outputs.py','split':'original-nine-checks'}, | |
| 'method':'LoRA SFT','parameters':{'steps':20,'rate':0.0001,'rank':16,'seed':42},'metric':'pass_rate','evaluation_scope':'seen'} | |
| return {'id':V1,'package_repo':PACKAGE,'package_revision':REVISION,'environment_path':TASK,'model':{'repo_id':jobs.MODEL,'revision':jobs.REV},'method':'LoRA SFT','evaluation_scope':'seen','parameters':{'steps':list(STEPS),'rate':list(RATES),'rank':list(RANKS),'seed':42},'experiment_template':template, | |
| 'sandbox_image':IMAGE,'sandbox_image_digest':IMAGE_DIGEST,'sandbox_image_revision':IMAGE_REVISION,'sandbox_image_source':'https://huggingface.co/spaces/benchflow/posttrain-shift-schedule-eval-image/tree/'+IMAGE_REVISION, | |
| 'decoding':{'do_sample':False,'max_new_tokens':4096},'output_format':'JSON files: violations.json and schedule.json','limits':'Legacy pinned task; oracle and empty controls, base-model evaluation, SFT, saved-adapter reload and original nine-check verifier.','run_url':'/api/experiments/{id}/run','budget_url':'/api/arena/budget'} | |
| def profiles(): | |
| hosted=[{**images.public(p),'model':{'repo_id':jobs.MODEL,'revision':jobs.REV},'method':'LoRA SFT','evaluation_scope':'seen','parameters':{'steps':list(STEPS),'rate':list(RATES),'rank':list(RANKS),'seed':42}, | |
| 'decoding':{'do_sample':False,'max_new_tokens':4096},'output_format':'One fenced bash script executed in the task environment; the package verifier scores the result.', | |
| 'limits':'One seen task per run: empty and oracle controls, base-model evaluation, oracle-supervised LoRA SFT, saved-adapter reload and the package verifier. Not held-out evaluation.', | |
| 'run_url':'/api/experiments/{id}/run','budget_url':'/api/arena/budget'} for p in images.ready()] | |
| return {'provider':'huggingface','profiles':[v1_profile()]+hosted,'build_url':'/api/v2/environments/{environment_id}/images', | |
| 'note':'Any validated environment task can get a hosted profile: build its sandbox image (BenchFlow editor), then register an experiment from the returned experiment_template.'} | |
| def parameters(p): | |
| if set(p)!={'steps','rate','rank','seed'} or type(p['steps']) is not int or p['steps'] not in STEPS or type(p['rank']) is not int or p['rank'] not in RANKS or type(p['rate']) not in (int,float) or p['rate'] not in RATES or type(p['seed']) is not int or p['seed']!=42: | |
| raise HTTPException(422,'Use profile parameters: steps 20/50/100, rate 0.0001/0.0002, rank 16/32, seed 42.') | |
| return p | |
| def checked_config(row): | |
| c=row['config'];source=next((r for r in env.environments() if r['id']==c['environment_id']),None) | |
| if not source:raise HTTPException(422,'Environment submission not found for this experiment.') | |
| if c['model']!={'repo_id':jobs.MODEL,'revision':jobs.REV} or c['method']!='LoRA SFT' or c['evaluation_scope']!='seen' or c['metric']!='pass_rate': | |
| raise HTTPException(422,'Hosted profiles require the pinned model, the oracle as training data, the original verifier, LoRA SFT and seen-task scope.') | |
| if (source.get('repo_type'),source.get('repo_id'),source.get('revision'),source.get('entry_path',''))==('dataset',PACKAGE,REVISION,'') and c['environment_revision']==REVISION: | |
| train={'repo_id':PACKAGE,'revision':REVISION,'path':TASK+'/oracle/solve.sh','split':'oracle-generated-seen'} | |
| evaluation={'repo_id':PACKAGE,'revision':REVISION,'path':TASK+'/verifier/test_outputs.py','split':'original-nine-checks'} | |
| if c['training_data']!=train or c['evaluation_data']!=evaluation: | |
| raise HTTPException(422,'This hosted profile requires its pinned oracle training source and original verifier.') | |
| p=parameters(c['parameters']) | |
| return {**p,'max_new_tokens':4096,'experiment_id':row['id'],'environment_id':c['environment_id'],'package_repo':PACKAGE,'package_revision':REVISION,'environment_path':TASK,'model':jobs.MODEL,'model_revision':jobs.REV,'sandbox_image':IMAGE,'sandbox_image_digest':IMAGE_DIGEST,'sandbox_image_revision':IMAGE_REVISION,'profile':V1} | |
| profile=next((p for p in images.ready() if p['environment_id']==c['environment_id'] and c['training_data']=={'repo_id':p['package']['repo_id'],'revision':p['package']['revision'],'path':p['package']['prefix']+'oracle/solve.sh','split':'oracle-generated-seen'}),None) | |
| if not profile: | |
| raise HTTPException(422,'No hosted execution profile for this environment task. Build its sandbox image at /api/v2/environments/{environment_id}/images, then register the experiment from its experiment_template.') | |
| pk=profile['package'] | |
| if c['environment_revision']!=pk['revision'] or c['evaluation_data']!={'repo_id':pk['repo_id'],'revision':pk['revision'],'path':pk['prefix']+'verifier/test.sh','split':'original-checks'}: | |
| raise HTTPException(422,'This hosted profile requires its pinned package revision and verifier/test.sh as evaluation data.') | |
| p=parameters(c['parameters']) | |
| return {**p,'max_new_tokens':4096,'experiment_id':row['id'],'environment_id':c['environment_id'],'package_repo':pk['repo_id'],'package_repo_type':pk['repo_type'],'package_revision':pk['revision'], | |
| 'environment_path':pk['prefix'].rstrip('/'),'task':profile['task'],'snapshot_path':profile['snapshot']['path'],'snapshot_revision':profile['snapshot']['revision'], | |
| 'model':jobs.MODEL,'model_revision':jobs.REV,'sandbox_image':profile['image_ref'],'sandbox_image_digest':profile['image_digest'],'sandbox_image_space':profile['image_space'],'sandbox_image_commit':profile['image_commit'], | |
| 'agent_timeout_sec':profile['hosted_timeouts']['agent_timeout_sec'],'verifier_timeout_sec':profile['hosted_timeouts']['verifier_timeout_sec'],'profile':'environment-v2:'+profile['id']} | |
| def owned_editor(experiment_id,request): | |
| user=jobs.session_operator(request);row=collab.experiment(experiment_id) | |
| if row['owner']!=user:raise HTTPException(403,'Only the experiment owner can allocate this workspace compute for it.') | |
| return user,row | |
| def recipe(experiment_id:str): | |
| row=collab.experiment(experiment_id) | |
| return {'launch':False,'config':checked_config(row),'allocation':jobs.quote(),'run_url':'/api/experiments/'+experiment_id+'/run'} | |
| def run(experiment_id:str,value:RunRequest,request:Request): | |
| user,row=owned_editor(experiment_id,request) | |
| if images.RETIRED: | |
| raise HTTPException(410,'Hosted per-task profiles are retired; their sandbox image Spaces were deleted on Sept 23. Run the collection against a challenge: /api/challenges.') | |
| previous=next((r for r in jobs.read()['runs'] if r.get('author')==user and r.get('request_key')==value.request_id),None) | |
| if previous: | |
| if previous.get('kind')!='environment-train' or previous.get('config',{}).get('experiment_id')!=experiment_id: | |
| raise HTTPException(409,'This request ID belongs to another execution. Reuse its original experiment or choose a new request ID.') | |
| return previous | |
| if row.get('result'):raise HTTPException(409,'This experiment already has an immutable result. Register a new experiment for another run.') | |
| config=checked_config(row) | |
| # Exact source revision is reserved with the job before any compute starts. | |
| return jobs.launch_environment(value.request_id,user,config,RUNNER if config['profile']==V1 else RUNNER_V2) | |
| def runs(experiment_id:str): | |
| collab.experiment(experiment_id) | |
| return [r for r in jobs.runs() if r.get('config',{}).get('experiment_id')==experiment_id] | |
| def checked_evaluation(value,total): | |
| return isinstance(value,dict) and value.get('status')=='completed' and value.get('sandbox_terminated') is True and type(value.get('passed_tests')) is int and value.get('total_tests')==total and 0<=value['passed_tests']<=total | |
| def collect(experiment_id:str,run_id:str,request:Request): | |
| user,row=owned_editor(experiment_id,request) | |
| existing=row.get('result') | |
| if existing: | |
| if existing.get('execution_run_id')==run_id and existing.get('payload',{}).get('request_id')=='collected-'+run_id:return row | |
| raise HTTPException(409,'This experiment already has another immutable result.') | |
| head=jobs.api().repo_info(jobs.REPO,repo_type='dataset').sha | |
| record=next((r for r in jobs.read(head)['runs'] if r['run_id']==run_id and r.get('config',{}).get('experiment_id')==experiment_id),None) | |
| if not record:raise HTTPException(404,'Run does not belong to this experiment.') | |
| if jobs.stage(record)!='COMPLETED':raise HTTPException(409,'Wait for the HF job to complete before collecting evidence.') | |
| try: | |
| report=json.loads(Path(hf_hub_download(jobs.REPO,record['report_path'],repo_type='dataset',revision=head,token=jobs.api().token)).read_text()) | |
| except Exception:raise HTTPException(409,'Completed runner report is not available.') from None | |
| if not isinstance(report,dict) or report.get('status')!='completed' or report.get('adapter_reloaded') is not True or report.get('sandbox_terminated') is not True or report.get('experiment_id')!=experiment_id: | |
| raise HTTPException(409,'Runner evidence is incomplete; do not rank this run.') | |
| if report.get('config')!={**record['config'],'run_id':run_id,'kind':record['kind'],'report_path':record['report_path']}:raise HTTPException(409,'Runner report configuration differs from its reserved job.') | |
| if type(report.get('global_steps')) is not int or report['global_steps']!=record['config']['steps']:raise HTTPException(409,'Runner optimizer steps do not match its reserved configuration.') | |
| if record['config'].get('max_new_tokens')!=4096:raise HTTPException(409,'This diagnostic run used a superseded generation budget. Run the current recipe before collecting a comparison result.') | |
| legacy=record['config'].get('profile')==V1 | |
| total=9 if legacy else report.get('total_tests') | |
| if type(total) is not int or not 1<=total<=500:raise HTTPException(409,'Runner report lacks a valid verifier check count.') | |
| controls=report.get('controls',{}) | |
| if not isinstance(controls,dict):raise HTTPException(409,'Runtime controls are incomplete or invalid.') | |
| empty=controls.get('empty',{});oracle=controls.get('oracle',{}) | |
| valid=checked_evaluation(empty,total) and checked_evaluation(oracle,total) and oracle['passed_tests']==total | |
| if legacy:valid=valid and empty['passed_tests']==0 | |
| else:valid=valid and empty['passed_tests']<total and empty.get('verifier_reward')==0 and oracle.get('verifier_reward')==1 | |
| if not valid:raise HTTPException(409,'Runtime controls are incomplete or invalid.') | |
| baseline=report.get('baseline',{});final=report.get('final',{}) | |
| if not (checked_evaluation(baseline,total) and checked_evaluation(final,total)): | |
| raise HTTPException(409,'Original verifier counts are missing or invalid.') | |
| adapter=report.get('adapter',{}) | |
| if not isinstance(adapter,dict) or adapter.get('repo')!=jobs.REPO or adapter.get('repo_type')!='dataset' or adapter.get('path')!='arena/adapters/'+run_id or not isinstance(adapter.get('revision'),str) or not re.fullmatch(r'[0-9a-f]{40}',adapter['revision']): | |
| raise HTTPException(409,'Runner adapter reference is missing or does not match this run.') | |
| payload=collab.Result(request_id='collected-'+run_id,baseline=baseline['passed_tests']/total,score=final['passed_tests']/total, | |
| report_url=f'https://huggingface.co/datasets/{jobs.REPO}/blob/{head}/{record["report_path"]}', | |
| job_url=record['job_url'],adapter_url=f'https://huggingface.co/datasets/{jobs.REPO}/tree/{adapter["revision"]}/{adapter["path"]}') | |
| # Store collector provenance atomically; participant Result input cannot set it. | |
| def store(rows): | |
| current=next((r for r in rows if r['id']==experiment_id),None) | |
| if not current:raise HTTPException(404,'Experiment not found in participant registry.') | |
| if current['owner']!=user:raise HTTPException(403,'Only the experiment owner may collect its result.') | |
| previous=current.get('result') | |
| if previous: | |
| if previous.get('execution_run_id')==run_id and previous.get('payload',{}).get('request_id')=='collected-'+run_id:return current | |
| raise HTTPException(409,'This experiment already has another immutable result.') | |
| if current['config']!=row['config']:raise HTTPException(409,'Experiment configuration changed while collecting evidence.') | |
| current['result']={'payload':payload.model_dump(),'verification':'pending','reported_at':collab.now(), | |
| 'verification_note':'Hosted runner evidence collected; awaiting organizer evidence review.','execution_run_id':run_id,'checks':{'passed':final['passed_tests'],'total':total,'baseline_passed':baseline['passed_tests']}} | |
| current['status']='reported' | |
| return current | |
| return env.replace_file(collab.EXPERIMENTS,store,[]) | |