Spaces:
Running
Running
Switch canceled H200 queue to HF A100 QLoRA with five dollar compute bound
Browse files- hf_gpu_backend.py +7 -7
- hf_gpu_train.py +5 -4
- index.html +4 -4
hf_gpu_backend.py
CHANGED
|
@@ -6,11 +6,11 @@ from huggingface_hub import HfApi,hf_hub_download
|
|
| 6 |
from huggingface_hub.errors import EntryNotFoundError
|
| 7 |
from fastapi import HTTPException
|
| 8 |
REPO='benchflow/posttrain-lab-20260920-artifacts'
|
| 9 |
-
LEDGER='training/hf-gpu-experiment-
|
| 10 |
-
PREVIOUS_JOB='
|
| 11 |
MODEL='Qwen/Qwen3.6-27B'
|
| 12 |
MODEL_REV='6a9e13bd6fc8f0983b9b99948120bc37f49c13e9'
|
| 13 |
-
FLAVOR='
|
| 14 |
TIMEOUT=7200
|
| 15 |
|
| 16 |
def hub(): return HfApi(token=os.environ.get('HF_TOKEN'))
|
|
@@ -23,7 +23,7 @@ def write_ledger(record,parent):
|
|
| 23 |
|
| 24 |
def recipe(model,steps,rate,rank):
|
| 25 |
if model!=MODEL or steps not in (20,50,100) or rate not in (2e-5,5e-5,1e-4) or rank not in (8,16,32):raise HTTPException(422,'Unsupported training settings')
|
| 26 |
-
return dict(model=model,model_revision=MODEL_REV,steps=steps,rate=rate,rank=rank,max_length=
|
| 27 |
|
| 28 |
def readiness():
|
| 29 |
blockers=[];record=None
|
|
@@ -36,7 +36,7 @@ def readiness():
|
|
| 36 |
def launch(model,steps,rate,rank):
|
| 37 |
config=recipe(model,steps,rate,rank)
|
| 38 |
previous=hub().inspect_job(job_id=PREVIOUS_JOB,namespace='benchflow')
|
| 39 |
-
if str(previous.status.stage)
|
| 40 |
if read_ledger():raise HTTPException(409,'GPU run already reserved. No duplicate submitted.')
|
| 41 |
with urllib.request.urlopen('https://huggingface.co/api/jobs/hardware',timeout=20) as r:hardware=json.load(r)
|
| 42 |
item=next(x for x in hardware if x['name']==FLAVOR)
|
|
@@ -49,11 +49,11 @@ def launch(model,steps,rate,rank):
|
|
| 49 |
config['data_revision']=head
|
| 50 |
if read_ledger():raise HTTPException(409,'GPU run already reserved')
|
| 51 |
run='hf-gpu-'+uuid.uuid4().hex[:12]
|
| 52 |
-
record=dict(run_id=run,provider='huggingface',model=model,status='reserved',recipe=config,budget_cap_usd=200,max_job_cost_usd=round(bound,4),prior_attempt_job_id=PREVIOUS_JOB,cumulative_compute_reservation_usd=round(bound+
|
| 53 |
try: reserved=write_ledger(record,head)
|
| 54 |
except Exception:raise HTTPException(409,'Could not atomically reserve job. Check history.') from None
|
| 55 |
bootstrap="from huggingface_hub import hf_hub_download; import runpy,os,json; c=json.loads(os.environ['TRAIN_CONFIG']); runpy.run_path(hf_hub_download(c['data_repo'],'training/hf-corpus/hf_gpu_train.py',repo_type='dataset',revision=c['data_revision']),run_name='__main__')"
|
| 56 |
-
command=['bash','-lc',"python -m pip install --break-system-packages --no-cache-dir transformers==5.3.0 peft==0.18.1 datasets==4.6.1 accelerate==1.12.0 huggingface_hub==1.32.0 && python -u -c "+__import__('shlex').quote(bootstrap)]
|
| 57 |
try:
|
| 58 |
job=api.run_job(image='pytorch/pytorch:2.10.0-cuda12.8-cudnn9-runtime',command=command,env={'TRAIN_CONFIG':json.dumps(config),'RUN_ID':run,'PYTHONUNBUFFERED':'1'},secrets={'HF_TOKEN':os.environ['HF_TOKEN']},flavor=FLAVOR,timeout=TIMEOUT,namespace='benchflow',name=run,labels={'experiment':'posttrain-hf-gpu','run_id':run})
|
| 59 |
record.update(job_id=job.id,job_url=f'https://huggingface.co/jobs/benchflow/{job.id}',status=str(job.status.stage))
|
|
|
|
| 6 |
from huggingface_hub.errors import EntryNotFoundError
|
| 7 |
from fastapi import HTTPException
|
| 8 |
REPO='benchflow/posttrain-lab-20260920-artifacts'
|
| 9 |
+
LEDGER='training/hf-gpu-experiment-v3.json'
|
| 10 |
+
PREVIOUS_JOB='6ab0c03552d0dbd7f1d77567'
|
| 11 |
MODEL='Qwen/Qwen3.6-27B'
|
| 12 |
MODEL_REV='6a9e13bd6fc8f0983b9b99948120bc37f49c13e9'
|
| 13 |
+
FLAVOR='a100-large'
|
| 14 |
TIMEOUT=7200
|
| 15 |
|
| 16 |
def hub(): return HfApi(token=os.environ.get('HF_TOKEN'))
|
|
|
|
| 23 |
|
| 24 |
def recipe(model,steps,rate,rank):
|
| 25 |
if model!=MODEL or steps not in (20,50,100) or rate not in (2e-5,5e-5,1e-4) or rank not in (8,16,32):raise HTTPException(422,'Unsupported training settings')
|
| 26 |
+
return dict(model=model,model_revision=MODEL_REV,steps=steps,rate=rate,rank=rank,max_length=8192,quantization="nf4",flavor=FLAVOR,timeout_seconds=TIMEOUT,data_repo=REPO)
|
| 27 |
|
| 28 |
def readiness():
|
| 29 |
blockers=[];record=None
|
|
|
|
| 36 |
def launch(model,steps,rate,rank):
|
| 37 |
config=recipe(model,steps,rate,rank)
|
| 38 |
previous=hub().inspect_job(job_id=PREVIOUS_JOB,namespace='benchflow')
|
| 39 |
+
if str(previous.status.stage)not in ('ERROR','CANCELED'):raise HTTPException(409,'Prior attempt is not confirmed failed; retry blocked.')
|
| 40 |
if read_ledger():raise HTTPException(409,'GPU run already reserved. No duplicate submitted.')
|
| 41 |
with urllib.request.urlopen('https://huggingface.co/api/jobs/hardware',timeout=20) as r:hardware=json.load(r)
|
| 42 |
item=next(x for x in hardware if x['name']==FLAVOR)
|
|
|
|
| 49 |
config['data_revision']=head
|
| 50 |
if read_ledger():raise HTTPException(409,'GPU run already reserved')
|
| 51 |
run='hf-gpu-'+uuid.uuid4().hex[:12]
|
| 52 |
+
record=dict(run_id=run,provider='huggingface',model=model,status='reserved',recipe=config,budget_cap_usd=200,max_job_cost_usd=round(bound,4),prior_attempt_job_id=PREVIOUS_JOB,cumulative_compute_reservation_usd=round(bound+20,4),updated_at=datetime.now(timezone.utc).isoformat(),artifact_url=f'https://huggingface.co/datasets/{REPO}/tree/main/training/{run}/adapter')
|
| 53 |
try: reserved=write_ledger(record,head)
|
| 54 |
except Exception:raise HTTPException(409,'Could not atomically reserve job. Check history.') from None
|
| 55 |
bootstrap="from huggingface_hub import hf_hub_download; import runpy,os,json; c=json.loads(os.environ['TRAIN_CONFIG']); runpy.run_path(hf_hub_download(c['data_repo'],'training/hf-corpus/hf_gpu_train.py',repo_type='dataset',revision=c['data_revision']),run_name='__main__')"
|
| 56 |
+
command=['bash','-lc',"python -m pip install --break-system-packages --no-cache-dir transformers==5.3.0 peft==0.18.1 datasets==4.6.1 accelerate==1.12.0 huggingface_hub==1.32.0 bitsandbytes==0.50.2 && python -u -c "+__import__('shlex').quote(bootstrap)]
|
| 57 |
try:
|
| 58 |
job=api.run_job(image='pytorch/pytorch:2.10.0-cuda12.8-cudnn9-runtime',command=command,env={'TRAIN_CONFIG':json.dumps(config),'RUN_ID':run,'PYTHONUNBUFFERED':'1'},secrets={'HF_TOKEN':os.environ['HF_TOKEN']},flavor=FLAVOR,timeout=TIMEOUT,namespace='benchflow',name=run,labels={'experiment':'posttrain-hf-gpu','run_id':run})
|
| 59 |
record.update(job_id=job.id,job_url=f'https://huggingface.co/jobs/benchflow/{job.id}',status=str(job.status.stage))
|
hf_gpu_train.py
CHANGED
|
@@ -4,7 +4,7 @@ from collections.abc import Mapping
|
|
| 4 |
from pathlib import Path
|
| 5 |
from huggingface_hub import HfApi,hf_hub_download
|
| 6 |
|
| 7 |
-
def tokenize_rows(rows, tokenizer, max_length=
|
| 8 |
output=[]; skipped=0
|
| 9 |
for original in rows:
|
| 10 |
row=json.loads(json.dumps(original))
|
|
@@ -30,8 +30,8 @@ def tokenize_rows(rows, tokenizer, max_length=16384):
|
|
| 30 |
|
| 31 |
def main():
|
| 32 |
import torch
|
| 33 |
-
from transformers import AutoTokenizer,AutoModelForImageTextToText,Trainer,TrainingArguments,DataCollatorForSeq2Seq
|
| 34 |
-
from peft import LoraConfig,get_peft_model
|
| 35 |
from datasets import Dataset
|
| 36 |
cfg=json.loads(os.environ['TRAIN_CONFIG']); api=HfApi(); run=os.environ['RUN_ID']
|
| 37 |
out=Path('/tmp/adapter'); out.mkdir()
|
|
@@ -42,7 +42,8 @@ def main():
|
|
| 42 |
examples,skipped=tokenize_rows(rows,tokenizer)
|
| 43 |
if not examples: raise RuntimeError('No valid full-context training examples')
|
| 44 |
print(json.dumps({'stage':'data-ready','examples':len(examples),'skipped_overlength_or_template':skipped,'gpu':torch.cuda.get_device_name(0)}),flush=True)
|
| 45 |
-
model=AutoModelForImageTextToText.from_pretrained(cfg['model'],revision=cfg['model_revision'],dtype=torch.bfloat16,attn_implementation='sdpa',device_map={'':'cuda:0'})
|
|
|
|
| 46 |
model.config.use_cache=False
|
| 47 |
model=get_peft_model(model,LoraConfig(r=cfg['rank'],lora_alpha=cfg['rank']*2,lora_dropout=0.05,bias='none',task_type='CAUSAL_LM',target_modules=['q_proj','k_proj','v_proj','o_proj']))
|
| 48 |
model.print_trainable_parameters()
|
|
|
|
| 4 |
from pathlib import Path
|
| 5 |
from huggingface_hub import HfApi,hf_hub_download
|
| 6 |
|
| 7 |
+
def tokenize_rows(rows, tokenizer, max_length=8192):
|
| 8 |
output=[]; skipped=0
|
| 9 |
for original in rows:
|
| 10 |
row=json.loads(json.dumps(original))
|
|
|
|
| 30 |
|
| 31 |
def main():
|
| 32 |
import torch
|
| 33 |
+
from transformers import AutoTokenizer,AutoModelForImageTextToText,Trainer,TrainingArguments,DataCollatorForSeq2Seq,BitsAndBytesConfig
|
| 34 |
+
from peft import LoraConfig,get_peft_model,prepare_model_for_kbit_training
|
| 35 |
from datasets import Dataset
|
| 36 |
cfg=json.loads(os.environ['TRAIN_CONFIG']); api=HfApi(); run=os.environ['RUN_ID']
|
| 37 |
out=Path('/tmp/adapter'); out.mkdir()
|
|
|
|
| 42 |
examples,skipped=tokenize_rows(rows,tokenizer)
|
| 43 |
if not examples: raise RuntimeError('No valid full-context training examples')
|
| 44 |
print(json.dumps({'stage':'data-ready','examples':len(examples),'skipped_overlength_or_template':skipped,'gpu':torch.cuda.get_device_name(0)}),flush=True)
|
| 45 |
+
model=AutoModelForImageTextToText.from_pretrained(cfg['model'],revision=cfg['model_revision'],dtype=torch.bfloat16,attn_implementation='sdpa',device_map={'':'cuda:0'},quantization_config=BitsAndBytesConfig(load_in_4bit=True,bnb_4bit_quant_type='nf4',bnb_4bit_compute_dtype=torch.bfloat16,bnb_4bit_use_double_quant=True))
|
| 46 |
+
model=prepare_model_for_kbit_training(model,use_gradient_checkpointing=True)
|
| 47 |
model.config.use_cache=False
|
| 48 |
model=get_peft_model(model,LoraConfig(r=cfg['rank'],lora_alpha=cfg['rank']*2,lora_dropout=0.05,bias='none',task_type='CAUSAL_LM',target_modules=['q_proj','k_proj','v_proj','o_proj']))
|
| 49 |
model.print_trainable_parameters()
|
index.html
CHANGED
|
@@ -18,13 +18,13 @@
|
|
| 18 |
<div class="layout">
|
| 19 |
<section class="panel" aria-labelledby="setup-heading"><div class="section-title"><h2 id="setup-heading">Your experiment</h2><span class="section-label">Configure</span></div>
|
| 20 |
<div class="field-row"><span class="step">01</span><div><label class="field-title" for="model">Start with a model</label><select id="model"><option value="Qwen/Qwen3.6-27B">Qwen3.6 · 27B</option></select><div class="model-info"><p class="help" id="model-help">LoRA fine-tuning · Hugging Face GPU Jobs</p><a class="help" id="model-link" href="https://huggingface.co/Qwen/Qwen3.6-27B" target="_blank" rel="noopener">Model card ↗</a></div></div></div>
|
| 21 |
-
<div class="field-row"><span class="step">02</span><div><div class="field-title">Training dataset</div><div class="env"><div><strong>SkillsBench / Verified teacher traces</strong><p>10 verified trajectories · 394 teacher exchanges</p></div><span class="tag">SFT corpus</span></div><p class="help">Available HF subset of Carrie’s corpus. Full-context examples up to
|
| 22 |
-
<div class="field-row"><span class="step">03</span><div><div class="field-title">Shape the training recipe</div><div class="method">Supervised fine-tuning <span>LoRA · GRPO off</span></div><div class="parameters"><div><label for="steps">Optimizer steps</label><select id="steps"><option selected>20</option><option>50</option><option>100</option></select></div><div><label for="rate">Learning rate</label><select id="rate"><option value="0.00002">2 × 10⁻⁵</option><option value="0.00005">5 × 10⁻⁵</option><option selected value="0.0001">1 × 10⁻⁴</option></select></div><div><label for="rank">LoRA rank</label><select id="rank"><option>8</option><option>16</option><option selected>32</option></select></div></div><p class="help">20 optimizer steps for the first run. A two-hour HF timeout bounds compute.</p></div></div>
|
| 23 |
<div class="recipe-footer"><span class="help">Existing corpus. Traceable training receipt.</span><button class="quiet" id="recipe-download">Download recipe ↓</button></div><details id="recipe-details"><summary>Inspect recipe</summary><pre id="recipe">Loading recipe…</pre></details>
|
| 24 |
</section>
|
| 25 |
<aside>
|
| 26 |
-
<section class="panel summary"><div class="summary-head"><span class="eyebrow">Run preview</span><span class="tag">Real SFT</span></div><h2>Train the model.<br>Keep the evidence.</h2><dl><div><dt>Model</dt><dd id="chosen-model">Qwen3.6 · 27B</dd></div><div><dt>Recipe</dt><dd id="chosen-recipe">20 steps · rank 32</dd></div><div><dt>Compute</dt><dd>Hugging Face
|
| 27 |
-
<section class="training-panel" aria-labelledby="training-heading"><div class="training-title"><h2 id="training-heading">Connection & budget</h2><span class="tag" id="training-badge">Checking</span></div><details open><summary>Training readiness</summary><ul id="training-blockers" class="help"><li>Checking…</li></ul></details><p id="training-status" role="status" class="status"></p><div class="budget"><span>Total experiment cap</span><strong>$200</strong></div><p class="help">First run:
|
| 28 |
</aside></div>
|
| 29 |
<section aria-labelledby="runs-heading"><div class="runs-header"><div><div class="eyebrow">Your experiment log</div><h2 id="runs-heading">Run history</h2></div><button class="quiet" id="refresh">Refresh ↻</button></div><div id="run-status" role="status" aria-live="polite"></div><div id="runs"><div class="empty">Loading runs…</div></div></section>
|
| 30 |
<footer><span>PostTrain Arena / BenchFlow</span><span class="metric"><strong>Δ = score after − score before.</strong> Measured only after training and evaluation.</span></footer>
|
|
|
|
| 18 |
<div class="layout">
|
| 19 |
<section class="panel" aria-labelledby="setup-heading"><div class="section-title"><h2 id="setup-heading">Your experiment</h2><span class="section-label">Configure</span></div>
|
| 20 |
<div class="field-row"><span class="step">01</span><div><label class="field-title" for="model">Start with a model</label><select id="model"><option value="Qwen/Qwen3.6-27B">Qwen3.6 · 27B</option></select><div class="model-info"><p class="help" id="model-help">LoRA fine-tuning · Hugging Face GPU Jobs</p><a class="help" id="model-link" href="https://huggingface.co/Qwen/Qwen3.6-27B" target="_blank" rel="noopener">Model card ↗</a></div></div></div>
|
| 21 |
+
<div class="field-row"><span class="step">02</span><div><div class="field-title">Training dataset</div><div class="env"><div><strong>SkillsBench / Verified teacher traces</strong><p>10 verified trajectories · 394 teacher exchanges</p></div><span class="tag">SFT corpus</span></div><p class="help">Available HF subset of Carrie’s corpus. Full-context examples up to 8,192 tokens are used. This is a pipeline test; no held-out performance claim.</p></div></div>
|
| 22 |
+
<div class="field-row"><span class="step">03</span><div><div class="field-title">Shape the training recipe</div><div class="method">Supervised fine-tuning <span>4-bit LoRA · GRPO off</span></div><div class="parameters"><div><label for="steps">Optimizer steps</label><select id="steps"><option selected>20</option><option>50</option><option>100</option></select></div><div><label for="rate">Learning rate</label><select id="rate"><option value="0.00002">2 × 10⁻⁵</option><option value="0.00005">5 × 10⁻⁵</option><option selected value="0.0001">1 × 10⁻⁴</option></select></div><div><label for="rank">LoRA rank</label><select id="rank"><option>8</option><option>16</option><option selected>32</option></select></div></div><p class="help">20 optimizer steps for the first run. A two-hour HF timeout bounds compute.</p></div></div>
|
| 23 |
<div class="recipe-footer"><span class="help">Existing corpus. Traceable training receipt.</span><button class="quiet" id="recipe-download">Download recipe ↓</button></div><details id="recipe-details"><summary>Inspect recipe</summary><pre id="recipe">Loading recipe…</pre></details>
|
| 24 |
</section>
|
| 25 |
<aside>
|
| 26 |
+
<section class="panel summary"><div class="summary-head"><span class="eyebrow">Run preview</span><span class="tag">Real SFT</span></div><h2>Train the model.<br>Keep the evidence.</h2><dl><div><dt>Model</dt><dd id="chosen-model">Qwen3.6 · 27B</dd></div><div><dt>Recipe</dt><dd id="chosen-recipe">20 steps · rank 32</dd></div><div><dt>Compute</dt><dd>Hugging Face A100 80GB GPU</dd></div><div><dt>You’ll receive</dt><dd>Trained adapter & job receipt</dd></div></dl><button class="primary" id="launch" disabled>Checking training access…</button><p class="footnote">Starts paid model training. Does not deploy a serving endpoint or run benchmark evaluation.</p><div id="launch-status" class="status" role="status" aria-live="polite"></div></section>
|
| 27 |
+
<section class="training-panel" aria-labelledby="training-heading"><div class="training-title"><h2 id="training-heading">Connection & budget</h2><span class="tag" id="training-badge">Checking</span></div><details open><summary>Training readiness</summary><ul id="training-blockers" class="help"><li>Checking…</li></ul></details><p id="training-status" role="status" class="status"></p><div class="budget"><span>Total experiment cap</span><strong>$200</strong></div><p class="help">First run: A100 80GB, about $2.50/hour, two-hour timeout (about $5 maximum compute). Remaining budget stays unallocated. Repeat submissions are blocked, including after a restart. Reservation is not a billing charge.</p></section>
|
| 28 |
</aside></div>
|
| 29 |
<section aria-labelledby="runs-heading"><div class="runs-header"><div><div class="eyebrow">Your experiment log</div><h2 id="runs-heading">Run history</h2></div><button class="quiet" id="refresh">Refresh ↻</button></div><div id="run-status" role="status" aria-live="polite"></div><div id="runs"><div class="empty">Loading runs…</div></div></section>
|
| 30 |
<footer><span>PostTrain Arena / BenchFlow</span><span class="metric"><strong>Δ = score after − score before.</strong> Measured only after training and evaluation.</span></footer>
|