xdotli commited on
Commit
73be92c
·
verified ·
1 Parent(s): a5a8001

Switch canceled H200 queue to HF A100 QLoRA with five dollar compute bound

Browse files
Files changed (3) hide show
  1. hf_gpu_backend.py +7 -7
  2. hf_gpu_train.py +5 -4
  3. index.html +4 -4
hf_gpu_backend.py CHANGED
@@ -6,11 +6,11 @@ from huggingface_hub import HfApi,hf_hub_download
6
  from huggingface_hub.errors import EntryNotFoundError
7
  from fastapi import HTTPException
8
  REPO='benchflow/posttrain-lab-20260920-artifacts'
9
- LEDGER='training/hf-gpu-experiment-v2.json'
10
- PREVIOUS_JOB='6ab0bf0a52d0dbd7f1d77521'
11
  MODEL='Qwen/Qwen3.6-27B'
12
  MODEL_REV='6a9e13bd6fc8f0983b9b99948120bc37f49c13e9'
13
- FLAVOR='h200'
14
  TIMEOUT=7200
15
 
16
  def hub(): return HfApi(token=os.environ.get('HF_TOKEN'))
@@ -23,7 +23,7 @@ def write_ledger(record,parent):
23
 
24
  def recipe(model,steps,rate,rank):
25
  if model!=MODEL or steps not in (20,50,100) or rate not in (2e-5,5e-5,1e-4) or rank not in (8,16,32):raise HTTPException(422,'Unsupported training settings')
26
- return dict(model=model,model_revision=MODEL_REV,steps=steps,rate=rate,rank=rank,max_length=16384,flavor=FLAVOR,timeout_seconds=TIMEOUT,data_repo=REPO)
27
 
28
  def readiness():
29
  blockers=[];record=None
@@ -36,7 +36,7 @@ def readiness():
36
  def launch(model,steps,rate,rank):
37
  config=recipe(model,steps,rate,rank)
38
  previous=hub().inspect_job(job_id=PREVIOUS_JOB,namespace='benchflow')
39
- if str(previous.status.stage)!='ERROR':raise HTTPException(409,'Prior attempt is not confirmed failed; retry blocked.')
40
  if read_ledger():raise HTTPException(409,'GPU run already reserved. No duplicate submitted.')
41
  with urllib.request.urlopen('https://huggingface.co/api/jobs/hardware',timeout=20) as r:hardware=json.load(r)
42
  item=next(x for x in hardware if x['name']==FLAVOR)
@@ -49,11 +49,11 @@ def launch(model,steps,rate,rank):
49
  config['data_revision']=head
50
  if read_ledger():raise HTTPException(409,'GPU run already reserved')
51
  run='hf-gpu-'+uuid.uuid4().hex[:12]
52
- record=dict(run_id=run,provider='huggingface',model=model,status='reserved',recipe=config,budget_cap_usd=200,max_job_cost_usd=round(bound,4),prior_attempt_job_id=PREVIOUS_JOB,cumulative_compute_reservation_usd=round(bound+10,4),updated_at=datetime.now(timezone.utc).isoformat(),artifact_url=f'https://huggingface.co/datasets/{REPO}/tree/main/training/{run}/adapter')
53
  try: reserved=write_ledger(record,head)
54
  except Exception:raise HTTPException(409,'Could not atomically reserve job. Check history.') from None
55
  bootstrap="from huggingface_hub import hf_hub_download; import runpy,os,json; c=json.loads(os.environ['TRAIN_CONFIG']); runpy.run_path(hf_hub_download(c['data_repo'],'training/hf-corpus/hf_gpu_train.py',repo_type='dataset',revision=c['data_revision']),run_name='__main__')"
56
- command=['bash','-lc',"python -m pip install --break-system-packages --no-cache-dir transformers==5.3.0 peft==0.18.1 datasets==4.6.1 accelerate==1.12.0 huggingface_hub==1.32.0 && python -u -c "+__import__('shlex').quote(bootstrap)]
57
  try:
58
  job=api.run_job(image='pytorch/pytorch:2.10.0-cuda12.8-cudnn9-runtime',command=command,env={'TRAIN_CONFIG':json.dumps(config),'RUN_ID':run,'PYTHONUNBUFFERED':'1'},secrets={'HF_TOKEN':os.environ['HF_TOKEN']},flavor=FLAVOR,timeout=TIMEOUT,namespace='benchflow',name=run,labels={'experiment':'posttrain-hf-gpu','run_id':run})
59
  record.update(job_id=job.id,job_url=f'https://huggingface.co/jobs/benchflow/{job.id}',status=str(job.status.stage))
 
6
  from huggingface_hub.errors import EntryNotFoundError
7
  from fastapi import HTTPException
8
  REPO='benchflow/posttrain-lab-20260920-artifacts'
9
+ LEDGER='training/hf-gpu-experiment-v3.json'
10
+ PREVIOUS_JOB='6ab0c03552d0dbd7f1d77567'
11
  MODEL='Qwen/Qwen3.6-27B'
12
  MODEL_REV='6a9e13bd6fc8f0983b9b99948120bc37f49c13e9'
13
+ FLAVOR='a100-large'
14
  TIMEOUT=7200
15
 
16
  def hub(): return HfApi(token=os.environ.get('HF_TOKEN'))
 
23
 
24
  def recipe(model,steps,rate,rank):
25
  if model!=MODEL or steps not in (20,50,100) or rate not in (2e-5,5e-5,1e-4) or rank not in (8,16,32):raise HTTPException(422,'Unsupported training settings')
26
+ return dict(model=model,model_revision=MODEL_REV,steps=steps,rate=rate,rank=rank,max_length=8192,quantization="nf4",flavor=FLAVOR,timeout_seconds=TIMEOUT,data_repo=REPO)
27
 
28
  def readiness():
29
  blockers=[];record=None
 
36
  def launch(model,steps,rate,rank):
37
  config=recipe(model,steps,rate,rank)
38
  previous=hub().inspect_job(job_id=PREVIOUS_JOB,namespace='benchflow')
39
+ if str(previous.status.stage)not in ('ERROR','CANCELED'):raise HTTPException(409,'Prior attempt is not confirmed failed; retry blocked.')
40
  if read_ledger():raise HTTPException(409,'GPU run already reserved. No duplicate submitted.')
41
  with urllib.request.urlopen('https://huggingface.co/api/jobs/hardware',timeout=20) as r:hardware=json.load(r)
42
  item=next(x for x in hardware if x['name']==FLAVOR)
 
49
  config['data_revision']=head
50
  if read_ledger():raise HTTPException(409,'GPU run already reserved')
51
  run='hf-gpu-'+uuid.uuid4().hex[:12]
52
+ record=dict(run_id=run,provider='huggingface',model=model,status='reserved',recipe=config,budget_cap_usd=200,max_job_cost_usd=round(bound,4),prior_attempt_job_id=PREVIOUS_JOB,cumulative_compute_reservation_usd=round(bound+20,4),updated_at=datetime.now(timezone.utc).isoformat(),artifact_url=f'https://huggingface.co/datasets/{REPO}/tree/main/training/{run}/adapter')
53
  try: reserved=write_ledger(record,head)
54
  except Exception:raise HTTPException(409,'Could not atomically reserve job. Check history.') from None
55
  bootstrap="from huggingface_hub import hf_hub_download; import runpy,os,json; c=json.loads(os.environ['TRAIN_CONFIG']); runpy.run_path(hf_hub_download(c['data_repo'],'training/hf-corpus/hf_gpu_train.py',repo_type='dataset',revision=c['data_revision']),run_name='__main__')"
56
+ command=['bash','-lc',"python -m pip install --break-system-packages --no-cache-dir transformers==5.3.0 peft==0.18.1 datasets==4.6.1 accelerate==1.12.0 huggingface_hub==1.32.0 bitsandbytes==0.50.2 && python -u -c "+__import__('shlex').quote(bootstrap)]
57
  try:
58
  job=api.run_job(image='pytorch/pytorch:2.10.0-cuda12.8-cudnn9-runtime',command=command,env={'TRAIN_CONFIG':json.dumps(config),'RUN_ID':run,'PYTHONUNBUFFERED':'1'},secrets={'HF_TOKEN':os.environ['HF_TOKEN']},flavor=FLAVOR,timeout=TIMEOUT,namespace='benchflow',name=run,labels={'experiment':'posttrain-hf-gpu','run_id':run})
59
  record.update(job_id=job.id,job_url=f'https://huggingface.co/jobs/benchflow/{job.id}',status=str(job.status.stage))
hf_gpu_train.py CHANGED
@@ -4,7 +4,7 @@ from collections.abc import Mapping
4
  from pathlib import Path
5
  from huggingface_hub import HfApi,hf_hub_download
6
 
7
- def tokenize_rows(rows, tokenizer, max_length=16384):
8
  output=[]; skipped=0
9
  for original in rows:
10
  row=json.loads(json.dumps(original))
@@ -30,8 +30,8 @@ def tokenize_rows(rows, tokenizer, max_length=16384):
30
 
31
  def main():
32
  import torch
33
- from transformers import AutoTokenizer,AutoModelForImageTextToText,Trainer,TrainingArguments,DataCollatorForSeq2Seq
34
- from peft import LoraConfig,get_peft_model
35
  from datasets import Dataset
36
  cfg=json.loads(os.environ['TRAIN_CONFIG']); api=HfApi(); run=os.environ['RUN_ID']
37
  out=Path('/tmp/adapter'); out.mkdir()
@@ -42,7 +42,8 @@ def main():
42
  examples,skipped=tokenize_rows(rows,tokenizer)
43
  if not examples: raise RuntimeError('No valid full-context training examples')
44
  print(json.dumps({'stage':'data-ready','examples':len(examples),'skipped_overlength_or_template':skipped,'gpu':torch.cuda.get_device_name(0)}),flush=True)
45
- model=AutoModelForImageTextToText.from_pretrained(cfg['model'],revision=cfg['model_revision'],dtype=torch.bfloat16,attn_implementation='sdpa',device_map={'':'cuda:0'})
 
46
  model.config.use_cache=False
47
  model=get_peft_model(model,LoraConfig(r=cfg['rank'],lora_alpha=cfg['rank']*2,lora_dropout=0.05,bias='none',task_type='CAUSAL_LM',target_modules=['q_proj','k_proj','v_proj','o_proj']))
48
  model.print_trainable_parameters()
 
4
  from pathlib import Path
5
  from huggingface_hub import HfApi,hf_hub_download
6
 
7
+ def tokenize_rows(rows, tokenizer, max_length=8192):
8
  output=[]; skipped=0
9
  for original in rows:
10
  row=json.loads(json.dumps(original))
 
30
 
31
  def main():
32
  import torch
33
+ from transformers import AutoTokenizer,AutoModelForImageTextToText,Trainer,TrainingArguments,DataCollatorForSeq2Seq,BitsAndBytesConfig
34
+ from peft import LoraConfig,get_peft_model,prepare_model_for_kbit_training
35
  from datasets import Dataset
36
  cfg=json.loads(os.environ['TRAIN_CONFIG']); api=HfApi(); run=os.environ['RUN_ID']
37
  out=Path('/tmp/adapter'); out.mkdir()
 
42
  examples,skipped=tokenize_rows(rows,tokenizer)
43
  if not examples: raise RuntimeError('No valid full-context training examples')
44
  print(json.dumps({'stage':'data-ready','examples':len(examples),'skipped_overlength_or_template':skipped,'gpu':torch.cuda.get_device_name(0)}),flush=True)
45
+ model=AutoModelForImageTextToText.from_pretrained(cfg['model'],revision=cfg['model_revision'],dtype=torch.bfloat16,attn_implementation='sdpa',device_map={'':'cuda:0'},quantization_config=BitsAndBytesConfig(load_in_4bit=True,bnb_4bit_quant_type='nf4',bnb_4bit_compute_dtype=torch.bfloat16,bnb_4bit_use_double_quant=True))
46
+ model=prepare_model_for_kbit_training(model,use_gradient_checkpointing=True)
47
  model.config.use_cache=False
48
  model=get_peft_model(model,LoraConfig(r=cfg['rank'],lora_alpha=cfg['rank']*2,lora_dropout=0.05,bias='none',task_type='CAUSAL_LM',target_modules=['q_proj','k_proj','v_proj','o_proj']))
49
  model.print_trainable_parameters()
index.html CHANGED
@@ -18,13 +18,13 @@
18
  <div class="layout">
19
  <section class="panel" aria-labelledby="setup-heading"><div class="section-title"><h2 id="setup-heading">Your experiment</h2><span class="section-label">Configure</span></div>
20
  <div class="field-row"><span class="step">01</span><div><label class="field-title" for="model">Start with a model</label><select id="model"><option value="Qwen/Qwen3.6-27B">Qwen3.6 · 27B</option></select><div class="model-info"><p class="help" id="model-help">LoRA fine-tuning · Hugging Face GPU Jobs</p><a class="help" id="model-link" href="https://huggingface.co/Qwen/Qwen3.6-27B" target="_blank" rel="noopener">Model card ↗</a></div></div></div>
21
- <div class="field-row"><span class="step">02</span><div><div class="field-title">Training dataset</div><div class="env"><div><strong>SkillsBench / Verified teacher traces</strong><p>10 verified trajectories · 394 teacher exchanges</p></div><span class="tag">SFT corpus</span></div><p class="help">Available HF subset of Carrie’s corpus. Full-context examples up to 16,384 tokens are used. This is a pipeline test; no held-out performance claim.</p></div></div>
22
- <div class="field-row"><span class="step">03</span><div><div class="field-title">Shape the training recipe</div><div class="method">Supervised fine-tuning <span>LoRA · GRPO off</span></div><div class="parameters"><div><label for="steps">Optimizer steps</label><select id="steps"><option selected>20</option><option>50</option><option>100</option></select></div><div><label for="rate">Learning rate</label><select id="rate"><option value="0.00002">2 × 10⁻⁵</option><option value="0.00005">5 × 10⁻⁵</option><option selected value="0.0001">1 × 10⁻⁴</option></select></div><div><label for="rank">LoRA rank</label><select id="rank"><option>8</option><option>16</option><option selected>32</option></select></div></div><p class="help">20 optimizer steps for the first run. A two-hour HF timeout bounds compute.</p></div></div>
23
  <div class="recipe-footer"><span class="help">Existing corpus. Traceable training receipt.</span><button class="quiet" id="recipe-download">Download recipe ↓</button></div><details id="recipe-details"><summary>Inspect recipe</summary><pre id="recipe">Loading recipe…</pre></details>
24
  </section>
25
  <aside>
26
- <section class="panel summary"><div class="summary-head"><span class="eyebrow">Run preview</span><span class="tag">Real SFT</span></div><h2>Train the model.<br>Keep the evidence.</h2><dl><div><dt>Model</dt><dd id="chosen-model">Qwen3.6 · 27B</dd></div><div><dt>Recipe</dt><dd id="chosen-recipe">20 steps · rank 32</dd></div><div><dt>Compute</dt><dd>Hugging Face H200 GPU</dd></div><div><dt>You’ll receive</dt><dd>Trained adapter & job receipt</dd></div></dl><button class="primary" id="launch" disabled>Checking training access…</button><p class="footnote">Starts paid model training. Does not deploy a serving endpoint or run benchmark evaluation.</p><div id="launch-status" class="status" role="status" aria-live="polite"></div></section>
27
- <section class="training-panel" aria-labelledby="training-heading"><div class="training-title"><h2 id="training-heading">Connection & budget</h2><span class="tag" id="training-badge">Checking</span></div><details open><summary>Training readiness</summary><ul id="training-blockers" class="help"><li>Checking…</li></ul></details><p id="training-status" role="status" class="status"></p><div class="budget"><span>Total experiment cap</span><strong>$200</strong></div><p class="help">First run: H200, about $5/hour, two-hour timeout (about $10 maximum compute). Remaining budget stays unallocated. Repeat submissions are blocked, including after a restart. Reservation is not a billing charge.</p></section>
28
  </aside></div>
29
  <section aria-labelledby="runs-heading"><div class="runs-header"><div><div class="eyebrow">Your experiment log</div><h2 id="runs-heading">Run history</h2></div><button class="quiet" id="refresh">Refresh ↻</button></div><div id="run-status" role="status" aria-live="polite"></div><div id="runs"><div class="empty">Loading runs…</div></div></section>
30
  <footer><span>PostTrain Arena / BenchFlow</span><span class="metric"><strong>Δ = score after − score before.</strong> Measured only after training and evaluation.</span></footer>
 
18
  <div class="layout">
19
  <section class="panel" aria-labelledby="setup-heading"><div class="section-title"><h2 id="setup-heading">Your experiment</h2><span class="section-label">Configure</span></div>
20
  <div class="field-row"><span class="step">01</span><div><label class="field-title" for="model">Start with a model</label><select id="model"><option value="Qwen/Qwen3.6-27B">Qwen3.6 · 27B</option></select><div class="model-info"><p class="help" id="model-help">LoRA fine-tuning · Hugging Face GPU Jobs</p><a class="help" id="model-link" href="https://huggingface.co/Qwen/Qwen3.6-27B" target="_blank" rel="noopener">Model card ↗</a></div></div></div>
21
+ <div class="field-row"><span class="step">02</span><div><div class="field-title">Training dataset</div><div class="env"><div><strong>SkillsBench / Verified teacher traces</strong><p>10 verified trajectories · 394 teacher exchanges</p></div><span class="tag">SFT corpus</span></div><p class="help">Available HF subset of Carrie’s corpus. Full-context examples up to 8,192 tokens are used. This is a pipeline test; no held-out performance claim.</p></div></div>
22
+ <div class="field-row"><span class="step">03</span><div><div class="field-title">Shape the training recipe</div><div class="method">Supervised fine-tuning <span>4-bit LoRA · GRPO off</span></div><div class="parameters"><div><label for="steps">Optimizer steps</label><select id="steps"><option selected>20</option><option>50</option><option>100</option></select></div><div><label for="rate">Learning rate</label><select id="rate"><option value="0.00002">2 × 10⁻⁵</option><option value="0.00005">5 × 10⁻⁵</option><option selected value="0.0001">1 × 10⁻⁴</option></select></div><div><label for="rank">LoRA rank</label><select id="rank"><option>8</option><option>16</option><option selected>32</option></select></div></div><p class="help">20 optimizer steps for the first run. A two-hour HF timeout bounds compute.</p></div></div>
23
  <div class="recipe-footer"><span class="help">Existing corpus. Traceable training receipt.</span><button class="quiet" id="recipe-download">Download recipe ↓</button></div><details id="recipe-details"><summary>Inspect recipe</summary><pre id="recipe">Loading recipe…</pre></details>
24
  </section>
25
  <aside>
26
+ <section class="panel summary"><div class="summary-head"><span class="eyebrow">Run preview</span><span class="tag">Real SFT</span></div><h2>Train the model.<br>Keep the evidence.</h2><dl><div><dt>Model</dt><dd id="chosen-model">Qwen3.6 · 27B</dd></div><div><dt>Recipe</dt><dd id="chosen-recipe">20 steps · rank 32</dd></div><div><dt>Compute</dt><dd>Hugging Face A100 80GB GPU</dd></div><div><dt>You’ll receive</dt><dd>Trained adapter & job receipt</dd></div></dl><button class="primary" id="launch" disabled>Checking training access…</button><p class="footnote">Starts paid model training. Does not deploy a serving endpoint or run benchmark evaluation.</p><div id="launch-status" class="status" role="status" aria-live="polite"></div></section>
27
+ <section class="training-panel" aria-labelledby="training-heading"><div class="training-title"><h2 id="training-heading">Connection & budget</h2><span class="tag" id="training-badge">Checking</span></div><details open><summary>Training readiness</summary><ul id="training-blockers" class="help"><li>Checking…</li></ul></details><p id="training-status" role="status" class="status"></p><div class="budget"><span>Total experiment cap</span><strong>$200</strong></div><p class="help">First run: A100 80GB, about $2.50/hour, two-hour timeout (about $5 maximum compute). Remaining budget stays unallocated. Repeat submissions are blocked, including after a restart. Reservation is not a billing charge.</p></section>
28
  </aside></div>
29
  <section aria-labelledby="runs-heading"><div class="runs-header"><div><div class="eyebrow">Your experiment log</div><h2 id="runs-heading">Run history</h2></div><button class="quiet" id="refresh">Refresh ↻</button></div><div id="run-status" role="status" aria-live="polite"></div><div id="runs"><div class="empty">Loading runs…</div></div></section>
30
  <footer><span>PostTrain Arena / BenchFlow</span><span class="metric"><strong>Δ = score after − score before.</strong> Measured only after training and evaluation.</span></footer>