jebadiah-9b-v2 / eval /train_summary.json
jbrashear's picture
Jebadiah 9B v2: merged bf16 weights, tokenizer, temperatures, scripts, eval records
b1d222a verified
Raw History Blame Contribute Delete
3.18 kB
{
"config": {
"run_name": "9b-chat-v1",
"description": "The v1 recipe on the CHAT checkpoint Qwen/Qwen3.5-9B (thinking off). 4b-chat-v1 scored 72.49 (best 4B ever, +2.2 over 4b-v1) with no synthetic data, so the base checkpoint is the lever. Read against 9b-v1 (73.29): the candidate for the published 9B v2.",
"base_model": "Qwen/Qwen3.5-9B",
"base_revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a",
"data_dir": "/workspace/jeb/data-v1",
"dataset_path": "/workspace/jeb/data-v1/train.jsonl",
"eval_dataset_path": "/workspace/jeb/data-v1/calib.jsonl",
"output_dir": "/workspace/jeb/runs/9b-chat-v1",
"method": "lora",
"num_epochs": 1,
"batch_size": 8,
"gradient_accumulation_steps": 1,
"learning_rate": 0.0001,
"lr_scheduler_type": "cosine",
"lora_rank": 16,
"lora_alpha": 32,
"max_seq_length": 2048,
"warmup_steps": 30,
"weight_decay": 0.0,
"max_grad_norm": 1.0,
"use_gradient_checkpointing": true,
"attn_implementation": "sdpa",
"eval_steps": 200,
"logging_steps": 10,
"seed": 17,
"decide": {
"objective": "candidate_ce",
"shuffle_choice_options": true,
"lora_dropout": 0.05,
"calib_eval_limit": 358,
"target_modules": "all-linear",
"score_targets": "ordinal",
"score_ordinal_adjacent": 0.2
},
"prompt_source_sha256": "d2660ebec28bd3f1704235bda88d24a397c1c62475e740519cb8ef2d08f25fdd",
"prompt_source_commit": "e5c089386e0239c9eb270eeb490d181722b8da5b",
"sweep": {
"letter": "h",
"smoke": false,
"eval_only": false,
"max_steps": -1,
"eval_limit": 0,
"eval_repeats": 5,
"eval_repeats_for": "jevals-=5,nimble-eval=3,kev-transfer=3,typed-decisions=2,kev-decision=1,nimble-public=1",
"eval_batch_size": 16,
"eval_latency_sample": 30,
"temperature_target": "train",
"notes": []
}
},
"target_modules": "all-linear",
"lora_modules": [
"down_proj",
"gate_proj",
"in_proj_a",
"in_proj_b",
"in_proj_qkv",
"in_proj_z",
"k_proj",
"o_proj",
"out_proj",
"q_proj",
"up_proj",
"v_proj"
],
"trainable_params": 43278336,
"prompt_source_commit": "e5c089386e0239c9eb270eeb490d181722b8da5b",
"prompt_source_sha256": "d2660ebec28bd3f1704235bda88d24a397c1c62475e740519cb8ef2d08f25fdd",
"chat_template_sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715",
"single_token_labels": 68,
"chat_template_kwargs": {
"add_generation_prompt": true,
"enable_thinking": false,
"thinking": false
},
"train_examples": 14892,
"calib_examples": 689,
"truncated_prompts": 6,
"steps": 1862,
"wall_clock_s": 5333.4,
"load_s": 50.6,
"peak_memory_allocated_gb": 31.59,
"final_calib": {
"calib_accuracy": 0.8603351955307262,
"calib_nll": 0.4381291315885571,
"calib_accuracy_choice": 0.875,
"calib_nll_choice": 0.4329145725041313,
"calib_accuracy_noul": 0.8490566037735849,
"calib_nll_noul": 0.34073033509579664,
"calib_accuracy_score": 0.8548387096774194,
"calib_nll_score": 0.5267721637743564
},
"hardware": "NVIDIA A100 80GB PCIe",
"torch": "2.11.0+cu128",
"adapter_dir": "/workspace/jeb/runs/9b-chat-v1/adapter",
"fp32_candidate_logits": true,
"score_targets": "ordinal",
"score_ordinal_adjacent": 0.2
}