{ "run_name": "9b-chat-v1", "description": "The v1 recipe on the CHAT checkpoint Qwen/Qwen3.5-9B (thinking off). 4b-chat-v1 scored 72.49 (best 4B ever, +2.2 over 4b-v1) with no synthetic data, so the base checkpoint is the lever. Read against 9b-v1 (73.29): the candidate for the published 9B v2.", "base_model": "Qwen/Qwen3.5-9B", "base_revision": "c202236235762e1c871ad0ccb60c8ee5ba337b9a", "data_dir": "/workspace/jeb/data-v1", "dataset_path": "/workspace/jeb/data-v1/train.jsonl", "eval_dataset_path": "/workspace/jeb/data-v1/calib.jsonl", "output_dir": "/workspace/jeb/runs/9b-chat-v1", "method": "lora", "num_epochs": 1, "batch_size": 8, "gradient_accumulation_steps": 1, "learning_rate": 0.0001, "lr_scheduler_type": "cosine", "lora_rank": 16, "lora_alpha": 32, "max_seq_length": 2048, "warmup_steps": 30, "weight_decay": 0.0, "max_grad_norm": 1.0, "use_gradient_checkpointing": true, "attn_implementation": "sdpa", "eval_steps": 200, "logging_steps": 10, "seed": 17, "decide": { "objective": "candidate_ce", "shuffle_choice_options": true, "lora_dropout": 0.05, "calib_eval_limit": 358, "target_modules": "all-linear", "score_targets": "ordinal", "score_ordinal_adjacent": 0.2 }, "prompt_source_sha256": "d2660ebec28bd3f1704235bda88d24a397c1c62475e740519cb8ef2d08f25fdd", "prompt_source_commit": "e5c089386e0239c9eb270eeb490d181722b8da5b", "sweep": { "letter": "h", "smoke": false, "eval_only": false, "max_steps": -1, "eval_limit": 0, "eval_repeats": 5, "eval_repeats_for": "jevals-=5,nimble-eval=3,kev-transfer=3,typed-decisions=2,kev-decision=1,nimble-public=1", "eval_batch_size": 16, "eval_latency_sample": 30, "temperature_target": "train", "notes": [] } }