Qev-2B / initial_training_config.json
AustinFu's picture
Release Qev-2B decision model and distillation recipe
00da497 verified
Raw History Blame Contribute Delete
1.35 kB
{
"model": {
"base": "Qwen/Qwen3.5-2B-Base",
"revision": "b1485b2fa6dfa1287294f269f5fb618e03d52d7c",
"head_dim": 256,
"head_heads": 4,
"head_layers": 2,
"lora_rank": 64,
"lora_dropout": 0.0,
"weights_dtype": "bf16",
"attention": "sdpa",
"rows_per_forward": 32,
"layout": "state-question-candidate.v1",
"max_padding_ratio": 1.25,
"max_padded_tokens": 8192,
"choice_none_policy": "as-provided",
"candidate_interaction": "last-full-attention"
},
"limits": {
"max_state": 1024,
"max_question": 512,
"max_candidate": 256,
"max_path": 2048,
"max_candidates": 128
},
"training": {
"epochs": 2,
"batch_size": 8,
"accum": 2,
"seed": 17,
"lr": 2e-05,
"head_lr": 0.0001,
"weight_decay": 0.01,
"warmup_steps": 20,
"head_warmup_steps": 100,
"gradient_checkpointing": true,
"save_every": 100,
"autocast": "bf16",
"batch_assignment": "leaf-balanced-v1",
"prefix_execution": "tree-batched",
"joint_gate_lr": 0.01,
"none_insert_prob": 0.2,
"none_insert_absent_frac": 0.5,
"none_insert_exempt_sources": [
"jev_distill/"
],
"late_split": "late_train",
"late_fraction": 0.5,
"late_repeats": 3,
"distillation": {
"cache": "data/teacher-logits",
"weight": 1.0
}
}
}