Download initial_training_config.json from AustinFu/Qev-2B: direct link, hf CLI and curl.
- Browser
- Download file 1.35 kB
-
https://huggingface.co/AustinFu/Qev-2B/resolve/main/initial_training_config.json
- Command line
-
hf download hf://AustinFu/Qev-2B/initial_training_config.json
-
curl -L -o initial_training_config.json https://huggingface.co/AustinFu/Qev-2B/resolve/main/initial_training_config.json
1.35 kB
| { | |
| "model": { | |
| "base": "Qwen/Qwen3.5-2B-Base", | |
| "revision": "b1485b2fa6dfa1287294f269f5fb618e03d52d7c", | |
| "head_dim": 256, | |
| "head_heads": 4, | |
| "head_layers": 2, | |
| "lora_rank": 64, | |
| "lora_dropout": 0.0, | |
| "weights_dtype": "bf16", | |
| "attention": "sdpa", | |
| "rows_per_forward": 32, | |
| "layout": "state-question-candidate.v1", | |
| "max_padding_ratio": 1.25, | |
| "max_padded_tokens": 8192, | |
| "choice_none_policy": "as-provided", | |
| "candidate_interaction": "last-full-attention" | |
| }, | |
| "limits": { | |
| "max_state": 1024, | |
| "max_question": 512, | |
| "max_candidate": 256, | |
| "max_path": 2048, | |
| "max_candidates": 128 | |
| }, | |
| "training": { | |
| "epochs": 2, | |
| "batch_size": 8, | |
| "accum": 2, | |
| "seed": 17, | |
| "lr": 2e-05, | |
| "head_lr": 0.0001, | |
| "weight_decay": 0.01, | |
| "warmup_steps": 20, | |
| "head_warmup_steps": 100, | |
| "gradient_checkpointing": true, | |
| "save_every": 100, | |
| "autocast": "bf16", | |
| "batch_assignment": "leaf-balanced-v1", | |
| "prefix_execution": "tree-batched", | |
| "joint_gate_lr": 0.01, | |
| "none_insert_prob": 0.2, | |
| "none_insert_absent_frac": 0.5, | |
| "none_insert_exempt_sources": [ | |
| "jev_distill/" | |
| ], | |
| "late_split": "late_train", | |
| "late_fraction": 0.5, | |
| "late_repeats": 3, | |
| "distillation": { | |
| "cache": "data/teacher-logits", | |
| "weight": 1.0 | |
| } | |
| } | |
| } | |