File size: 1,350 Bytes
00da497
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
{
  "model": {
    "base": "Qwen/Qwen3.5-2B-Base",
    "revision": "b1485b2fa6dfa1287294f269f5fb618e03d52d7c",
    "head_dim": 256,
    "head_heads": 4,
    "head_layers": 2,
    "lora_rank": 64,
    "lora_dropout": 0.0,
    "weights_dtype": "bf16",
    "attention": "sdpa",
    "rows_per_forward": 32,
    "layout": "state-question-candidate.v1",
    "max_padding_ratio": 1.25,
    "max_padded_tokens": 8192,
    "choice_none_policy": "as-provided",
    "candidate_interaction": "last-full-attention"
  },
  "limits": {
    "max_state": 1024,
    "max_question": 512,
    "max_candidate": 256,
    "max_path": 2048,
    "max_candidates": 128
  },
  "training": {
    "epochs": 2,
    "batch_size": 8,
    "accum": 2,
    "seed": 17,
    "lr": 2e-05,
    "head_lr": 0.0001,
    "weight_decay": 0.01,
    "warmup_steps": 20,
    "head_warmup_steps": 100,
    "gradient_checkpointing": true,
    "save_every": 100,
    "autocast": "bf16",
    "batch_assignment": "leaf-balanced-v1",
    "prefix_execution": "tree-batched",
    "joint_gate_lr": 0.01,
    "none_insert_prob": 0.2,
    "none_insert_absent_frac": 0.5,
    "none_insert_exempt_sources": [
      "jev_distill/"
    ],
    "late_split": "late_train",
    "late_fraction": 0.5,
    "late_repeats": 3,
    "distillation": {
      "cache": "data/teacher-logits",
      "weight": 1.0
    }
  }
}