File size: 3,891 Bytes
2f6b87a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
{
  "schema_version": 1,
  "study": "lr_capacity_search",
  "method": "lora",
  "protocol": "frame_block_cv_search",
  "model_seed": 42,
  "split_seed": 42,
  "init_from_sha256": "3094f103c17bd13558f960c22b91ed3316679cadeab88ba269d862019c4dd58a",
  "fold_subset": [
    0,
    1,
    2,
    3,
    4,
    5,
    6,
    7,
    8,
    9
  ],
  "smoke": false,
  "search_space": {
    "lr_candidates": [
      0.0001,
      0.0003,
      0.001
    ],
    "capacity_axis": "rank",
    "reference_capacity": {
      "rank": 4
    },
    "capacity_candidates": [
      4,
      8,
      16,
      32
    ]
  },
  "lr_selection": {
    "axis": "lr",
    "candidates": [
      0.0001,
      0.0003,
      0.001
    ],
    "held_at": {
      "rank": 4
    },
    "fold_subset": [
      0,
      1,
      2,
      3,
      4,
      5,
      6,
      7,
      8,
      9
    ],
    "mean_accuracy_by_candidate": [
      {
        "lr": 0.0001,
        "config_id": "lora__lr1.0e-04__rank0004",
        "mean_accuracy": 0.9563636363636363
      },
      {
        "lr": 0.0003,
        "config_id": "lora__lr3.0e-04__rank0004",
        "mean_accuracy": 0.9781818181818182
      },
      {
        "lr": 0.001,
        "config_id": "lora__lr1.0e-03__rank0004",
        "mean_accuracy": 0.99
      }
    ],
    "highest_mean_accuracy": 0.99,
    "tied_highest": [
      0.001
    ],
    "tie_rule": "highest ten-fold mean held-out accuracy, then the numerically lowest learning rate on an exact tie. Accuracy alone: no UAR, no weighted F1, no loss and no tolerance band enters it",
    "selected": 0.001
  },
  "capacity_selection": {
    "axis": "rank",
    "candidates": [
      4,
      8,
      16,
      32
    ],
    "at_lr": 0.001,
    "fold_subset": [
      0,
      1,
      2,
      3,
      4,
      5,
      6,
      7,
      8,
      9
    ],
    "mean_accuracy_by_candidate": [
      {
        "rank": 4,
        "config_id": "lora__lr1.0e-03__rank0004",
        "mean_accuracy": 0.99,
        "trainable_params": 152070
      },
      {
        "rank": 8,
        "config_id": "lora__lr1.0e-03__rank0008",
        "mean_accuracy": 0.9881818181818183,
        "trainable_params": 299526
      },
      {
        "rank": 16,
        "config_id": "lora__lr1.0e-03__rank0016",
        "mean_accuracy": 0.9818181818181818,
        "trainable_params": 594438
      },
      {
        "rank": 32,
        "config_id": "lora__lr1.0e-03__rank0032",
        "mean_accuracy": 0.9800000000000001,
        "trainable_params": 1184262
      }
    ],
    "highest_mean_accuracy": 0.99,
    "tied_highest": [
      4
    ],
    "tie_rule": "highest ten-fold mean held-out accuracy, then fewer trainable parameters, then the lexicographically smallest configuration id. Accuracy alone decides first: no UAR, no weighted F1, no loss and no tolerance band enters it",
    "selected": 4
  },
  "selected": {
    "config_id": "lora__lr1.0e-03__rank0004",
    "values": {
      "rank": 4,
      "lr": 0.001
    },
    "trainable_params": 152070,
    "mean_accuracy": 0.99,
    "cell_dir": "outputs/search__lora__src-ferplus__seed42/cells/lora__lr1.0e-03__rank0004"
  },
  "optimism": "the configuration is selected on the same ten held-out folds whose mean accuracy is then reported, so the comparison is optimistically biased by hyperparameter selection on top of the best-epoch-on-the-held-out-block optimism every cell already carries. It is not nested cross-validation, not independent validation, not an unbiased estimate and not a like-for-like comparison with the published figures",
  "design": "a sequential hyperparameter study: learning-rate selection, then capacity selection at the selected rate. The best-observed configuration of each strategy is the configuration that enters the five-method comparison",
  "comparison_eligible": true,
  "comparison_eligibility": "selected over all ten folds"
}