xingxm commited on
Commit
a7ae526
·
verified ·
1 Parent(s): 3ea479e

Add designcoder_evaluator_qwen3.8_27b_adamw_bs256_data37847_step296/trainer_state.json

Browse files
designcoder_evaluator_qwen3.8_27b_adamw_bs256_data37847_step296/trainer_state.json ADDED
@@ -0,0 +1,246 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 2.0,
6
+ "eval_steps": 500,
7
+ "global_step": 296,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.06756756756756757,
14
+ "grad_norm": 1.878107082674217,
15
+ "learning_rate": 1.5e-06,
16
+ "loss": 1.0629616737365724,
17
+ "step": 10
18
+ },
19
+ {
20
+ "epoch": 0.13513513513513514,
21
+ "grad_norm": 1.0194370889139108,
22
+ "learning_rate": 3.1666666666666667e-06,
23
+ "loss": 0.8560467720031738,
24
+ "step": 20
25
+ },
26
+ {
27
+ "epoch": 0.20270270270270271,
28
+ "grad_norm": 0.6544153776775151,
29
+ "learning_rate": 4.833333333333333e-06,
30
+ "loss": 0.6546797752380371,
31
+ "step": 30
32
+ },
33
+ {
34
+ "epoch": 0.2702702702702703,
35
+ "grad_norm": 0.49647934999629595,
36
+ "learning_rate": 4.9858901447482924e-06,
37
+ "loss": 0.5755151748657227,
38
+ "step": 40
39
+ },
40
+ {
41
+ "epoch": 0.33783783783783783,
42
+ "grad_norm": 0.4304240879222198,
43
+ "learning_rate": 4.937319780454559e-06,
44
+ "loss": 0.5334901809692383,
45
+ "step": 50
46
+ },
47
+ {
48
+ "epoch": 0.40540540540540543,
49
+ "grad_norm": 0.42756601260151317,
50
+ "learning_rate": 4.854791259851735e-06,
51
+ "loss": 0.508150577545166,
52
+ "step": 60
53
+ },
54
+ {
55
+ "epoch": 0.47297297297297297,
56
+ "grad_norm": 0.46056131786658294,
57
+ "learning_rate": 4.739454418273314e-06,
58
+ "loss": 0.49311065673828125,
59
+ "step": 70
60
+ },
61
+ {
62
+ "epoch": 0.5405405405405406,
63
+ "grad_norm": 0.746908409586495,
64
+ "learning_rate": 4.592916195656322e-06,
65
+ "loss": 0.4820195198059082,
66
+ "step": 80
67
+ },
68
+ {
69
+ "epoch": 0.6081081081081081,
70
+ "grad_norm": 0.5518982067901123,
71
+ "learning_rate": 4.417218247719794e-06,
72
+ "loss": 0.47684478759765625,
73
+ "step": 90
74
+ },
75
+ {
76
+ "epoch": 0.6756756756756757,
77
+ "grad_norm": 0.4754958774717991,
78
+ "learning_rate": 4.2148085004302205e-06,
79
+ "loss": 0.4692880153656006,
80
+ "step": 100
81
+ },
82
+ {
83
+ "epoch": 0.7432432432432432,
84
+ "grad_norm": 0.39636201567199464,
85
+ "learning_rate": 3.988507044073687e-06,
86
+ "loss": 0.463223934173584,
87
+ "step": 110
88
+ },
89
+ {
90
+ "epoch": 0.8108108108108109,
91
+ "grad_norm": 0.5377841789491057,
92
+ "learning_rate": 3.741466842118327e-06,
93
+ "loss": 0.4611557960510254,
94
+ "step": 120
95
+ },
96
+ {
97
+ "epoch": 0.8783783783783784,
98
+ "grad_norm": 0.5463097603447346,
99
+ "learning_rate": 3.477129802294057e-06,
100
+ "loss": 0.45296297073364256,
101
+ "step": 130
102
+ },
103
+ {
104
+ "epoch": 0.9459459459459459,
105
+ "grad_norm": 0.4457573017627412,
106
+ "learning_rate": 3.1991788219328657e-06,
107
+ "loss": 0.4486509323120117,
108
+ "step": 140
109
+ },
110
+ {
111
+ "epoch": 1.0135135135135136,
112
+ "grad_norm": 0.44983735301350025,
113
+ "learning_rate": 2.911486475701835e-06,
114
+ "loss": 0.4460547924041748,
115
+ "step": 150
116
+ },
117
+ {
118
+ "epoch": 1.0810810810810811,
119
+ "grad_norm": 0.4679423900616534,
120
+ "learning_rate": 2.6180610606412587e-06,
121
+ "loss": 0.43609347343444826,
122
+ "step": 160
123
+ },
124
+ {
125
+ "epoch": 1.1486486486486487,
126
+ "grad_norm": 0.46004275728094907,
127
+ "learning_rate": 2.322990750239733e-06,
128
+ "loss": 0.43337116241455076,
129
+ "step": 170
130
+ },
131
+ {
132
+ "epoch": 1.2162162162162162,
133
+ "grad_norm": 0.4549358577066241,
134
+ "learning_rate": 2.030386635624135e-06,
135
+ "loss": 0.4316365718841553,
136
+ "step": 180
137
+ },
138
+ {
139
+ "epoch": 1.2837837837837838,
140
+ "grad_norm": 0.596917314782809,
141
+ "learning_rate": 1.7443254474477328e-06,
142
+ "loss": 0.4303096294403076,
143
+ "step": 190
144
+ },
145
+ {
146
+ "epoch": 1.3513513513513513,
147
+ "grad_norm": 0.47738926870283904,
148
+ "learning_rate": 1.4687927565084023e-06,
149
+ "loss": 0.4278303623199463,
150
+ "step": 200
151
+ },
152
+ {
153
+ "epoch": 1.4189189189189189,
154
+ "grad_norm": 1.315544608495029,
155
+ "learning_rate": 1.2076274444589361e-06,
156
+ "loss": 0.4261355400085449,
157
+ "step": 210
158
+ },
159
+ {
160
+ "epoch": 1.4864864864864864,
161
+ "grad_norm": 0.45886105321791265,
162
+ "learning_rate": 9.644682182758305e-07,
163
+ "loss": 0.42614049911499025,
164
+ "step": 220
165
+ },
166
+ {
167
+ "epoch": 1.554054054054054,
168
+ "grad_norm": 0.4560051449943854,
169
+ "learning_rate": 7.427029136780333e-07,
170
+ "loss": 0.42502822875976565,
171
+ "step": 230
172
+ },
173
+ {
174
+ "epoch": 1.6216216216216215,
175
+ "grad_norm": 0.4118149029932589,
176
+ "learning_rate": 5.454212938299256e-07,
177
+ "loss": 0.42264533042907715,
178
+ "step": 240
179
+ },
180
+ {
181
+ "epoch": 1.689189189189189,
182
+ "grad_norm": 0.4756231741121155,
183
+ "learning_rate": 3.753720009644371e-07,
184
+ "loss": 0.4201972961425781,
185
+ "step": 250
186
+ },
187
+ {
188
+ "epoch": 1.7567567567567568,
189
+ "grad_norm": 0.39022848583695624,
190
+ "learning_rate": 2.3492426070131746e-07,
191
+ "loss": 0.42458620071411135,
192
+ "step": 260
193
+ },
194
+ {
195
+ "epoch": 1.8243243243243243,
196
+ "grad_norm": 0.9959555152127557,
197
+ "learning_rate": 1.2603487261826726e-07,
198
+ "loss": 0.42471790313720703,
199
+ "step": 270
200
+ },
201
+ {
202
+ "epoch": 1.8918918918918919,
203
+ "grad_norm": 0.4178560571393511,
204
+ "learning_rate": 5.022094698148072e-08,
205
+ "loss": 0.42223272323608396,
206
+ "step": 280
207
+ },
208
+ {
209
+ "epoch": 1.9594594594594594,
210
+ "grad_norm": 0.39226660688001813,
211
+ "learning_rate": 8.538767483325384e-09,
212
+ "loss": 0.4246257781982422,
213
+ "step": 290
214
+ },
215
+ {
216
+ "epoch": 2.0,
217
+ "step": 296,
218
+ "total_flos": 2.3822562326657106e+18,
219
+ "train_loss": 0.49392863383164276,
220
+ "train_runtime": 16700.8149,
221
+ "train_samples_per_second": 4.532,
222
+ "train_steps_per_second": 0.018
223
+ }
224
+ ],
225
+ "logging_steps": 10,
226
+ "max_steps": 296,
227
+ "num_input_tokens_seen": 0,
228
+ "num_train_epochs": 2,
229
+ "save_steps": 100,
230
+ "stateful_callbacks": {
231
+ "TrainerControl": {
232
+ "args": {
233
+ "should_epoch_stop": false,
234
+ "should_evaluate": false,
235
+ "should_log": false,
236
+ "should_save": true,
237
+ "should_training_stop": true
238
+ },
239
+ "attributes": {}
240
+ }
241
+ },
242
+ "total_flos": 2.3822562326657106e+18,
243
+ "train_batch_size": 1,
244
+ "trial_name": null,
245
+ "trial_params": null
246
+ }