devoppro commited on
Commit
d708a29
·
verified ·
1 Parent(s): 04fcca1

Training in progress, step 300, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:eff5893f6d1018c76da56d8c4a5822e53399dec8cbf09c3f35c6520dd57fee7a
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ba8406c53f83c09c1ffeabb8ce1224cbfb422c48a8d00576b75192664a252c36
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7ccf1805134bd0c674a65a5ac91541131c5ca34d46b4d69f086cda162433a665
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8a0eee39df69ccbd8f983b1f8a785ab0e689171b4e4bf4b721400c5924aa6fe6
3
  size 2471218763
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0e8bef6dda512071503ea3bab68f0960919f4ba9156b465cc1853aaa448a81f7
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b31b319d83cac2dd433a790a8abad45ac5c816140a98898786a198f16bd883cd
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:03d14372590aaa8a694124e857b900ce7df2b5fee119f58bc35688c4616f5961
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9cf04c7f94b59cd606c42cf0fc828e78ab899584bb7bbb628645ada4c3c807cf
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.004,
6
  "eval_steps": 500,
7
- "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -148,6 +148,76 @@
148
  "learning_rate": 0.0002994048096192384,
149
  "loss": 9.44781723022461,
150
  "step": 200
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
151
  }
152
  ],
153
  "logging_steps": 10,
@@ -167,7 +237,7 @@
167
  "attributes": {}
168
  }
169
  },
170
- "total_flos": 3778874966016000.0,
171
  "train_batch_size": 2,
172
  "trial_name": null,
173
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.006,
6
  "eval_steps": 500,
7
+ "global_step": 300,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
148
  "learning_rate": 0.0002994048096192384,
149
  "loss": 9.44781723022461,
150
  "step": 200
151
+ },
152
+ {
153
+ "epoch": 0.0042,
154
+ "grad_norm": 1.9302258491516113,
155
+ "learning_rate": 0.0002993446893787575,
156
+ "loss": 11.585807800292969,
157
+ "step": 210
158
+ },
159
+ {
160
+ "epoch": 0.0044,
161
+ "grad_norm": 1.4165903329849243,
162
+ "learning_rate": 0.0002992845691382765,
163
+ "loss": 10.411253356933594,
164
+ "step": 220
165
+ },
166
+ {
167
+ "epoch": 0.0046,
168
+ "grad_norm": 2.6108484268188477,
169
+ "learning_rate": 0.00029922444889779557,
170
+ "loss": 13.10027618408203,
171
+ "step": 230
172
+ },
173
+ {
174
+ "epoch": 0.0048,
175
+ "grad_norm": 2.317113161087036,
176
+ "learning_rate": 0.00029916432865731456,
177
+ "loss": 10.390520477294922,
178
+ "step": 240
179
+ },
180
+ {
181
+ "epoch": 0.005,
182
+ "grad_norm": 2.3662290573120117,
183
+ "learning_rate": 0.00029910420841683367,
184
+ "loss": 11.516622924804688,
185
+ "step": 250
186
+ },
187
+ {
188
+ "epoch": 0.0052,
189
+ "grad_norm": 1.1160132884979248,
190
+ "learning_rate": 0.0002990440881763527,
191
+ "loss": 10.796371459960938,
192
+ "step": 260
193
+ },
194
+ {
195
+ "epoch": 0.0054,
196
+ "grad_norm": 2.113326072692871,
197
+ "learning_rate": 0.0002989839679358717,
198
+ "loss": 12.288037109375,
199
+ "step": 270
200
+ },
201
+ {
202
+ "epoch": 0.0056,
203
+ "grad_norm": 1.6802517175674438,
204
+ "learning_rate": 0.00029892384769539075,
205
+ "loss": 9.989302062988282,
206
+ "step": 280
207
+ },
208
+ {
209
+ "epoch": 0.0058,
210
+ "grad_norm": 0.667611300945282,
211
+ "learning_rate": 0.0002988637274549098,
212
+ "loss": 10.866537475585938,
213
+ "step": 290
214
+ },
215
+ {
216
+ "epoch": 0.006,
217
+ "grad_norm": 0.9238641858100891,
218
+ "learning_rate": 0.00029880360721442885,
219
+ "loss": 10.386916351318359,
220
+ "step": 300
221
  }
222
  ],
223
  "logging_steps": 10,
 
237
  "attributes": {}
238
  }
239
  },
240
+ "total_flos": 5668312449024000.0,
241
  "train_batch_size": 2,
242
  "trial_name": null,
243
  "trial_params": null