devoppro commited on
Commit
0b599e8
·
verified ·
1 Parent(s): 7975ad8

Training in progress, step 1700, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e5057c9842c32a69d89842ae1ea0f94292f62299087952cbb292a8939dab162b
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e62f7ec49e337899fe60ae17d255db8f70179b4e39b6c2c63d4369c4c9c02991
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:40095725fc0786a3c912798bc42b665ef77af7017cc5a4773c5369b9374cfaaf
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e7a9212260f8bf2a398b78af8f740ee3206cd998eb80d6106578b3826d7a7f64
3
  size 2471218763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:098b29492211804ab324a36f37466821d948280bb74fce4ba895c03f13ecd878
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f4a9f217e852f439efa6bd32fde98d6867f11aa6ea13ddc021ba10af6a0b0934
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:684e33945d5e9615d67982f907b4bd5b55d412c2b48b33613fdd4ef1c4d053c4
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:730fa31435cb48457d9bff27c4dd280598a01503fe2e70600467d96a61b92ee0
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:12b3905bb99561ce66746740b1de5588d55bd3f6f12bc854dc34c827aebf5528
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d4e47d1d4f05a797c13ba0c0cf4a8d6bdae4ef929970ec2744f85af7b8f98bab
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -4,7 +4,7 @@
4
  "best_model_checkpoint": null,
5
  "epoch": 1.002,
6
  "eval_steps": 500,
7
- "global_step": 1600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1128,6 +1128,76 @@
1128
  "learning_rate": 0.00029098797595190375,
1129
  "loss": 3103321489408.0,
1130
  "step": 1600
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1131
  }
1132
  ],
1133
  "logging_steps": 10,
@@ -1147,7 +1217,7 @@
1147
  "attributes": {}
1148
  }
1149
  },
1150
- "total_flos": 2.66012131885056e+16,
1151
  "train_batch_size": 2,
1152
  "trial_name": null,
1153
  "trial_params": null
 
4
  "best_model_checkpoint": null,
5
  "epoch": 1.002,
6
  "eval_steps": 500,
7
+ "global_step": 1700,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1128
  "learning_rate": 0.00029098797595190375,
1129
  "loss": 3103321489408.0,
1130
  "step": 1600
1131
+ },
1132
+ {
1133
+ "epoch": 1.0002,
1134
+ "grad_norm": 0.0,
1135
+ "learning_rate": 0.0002909278557114228,
1136
+ "loss": 95.5510498046875,
1137
+ "step": 1610
1138
+ },
1139
+ {
1140
+ "epoch": 1.0004,
1141
+ "grad_norm": 0.0,
1142
+ "learning_rate": 0.00029086773547094184,
1143
+ "loss": 98.69436645507812,
1144
+ "step": 1620
1145
+ },
1146
+ {
1147
+ "epoch": 1.0006,
1148
+ "grad_norm": 0.0,
1149
+ "learning_rate": 0.0002908076152304609,
1150
+ "loss": 99.0325439453125,
1151
+ "step": 1630
1152
+ },
1153
+ {
1154
+ "epoch": 1.0008,
1155
+ "grad_norm": 0.0,
1156
+ "learning_rate": 0.00029074749498997994,
1157
+ "loss": 97.31904296875,
1158
+ "step": 1640
1159
+ },
1160
+ {
1161
+ "epoch": 1.001,
1162
+ "grad_norm": 0.0,
1163
+ "learning_rate": 0.000290687374749499,
1164
+ "loss": 98.40114135742188,
1165
+ "step": 1650
1166
+ },
1167
+ {
1168
+ "epoch": 1.0012,
1169
+ "grad_norm": 0.0,
1170
+ "learning_rate": 0.00029062725450901803,
1171
+ "loss": 99.31021728515626,
1172
+ "step": 1660
1173
+ },
1174
+ {
1175
+ "epoch": 1.0014,
1176
+ "grad_norm": 0.0,
1177
+ "learning_rate": 0.00029056713426853703,
1178
+ "loss": 99.4878662109375,
1179
+ "step": 1670
1180
+ },
1181
+ {
1182
+ "epoch": 1.0016,
1183
+ "grad_norm": 0.0,
1184
+ "learning_rate": 0.0002905070140280561,
1185
+ "loss": 99.3884521484375,
1186
+ "step": 1680
1187
+ },
1188
+ {
1189
+ "epoch": 1.0018,
1190
+ "grad_norm": 0.0,
1191
+ "learning_rate": 0.0002904468937875751,
1192
+ "loss": 99.46731567382812,
1193
+ "step": 1690
1194
+ },
1195
+ {
1196
+ "epoch": 1.002,
1197
+ "grad_norm": 0.0,
1198
+ "learning_rate": 0.00029038677354709417,
1199
+ "loss": 99.4464599609375,
1200
+ "step": 1700
1201
  }
1202
  ],
1203
  "logging_steps": 10,
 
1217
  "attributes": {}
1218
  }
1219
  },
1220
+ "total_flos": 2.666982292581888e+16,
1221
  "train_batch_size": 2,
1222
  "trial_name": null,
1223
  "trial_params": null