devoppro commited on
Commit
a5a575e
·
verified ·
1 Parent(s): 0b599e8

Training in progress, step 1900

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e62f7ec49e337899fe60ae17d255db8f70179b4e39b6c2c63d4369c4c9c02991
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e5057c9842c32a69d89842ae1ea0f94292f62299087952cbb292a8939dab162b
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e7a9212260f8bf2a398b78af8f740ee3206cd998eb80d6106578b3826d7a7f64
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:40095725fc0786a3c912798bc42b665ef77af7017cc5a4773c5369b9374cfaaf
3
  size 2471218763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f4a9f217e852f439efa6bd32fde98d6867f11aa6ea13ddc021ba10af6a0b0934
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:098b29492211804ab324a36f37466821d948280bb74fce4ba895c03f13ecd878
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:730fa31435cb48457d9bff27c4dd280598a01503fe2e70600467d96a61b92ee0
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:684e33945d5e9615d67982f907b4bd5b55d412c2b48b33613fdd4ef1c4d053c4
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d4e47d1d4f05a797c13ba0c0cf4a8d6bdae4ef929970ec2744f85af7b8f98bab
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:12b3905bb99561ce66746740b1de5588d55bd3f6f12bc854dc34c827aebf5528
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -4,7 +4,7 @@
4
  "best_model_checkpoint": null,
5
  "epoch": 1.002,
6
  "eval_steps": 500,
7
- "global_step": 1700,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1128,76 +1128,6 @@
1128
  "learning_rate": 0.00029098797595190375,
1129
  "loss": 3103321489408.0,
1130
  "step": 1600
1131
- },
1132
- {
1133
- "epoch": 1.0002,
1134
- "grad_norm": 0.0,
1135
- "learning_rate": 0.0002909278557114228,
1136
- "loss": 95.5510498046875,
1137
- "step": 1610
1138
- },
1139
- {
1140
- "epoch": 1.0004,
1141
- "grad_norm": 0.0,
1142
- "learning_rate": 0.00029086773547094184,
1143
- "loss": 98.69436645507812,
1144
- "step": 1620
1145
- },
1146
- {
1147
- "epoch": 1.0006,
1148
- "grad_norm": 0.0,
1149
- "learning_rate": 0.0002908076152304609,
1150
- "loss": 99.0325439453125,
1151
- "step": 1630
1152
- },
1153
- {
1154
- "epoch": 1.0008,
1155
- "grad_norm": 0.0,
1156
- "learning_rate": 0.00029074749498997994,
1157
- "loss": 97.31904296875,
1158
- "step": 1640
1159
- },
1160
- {
1161
- "epoch": 1.001,
1162
- "grad_norm": 0.0,
1163
- "learning_rate": 0.000290687374749499,
1164
- "loss": 98.40114135742188,
1165
- "step": 1650
1166
- },
1167
- {
1168
- "epoch": 1.0012,
1169
- "grad_norm": 0.0,
1170
- "learning_rate": 0.00029062725450901803,
1171
- "loss": 99.31021728515626,
1172
- "step": 1660
1173
- },
1174
- {
1175
- "epoch": 1.0014,
1176
- "grad_norm": 0.0,
1177
- "learning_rate": 0.00029056713426853703,
1178
- "loss": 99.4878662109375,
1179
- "step": 1670
1180
- },
1181
- {
1182
- "epoch": 1.0016,
1183
- "grad_norm": 0.0,
1184
- "learning_rate": 0.0002905070140280561,
1185
- "loss": 99.3884521484375,
1186
- "step": 1680
1187
- },
1188
- {
1189
- "epoch": 1.0018,
1190
- "grad_norm": 0.0,
1191
- "learning_rate": 0.0002904468937875751,
1192
- "loss": 99.46731567382812,
1193
- "step": 1690
1194
- },
1195
- {
1196
- "epoch": 1.002,
1197
- "grad_norm": 0.0,
1198
- "learning_rate": 0.00029038677354709417,
1199
- "loss": 99.4464599609375,
1200
- "step": 1700
1201
  }
1202
  ],
1203
  "logging_steps": 10,
@@ -1217,7 +1147,7 @@
1217
  "attributes": {}
1218
  }
1219
  },
1220
- "total_flos": 2.666982292581888e+16,
1221
  "train_batch_size": 2,
1222
  "trial_name": null,
1223
  "trial_params": null
 
4
  "best_model_checkpoint": null,
5
  "epoch": 1.002,
6
  "eval_steps": 500,
7
+ "global_step": 1600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1128
  "learning_rate": 0.00029098797595190375,
1129
  "loss": 3103321489408.0,
1130
  "step": 1600
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1131
  }
1132
  ],
1133
  "logging_steps": 10,
 
1147
  "attributes": {}
1148
  }
1149
  },
1150
+ "total_flos": 2.66012131885056e+16,
1151
  "train_batch_size": 2,
1152
  "trial_name": null,
1153
  "trial_params": null
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e62f7ec49e337899fe60ae17d255db8f70179b4e39b6c2c63d4369c4c9c02991
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:67ce32f079c2d9383c97ce4438a5a5face034e7833873ae3afffd266f5462b3e
3
  size 1235573136