devoppro commited on
Commit
838f5aa
·
verified ·
1 Parent(s): d447039

Training in progress, step 1600, checkpoint

Browse files
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d6b1414405c6569082ce4c89ffc35e400efabe527d1fbfaf811626cb9b7be419
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:40095725fc0786a3c912798bc42b665ef77af7017cc5a4773c5369b9374cfaaf
3
  size 2471218763
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:61c19bab1174704a4a4441475683bf1270277af15d2e2c95e964789128e482c4
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:098b29492211804ab324a36f37466821d948280bb74fce4ba895c03f13ecd878
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7ad85e280d5a5bb25ea74c35cf3277ee0dc8ea6fc4405d02f6e2c57f852fa261
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:684e33945d5e9615d67982f907b4bd5b55d412c2b48b33613fdd4ef1c4d053c4
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bf4d4bc05efa244f95b3aea452c294bae8df3a3f92048af90e4124e1d08e77d7
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:12b3905bb99561ce66746740b1de5588d55bd3f6f12bc854dc34c827aebf5528
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.03,
6
  "eval_steps": 500,
7
- "global_step": 1500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1058,6 +1058,76 @@
1058
  "learning_rate": 0.0002915891783567134,
1059
  "loss": 77.63518676757812,
1060
  "step": 1500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1061
  }
1062
  ],
1063
  "logging_steps": 10,
@@ -1077,7 +1147,7 @@
1077
  "attributes": {}
1078
  }
1079
  },
1080
- "total_flos": 2.503701404508672e+16,
1081
  "train_batch_size": 2,
1082
  "trial_name": null,
1083
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 1.002,
6
  "eval_steps": 500,
7
+ "global_step": 1600,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1058
  "learning_rate": 0.0002915891783567134,
1059
  "loss": 77.63518676757812,
1060
  "step": 1500
1061
+ },
1062
+ {
1063
+ "epoch": 1.0002,
1064
+ "grad_norm": NaN,
1065
+ "learning_rate": 0.00029152905811623243,
1066
+ "loss": 107136583467008.0,
1067
+ "step": 1510
1068
+ },
1069
+ {
1070
+ "epoch": 1.0004,
1071
+ "grad_norm": NaN,
1072
+ "learning_rate": 0.0002914689378757515,
1073
+ "loss": 3583383070310.4,
1074
+ "step": 1520
1075
+ },
1076
+ {
1077
+ "epoch": 1.0006,
1078
+ "grad_norm": NaN,
1079
+ "learning_rate": 0.0002914088176352705,
1080
+ "loss": 36528589478297.6,
1081
+ "step": 1530
1082
+ },
1083
+ {
1084
+ "epoch": 1.0008,
1085
+ "grad_norm": NaN,
1086
+ "learning_rate": 0.00029134869739478957,
1087
+ "loss": 1132657744281.6,
1088
+ "step": 1540
1089
+ },
1090
+ {
1091
+ "epoch": 1.001,
1092
+ "grad_norm": NaN,
1093
+ "learning_rate": 0.0002912885771543086,
1094
+ "loss": 9138627379.2,
1095
+ "step": 1550
1096
+ },
1097
+ {
1098
+ "epoch": 1.0012,
1099
+ "grad_norm": NaN,
1100
+ "learning_rate": 0.0002912284569138276,
1101
+ "loss": 1681122891975884.8,
1102
+ "step": 1560
1103
+ },
1104
+ {
1105
+ "epoch": 1.0014,
1106
+ "grad_norm": NaN,
1107
+ "learning_rate": 0.00029116833667334666,
1108
+ "loss": 929300570596966.4,
1109
+ "step": 1570
1110
+ },
1111
+ {
1112
+ "epoch": 1.0016,
1113
+ "grad_norm": NaN,
1114
+ "learning_rate": 0.0002911082164328657,
1115
+ "loss": 9330961573478.4,
1116
+ "step": 1580
1117
+ },
1118
+ {
1119
+ "epoch": 1.0018,
1120
+ "grad_norm": NaN,
1121
+ "learning_rate": 0.00029104809619238476,
1122
+ "loss": 63892658271027.2,
1123
+ "step": 1590
1124
+ },
1125
+ {
1126
+ "epoch": 1.002,
1127
+ "grad_norm": NaN,
1128
+ "learning_rate": 0.00029098797595190375,
1129
+ "loss": 3103321489408.0,
1130
+ "step": 1600
1131
  }
1132
  ],
1133
  "logging_steps": 10,
 
1147
  "attributes": {}
1148
  }
1149
  },
1150
+ "total_flos": 2.66012131885056e+16,
1151
  "train_batch_size": 2,
1152
  "trial_name": null,
1153
  "trial_params": null