devoppro commited on
Commit
aba239b
·
verified ·
1 Parent(s): ecef9aa

Training in progress, step 200, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a746d1236d3fa2278997b4fe8b4862ddd7e6f6943a953a1e3a3dcb4e53feef47
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eff5893f6d1018c76da56d8c4a5822e53399dec8cbf09c3f35c6520dd57fee7a
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f22e2f31d8b844f3b49fcf0dfcfd7d49a319f0f4f83070bfae39bda9c1efaa59
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7ccf1805134bd0c674a65a5ac91541131c5ca34d46b4d69f086cda162433a665
3
  size 2471218763
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d15f80eff8e7aa3a49c0408f3d31fa2e6e9e75a2fbae063b0cc83bfa545cc3d7
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0e8bef6dda512071503ea3bab68f0960919f4ba9156b465cc1853aaa448a81f7
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:116b971bb63560610677655e38ba8be2afb415c6fd29c1d069804301bcde855e
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:03d14372590aaa8a694124e857b900ce7df2b5fee119f58bc35688c4616f5961
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.002,
6
  "eval_steps": 500,
7
- "global_step": 100,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -78,6 +78,76 @@
78
  "learning_rate": 0.00029699999999999996,
79
  "loss": 12.466087341308594,
80
  "step": 100
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
81
  }
82
  ],
83
  "logging_steps": 10,
@@ -97,7 +167,7 @@
97
  "attributes": {}
98
  }
99
  },
100
- "total_flos": 1889437483008000.0,
101
  "train_batch_size": 2,
102
  "trial_name": null,
103
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.004,
6
  "eval_steps": 500,
7
+ "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
78
  "learning_rate": 0.00029699999999999996,
79
  "loss": 12.466087341308594,
80
  "step": 100
81
+ },
82
+ {
83
+ "epoch": 0.0022,
84
+ "grad_norm": 0.9064206480979919,
85
+ "learning_rate": 0.0002999458917835671,
86
+ "loss": 12.170447540283202,
87
+ "step": 110
88
+ },
89
+ {
90
+ "epoch": 0.0024,
91
+ "grad_norm": 4.273526191711426,
92
+ "learning_rate": 0.00029988577154308615,
93
+ "loss": 10.753311157226562,
94
+ "step": 120
95
+ },
96
+ {
97
+ "epoch": 0.0026,
98
+ "grad_norm": 1.101340889930725,
99
+ "learning_rate": 0.0002998256513026052,
100
+ "loss": 12.582955932617187,
101
+ "step": 130
102
+ },
103
+ {
104
+ "epoch": 0.0028,
105
+ "grad_norm": 0.7963618040084839,
106
+ "learning_rate": 0.0002997655310621242,
107
+ "loss": 12.283400726318359,
108
+ "step": 140
109
+ },
110
+ {
111
+ "epoch": 0.003,
112
+ "grad_norm": 1.5654473304748535,
113
+ "learning_rate": 0.00029970541082164324,
114
+ "loss": 11.285658264160157,
115
+ "step": 150
116
+ },
117
+ {
118
+ "epoch": 0.0032,
119
+ "grad_norm": 1.8542180061340332,
120
+ "learning_rate": 0.0002996452905811623,
121
+ "loss": 10.847380065917969,
122
+ "step": 160
123
+ },
124
+ {
125
+ "epoch": 0.0034,
126
+ "grad_norm": 2.580841302871704,
127
+ "learning_rate": 0.00029958517034068134,
128
+ "loss": 9.636419677734375,
129
+ "step": 170
130
+ },
131
+ {
132
+ "epoch": 0.0036,
133
+ "grad_norm": 1.7422585487365723,
134
+ "learning_rate": 0.0002995250501002004,
135
+ "loss": 12.142258453369141,
136
+ "step": 180
137
+ },
138
+ {
139
+ "epoch": 0.0038,
140
+ "grad_norm": 1.7994998693466187,
141
+ "learning_rate": 0.00029946492985971943,
142
+ "loss": 11.316072845458985,
143
+ "step": 190
144
+ },
145
+ {
146
+ "epoch": 0.004,
147
+ "grad_norm": 0.8520310521125793,
148
+ "learning_rate": 0.0002994048096192384,
149
+ "loss": 9.44781723022461,
150
+ "step": 200
151
  }
152
  ],
153
  "logging_steps": 10,
 
167
  "attributes": {}
168
  }
169
  },
170
+ "total_flos": 3778874966016000.0,
171
  "train_batch_size": 2,
172
  "trial_name": null,
173
  "trial_params": null