CodeIsAbstract commited on
Commit
5712ed1
·
verified ·
1 Parent(s): 6aea879

Training in progress, step 3000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d81ad66b4930d7b2dd698095062a3016fe78a7fa017e78cd010fb6336a5852a1
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:af5354d0e29ddec9e27730302b091cb0479fe966a93828191865da85fb63ce05
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:45c86d2326cf881f6bfca4782a087b281584d3ed7d1eb8ee04cb9929d154a36b
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7f761042663de134db3875290b47fa4678d163e6a29c60d847213bd9e9bf820e
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:325c636e6311e0fe4ca41b46abaf69ab835c7fa1be4ee8051f00fed0ade88598
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:83d5db11e093d39ad6884cbfd0832c030b7cb2063c1e8ae3708e8e5be7a44ced
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b652e5527d5065c8d23b740b7553715cb3c1ddbdfe1b6360339df9f41f3fe84c
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:623efa011285242890286b7518de62c3b1d6ed1f56707c7ec58c501e60c532b6
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.5,
6
  "eval_steps": 100,
7
- "global_step": 2500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1958,6 +1958,396 @@
1958
  "eval_samples_per_second": 66.962,
1959
  "eval_steps_per_second": 3.82,
1960
  "step": 2500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1961
  }
1962
  ],
1963
  "logging_steps": 10,
@@ -1977,7 +2367,7 @@
1977
  "attributes": {}
1978
  }
1979
  },
1980
- "total_flos": 8.1538628714496e+17,
1981
  "train_batch_size": 18,
1982
  "trial_name": null,
1983
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.6,
6
  "eval_steps": 100,
7
+ "global_step": 3000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1958
  "eval_samples_per_second": 66.962,
1959
  "eval_steps_per_second": 3.82,
1960
  "step": 2500
1961
+ },
1962
+ {
1963
+ "epoch": 0.502,
1964
+ "grad_norm": 0.048095703125,
1965
+ "learning_rate": 0.0003,
1966
+ "loss": 2.600408363342285,
1967
+ "step": 2510
1968
+ },
1969
+ {
1970
+ "epoch": 0.504,
1971
+ "grad_norm": 0.052001953125,
1972
+ "learning_rate": 0.0003,
1973
+ "loss": 2.6103189468383787,
1974
+ "step": 2520
1975
+ },
1976
+ {
1977
+ "epoch": 0.506,
1978
+ "grad_norm": 0.05712890625,
1979
+ "learning_rate": 0.0003,
1980
+ "loss": 2.5979167938232424,
1981
+ "step": 2530
1982
+ },
1983
+ {
1984
+ "epoch": 0.508,
1985
+ "grad_norm": 0.049560546875,
1986
+ "learning_rate": 0.0003,
1987
+ "loss": 2.5972728729248047,
1988
+ "step": 2540
1989
+ },
1990
+ {
1991
+ "epoch": 0.51,
1992
+ "grad_norm": 0.050048828125,
1993
+ "learning_rate": 0.0003,
1994
+ "loss": 2.608352851867676,
1995
+ "step": 2550
1996
+ },
1997
+ {
1998
+ "epoch": 0.512,
1999
+ "grad_norm": 0.07275390625,
2000
+ "learning_rate": 0.0003,
2001
+ "loss": 2.600900650024414,
2002
+ "step": 2560
2003
+ },
2004
+ {
2005
+ "epoch": 0.514,
2006
+ "grad_norm": 0.052734375,
2007
+ "learning_rate": 0.0003,
2008
+ "loss": 2.638876533508301,
2009
+ "step": 2570
2010
+ },
2011
+ {
2012
+ "epoch": 0.516,
2013
+ "grad_norm": 0.05126953125,
2014
+ "learning_rate": 0.0003,
2015
+ "loss": 2.610831069946289,
2016
+ "step": 2580
2017
+ },
2018
+ {
2019
+ "epoch": 0.518,
2020
+ "grad_norm": 0.051513671875,
2021
+ "learning_rate": 0.0003,
2022
+ "loss": 2.6283756256103517,
2023
+ "step": 2590
2024
+ },
2025
+ {
2026
+ "epoch": 0.52,
2027
+ "grad_norm": 0.0791015625,
2028
+ "learning_rate": 0.0003,
2029
+ "loss": 2.6440673828125,
2030
+ "step": 2600
2031
+ },
2032
+ {
2033
+ "epoch": 0.52,
2034
+ "eval_loss": 3.0254361629486084,
2035
+ "eval_runtime": 4.3572,
2036
+ "eval_samples_per_second": 68.392,
2037
+ "eval_steps_per_second": 3.902,
2038
+ "step": 2600
2039
+ },
2040
+ {
2041
+ "epoch": 0.522,
2042
+ "grad_norm": 0.05029296875,
2043
+ "learning_rate": 0.0003,
2044
+ "loss": 2.625893211364746,
2045
+ "step": 2610
2046
+ },
2047
+ {
2048
+ "epoch": 0.524,
2049
+ "grad_norm": 0.061767578125,
2050
+ "learning_rate": 0.0003,
2051
+ "loss": 2.6046930313110352,
2052
+ "step": 2620
2053
+ },
2054
+ {
2055
+ "epoch": 0.526,
2056
+ "grad_norm": 0.055419921875,
2057
+ "learning_rate": 0.0003,
2058
+ "loss": 2.623915100097656,
2059
+ "step": 2630
2060
+ },
2061
+ {
2062
+ "epoch": 0.528,
2063
+ "grad_norm": 0.058349609375,
2064
+ "learning_rate": 0.0003,
2065
+ "loss": 2.6163360595703127,
2066
+ "step": 2640
2067
+ },
2068
+ {
2069
+ "epoch": 0.53,
2070
+ "grad_norm": 0.04736328125,
2071
+ "learning_rate": 0.0003,
2072
+ "loss": 2.5871484756469725,
2073
+ "step": 2650
2074
+ },
2075
+ {
2076
+ "epoch": 0.532,
2077
+ "grad_norm": 0.059326171875,
2078
+ "learning_rate": 0.0003,
2079
+ "loss": 2.6428932189941405,
2080
+ "step": 2660
2081
+ },
2082
+ {
2083
+ "epoch": 0.534,
2084
+ "grad_norm": 0.0498046875,
2085
+ "learning_rate": 0.0003,
2086
+ "loss": 2.608722686767578,
2087
+ "step": 2670
2088
+ },
2089
+ {
2090
+ "epoch": 0.536,
2091
+ "grad_norm": 0.224609375,
2092
+ "learning_rate": 0.0003,
2093
+ "loss": 2.6207298278808593,
2094
+ "step": 2680
2095
+ },
2096
+ {
2097
+ "epoch": 0.538,
2098
+ "grad_norm": 0.056884765625,
2099
+ "learning_rate": 0.0003,
2100
+ "loss": 2.598964309692383,
2101
+ "step": 2690
2102
+ },
2103
+ {
2104
+ "epoch": 0.54,
2105
+ "grad_norm": 0.185546875,
2106
+ "learning_rate": 0.0003,
2107
+ "loss": 2.6367502212524414,
2108
+ "step": 2700
2109
+ },
2110
+ {
2111
+ "epoch": 0.54,
2112
+ "eval_loss": 3.0251870155334473,
2113
+ "eval_runtime": 4.446,
2114
+ "eval_samples_per_second": 67.027,
2115
+ "eval_steps_per_second": 3.824,
2116
+ "step": 2700
2117
+ },
2118
+ {
2119
+ "epoch": 0.542,
2120
+ "grad_norm": 0.0732421875,
2121
+ "learning_rate": 0.0003,
2122
+ "loss": 2.625163269042969,
2123
+ "step": 2710
2124
+ },
2125
+ {
2126
+ "epoch": 0.544,
2127
+ "grad_norm": 0.054931640625,
2128
+ "learning_rate": 0.0003,
2129
+ "loss": 2.641737365722656,
2130
+ "step": 2720
2131
+ },
2132
+ {
2133
+ "epoch": 0.546,
2134
+ "grad_norm": 0.1845703125,
2135
+ "learning_rate": 0.0003,
2136
+ "loss": 2.6120086669921876,
2137
+ "step": 2730
2138
+ },
2139
+ {
2140
+ "epoch": 0.548,
2141
+ "grad_norm": 0.049560546875,
2142
+ "learning_rate": 0.0003,
2143
+ "loss": 2.5945173263549806,
2144
+ "step": 2740
2145
+ },
2146
+ {
2147
+ "epoch": 0.55,
2148
+ "grad_norm": 0.060302734375,
2149
+ "learning_rate": 0.0003,
2150
+ "loss": 2.635161590576172,
2151
+ "step": 2750
2152
+ },
2153
+ {
2154
+ "epoch": 0.552,
2155
+ "grad_norm": 1.8046875,
2156
+ "learning_rate": 0.0003,
2157
+ "loss": 2.6317985534667967,
2158
+ "step": 2760
2159
+ },
2160
+ {
2161
+ "epoch": 0.554,
2162
+ "grad_norm": 0.0517578125,
2163
+ "learning_rate": 0.0003,
2164
+ "loss": 2.6268796920776367,
2165
+ "step": 2770
2166
+ },
2167
+ {
2168
+ "epoch": 0.556,
2169
+ "grad_norm": 2.015625,
2170
+ "learning_rate": 0.0003,
2171
+ "loss": 2.639748382568359,
2172
+ "step": 2780
2173
+ },
2174
+ {
2175
+ "epoch": 0.558,
2176
+ "grad_norm": 0.061767578125,
2177
+ "learning_rate": 0.0003,
2178
+ "loss": 2.6535463333129883,
2179
+ "step": 2790
2180
+ },
2181
+ {
2182
+ "epoch": 0.56,
2183
+ "grad_norm": 0.06396484375,
2184
+ "learning_rate": 0.0003,
2185
+ "loss": 2.609045219421387,
2186
+ "step": 2800
2187
+ },
2188
+ {
2189
+ "epoch": 0.56,
2190
+ "eval_loss": 3.024867534637451,
2191
+ "eval_runtime": 4.4522,
2192
+ "eval_samples_per_second": 66.933,
2193
+ "eval_steps_per_second": 3.818,
2194
+ "step": 2800
2195
+ },
2196
+ {
2197
+ "epoch": 0.562,
2198
+ "grad_norm": 0.064453125,
2199
+ "learning_rate": 0.0003,
2200
+ "loss": 2.6145769119262696,
2201
+ "step": 2810
2202
+ },
2203
+ {
2204
+ "epoch": 0.564,
2205
+ "grad_norm": 0.052978515625,
2206
+ "learning_rate": 0.0003,
2207
+ "loss": 2.603018379211426,
2208
+ "step": 2820
2209
+ },
2210
+ {
2211
+ "epoch": 0.566,
2212
+ "grad_norm": 0.05615234375,
2213
+ "learning_rate": 0.0003,
2214
+ "loss": 2.6036045074462892,
2215
+ "step": 2830
2216
+ },
2217
+ {
2218
+ "epoch": 0.568,
2219
+ "grad_norm": 0.049560546875,
2220
+ "learning_rate": 0.0003,
2221
+ "loss": 2.610326385498047,
2222
+ "step": 2840
2223
+ },
2224
+ {
2225
+ "epoch": 0.57,
2226
+ "grad_norm": 0.060546875,
2227
+ "learning_rate": 0.0003,
2228
+ "loss": 2.5892526626586916,
2229
+ "step": 2850
2230
+ },
2231
+ {
2232
+ "epoch": 0.572,
2233
+ "grad_norm": 0.06103515625,
2234
+ "learning_rate": 0.0003,
2235
+ "loss": 2.6054714202880858,
2236
+ "step": 2860
2237
+ },
2238
+ {
2239
+ "epoch": 0.574,
2240
+ "grad_norm": 0.05810546875,
2241
+ "learning_rate": 0.0003,
2242
+ "loss": 2.622623062133789,
2243
+ "step": 2870
2244
+ },
2245
+ {
2246
+ "epoch": 0.576,
2247
+ "grad_norm": 0.384765625,
2248
+ "learning_rate": 0.0003,
2249
+ "loss": 2.6183904647827148,
2250
+ "step": 2880
2251
+ },
2252
+ {
2253
+ "epoch": 0.578,
2254
+ "grad_norm": 4.875,
2255
+ "learning_rate": 0.0003,
2256
+ "loss": 2.5992977142333986,
2257
+ "step": 2890
2258
+ },
2259
+ {
2260
+ "epoch": 0.58,
2261
+ "grad_norm": 0.154296875,
2262
+ "learning_rate": 0.0003,
2263
+ "loss": 2.6220735549926757,
2264
+ "step": 2900
2265
+ },
2266
+ {
2267
+ "epoch": 0.58,
2268
+ "eval_loss": 3.024597644805908,
2269
+ "eval_runtime": 4.4485,
2270
+ "eval_samples_per_second": 66.989,
2271
+ "eval_steps_per_second": 3.822,
2272
+ "step": 2900
2273
+ },
2274
+ {
2275
+ "epoch": 0.582,
2276
+ "grad_norm": 0.046875,
2277
+ "learning_rate": 0.0003,
2278
+ "loss": 2.6097631454467773,
2279
+ "step": 2910
2280
+ },
2281
+ {
2282
+ "epoch": 0.584,
2283
+ "grad_norm": 0.435546875,
2284
+ "learning_rate": 0.0003,
2285
+ "loss": 2.6088932037353514,
2286
+ "step": 2920
2287
+ },
2288
+ {
2289
+ "epoch": 0.586,
2290
+ "grad_norm": 0.06494140625,
2291
+ "learning_rate": 0.0003,
2292
+ "loss": 2.6292213439941405,
2293
+ "step": 2930
2294
+ },
2295
+ {
2296
+ "epoch": 0.588,
2297
+ "grad_norm": 0.052734375,
2298
+ "learning_rate": 0.0003,
2299
+ "loss": 2.6273422241210938,
2300
+ "step": 2940
2301
+ },
2302
+ {
2303
+ "epoch": 0.59,
2304
+ "grad_norm": 0.09716796875,
2305
+ "learning_rate": 0.0003,
2306
+ "loss": 2.622287368774414,
2307
+ "step": 2950
2308
+ },
2309
+ {
2310
+ "epoch": 0.592,
2311
+ "grad_norm": 0.0498046875,
2312
+ "learning_rate": 0.0003,
2313
+ "loss": 2.5906503677368162,
2314
+ "step": 2960
2315
+ },
2316
+ {
2317
+ "epoch": 0.594,
2318
+ "grad_norm": 0.057373046875,
2319
+ "learning_rate": 0.0003,
2320
+ "loss": 2.631303596496582,
2321
+ "step": 2970
2322
+ },
2323
+ {
2324
+ "epoch": 0.596,
2325
+ "grad_norm": 0.054443359375,
2326
+ "learning_rate": 0.0003,
2327
+ "loss": 2.6114727020263673,
2328
+ "step": 2980
2329
+ },
2330
+ {
2331
+ "epoch": 0.598,
2332
+ "grad_norm": 0.0908203125,
2333
+ "learning_rate": 0.0003,
2334
+ "loss": 2.603567886352539,
2335
+ "step": 2990
2336
+ },
2337
+ {
2338
+ "epoch": 0.6,
2339
+ "grad_norm": 0.0556640625,
2340
+ "learning_rate": 0.0003,
2341
+ "loss": 2.585314178466797,
2342
+ "step": 3000
2343
+ },
2344
+ {
2345
+ "epoch": 0.6,
2346
+ "eval_loss": 3.0278046131134033,
2347
+ "eval_runtime": 4.4444,
2348
+ "eval_samples_per_second": 67.051,
2349
+ "eval_steps_per_second": 3.825,
2350
+ "step": 3000
2351
  }
2352
  ],
2353
  "logging_steps": 10,
 
2367
  "attributes": {}
2368
  }
2369
  },
2370
+ "total_flos": 9.78463544573952e+17,
2371
  "train_batch_size": 18,
2372
  "trial_name": null,
2373
  "trial_params": null