CodeIsAbstract commited on
Commit
51db28b
·
verified ·
1 Parent(s): d570bdd

Training in progress, step 28000, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ac08b42599d1bde48be74806637c6c061e7123fe79086fdd5f3704e9fd3d6ce1
3
  size 541154336
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:32cc85238f9b318459c5a80206968432f8685331234104eda2016524ba706b75
3
  size 541154336
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a587cba38b05004cfdcf48f4bea4b6d068091255cf02ec07324ae8e96318e4f0
3
  size 1082379659
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5e66d55b95606084af7dc0fbf657d700ecafd09b008231850874bdde638deda6
3
  size 1082379659
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a804dd9b4962bc1e7c8e5b51c83ce95f04ab0a366340b47fc4849e7d4ecffd6d
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8553d3987609520495680f36a4682a760a418a930648ce46178a1399cde440b0
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:668b318436d635cfc651258a221542355f21124096cff5d5c7e78a7d5c81ab33
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:71903ea651461945bbccce0e330f120093c2124a33f5b93ba51f8b9e386de7f9
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.3116883116883117,
6
  "eval_steps": 1000,
7
- "global_step": 24000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -1880,6 +1880,318 @@
1880
  "eval_samples_per_second": 35.576,
1881
  "eval_steps_per_second": 8.894,
1882
  "step": 24000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1883
  }
1884
  ],
1885
  "logging_steps": 100,
@@ -1899,7 +2211,7 @@
1899
  "attributes": {}
1900
  }
1901
  },
1902
- "total_flos": 7.1446353149952e+17,
1903
  "train_batch_size": 22,
1904
  "trial_name": null,
1905
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.36363636363636365,
6
  "eval_steps": 1000,
7
+ "global_step": 28000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
1880
  "eval_samples_per_second": 35.576,
1881
  "eval_steps_per_second": 8.894,
1882
  "step": 24000
1883
+ },
1884
+ {
1885
+ "epoch": 0.312987012987013,
1886
+ "grad_norm": 0.21797960996627808,
1887
+ "learning_rate": 0.0008935749824785922,
1888
+ "loss": 2.8992,
1889
+ "step": 24100
1890
+ },
1891
+ {
1892
+ "epoch": 0.3142857142857143,
1893
+ "grad_norm": 0.19554930925369263,
1894
+ "learning_rate": 0.0008926841258872131,
1895
+ "loss": 2.8771,
1896
+ "step": 24200
1897
+ },
1898
+ {
1899
+ "epoch": 0.3155844155844156,
1900
+ "grad_norm": 0.18446357548236847,
1901
+ "learning_rate": 0.0008917900042572283,
1902
+ "loss": 2.8753,
1903
+ "step": 24300
1904
+ },
1905
+ {
1906
+ "epoch": 0.3168831168831169,
1907
+ "grad_norm": 0.18406012654304504,
1908
+ "learning_rate": 0.0008908926250229633,
1909
+ "loss": 2.8819,
1910
+ "step": 24400
1911
+ },
1912
+ {
1913
+ "epoch": 0.3181818181818182,
1914
+ "grad_norm": 0.21075041592121124,
1915
+ "learning_rate": 0.0008899919956458296,
1916
+ "loss": 2.8796,
1917
+ "step": 24500
1918
+ },
1919
+ {
1920
+ "epoch": 0.3194805194805195,
1921
+ "grad_norm": 0.20150282979011536,
1922
+ "learning_rate": 0.0008890881236142624,
1923
+ "loss": 2.8611,
1924
+ "step": 24600
1925
+ },
1926
+ {
1927
+ "epoch": 0.3207792207792208,
1928
+ "grad_norm": 0.43753939867019653,
1929
+ "learning_rate": 0.0008881810164436588,
1930
+ "loss": 2.8587,
1931
+ "step": 24700
1932
+ },
1933
+ {
1934
+ "epoch": 0.3220779220779221,
1935
+ "grad_norm": 0.1800195425748825,
1936
+ "learning_rate": 0.0008872706816763148,
1937
+ "loss": 2.8797,
1938
+ "step": 24800
1939
+ },
1940
+ {
1941
+ "epoch": 0.3233766233766234,
1942
+ "grad_norm": 0.2071843296289444,
1943
+ "learning_rate": 0.0008863571268813628,
1944
+ "loss": 2.8697,
1945
+ "step": 24900
1946
+ },
1947
+ {
1948
+ "epoch": 0.3246753246753247,
1949
+ "grad_norm": 0.19320254027843475,
1950
+ "learning_rate": 0.0008854403596547088,
1951
+ "loss": 2.8938,
1952
+ "step": 25000
1953
+ },
1954
+ {
1955
+ "epoch": 0.3246753246753247,
1956
+ "eval_loss": 3.232848882675171,
1957
+ "eval_runtime": 15.6644,
1958
+ "eval_samples_per_second": 36.771,
1959
+ "eval_steps_per_second": 9.193,
1960
+ "step": 25000
1961
+ },
1962
+ {
1963
+ "epoch": 0.32597402597402597,
1964
+ "grad_norm": 0.21449397504329681,
1965
+ "learning_rate": 0.000884520387618969,
1966
+ "loss": 2.8374,
1967
+ "step": 25100
1968
+ },
1969
+ {
1970
+ "epoch": 0.32727272727272727,
1971
+ "grad_norm": 0.2025729864835739,
1972
+ "learning_rate": 0.0008835972184234066,
1973
+ "loss": 2.845,
1974
+ "step": 25200
1975
+ },
1976
+ {
1977
+ "epoch": 0.32857142857142857,
1978
+ "grad_norm": 0.22090762853622437,
1979
+ "learning_rate": 0.000882670859743868,
1980
+ "loss": 2.8518,
1981
+ "step": 25300
1982
+ },
1983
+ {
1984
+ "epoch": 0.32987012987012987,
1985
+ "grad_norm": 0.20626656711101532,
1986
+ "learning_rate": 0.0008817413192827191,
1987
+ "loss": 2.8299,
1988
+ "step": 25400
1989
+ },
1990
+ {
1991
+ "epoch": 0.33116883116883117,
1992
+ "grad_norm": 0.18527238070964813,
1993
+ "learning_rate": 0.0008808086047687813,
1994
+ "loss": 2.8532,
1995
+ "step": 25500
1996
+ },
1997
+ {
1998
+ "epoch": 0.33246753246753247,
1999
+ "grad_norm": 0.20959258079528809,
2000
+ "learning_rate": 0.0008798727239572676,
2001
+ "loss": 2.8656,
2002
+ "step": 25600
2003
+ },
2004
+ {
2005
+ "epoch": 0.33376623376623377,
2006
+ "grad_norm": 0.2122378945350647,
2007
+ "learning_rate": 0.000878933684629717,
2008
+ "loss": 2.8532,
2009
+ "step": 25700
2010
+ },
2011
+ {
2012
+ "epoch": 0.33506493506493507,
2013
+ "grad_norm": 0.18408456444740295,
2014
+ "learning_rate": 0.0008779914945939311,
2015
+ "loss": 2.8639,
2016
+ "step": 25800
2017
+ },
2018
+ {
2019
+ "epoch": 0.33636363636363636,
2020
+ "grad_norm": 0.20426695048809052,
2021
+ "learning_rate": 0.0008770461616839082,
2022
+ "loss": 2.865,
2023
+ "step": 25900
2024
+ },
2025
+ {
2026
+ "epoch": 0.33766233766233766,
2027
+ "grad_norm": 0.1808163821697235,
2028
+ "learning_rate": 0.0008760976937597787,
2029
+ "loss": 2.8481,
2030
+ "step": 26000
2031
+ },
2032
+ {
2033
+ "epoch": 0.33766233766233766,
2034
+ "eval_loss": 3.214432716369629,
2035
+ "eval_runtime": 15.0454,
2036
+ "eval_samples_per_second": 38.284,
2037
+ "eval_steps_per_second": 9.571,
2038
+ "step": 26000
2039
+ },
2040
+ {
2041
+ "epoch": 0.33896103896103896,
2042
+ "grad_norm": 0.20782601833343506,
2043
+ "learning_rate": 0.0008751460987077398,
2044
+ "loss": 2.8405,
2045
+ "step": 26100
2046
+ },
2047
+ {
2048
+ "epoch": 0.34025974025974026,
2049
+ "grad_norm": 0.2266394942998886,
2050
+ "learning_rate": 0.0008741913844399896,
2051
+ "loss": 2.8304,
2052
+ "step": 26200
2053
+ },
2054
+ {
2055
+ "epoch": 0.34155844155844156,
2056
+ "grad_norm": 0.215355783700943,
2057
+ "learning_rate": 0.0008732335588946612,
2058
+ "loss": 2.8749,
2059
+ "step": 26300
2060
+ },
2061
+ {
2062
+ "epoch": 0.34285714285714286,
2063
+ "grad_norm": 0.17590366303920746,
2064
+ "learning_rate": 0.0008722726300357573,
2065
+ "loss": 2.8542,
2066
+ "step": 26400
2067
+ },
2068
+ {
2069
+ "epoch": 0.34415584415584416,
2070
+ "grad_norm": 0.19376814365386963,
2071
+ "learning_rate": 0.0008713086058530832,
2072
+ "loss": 2.8636,
2073
+ "step": 26500
2074
+ },
2075
+ {
2076
+ "epoch": 0.34545454545454546,
2077
+ "grad_norm": 0.19086168706417084,
2078
+ "learning_rate": 0.0008703414943621817,
2079
+ "loss": 2.8752,
2080
+ "step": 26600
2081
+ },
2082
+ {
2083
+ "epoch": 0.34675324675324676,
2084
+ "grad_norm": 0.19386428594589233,
2085
+ "learning_rate": 0.0008693713036042643,
2086
+ "loss": 2.8595,
2087
+ "step": 26700
2088
+ },
2089
+ {
2090
+ "epoch": 0.34805194805194806,
2091
+ "grad_norm": 0.20364010334014893,
2092
+ "learning_rate": 0.0008683980416461465,
2093
+ "loss": 2.8322,
2094
+ "step": 26800
2095
+ },
2096
+ {
2097
+ "epoch": 0.34935064935064936,
2098
+ "grad_norm": 0.20115472376346588,
2099
+ "learning_rate": 0.0008674217165801797,
2100
+ "loss": 2.8271,
2101
+ "step": 26900
2102
+ },
2103
+ {
2104
+ "epoch": 0.35064935064935066,
2105
+ "grad_norm": 0.19452853500843048,
2106
+ "learning_rate": 0.0008664423365241833,
2107
+ "loss": 2.8445,
2108
+ "step": 27000
2109
+ },
2110
+ {
2111
+ "epoch": 0.35064935064935066,
2112
+ "eval_loss": 3.206655263900757,
2113
+ "eval_runtime": 14.9037,
2114
+ "eval_samples_per_second": 38.648,
2115
+ "eval_steps_per_second": 9.662,
2116
+ "step": 27000
2117
+ },
2118
+ {
2119
+ "epoch": 0.35194805194805195,
2120
+ "grad_norm": 0.29245325922966003,
2121
+ "learning_rate": 0.0008654599096213792,
2122
+ "loss": 2.8584,
2123
+ "step": 27100
2124
+ },
2125
+ {
2126
+ "epoch": 0.35324675324675325,
2127
+ "grad_norm": 0.1941668838262558,
2128
+ "learning_rate": 0.0008644744440403214,
2129
+ "loss": 2.8146,
2130
+ "step": 27200
2131
+ },
2132
+ {
2133
+ "epoch": 0.35454545454545455,
2134
+ "grad_norm": 0.20218811929225922,
2135
+ "learning_rate": 0.0008634859479748306,
2136
+ "loss": 2.8623,
2137
+ "step": 27300
2138
+ },
2139
+ {
2140
+ "epoch": 0.35584415584415585,
2141
+ "grad_norm": 0.19882351160049438,
2142
+ "learning_rate": 0.0008624944296439246,
2143
+ "loss": 2.8491,
2144
+ "step": 27400
2145
+ },
2146
+ {
2147
+ "epoch": 0.35714285714285715,
2148
+ "grad_norm": 0.19472059607505798,
2149
+ "learning_rate": 0.0008614998972917503,
2150
+ "loss": 2.8372,
2151
+ "step": 27500
2152
+ },
2153
+ {
2154
+ "epoch": 0.35844155844155845,
2155
+ "grad_norm": 0.20453399419784546,
2156
+ "learning_rate": 0.000860502359187515,
2157
+ "loss": 2.8228,
2158
+ "step": 27600
2159
+ },
2160
+ {
2161
+ "epoch": 0.35974025974025975,
2162
+ "grad_norm": 0.20085613429546356,
2163
+ "learning_rate": 0.0008595018236254182,
2164
+ "loss": 2.8416,
2165
+ "step": 27700
2166
+ },
2167
+ {
2168
+ "epoch": 0.36103896103896105,
2169
+ "grad_norm": 0.1953079253435135,
2170
+ "learning_rate": 0.0008584982989245822,
2171
+ "loss": 2.844,
2172
+ "step": 27800
2173
+ },
2174
+ {
2175
+ "epoch": 0.36233766233766235,
2176
+ "grad_norm": 0.18861424922943115,
2177
+ "learning_rate": 0.0008574917934289829,
2178
+ "loss": 2.8586,
2179
+ "step": 27900
2180
+ },
2181
+ {
2182
+ "epoch": 0.36363636363636365,
2183
+ "grad_norm": 0.2032879889011383,
2184
+ "learning_rate": 0.0008564823155073804,
2185
+ "loss": 2.8676,
2186
+ "step": 28000
2187
+ },
2188
+ {
2189
+ "epoch": 0.36363636363636365,
2190
+ "eval_loss": 3.213535785675049,
2191
+ "eval_runtime": 15.9519,
2192
+ "eval_samples_per_second": 36.109,
2193
+ "eval_steps_per_second": 9.027,
2194
+ "step": 28000
2195
  }
2196
  ],
2197
  "logging_steps": 100,
 
2211
  "attributes": {}
2212
  }
2213
  },
2214
+ "total_flos": 8.3354078674944e+17,
2215
  "train_batch_size": 22,
2216
  "trial_name": null,
2217
  "trial_params": null