michaelc0des commited on
Commit
cf93d2b
·
verified ·
1 Parent(s): 4dfcc01

Upload checkpoint-1039

Browse files
checkpoint-1039/rng_state_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ab57b6f0fc9c83b89ae996a7c767658a6cd61e987d6078df318a8206fe30f546
3
+ size 16389
checkpoint-1039/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8e20e8455548e074938a3ef8c1dd1413440a4755c4fbfba6c8db9bbfbf3d5381
3
+ size 1465
checkpoint-1039/trainer_state.json ADDED
@@ -0,0 +1,336 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": 1039,
3
+ "best_metric": 2.934025287628174,
4
+ "best_model_checkpoint": "outputs/chess-explainer/checkpoint-1039",
5
+ "epoch": 1.0,
6
+ "eval_steps": 500,
7
+ "global_step": 1039,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.0009624639076034649,
14
+ "grad_norm": 883.1925048828125,
15
+ "learning_rate": 0.0,
16
+ "loss": 17.491071701049805,
17
+ "step": 1
18
+ },
19
+ {
20
+ "epoch": 0.02406159769008662,
21
+ "grad_norm": 55.103694915771484,
22
+ "learning_rate": 1.5384615384615387e-05,
23
+ "loss": 11.750035603841146,
24
+ "step": 25
25
+ },
26
+ {
27
+ "epoch": 0.04812319538017324,
28
+ "grad_norm": 9.345708847045898,
29
+ "learning_rate": 3.141025641025641e-05,
30
+ "loss": 4.889132080078125,
31
+ "step": 50
32
+ },
33
+ {
34
+ "epoch": 0.07218479307025986,
35
+ "grad_norm": 10.196633338928223,
36
+ "learning_rate": 4.7435897435897435e-05,
37
+ "loss": 3.7856689453125,
38
+ "step": 75
39
+ },
40
+ {
41
+ "epoch": 0.09624639076034648,
42
+ "grad_norm": 8.611700057983398,
43
+ "learning_rate": 6.346153846153847e-05,
44
+ "loss": 3.5517022705078123,
45
+ "step": 100
46
+ },
47
+ {
48
+ "epoch": 0.12030798845043311,
49
+ "grad_norm": 7.812922477722168,
50
+ "learning_rate": 7.948717948717948e-05,
51
+ "loss": 3.4541973876953125,
52
+ "step": 125
53
+ },
54
+ {
55
+ "epoch": 0.14436958614051973,
56
+ "grad_norm": 7.820157527923584,
57
+ "learning_rate": 9.551282051282052e-05,
58
+ "loss": 3.3625674438476563,
59
+ "step": 150
60
+ },
61
+ {
62
+ "epoch": 0.16843118383060635,
63
+ "grad_norm": 7.358847141265869,
64
+ "learning_rate": 0.00011153846153846154,
65
+ "loss": 3.33571533203125,
66
+ "step": 175
67
+ },
68
+ {
69
+ "epoch": 0.19249278152069296,
70
+ "grad_norm": 7.274508953094482,
71
+ "learning_rate": 0.00012756410256410257,
72
+ "loss": 3.3122216796875,
73
+ "step": 200
74
+ },
75
+ {
76
+ "epoch": 0.2165543792107796,
77
+ "grad_norm": 7.056718349456787,
78
+ "learning_rate": 0.0001435897435897436,
79
+ "loss": 3.28180419921875,
80
+ "step": 225
81
+ },
82
+ {
83
+ "epoch": 0.24061597690086622,
84
+ "grad_norm": 6.19630241394043,
85
+ "learning_rate": 0.00015961538461538462,
86
+ "loss": 3.2475381469726563,
87
+ "step": 250
88
+ },
89
+ {
90
+ "epoch": 0.2646775745909528,
91
+ "grad_norm": 6.5128397941589355,
92
+ "learning_rate": 0.00017564102564102566,
93
+ "loss": 3.234718017578125,
94
+ "step": 275
95
+ },
96
+ {
97
+ "epoch": 0.28873917228103946,
98
+ "grad_norm": 6.359344482421875,
99
+ "learning_rate": 0.00019166666666666667,
100
+ "loss": 3.2271856689453124,
101
+ "step": 300
102
+ },
103
+ {
104
+ "epoch": 0.3128007699711261,
105
+ "grad_norm": 6.695319175720215,
106
+ "learning_rate": 0.0001999993003464737,
107
+ "loss": 3.1996661376953126,
108
+ "step": 325
109
+ },
110
+ {
111
+ "epoch": 0.3368623676612127,
112
+ "grad_norm": 5.9124321937561035,
113
+ "learning_rate": 0.00019999334849877745,
114
+ "loss": 3.1934442138671875,
115
+ "step": 350
116
+ },
117
+ {
118
+ "epoch": 0.36092396535129934,
119
+ "grad_norm": 5.623684406280518,
120
+ "learning_rate": 0.00019998132369740206,
121
+ "loss": 3.1889007568359373,
122
+ "step": 375
123
+ },
124
+ {
125
+ "epoch": 0.3849855630413859,
126
+ "grad_norm": 5.664840221405029,
127
+ "learning_rate": 0.00019996322667265672,
128
+ "loss": 3.173824462890625,
129
+ "step": 400
130
+ },
131
+ {
132
+ "epoch": 0.40904716073147257,
133
+ "grad_norm": 6.227374076843262,
134
+ "learning_rate": 0.0001999390585236385,
135
+ "loss": 3.15881591796875,
136
+ "step": 425
137
+ },
138
+ {
139
+ "epoch": 0.4331087584215592,
140
+ "grad_norm": 5.214109420776367,
141
+ "learning_rate": 0.0001999088207181655,
142
+ "loss": 3.09910888671875,
143
+ "step": 450
144
+ },
145
+ {
146
+ "epoch": 0.4571703561116458,
147
+ "grad_norm": 5.23874568939209,
148
+ "learning_rate": 0.0001998725150926878,
149
+ "loss": 3.1576531982421874,
150
+ "step": 475
151
+ },
152
+ {
153
+ "epoch": 0.48123195380173245,
154
+ "grad_norm": 5.108809471130371,
155
+ "learning_rate": 0.0001998301438521759,
156
+ "loss": 3.1337042236328125,
157
+ "step": 500
158
+ },
159
+ {
160
+ "epoch": 0.5052935514918191,
161
+ "grad_norm": 5.708388805389404,
162
+ "learning_rate": 0.00019978170956998676,
163
+ "loss": 3.128594970703125,
164
+ "step": 525
165
+ },
166
+ {
167
+ "epoch": 0.5293551491819056,
168
+ "grad_norm": 5.174871921539307,
169
+ "learning_rate": 0.00019972721518770757,
170
+ "loss": 3.1022515869140626,
171
+ "step": 550
172
+ },
173
+ {
174
+ "epoch": 0.5534167468719923,
175
+ "grad_norm": 5.113072872161865,
176
+ "learning_rate": 0.00019966666401497705,
177
+ "loss": 3.063781433105469,
178
+ "step": 575
179
+ },
180
+ {
181
+ "epoch": 0.5774783445620789,
182
+ "grad_norm": 5.224897384643555,
183
+ "learning_rate": 0.00019960005972928445,
184
+ "loss": 3.0889013671875,
185
+ "step": 600
186
+ },
187
+ {
188
+ "epoch": 0.6015399422521656,
189
+ "grad_norm": 5.108465671539307,
190
+ "learning_rate": 0.00019952740637574633,
191
+ "loss": 3.0858636474609376,
192
+ "step": 625
193
+ },
194
+ {
195
+ "epoch": 0.6256015399422522,
196
+ "grad_norm": 5.350228786468506,
197
+ "learning_rate": 0.00019944870836686063,
198
+ "loss": 3.0400531005859377,
199
+ "step": 650
200
+ },
201
+ {
202
+ "epoch": 0.6496631376323387,
203
+ "grad_norm": 4.423816204071045,
204
+ "learning_rate": 0.0001993639704822389,
205
+ "loss": 3.06386474609375,
206
+ "step": 675
207
+ },
208
+ {
209
+ "epoch": 0.6737247353224254,
210
+ "grad_norm": 5.499719619750977,
211
+ "learning_rate": 0.00019927319786831594,
212
+ "loss": 3.0412890625,
213
+ "step": 700
214
+ },
215
+ {
216
+ "epoch": 0.697786333012512,
217
+ "grad_norm": 4.638278007507324,
218
+ "learning_rate": 0.0001991763960380373,
219
+ "loss": 3.0481976318359374,
220
+ "step": 725
221
+ },
222
+ {
223
+ "epoch": 0.7218479307025987,
224
+ "grad_norm": 5.008415222167969,
225
+ "learning_rate": 0.00019907357087052424,
226
+ "loss": 3.0042901611328126,
227
+ "step": 750
228
+ },
229
+ {
230
+ "epoch": 0.7459095283926853,
231
+ "grad_norm": 4.793009281158447,
232
+ "learning_rate": 0.00019896472861071702,
233
+ "loss": 2.9917413330078126,
234
+ "step": 775
235
+ },
236
+ {
237
+ "epoch": 0.7699711260827719,
238
+ "grad_norm": 5.019183158874512,
239
+ "learning_rate": 0.0001988498758689953,
240
+ "loss": 3.0368035888671874,
241
+ "step": 800
242
+ },
243
+ {
244
+ "epoch": 0.7940327237728585,
245
+ "grad_norm": 4.315158843994141,
246
+ "learning_rate": 0.00019872901962077687,
247
+ "loss": 2.9863958740234375,
248
+ "step": 825
249
+ },
250
+ {
251
+ "epoch": 0.8180943214629451,
252
+ "grad_norm": 5.23874044418335,
253
+ "learning_rate": 0.00019860216720609395,
254
+ "loss": 2.9992340087890623,
255
+ "step": 850
256
+ },
257
+ {
258
+ "epoch": 0.8421559191530318,
259
+ "grad_norm": 4.575445652008057,
260
+ "learning_rate": 0.00019846932632914733,
261
+ "loss": 3.0112255859375,
262
+ "step": 875
263
+ },
264
+ {
265
+ "epoch": 0.8662175168431184,
266
+ "grad_norm": 4.955294609069824,
267
+ "learning_rate": 0.0001983305050578386,
268
+ "loss": 3.0229232788085936,
269
+ "step": 900
270
+ },
271
+ {
272
+ "epoch": 0.890279114533205,
273
+ "grad_norm": 4.7536773681640625,
274
+ "learning_rate": 0.00019818571182328003,
275
+ "loss": 3.0158026123046877,
276
+ "step": 925
277
+ },
278
+ {
279
+ "epoch": 0.9143407122232916,
280
+ "grad_norm": 4.718550205230713,
281
+ "learning_rate": 0.00019803495541928262,
282
+ "loss": 2.983572998046875,
283
+ "step": 950
284
+ },
285
+ {
286
+ "epoch": 0.9384023099133783,
287
+ "grad_norm": 4.429309844970703,
288
+ "learning_rate": 0.000197878245001822,
289
+ "loss": 2.9714495849609377,
290
+ "step": 975
291
+ },
292
+ {
293
+ "epoch": 0.9624639076034649,
294
+ "grad_norm": 4.627259254455566,
295
+ "learning_rate": 0.00019771559008848223,
296
+ "loss": 2.961048889160156,
297
+ "step": 1000
298
+ },
299
+ {
300
+ "epoch": 0.9865255052935515,
301
+ "grad_norm": 5.209016799926758,
302
+ "learning_rate": 0.00019754700055787797,
303
+ "loss": 2.9716583251953126,
304
+ "step": 1025
305
+ },
306
+ {
307
+ "epoch": 1.0,
308
+ "eval_loss": 2.934025287628174,
309
+ "eval_runtime": 12.8238,
310
+ "eval_samples_per_second": 116.97,
311
+ "eval_steps_per_second": 1.872,
312
+ "step": 1039
313
+ }
314
+ ],
315
+ "logging_steps": 25,
316
+ "max_steps": 10390,
317
+ "num_input_tokens_seen": 0,
318
+ "num_train_epochs": 10,
319
+ "save_steps": 500,
320
+ "stateful_callbacks": {
321
+ "TrainerControl": {
322
+ "args": {
323
+ "should_epoch_stop": false,
324
+ "should_evaluate": false,
325
+ "should_log": false,
326
+ "should_save": true,
327
+ "should_training_stop": false
328
+ },
329
+ "attributes": {}
330
+ }
331
+ },
332
+ "total_flos": 0.0,
333
+ "train_batch_size": 8,
334
+ "trial_name": null,
335
+ "trial_params": null
336
+ }