rickysk commited on
Commit
f8c59ac
1 Parent(s): 1978f69

End of training

Browse files
Files changed (3) hide show
  1. all_results.json +8 -0
  2. test_results.json +8 -0
  3. trainer_state.json +907 -0
all_results.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "epoch": 15.06,
3
+ "eval_accuracy": 0.896774193548387,
4
+ "eval_loss": 0.3229670226573944,
5
+ "eval_runtime": 15.21,
6
+ "eval_samples_per_second": 10.191,
7
+ "eval_steps_per_second": 2.564
8
+ }
test_results.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "epoch": 15.06,
3
+ "eval_accuracy": 0.896774193548387,
4
+ "eval_loss": 0.3229670226573944,
5
+ "eval_runtime": 15.21,
6
+ "eval_samples_per_second": 10.191,
7
+ "eval_steps_per_second": 2.564
8
+ }
trainer_state.json ADDED
@@ -0,0 +1,907 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": 0.9857142857142858,
3
+ "best_model_checkpoint": "videomae-base-finetuned-ucf101-subset/checkpoint-975",
4
+ "epoch": 15.0625,
5
+ "global_step": 1200,
6
+ "is_hyper_param_search": false,
7
+ "is_local_process_zero": true,
8
+ "is_world_process_zero": true,
9
+ "log_history": [
10
+ {
11
+ "epoch": 0.01,
12
+ "learning_rate": 4.166666666666667e-06,
13
+ "loss": 2.4162,
14
+ "step": 10
15
+ },
16
+ {
17
+ "epoch": 0.02,
18
+ "learning_rate": 8.333333333333334e-06,
19
+ "loss": 2.3905,
20
+ "step": 20
21
+ },
22
+ {
23
+ "epoch": 0.03,
24
+ "learning_rate": 1.25e-05,
25
+ "loss": 2.3538,
26
+ "step": 30
27
+ },
28
+ {
29
+ "epoch": 0.03,
30
+ "learning_rate": 1.6666666666666667e-05,
31
+ "loss": 2.3149,
32
+ "step": 40
33
+ },
34
+ {
35
+ "epoch": 0.04,
36
+ "learning_rate": 2.0833333333333336e-05,
37
+ "loss": 2.2402,
38
+ "step": 50
39
+ },
40
+ {
41
+ "epoch": 0.05,
42
+ "learning_rate": 2.5e-05,
43
+ "loss": 2.1902,
44
+ "step": 60
45
+ },
46
+ {
47
+ "epoch": 0.06,
48
+ "learning_rate": 2.916666666666667e-05,
49
+ "loss": 2.1382,
50
+ "step": 70
51
+ },
52
+ {
53
+ "epoch": 0.06,
54
+ "eval_accuracy": 0.2571428571428571,
55
+ "eval_loss": 2.105558156967163,
56
+ "eval_runtime": 6.8406,
57
+ "eval_samples_per_second": 10.233,
58
+ "eval_steps_per_second": 2.631,
59
+ "step": 75
60
+ },
61
+ {
62
+ "epoch": 1.0,
63
+ "learning_rate": 3.3333333333333335e-05,
64
+ "loss": 2.1216,
65
+ "step": 80
66
+ },
67
+ {
68
+ "epoch": 1.01,
69
+ "learning_rate": 3.7500000000000003e-05,
70
+ "loss": 1.8205,
71
+ "step": 90
72
+ },
73
+ {
74
+ "epoch": 1.02,
75
+ "learning_rate": 4.166666666666667e-05,
76
+ "loss": 1.7791,
77
+ "step": 100
78
+ },
79
+ {
80
+ "epoch": 1.03,
81
+ "learning_rate": 4.5833333333333334e-05,
82
+ "loss": 1.4561,
83
+ "step": 110
84
+ },
85
+ {
86
+ "epoch": 1.04,
87
+ "learning_rate": 5e-05,
88
+ "loss": 1.3894,
89
+ "step": 120
90
+ },
91
+ {
92
+ "epoch": 1.05,
93
+ "learning_rate": 4.9537037037037035e-05,
94
+ "loss": 0.9883,
95
+ "step": 130
96
+ },
97
+ {
98
+ "epoch": 1.05,
99
+ "learning_rate": 4.9074074074074075e-05,
100
+ "loss": 0.9602,
101
+ "step": 140
102
+ },
103
+ {
104
+ "epoch": 1.06,
105
+ "learning_rate": 4.8611111111111115e-05,
106
+ "loss": 0.8185,
107
+ "step": 150
108
+ },
109
+ {
110
+ "epoch": 1.06,
111
+ "eval_accuracy": 0.8,
112
+ "eval_loss": 0.6269657611846924,
113
+ "eval_runtime": 6.7548,
114
+ "eval_samples_per_second": 10.363,
115
+ "eval_steps_per_second": 2.665,
116
+ "step": 150
117
+ },
118
+ {
119
+ "epoch": 2.01,
120
+ "learning_rate": 4.814814814814815e-05,
121
+ "loss": 0.6751,
122
+ "step": 160
123
+ },
124
+ {
125
+ "epoch": 2.02,
126
+ "learning_rate": 4.768518518518519e-05,
127
+ "loss": 0.8774,
128
+ "step": 170
129
+ },
130
+ {
131
+ "epoch": 2.02,
132
+ "learning_rate": 4.722222222222222e-05,
133
+ "loss": 0.5025,
134
+ "step": 180
135
+ },
136
+ {
137
+ "epoch": 2.03,
138
+ "learning_rate": 4.675925925925926e-05,
139
+ "loss": 0.7617,
140
+ "step": 190
141
+ },
142
+ {
143
+ "epoch": 2.04,
144
+ "learning_rate": 4.62962962962963e-05,
145
+ "loss": 0.5016,
146
+ "step": 200
147
+ },
148
+ {
149
+ "epoch": 2.05,
150
+ "learning_rate": 4.5833333333333334e-05,
151
+ "loss": 0.7137,
152
+ "step": 210
153
+ },
154
+ {
155
+ "epoch": 2.06,
156
+ "learning_rate": 4.5370370370370374e-05,
157
+ "loss": 0.5221,
158
+ "step": 220
159
+ },
160
+ {
161
+ "epoch": 2.06,
162
+ "eval_accuracy": 0.8,
163
+ "eval_loss": 0.4340825378894806,
164
+ "eval_runtime": 6.8255,
165
+ "eval_samples_per_second": 10.256,
166
+ "eval_steps_per_second": 2.637,
167
+ "step": 225
168
+ },
169
+ {
170
+ "epoch": 3.0,
171
+ "learning_rate": 4.490740740740741e-05,
172
+ "loss": 0.2861,
173
+ "step": 230
174
+ },
175
+ {
176
+ "epoch": 3.01,
177
+ "learning_rate": 4.4444444444444447e-05,
178
+ "loss": 0.3487,
179
+ "step": 240
180
+ },
181
+ {
182
+ "epoch": 3.02,
183
+ "learning_rate": 4.3981481481481486e-05,
184
+ "loss": 0.41,
185
+ "step": 250
186
+ },
187
+ {
188
+ "epoch": 3.03,
189
+ "learning_rate": 4.351851851851852e-05,
190
+ "loss": 0.2375,
191
+ "step": 260
192
+ },
193
+ {
194
+ "epoch": 3.04,
195
+ "learning_rate": 4.305555555555556e-05,
196
+ "loss": 0.3549,
197
+ "step": 270
198
+ },
199
+ {
200
+ "epoch": 3.05,
201
+ "learning_rate": 4.259259259259259e-05,
202
+ "loss": 0.338,
203
+ "step": 280
204
+ },
205
+ {
206
+ "epoch": 3.05,
207
+ "learning_rate": 4.212962962962963e-05,
208
+ "loss": 0.4893,
209
+ "step": 290
210
+ },
211
+ {
212
+ "epoch": 3.06,
213
+ "learning_rate": 4.166666666666667e-05,
214
+ "loss": 0.1069,
215
+ "step": 300
216
+ },
217
+ {
218
+ "epoch": 3.06,
219
+ "eval_accuracy": 0.8714285714285714,
220
+ "eval_loss": 0.4389760196208954,
221
+ "eval_runtime": 6.8626,
222
+ "eval_samples_per_second": 10.2,
223
+ "eval_steps_per_second": 2.623,
224
+ "step": 300
225
+ },
226
+ {
227
+ "epoch": 4.01,
228
+ "learning_rate": 4.1203703703703705e-05,
229
+ "loss": 0.3017,
230
+ "step": 310
231
+ },
232
+ {
233
+ "epoch": 4.02,
234
+ "learning_rate": 4.074074074074074e-05,
235
+ "loss": 0.2006,
236
+ "step": 320
237
+ },
238
+ {
239
+ "epoch": 4.03,
240
+ "learning_rate": 4.027777777777778e-05,
241
+ "loss": 0.8863,
242
+ "step": 330
243
+ },
244
+ {
245
+ "epoch": 4.03,
246
+ "learning_rate": 3.981481481481482e-05,
247
+ "loss": 0.0937,
248
+ "step": 340
249
+ },
250
+ {
251
+ "epoch": 4.04,
252
+ "learning_rate": 3.935185185185186e-05,
253
+ "loss": 0.1638,
254
+ "step": 350
255
+ },
256
+ {
257
+ "epoch": 4.05,
258
+ "learning_rate": 3.888888888888889e-05,
259
+ "loss": 0.1285,
260
+ "step": 360
261
+ },
262
+ {
263
+ "epoch": 4.06,
264
+ "learning_rate": 3.8425925925925924e-05,
265
+ "loss": 0.0195,
266
+ "step": 370
267
+ },
268
+ {
269
+ "epoch": 4.06,
270
+ "eval_accuracy": 0.8571428571428571,
271
+ "eval_loss": 0.29379311203956604,
272
+ "eval_runtime": 6.6946,
273
+ "eval_samples_per_second": 10.456,
274
+ "eval_steps_per_second": 2.689,
275
+ "step": 375
276
+ },
277
+ {
278
+ "epoch": 5.0,
279
+ "learning_rate": 3.7962962962962964e-05,
280
+ "loss": 0.1345,
281
+ "step": 380
282
+ },
283
+ {
284
+ "epoch": 5.01,
285
+ "learning_rate": 3.7500000000000003e-05,
286
+ "loss": 0.0226,
287
+ "step": 390
288
+ },
289
+ {
290
+ "epoch": 5.02,
291
+ "learning_rate": 3.7037037037037037e-05,
292
+ "loss": 0.0097,
293
+ "step": 400
294
+ },
295
+ {
296
+ "epoch": 5.03,
297
+ "learning_rate": 3.6574074074074076e-05,
298
+ "loss": 0.3541,
299
+ "step": 410
300
+ },
301
+ {
302
+ "epoch": 5.04,
303
+ "learning_rate": 3.611111111111111e-05,
304
+ "loss": 0.2285,
305
+ "step": 420
306
+ },
307
+ {
308
+ "epoch": 5.05,
309
+ "learning_rate": 3.564814814814815e-05,
310
+ "loss": 0.2023,
311
+ "step": 430
312
+ },
313
+ {
314
+ "epoch": 5.05,
315
+ "learning_rate": 3.518518518518519e-05,
316
+ "loss": 0.0792,
317
+ "step": 440
318
+ },
319
+ {
320
+ "epoch": 5.06,
321
+ "learning_rate": 3.472222222222222e-05,
322
+ "loss": 0.0097,
323
+ "step": 450
324
+ },
325
+ {
326
+ "epoch": 5.06,
327
+ "eval_accuracy": 0.9,
328
+ "eval_loss": 0.21138475835323334,
329
+ "eval_runtime": 6.7377,
330
+ "eval_samples_per_second": 10.389,
331
+ "eval_steps_per_second": 2.672,
332
+ "step": 450
333
+ },
334
+ {
335
+ "epoch": 6.01,
336
+ "learning_rate": 3.425925925925926e-05,
337
+ "loss": 0.2802,
338
+ "step": 460
339
+ },
340
+ {
341
+ "epoch": 6.02,
342
+ "learning_rate": 3.3796296296296295e-05,
343
+ "loss": 0.2688,
344
+ "step": 470
345
+ },
346
+ {
347
+ "epoch": 6.03,
348
+ "learning_rate": 3.3333333333333335e-05,
349
+ "loss": 0.0271,
350
+ "step": 480
351
+ },
352
+ {
353
+ "epoch": 6.03,
354
+ "learning_rate": 3.2870370370370375e-05,
355
+ "loss": 0.0278,
356
+ "step": 490
357
+ },
358
+ {
359
+ "epoch": 6.04,
360
+ "learning_rate": 3.240740740740741e-05,
361
+ "loss": 0.0655,
362
+ "step": 500
363
+ },
364
+ {
365
+ "epoch": 6.05,
366
+ "learning_rate": 3.194444444444444e-05,
367
+ "loss": 0.063,
368
+ "step": 510
369
+ },
370
+ {
371
+ "epoch": 6.06,
372
+ "learning_rate": 3.148148148148148e-05,
373
+ "loss": 0.0076,
374
+ "step": 520
375
+ },
376
+ {
377
+ "epoch": 6.06,
378
+ "eval_accuracy": 0.9428571428571428,
379
+ "eval_loss": 0.1508892923593521,
380
+ "eval_runtime": 6.6721,
381
+ "eval_samples_per_second": 10.491,
382
+ "eval_steps_per_second": 2.698,
383
+ "step": 525
384
+ },
385
+ {
386
+ "epoch": 7.0,
387
+ "learning_rate": 3.101851851851852e-05,
388
+ "loss": 0.157,
389
+ "step": 530
390
+ },
391
+ {
392
+ "epoch": 7.01,
393
+ "learning_rate": 3.055555555555556e-05,
394
+ "loss": 0.0048,
395
+ "step": 540
396
+ },
397
+ {
398
+ "epoch": 7.02,
399
+ "learning_rate": 3.0092592592592593e-05,
400
+ "loss": 0.0622,
401
+ "step": 550
402
+ },
403
+ {
404
+ "epoch": 7.03,
405
+ "learning_rate": 2.962962962962963e-05,
406
+ "loss": 0.0213,
407
+ "step": 560
408
+ },
409
+ {
410
+ "epoch": 7.04,
411
+ "learning_rate": 2.916666666666667e-05,
412
+ "loss": 0.0034,
413
+ "step": 570
414
+ },
415
+ {
416
+ "epoch": 7.05,
417
+ "learning_rate": 2.8703703703703706e-05,
418
+ "loss": 0.0495,
419
+ "step": 580
420
+ },
421
+ {
422
+ "epoch": 7.05,
423
+ "learning_rate": 2.824074074074074e-05,
424
+ "loss": 0.0037,
425
+ "step": 590
426
+ },
427
+ {
428
+ "epoch": 7.06,
429
+ "learning_rate": 2.777777777777778e-05,
430
+ "loss": 0.1686,
431
+ "step": 600
432
+ },
433
+ {
434
+ "epoch": 7.06,
435
+ "eval_accuracy": 0.9571428571428572,
436
+ "eval_loss": 0.2527245581150055,
437
+ "eval_runtime": 6.9358,
438
+ "eval_samples_per_second": 10.093,
439
+ "eval_steps_per_second": 2.595,
440
+ "step": 600
441
+ },
442
+ {
443
+ "epoch": 8.01,
444
+ "learning_rate": 2.7314814814814816e-05,
445
+ "loss": 0.0044,
446
+ "step": 610
447
+ },
448
+ {
449
+ "epoch": 8.02,
450
+ "learning_rate": 2.6851851851851855e-05,
451
+ "loss": 0.0034,
452
+ "step": 620
453
+ },
454
+ {
455
+ "epoch": 8.03,
456
+ "learning_rate": 2.6388888888888892e-05,
457
+ "loss": 0.0038,
458
+ "step": 630
459
+ },
460
+ {
461
+ "epoch": 8.03,
462
+ "learning_rate": 2.5925925925925925e-05,
463
+ "loss": 0.0055,
464
+ "step": 640
465
+ },
466
+ {
467
+ "epoch": 8.04,
468
+ "learning_rate": 2.5462962962962965e-05,
469
+ "loss": 0.2185,
470
+ "step": 650
471
+ },
472
+ {
473
+ "epoch": 8.05,
474
+ "learning_rate": 2.5e-05,
475
+ "loss": 0.1292,
476
+ "step": 660
477
+ },
478
+ {
479
+ "epoch": 8.06,
480
+ "learning_rate": 2.4537037037037038e-05,
481
+ "loss": 0.0679,
482
+ "step": 670
483
+ },
484
+ {
485
+ "epoch": 8.06,
486
+ "eval_accuracy": 0.9714285714285714,
487
+ "eval_loss": 0.061473362147808075,
488
+ "eval_runtime": 6.7398,
489
+ "eval_samples_per_second": 10.386,
490
+ "eval_steps_per_second": 2.671,
491
+ "step": 675
492
+ },
493
+ {
494
+ "epoch": 9.0,
495
+ "learning_rate": 2.4074074074074074e-05,
496
+ "loss": 0.0031,
497
+ "step": 680
498
+ },
499
+ {
500
+ "epoch": 9.01,
501
+ "learning_rate": 2.361111111111111e-05,
502
+ "loss": 0.008,
503
+ "step": 690
504
+ },
505
+ {
506
+ "epoch": 9.02,
507
+ "learning_rate": 2.314814814814815e-05,
508
+ "loss": 0.0656,
509
+ "step": 700
510
+ },
511
+ {
512
+ "epoch": 9.03,
513
+ "learning_rate": 2.2685185185185187e-05,
514
+ "loss": 0.0026,
515
+ "step": 710
516
+ },
517
+ {
518
+ "epoch": 9.04,
519
+ "learning_rate": 2.2222222222222223e-05,
520
+ "loss": 0.0029,
521
+ "step": 720
522
+ },
523
+ {
524
+ "epoch": 9.05,
525
+ "learning_rate": 2.175925925925926e-05,
526
+ "loss": 0.0035,
527
+ "step": 730
528
+ },
529
+ {
530
+ "epoch": 9.05,
531
+ "learning_rate": 2.1296296296296296e-05,
532
+ "loss": 0.0029,
533
+ "step": 740
534
+ },
535
+ {
536
+ "epoch": 9.06,
537
+ "learning_rate": 2.0833333333333336e-05,
538
+ "loss": 0.0024,
539
+ "step": 750
540
+ },
541
+ {
542
+ "epoch": 9.06,
543
+ "eval_accuracy": 0.9428571428571428,
544
+ "eval_loss": 0.15893925726413727,
545
+ "eval_runtime": 6.754,
546
+ "eval_samples_per_second": 10.364,
547
+ "eval_steps_per_second": 2.665,
548
+ "step": 750
549
+ },
550
+ {
551
+ "epoch": 10.01,
552
+ "learning_rate": 2.037037037037037e-05,
553
+ "loss": 0.0027,
554
+ "step": 760
555
+ },
556
+ {
557
+ "epoch": 10.02,
558
+ "learning_rate": 1.990740740740741e-05,
559
+ "loss": 0.0024,
560
+ "step": 770
561
+ },
562
+ {
563
+ "epoch": 10.03,
564
+ "learning_rate": 1.9444444444444445e-05,
565
+ "loss": 0.0025,
566
+ "step": 780
567
+ },
568
+ {
569
+ "epoch": 10.03,
570
+ "learning_rate": 1.8981481481481482e-05,
571
+ "loss": 0.0023,
572
+ "step": 790
573
+ },
574
+ {
575
+ "epoch": 10.04,
576
+ "learning_rate": 1.8518518518518518e-05,
577
+ "loss": 0.0873,
578
+ "step": 800
579
+ },
580
+ {
581
+ "epoch": 10.05,
582
+ "learning_rate": 1.8055555555555555e-05,
583
+ "loss": 0.1121,
584
+ "step": 810
585
+ },
586
+ {
587
+ "epoch": 10.06,
588
+ "learning_rate": 1.7592592592592595e-05,
589
+ "loss": 0.1946,
590
+ "step": 820
591
+ },
592
+ {
593
+ "epoch": 10.06,
594
+ "eval_accuracy": 0.9,
595
+ "eval_loss": 0.40144404768943787,
596
+ "eval_runtime": 6.6936,
597
+ "eval_samples_per_second": 10.458,
598
+ "eval_steps_per_second": 2.689,
599
+ "step": 825
600
+ },
601
+ {
602
+ "epoch": 11.0,
603
+ "learning_rate": 1.712962962962963e-05,
604
+ "loss": 0.1594,
605
+ "step": 830
606
+ },
607
+ {
608
+ "epoch": 11.01,
609
+ "learning_rate": 1.6666666666666667e-05,
610
+ "loss": 0.003,
611
+ "step": 840
612
+ },
613
+ {
614
+ "epoch": 11.02,
615
+ "learning_rate": 1.6203703703703704e-05,
616
+ "loss": 0.0025,
617
+ "step": 850
618
+ },
619
+ {
620
+ "epoch": 11.03,
621
+ "learning_rate": 1.574074074074074e-05,
622
+ "loss": 0.0093,
623
+ "step": 860
624
+ },
625
+ {
626
+ "epoch": 11.04,
627
+ "learning_rate": 1.527777777777778e-05,
628
+ "loss": 0.0034,
629
+ "step": 870
630
+ },
631
+ {
632
+ "epoch": 11.05,
633
+ "learning_rate": 1.4814814814814815e-05,
634
+ "loss": 0.0251,
635
+ "step": 880
636
+ },
637
+ {
638
+ "epoch": 11.05,
639
+ "learning_rate": 1.4351851851851853e-05,
640
+ "loss": 0.1587,
641
+ "step": 890
642
+ },
643
+ {
644
+ "epoch": 11.06,
645
+ "learning_rate": 1.388888888888889e-05,
646
+ "loss": 0.154,
647
+ "step": 900
648
+ },
649
+ {
650
+ "epoch": 11.06,
651
+ "eval_accuracy": 0.9428571428571428,
652
+ "eval_loss": 0.1862240433692932,
653
+ "eval_runtime": 6.664,
654
+ "eval_samples_per_second": 10.504,
655
+ "eval_steps_per_second": 2.701,
656
+ "step": 900
657
+ },
658
+ {
659
+ "epoch": 12.01,
660
+ "learning_rate": 1.3425925925925928e-05,
661
+ "loss": 0.0023,
662
+ "step": 910
663
+ },
664
+ {
665
+ "epoch": 12.02,
666
+ "learning_rate": 1.2962962962962962e-05,
667
+ "loss": 0.0025,
668
+ "step": 920
669
+ },
670
+ {
671
+ "epoch": 12.03,
672
+ "learning_rate": 1.25e-05,
673
+ "loss": 0.0027,
674
+ "step": 930
675
+ },
676
+ {
677
+ "epoch": 12.03,
678
+ "learning_rate": 1.2037037037037037e-05,
679
+ "loss": 0.002,
680
+ "step": 940
681
+ },
682
+ {
683
+ "epoch": 12.04,
684
+ "learning_rate": 1.1574074074074075e-05,
685
+ "loss": 0.002,
686
+ "step": 950
687
+ },
688
+ {
689
+ "epoch": 12.05,
690
+ "learning_rate": 1.1111111111111112e-05,
691
+ "loss": 0.0025,
692
+ "step": 960
693
+ },
694
+ {
695
+ "epoch": 12.06,
696
+ "learning_rate": 1.0648148148148148e-05,
697
+ "loss": 0.0021,
698
+ "step": 970
699
+ },
700
+ {
701
+ "epoch": 12.06,
702
+ "eval_accuracy": 0.9857142857142858,
703
+ "eval_loss": 0.06825670599937439,
704
+ "eval_runtime": 6.9976,
705
+ "eval_samples_per_second": 10.003,
706
+ "eval_steps_per_second": 2.572,
707
+ "step": 975
708
+ },
709
+ {
710
+ "epoch": 13.0,
711
+ "learning_rate": 1.0185185185185185e-05,
712
+ "loss": 0.002,
713
+ "step": 980
714
+ },
715
+ {
716
+ "epoch": 13.01,
717
+ "learning_rate": 9.722222222222223e-06,
718
+ "loss": 0.0018,
719
+ "step": 990
720
+ },
721
+ {
722
+ "epoch": 13.02,
723
+ "learning_rate": 9.259259259259259e-06,
724
+ "loss": 0.002,
725
+ "step": 1000
726
+ },
727
+ {
728
+ "epoch": 13.03,
729
+ "learning_rate": 8.796296296296297e-06,
730
+ "loss": 0.0022,
731
+ "step": 1010
732
+ },
733
+ {
734
+ "epoch": 13.04,
735
+ "learning_rate": 8.333333333333334e-06,
736
+ "loss": 0.0021,
737
+ "step": 1020
738
+ },
739
+ {
740
+ "epoch": 13.05,
741
+ "learning_rate": 7.87037037037037e-06,
742
+ "loss": 0.0019,
743
+ "step": 1030
744
+ },
745
+ {
746
+ "epoch": 13.05,
747
+ "learning_rate": 7.4074074074074075e-06,
748
+ "loss": 0.0021,
749
+ "step": 1040
750
+ },
751
+ {
752
+ "epoch": 13.06,
753
+ "learning_rate": 6.944444444444445e-06,
754
+ "loss": 0.0019,
755
+ "step": 1050
756
+ },
757
+ {
758
+ "epoch": 13.06,
759
+ "eval_accuracy": 0.9857142857142858,
760
+ "eval_loss": 0.05410011112689972,
761
+ "eval_runtime": 6.8574,
762
+ "eval_samples_per_second": 10.208,
763
+ "eval_steps_per_second": 2.625,
764
+ "step": 1050
765
+ },
766
+ {
767
+ "epoch": 14.01,
768
+ "learning_rate": 6.481481481481481e-06,
769
+ "loss": 0.0019,
770
+ "step": 1060
771
+ },
772
+ {
773
+ "epoch": 14.02,
774
+ "learning_rate": 6.0185185185185185e-06,
775
+ "loss": 0.0019,
776
+ "step": 1070
777
+ },
778
+ {
779
+ "epoch": 14.03,
780
+ "learning_rate": 5.555555555555556e-06,
781
+ "loss": 0.002,
782
+ "step": 1080
783
+ },
784
+ {
785
+ "epoch": 14.03,
786
+ "learning_rate": 5.092592592592592e-06,
787
+ "loss": 0.0019,
788
+ "step": 1090
789
+ },
790
+ {
791
+ "epoch": 14.04,
792
+ "learning_rate": 4.6296296296296296e-06,
793
+ "loss": 0.002,
794
+ "step": 1100
795
+ },
796
+ {
797
+ "epoch": 14.05,
798
+ "learning_rate": 4.166666666666667e-06,
799
+ "loss": 0.0018,
800
+ "step": 1110
801
+ },
802
+ {
803
+ "epoch": 14.06,
804
+ "learning_rate": 3.7037037037037037e-06,
805
+ "loss": 0.002,
806
+ "step": 1120
807
+ },
808
+ {
809
+ "epoch": 14.06,
810
+ "eval_accuracy": 0.9857142857142858,
811
+ "eval_loss": 0.04728488251566887,
812
+ "eval_runtime": 7.0728,
813
+ "eval_samples_per_second": 9.897,
814
+ "eval_steps_per_second": 2.545,
815
+ "step": 1125
816
+ },
817
+ {
818
+ "epoch": 15.0,
819
+ "learning_rate": 3.2407407407407406e-06,
820
+ "loss": 0.0022,
821
+ "step": 1130
822
+ },
823
+ {
824
+ "epoch": 15.01,
825
+ "learning_rate": 2.777777777777778e-06,
826
+ "loss": 0.002,
827
+ "step": 1140
828
+ },
829
+ {
830
+ "epoch": 15.02,
831
+ "learning_rate": 2.3148148148148148e-06,
832
+ "loss": 0.0019,
833
+ "step": 1150
834
+ },
835
+ {
836
+ "epoch": 15.03,
837
+ "learning_rate": 1.8518518518518519e-06,
838
+ "loss": 0.002,
839
+ "step": 1160
840
+ },
841
+ {
842
+ "epoch": 15.04,
843
+ "learning_rate": 1.388888888888889e-06,
844
+ "loss": 0.0019,
845
+ "step": 1170
846
+ },
847
+ {
848
+ "epoch": 15.05,
849
+ "learning_rate": 9.259259259259259e-07,
850
+ "loss": 0.0018,
851
+ "step": 1180
852
+ },
853
+ {
854
+ "epoch": 15.05,
855
+ "learning_rate": 4.6296296296296297e-07,
856
+ "loss": 0.0019,
857
+ "step": 1190
858
+ },
859
+ {
860
+ "epoch": 15.06,
861
+ "learning_rate": 0.0,
862
+ "loss": 0.0018,
863
+ "step": 1200
864
+ },
865
+ {
866
+ "epoch": 15.06,
867
+ "eval_accuracy": 0.9857142857142858,
868
+ "eval_loss": 0.047529980540275574,
869
+ "eval_runtime": 6.8222,
870
+ "eval_samples_per_second": 10.261,
871
+ "eval_steps_per_second": 2.638,
872
+ "step": 1200
873
+ },
874
+ {
875
+ "epoch": 15.06,
876
+ "step": 1200,
877
+ "total_flos": 5.981536752500736e+18,
878
+ "train_loss": 0.3337698955554515,
879
+ "train_runtime": 902.4336,
880
+ "train_samples_per_second": 5.319,
881
+ "train_steps_per_second": 1.33
882
+ },
883
+ {
884
+ "epoch": 15.06,
885
+ "eval_accuracy": 0.896774193548387,
886
+ "eval_loss": 0.3229670226573944,
887
+ "eval_runtime": 15.168,
888
+ "eval_samples_per_second": 10.219,
889
+ "eval_steps_per_second": 2.571,
890
+ "step": 1200
891
+ },
892
+ {
893
+ "epoch": 15.06,
894
+ "eval_accuracy": 0.896774193548387,
895
+ "eval_loss": 0.3229670226573944,
896
+ "eval_runtime": 15.21,
897
+ "eval_samples_per_second": 10.191,
898
+ "eval_steps_per_second": 2.564,
899
+ "step": 1200
900
+ }
901
+ ],
902
+ "max_steps": 1200,
903
+ "num_train_epochs": 9223372036854775807,
904
+ "total_flos": 5.981536752500736e+18,
905
+ "trial_name": null,
906
+ "trial_params": null
907
+ }