devoppro commited on
Commit
1114313
·
verified ·
1 Parent(s): a451e19

Training in progress, step 500, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:68d10d6c77661a22b49f687a60f7c0595cbbc5bdafd8820a654f5e22542b0e33
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:224bbc547320cbc7a57a274bef9f525f2e4be737031df14b601959f9380cb929
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:34eb8fd94484f12828f3079ddb1062719a88baf43040ad9bfe0255470c6e514a
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f014f1f8cf2eb9a4ff44ad059ad7d7ab8356457d91805882d791c32ab849f992
3
  size 2471218763
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:526edfc17cc77c8e1a37fdb81a9a703de4513b58ba4b0dbf11906a54aab78396
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:68b03f74e10591ff22477365642275703166a34b306471da2a9df3e87497356f
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c692682bd43a00387bd41dcd12b266427cdc8a4657c91543547ea08ad6b10378
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0471ce3745a36e49a3e0d19eac020ddc42dff8b92447d06fa38e6a682e228b19
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.024,
6
  "eval_steps": 500,
7
- "global_step": 1200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -291,563 +291,73 @@
291
  },
292
  {
293
  "epoch": 0.0082,
294
- "grad_norm": 28.452800750732422,
295
  "learning_rate": 0.00029814228456913827,
296
- "loss": 102.99000244140625,
297
  "step": 410
298
  },
299
  {
300
  "epoch": 0.0084,
301
- "grad_norm": 12.671554565429688,
302
  "learning_rate": 0.00029808216432865726,
303
- "loss": 83.27249755859376,
304
  "step": 420
305
  },
306
  {
307
  "epoch": 0.0086,
308
- "grad_norm": 7.306222438812256,
309
  "learning_rate": 0.0002980220440881763,
310
- "loss": 73.54000854492188,
311
  "step": 430
312
  },
313
  {
314
  "epoch": 0.0088,
315
- "grad_norm": 11.418675422668457,
316
  "learning_rate": 0.00029796192384769535,
317
- "loss": 77.23857421875,
318
  "step": 440
319
  },
320
  {
321
  "epoch": 0.009,
322
- "grad_norm": 8.655750274658203,
323
  "learning_rate": 0.0002979018036072144,
324
- "loss": 77.81395263671875,
325
  "step": 450
326
  },
327
  {
328
  "epoch": 0.0092,
329
- "grad_norm": 11.155708312988281,
330
  "learning_rate": 0.00029784168336673345,
331
- "loss": 78.6078125,
332
  "step": 460
333
  },
334
  {
335
  "epoch": 0.0094,
336
- "grad_norm": 9.934618949890137,
337
  "learning_rate": 0.0002977815631262525,
338
- "loss": 71.70912475585938,
339
  "step": 470
340
  },
341
  {
342
  "epoch": 0.0096,
343
- "grad_norm": 7.762029647827148,
344
  "learning_rate": 0.00029772144288577155,
345
- "loss": 73.4228515625,
346
  "step": 480
347
  },
348
  {
349
  "epoch": 0.0098,
350
- "grad_norm": 9.61263370513916,
351
  "learning_rate": 0.00029766132264529054,
352
- "loss": 75.76692504882813,
353
  "step": 490
354
  },
355
  {
356
  "epoch": 0.01,
357
- "grad_norm": 11.183115005493164,
358
  "learning_rate": 0.0002976012024048096,
359
- "loss": 78.96190795898437,
360
  "step": 500
361
- },
362
- {
363
- "epoch": 0.0102,
364
- "grad_norm": 9.10632610321045,
365
- "learning_rate": 0.00029754108216432863,
366
- "loss": 78.6921875,
367
- "step": 510
368
- },
369
- {
370
- "epoch": 0.0104,
371
- "grad_norm": 9.557915687561035,
372
- "learning_rate": 0.0002974809619238477,
373
- "loss": 71.66650390625,
374
- "step": 520
375
- },
376
- {
377
- "epoch": 0.0106,
378
- "grad_norm": 8.777642250061035,
379
- "learning_rate": 0.0002974208416833667,
380
- "loss": 70.4987548828125,
381
- "step": 530
382
- },
383
- {
384
- "epoch": 0.0108,
385
- "grad_norm": 13.94754409790039,
386
- "learning_rate": 0.0002973607214428857,
387
- "loss": 72.32754516601562,
388
- "step": 540
389
- },
390
- {
391
- "epoch": 0.011,
392
- "grad_norm": 8.856101989746094,
393
- "learning_rate": 0.0002973006012024048,
394
- "loss": 70.21427612304687,
395
- "step": 550
396
- },
397
- {
398
- "epoch": 0.0112,
399
- "grad_norm": 9.487411499023438,
400
- "learning_rate": 0.0002972404809619238,
401
- "loss": 76.47840576171875,
402
- "step": 560
403
- },
404
- {
405
- "epoch": 0.0114,
406
- "grad_norm": 8.979376792907715,
407
- "learning_rate": 0.00029718036072144287,
408
- "loss": 77.88162231445312,
409
- "step": 570
410
- },
411
- {
412
- "epoch": 0.0116,
413
- "grad_norm": 9.910178184509277,
414
- "learning_rate": 0.0002971202404809619,
415
- "loss": 77.7337646484375,
416
- "step": 580
417
- },
418
- {
419
- "epoch": 0.0118,
420
- "grad_norm": 5.745482444763184,
421
- "learning_rate": 0.00029706012024048096,
422
- "loss": 71.672021484375,
423
- "step": 590
424
- },
425
- {
426
- "epoch": 0.012,
427
- "grad_norm": 9.654470443725586,
428
- "learning_rate": 0.00029699999999999996,
429
- "loss": 74.40623168945312,
430
- "step": 600
431
- },
432
- {
433
- "epoch": 0.0122,
434
- "grad_norm": 8.696991920471191,
435
- "learning_rate": 0.000296939879759519,
436
- "loss": 74.297412109375,
437
- "step": 610
438
- },
439
- {
440
- "epoch": 0.0124,
441
- "grad_norm": 11.501073837280273,
442
- "learning_rate": 0.00029687975951903805,
443
- "loss": 70.46600952148438,
444
- "step": 620
445
- },
446
- {
447
- "epoch": 0.0126,
448
- "grad_norm": 6.01923131942749,
449
- "learning_rate": 0.0002968196392785571,
450
- "loss": 68.65836181640626,
451
- "step": 630
452
- },
453
- {
454
- "epoch": 0.0128,
455
- "grad_norm": 7.358704090118408,
456
- "learning_rate": 0.0002967595190380761,
457
- "loss": 69.74998779296875,
458
- "step": 640
459
- },
460
- {
461
- "epoch": 0.013,
462
- "grad_norm": 9.420614242553711,
463
- "learning_rate": 0.0002966993987975952,
464
- "loss": 76.11289672851562,
465
- "step": 650
466
- },
467
- {
468
- "epoch": 0.0132,
469
- "grad_norm": 8.627158164978027,
470
- "learning_rate": 0.0002966392785571142,
471
- "loss": 77.244384765625,
472
- "step": 660
473
- },
474
- {
475
- "epoch": 0.0134,
476
- "grad_norm": 8.27780818939209,
477
- "learning_rate": 0.00029657915831663323,
478
- "loss": 72.63546142578124,
479
- "step": 670
480
- },
481
- {
482
- "epoch": 0.0136,
483
- "grad_norm": 8.197218894958496,
484
- "learning_rate": 0.0002965190380761523,
485
- "loss": 75.8551513671875,
486
- "step": 680
487
- },
488
- {
489
- "epoch": 0.0138,
490
- "grad_norm": 9.352670669555664,
491
- "learning_rate": 0.00029645891783567133,
492
- "loss": 70.17238159179688,
493
- "step": 690
494
- },
495
- {
496
- "epoch": 0.014,
497
- "grad_norm": 10.278020858764648,
498
- "learning_rate": 0.0002963987975951904,
499
- "loss": 72.43441772460938,
500
- "step": 700
501
- },
502
- {
503
- "epoch": 0.0142,
504
- "grad_norm": 8.186079978942871,
505
- "learning_rate": 0.00029633867735470937,
506
- "loss": 69.34667358398437,
507
- "step": 710
508
- },
509
- {
510
- "epoch": 0.0144,
511
- "grad_norm": 8.636640548706055,
512
- "learning_rate": 0.0002962785571142284,
513
- "loss": 71.52565307617188,
514
- "step": 720
515
- },
516
- {
517
- "epoch": 0.0146,
518
- "grad_norm": 8.615554809570312,
519
- "learning_rate": 0.00029621843687374747,
520
- "loss": 66.20573120117187,
521
- "step": 730
522
- },
523
- {
524
- "epoch": 0.0148,
525
- "grad_norm": 9.266575813293457,
526
- "learning_rate": 0.0002961583166332665,
527
- "loss": 66.66327514648438,
528
- "step": 740
529
- },
530
- {
531
- "epoch": 0.015,
532
- "grad_norm": 6.957458972930908,
533
- "learning_rate": 0.00029609819639278556,
534
- "loss": 69.37461547851562,
535
- "step": 750
536
- },
537
- {
538
- "epoch": 0.0152,
539
- "grad_norm": 10.149578094482422,
540
- "learning_rate": 0.0002960380761523046,
541
- "loss": 75.38046264648438,
542
- "step": 760
543
- },
544
- {
545
- "epoch": 0.0154,
546
- "grad_norm": 8.38595962524414,
547
- "learning_rate": 0.0002959779559118236,
548
- "loss": 72.13077392578126,
549
- "step": 770
550
- },
551
- {
552
- "epoch": 0.0156,
553
- "grad_norm": 9.602531433105469,
554
- "learning_rate": 0.00029591783567134265,
555
- "loss": 73.194677734375,
556
- "step": 780
557
- },
558
- {
559
- "epoch": 0.0158,
560
- "grad_norm": 11.321459770202637,
561
- "learning_rate": 0.0002958577154308617,
562
- "loss": 67.77056884765625,
563
- "step": 790
564
- },
565
- {
566
- "epoch": 0.016,
567
- "grad_norm": 9.131096839904785,
568
- "learning_rate": 0.00029579759519038075,
569
- "loss": 68.58425903320312,
570
- "step": 800
571
- },
572
- {
573
- "epoch": 0.0162,
574
- "grad_norm": 9.43658447265625,
575
- "learning_rate": 0.00029573747494989974,
576
- "loss": 71.80587158203124,
577
- "step": 810
578
- },
579
- {
580
- "epoch": 0.0164,
581
- "grad_norm": 7.688558578491211,
582
- "learning_rate": 0.0002956773547094188,
583
- "loss": 73.39664916992187,
584
- "step": 820
585
- },
586
- {
587
- "epoch": 0.0166,
588
- "grad_norm": 9.346514701843262,
589
- "learning_rate": 0.00029561723446893784,
590
- "loss": 73.8046142578125,
591
- "step": 830
592
- },
593
- {
594
- "epoch": 0.0168,
595
- "grad_norm": 7.071310997009277,
596
- "learning_rate": 0.0002955571142284569,
597
- "loss": 69.04111938476562,
598
- "step": 840
599
- },
600
- {
601
- "epoch": 0.017,
602
- "grad_norm": 10.125195503234863,
603
- "learning_rate": 0.00029549699398797593,
604
- "loss": 69.92824096679688,
605
- "step": 850
606
- },
607
- {
608
- "epoch": 0.0172,
609
- "grad_norm": 8.768003463745117,
610
- "learning_rate": 0.000295436873747495,
611
- "loss": 74.65200805664062,
612
- "step": 860
613
- },
614
- {
615
- "epoch": 0.0174,
616
- "grad_norm": 8.981837272644043,
617
- "learning_rate": 0.000295376753507014,
618
- "loss": 72.95183715820312,
619
- "step": 870
620
- },
621
- {
622
- "epoch": 0.0176,
623
- "grad_norm": 9.743247032165527,
624
- "learning_rate": 0.000295316633266533,
625
- "loss": 68.19429321289063,
626
- "step": 880
627
- },
628
- {
629
- "epoch": 0.0178,
630
- "grad_norm": 7.722388744354248,
631
- "learning_rate": 0.00029525651302605207,
632
- "loss": 71.07997436523438,
633
- "step": 890
634
- },
635
- {
636
- "epoch": 0.018,
637
- "grad_norm": 7.273824214935303,
638
- "learning_rate": 0.0002951963927855711,
639
- "loss": 71.75267333984375,
640
- "step": 900
641
- },
642
- {
643
- "epoch": 0.0182,
644
- "grad_norm": 7.856244087219238,
645
- "learning_rate": 0.00029513627254509016,
646
- "loss": 73.38182983398437,
647
- "step": 910
648
- },
649
- {
650
- "epoch": 0.0184,
651
- "grad_norm": 7.895908355712891,
652
- "learning_rate": 0.00029507615230460916,
653
- "loss": 70.02378540039062,
654
- "step": 920
655
- },
656
- {
657
- "epoch": 0.0186,
658
- "grad_norm": 9.500384330749512,
659
- "learning_rate": 0.0002950160320641282,
660
- "loss": 69.76512451171875,
661
- "step": 930
662
- },
663
- {
664
- "epoch": 0.0188,
665
- "grad_norm": 8.77175235748291,
666
- "learning_rate": 0.0002949559118236473,
667
- "loss": 69.60545654296875,
668
- "step": 940
669
- },
670
- {
671
- "epoch": 0.019,
672
- "grad_norm": 9.102203369140625,
673
- "learning_rate": 0.0002948957915831663,
674
- "loss": 71.84405517578125,
675
- "step": 950
676
- },
677
- {
678
- "epoch": 0.0192,
679
- "grad_norm": 8.29865837097168,
680
- "learning_rate": 0.00029483567134268535,
681
- "loss": 67.64026489257813,
682
- "step": 960
683
- },
684
- {
685
- "epoch": 0.0194,
686
- "grad_norm": 6.868947505950928,
687
- "learning_rate": 0.0002947755511022044,
688
- "loss": 65.34761962890624,
689
- "step": 970
690
- },
691
- {
692
- "epoch": 0.0196,
693
- "grad_norm": 9.127030372619629,
694
- "learning_rate": 0.00029471543086172344,
695
- "loss": 69.42554321289063,
696
- "step": 980
697
- },
698
- {
699
- "epoch": 0.0198,
700
- "grad_norm": 7.779689311981201,
701
- "learning_rate": 0.00029465531062124244,
702
- "loss": 72.64813842773438,
703
- "step": 990
704
- },
705
- {
706
- "epoch": 0.02,
707
- "grad_norm": 9.583319664001465,
708
- "learning_rate": 0.0002945951903807615,
709
- "loss": 71.14970703125,
710
- "step": 1000
711
- },
712
- {
713
- "epoch": 0.0202,
714
- "grad_norm": 9.4298677444458,
715
- "learning_rate": 0.00029453507014028053,
716
- "loss": 68.8579345703125,
717
- "step": 1010
718
- },
719
- {
720
- "epoch": 0.0204,
721
- "grad_norm": 11.69355583190918,
722
- "learning_rate": 0.0002944749498997996,
723
- "loss": 66.29926147460938,
724
- "step": 1020
725
- },
726
- {
727
- "epoch": 0.0206,
728
- "grad_norm": 9.638633728027344,
729
- "learning_rate": 0.00029441482965931857,
730
- "loss": 69.4812744140625,
731
- "step": 1030
732
- },
733
- {
734
- "epoch": 0.0208,
735
- "grad_norm": 7.3949384689331055,
736
- "learning_rate": 0.0002943547094188377,
737
- "loss": 69.908984375,
738
- "step": 1040
739
- },
740
- {
741
- "epoch": 0.021,
742
- "grad_norm": 9.138998031616211,
743
- "learning_rate": 0.0002942945891783567,
744
- "loss": 68.20003662109374,
745
- "step": 1050
746
- },
747
- {
748
- "epoch": 0.0212,
749
- "grad_norm": 9.650430679321289,
750
- "learning_rate": 0.0002942344689378757,
751
- "loss": 64.76406860351562,
752
- "step": 1060
753
- },
754
- {
755
- "epoch": 0.0214,
756
- "grad_norm": 8.263360977172852,
757
- "learning_rate": 0.00029417434869739476,
758
- "loss": 73.39156494140624,
759
- "step": 1070
760
- },
761
- {
762
- "epoch": 0.0216,
763
- "grad_norm": 8.509857177734375,
764
- "learning_rate": 0.0002941142284569138,
765
- "loss": 65.55498657226562,
766
- "step": 1080
767
- },
768
- {
769
- "epoch": 0.0218,
770
- "grad_norm": 10.162703514099121,
771
- "learning_rate": 0.00029405410821643286,
772
- "loss": 71.94191284179688,
773
- "step": 1090
774
- },
775
- {
776
- "epoch": 0.022,
777
- "grad_norm": 10.938014030456543,
778
- "learning_rate": 0.00029399398797595185,
779
- "loss": 68.12153930664063,
780
- "step": 1100
781
- },
782
- {
783
- "epoch": 0.0222,
784
- "grad_norm": 9.84087085723877,
785
- "learning_rate": 0.0002939338677354709,
786
- "loss": 73.03311767578126,
787
- "step": 1110
788
- },
789
- {
790
- "epoch": 0.0224,
791
- "grad_norm": 8.5576171875,
792
- "learning_rate": 0.00029387374749498995,
793
- "loss": 67.728662109375,
794
- "step": 1120
795
- },
796
- {
797
- "epoch": 0.0226,
798
- "grad_norm": 6.275845050811768,
799
- "learning_rate": 0.000293813627254509,
800
- "loss": 59.14338989257813,
801
- "step": 1130
802
- },
803
- {
804
- "epoch": 0.0228,
805
- "grad_norm": 10.62365436553955,
806
- "learning_rate": 0.00029375350701402804,
807
- "loss": 68.015625,
808
- "step": 1140
809
- },
810
- {
811
- "epoch": 0.023,
812
- "grad_norm": 10.124994277954102,
813
- "learning_rate": 0.0002936933867735471,
814
- "loss": 67.996630859375,
815
- "step": 1150
816
- },
817
- {
818
- "epoch": 0.0232,
819
- "grad_norm": 9.192570686340332,
820
- "learning_rate": 0.00029363326653306614,
821
- "loss": 69.9107177734375,
822
- "step": 1160
823
- },
824
- {
825
- "epoch": 0.0234,
826
- "grad_norm": 11.012198448181152,
827
- "learning_rate": 0.00029357314629258513,
828
- "loss": 72.0302978515625,
829
- "step": 1170
830
- },
831
- {
832
- "epoch": 0.0236,
833
- "grad_norm": 8.58188533782959,
834
- "learning_rate": 0.0002935130260521042,
835
- "loss": 73.009423828125,
836
- "step": 1180
837
- },
838
- {
839
- "epoch": 0.0238,
840
- "grad_norm": 11.214829444885254,
841
- "learning_rate": 0.0002934529058116232,
842
- "loss": 69.40020141601562,
843
- "step": 1190
844
- },
845
- {
846
- "epoch": 0.024,
847
- "grad_norm": 8.615409851074219,
848
- "learning_rate": 0.0002933927855711423,
849
- "loss": 73.381103515625,
850
- "step": 1200
851
  }
852
  ],
853
  "logging_steps": 10,
@@ -867,7 +377,7 @@
867
  "attributes": {}
868
  }
869
  },
870
- "total_flos": 1.253963749312512e+16,
871
  "train_batch_size": 2,
872
  "trial_name": null,
873
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.01,
6
  "eval_steps": 500,
7
+ "global_step": 500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
291
  },
292
  {
293
  "epoch": 0.0082,
294
+ "grad_norm": 28.707782745361328,
295
  "learning_rate": 0.00029814228456913827,
296
+ "loss": 103.1381591796875,
297
  "step": 410
298
  },
299
  {
300
  "epoch": 0.0084,
301
+ "grad_norm": 12.400252342224121,
302
  "learning_rate": 0.00029808216432865726,
303
+ "loss": 83.130029296875,
304
  "step": 420
305
  },
306
  {
307
  "epoch": 0.0086,
308
+ "grad_norm": 7.325169563293457,
309
  "learning_rate": 0.0002980220440881763,
310
+ "loss": 73.54479370117187,
311
  "step": 430
312
  },
313
  {
314
  "epoch": 0.0088,
315
+ "grad_norm": 11.358526229858398,
316
  "learning_rate": 0.00029796192384769535,
317
+ "loss": 77.26640625,
318
  "step": 440
319
  },
320
  {
321
  "epoch": 0.009,
322
+ "grad_norm": 8.655184745788574,
323
  "learning_rate": 0.0002979018036072144,
324
+ "loss": 77.83007202148437,
325
  "step": 450
326
  },
327
  {
328
  "epoch": 0.0092,
329
+ "grad_norm": 10.999149322509766,
330
  "learning_rate": 0.00029784168336673345,
331
+ "loss": 78.61140747070313,
332
  "step": 460
333
  },
334
  {
335
  "epoch": 0.0094,
336
+ "grad_norm": 9.994011878967285,
337
  "learning_rate": 0.0002977815631262525,
338
+ "loss": 71.70507202148437,
339
  "step": 470
340
  },
341
  {
342
  "epoch": 0.0096,
343
+ "grad_norm": 7.690250873565674,
344
  "learning_rate": 0.00029772144288577155,
345
+ "loss": 73.41466674804687,
346
  "step": 480
347
  },
348
  {
349
  "epoch": 0.0098,
350
+ "grad_norm": 9.543883323669434,
351
  "learning_rate": 0.00029766132264529054,
352
+ "loss": 75.77626953125,
353
  "step": 490
354
  },
355
  {
356
  "epoch": 0.01,
357
+ "grad_norm": 11.137728691101074,
358
  "learning_rate": 0.0002976012024048096,
359
+ "loss": 78.9918212890625,
360
  "step": 500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
361
  }
362
  ],
363
  "logging_steps": 10,
 
377
  "attributes": {}
378
  }
379
  },
380
+ "total_flos": 8161860202859520.0,
381
  "train_batch_size": 2,
382
  "trial_name": null,
383
  "trial_params": null