devoppro commited on
Commit
a275287
·
verified ·
1 Parent(s): 0c28734

Training in progress, step 1100

Browse files
last-checkpoint/config.json CHANGED
@@ -2,13 +2,9 @@
2
  "architectures": [
3
  "ModernLLMForCausalLM"
4
  ],
5
- "auto_map": {
6
- "AutoConfig": "configuration_modern_llm.ModernLLMConfig",
7
- "AutoModelForCausalLM": "modeling_modern_llm.ModernLLMForCausalLM"
8
- },
9
- "bos_token_id": 151643,
10
  "dtype": "float32",
11
- "eos_token_id": 151643,
12
  "hidden_size": 768,
13
  "intermediate_size": 2048,
14
  "max_position_embeddings": 2048,
@@ -16,7 +12,7 @@
16
  "num_attention_heads": 12,
17
  "num_hidden_layers": 12,
18
  "num_key_value_heads": 4,
19
- "pad_token_id": 151643,
20
  "rms_norm_eps": 1e-06,
21
  "rope_theta": 1000000.0,
22
  "transformers_version": "5.15.1",
 
2
  "architectures": [
3
  "ModernLLMForCausalLM"
4
  ],
5
+ "bos_token_id": 1,
 
 
 
 
6
  "dtype": "float32",
7
+ "eos_token_id": 2,
8
  "hidden_size": 768,
9
  "intermediate_size": 2048,
10
  "max_position_embeddings": 2048,
 
12
  "num_attention_heads": 12,
13
  "num_hidden_layers": 12,
14
  "num_key_value_heads": 4,
15
+ "pad_token_id": 0,
16
  "rms_norm_eps": 1e-06,
17
  "rope_theta": 1000000.0,
18
  "transformers_version": "5.15.1",
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5a3ad66771ef5b8246c59acc3ea2a9104f762ec13d612e6e588d54d7c66836e3
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:224bbc547320cbc7a57a274bef9f525f2e4be737031df14b601959f9380cb929
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:87af1234b6b81d7ca503ab8b876edeee1375b5e739dae1d471d4d0c94c2cdc13
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f014f1f8cf2eb9a4ff44ad059ad7d7ab8356457d91805882d791c32ab849f992
3
  size 2471218763
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9aa7e5f7f3366b7db6f25a9e3fa739116674bd641ce589a5940ff73f382fec0c
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:68b03f74e10591ff22477365642275703166a34b306471da2a9df3e87497356f
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4835dfebca207db20e97bc75a0cdfb3c9040e987a06d53d84bc51055c8c81f56
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0471ce3745a36e49a3e0d19eac020ddc42dff8b92447d06fa38e6a682e228b19
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.02,
6
  "eval_steps": 500,
7
- "global_step": 1000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -358,356 +358,6 @@
358
  "learning_rate": 0.0002976012024048096,
359
  "loss": 78.9918212890625,
360
  "step": 500
361
- },
362
- {
363
- "epoch": 0.0102,
364
- "grad_norm": 15.467362403869629,
365
- "learning_rate": 0.00029754108216432863,
366
- "loss": 124.401513671875,
367
- "step": 510
368
- },
369
- {
370
- "epoch": 0.0104,
371
- "grad_norm": 13.991318702697754,
372
- "learning_rate": 0.0002974809619238477,
373
- "loss": 110.567431640625,
374
- "step": 520
375
- },
376
- {
377
- "epoch": 0.0106,
378
- "grad_norm": 17.036069869995117,
379
- "learning_rate": 0.0002974208416833667,
380
- "loss": 104.7319580078125,
381
- "step": 530
382
- },
383
- {
384
- "epoch": 0.0108,
385
- "grad_norm": 13.922304153442383,
386
- "learning_rate": 0.0002973607214428857,
387
- "loss": 100.63988647460937,
388
- "step": 540
389
- },
390
- {
391
- "epoch": 0.011,
392
- "grad_norm": 13.223531723022461,
393
- "learning_rate": 0.0002973006012024048,
394
- "loss": 99.96443481445313,
395
- "step": 550
396
- },
397
- {
398
- "epoch": 0.0112,
399
- "grad_norm": 14.53707504272461,
400
- "learning_rate": 0.0002972404809619238,
401
- "loss": 95.74251098632813,
402
- "step": 560
403
- },
404
- {
405
- "epoch": 0.0114,
406
- "grad_norm": 14.91770076751709,
407
- "learning_rate": 0.00029718036072144287,
408
- "loss": 94.661767578125,
409
- "step": 570
410
- },
411
- {
412
- "epoch": 0.0116,
413
- "grad_norm": 13.45614242553711,
414
- "learning_rate": 0.0002971202404809619,
415
- "loss": 90.941796875,
416
- "step": 580
417
- },
418
- {
419
- "epoch": 0.0118,
420
- "grad_norm": 13.695131301879883,
421
- "learning_rate": 0.00029706012024048096,
422
- "loss": 92.9096923828125,
423
- "step": 590
424
- },
425
- {
426
- "epoch": 0.012,
427
- "grad_norm": 16.406902313232422,
428
- "learning_rate": 0.00029699999999999996,
429
- "loss": 90.14566650390626,
430
- "step": 600
431
- },
432
- {
433
- "epoch": 0.0122,
434
- "grad_norm": 11.779080390930176,
435
- "learning_rate": 0.000296939879759519,
436
- "loss": 86.8746826171875,
437
- "step": 610
438
- },
439
- {
440
- "epoch": 0.0124,
441
- "grad_norm": 13.758212089538574,
442
- "learning_rate": 0.00029687975951903805,
443
- "loss": 87.3355224609375,
444
- "step": 620
445
- },
446
- {
447
- "epoch": 0.0126,
448
- "grad_norm": 12.833357810974121,
449
- "learning_rate": 0.0002968196392785571,
450
- "loss": 89.81148071289063,
451
- "step": 630
452
- },
453
- {
454
- "epoch": 0.0128,
455
- "grad_norm": 14.684094429016113,
456
- "learning_rate": 0.0002967595190380761,
457
- "loss": 89.44540405273438,
458
- "step": 640
459
- },
460
- {
461
- "epoch": 0.013,
462
- "grad_norm": 10.780083656311035,
463
- "learning_rate": 0.0002966993987975952,
464
- "loss": 88.07953491210938,
465
- "step": 650
466
- },
467
- {
468
- "epoch": 0.0132,
469
- "grad_norm": 11.908385276794434,
470
- "learning_rate": 0.0002966392785571142,
471
- "loss": 87.04054565429688,
472
- "step": 660
473
- },
474
- {
475
- "epoch": 0.0134,
476
- "grad_norm": 15.518954277038574,
477
- "learning_rate": 0.00029657915831663323,
478
- "loss": 85.9423583984375,
479
- "step": 670
480
- },
481
- {
482
- "epoch": 0.0136,
483
- "grad_norm": 10.570508003234863,
484
- "learning_rate": 0.0002965190380761523,
485
- "loss": 86.79979858398437,
486
- "step": 680
487
- },
488
- {
489
- "epoch": 0.0138,
490
- "grad_norm": 14.696393013000488,
491
- "learning_rate": 0.00029645891783567133,
492
- "loss": 79.76475219726562,
493
- "step": 690
494
- },
495
- {
496
- "epoch": 0.014,
497
- "grad_norm": 12.92392349243164,
498
- "learning_rate": 0.0002963987975951904,
499
- "loss": 87.0448974609375,
500
- "step": 700
501
- },
502
- {
503
- "epoch": 0.0142,
504
- "grad_norm": 13.465226173400879,
505
- "learning_rate": 0.00029633867735470937,
506
- "loss": 90.22592163085938,
507
- "step": 710
508
- },
509
- {
510
- "epoch": 0.0144,
511
- "grad_norm": 12.435464859008789,
512
- "learning_rate": 0.0002962785571142284,
513
- "loss": 86.32198486328124,
514
- "step": 720
515
- },
516
- {
517
- "epoch": 0.0146,
518
- "grad_norm": 12.340784072875977,
519
- "learning_rate": 0.00029621843687374747,
520
- "loss": 89.89630737304688,
521
- "step": 730
522
- },
523
- {
524
- "epoch": 0.0148,
525
- "grad_norm": 14.63569164276123,
526
- "learning_rate": 0.0002961583166332665,
527
- "loss": 86.31669921875,
528
- "step": 740
529
- },
530
- {
531
- "epoch": 0.015,
532
- "grad_norm": 10.646244049072266,
533
- "learning_rate": 0.00029609819639278556,
534
- "loss": 85.20569458007813,
535
- "step": 750
536
- },
537
- {
538
- "epoch": 0.0152,
539
- "grad_norm": 8.930062294006348,
540
- "learning_rate": 0.0002960380761523046,
541
- "loss": 86.44678955078125,
542
- "step": 760
543
- },
544
- {
545
- "epoch": 0.0154,
546
- "grad_norm": 12.228261947631836,
547
- "learning_rate": 0.0002959779559118236,
548
- "loss": 84.96986083984375,
549
- "step": 770
550
- },
551
- {
552
- "epoch": 0.0156,
553
- "grad_norm": 9.189741134643555,
554
- "learning_rate": 0.00029591783567134265,
555
- "loss": 86.03868408203125,
556
- "step": 780
557
- },
558
- {
559
- "epoch": 0.0158,
560
- "grad_norm": 12.331954956054688,
561
- "learning_rate": 0.0002958577154308617,
562
- "loss": 85.13253784179688,
563
- "step": 790
564
- },
565
- {
566
- "epoch": 0.016,
567
- "grad_norm": 14.345333099365234,
568
- "learning_rate": 0.00029579759519038075,
569
- "loss": 86.58322143554688,
570
- "step": 800
571
- },
572
- {
573
- "epoch": 0.0162,
574
- "grad_norm": 10.961579322814941,
575
- "learning_rate": 0.00029573747494989974,
576
- "loss": 84.83054809570312,
577
- "step": 810
578
- },
579
- {
580
- "epoch": 0.0164,
581
- "grad_norm": 11.578486442565918,
582
- "learning_rate": 0.0002956773547094188,
583
- "loss": 87.94888305664062,
584
- "step": 820
585
- },
586
- {
587
- "epoch": 0.0166,
588
- "grad_norm": 13.537806510925293,
589
- "learning_rate": 0.00029561723446893784,
590
- "loss": 87.522900390625,
591
- "step": 830
592
- },
593
- {
594
- "epoch": 0.0168,
595
- "grad_norm": 10.907570838928223,
596
- "learning_rate": 0.0002955571142284569,
597
- "loss": 84.80013427734374,
598
- "step": 840
599
- },
600
- {
601
- "epoch": 0.017,
602
- "grad_norm": 13.713836669921875,
603
- "learning_rate": 0.00029549699398797593,
604
- "loss": 86.79594116210937,
605
- "step": 850
606
- },
607
- {
608
- "epoch": 0.0172,
609
- "grad_norm": 12.571099281311035,
610
- "learning_rate": 0.000295436873747495,
611
- "loss": 84.85569458007812,
612
- "step": 860
613
- },
614
- {
615
- "epoch": 0.0174,
616
- "grad_norm": 9.433479309082031,
617
- "learning_rate": 0.000295376753507014,
618
- "loss": 85.20677490234375,
619
- "step": 870
620
- },
621
- {
622
- "epoch": 0.0176,
623
- "grad_norm": 12.85809326171875,
624
- "learning_rate": 0.000295316633266533,
625
- "loss": 79.00835571289062,
626
- "step": 880
627
- },
628
- {
629
- "epoch": 0.0178,
630
- "grad_norm": 11.075084686279297,
631
- "learning_rate": 0.00029525651302605207,
632
- "loss": 80.06212158203125,
633
- "step": 890
634
- },
635
- {
636
- "epoch": 0.018,
637
- "grad_norm": 14.64054012298584,
638
- "learning_rate": 0.0002951963927855711,
639
- "loss": 83.31229248046876,
640
- "step": 900
641
- },
642
- {
643
- "epoch": 0.0182,
644
- "grad_norm": 11.558423042297363,
645
- "learning_rate": 0.00029513627254509016,
646
- "loss": 78.82657470703126,
647
- "step": 910
648
- },
649
- {
650
- "epoch": 0.0184,
651
- "grad_norm": 22.44339942932129,
652
- "learning_rate": 0.00029507615230460916,
653
- "loss": 81.31973266601562,
654
- "step": 920
655
- },
656
- {
657
- "epoch": 0.0186,
658
- "grad_norm": 10.331223487854004,
659
- "learning_rate": 0.0002950160320641282,
660
- "loss": 83.343115234375,
661
- "step": 930
662
- },
663
- {
664
- "epoch": 0.0188,
665
- "grad_norm": 9.869694709777832,
666
- "learning_rate": 0.0002949559118236473,
667
- "loss": 79.85993041992188,
668
- "step": 940
669
- },
670
- {
671
- "epoch": 0.019,
672
- "grad_norm": 10.701475143432617,
673
- "learning_rate": 0.0002948957915831663,
674
- "loss": 84.02029418945312,
675
- "step": 950
676
- },
677
- {
678
- "epoch": 0.0192,
679
- "grad_norm": 12.234188079833984,
680
- "learning_rate": 0.00029483567134268535,
681
- "loss": 83.42827758789062,
682
- "step": 960
683
- },
684
- {
685
- "epoch": 0.0194,
686
- "grad_norm": 11.610219955444336,
687
- "learning_rate": 0.0002947755511022044,
688
- "loss": 80.161669921875,
689
- "step": 970
690
- },
691
- {
692
- "epoch": 0.0196,
693
- "grad_norm": 8.672530174255371,
694
- "learning_rate": 0.00029471543086172344,
695
- "loss": 81.43499145507812,
696
- "step": 980
697
- },
698
- {
699
- "epoch": 0.0198,
700
- "grad_norm": 11.546252250671387,
701
- "learning_rate": 0.00029465531062124244,
702
- "loss": 79.0343017578125,
703
- "step": 990
704
- },
705
- {
706
- "epoch": 0.02,
707
- "grad_norm": 11.632270812988281,
708
- "learning_rate": 0.0002945951903807615,
709
- "loss": 80.40287475585937,
710
- "step": 1000
711
  }
712
  ],
713
  "logging_steps": 10,
@@ -727,7 +377,7 @@
727
  "attributes": {}
728
  }
729
  },
730
- "total_flos": 1.65498647609088e+16,
731
  "train_batch_size": 2,
732
  "trial_name": null,
733
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.01,
6
  "eval_steps": 500,
7
+ "global_step": 500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
358
  "learning_rate": 0.0002976012024048096,
359
  "loss": 78.9918212890625,
360
  "step": 500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
361
  }
362
  ],
363
  "logging_steps": 10,
 
377
  "attributes": {}
378
  }
379
  },
380
+ "total_flos": 8161860202859520.0,
381
  "train_batch_size": 2,
382
  "trial_name": null,
383
  "trial_params": null
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cba5aee8ada95f513ddc2ee5204de606b5a10fa33758d2e9c8bcbc512ffe2d6b
3
  size 5201
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3ead25e90961f39a7cf235a53c52302322d446327bb423dad890f04c47276211
3
  size 5201
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5a3ad66771ef5b8246c59acc3ea2a9104f762ec13d612e6e588d54d7c66836e3
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5115379d755d2923b9e3aee82fa06ce8b6ba2890c3351d2cad9f1e33c87b9712
3
  size 1235573136