devoppro commited on
Commit
276227a
·
verified ·
1 Parent(s): 0e1badc

Training in progress, step 900

Browse files
last-checkpoint/config.json CHANGED
@@ -2,13 +2,9 @@
2
  "architectures": [
3
  "ModernLLMForCausalLM"
4
  ],
5
- "auto_map": {
6
- "AutoConfig": "configuration_modern_llm.ModernLLMConfig",
7
- "AutoModelForCausalLM": "modeling_modern_llm.ModernLLMForCausalLM"
8
- },
9
- "bos_token_id": 151643,
10
  "dtype": "float32",
11
- "eos_token_id": 151643,
12
  "hidden_size": 768,
13
  "intermediate_size": 2048,
14
  "max_position_embeddings": 2048,
@@ -16,7 +12,7 @@
16
  "num_attention_heads": 12,
17
  "num_hidden_layers": 12,
18
  "num_key_value_heads": 4,
19
- "pad_token_id": 151643,
20
  "rms_norm_eps": 1e-06,
21
  "rope_theta": 1000000.0,
22
  "transformers_version": "5.15.1",
 
2
  "architectures": [
3
  "ModernLLMForCausalLM"
4
  ],
5
+ "bos_token_id": 1,
 
 
 
 
6
  "dtype": "float32",
7
+ "eos_token_id": 2,
8
  "hidden_size": 768,
9
  "intermediate_size": 2048,
10
  "max_position_embeddings": 2048,
 
12
  "num_attention_heads": 12,
13
  "num_hidden_layers": 12,
14
  "num_key_value_heads": 4,
15
+ "pad_token_id": 0,
16
  "rms_norm_eps": 1e-06,
17
  "rope_theta": 1000000.0,
18
  "transformers_version": "5.15.1",
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f84b17a798e584a5f20295c2ca6d0e22796c96f14117857521b9e1b9a6036999
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:224bbc547320cbc7a57a274bef9f525f2e4be737031df14b601959f9380cb929
3
  size 1235573136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dccafa71c45e0091721a7b6ad53e58f216fa918836a42eada20ce086a2164460
3
  size 2471218763
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f014f1f8cf2eb9a4ff44ad059ad7d7ab8356457d91805882d791c32ab849f992
3
  size 2471218763
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:1347e41739e6d97d50d223900089b55b6dd4a02d2dc85ec6e9aa6a312e93005a
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:68b03f74e10591ff22477365642275703166a34b306471da2a9df3e87497356f
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e10f9e65a9c24c381ff534d5df3cf5fdc8a240cae768aed7007c78e044f60c2a
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0471ce3745a36e49a3e0d19eac020ddc42dff8b92447d06fa38e6a682e228b19
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.016,
6
  "eval_steps": 500,
7
- "global_step": 800,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -358,216 +358,6 @@
358
  "learning_rate": 0.0002976012024048096,
359
  "loss": 78.9918212890625,
360
  "step": 500
361
- },
362
- {
363
- "epoch": 0.0102,
364
- "grad_norm": 15.467362403869629,
365
- "learning_rate": 0.00029754108216432863,
366
- "loss": 124.401513671875,
367
- "step": 510
368
- },
369
- {
370
- "epoch": 0.0104,
371
- "grad_norm": 13.991318702697754,
372
- "learning_rate": 0.0002974809619238477,
373
- "loss": 110.567431640625,
374
- "step": 520
375
- },
376
- {
377
- "epoch": 0.0106,
378
- "grad_norm": 17.036069869995117,
379
- "learning_rate": 0.0002974208416833667,
380
- "loss": 104.7319580078125,
381
- "step": 530
382
- },
383
- {
384
- "epoch": 0.0108,
385
- "grad_norm": 13.922304153442383,
386
- "learning_rate": 0.0002973607214428857,
387
- "loss": 100.63988647460937,
388
- "step": 540
389
- },
390
- {
391
- "epoch": 0.011,
392
- "grad_norm": 13.223531723022461,
393
- "learning_rate": 0.0002973006012024048,
394
- "loss": 99.96443481445313,
395
- "step": 550
396
- },
397
- {
398
- "epoch": 0.0112,
399
- "grad_norm": 14.53707504272461,
400
- "learning_rate": 0.0002972404809619238,
401
- "loss": 95.74251098632813,
402
- "step": 560
403
- },
404
- {
405
- "epoch": 0.0114,
406
- "grad_norm": 14.91770076751709,
407
- "learning_rate": 0.00029718036072144287,
408
- "loss": 94.661767578125,
409
- "step": 570
410
- },
411
- {
412
- "epoch": 0.0116,
413
- "grad_norm": 13.45614242553711,
414
- "learning_rate": 0.0002971202404809619,
415
- "loss": 90.941796875,
416
- "step": 580
417
- },
418
- {
419
- "epoch": 0.0118,
420
- "grad_norm": 13.695131301879883,
421
- "learning_rate": 0.00029706012024048096,
422
- "loss": 92.9096923828125,
423
- "step": 590
424
- },
425
- {
426
- "epoch": 0.012,
427
- "grad_norm": 16.406902313232422,
428
- "learning_rate": 0.00029699999999999996,
429
- "loss": 90.14566650390626,
430
- "step": 600
431
- },
432
- {
433
- "epoch": 0.0122,
434
- "grad_norm": 11.779080390930176,
435
- "learning_rate": 0.000296939879759519,
436
- "loss": 86.8746826171875,
437
- "step": 610
438
- },
439
- {
440
- "epoch": 0.0124,
441
- "grad_norm": 13.758212089538574,
442
- "learning_rate": 0.00029687975951903805,
443
- "loss": 87.3355224609375,
444
- "step": 620
445
- },
446
- {
447
- "epoch": 0.0126,
448
- "grad_norm": 12.833357810974121,
449
- "learning_rate": 0.0002968196392785571,
450
- "loss": 89.81148071289063,
451
- "step": 630
452
- },
453
- {
454
- "epoch": 0.0128,
455
- "grad_norm": 14.684094429016113,
456
- "learning_rate": 0.0002967595190380761,
457
- "loss": 89.44540405273438,
458
- "step": 640
459
- },
460
- {
461
- "epoch": 0.013,
462
- "grad_norm": 10.780083656311035,
463
- "learning_rate": 0.0002966993987975952,
464
- "loss": 88.07953491210938,
465
- "step": 650
466
- },
467
- {
468
- "epoch": 0.0132,
469
- "grad_norm": 11.908385276794434,
470
- "learning_rate": 0.0002966392785571142,
471
- "loss": 87.04054565429688,
472
- "step": 660
473
- },
474
- {
475
- "epoch": 0.0134,
476
- "grad_norm": 15.518954277038574,
477
- "learning_rate": 0.00029657915831663323,
478
- "loss": 85.9423583984375,
479
- "step": 670
480
- },
481
- {
482
- "epoch": 0.0136,
483
- "grad_norm": 10.570508003234863,
484
- "learning_rate": 0.0002965190380761523,
485
- "loss": 86.79979858398437,
486
- "step": 680
487
- },
488
- {
489
- "epoch": 0.0138,
490
- "grad_norm": 14.696393013000488,
491
- "learning_rate": 0.00029645891783567133,
492
- "loss": 79.76475219726562,
493
- "step": 690
494
- },
495
- {
496
- "epoch": 0.014,
497
- "grad_norm": 12.92392349243164,
498
- "learning_rate": 0.0002963987975951904,
499
- "loss": 87.0448974609375,
500
- "step": 700
501
- },
502
- {
503
- "epoch": 0.0142,
504
- "grad_norm": 13.465226173400879,
505
- "learning_rate": 0.00029633867735470937,
506
- "loss": 90.22592163085938,
507
- "step": 710
508
- },
509
- {
510
- "epoch": 0.0144,
511
- "grad_norm": 12.435464859008789,
512
- "learning_rate": 0.0002962785571142284,
513
- "loss": 86.32198486328124,
514
- "step": 720
515
- },
516
- {
517
- "epoch": 0.0146,
518
- "grad_norm": 12.340784072875977,
519
- "learning_rate": 0.00029621843687374747,
520
- "loss": 89.89630737304688,
521
- "step": 730
522
- },
523
- {
524
- "epoch": 0.0148,
525
- "grad_norm": 14.63569164276123,
526
- "learning_rate": 0.0002961583166332665,
527
- "loss": 86.31669921875,
528
- "step": 740
529
- },
530
- {
531
- "epoch": 0.015,
532
- "grad_norm": 10.646244049072266,
533
- "learning_rate": 0.00029609819639278556,
534
- "loss": 85.20569458007813,
535
- "step": 750
536
- },
537
- {
538
- "epoch": 0.0152,
539
- "grad_norm": 8.930062294006348,
540
- "learning_rate": 0.0002960380761523046,
541
- "loss": 86.44678955078125,
542
- "step": 760
543
- },
544
- {
545
- "epoch": 0.0154,
546
- "grad_norm": 12.228261947631836,
547
- "learning_rate": 0.0002959779559118236,
548
- "loss": 84.96986083984375,
549
- "step": 770
550
- },
551
- {
552
- "epoch": 0.0156,
553
- "grad_norm": 9.189741134643555,
554
- "learning_rate": 0.00029591783567134265,
555
- "loss": 86.03868408203125,
556
- "step": 780
557
- },
558
- {
559
- "epoch": 0.0158,
560
- "grad_norm": 12.331954956054688,
561
- "learning_rate": 0.0002958577154308617,
562
- "loss": 85.13253784179688,
563
- "step": 790
564
- },
565
- {
566
- "epoch": 0.016,
567
- "grad_norm": 14.345333099365234,
568
- "learning_rate": 0.00029579759519038075,
569
- "loss": 86.58322143554688,
570
- "step": 800
571
  }
572
  ],
573
  "logging_steps": 10,
@@ -587,7 +377,7 @@
587
  "attributes": {}
588
  }
589
  },
590
- "total_flos": 1.315412906047488e+16,
591
  "train_batch_size": 2,
592
  "trial_name": null,
593
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.01,
6
  "eval_steps": 500,
7
+ "global_step": 500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
358
  "learning_rate": 0.0002976012024048096,
359
  "loss": 78.9918212890625,
360
  "step": 500
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
361
  }
362
  ],
363
  "logging_steps": 10,
 
377
  "attributes": {}
378
  }
379
  },
380
+ "total_flos": 8161860202859520.0,
381
  "train_batch_size": 2,
382
  "trial_name": null,
383
  "trial_params": null
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cba5aee8ada95f513ddc2ee5204de606b5a10fa33758d2e9c8bcbc512ffe2d6b
3
  size 5201
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3ead25e90961f39a7cf235a53c52302322d446327bb423dad890f04c47276211
3
  size 5201
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f84b17a798e584a5f20295c2ca6d0e22796c96f14117857521b9e1b9a6036999
3
  size 1235573136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ffff9e975d47376ee2def7076b59f815ce93ea1f571a2f266564c97b76a192a4
3
  size 1235573136