CodeIsAbstract commited on
Commit
695730b
·
verified ·
1 Parent(s): 1023101

Training in progress, step 3500, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:af5354d0e29ddec9e27730302b091cb0479fe966a93828191865da85fb63ce05
3
  size 234681136
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:897a290be0c0b16790ad7b090d09b56b33c5889810a7ea5907d4e0fed1bd51ab
3
  size 234681136
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7f761042663de134db3875290b47fa4678d163e6a29c60d847213bd9e9bf820e
3
  size 469516363
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ba7808784e33b510417a0442d2514a5dc45d8fa9986862d18084d909efcb30a7
3
  size 469516363
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:83d5db11e093d39ad6884cbfd0832c030b7cb2063c1e8ae3708e8e5be7a44ced
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7a9fa5fc86e7c65bf8b2609ce132c422acd30ff79c1395383385a52bfa9deec0
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:623efa011285242890286b7518de62c3b1d6ed1f56707c7ec58c501e60c532b6
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3e1c5dfea78108b1145382b3ab5baa3a9f481a062e702dff052035bc19b0def5
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.6,
6
  "eval_steps": 100,
7
- "global_step": 3000,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -2348,6 +2348,396 @@
2348
  "eval_samples_per_second": 67.051,
2349
  "eval_steps_per_second": 3.825,
2350
  "step": 3000
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2351
  }
2352
  ],
2353
  "logging_steps": 10,
@@ -2367,7 +2757,7 @@
2367
  "attributes": {}
2368
  }
2369
  },
2370
- "total_flos": 9.78463544573952e+17,
2371
  "train_batch_size": 18,
2372
  "trial_name": null,
2373
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.7,
6
  "eval_steps": 100,
7
+ "global_step": 3500,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
2348
  "eval_samples_per_second": 67.051,
2349
  "eval_steps_per_second": 3.825,
2350
  "step": 3000
2351
+ },
2352
+ {
2353
+ "epoch": 0.602,
2354
+ "grad_norm": 0.055908203125,
2355
+ "learning_rate": 0.0003,
2356
+ "loss": 2.6127801895141602,
2357
+ "step": 3010
2358
+ },
2359
+ {
2360
+ "epoch": 0.604,
2361
+ "grad_norm": 0.1455078125,
2362
+ "learning_rate": 0.0003,
2363
+ "loss": 2.5975704193115234,
2364
+ "step": 3020
2365
+ },
2366
+ {
2367
+ "epoch": 0.606,
2368
+ "grad_norm": 15.0625,
2369
+ "learning_rate": 0.0003,
2370
+ "loss": 2.5873924255371095,
2371
+ "step": 3030
2372
+ },
2373
+ {
2374
+ "epoch": 0.608,
2375
+ "grad_norm": 0.0478515625,
2376
+ "learning_rate": 0.0003,
2377
+ "loss": 2.5901187896728515,
2378
+ "step": 3040
2379
+ },
2380
+ {
2381
+ "epoch": 0.61,
2382
+ "grad_norm": 0.166015625,
2383
+ "learning_rate": 0.0003,
2384
+ "loss": 2.585931968688965,
2385
+ "step": 3050
2386
+ },
2387
+ {
2388
+ "epoch": 0.612,
2389
+ "grad_norm": 0.053466796875,
2390
+ "learning_rate": 0.0003,
2391
+ "loss": 2.6187633514404296,
2392
+ "step": 3060
2393
+ },
2394
+ {
2395
+ "epoch": 0.614,
2396
+ "grad_norm": 0.0556640625,
2397
+ "learning_rate": 0.0003,
2398
+ "loss": 2.6473546981811524,
2399
+ "step": 3070
2400
+ },
2401
+ {
2402
+ "epoch": 0.616,
2403
+ "grad_norm": 0.055908203125,
2404
+ "learning_rate": 0.0003,
2405
+ "loss": 2.606519317626953,
2406
+ "step": 3080
2407
+ },
2408
+ {
2409
+ "epoch": 0.618,
2410
+ "grad_norm": 12.0,
2411
+ "learning_rate": 0.0003,
2412
+ "loss": 2.6005083084106446,
2413
+ "step": 3090
2414
+ },
2415
+ {
2416
+ "epoch": 0.62,
2417
+ "grad_norm": 0.0615234375,
2418
+ "learning_rate": 0.0003,
2419
+ "loss": 2.6309852600097656,
2420
+ "step": 3100
2421
+ },
2422
+ {
2423
+ "epoch": 0.62,
2424
+ "eval_loss": 3.0271658897399902,
2425
+ "eval_runtime": 4.4454,
2426
+ "eval_samples_per_second": 67.036,
2427
+ "eval_steps_per_second": 3.824,
2428
+ "step": 3100
2429
+ },
2430
+ {
2431
+ "epoch": 0.622,
2432
+ "grad_norm": 0.107421875,
2433
+ "learning_rate": 0.0003,
2434
+ "loss": 2.6178503036499023,
2435
+ "step": 3110
2436
+ },
2437
+ {
2438
+ "epoch": 0.624,
2439
+ "grad_norm": 0.06640625,
2440
+ "learning_rate": 0.0003,
2441
+ "loss": 2.614837646484375,
2442
+ "step": 3120
2443
+ },
2444
+ {
2445
+ "epoch": 0.626,
2446
+ "grad_norm": 1.203125,
2447
+ "learning_rate": 0.0003,
2448
+ "loss": 2.62316837310791,
2449
+ "step": 3130
2450
+ },
2451
+ {
2452
+ "epoch": 0.628,
2453
+ "grad_norm": 0.0517578125,
2454
+ "learning_rate": 0.0003,
2455
+ "loss": 2.59710693359375,
2456
+ "step": 3140
2457
+ },
2458
+ {
2459
+ "epoch": 0.63,
2460
+ "grad_norm": 0.058349609375,
2461
+ "learning_rate": 0.0003,
2462
+ "loss": 2.6030540466308594,
2463
+ "step": 3150
2464
+ },
2465
+ {
2466
+ "epoch": 0.632,
2467
+ "grad_norm": 0.05029296875,
2468
+ "learning_rate": 0.0003,
2469
+ "loss": 2.5992559432983398,
2470
+ "step": 3160
2471
+ },
2472
+ {
2473
+ "epoch": 0.634,
2474
+ "grad_norm": 0.052490234375,
2475
+ "learning_rate": 0.0003,
2476
+ "loss": 2.6360668182373046,
2477
+ "step": 3170
2478
+ },
2479
+ {
2480
+ "epoch": 0.636,
2481
+ "grad_norm": 0.11279296875,
2482
+ "learning_rate": 0.0003,
2483
+ "loss": 2.6165666580200195,
2484
+ "step": 3180
2485
+ },
2486
+ {
2487
+ "epoch": 0.638,
2488
+ "grad_norm": 0.052978515625,
2489
+ "learning_rate": 0.0003,
2490
+ "loss": 2.636544036865234,
2491
+ "step": 3190
2492
+ },
2493
+ {
2494
+ "epoch": 0.64,
2495
+ "grad_norm": 0.0947265625,
2496
+ "learning_rate": 0.0003,
2497
+ "loss": 2.6081783294677736,
2498
+ "step": 3200
2499
+ },
2500
+ {
2501
+ "epoch": 0.64,
2502
+ "eval_loss": 3.0295300483703613,
2503
+ "eval_runtime": 4.4243,
2504
+ "eval_samples_per_second": 67.356,
2505
+ "eval_steps_per_second": 3.842,
2506
+ "step": 3200
2507
+ },
2508
+ {
2509
+ "epoch": 0.642,
2510
+ "grad_norm": 0.07470703125,
2511
+ "learning_rate": 0.0003,
2512
+ "loss": 2.6136301040649412,
2513
+ "step": 3210
2514
+ },
2515
+ {
2516
+ "epoch": 0.644,
2517
+ "grad_norm": 5.59375,
2518
+ "learning_rate": 0.0003,
2519
+ "loss": 2.611122703552246,
2520
+ "step": 3220
2521
+ },
2522
+ {
2523
+ "epoch": 0.646,
2524
+ "grad_norm": 0.0888671875,
2525
+ "learning_rate": 0.0003,
2526
+ "loss": 2.611610984802246,
2527
+ "step": 3230
2528
+ },
2529
+ {
2530
+ "epoch": 0.648,
2531
+ "grad_norm": 6.4375,
2532
+ "learning_rate": 0.0003,
2533
+ "loss": 2.6010536193847655,
2534
+ "step": 3240
2535
+ },
2536
+ {
2537
+ "epoch": 0.65,
2538
+ "grad_norm": 0.051513671875,
2539
+ "learning_rate": 0.0003,
2540
+ "loss": 2.623811149597168,
2541
+ "step": 3250
2542
+ },
2543
+ {
2544
+ "epoch": 0.652,
2545
+ "grad_norm": 0.06005859375,
2546
+ "learning_rate": 0.0003,
2547
+ "loss": 2.5976247787475586,
2548
+ "step": 3260
2549
+ },
2550
+ {
2551
+ "epoch": 0.654,
2552
+ "grad_norm": 0.05859375,
2553
+ "learning_rate": 0.0003,
2554
+ "loss": 2.593853569030762,
2555
+ "step": 3270
2556
+ },
2557
+ {
2558
+ "epoch": 0.656,
2559
+ "grad_norm": 0.060546875,
2560
+ "learning_rate": 0.0003,
2561
+ "loss": 2.645317268371582,
2562
+ "step": 3280
2563
+ },
2564
+ {
2565
+ "epoch": 0.658,
2566
+ "grad_norm": 0.0595703125,
2567
+ "learning_rate": 0.0003,
2568
+ "loss": 2.592297172546387,
2569
+ "step": 3290
2570
+ },
2571
+ {
2572
+ "epoch": 0.66,
2573
+ "grad_norm": 2.21875,
2574
+ "learning_rate": 0.0003,
2575
+ "loss": 2.60241641998291,
2576
+ "step": 3300
2577
+ },
2578
+ {
2579
+ "epoch": 0.66,
2580
+ "eval_loss": 3.0250847339630127,
2581
+ "eval_runtime": 4.4783,
2582
+ "eval_samples_per_second": 66.543,
2583
+ "eval_steps_per_second": 3.796,
2584
+ "step": 3300
2585
+ },
2586
+ {
2587
+ "epoch": 0.662,
2588
+ "grad_norm": 0.052490234375,
2589
+ "learning_rate": 0.0003,
2590
+ "loss": 2.600591278076172,
2591
+ "step": 3310
2592
+ },
2593
+ {
2594
+ "epoch": 0.664,
2595
+ "grad_norm": 0.052978515625,
2596
+ "learning_rate": 0.0003,
2597
+ "loss": 2.6144927978515624,
2598
+ "step": 3320
2599
+ },
2600
+ {
2601
+ "epoch": 0.666,
2602
+ "grad_norm": 0.057861328125,
2603
+ "learning_rate": 0.0003,
2604
+ "loss": 2.593229866027832,
2605
+ "step": 3330
2606
+ },
2607
+ {
2608
+ "epoch": 0.668,
2609
+ "grad_norm": 0.0546875,
2610
+ "learning_rate": 0.0003,
2611
+ "loss": 2.6041738510131838,
2612
+ "step": 3340
2613
+ },
2614
+ {
2615
+ "epoch": 0.67,
2616
+ "grad_norm": 0.052978515625,
2617
+ "learning_rate": 0.0003,
2618
+ "loss": 2.607752227783203,
2619
+ "step": 3350
2620
+ },
2621
+ {
2622
+ "epoch": 0.672,
2623
+ "grad_norm": 1.671875,
2624
+ "learning_rate": 0.0003,
2625
+ "loss": 2.6296701431274414,
2626
+ "step": 3360
2627
+ },
2628
+ {
2629
+ "epoch": 0.674,
2630
+ "grad_norm": 0.55859375,
2631
+ "learning_rate": 0.0003,
2632
+ "loss": 2.5772432327270507,
2633
+ "step": 3370
2634
+ },
2635
+ {
2636
+ "epoch": 0.676,
2637
+ "grad_norm": 0.052734375,
2638
+ "learning_rate": 0.0003,
2639
+ "loss": 2.6260414123535156,
2640
+ "step": 3380
2641
+ },
2642
+ {
2643
+ "epoch": 0.678,
2644
+ "grad_norm": 0.062255859375,
2645
+ "learning_rate": 0.0003,
2646
+ "loss": 2.5910011291503907,
2647
+ "step": 3390
2648
+ },
2649
+ {
2650
+ "epoch": 0.68,
2651
+ "grad_norm": 0.0546875,
2652
+ "learning_rate": 0.0003,
2653
+ "loss": 2.5773406982421876,
2654
+ "step": 3400
2655
+ },
2656
+ {
2657
+ "epoch": 0.68,
2658
+ "eval_loss": 3.0262410640716553,
2659
+ "eval_runtime": 4.4195,
2660
+ "eval_samples_per_second": 67.428,
2661
+ "eval_steps_per_second": 3.847,
2662
+ "step": 3400
2663
+ },
2664
+ {
2665
+ "epoch": 0.682,
2666
+ "grad_norm": 0.056640625,
2667
+ "learning_rate": 0.0003,
2668
+ "loss": 2.602094268798828,
2669
+ "step": 3410
2670
+ },
2671
+ {
2672
+ "epoch": 0.684,
2673
+ "grad_norm": 0.04931640625,
2674
+ "learning_rate": 0.0003,
2675
+ "loss": 2.6283613204956056,
2676
+ "step": 3420
2677
+ },
2678
+ {
2679
+ "epoch": 0.686,
2680
+ "grad_norm": 0.26171875,
2681
+ "learning_rate": 0.0003,
2682
+ "loss": 2.58358211517334,
2683
+ "step": 3430
2684
+ },
2685
+ {
2686
+ "epoch": 0.688,
2687
+ "grad_norm": 0.052978515625,
2688
+ "learning_rate": 0.0003,
2689
+ "loss": 2.5964986801147463,
2690
+ "step": 3440
2691
+ },
2692
+ {
2693
+ "epoch": 0.69,
2694
+ "grad_norm": 0.05322265625,
2695
+ "learning_rate": 0.0003,
2696
+ "loss": 2.5851970672607423,
2697
+ "step": 3450
2698
+ },
2699
+ {
2700
+ "epoch": 0.692,
2701
+ "grad_norm": 0.051025390625,
2702
+ "learning_rate": 0.0003,
2703
+ "loss": 2.6003726959228515,
2704
+ "step": 3460
2705
+ },
2706
+ {
2707
+ "epoch": 0.694,
2708
+ "grad_norm": 0.050048828125,
2709
+ "learning_rate": 0.0003,
2710
+ "loss": 2.6186376571655274,
2711
+ "step": 3470
2712
+ },
2713
+ {
2714
+ "epoch": 0.696,
2715
+ "grad_norm": 0.052001953125,
2716
+ "learning_rate": 0.0003,
2717
+ "loss": 2.624268341064453,
2718
+ "step": 3480
2719
+ },
2720
+ {
2721
+ "epoch": 0.698,
2722
+ "grad_norm": 0.05029296875,
2723
+ "learning_rate": 0.0003,
2724
+ "loss": 2.5755191802978517,
2725
+ "step": 3490
2726
+ },
2727
+ {
2728
+ "epoch": 0.7,
2729
+ "grad_norm": 0.244140625,
2730
+ "learning_rate": 0.0003,
2731
+ "loss": 2.593659782409668,
2732
+ "step": 3500
2733
+ },
2734
+ {
2735
+ "epoch": 0.7,
2736
+ "eval_loss": 3.0274429321289062,
2737
+ "eval_runtime": 4.4166,
2738
+ "eval_samples_per_second": 67.472,
2739
+ "eval_steps_per_second": 3.849,
2740
+ "step": 3500
2741
  }
2742
  ],
2743
  "logging_steps": 10,
 
2757
  "attributes": {}
2758
  }
2759
  },
2760
+ "total_flos": 1.141540802002944e+18,
2761
  "train_batch_size": 18,
2762
  "trial_name": null,
2763
  "trial_params": null