aixk commited on
Commit
eb192bc
·
verified ·
1 Parent(s): 8b61aff

Upload folder using huggingface_hub

Browse files
checkpoint-600/hybrid_tokenizer_mapping.json CHANGED
The diff for this file is too large to render. See raw diff
 
checkpoint-600/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:349414166cba298420b7d258a8646e5a2739f6f68b6de484bf0ee534f9ab641e
3
  size 501759848
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9072726981ec7e8bf7d7756cc5487ad55281d66945a89c30b6a780ee878272ba
3
  size 501759848
checkpoint-600/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fa6f2bf80d3aad8ca62b05862952bf4bdd06c94c4aec076c56ebef337e100749
3
  size 254410983
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e0e65d287cc1248788f59572bcae4537952068dbb45d16433f12032db06b0122
3
  size 254410983
checkpoint-600/trainer_state.json CHANGED
@@ -361,72 +361,72 @@
361
  },
362
  {
363
  "epoch": 0.3138340833044883,
364
- "grad_norm": 63.695133209228516,
365
  "learning_rate": 5.0199999999999994e-05,
366
  "loss": 6.0088,
367
  "step": 510
368
  },
369
  {
370
  "epoch": 0.3199876927810469,
371
- "grad_norm": 50.28738784790039,
372
  "learning_rate": 5.119999999999999e-05,
373
  "loss": 5.9469,
374
  "step": 520
375
  },
376
  {
377
  "epoch": 0.32614130225760546,
378
- "grad_norm": 43.3399543762207,
379
  "learning_rate": 5.2199999999999995e-05,
380
  "loss": 5.9464,
381
  "step": 530
382
  },
383
  {
384
  "epoch": 0.3322949117341641,
385
- "grad_norm": 48.50353240966797,
386
  "learning_rate": 5.32e-05,
387
- "loss": 5.9137,
388
  "step": 540
389
  },
390
  {
391
  "epoch": 0.33844852121072266,
392
- "grad_norm": 97.7204360961914,
393
  "learning_rate": 5.4199999999999996e-05,
394
- "loss": 5.9181,
395
  "step": 550
396
  },
397
  {
398
  "epoch": 0.34460213068728124,
399
- "grad_norm": 43.342079162597656,
400
  "learning_rate": 5.519999999999999e-05,
401
- "loss": 5.8822,
402
  "step": 560
403
  },
404
  {
405
  "epoch": 0.35075574016383987,
406
- "grad_norm": 51.02119827270508,
407
  "learning_rate": 5.619999999999999e-05,
408
- "loss": 5.8766,
409
  "step": 570
410
  },
411
  {
412
  "epoch": 0.35690934964039844,
413
- "grad_norm": 54.75588607788086,
414
  "learning_rate": 5.72e-05,
415
- "loss": 5.842,
416
  "step": 580
417
  },
418
  {
419
  "epoch": 0.363062959116957,
420
- "grad_norm": 49.630027770996094,
421
  "learning_rate": 5.82e-05,
422
- "loss": 5.8319,
423
  "step": 590
424
  },
425
  {
426
  "epoch": 0.36921656859351565,
427
- "grad_norm": 50.86918258666992,
428
  "learning_rate": 5.9199999999999996e-05,
429
- "loss": 5.7875,
430
  "step": 600
431
  }
432
  ],
 
361
  },
362
  {
363
  "epoch": 0.3138340833044883,
364
+ "grad_norm": 63.603389739990234,
365
  "learning_rate": 5.0199999999999994e-05,
366
  "loss": 6.0088,
367
  "step": 510
368
  },
369
  {
370
  "epoch": 0.3199876927810469,
371
+ "grad_norm": 50.28224563598633,
372
  "learning_rate": 5.119999999999999e-05,
373
  "loss": 5.9469,
374
  "step": 520
375
  },
376
  {
377
  "epoch": 0.32614130225760546,
378
+ "grad_norm": 43.31625747680664,
379
  "learning_rate": 5.2199999999999995e-05,
380
  "loss": 5.9464,
381
  "step": 530
382
  },
383
  {
384
  "epoch": 0.3322949117341641,
385
+ "grad_norm": 56.94966506958008,
386
  "learning_rate": 5.32e-05,
387
+ "loss": 5.9139,
388
  "step": 540
389
  },
390
  {
391
  "epoch": 0.33844852121072266,
392
+ "grad_norm": 97.49286651611328,
393
  "learning_rate": 5.4199999999999996e-05,
394
+ "loss": 5.9177,
395
  "step": 550
396
  },
397
  {
398
  "epoch": 0.34460213068728124,
399
+ "grad_norm": 57.25851058959961,
400
  "learning_rate": 5.519999999999999e-05,
401
+ "loss": 5.8818,
402
  "step": 560
403
  },
404
  {
405
  "epoch": 0.35075574016383987,
406
+ "grad_norm": 40.23714065551758,
407
  "learning_rate": 5.619999999999999e-05,
408
+ "loss": 5.8755,
409
  "step": 570
410
  },
411
  {
412
  "epoch": 0.35690934964039844,
413
+ "grad_norm": 39.241878509521484,
414
  "learning_rate": 5.72e-05,
415
+ "loss": 5.8417,
416
  "step": 580
417
  },
418
  {
419
  "epoch": 0.363062959116957,
420
+ "grad_norm": 46.391117095947266,
421
  "learning_rate": 5.82e-05,
422
+ "loss": 5.8327,
423
  "step": 590
424
  },
425
  {
426
  "epoch": 0.36921656859351565,
427
+ "grad_norm": 51.62196350097656,
428
  "learning_rate": 5.9199999999999996e-05,
429
+ "loss": 5.7876,
430
  "step": 600
431
  }
432
  ],