Instructions to use CodeIsAbstract/LLAMA_RoPE_Baseline with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use CodeIsAbstract/LLAMA_RoPE_Baseline with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="CodeIsAbstract/LLAMA_RoPE_Baseline")# Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("CodeIsAbstract/LLAMA_RoPE_Baseline") model = AutoModelForCausalLM.from_pretrained("CodeIsAbstract/LLAMA_RoPE_Baseline", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use CodeIsAbstract/LLAMA_RoPE_Baseline with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "CodeIsAbstract/LLAMA_RoPE_Baseline" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "CodeIsAbstract/LLAMA_RoPE_Baseline", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/CodeIsAbstract/LLAMA_RoPE_Baseline
- SGLang
How to use CodeIsAbstract/LLAMA_RoPE_Baseline with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "CodeIsAbstract/LLAMA_RoPE_Baseline" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "CodeIsAbstract/LLAMA_RoPE_Baseline", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "CodeIsAbstract/LLAMA_RoPE_Baseline" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "CodeIsAbstract/LLAMA_RoPE_Baseline", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use CodeIsAbstract/LLAMA_RoPE_Baseline with Docker Model Runner:
docker model run hf.co/CodeIsAbstract/LLAMA_RoPE_Baseline
Download last-checkpoint/trainer_state.json from CodeIsAbstract/LLAMA_RoPE_Baseline: direct link, hf CLI and curl.
- Browser
- Download file 103 kB
-
https://huggingface.co/CodeIsAbstract/LLAMA_RoPE_Baseline/resolve/main/last-checkpoint/trainer_state.json
- Command line
-
hf download hf://CodeIsAbstract/LLAMA_RoPE_Baseline/last-checkpoint/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/CodeIsAbstract/LLAMA_RoPE_Baseline/resolve/main/last-checkpoint/trainer_state.json
103 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 0.6753246753246753, | |
| "eval_steps": 1000, | |
| "global_step": 52000, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.0012987012987012987, | |
| "grad_norm": 0.8535653352737427, | |
| "learning_rate": 9.42857142857143e-05, | |
| "loss": 8.6239, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.0025974025974025974, | |
| "grad_norm": 0.9884688258171082, | |
| "learning_rate": 0.0001895238095238095, | |
| "loss": 6.7685, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.003896103896103896, | |
| "grad_norm": 0.5704336762428284, | |
| "learning_rate": 0.0002847619047619048, | |
| "loss": 5.9666, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.005194805194805195, | |
| "grad_norm": 0.6232308149337769, | |
| "learning_rate": 0.00038, | |
| "loss": 5.518, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.006493506493506494, | |
| "grad_norm": 0.6595708131790161, | |
| "learning_rate": 0.00047523809523809525, | |
| "loss": 5.205, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 0.007792207792207792, | |
| "grad_norm": 0.5402975678443909, | |
| "learning_rate": 0.0005704761904761905, | |
| "loss": 4.9366, | |
| "step": 600 | |
| }, | |
| { | |
| "epoch": 0.00909090909090909, | |
| "grad_norm": 0.5196501016616821, | |
| "learning_rate": 0.0006657142857142858, | |
| "loss": 4.7613, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 0.01038961038961039, | |
| "grad_norm": 0.6473610401153564, | |
| "learning_rate": 0.0007609523809523809, | |
| "loss": 4.6361, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 0.011688311688311689, | |
| "grad_norm": 0.4435962736606598, | |
| "learning_rate": 0.0008561904761904762, | |
| "loss": 4.5062, | |
| "step": 900 | |
| }, | |
| { | |
| "epoch": 0.012987012987012988, | |
| "grad_norm": 0.46511825919151306, | |
| "learning_rate": 0.0009514285714285714, | |
| "loss": 4.3812, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 0.012987012987012988, | |
| "eval_loss": 4.666484355926514, | |
| "eval_runtime": 15.6105, | |
| "eval_samples_per_second": 36.898, | |
| "eval_steps_per_second": 9.225, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 0.014285714285714285, | |
| "grad_norm": 0.38127702474594116, | |
| "learning_rate": 0.0009999995009117187, | |
| "loss": 4.2976, | |
| "step": 1100 | |
| }, | |
| { | |
| "epoch": 0.015584415584415584, | |
| "grad_norm": 0.4769284129142761, | |
| "learning_rate": 0.0009999953851546302, | |
| "loss": 4.1888, | |
| "step": 1200 | |
| }, | |
| { | |
| "epoch": 0.016883116883116882, | |
| "grad_norm": 0.4059095084667206, | |
| "learning_rate": 0.000999987112101314, | |
| "loss": 4.0874, | |
| "step": 1300 | |
| }, | |
| { | |
| "epoch": 0.01818181818181818, | |
| "grad_norm": 0.34086140990257263, | |
| "learning_rate": 0.0009999746818205575, | |
| "loss": 3.974, | |
| "step": 1400 | |
| }, | |
| { | |
| "epoch": 0.01948051948051948, | |
| "grad_norm": 0.32273247838020325, | |
| "learning_rate": 0.0009999580944157148, | |
| "loss": 3.9202, | |
| "step": 1500 | |
| }, | |
| { | |
| "epoch": 0.02077922077922078, | |
| "grad_norm": 0.36565664410591125, | |
| "learning_rate": 0.000999937350024705, | |
| "loss": 3.8821, | |
| "step": 1600 | |
| }, | |
| { | |
| "epoch": 0.02207792207792208, | |
| "grad_norm": 0.3269343376159668, | |
| "learning_rate": 0.0009999124488200097, | |
| "loss": 3.7686, | |
| "step": 1700 | |
| }, | |
| { | |
| "epoch": 0.023376623376623377, | |
| "grad_norm": 0.3859221339225769, | |
| "learning_rate": 0.0009998833910086751, | |
| "loss": 3.7079, | |
| "step": 1800 | |
| }, | |
| { | |
| "epoch": 0.024675324675324677, | |
| "grad_norm": 0.3663485646247864, | |
| "learning_rate": 0.000999850176832307, | |
| "loss": 3.6803, | |
| "step": 1900 | |
| }, | |
| { | |
| "epoch": 0.025974025974025976, | |
| "grad_norm": 0.46907544136047363, | |
| "learning_rate": 0.0009998128065670706, | |
| "loss": 3.6502, | |
| "step": 2000 | |
| }, | |
| { | |
| "epoch": 0.025974025974025976, | |
| "eval_loss": 3.9888415336608887, | |
| "eval_runtime": 15.6284, | |
| "eval_samples_per_second": 36.856, | |
| "eval_steps_per_second": 9.214, | |
| "step": 2000 | |
| }, | |
| { | |
| "epoch": 0.02727272727272727, | |
| "grad_norm": 0.3356938362121582, | |
| "learning_rate": 0.0009997712805236867, | |
| "loss": 3.6339, | |
| "step": 2100 | |
| }, | |
| { | |
| "epoch": 0.02857142857142857, | |
| "grad_norm": 0.2554614543914795, | |
| "learning_rate": 0.0009997255990474312, | |
| "loss": 3.601, | |
| "step": 2200 | |
| }, | |
| { | |
| "epoch": 0.02987012987012987, | |
| "grad_norm": 0.2642477750778198, | |
| "learning_rate": 0.00099967576251813, | |
| "loss": 3.5622, | |
| "step": 2300 | |
| }, | |
| { | |
| "epoch": 0.03116883116883117, | |
| "grad_norm": 0.2671547532081604, | |
| "learning_rate": 0.0009996217713501576, | |
| "loss": 3.517, | |
| "step": 2400 | |
| }, | |
| { | |
| "epoch": 0.032467532467532464, | |
| "grad_norm": 0.26076218485832214, | |
| "learning_rate": 0.0009995636259924328, | |
| "loss": 3.5058, | |
| "step": 2500 | |
| }, | |
| { | |
| "epoch": 0.033766233766233764, | |
| "grad_norm": 0.39065253734588623, | |
| "learning_rate": 0.0009995013269284145, | |
| "loss": 3.5262, | |
| "step": 2600 | |
| }, | |
| { | |
| "epoch": 0.03506493506493506, | |
| "grad_norm": 0.23748748004436493, | |
| "learning_rate": 0.0009994348746760996, | |
| "loss": 3.4818, | |
| "step": 2700 | |
| }, | |
| { | |
| "epoch": 0.03636363636363636, | |
| "grad_norm": 0.23590387403964996, | |
| "learning_rate": 0.000999364269788016, | |
| "loss": 3.4639, | |
| "step": 2800 | |
| }, | |
| { | |
| "epoch": 0.03766233766233766, | |
| "grad_norm": 0.2318197786808014, | |
| "learning_rate": 0.00099928951285122, | |
| "loss": 3.4444, | |
| "step": 2900 | |
| }, | |
| { | |
| "epoch": 0.03896103896103896, | |
| "grad_norm": 0.2852722406387329, | |
| "learning_rate": 0.0009992106044872913, | |
| "loss": 3.4227, | |
| "step": 3000 | |
| }, | |
| { | |
| "epoch": 0.03896103896103896, | |
| "eval_loss": 3.778900623321533, | |
| "eval_runtime": 15.1585, | |
| "eval_samples_per_second": 37.998, | |
| "eval_steps_per_second": 9.5, | |
| "step": 3000 | |
| }, | |
| { | |
| "epoch": 0.04025974025974026, | |
| "grad_norm": 0.23955008387565613, | |
| "learning_rate": 0.0009991275453523263, | |
| "loss": 3.3974, | |
| "step": 3100 | |
| }, | |
| { | |
| "epoch": 0.04155844155844156, | |
| "grad_norm": 0.2509465217590332, | |
| "learning_rate": 0.0009990403361369346, | |
| "loss": 3.4157, | |
| "step": 3200 | |
| }, | |
| { | |
| "epoch": 0.04285714285714286, | |
| "grad_norm": 0.24335195124149323, | |
| "learning_rate": 0.0009989489775662315, | |
| "loss": 3.3972, | |
| "step": 3300 | |
| }, | |
| { | |
| "epoch": 0.04415584415584416, | |
| "grad_norm": 0.7076203227043152, | |
| "learning_rate": 0.000998853470399834, | |
| "loss": 3.4082, | |
| "step": 3400 | |
| }, | |
| { | |
| "epoch": 0.045454545454545456, | |
| "grad_norm": 0.25543642044067383, | |
| "learning_rate": 0.0009987538154318518, | |
| "loss": 3.373, | |
| "step": 3500 | |
| }, | |
| { | |
| "epoch": 0.046753246753246755, | |
| "grad_norm": 0.7122920155525208, | |
| "learning_rate": 0.0009986500134908838, | |
| "loss": 3.3898, | |
| "step": 3600 | |
| }, | |
| { | |
| "epoch": 0.048051948051948054, | |
| "grad_norm": 0.2252849042415619, | |
| "learning_rate": 0.000998542065440008, | |
| "loss": 3.3739, | |
| "step": 3700 | |
| }, | |
| { | |
| "epoch": 0.04935064935064935, | |
| "grad_norm": 0.20722781121730804, | |
| "learning_rate": 0.0009984299721767773, | |
| "loss": 3.353, | |
| "step": 3800 | |
| }, | |
| { | |
| "epoch": 0.05064935064935065, | |
| "grad_norm": 0.2006598860025406, | |
| "learning_rate": 0.00099831373463321, | |
| "loss": 3.3124, | |
| "step": 3900 | |
| }, | |
| { | |
| "epoch": 0.05194805194805195, | |
| "grad_norm": 0.2024812549352646, | |
| "learning_rate": 0.0009981933537757828, | |
| "loss": 3.3367, | |
| "step": 4000 | |
| }, | |
| { | |
| "epoch": 0.05194805194805195, | |
| "eval_loss": 3.681532144546509, | |
| "eval_runtime": 16.9008, | |
| "eval_samples_per_second": 34.081, | |
| "eval_steps_per_second": 8.52, | |
| "step": 4000 | |
| }, | |
| { | |
| "epoch": 0.053246753246753244, | |
| "grad_norm": 0.24371765553951263, | |
| "learning_rate": 0.0009980688306054227, | |
| "loss": 3.2775, | |
| "step": 4100 | |
| }, | |
| { | |
| "epoch": 0.05454545454545454, | |
| "grad_norm": 0.20401820540428162, | |
| "learning_rate": 0.0009979401661574985, | |
| "loss": 3.2952, | |
| "step": 4200 | |
| }, | |
| { | |
| "epoch": 0.05584415584415584, | |
| "grad_norm": 0.2212119996547699, | |
| "learning_rate": 0.0009978073615018127, | |
| "loss": 3.257, | |
| "step": 4300 | |
| }, | |
| { | |
| "epoch": 0.05714285714285714, | |
| "grad_norm": 0.2384602129459381, | |
| "learning_rate": 0.000997670417742592, | |
| "loss": 3.2859, | |
| "step": 4400 | |
| }, | |
| { | |
| "epoch": 0.05844155844155844, | |
| "grad_norm": 0.21143975853919983, | |
| "learning_rate": 0.0009975293360184785, | |
| "loss": 3.2475, | |
| "step": 4500 | |
| }, | |
| { | |
| "epoch": 0.05974025974025974, | |
| "grad_norm": 0.20411203801631927, | |
| "learning_rate": 0.00099738411750252, | |
| "loss": 3.2443, | |
| "step": 4600 | |
| }, | |
| { | |
| "epoch": 0.06103896103896104, | |
| "grad_norm": 0.2097695916891098, | |
| "learning_rate": 0.0009972347634021606, | |
| "loss": 3.2703, | |
| "step": 4700 | |
| }, | |
| { | |
| "epoch": 0.06233766233766234, | |
| "grad_norm": 0.2140115350484848, | |
| "learning_rate": 0.00099708127495923, | |
| "loss": 3.2569, | |
| "step": 4800 | |
| }, | |
| { | |
| "epoch": 0.06363636363636363, | |
| "grad_norm": 0.2163541465997696, | |
| "learning_rate": 0.000996923653449934, | |
| "loss": 3.2646, | |
| "step": 4900 | |
| }, | |
| { | |
| "epoch": 0.06493506493506493, | |
| "grad_norm": 0.1931372582912445, | |
| "learning_rate": 0.0009967619001848434, | |
| "loss": 3.2224, | |
| "step": 5000 | |
| }, | |
| { | |
| "epoch": 0.06493506493506493, | |
| "eval_loss": 3.59517240524292, | |
| "eval_runtime": 15.382, | |
| "eval_samples_per_second": 37.446, | |
| "eval_steps_per_second": 9.362, | |
| "step": 5000 | |
| }, | |
| { | |
| "epoch": 0.06623376623376623, | |
| "grad_norm": 0.2017965465784073, | |
| "learning_rate": 0.0009965960165088828, | |
| "loss": 3.2159, | |
| "step": 5100 | |
| }, | |
| { | |
| "epoch": 0.06753246753246753, | |
| "grad_norm": 0.2096720039844513, | |
| "learning_rate": 0.0009964260038013203, | |
| "loss": 3.1967, | |
| "step": 5200 | |
| }, | |
| { | |
| "epoch": 0.06883116883116883, | |
| "grad_norm": 0.2119290977716446, | |
| "learning_rate": 0.000996251863475755, | |
| "loss": 3.1962, | |
| "step": 5300 | |
| }, | |
| { | |
| "epoch": 0.07012987012987013, | |
| "grad_norm": 0.27017727494239807, | |
| "learning_rate": 0.0009960735969801067, | |
| "loss": 3.1672, | |
| "step": 5400 | |
| }, | |
| { | |
| "epoch": 0.07142857142857142, | |
| "grad_norm": 0.21126972138881683, | |
| "learning_rate": 0.0009958912057966018, | |
| "loss": 3.2301, | |
| "step": 5500 | |
| }, | |
| { | |
| "epoch": 0.07272727272727272, | |
| "grad_norm": 0.238760307431221, | |
| "learning_rate": 0.0009957046914417628, | |
| "loss": 3.2367, | |
| "step": 5600 | |
| }, | |
| { | |
| "epoch": 0.07402597402597402, | |
| "grad_norm": 0.19038313627243042, | |
| "learning_rate": 0.000995514055466395, | |
| "loss": 3.2026, | |
| "step": 5700 | |
| }, | |
| { | |
| "epoch": 0.07532467532467532, | |
| "grad_norm": 0.20789292454719543, | |
| "learning_rate": 0.0009953192994555736, | |
| "loss": 3.1718, | |
| "step": 5800 | |
| }, | |
| { | |
| "epoch": 0.07662337662337662, | |
| "grad_norm": 0.18676762282848358, | |
| "learning_rate": 0.00099512042502863, | |
| "loss": 3.167, | |
| "step": 5900 | |
| }, | |
| { | |
| "epoch": 0.07792207792207792, | |
| "grad_norm": 0.19430765509605408, | |
| "learning_rate": 0.0009949174338391394, | |
| "loss": 3.1721, | |
| "step": 6000 | |
| }, | |
| { | |
| "epoch": 0.07792207792207792, | |
| "eval_loss": 3.5258212089538574, | |
| "eval_runtime": 15.3635, | |
| "eval_samples_per_second": 37.491, | |
| "eval_steps_per_second": 9.373, | |
| "step": 6000 | |
| }, | |
| { | |
| "epoch": 0.07922077922077922, | |
| "grad_norm": 0.20157617330551147, | |
| "learning_rate": 0.0009947103275749064, | |
| "loss": 3.1696, | |
| "step": 6100 | |
| }, | |
| { | |
| "epoch": 0.08051948051948052, | |
| "grad_norm": 0.18047969043254852, | |
| "learning_rate": 0.0009944991079579514, | |
| "loss": 3.1601, | |
| "step": 6200 | |
| }, | |
| { | |
| "epoch": 0.08181818181818182, | |
| "grad_norm": 0.2030571699142456, | |
| "learning_rate": 0.0009942837767444952, | |
| "loss": 3.1709, | |
| "step": 6300 | |
| }, | |
| { | |
| "epoch": 0.08311688311688312, | |
| "grad_norm": 0.18791615962982178, | |
| "learning_rate": 0.000994064335724946, | |
| "loss": 3.1808, | |
| "step": 6400 | |
| }, | |
| { | |
| "epoch": 0.08441558441558442, | |
| "grad_norm": 0.1927482783794403, | |
| "learning_rate": 0.0009938407867238828, | |
| "loss": 3.1272, | |
| "step": 6500 | |
| }, | |
| { | |
| "epoch": 0.08571428571428572, | |
| "grad_norm": 0.1764034628868103, | |
| "learning_rate": 0.0009936131316000418, | |
| "loss": 3.1407, | |
| "step": 6600 | |
| }, | |
| { | |
| "epoch": 0.08701298701298701, | |
| "grad_norm": 0.1786874383687973, | |
| "learning_rate": 0.0009933813722463004, | |
| "loss": 3.107, | |
| "step": 6700 | |
| }, | |
| { | |
| "epoch": 0.08831168831168831, | |
| "grad_norm": 0.2685558497905731, | |
| "learning_rate": 0.0009931455105896606, | |
| "loss": 3.1609, | |
| "step": 6800 | |
| }, | |
| { | |
| "epoch": 0.08961038961038961, | |
| "grad_norm": 0.19291436672210693, | |
| "learning_rate": 0.000992905548591234, | |
| "loss": 3.1444, | |
| "step": 6900 | |
| }, | |
| { | |
| "epoch": 0.09090909090909091, | |
| "grad_norm": 0.1896260678768158, | |
| "learning_rate": 0.0009926614882462253, | |
| "loss": 3.1329, | |
| "step": 7000 | |
| }, | |
| { | |
| "epoch": 0.09090909090909091, | |
| "eval_loss": 3.4966745376586914, | |
| "eval_runtime": 14.4834, | |
| "eval_samples_per_second": 39.77, | |
| "eval_steps_per_second": 9.942, | |
| "step": 7000 | |
| }, | |
| { | |
| "epoch": 0.09220779220779221, | |
| "grad_norm": 0.17750093340873718, | |
| "learning_rate": 0.0009924133315839156, | |
| "loss": 3.1602, | |
| "step": 7100 | |
| }, | |
| { | |
| "epoch": 0.09350649350649351, | |
| "grad_norm": 0.18574649095535278, | |
| "learning_rate": 0.0009921610806676456, | |
| "loss": 3.1451, | |
| "step": 7200 | |
| }, | |
| { | |
| "epoch": 0.09480519480519481, | |
| "grad_norm": 0.18256109952926636, | |
| "learning_rate": 0.000991904737594798, | |
| "loss": 3.1305, | |
| "step": 7300 | |
| }, | |
| { | |
| "epoch": 0.09610389610389611, | |
| "grad_norm": 0.17488889396190643, | |
| "learning_rate": 0.0009916443044967807, | |
| "loss": 3.1211, | |
| "step": 7400 | |
| }, | |
| { | |
| "epoch": 0.09740259740259741, | |
| "grad_norm": 0.205893412232399, | |
| "learning_rate": 0.0009913797835390088, | |
| "loss": 3.1039, | |
| "step": 7500 | |
| }, | |
| { | |
| "epoch": 0.0987012987012987, | |
| "grad_norm": 0.20423896610736847, | |
| "learning_rate": 0.0009911111769208864, | |
| "loss": 3.1159, | |
| "step": 7600 | |
| }, | |
| { | |
| "epoch": 0.1, | |
| "grad_norm": 0.18055465817451477, | |
| "learning_rate": 0.000990838486875789, | |
| "loss": 3.0872, | |
| "step": 7700 | |
| }, | |
| { | |
| "epoch": 0.1012987012987013, | |
| "grad_norm": 0.18993014097213745, | |
| "learning_rate": 0.0009905617156710437, | |
| "loss": 3.0976, | |
| "step": 7800 | |
| }, | |
| { | |
| "epoch": 0.1025974025974026, | |
| "grad_norm": 0.1896851807832718, | |
| "learning_rate": 0.000990280865607912, | |
| "loss": 3.0932, | |
| "step": 7900 | |
| }, | |
| { | |
| "epoch": 0.1038961038961039, | |
| "grad_norm": 0.18168048560619354, | |
| "learning_rate": 0.0009899959390215689, | |
| "loss": 3.0862, | |
| "step": 8000 | |
| }, | |
| { | |
| "epoch": 0.1038961038961039, | |
| "eval_loss": 3.4424216747283936, | |
| "eval_runtime": 12.6555, | |
| "eval_samples_per_second": 45.514, | |
| "eval_steps_per_second": 11.378, | |
| "step": 8000 | |
| }, | |
| { | |
| "epoch": 0.10519480519480519, | |
| "grad_norm": 0.19026321172714233, | |
| "learning_rate": 0.000989706938281085, | |
| "loss": 3.0792, | |
| "step": 8100 | |
| }, | |
| { | |
| "epoch": 0.10649350649350649, | |
| "grad_norm": 0.22636525332927704, | |
| "learning_rate": 0.0009894138657894054, | |
| "loss": 3.0808, | |
| "step": 8200 | |
| }, | |
| { | |
| "epoch": 0.10779220779220779, | |
| "grad_norm": 0.1759577989578247, | |
| "learning_rate": 0.0009891167239833311, | |
| "loss": 3.0651, | |
| "step": 8300 | |
| }, | |
| { | |
| "epoch": 0.10909090909090909, | |
| "grad_norm": 0.19504410028457642, | |
| "learning_rate": 0.0009888155153334984, | |
| "loss": 3.0701, | |
| "step": 8400 | |
| }, | |
| { | |
| "epoch": 0.11038961038961038, | |
| "grad_norm": 0.20704291760921478, | |
| "learning_rate": 0.000988510242344357, | |
| "loss": 3.072, | |
| "step": 8500 | |
| }, | |
| { | |
| "epoch": 0.11168831168831168, | |
| "grad_norm": 0.19917143881320953, | |
| "learning_rate": 0.000988200907554151, | |
| "loss": 3.0716, | |
| "step": 8600 | |
| }, | |
| { | |
| "epoch": 0.11298701298701298, | |
| "grad_norm": 0.1709250509738922, | |
| "learning_rate": 0.000987887513534897, | |
| "loss": 3.0622, | |
| "step": 8700 | |
| }, | |
| { | |
| "epoch": 0.11428571428571428, | |
| "grad_norm": 0.2797200381755829, | |
| "learning_rate": 0.0009875700628923622, | |
| "loss": 3.0743, | |
| "step": 8800 | |
| }, | |
| { | |
| "epoch": 0.11558441558441558, | |
| "grad_norm": 0.172173872590065, | |
| "learning_rate": 0.000987248558266044, | |
| "loss": 3.0676, | |
| "step": 8900 | |
| }, | |
| { | |
| "epoch": 0.11688311688311688, | |
| "grad_norm": 0.1997879594564438, | |
| "learning_rate": 0.000986923002329147, | |
| "loss": 3.0591, | |
| "step": 9000 | |
| }, | |
| { | |
| "epoch": 0.11688311688311688, | |
| "eval_loss": 3.424792766571045, | |
| "eval_runtime": 15.3808, | |
| "eval_samples_per_second": 37.449, | |
| "eval_steps_per_second": 9.362, | |
| "step": 9000 | |
| }, | |
| { | |
| "epoch": 0.11818181818181818, | |
| "grad_norm": 0.16804350912570953, | |
| "learning_rate": 0.0009865933977885612, | |
| "loss": 3.0642, | |
| "step": 9100 | |
| }, | |
| { | |
| "epoch": 0.11948051948051948, | |
| "grad_norm": 0.1866862177848816, | |
| "learning_rate": 0.0009862597473848393, | |
| "loss": 3.0378, | |
| "step": 9200 | |
| }, | |
| { | |
| "epoch": 0.12077922077922078, | |
| "grad_norm": 0.2142857313156128, | |
| "learning_rate": 0.000985922053892174, | |
| "loss": 3.057, | |
| "step": 9300 | |
| }, | |
| { | |
| "epoch": 0.12207792207792208, | |
| "grad_norm": 0.1772162914276123, | |
| "learning_rate": 0.0009855803201183743, | |
| "loss": 3.0585, | |
| "step": 9400 | |
| }, | |
| { | |
| "epoch": 0.12337662337662338, | |
| "grad_norm": 0.1796869933605194, | |
| "learning_rate": 0.0009852345489048447, | |
| "loss": 3.0228, | |
| "step": 9500 | |
| }, | |
| { | |
| "epoch": 0.12467532467532468, | |
| "grad_norm": 0.19519853591918945, | |
| "learning_rate": 0.0009848847431265576, | |
| "loss": 3.0493, | |
| "step": 9600 | |
| }, | |
| { | |
| "epoch": 0.12597402597402596, | |
| "grad_norm": 0.1866455227136612, | |
| "learning_rate": 0.0009845309056920326, | |
| "loss": 3.0369, | |
| "step": 9700 | |
| }, | |
| { | |
| "epoch": 0.12727272727272726, | |
| "grad_norm": 0.1888640969991684, | |
| "learning_rate": 0.000984173039543311, | |
| "loss": 3.0388, | |
| "step": 9800 | |
| }, | |
| { | |
| "epoch": 0.12857142857142856, | |
| "grad_norm": 0.1753017008304596, | |
| "learning_rate": 0.0009838111476559313, | |
| "loss": 3.0533, | |
| "step": 9900 | |
| }, | |
| { | |
| "epoch": 0.12987012987012986, | |
| "grad_norm": 0.17457301914691925, | |
| "learning_rate": 0.000983445233038905, | |
| "loss": 3.0569, | |
| "step": 10000 | |
| }, | |
| { | |
| "epoch": 0.12987012987012986, | |
| "eval_loss": 3.401273012161255, | |
| "eval_runtime": 14.6757, | |
| "eval_samples_per_second": 39.248, | |
| "eval_steps_per_second": 9.812, | |
| "step": 10000 | |
| }, | |
| { | |
| "epoch": 0.13116883116883116, | |
| "grad_norm": 0.17421157658100128, | |
| "learning_rate": 0.0009830752987346908, | |
| "loss": 3.0596, | |
| "step": 10100 | |
| }, | |
| { | |
| "epoch": 0.13246753246753246, | |
| "grad_norm": 0.1772899180650711, | |
| "learning_rate": 0.0009827013478191703, | |
| "loss": 3.016, | |
| "step": 10200 | |
| }, | |
| { | |
| "epoch": 0.13376623376623376, | |
| "grad_norm": 0.169020414352417, | |
| "learning_rate": 0.0009823233834016214, | |
| "loss": 3.0173, | |
| "step": 10300 | |
| }, | |
| { | |
| "epoch": 0.13506493506493505, | |
| "grad_norm": 0.34023940563201904, | |
| "learning_rate": 0.0009819414086246938, | |
| "loss": 3.0004, | |
| "step": 10400 | |
| }, | |
| { | |
| "epoch": 0.13636363636363635, | |
| "grad_norm": 0.19924494624137878, | |
| "learning_rate": 0.0009815554266643808, | |
| "loss": 3.003, | |
| "step": 10500 | |
| }, | |
| { | |
| "epoch": 0.13766233766233765, | |
| "grad_norm": 0.1713639199733734, | |
| "learning_rate": 0.0009811654407299948, | |
| "loss": 2.9895, | |
| "step": 10600 | |
| }, | |
| { | |
| "epoch": 0.13896103896103895, | |
| "grad_norm": 0.16809602081775665, | |
| "learning_rate": 0.00098077145406414, | |
| "loss": 3.0078, | |
| "step": 10700 | |
| }, | |
| { | |
| "epoch": 0.14025974025974025, | |
| "grad_norm": 0.18249379098415375, | |
| "learning_rate": 0.0009803734699426853, | |
| "loss": 3.0379, | |
| "step": 10800 | |
| }, | |
| { | |
| "epoch": 0.14155844155844155, | |
| "grad_norm": 0.19280089437961578, | |
| "learning_rate": 0.0009799714916747368, | |
| "loss": 2.9917, | |
| "step": 10900 | |
| }, | |
| { | |
| "epoch": 0.14285714285714285, | |
| "grad_norm": 0.18399550020694733, | |
| "learning_rate": 0.000979565522602611, | |
| "loss": 3.0597, | |
| "step": 11000 | |
| }, | |
| { | |
| "epoch": 0.14285714285714285, | |
| "eval_loss": 3.4125261306762695, | |
| "eval_runtime": 15.5875, | |
| "eval_samples_per_second": 36.953, | |
| "eval_steps_per_second": 9.238, | |
| "step": 11000 | |
| }, | |
| { | |
| "epoch": 0.14415584415584415, | |
| "grad_norm": 0.17574363946914673, | |
| "learning_rate": 0.000979155566101806, | |
| "loss": 3.0168, | |
| "step": 11100 | |
| }, | |
| { | |
| "epoch": 0.14545454545454545, | |
| "grad_norm": 0.17517109215259552, | |
| "learning_rate": 0.0009787416255809752, | |
| "loss": 2.9817, | |
| "step": 11200 | |
| }, | |
| { | |
| "epoch": 0.14675324675324675, | |
| "grad_norm": 0.18359977006912231, | |
| "learning_rate": 0.0009783237044818968, | |
| "loss": 2.9916, | |
| "step": 11300 | |
| }, | |
| { | |
| "epoch": 0.14805194805194805, | |
| "grad_norm": 0.17518875002861023, | |
| "learning_rate": 0.000977901806279446, | |
| "loss": 2.9567, | |
| "step": 11400 | |
| }, | |
| { | |
| "epoch": 0.14935064935064934, | |
| "grad_norm": 0.19247221946716309, | |
| "learning_rate": 0.0009774759344815674, | |
| "loss": 2.9901, | |
| "step": 11500 | |
| }, | |
| { | |
| "epoch": 0.15064935064935064, | |
| "grad_norm": 0.18513117730617523, | |
| "learning_rate": 0.000977046092629244, | |
| "loss": 2.992, | |
| "step": 11600 | |
| }, | |
| { | |
| "epoch": 0.15194805194805194, | |
| "grad_norm": 0.18412470817565918, | |
| "learning_rate": 0.0009766122842964683, | |
| "loss": 2.9985, | |
| "step": 11700 | |
| }, | |
| { | |
| "epoch": 0.15324675324675324, | |
| "grad_norm": 0.19853200018405914, | |
| "learning_rate": 0.0009761745130902134, | |
| "loss": 2.978, | |
| "step": 11800 | |
| }, | |
| { | |
| "epoch": 0.15454545454545454, | |
| "grad_norm": 0.19611628353595734, | |
| "learning_rate": 0.0009757327826504022, | |
| "loss": 2.9771, | |
| "step": 11900 | |
| }, | |
| { | |
| "epoch": 0.15584415584415584, | |
| "grad_norm": 0.17522920668125153, | |
| "learning_rate": 0.0009752870966498766, | |
| "loss": 2.9641, | |
| "step": 12000 | |
| }, | |
| { | |
| "epoch": 0.15584415584415584, | |
| "eval_loss": 3.3487231731414795, | |
| "eval_runtime": 13.9045, | |
| "eval_samples_per_second": 41.425, | |
| "eval_steps_per_second": 10.356, | |
| "step": 12000 | |
| }, | |
| { | |
| "epoch": 0.15714285714285714, | |
| "grad_norm": 0.18391777575016022, | |
| "learning_rate": 0.0009748374587943688, | |
| "loss": 2.9694, | |
| "step": 12100 | |
| }, | |
| { | |
| "epoch": 0.15844155844155844, | |
| "grad_norm": 0.20561572909355164, | |
| "learning_rate": 0.0009743838728224687, | |
| "loss": 3.0187, | |
| "step": 12200 | |
| }, | |
| { | |
| "epoch": 0.15974025974025974, | |
| "grad_norm": 0.17854483425617218, | |
| "learning_rate": 0.0009739263425055934, | |
| "loss": 2.9687, | |
| "step": 12300 | |
| }, | |
| { | |
| "epoch": 0.16103896103896104, | |
| "grad_norm": 0.18647781014442444, | |
| "learning_rate": 0.0009734648716479563, | |
| "loss": 2.9574, | |
| "step": 12400 | |
| }, | |
| { | |
| "epoch": 0.16233766233766234, | |
| "grad_norm": 0.1802452653646469, | |
| "learning_rate": 0.0009729994640865349, | |
| "loss": 2.9812, | |
| "step": 12500 | |
| }, | |
| { | |
| "epoch": 0.16363636363636364, | |
| "grad_norm": 0.18787842988967896, | |
| "learning_rate": 0.0009725301236910393, | |
| "loss": 2.9771, | |
| "step": 12600 | |
| }, | |
| { | |
| "epoch": 0.16493506493506493, | |
| "grad_norm": 0.18528838455677032, | |
| "learning_rate": 0.0009720568543638793, | |
| "loss": 2.9525, | |
| "step": 12700 | |
| }, | |
| { | |
| "epoch": 0.16623376623376623, | |
| "grad_norm": 0.17883966863155365, | |
| "learning_rate": 0.000971579660040133, | |
| "loss": 2.9624, | |
| "step": 12800 | |
| }, | |
| { | |
| "epoch": 0.16753246753246753, | |
| "grad_norm": 0.17459270358085632, | |
| "learning_rate": 0.0009710985446875134, | |
| "loss": 2.9696, | |
| "step": 12900 | |
| }, | |
| { | |
| "epoch": 0.16883116883116883, | |
| "grad_norm": 0.1830231249332428, | |
| "learning_rate": 0.0009706135123063352, | |
| "loss": 2.9609, | |
| "step": 13000 | |
| }, | |
| { | |
| "epoch": 0.16883116883116883, | |
| "eval_loss": 3.337019920349121, | |
| "eval_runtime": 15.8318, | |
| "eval_samples_per_second": 36.383, | |
| "eval_steps_per_second": 9.096, | |
| "step": 13000 | |
| }, | |
| { | |
| "epoch": 0.17012987012987013, | |
| "grad_norm": 0.20791789889335632, | |
| "learning_rate": 0.0009701245669294825, | |
| "loss": 2.9953, | |
| "step": 13100 | |
| }, | |
| { | |
| "epoch": 0.17142857142857143, | |
| "grad_norm": 0.20302486419677734, | |
| "learning_rate": 0.0009696317126223742, | |
| "loss": 3.0519, | |
| "step": 13200 | |
| }, | |
| { | |
| "epoch": 0.17272727272727273, | |
| "grad_norm": 0.17939983308315277, | |
| "learning_rate": 0.000969134953482931, | |
| "loss": 2.9618, | |
| "step": 13300 | |
| }, | |
| { | |
| "epoch": 0.17402597402597403, | |
| "grad_norm": 0.2437734603881836, | |
| "learning_rate": 0.0009686342936415407, | |
| "loss": 2.955, | |
| "step": 13400 | |
| }, | |
| { | |
| "epoch": 0.17532467532467533, | |
| "grad_norm": 0.19909310340881348, | |
| "learning_rate": 0.000968129737261024, | |
| "loss": 2.9898, | |
| "step": 13500 | |
| }, | |
| { | |
| "epoch": 0.17662337662337663, | |
| "grad_norm": 0.1696212738752365, | |
| "learning_rate": 0.000967621288536601, | |
| "loss": 2.9793, | |
| "step": 13600 | |
| }, | |
| { | |
| "epoch": 0.17792207792207793, | |
| "grad_norm": 0.171333447098732, | |
| "learning_rate": 0.0009671089516958538, | |
| "loss": 2.942, | |
| "step": 13700 | |
| }, | |
| { | |
| "epoch": 0.17922077922077922, | |
| "grad_norm": 0.17436903715133667, | |
| "learning_rate": 0.0009665927309986944, | |
| "loss": 2.9574, | |
| "step": 13800 | |
| }, | |
| { | |
| "epoch": 0.18051948051948052, | |
| "grad_norm": 0.20607399940490723, | |
| "learning_rate": 0.0009660726307373266, | |
| "loss": 2.965, | |
| "step": 13900 | |
| }, | |
| { | |
| "epoch": 0.18181818181818182, | |
| "grad_norm": 0.17206591367721558, | |
| "learning_rate": 0.0009655486552362127, | |
| "loss": 2.9532, | |
| "step": 14000 | |
| }, | |
| { | |
| "epoch": 0.18181818181818182, | |
| "eval_loss": 3.327735424041748, | |
| "eval_runtime": 13.5364, | |
| "eval_samples_per_second": 42.552, | |
| "eval_steps_per_second": 10.638, | |
| "step": 14000 | |
| }, | |
| { | |
| "epoch": 0.18311688311688312, | |
| "grad_norm": 0.20287004113197327, | |
| "learning_rate": 0.000965020808852035, | |
| "loss": 2.9153, | |
| "step": 14100 | |
| }, | |
| { | |
| "epoch": 0.18441558441558442, | |
| "grad_norm": 0.18036919832229614, | |
| "learning_rate": 0.0009644890959736619, | |
| "loss": 2.933, | |
| "step": 14200 | |
| }, | |
| { | |
| "epoch": 0.18571428571428572, | |
| "grad_norm": 0.21330875158309937, | |
| "learning_rate": 0.0009639535210221099, | |
| "loss": 2.9773, | |
| "step": 14300 | |
| }, | |
| { | |
| "epoch": 0.18701298701298702, | |
| "grad_norm": 0.20942148566246033, | |
| "learning_rate": 0.000963414088450508, | |
| "loss": 2.9225, | |
| "step": 14400 | |
| }, | |
| { | |
| "epoch": 0.18831168831168832, | |
| "grad_norm": 0.188313826918602, | |
| "learning_rate": 0.0009628708027440592, | |
| "loss": 2.9409, | |
| "step": 14500 | |
| }, | |
| { | |
| "epoch": 0.18961038961038962, | |
| "grad_norm": 0.17884506285190582, | |
| "learning_rate": 0.0009623236684200043, | |
| "loss": 2.9635, | |
| "step": 14600 | |
| }, | |
| { | |
| "epoch": 0.19090909090909092, | |
| "grad_norm": 0.17130698263645172, | |
| "learning_rate": 0.0009617726900275845, | |
| "loss": 2.9447, | |
| "step": 14700 | |
| }, | |
| { | |
| "epoch": 0.19220779220779222, | |
| "grad_norm": 0.19943124055862427, | |
| "learning_rate": 0.0009612178721480027, | |
| "loss": 2.9525, | |
| "step": 14800 | |
| }, | |
| { | |
| "epoch": 0.19350649350649352, | |
| "grad_norm": 0.18249951303005219, | |
| "learning_rate": 0.0009606592193943861, | |
| "loss": 2.9405, | |
| "step": 14900 | |
| }, | |
| { | |
| "epoch": 0.19480519480519481, | |
| "grad_norm": 0.17710338532924652, | |
| "learning_rate": 0.0009600967364117478, | |
| "loss": 2.921, | |
| "step": 15000 | |
| }, | |
| { | |
| "epoch": 0.19480519480519481, | |
| "eval_loss": 3.307467460632324, | |
| "eval_runtime": 15.2964, | |
| "eval_samples_per_second": 37.656, | |
| "eval_steps_per_second": 9.414, | |
| "step": 15000 | |
| }, | |
| { | |
| "epoch": 0.1961038961038961, | |
| "grad_norm": 0.2033987194299698, | |
| "learning_rate": 0.0009595304278769472, | |
| "loss": 2.9253, | |
| "step": 15100 | |
| }, | |
| { | |
| "epoch": 0.1974025974025974, | |
| "grad_norm": 0.17457351088523865, | |
| "learning_rate": 0.000958960298498653, | |
| "loss": 2.9183, | |
| "step": 15200 | |
| }, | |
| { | |
| "epoch": 0.1987012987012987, | |
| "grad_norm": 0.2011554092168808, | |
| "learning_rate": 0.0009583863530173018, | |
| "loss": 2.9254, | |
| "step": 15300 | |
| }, | |
| { | |
| "epoch": 0.2, | |
| "grad_norm": 0.18333646655082703, | |
| "learning_rate": 0.0009578085962050609, | |
| "loss": 2.9383, | |
| "step": 15400 | |
| }, | |
| { | |
| "epoch": 0.2012987012987013, | |
| "grad_norm": 0.167351633310318, | |
| "learning_rate": 0.0009572270328657869, | |
| "loss": 2.9435, | |
| "step": 15500 | |
| }, | |
| { | |
| "epoch": 0.2025974025974026, | |
| "grad_norm": 0.17198020219802856, | |
| "learning_rate": 0.0009566416678349864, | |
| "loss": 2.9336, | |
| "step": 15600 | |
| }, | |
| { | |
| "epoch": 0.2038961038961039, | |
| "grad_norm": 0.19012530148029327, | |
| "learning_rate": 0.0009560525059797762, | |
| "loss": 2.9353, | |
| "step": 15700 | |
| }, | |
| { | |
| "epoch": 0.2051948051948052, | |
| "grad_norm": 0.18222008645534515, | |
| "learning_rate": 0.0009554595521988423, | |
| "loss": 2.934, | |
| "step": 15800 | |
| }, | |
| { | |
| "epoch": 0.2064935064935065, | |
| "grad_norm": 0.18934360146522522, | |
| "learning_rate": 0.0009548628114223989, | |
| "loss": 2.902, | |
| "step": 15900 | |
| }, | |
| { | |
| "epoch": 0.2077922077922078, | |
| "grad_norm": 0.17503845691680908, | |
| "learning_rate": 0.0009542622886121486, | |
| "loss": 2.9485, | |
| "step": 16000 | |
| }, | |
| { | |
| "epoch": 0.2077922077922078, | |
| "eval_loss": 3.305332660675049, | |
| "eval_runtime": 14.5112, | |
| "eval_samples_per_second": 39.693, | |
| "eval_steps_per_second": 9.923, | |
| "step": 16000 | |
| }, | |
| { | |
| "epoch": 0.20909090909090908, | |
| "grad_norm": 0.19923605024814606, | |
| "learning_rate": 0.0009536579887612396, | |
| "loss": 2.9049, | |
| "step": 16100 | |
| }, | |
| { | |
| "epoch": 0.21038961038961038, | |
| "grad_norm": 0.1969510018825531, | |
| "learning_rate": 0.0009530499168942252, | |
| "loss": 2.9263, | |
| "step": 16200 | |
| }, | |
| { | |
| "epoch": 0.21168831168831168, | |
| "grad_norm": 0.3835534155368805, | |
| "learning_rate": 0.0009524380780670223, | |
| "loss": 2.9207, | |
| "step": 16300 | |
| }, | |
| { | |
| "epoch": 0.21298701298701297, | |
| "grad_norm": 0.18829414248466492, | |
| "learning_rate": 0.0009518224773668678, | |
| "loss": 2.9247, | |
| "step": 16400 | |
| }, | |
| { | |
| "epoch": 0.21428571428571427, | |
| "grad_norm": 0.18950419127941132, | |
| "learning_rate": 0.0009512031199122779, | |
| "loss": 2.914, | |
| "step": 16500 | |
| }, | |
| { | |
| "epoch": 0.21558441558441557, | |
| "grad_norm": 0.23597006499767303, | |
| "learning_rate": 0.0009505800108530054, | |
| "loss": 2.9064, | |
| "step": 16600 | |
| }, | |
| { | |
| "epoch": 0.21688311688311687, | |
| "grad_norm": 0.18245990574359894, | |
| "learning_rate": 0.0009499531553699957, | |
| "loss": 2.9219, | |
| "step": 16700 | |
| }, | |
| { | |
| "epoch": 0.21818181818181817, | |
| "grad_norm": 0.20525243878364563, | |
| "learning_rate": 0.0009493225586753449, | |
| "loss": 2.9373, | |
| "step": 16800 | |
| }, | |
| { | |
| "epoch": 0.21948051948051947, | |
| "grad_norm": 0.17929033935070038, | |
| "learning_rate": 0.0009486882260122557, | |
| "loss": 2.8947, | |
| "step": 16900 | |
| }, | |
| { | |
| "epoch": 0.22077922077922077, | |
| "grad_norm": 0.18927592039108276, | |
| "learning_rate": 0.0009480501626549945, | |
| "loss": 2.903, | |
| "step": 17000 | |
| }, | |
| { | |
| "epoch": 0.22077922077922077, | |
| "eval_loss": 3.2817673683166504, | |
| "eval_runtime": 16.3954, | |
| "eval_samples_per_second": 35.132, | |
| "eval_steps_per_second": 8.783, | |
| "step": 17000 | |
| }, | |
| { | |
| "epoch": 0.22207792207792207, | |
| "grad_norm": 0.22612017393112183, | |
| "learning_rate": 0.0009474083739088471, | |
| "loss": 2.91, | |
| "step": 17100 | |
| }, | |
| { | |
| "epoch": 0.22337662337662337, | |
| "grad_norm": 0.19862578809261322, | |
| "learning_rate": 0.0009467628651100745, | |
| "loss": 2.9319, | |
| "step": 17200 | |
| }, | |
| { | |
| "epoch": 0.22467532467532467, | |
| "grad_norm": 0.19449101388454437, | |
| "learning_rate": 0.0009461136416258689, | |
| "loss": 2.9244, | |
| "step": 17300 | |
| }, | |
| { | |
| "epoch": 0.22597402597402597, | |
| "grad_norm": 0.18568117916584015, | |
| "learning_rate": 0.0009454607088543088, | |
| "loss": 2.9118, | |
| "step": 17400 | |
| }, | |
| { | |
| "epoch": 0.22727272727272727, | |
| "grad_norm": 0.174198180437088, | |
| "learning_rate": 0.000944804072224314, | |
| "loss": 2.9134, | |
| "step": 17500 | |
| }, | |
| { | |
| "epoch": 0.22857142857142856, | |
| "grad_norm": 0.18078738451004028, | |
| "learning_rate": 0.0009441437371956013, | |
| "loss": 2.9039, | |
| "step": 17600 | |
| }, | |
| { | |
| "epoch": 0.22987012987012986, | |
| "grad_norm": 0.20317985117435455, | |
| "learning_rate": 0.0009434797092586377, | |
| "loss": 2.8848, | |
| "step": 17700 | |
| }, | |
| { | |
| "epoch": 0.23116883116883116, | |
| "grad_norm": 0.17165842652320862, | |
| "learning_rate": 0.0009428119939345959, | |
| "loss": 2.8846, | |
| "step": 17800 | |
| }, | |
| { | |
| "epoch": 0.23246753246753246, | |
| "grad_norm": 0.21153512597084045, | |
| "learning_rate": 0.0009421405967753078, | |
| "loss": 2.8944, | |
| "step": 17900 | |
| }, | |
| { | |
| "epoch": 0.23376623376623376, | |
| "grad_norm": 0.22300300002098083, | |
| "learning_rate": 0.000941465523363219, | |
| "loss": 2.8981, | |
| "step": 18000 | |
| }, | |
| { | |
| "epoch": 0.23376623376623376, | |
| "eval_loss": 3.2792882919311523, | |
| "eval_runtime": 15.157, | |
| "eval_samples_per_second": 38.002, | |
| "eval_steps_per_second": 9.501, | |
| "step": 18000 | |
| }, | |
| { | |
| "epoch": 0.23506493506493506, | |
| "grad_norm": 0.17719519138336182, | |
| "learning_rate": 0.0009407867793113414, | |
| "loss": 2.9096, | |
| "step": 18100 | |
| }, | |
| { | |
| "epoch": 0.23636363636363636, | |
| "grad_norm": 0.17834420502185822, | |
| "learning_rate": 0.0009401043702632071, | |
| "loss": 2.8922, | |
| "step": 18200 | |
| }, | |
| { | |
| "epoch": 0.23766233766233766, | |
| "grad_norm": 0.20472975075244904, | |
| "learning_rate": 0.000939418301892822, | |
| "loss": 2.8927, | |
| "step": 18300 | |
| }, | |
| { | |
| "epoch": 0.23896103896103896, | |
| "grad_norm": 0.1972654014825821, | |
| "learning_rate": 0.0009387285799046174, | |
| "loss": 2.904, | |
| "step": 18400 | |
| }, | |
| { | |
| "epoch": 0.24025974025974026, | |
| "grad_norm": 0.1871407926082611, | |
| "learning_rate": 0.0009380352100334035, | |
| "loss": 2.8829, | |
| "step": 18500 | |
| }, | |
| { | |
| "epoch": 0.24155844155844156, | |
| "grad_norm": 0.1807365119457245, | |
| "learning_rate": 0.0009373381980443212, | |
| "loss": 2.8977, | |
| "step": 18600 | |
| }, | |
| { | |
| "epoch": 0.24285714285714285, | |
| "grad_norm": 0.19222930073738098, | |
| "learning_rate": 0.000936637549732795, | |
| "loss": 2.8991, | |
| "step": 18700 | |
| }, | |
| { | |
| "epoch": 0.24415584415584415, | |
| "grad_norm": 0.2098565548658371, | |
| "learning_rate": 0.0009359332709244837, | |
| "loss": 2.8827, | |
| "step": 18800 | |
| }, | |
| { | |
| "epoch": 0.24545454545454545, | |
| "grad_norm": 0.2327863723039627, | |
| "learning_rate": 0.0009352253674752325, | |
| "loss": 2.8909, | |
| "step": 18900 | |
| }, | |
| { | |
| "epoch": 0.24675324675324675, | |
| "grad_norm": 0.1913413554430008, | |
| "learning_rate": 0.0009345138452710245, | |
| "loss": 2.8913, | |
| "step": 19000 | |
| }, | |
| { | |
| "epoch": 0.24675324675324675, | |
| "eval_loss": 3.2709906101226807, | |
| "eval_runtime": 14.4105, | |
| "eval_samples_per_second": 39.971, | |
| "eval_steps_per_second": 9.993, | |
| "step": 19000 | |
| }, | |
| { | |
| "epoch": 0.24805194805194805, | |
| "grad_norm": 0.17969286441802979, | |
| "learning_rate": 0.0009337987102279313, | |
| "loss": 2.8998, | |
| "step": 19100 | |
| }, | |
| { | |
| "epoch": 0.24935064935064935, | |
| "grad_norm": 0.18576957285404205, | |
| "learning_rate": 0.0009330799682920645, | |
| "loss": 2.8886, | |
| "step": 19200 | |
| }, | |
| { | |
| "epoch": 0.2506493506493506, | |
| "grad_norm": 0.17968888580799103, | |
| "learning_rate": 0.0009323576254395253, | |
| "loss": 2.8998, | |
| "step": 19300 | |
| }, | |
| { | |
| "epoch": 0.2519480519480519, | |
| "grad_norm": 0.1743050366640091, | |
| "learning_rate": 0.0009316316876763557, | |
| "loss": 2.8837, | |
| "step": 19400 | |
| }, | |
| { | |
| "epoch": 0.2532467532467532, | |
| "grad_norm": 0.17573101818561554, | |
| "learning_rate": 0.0009309021610384879, | |
| "loss": 2.8681, | |
| "step": 19500 | |
| }, | |
| { | |
| "epoch": 0.2545454545454545, | |
| "grad_norm": 0.199496790766716, | |
| "learning_rate": 0.0009301690515916948, | |
| "loss": 2.8717, | |
| "step": 19600 | |
| }, | |
| { | |
| "epoch": 0.2558441558441558, | |
| "grad_norm": 0.1846366673707962, | |
| "learning_rate": 0.0009294323654315387, | |
| "loss": 2.9166, | |
| "step": 19700 | |
| }, | |
| { | |
| "epoch": 0.2571428571428571, | |
| "grad_norm": 0.18776021897792816, | |
| "learning_rate": 0.0009286921086833214, | |
| "loss": 2.8916, | |
| "step": 19800 | |
| }, | |
| { | |
| "epoch": 0.2584415584415584, | |
| "grad_norm": 0.1786223202943802, | |
| "learning_rate": 0.000927948287502033, | |
| "loss": 2.9013, | |
| "step": 19900 | |
| }, | |
| { | |
| "epoch": 0.2597402597402597, | |
| "grad_norm": 0.18665452301502228, | |
| "learning_rate": 0.0009272009080723005, | |
| "loss": 2.9108, | |
| "step": 20000 | |
| }, | |
| { | |
| "epoch": 0.2597402597402597, | |
| "eval_loss": 3.2676138877868652, | |
| "eval_runtime": 15.0403, | |
| "eval_samples_per_second": 38.297, | |
| "eval_steps_per_second": 9.574, | |
| "step": 20000 | |
| }, | |
| { | |
| "epoch": 0.261038961038961, | |
| "grad_norm": 0.1809331625699997, | |
| "learning_rate": 0.0009264499766083365, | |
| "loss": 2.8836, | |
| "step": 20100 | |
| }, | |
| { | |
| "epoch": 0.2623376623376623, | |
| "grad_norm": 0.19160114228725433, | |
| "learning_rate": 0.0009256954993538877, | |
| "loss": 2.8881, | |
| "step": 20200 | |
| }, | |
| { | |
| "epoch": 0.2636363636363636, | |
| "grad_norm": 0.1818685233592987, | |
| "learning_rate": 0.0009249374825821832, | |
| "loss": 2.8543, | |
| "step": 20300 | |
| }, | |
| { | |
| "epoch": 0.2649350649350649, | |
| "grad_norm": 0.1778269112110138, | |
| "learning_rate": 0.0009241759325958815, | |
| "loss": 2.8741, | |
| "step": 20400 | |
| }, | |
| { | |
| "epoch": 0.2662337662337662, | |
| "grad_norm": 0.1888059824705124, | |
| "learning_rate": 0.0009234108557270187, | |
| "loss": 2.8772, | |
| "step": 20500 | |
| }, | |
| { | |
| "epoch": 0.2675324675324675, | |
| "grad_norm": 0.22666007280349731, | |
| "learning_rate": 0.000922642258336956, | |
| "loss": 2.9021, | |
| "step": 20600 | |
| }, | |
| { | |
| "epoch": 0.2688311688311688, | |
| "grad_norm": 0.18350407481193542, | |
| "learning_rate": 0.0009218701468163267, | |
| "loss": 2.8845, | |
| "step": 20700 | |
| }, | |
| { | |
| "epoch": 0.2701298701298701, | |
| "grad_norm": 0.17511753737926483, | |
| "learning_rate": 0.000921094527584982, | |
| "loss": 2.9005, | |
| "step": 20800 | |
| }, | |
| { | |
| "epoch": 0.2714285714285714, | |
| "grad_norm": 0.18808013200759888, | |
| "learning_rate": 0.0009203154070919398, | |
| "loss": 2.8658, | |
| "step": 20900 | |
| }, | |
| { | |
| "epoch": 0.2727272727272727, | |
| "grad_norm": 0.18752652406692505, | |
| "learning_rate": 0.0009195327918153292, | |
| "loss": 2.8649, | |
| "step": 21000 | |
| }, | |
| { | |
| "epoch": 0.2727272727272727, | |
| "eval_loss": 3.2498960494995117, | |
| "eval_runtime": 15.6914, | |
| "eval_samples_per_second": 36.708, | |
| "eval_steps_per_second": 9.177, | |
| "step": 21000 | |
| }, | |
| { | |
| "epoch": 0.274025974025974, | |
| "grad_norm": 0.19540126621723175, | |
| "learning_rate": 0.0009187466882623372, | |
| "loss": 2.8755, | |
| "step": 21100 | |
| }, | |
| { | |
| "epoch": 0.2753246753246753, | |
| "grad_norm": 0.18758288025856018, | |
| "learning_rate": 0.0009179571029691546, | |
| "loss": 2.8737, | |
| "step": 21200 | |
| }, | |
| { | |
| "epoch": 0.2766233766233766, | |
| "grad_norm": 0.18555940687656403, | |
| "learning_rate": 0.0009171640425009224, | |
| "loss": 2.8626, | |
| "step": 21300 | |
| }, | |
| { | |
| "epoch": 0.2779220779220779, | |
| "grad_norm": 0.18831279873847961, | |
| "learning_rate": 0.0009163675134516758, | |
| "loss": 2.883, | |
| "step": 21400 | |
| }, | |
| { | |
| "epoch": 0.2792207792207792, | |
| "grad_norm": 0.19656601548194885, | |
| "learning_rate": 0.0009155675224442904, | |
| "loss": 2.8959, | |
| "step": 21500 | |
| }, | |
| { | |
| "epoch": 0.2805194805194805, | |
| "grad_norm": 0.18778946995735168, | |
| "learning_rate": 0.0009147640761304266, | |
| "loss": 2.8723, | |
| "step": 21600 | |
| }, | |
| { | |
| "epoch": 0.2818181818181818, | |
| "grad_norm": 0.1975911557674408, | |
| "learning_rate": 0.000913957181190475, | |
| "loss": 2.8724, | |
| "step": 21700 | |
| }, | |
| { | |
| "epoch": 0.2831168831168831, | |
| "grad_norm": 0.18618452548980713, | |
| "learning_rate": 0.0009131468443334998, | |
| "loss": 2.8597, | |
| "step": 21800 | |
| }, | |
| { | |
| "epoch": 0.2844155844155844, | |
| "grad_norm": 0.19419926404953003, | |
| "learning_rate": 0.0009123330722971841, | |
| "loss": 2.8398, | |
| "step": 21900 | |
| }, | |
| { | |
| "epoch": 0.2857142857142857, | |
| "grad_norm": 0.1885763257741928, | |
| "learning_rate": 0.0009115158718477732, | |
| "loss": 2.875, | |
| "step": 22000 | |
| }, | |
| { | |
| "epoch": 0.2857142857142857, | |
| "eval_loss": 3.2455015182495117, | |
| "eval_runtime": 14.7497, | |
| "eval_samples_per_second": 39.052, | |
| "eval_steps_per_second": 9.763, | |
| "step": 22000 | |
| }, | |
| { | |
| "epoch": 0.287012987012987, | |
| "grad_norm": 0.17821335792541504, | |
| "learning_rate": 0.0009106952497800183, | |
| "loss": 2.8378, | |
| "step": 22100 | |
| }, | |
| { | |
| "epoch": 0.2883116883116883, | |
| "grad_norm": 0.20162396132946014, | |
| "learning_rate": 0.0009098712129171207, | |
| "loss": 2.8512, | |
| "step": 22200 | |
| }, | |
| { | |
| "epoch": 0.2896103896103896, | |
| "grad_norm": 0.21083980798721313, | |
| "learning_rate": 0.0009090437681106742, | |
| "loss": 2.8587, | |
| "step": 22300 | |
| }, | |
| { | |
| "epoch": 0.2909090909090909, | |
| "grad_norm": 0.1923542320728302, | |
| "learning_rate": 0.0009082129222406087, | |
| "loss": 2.843, | |
| "step": 22400 | |
| }, | |
| { | |
| "epoch": 0.2922077922077922, | |
| "grad_norm": 0.19237105548381805, | |
| "learning_rate": 0.0009073786822151326, | |
| "loss": 2.8554, | |
| "step": 22500 | |
| }, | |
| { | |
| "epoch": 0.2935064935064935, | |
| "grad_norm": 0.1831800788640976, | |
| "learning_rate": 0.000906541054970676, | |
| "loss": 2.8928, | |
| "step": 22600 | |
| }, | |
| { | |
| "epoch": 0.2948051948051948, | |
| "grad_norm": 0.20052975416183472, | |
| "learning_rate": 0.000905700047471832, | |
| "loss": 2.8462, | |
| "step": 22700 | |
| }, | |
| { | |
| "epoch": 0.2961038961038961, | |
| "grad_norm": 0.20518037676811218, | |
| "learning_rate": 0.0009048556667113002, | |
| "loss": 2.8624, | |
| "step": 22800 | |
| }, | |
| { | |
| "epoch": 0.2974025974025974, | |
| "grad_norm": 0.18524277210235596, | |
| "learning_rate": 0.0009040079197098268, | |
| "loss": 2.8683, | |
| "step": 22900 | |
| }, | |
| { | |
| "epoch": 0.2987012987012987, | |
| "grad_norm": 0.18116071820259094, | |
| "learning_rate": 0.000903156813516148, | |
| "loss": 2.8335, | |
| "step": 23000 | |
| }, | |
| { | |
| "epoch": 0.2987012987012987, | |
| "eval_loss": 3.236686944961548, | |
| "eval_runtime": 16.906, | |
| "eval_samples_per_second": 34.071, | |
| "eval_steps_per_second": 8.518, | |
| "step": 23000 | |
| }, | |
| { | |
| "epoch": 0.3, | |
| "grad_norm": 0.2058299481868744, | |
| "learning_rate": 0.0009023023552069303, | |
| "loss": 2.8548, | |
| "step": 23100 | |
| }, | |
| { | |
| "epoch": 0.3012987012987013, | |
| "grad_norm": 0.21021772921085358, | |
| "learning_rate": 0.0009014445518867116, | |
| "loss": 2.8435, | |
| "step": 23200 | |
| }, | |
| { | |
| "epoch": 0.3025974025974026, | |
| "grad_norm": 0.18895122408866882, | |
| "learning_rate": 0.000900583410687843, | |
| "loss": 2.839, | |
| "step": 23300 | |
| }, | |
| { | |
| "epoch": 0.3038961038961039, | |
| "grad_norm": 0.18983018398284912, | |
| "learning_rate": 0.0008997189387704286, | |
| "loss": 2.843, | |
| "step": 23400 | |
| }, | |
| { | |
| "epoch": 0.3051948051948052, | |
| "grad_norm": 0.21453820168972015, | |
| "learning_rate": 0.0008988511433222664, | |
| "loss": 2.8443, | |
| "step": 23500 | |
| }, | |
| { | |
| "epoch": 0.3064935064935065, | |
| "grad_norm": 0.1873568445444107, | |
| "learning_rate": 0.0008979800315587885, | |
| "loss": 2.8473, | |
| "step": 23600 | |
| }, | |
| { | |
| "epoch": 0.3077922077922078, | |
| "grad_norm": 0.19516156613826752, | |
| "learning_rate": 0.000897105610723001, | |
| "loss": 2.8579, | |
| "step": 23700 | |
| }, | |
| { | |
| "epoch": 0.3090909090909091, | |
| "grad_norm": 0.1961786150932312, | |
| "learning_rate": 0.000896227888085424, | |
| "loss": 2.8282, | |
| "step": 23800 | |
| }, | |
| { | |
| "epoch": 0.3103896103896104, | |
| "grad_norm": 0.19147805869579315, | |
| "learning_rate": 0.0008953468709440309, | |
| "loss": 2.8734, | |
| "step": 23900 | |
| }, | |
| { | |
| "epoch": 0.3116883116883117, | |
| "grad_norm": 0.20040926337242126, | |
| "learning_rate": 0.0008944625666241877, | |
| "loss": 2.8491, | |
| "step": 24000 | |
| }, | |
| { | |
| "epoch": 0.3116883116883117, | |
| "eval_loss": 3.2374658584594727, | |
| "eval_runtime": 16.1905, | |
| "eval_samples_per_second": 35.576, | |
| "eval_steps_per_second": 8.894, | |
| "step": 24000 | |
| }, | |
| { | |
| "epoch": 0.312987012987013, | |
| "grad_norm": 0.21797960996627808, | |
| "learning_rate": 0.0008935749824785922, | |
| "loss": 2.8992, | |
| "step": 24100 | |
| }, | |
| { | |
| "epoch": 0.3142857142857143, | |
| "grad_norm": 0.19554930925369263, | |
| "learning_rate": 0.0008926841258872131, | |
| "loss": 2.8771, | |
| "step": 24200 | |
| }, | |
| { | |
| "epoch": 0.3155844155844156, | |
| "grad_norm": 0.18446357548236847, | |
| "learning_rate": 0.0008917900042572283, | |
| "loss": 2.8753, | |
| "step": 24300 | |
| }, | |
| { | |
| "epoch": 0.3168831168831169, | |
| "grad_norm": 0.18406012654304504, | |
| "learning_rate": 0.0008908926250229633, | |
| "loss": 2.8819, | |
| "step": 24400 | |
| }, | |
| { | |
| "epoch": 0.3181818181818182, | |
| "grad_norm": 0.21075041592121124, | |
| "learning_rate": 0.0008899919956458296, | |
| "loss": 2.8796, | |
| "step": 24500 | |
| }, | |
| { | |
| "epoch": 0.3194805194805195, | |
| "grad_norm": 0.20150282979011536, | |
| "learning_rate": 0.0008890881236142624, | |
| "loss": 2.8611, | |
| "step": 24600 | |
| }, | |
| { | |
| "epoch": 0.3207792207792208, | |
| "grad_norm": 0.43753939867019653, | |
| "learning_rate": 0.0008881810164436588, | |
| "loss": 2.8587, | |
| "step": 24700 | |
| }, | |
| { | |
| "epoch": 0.3220779220779221, | |
| "grad_norm": 0.1800195425748825, | |
| "learning_rate": 0.0008872706816763148, | |
| "loss": 2.8797, | |
| "step": 24800 | |
| }, | |
| { | |
| "epoch": 0.3233766233766234, | |
| "grad_norm": 0.2071843296289444, | |
| "learning_rate": 0.0008863571268813628, | |
| "loss": 2.8697, | |
| "step": 24900 | |
| }, | |
| { | |
| "epoch": 0.3246753246753247, | |
| "grad_norm": 0.19320254027843475, | |
| "learning_rate": 0.0008854403596547088, | |
| "loss": 2.8938, | |
| "step": 25000 | |
| }, | |
| { | |
| "epoch": 0.3246753246753247, | |
| "eval_loss": 3.232848882675171, | |
| "eval_runtime": 15.6644, | |
| "eval_samples_per_second": 36.771, | |
| "eval_steps_per_second": 9.193, | |
| "step": 25000 | |
| }, | |
| { | |
| "epoch": 0.32597402597402597, | |
| "grad_norm": 0.21449397504329681, | |
| "learning_rate": 0.000884520387618969, | |
| "loss": 2.8374, | |
| "step": 25100 | |
| }, | |
| { | |
| "epoch": 0.32727272727272727, | |
| "grad_norm": 0.2025729864835739, | |
| "learning_rate": 0.0008835972184234066, | |
| "loss": 2.845, | |
| "step": 25200 | |
| }, | |
| { | |
| "epoch": 0.32857142857142857, | |
| "grad_norm": 0.22090762853622437, | |
| "learning_rate": 0.000882670859743868, | |
| "loss": 2.8518, | |
| "step": 25300 | |
| }, | |
| { | |
| "epoch": 0.32987012987012987, | |
| "grad_norm": 0.20626656711101532, | |
| "learning_rate": 0.0008817413192827191, | |
| "loss": 2.8299, | |
| "step": 25400 | |
| }, | |
| { | |
| "epoch": 0.33116883116883117, | |
| "grad_norm": 0.18527238070964813, | |
| "learning_rate": 0.0008808086047687813, | |
| "loss": 2.8532, | |
| "step": 25500 | |
| }, | |
| { | |
| "epoch": 0.33246753246753247, | |
| "grad_norm": 0.20959258079528809, | |
| "learning_rate": 0.0008798727239572676, | |
| "loss": 2.8656, | |
| "step": 25600 | |
| }, | |
| { | |
| "epoch": 0.33376623376623377, | |
| "grad_norm": 0.2122378945350647, | |
| "learning_rate": 0.000878933684629717, | |
| "loss": 2.8532, | |
| "step": 25700 | |
| }, | |
| { | |
| "epoch": 0.33506493506493507, | |
| "grad_norm": 0.18408456444740295, | |
| "learning_rate": 0.0008779914945939311, | |
| "loss": 2.8639, | |
| "step": 25800 | |
| }, | |
| { | |
| "epoch": 0.33636363636363636, | |
| "grad_norm": 0.20426695048809052, | |
| "learning_rate": 0.0008770461616839082, | |
| "loss": 2.865, | |
| "step": 25900 | |
| }, | |
| { | |
| "epoch": 0.33766233766233766, | |
| "grad_norm": 0.1808163821697235, | |
| "learning_rate": 0.0008760976937597787, | |
| "loss": 2.8481, | |
| "step": 26000 | |
| }, | |
| { | |
| "epoch": 0.33766233766233766, | |
| "eval_loss": 3.214432716369629, | |
| "eval_runtime": 15.0454, | |
| "eval_samples_per_second": 38.284, | |
| "eval_steps_per_second": 9.571, | |
| "step": 26000 | |
| }, | |
| { | |
| "epoch": 0.33896103896103896, | |
| "grad_norm": 0.20782601833343506, | |
| "learning_rate": 0.0008751460987077398, | |
| "loss": 2.8405, | |
| "step": 26100 | |
| }, | |
| { | |
| "epoch": 0.34025974025974026, | |
| "grad_norm": 0.2266394942998886, | |
| "learning_rate": 0.0008741913844399896, | |
| "loss": 2.8304, | |
| "step": 26200 | |
| }, | |
| { | |
| "epoch": 0.34155844155844156, | |
| "grad_norm": 0.215355783700943, | |
| "learning_rate": 0.0008732335588946612, | |
| "loss": 2.8749, | |
| "step": 26300 | |
| }, | |
| { | |
| "epoch": 0.34285714285714286, | |
| "grad_norm": 0.17590366303920746, | |
| "learning_rate": 0.0008722726300357573, | |
| "loss": 2.8542, | |
| "step": 26400 | |
| }, | |
| { | |
| "epoch": 0.34415584415584416, | |
| "grad_norm": 0.19376814365386963, | |
| "learning_rate": 0.0008713086058530832, | |
| "loss": 2.8636, | |
| "step": 26500 | |
| }, | |
| { | |
| "epoch": 0.34545454545454546, | |
| "grad_norm": 0.19086168706417084, | |
| "learning_rate": 0.0008703414943621817, | |
| "loss": 2.8752, | |
| "step": 26600 | |
| }, | |
| { | |
| "epoch": 0.34675324675324676, | |
| "grad_norm": 0.19386428594589233, | |
| "learning_rate": 0.0008693713036042643, | |
| "loss": 2.8595, | |
| "step": 26700 | |
| }, | |
| { | |
| "epoch": 0.34805194805194806, | |
| "grad_norm": 0.20364010334014893, | |
| "learning_rate": 0.0008683980416461465, | |
| "loss": 2.8322, | |
| "step": 26800 | |
| }, | |
| { | |
| "epoch": 0.34935064935064936, | |
| "grad_norm": 0.20115472376346588, | |
| "learning_rate": 0.0008674217165801797, | |
| "loss": 2.8271, | |
| "step": 26900 | |
| }, | |
| { | |
| "epoch": 0.35064935064935066, | |
| "grad_norm": 0.19452853500843048, | |
| "learning_rate": 0.0008664423365241833, | |
| "loss": 2.8445, | |
| "step": 27000 | |
| }, | |
| { | |
| "epoch": 0.35064935064935066, | |
| "eval_loss": 3.206655263900757, | |
| "eval_runtime": 14.9037, | |
| "eval_samples_per_second": 38.648, | |
| "eval_steps_per_second": 9.662, | |
| "step": 27000 | |
| }, | |
| { | |
| "epoch": 0.35194805194805195, | |
| "grad_norm": 0.29245325922966003, | |
| "learning_rate": 0.0008654599096213792, | |
| "loss": 2.8584, | |
| "step": 27100 | |
| }, | |
| { | |
| "epoch": 0.35324675324675325, | |
| "grad_norm": 0.1941668838262558, | |
| "learning_rate": 0.0008644744440403214, | |
| "loss": 2.8146, | |
| "step": 27200 | |
| }, | |
| { | |
| "epoch": 0.35454545454545455, | |
| "grad_norm": 0.20218811929225922, | |
| "learning_rate": 0.0008634859479748306, | |
| "loss": 2.8623, | |
| "step": 27300 | |
| }, | |
| { | |
| "epoch": 0.35584415584415585, | |
| "grad_norm": 0.19882351160049438, | |
| "learning_rate": 0.0008624944296439246, | |
| "loss": 2.8491, | |
| "step": 27400 | |
| }, | |
| { | |
| "epoch": 0.35714285714285715, | |
| "grad_norm": 0.19472059607505798, | |
| "learning_rate": 0.0008614998972917503, | |
| "loss": 2.8372, | |
| "step": 27500 | |
| }, | |
| { | |
| "epoch": 0.35844155844155845, | |
| "grad_norm": 0.20453399419784546, | |
| "learning_rate": 0.000860502359187515, | |
| "loss": 2.8228, | |
| "step": 27600 | |
| }, | |
| { | |
| "epoch": 0.35974025974025975, | |
| "grad_norm": 0.20085613429546356, | |
| "learning_rate": 0.0008595018236254182, | |
| "loss": 2.8416, | |
| "step": 27700 | |
| }, | |
| { | |
| "epoch": 0.36103896103896105, | |
| "grad_norm": 0.1953079253435135, | |
| "learning_rate": 0.0008584982989245822, | |
| "loss": 2.844, | |
| "step": 27800 | |
| }, | |
| { | |
| "epoch": 0.36233766233766235, | |
| "grad_norm": 0.18861424922943115, | |
| "learning_rate": 0.0008574917934289829, | |
| "loss": 2.8586, | |
| "step": 27900 | |
| }, | |
| { | |
| "epoch": 0.36363636363636365, | |
| "grad_norm": 0.2032879889011383, | |
| "learning_rate": 0.0008564823155073804, | |
| "loss": 2.8676, | |
| "step": 28000 | |
| }, | |
| { | |
| "epoch": 0.36363636363636365, | |
| "eval_loss": 3.213535785675049, | |
| "eval_runtime": 15.9519, | |
| "eval_samples_per_second": 36.109, | |
| "eval_steps_per_second": 9.027, | |
| "step": 28000 | |
| }, | |
| { | |
| "epoch": 0.36493506493506495, | |
| "grad_norm": 0.20181582868099213, | |
| "learning_rate": 0.00085546987355325, | |
| "loss": 2.835, | |
| "step": 28100 | |
| }, | |
| { | |
| "epoch": 0.36623376623376624, | |
| "grad_norm": 0.1909325122833252, | |
| "learning_rate": 0.0008544544759847112, | |
| "loss": 2.8291, | |
| "step": 28200 | |
| }, | |
| { | |
| "epoch": 0.36753246753246754, | |
| "grad_norm": 0.21319599449634552, | |
| "learning_rate": 0.000853436131244459, | |
| "loss": 2.8536, | |
| "step": 28300 | |
| }, | |
| { | |
| "epoch": 0.36883116883116884, | |
| "grad_norm": 0.200112447142601, | |
| "learning_rate": 0.0008524148477996933, | |
| "loss": 2.8732, | |
| "step": 28400 | |
| }, | |
| { | |
| "epoch": 0.37012987012987014, | |
| "grad_norm": 0.19826772809028625, | |
| "learning_rate": 0.0008513906341420478, | |
| "loss": 2.8208, | |
| "step": 28500 | |
| }, | |
| { | |
| "epoch": 0.37142857142857144, | |
| "grad_norm": 0.21064135432243347, | |
| "learning_rate": 0.0008503634987875206, | |
| "loss": 2.805, | |
| "step": 28600 | |
| }, | |
| { | |
| "epoch": 0.37272727272727274, | |
| "grad_norm": 0.2014266848564148, | |
| "learning_rate": 0.0008493334502764021, | |
| "loss": 2.8257, | |
| "step": 28700 | |
| }, | |
| { | |
| "epoch": 0.37402597402597404, | |
| "grad_norm": 0.21088480949401855, | |
| "learning_rate": 0.0008483004971732049, | |
| "loss": 2.8261, | |
| "step": 28800 | |
| }, | |
| { | |
| "epoch": 0.37532467532467534, | |
| "grad_norm": 0.20050457119941711, | |
| "learning_rate": 0.0008472646480665924, | |
| "loss": 2.8694, | |
| "step": 28900 | |
| }, | |
| { | |
| "epoch": 0.37662337662337664, | |
| "grad_norm": 0.21420925855636597, | |
| "learning_rate": 0.0008462259115693076, | |
| "loss": 2.8473, | |
| "step": 29000 | |
| }, | |
| { | |
| "epoch": 0.37662337662337664, | |
| "eval_loss": 3.195821523666382, | |
| "eval_runtime": 15.2121, | |
| "eval_samples_per_second": 37.865, | |
| "eval_steps_per_second": 9.466, | |
| "step": 29000 | |
| }, | |
| { | |
| "epoch": 0.37792207792207794, | |
| "grad_norm": 0.2159246951341629, | |
| "learning_rate": 0.0008451842963181003, | |
| "loss": 2.8032, | |
| "step": 29100 | |
| }, | |
| { | |
| "epoch": 0.37922077922077924, | |
| "grad_norm": 0.19643986225128174, | |
| "learning_rate": 0.0008441398109736571, | |
| "loss": 2.847, | |
| "step": 29200 | |
| }, | |
| { | |
| "epoch": 0.38051948051948054, | |
| "grad_norm": 0.1965179294347763, | |
| "learning_rate": 0.0008430924642205279, | |
| "loss": 2.8529, | |
| "step": 29300 | |
| }, | |
| { | |
| "epoch": 0.38181818181818183, | |
| "grad_norm": 0.21244043111801147, | |
| "learning_rate": 0.0008420422647670548, | |
| "loss": 2.8067, | |
| "step": 29400 | |
| }, | |
| { | |
| "epoch": 0.38311688311688313, | |
| "grad_norm": 0.20399615168571472, | |
| "learning_rate": 0.0008409892213452987, | |
| "loss": 2.8176, | |
| "step": 29500 | |
| }, | |
| { | |
| "epoch": 0.38441558441558443, | |
| "grad_norm": 0.20329082012176514, | |
| "learning_rate": 0.0008399333427109672, | |
| "loss": 2.8316, | |
| "step": 29600 | |
| }, | |
| { | |
| "epoch": 0.38571428571428573, | |
| "grad_norm": 0.19244763255119324, | |
| "learning_rate": 0.0008388746376433419, | |
| "loss": 2.8367, | |
| "step": 29700 | |
| }, | |
| { | |
| "epoch": 0.38701298701298703, | |
| "grad_norm": 0.18999354541301727, | |
| "learning_rate": 0.0008378131149452053, | |
| "loss": 2.8212, | |
| "step": 29800 | |
| }, | |
| { | |
| "epoch": 0.38831168831168833, | |
| "grad_norm": 0.20788761973381042, | |
| "learning_rate": 0.0008367487834427674, | |
| "loss": 2.7999, | |
| "step": 29900 | |
| }, | |
| { | |
| "epoch": 0.38961038961038963, | |
| "grad_norm": 0.21678058803081512, | |
| "learning_rate": 0.0008356816519855926, | |
| "loss": 2.81, | |
| "step": 30000 | |
| }, | |
| { | |
| "epoch": 0.38961038961038963, | |
| "eval_loss": 3.1891965866088867, | |
| "eval_runtime": 15.5166, | |
| "eval_samples_per_second": 37.122, | |
| "eval_steps_per_second": 9.28, | |
| "step": 30000 | |
| }, | |
| { | |
| "epoch": 0.39090909090909093, | |
| "grad_norm": 0.25863537192344666, | |
| "learning_rate": 0.0008346117294465258, | |
| "loss": 2.7919, | |
| "step": 30100 | |
| }, | |
| { | |
| "epoch": 0.3922077922077922, | |
| "grad_norm": 0.2082231491804123, | |
| "learning_rate": 0.0008335390247216193, | |
| "loss": 2.8058, | |
| "step": 30200 | |
| }, | |
| { | |
| "epoch": 0.3935064935064935, | |
| "grad_norm": 0.2484658658504486, | |
| "learning_rate": 0.0008324635467300578, | |
| "loss": 2.7913, | |
| "step": 30300 | |
| }, | |
| { | |
| "epoch": 0.3948051948051948, | |
| "grad_norm": 0.19809049367904663, | |
| "learning_rate": 0.000831385304414085, | |
| "loss": 2.8392, | |
| "step": 30400 | |
| }, | |
| { | |
| "epoch": 0.3961038961038961, | |
| "grad_norm": 0.19220763444900513, | |
| "learning_rate": 0.0008303043067389293, | |
| "loss": 2.8326, | |
| "step": 30500 | |
| }, | |
| { | |
| "epoch": 0.3974025974025974, | |
| "grad_norm": 0.19120201468467712, | |
| "learning_rate": 0.0008292205626927285, | |
| "loss": 2.8008, | |
| "step": 30600 | |
| }, | |
| { | |
| "epoch": 0.3987012987012987, | |
| "grad_norm": 0.2093675136566162, | |
| "learning_rate": 0.000828134081286456, | |
| "loss": 2.8228, | |
| "step": 30700 | |
| }, | |
| { | |
| "epoch": 0.4, | |
| "grad_norm": 0.2077227681875229, | |
| "learning_rate": 0.0008270448715538452, | |
| "loss": 2.8263, | |
| "step": 30800 | |
| }, | |
| { | |
| "epoch": 0.4012987012987013, | |
| "grad_norm": 0.2050541490316391, | |
| "learning_rate": 0.0008259529425513148, | |
| "loss": 2.8289, | |
| "step": 30900 | |
| }, | |
| { | |
| "epoch": 0.4025974025974026, | |
| "grad_norm": 0.19998504221439362, | |
| "learning_rate": 0.0008248583033578932, | |
| "loss": 2.8068, | |
| "step": 31000 | |
| }, | |
| { | |
| "epoch": 0.4025974025974026, | |
| "eval_loss": 3.1833224296569824, | |
| "eval_runtime": 11.0521, | |
| "eval_samples_per_second": 52.117, | |
| "eval_steps_per_second": 13.029, | |
| "step": 31000 | |
| }, | |
| { | |
| "epoch": 0.4038961038961039, | |
| "grad_norm": 0.20032985508441925, | |
| "learning_rate": 0.0008237609630751433, | |
| "loss": 2.8321, | |
| "step": 31100 | |
| }, | |
| { | |
| "epoch": 0.4051948051948052, | |
| "grad_norm": 0.2228243052959442, | |
| "learning_rate": 0.0008226609308270862, | |
| "loss": 2.8323, | |
| "step": 31200 | |
| }, | |
| { | |
| "epoch": 0.4064935064935065, | |
| "grad_norm": 0.22431324422359467, | |
| "learning_rate": 0.0008215582157601267, | |
| "loss": 2.8489, | |
| "step": 31300 | |
| }, | |
| { | |
| "epoch": 0.4077922077922078, | |
| "grad_norm": 0.20516513288021088, | |
| "learning_rate": 0.0008204528270429752, | |
| "loss": 2.8001, | |
| "step": 31400 | |
| }, | |
| { | |
| "epoch": 0.4090909090909091, | |
| "grad_norm": 0.20168229937553406, | |
| "learning_rate": 0.0008193447738665735, | |
| "loss": 2.8463, | |
| "step": 31500 | |
| }, | |
| { | |
| "epoch": 0.4103896103896104, | |
| "grad_norm": 0.20102174580097198, | |
| "learning_rate": 0.0008182340654440173, | |
| "loss": 2.8226, | |
| "step": 31600 | |
| }, | |
| { | |
| "epoch": 0.4116883116883117, | |
| "grad_norm": 0.2369510680437088, | |
| "learning_rate": 0.0008171207110104797, | |
| "loss": 2.8284, | |
| "step": 31700 | |
| }, | |
| { | |
| "epoch": 0.412987012987013, | |
| "grad_norm": 0.18998616933822632, | |
| "learning_rate": 0.0008160047198231344, | |
| "loss": 2.8506, | |
| "step": 31800 | |
| }, | |
| { | |
| "epoch": 0.4142857142857143, | |
| "grad_norm": 0.21406228840351105, | |
| "learning_rate": 0.000814886101161079, | |
| "loss": 2.81, | |
| "step": 31900 | |
| }, | |
| { | |
| "epoch": 0.4155844155844156, | |
| "grad_norm": 0.20903189480304718, | |
| "learning_rate": 0.0008137648643252575, | |
| "loss": 2.7976, | |
| "step": 32000 | |
| }, | |
| { | |
| "epoch": 0.4155844155844156, | |
| "eval_loss": 3.1785833835601807, | |
| "eval_runtime": 15.2674, | |
| "eval_samples_per_second": 37.727, | |
| "eval_steps_per_second": 9.432, | |
| "step": 32000 | |
| }, | |
| { | |
| "epoch": 0.41688311688311686, | |
| "grad_norm": 0.22132746875286102, | |
| "learning_rate": 0.0008126410186383836, | |
| "loss": 2.8195, | |
| "step": 32100 | |
| }, | |
| { | |
| "epoch": 0.41818181818181815, | |
| "grad_norm": 0.21235358715057373, | |
| "learning_rate": 0.0008115145734448622, | |
| "loss": 2.7824, | |
| "step": 32200 | |
| }, | |
| { | |
| "epoch": 0.41948051948051945, | |
| "grad_norm": 0.21031178534030914, | |
| "learning_rate": 0.0008103855381107122, | |
| "loss": 2.8135, | |
| "step": 32300 | |
| }, | |
| { | |
| "epoch": 0.42077922077922075, | |
| "grad_norm": 0.2186926305294037, | |
| "learning_rate": 0.0008092539220234894, | |
| "loss": 2.8166, | |
| "step": 32400 | |
| }, | |
| { | |
| "epoch": 0.42207792207792205, | |
| "grad_norm": 0.20855864882469177, | |
| "learning_rate": 0.0008081197345922069, | |
| "loss": 2.8595, | |
| "step": 32500 | |
| }, | |
| { | |
| "epoch": 0.42337662337662335, | |
| "grad_norm": 0.20601770281791687, | |
| "learning_rate": 0.0008069829852472583, | |
| "loss": 2.8201, | |
| "step": 32600 | |
| }, | |
| { | |
| "epoch": 0.42467532467532465, | |
| "grad_norm": 0.22806823253631592, | |
| "learning_rate": 0.000805843683440338, | |
| "loss": 2.8265, | |
| "step": 32700 | |
| }, | |
| { | |
| "epoch": 0.42597402597402595, | |
| "grad_norm": 0.21018485724925995, | |
| "learning_rate": 0.0008047018386443638, | |
| "loss": 2.8403, | |
| "step": 32800 | |
| }, | |
| { | |
| "epoch": 0.42727272727272725, | |
| "grad_norm": 0.218631774187088, | |
| "learning_rate": 0.0008035574603533975, | |
| "loss": 2.8052, | |
| "step": 32900 | |
| }, | |
| { | |
| "epoch": 0.42857142857142855, | |
| "grad_norm": 0.20622466504573822, | |
| "learning_rate": 0.000802410558082566, | |
| "loss": 2.824, | |
| "step": 33000 | |
| }, | |
| { | |
| "epoch": 0.42857142857142855, | |
| "eval_loss": 3.178421974182129, | |
| "eval_runtime": 15.6887, | |
| "eval_samples_per_second": 36.714, | |
| "eval_steps_per_second": 9.179, | |
| "step": 33000 | |
| }, | |
| { | |
| "epoch": 0.42987012987012985, | |
| "grad_norm": 0.24304358661174774, | |
| "learning_rate": 0.0008012611413679824, | |
| "loss": 2.8068, | |
| "step": 33100 | |
| }, | |
| { | |
| "epoch": 0.43116883116883115, | |
| "grad_norm": 0.20239757001399994, | |
| "learning_rate": 0.0008001092197666661, | |
| "loss": 2.7936, | |
| "step": 33200 | |
| }, | |
| { | |
| "epoch": 0.43246753246753245, | |
| "grad_norm": 0.21365991234779358, | |
| "learning_rate": 0.0007989548028564646, | |
| "loss": 2.8053, | |
| "step": 33300 | |
| }, | |
| { | |
| "epoch": 0.43376623376623374, | |
| "grad_norm": 0.19812121987342834, | |
| "learning_rate": 0.0007977979002359723, | |
| "loss": 2.8083, | |
| "step": 33400 | |
| }, | |
| { | |
| "epoch": 0.43506493506493504, | |
| "grad_norm": 0.20705953240394592, | |
| "learning_rate": 0.0007966385215244518, | |
| "loss": 2.7882, | |
| "step": 33500 | |
| }, | |
| { | |
| "epoch": 0.43636363636363634, | |
| "grad_norm": 0.23277799785137177, | |
| "learning_rate": 0.0007954766763617538, | |
| "loss": 2.8241, | |
| "step": 33600 | |
| }, | |
| { | |
| "epoch": 0.43766233766233764, | |
| "grad_norm": 0.2175544649362564, | |
| "learning_rate": 0.0007943123744082363, | |
| "loss": 2.816, | |
| "step": 33700 | |
| }, | |
| { | |
| "epoch": 0.43896103896103894, | |
| "grad_norm": 0.21744564175605774, | |
| "learning_rate": 0.000793145625344685, | |
| "loss": 2.7947, | |
| "step": 33800 | |
| }, | |
| { | |
| "epoch": 0.44025974025974024, | |
| "grad_norm": 0.2085115909576416, | |
| "learning_rate": 0.0007919764388722322, | |
| "loss": 2.8016, | |
| "step": 33900 | |
| }, | |
| { | |
| "epoch": 0.44155844155844154, | |
| "grad_norm": 0.21085582673549652, | |
| "learning_rate": 0.0007908048247122768, | |
| "loss": 2.7845, | |
| "step": 34000 | |
| }, | |
| { | |
| "epoch": 0.44155844155844154, | |
| "eval_loss": 3.1732594966888428, | |
| "eval_runtime": 14.9965, | |
| "eval_samples_per_second": 38.409, | |
| "eval_steps_per_second": 9.602, | |
| "step": 34000 | |
| }, | |
| { | |
| "epoch": 0.44285714285714284, | |
| "grad_norm": 0.1986636519432068, | |
| "learning_rate": 0.0007896307926064029, | |
| "loss": 2.8039, | |
| "step": 34100 | |
| }, | |
| { | |
| "epoch": 0.44415584415584414, | |
| "grad_norm": 0.2274903804063797, | |
| "learning_rate": 0.0007884543523162991, | |
| "loss": 2.7982, | |
| "step": 34200 | |
| }, | |
| { | |
| "epoch": 0.44545454545454544, | |
| "grad_norm": 0.23180291056632996, | |
| "learning_rate": 0.0007872755136236774, | |
| "loss": 2.8223, | |
| "step": 34300 | |
| }, | |
| { | |
| "epoch": 0.44675324675324674, | |
| "grad_norm": 0.21269173920154572, | |
| "learning_rate": 0.0007860942863301914, | |
| "loss": 2.8141, | |
| "step": 34400 | |
| }, | |
| { | |
| "epoch": 0.44805194805194803, | |
| "grad_norm": 0.20867326855659485, | |
| "learning_rate": 0.0007849106802573553, | |
| "loss": 2.7805, | |
| "step": 34500 | |
| }, | |
| { | |
| "epoch": 0.44935064935064933, | |
| "grad_norm": 1.6756339073181152, | |
| "learning_rate": 0.0007837247052464621, | |
| "loss": 2.7879, | |
| "step": 34600 | |
| }, | |
| { | |
| "epoch": 0.45064935064935063, | |
| "grad_norm": 0.24874623119831085, | |
| "learning_rate": 0.0007825363711585016, | |
| "loss": 2.8281, | |
| "step": 34700 | |
| }, | |
| { | |
| "epoch": 0.45194805194805193, | |
| "grad_norm": 0.19419661164283752, | |
| "learning_rate": 0.0007813456878740789, | |
| "loss": 2.7984, | |
| "step": 34800 | |
| }, | |
| { | |
| "epoch": 0.45324675324675323, | |
| "grad_norm": 0.21280324459075928, | |
| "learning_rate": 0.0007801526652933313, | |
| "loss": 2.834, | |
| "step": 34900 | |
| }, | |
| { | |
| "epoch": 0.45454545454545453, | |
| "grad_norm": 0.20888017117977142, | |
| "learning_rate": 0.0007789573133358471, | |
| "loss": 2.831, | |
| "step": 35000 | |
| }, | |
| { | |
| "epoch": 0.45454545454545453, | |
| "eval_loss": 3.1828813552856445, | |
| "eval_runtime": 14.929, | |
| "eval_samples_per_second": 38.583, | |
| "eval_steps_per_second": 9.646, | |
| "step": 35000 | |
| }, | |
| { | |
| "epoch": 0.45584415584415583, | |
| "grad_norm": 0.21340541541576385, | |
| "learning_rate": 0.0007777596419405823, | |
| "loss": 2.8065, | |
| "step": 35100 | |
| }, | |
| { | |
| "epoch": 0.45714285714285713, | |
| "grad_norm": 0.20886409282684326, | |
| "learning_rate": 0.0007765596610657783, | |
| "loss": 2.8423, | |
| "step": 35200 | |
| }, | |
| { | |
| "epoch": 0.4584415584415584, | |
| "grad_norm": 0.21619443595409393, | |
| "learning_rate": 0.0007753573806888795, | |
| "loss": 2.786, | |
| "step": 35300 | |
| }, | |
| { | |
| "epoch": 0.4597402597402597, | |
| "grad_norm": 0.21694345772266388, | |
| "learning_rate": 0.0007741528108064491, | |
| "loss": 2.7822, | |
| "step": 35400 | |
| }, | |
| { | |
| "epoch": 0.461038961038961, | |
| "grad_norm": 0.21181565523147583, | |
| "learning_rate": 0.0007729459614340872, | |
| "loss": 2.7599, | |
| "step": 35500 | |
| }, | |
| { | |
| "epoch": 0.4623376623376623, | |
| "grad_norm": 0.22892408072948456, | |
| "learning_rate": 0.0007717368426063475, | |
| "loss": 2.7934, | |
| "step": 35600 | |
| }, | |
| { | |
| "epoch": 0.4636363636363636, | |
| "grad_norm": 0.21864919364452362, | |
| "learning_rate": 0.0007705254643766527, | |
| "loss": 2.8059, | |
| "step": 35700 | |
| }, | |
| { | |
| "epoch": 0.4649350649350649, | |
| "grad_norm": 0.2145598828792572, | |
| "learning_rate": 0.000769311836817212, | |
| "loss": 2.8056, | |
| "step": 35800 | |
| }, | |
| { | |
| "epoch": 0.4662337662337662, | |
| "grad_norm": 0.21081195771694183, | |
| "learning_rate": 0.0007680959700189375, | |
| "loss": 2.7969, | |
| "step": 35900 | |
| }, | |
| { | |
| "epoch": 0.4675324675324675, | |
| "grad_norm": 0.22019855678081512, | |
| "learning_rate": 0.0007668778740913591, | |
| "loss": 2.7923, | |
| "step": 36000 | |
| }, | |
| { | |
| "epoch": 0.4675324675324675, | |
| "eval_loss": 3.161719560623169, | |
| "eval_runtime": 15.4038, | |
| "eval_samples_per_second": 37.393, | |
| "eval_steps_per_second": 9.348, | |
| "step": 36000 | |
| }, | |
| { | |
| "epoch": 0.4688311688311688, | |
| "grad_norm": 0.22442568838596344, | |
| "learning_rate": 0.0007656575591625415, | |
| "loss": 2.7954, | |
| "step": 36100 | |
| }, | |
| { | |
| "epoch": 0.4701298701298701, | |
| "grad_norm": 0.2147638201713562, | |
| "learning_rate": 0.0007644350353789997, | |
| "loss": 2.7897, | |
| "step": 36200 | |
| }, | |
| { | |
| "epoch": 0.4714285714285714, | |
| "grad_norm": 0.20484313368797302, | |
| "learning_rate": 0.0007632103129056145, | |
| "loss": 2.7859, | |
| "step": 36300 | |
| }, | |
| { | |
| "epoch": 0.4727272727272727, | |
| "grad_norm": 0.19691170752048492, | |
| "learning_rate": 0.0007619834019255482, | |
| "loss": 2.8251, | |
| "step": 36400 | |
| }, | |
| { | |
| "epoch": 0.474025974025974, | |
| "grad_norm": 0.2177974134683609, | |
| "learning_rate": 0.0007607543126401597, | |
| "loss": 2.7771, | |
| "step": 36500 | |
| }, | |
| { | |
| "epoch": 0.4753246753246753, | |
| "grad_norm": 0.24122121930122375, | |
| "learning_rate": 0.0007595230552689201, | |
| "loss": 2.7973, | |
| "step": 36600 | |
| }, | |
| { | |
| "epoch": 0.4766233766233766, | |
| "grad_norm": 0.2524343729019165, | |
| "learning_rate": 0.0007582896400493266, | |
| "loss": 2.82, | |
| "step": 36700 | |
| }, | |
| { | |
| "epoch": 0.4779220779220779, | |
| "grad_norm": 0.21695558726787567, | |
| "learning_rate": 0.000757054077236819, | |
| "loss": 2.802, | |
| "step": 36800 | |
| }, | |
| { | |
| "epoch": 0.4792207792207792, | |
| "grad_norm": 0.2205599695444107, | |
| "learning_rate": 0.0007558163771046934, | |
| "loss": 2.7509, | |
| "step": 36900 | |
| }, | |
| { | |
| "epoch": 0.4805194805194805, | |
| "grad_norm": 0.2173157036304474, | |
| "learning_rate": 0.0007545765499440169, | |
| "loss": 2.7978, | |
| "step": 37000 | |
| }, | |
| { | |
| "epoch": 0.4805194805194805, | |
| "eval_loss": 3.170949935913086, | |
| "eval_runtime": 16.3367, | |
| "eval_samples_per_second": 35.258, | |
| "eval_steps_per_second": 8.815, | |
| "step": 37000 | |
| }, | |
| { | |
| "epoch": 0.4818181818181818, | |
| "grad_norm": 0.21395540237426758, | |
| "learning_rate": 0.0007533346060635424, | |
| "loss": 2.775, | |
| "step": 37100 | |
| }, | |
| { | |
| "epoch": 0.4831168831168831, | |
| "grad_norm": 0.2262965440750122, | |
| "learning_rate": 0.0007520905557896221, | |
| "loss": 2.779, | |
| "step": 37200 | |
| }, | |
| { | |
| "epoch": 0.4844155844155844, | |
| "grad_norm": 0.2124248743057251, | |
| "learning_rate": 0.0007508444094661227, | |
| "loss": 2.8141, | |
| "step": 37300 | |
| }, | |
| { | |
| "epoch": 0.4857142857142857, | |
| "grad_norm": 0.24149860441684723, | |
| "learning_rate": 0.0007495961774543385, | |
| "loss": 2.8091, | |
| "step": 37400 | |
| }, | |
| { | |
| "epoch": 0.487012987012987, | |
| "grad_norm": 0.2356308251619339, | |
| "learning_rate": 0.0007483458701329059, | |
| "loss": 2.7898, | |
| "step": 37500 | |
| }, | |
| { | |
| "epoch": 0.4883116883116883, | |
| "grad_norm": 0.21705743670463562, | |
| "learning_rate": 0.0007470934978977167, | |
| "loss": 2.8022, | |
| "step": 37600 | |
| }, | |
| { | |
| "epoch": 0.4896103896103896, | |
| "grad_norm": 0.21993373334407806, | |
| "learning_rate": 0.0007458390711618315, | |
| "loss": 2.7881, | |
| "step": 37700 | |
| }, | |
| { | |
| "epoch": 0.4909090909090909, | |
| "grad_norm": 0.21296945214271545, | |
| "learning_rate": 0.0007445826003553939, | |
| "loss": 2.7987, | |
| "step": 37800 | |
| }, | |
| { | |
| "epoch": 0.4922077922077922, | |
| "grad_norm": 0.23273344337940216, | |
| "learning_rate": 0.000743324095925543, | |
| "loss": 2.8253, | |
| "step": 37900 | |
| }, | |
| { | |
| "epoch": 0.4935064935064935, | |
| "grad_norm": 0.2104436457157135, | |
| "learning_rate": 0.0007420635683363268, | |
| "loss": 2.7807, | |
| "step": 38000 | |
| }, | |
| { | |
| "epoch": 0.4935064935064935, | |
| "eval_loss": 3.151848316192627, | |
| "eval_runtime": 15.3437, | |
| "eval_samples_per_second": 37.54, | |
| "eval_steps_per_second": 9.385, | |
| "step": 38000 | |
| }, | |
| { | |
| "epoch": 0.4948051948051948, | |
| "grad_norm": 0.2078818380832672, | |
| "learning_rate": 0.0007408010280686151, | |
| "loss": 2.7504, | |
| "step": 38100 | |
| }, | |
| { | |
| "epoch": 0.4961038961038961, | |
| "grad_norm": 0.22127191722393036, | |
| "learning_rate": 0.0007395364856200127, | |
| "loss": 2.7698, | |
| "step": 38200 | |
| }, | |
| { | |
| "epoch": 0.4974025974025974, | |
| "grad_norm": 0.20785515010356903, | |
| "learning_rate": 0.0007382699515047715, | |
| "loss": 2.804, | |
| "step": 38300 | |
| }, | |
| { | |
| "epoch": 0.4987012987012987, | |
| "grad_norm": 0.267839640378952, | |
| "learning_rate": 0.0007370014362537041, | |
| "loss": 2.778, | |
| "step": 38400 | |
| }, | |
| { | |
| "epoch": 0.5, | |
| "grad_norm": 0.22827385365962982, | |
| "learning_rate": 0.0007357309504140948, | |
| "loss": 2.7786, | |
| "step": 38500 | |
| }, | |
| { | |
| "epoch": 0.5012987012987012, | |
| "grad_norm": 0.22568035125732422, | |
| "learning_rate": 0.0007344585045496133, | |
| "loss": 2.7655, | |
| "step": 38600 | |
| }, | |
| { | |
| "epoch": 0.5025974025974026, | |
| "grad_norm": 0.2355123907327652, | |
| "learning_rate": 0.000733184109240226, | |
| "loss": 2.7714, | |
| "step": 38700 | |
| }, | |
| { | |
| "epoch": 0.5038961038961038, | |
| "grad_norm": 0.2227976769208908, | |
| "learning_rate": 0.0007319077750821084, | |
| "loss": 2.7534, | |
| "step": 38800 | |
| }, | |
| { | |
| "epoch": 0.5051948051948052, | |
| "grad_norm": 0.25165653228759766, | |
| "learning_rate": 0.0007306295126875566, | |
| "loss": 2.7832, | |
| "step": 38900 | |
| }, | |
| { | |
| "epoch": 0.5064935064935064, | |
| "grad_norm": 0.24306100606918335, | |
| "learning_rate": 0.0007293493326848997, | |
| "loss": 2.774, | |
| "step": 39000 | |
| }, | |
| { | |
| "epoch": 0.5064935064935064, | |
| "eval_loss": 3.1421139240264893, | |
| "eval_runtime": 15.2375, | |
| "eval_samples_per_second": 37.801, | |
| "eval_steps_per_second": 9.45, | |
| "step": 39000 | |
| }, | |
| { | |
| "epoch": 0.5077922077922078, | |
| "grad_norm": 0.23936377465724945, | |
| "learning_rate": 0.0007280672457184108, | |
| "loss": 2.7569, | |
| "step": 39100 | |
| }, | |
| { | |
| "epoch": 0.509090909090909, | |
| "grad_norm": 0.2100268453359604, | |
| "learning_rate": 0.0007267832624482191, | |
| "loss": 2.7524, | |
| "step": 39200 | |
| }, | |
| { | |
| "epoch": 0.5103896103896104, | |
| "grad_norm": 0.21765153110027313, | |
| "learning_rate": 0.0007254973935502207, | |
| "loss": 2.7636, | |
| "step": 39300 | |
| }, | |
| { | |
| "epoch": 0.5116883116883116, | |
| "grad_norm": 0.22430171072483063, | |
| "learning_rate": 0.0007242096497159902, | |
| "loss": 2.7875, | |
| "step": 39400 | |
| }, | |
| { | |
| "epoch": 0.512987012987013, | |
| "grad_norm": 0.23513422906398773, | |
| "learning_rate": 0.0007229200416526911, | |
| "loss": 2.7606, | |
| "step": 39500 | |
| }, | |
| { | |
| "epoch": 0.5142857142857142, | |
| "grad_norm": 0.24426428973674774, | |
| "learning_rate": 0.0007216285800829885, | |
| "loss": 2.7867, | |
| "step": 39600 | |
| }, | |
| { | |
| "epoch": 0.5155844155844156, | |
| "grad_norm": 0.313528835773468, | |
| "learning_rate": 0.0007203352757449577, | |
| "loss": 2.7645, | |
| "step": 39700 | |
| }, | |
| { | |
| "epoch": 0.5168831168831168, | |
| "grad_norm": 0.23338325321674347, | |
| "learning_rate": 0.0007190401393919968, | |
| "loss": 2.7358, | |
| "step": 39800 | |
| }, | |
| { | |
| "epoch": 0.5181818181818182, | |
| "grad_norm": 0.2229086011648178, | |
| "learning_rate": 0.0007177431817927358, | |
| "loss": 2.7574, | |
| "step": 39900 | |
| }, | |
| { | |
| "epoch": 0.5194805194805194, | |
| "grad_norm": 0.22037512063980103, | |
| "learning_rate": 0.000716444413730948, | |
| "loss": 2.7507, | |
| "step": 40000 | |
| }, | |
| { | |
| "epoch": 0.5194805194805194, | |
| "eval_loss": 3.1465635299682617, | |
| "eval_runtime": 15.78, | |
| "eval_samples_per_second": 36.502, | |
| "eval_steps_per_second": 9.125, | |
| "step": 40000 | |
| }, | |
| { | |
| "epoch": 0.5207792207792208, | |
| "grad_norm": 0.22620290517807007, | |
| "learning_rate": 0.0007151438460054607, | |
| "loss": 2.7895, | |
| "step": 40100 | |
| }, | |
| { | |
| "epoch": 0.522077922077922, | |
| "grad_norm": 0.21930238604545593, | |
| "learning_rate": 0.000713841489430064, | |
| "loss": 2.7593, | |
| "step": 40200 | |
| }, | |
| { | |
| "epoch": 0.5233766233766234, | |
| "grad_norm": 0.232050359249115, | |
| "learning_rate": 0.0007125373548334219, | |
| "loss": 2.7341, | |
| "step": 40300 | |
| }, | |
| { | |
| "epoch": 0.5246753246753246, | |
| "grad_norm": 0.22093582153320312, | |
| "learning_rate": 0.0007112314530589825, | |
| "loss": 2.7415, | |
| "step": 40400 | |
| }, | |
| { | |
| "epoch": 0.525974025974026, | |
| "grad_norm": 0.21834588050842285, | |
| "learning_rate": 0.0007099237949648867, | |
| "loss": 2.7734, | |
| "step": 40500 | |
| }, | |
| { | |
| "epoch": 0.5272727272727272, | |
| "grad_norm": 0.2412509173154831, | |
| "learning_rate": 0.0007086143914238792, | |
| "loss": 2.7344, | |
| "step": 40600 | |
| }, | |
| { | |
| "epoch": 0.5285714285714286, | |
| "grad_norm": 0.2703639268875122, | |
| "learning_rate": 0.0007073032533232172, | |
| "loss": 2.7758, | |
| "step": 40700 | |
| }, | |
| { | |
| "epoch": 0.5298701298701298, | |
| "grad_norm": 0.26110807061195374, | |
| "learning_rate": 0.0007059903915645802, | |
| "loss": 2.7708, | |
| "step": 40800 | |
| }, | |
| { | |
| "epoch": 0.5311688311688312, | |
| "grad_norm": 0.2253560870885849, | |
| "learning_rate": 0.0007046758170639795, | |
| "loss": 2.7728, | |
| "step": 40900 | |
| }, | |
| { | |
| "epoch": 0.5324675324675324, | |
| "grad_norm": 0.2106214314699173, | |
| "learning_rate": 0.0007033595407516674, | |
| "loss": 2.7487, | |
| "step": 41000 | |
| }, | |
| { | |
| "epoch": 0.5324675324675324, | |
| "eval_loss": 3.1333227157592773, | |
| "eval_runtime": 15.8871, | |
| "eval_samples_per_second": 36.256, | |
| "eval_steps_per_second": 9.064, | |
| "step": 41000 | |
| }, | |
| { | |
| "epoch": 0.5337662337662338, | |
| "grad_norm": 0.21964064240455627, | |
| "learning_rate": 0.0007020415735720458, | |
| "loss": 2.7644, | |
| "step": 41100 | |
| }, | |
| { | |
| "epoch": 0.535064935064935, | |
| "grad_norm": 0.2143610268831253, | |
| "learning_rate": 0.0007007219264835758, | |
| "loss": 2.7476, | |
| "step": 41200 | |
| }, | |
| { | |
| "epoch": 0.5363636363636364, | |
| "grad_norm": 0.3317408263683319, | |
| "learning_rate": 0.0006994006104586865, | |
| "loss": 2.7193, | |
| "step": 41300 | |
| }, | |
| { | |
| "epoch": 0.5376623376623376, | |
| "grad_norm": 0.2196430265903473, | |
| "learning_rate": 0.0006980776364836834, | |
| "loss": 2.7699, | |
| "step": 41400 | |
| }, | |
| { | |
| "epoch": 0.538961038961039, | |
| "grad_norm": 0.2349650114774704, | |
| "learning_rate": 0.0006967530155586577, | |
| "loss": 2.7812, | |
| "step": 41500 | |
| }, | |
| { | |
| "epoch": 0.5402597402597402, | |
| "grad_norm": 0.20911413431167603, | |
| "learning_rate": 0.0006954267586973939, | |
| "loss": 2.7562, | |
| "step": 41600 | |
| }, | |
| { | |
| "epoch": 0.5415584415584416, | |
| "grad_norm": 0.2115071713924408, | |
| "learning_rate": 0.0006940988769272794, | |
| "loss": 2.7577, | |
| "step": 41700 | |
| }, | |
| { | |
| "epoch": 0.5428571428571428, | |
| "grad_norm": 0.24407769739627838, | |
| "learning_rate": 0.0006927693812892116, | |
| "loss": 2.7342, | |
| "step": 41800 | |
| }, | |
| { | |
| "epoch": 0.5441558441558442, | |
| "grad_norm": 0.21947546303272247, | |
| "learning_rate": 0.0006914382828375068, | |
| "loss": 2.7554, | |
| "step": 41900 | |
| }, | |
| { | |
| "epoch": 0.5454545454545454, | |
| "grad_norm": 0.2191598266363144, | |
| "learning_rate": 0.000690105592639809, | |
| "loss": 2.7256, | |
| "step": 42000 | |
| }, | |
| { | |
| "epoch": 0.5454545454545454, | |
| "eval_loss": 3.1299455165863037, | |
| "eval_runtime": 15.1552, | |
| "eval_samples_per_second": 38.007, | |
| "eval_steps_per_second": 9.502, | |
| "step": 42000 | |
| }, | |
| { | |
| "epoch": 0.5467532467532468, | |
| "grad_norm": 0.22920465469360352, | |
| "learning_rate": 0.0006887713217769954, | |
| "loss": 2.7328, | |
| "step": 42100 | |
| }, | |
| { | |
| "epoch": 0.548051948051948, | |
| "grad_norm": 0.23324090242385864, | |
| "learning_rate": 0.0006874354813430874, | |
| "loss": 2.7545, | |
| "step": 42200 | |
| }, | |
| { | |
| "epoch": 0.5493506493506494, | |
| "grad_norm": 0.21898794174194336, | |
| "learning_rate": 0.0006860980824451563, | |
| "loss": 2.7611, | |
| "step": 42300 | |
| }, | |
| { | |
| "epoch": 0.5506493506493506, | |
| "grad_norm": 0.21969111263751984, | |
| "learning_rate": 0.0006847591362032313, | |
| "loss": 2.7413, | |
| "step": 42400 | |
| }, | |
| { | |
| "epoch": 0.551948051948052, | |
| "grad_norm": 0.24275629222393036, | |
| "learning_rate": 0.0006834186537502076, | |
| "loss": 2.7421, | |
| "step": 42500 | |
| }, | |
| { | |
| "epoch": 0.5532467532467532, | |
| "grad_norm": 0.20918132364749908, | |
| "learning_rate": 0.0006820766462317534, | |
| "loss": 2.73, | |
| "step": 42600 | |
| }, | |
| { | |
| "epoch": 0.5545454545454546, | |
| "grad_norm": 0.2445412427186966, | |
| "learning_rate": 0.0006807331248062174, | |
| "loss": 2.7811, | |
| "step": 42700 | |
| }, | |
| { | |
| "epoch": 0.5558441558441558, | |
| "grad_norm": 0.24175776541233063, | |
| "learning_rate": 0.0006793881006445354, | |
| "loss": 2.7529, | |
| "step": 42800 | |
| }, | |
| { | |
| "epoch": 0.5571428571428572, | |
| "grad_norm": 0.234651118516922, | |
| "learning_rate": 0.0006780415849301388, | |
| "loss": 2.7592, | |
| "step": 42900 | |
| }, | |
| { | |
| "epoch": 0.5584415584415584, | |
| "grad_norm": 0.24966347217559814, | |
| "learning_rate": 0.0006766935888588599, | |
| "loss": 2.7314, | |
| "step": 43000 | |
| }, | |
| { | |
| "epoch": 0.5584415584415584, | |
| "eval_loss": 3.1258981227874756, | |
| "eval_runtime": 15.1058, | |
| "eval_samples_per_second": 38.131, | |
| "eval_steps_per_second": 9.533, | |
| "step": 43000 | |
| }, | |
| { | |
| "epoch": 0.5597402597402598, | |
| "grad_norm": 0.30740296840667725, | |
| "learning_rate": 0.0006753441236388405, | |
| "loss": 2.7467, | |
| "step": 43100 | |
| }, | |
| { | |
| "epoch": 0.561038961038961, | |
| "grad_norm": 0.2325826734304428, | |
| "learning_rate": 0.0006739932004904373, | |
| "loss": 2.7428, | |
| "step": 43200 | |
| }, | |
| { | |
| "epoch": 0.5623376623376624, | |
| "grad_norm": 0.2326454520225525, | |
| "learning_rate": 0.0006726408306461294, | |
| "loss": 2.7515, | |
| "step": 43300 | |
| }, | |
| { | |
| "epoch": 0.5636363636363636, | |
| "grad_norm": 0.24041685461997986, | |
| "learning_rate": 0.000671287025350425, | |
| "loss": 2.7339, | |
| "step": 43400 | |
| }, | |
| { | |
| "epoch": 0.564935064935065, | |
| "grad_norm": 0.21853885054588318, | |
| "learning_rate": 0.0006699317958597668, | |
| "loss": 2.7188, | |
| "step": 43500 | |
| }, | |
| { | |
| "epoch": 0.5662337662337662, | |
| "grad_norm": 0.32151710987091064, | |
| "learning_rate": 0.00066857515344244, | |
| "loss": 2.7317, | |
| "step": 43600 | |
| }, | |
| { | |
| "epoch": 0.5675324675324676, | |
| "grad_norm": 0.2511979639530182, | |
| "learning_rate": 0.0006672171093784773, | |
| "loss": 2.7714, | |
| "step": 43700 | |
| }, | |
| { | |
| "epoch": 0.5688311688311688, | |
| "grad_norm": 0.2406754046678543, | |
| "learning_rate": 0.0006658576749595663, | |
| "loss": 2.7762, | |
| "step": 43800 | |
| }, | |
| { | |
| "epoch": 0.5701298701298702, | |
| "grad_norm": 0.23970568180084229, | |
| "learning_rate": 0.0006644968614889538, | |
| "loss": 2.7163, | |
| "step": 43900 | |
| }, | |
| { | |
| "epoch": 0.5714285714285714, | |
| "grad_norm": 0.23587392270565033, | |
| "learning_rate": 0.0006631346802813543, | |
| "loss": 2.7272, | |
| "step": 44000 | |
| }, | |
| { | |
| "epoch": 0.5714285714285714, | |
| "eval_loss": 3.120995044708252, | |
| "eval_runtime": 15.2013, | |
| "eval_samples_per_second": 37.892, | |
| "eval_steps_per_second": 9.473, | |
| "step": 44000 | |
| }, | |
| { | |
| "epoch": 0.5727272727272728, | |
| "grad_norm": 0.2565126419067383, | |
| "learning_rate": 0.0006617711426628536, | |
| "loss": 2.7398, | |
| "step": 44100 | |
| }, | |
| { | |
| "epoch": 0.574025974025974, | |
| "grad_norm": 0.22110450267791748, | |
| "learning_rate": 0.0006604062599708158, | |
| "loss": 2.7204, | |
| "step": 44200 | |
| }, | |
| { | |
| "epoch": 0.5753246753246753, | |
| "grad_norm": 0.2212057262659073, | |
| "learning_rate": 0.0006590400435537894, | |
| "loss": 2.7467, | |
| "step": 44300 | |
| }, | |
| { | |
| "epoch": 0.5766233766233766, | |
| "grad_norm": 0.24286267161369324, | |
| "learning_rate": 0.0006576725047714116, | |
| "loss": 2.7193, | |
| "step": 44400 | |
| }, | |
| { | |
| "epoch": 0.577922077922078, | |
| "grad_norm": 0.23778927326202393, | |
| "learning_rate": 0.0006563036549943152, | |
| "loss": 2.7388, | |
| "step": 44500 | |
| }, | |
| { | |
| "epoch": 0.5792207792207792, | |
| "grad_norm": 0.22259333729743958, | |
| "learning_rate": 0.000654933505604033, | |
| "loss": 2.7296, | |
| "step": 44600 | |
| }, | |
| { | |
| "epoch": 0.5805194805194805, | |
| "grad_norm": 0.23351649940013885, | |
| "learning_rate": 0.0006535620679929045, | |
| "loss": 2.753, | |
| "step": 44700 | |
| }, | |
| { | |
| "epoch": 0.5818181818181818, | |
| "grad_norm": 0.22059041261672974, | |
| "learning_rate": 0.0006521893535639792, | |
| "loss": 2.7281, | |
| "step": 44800 | |
| }, | |
| { | |
| "epoch": 0.5831168831168831, | |
| "grad_norm": 0.23977316915988922, | |
| "learning_rate": 0.0006508153737309235, | |
| "loss": 2.7003, | |
| "step": 44900 | |
| }, | |
| { | |
| "epoch": 0.5844155844155844, | |
| "grad_norm": 0.25528889894485474, | |
| "learning_rate": 0.0006494401399179255, | |
| "loss": 2.7206, | |
| "step": 45000 | |
| }, | |
| { | |
| "epoch": 0.5844155844155844, | |
| "eval_loss": 3.1150028705596924, | |
| "eval_runtime": 16.5286, | |
| "eval_samples_per_second": 34.849, | |
| "eval_steps_per_second": 8.712, | |
| "step": 45000 | |
| }, | |
| { | |
| "epoch": 0.5857142857142857, | |
| "grad_norm": 0.24675515294075012, | |
| "learning_rate": 0.0006480636635595993, | |
| "loss": 2.7617, | |
| "step": 45100 | |
| }, | |
| { | |
| "epoch": 0.587012987012987, | |
| "grad_norm": 0.23121733963489532, | |
| "learning_rate": 0.0006466859561008905, | |
| "loss": 2.721, | |
| "step": 45200 | |
| }, | |
| { | |
| "epoch": 0.5883116883116883, | |
| "grad_norm": 0.24000558257102966, | |
| "learning_rate": 0.0006453070289969807, | |
| "loss": 2.7091, | |
| "step": 45300 | |
| }, | |
| { | |
| "epoch": 0.5896103896103896, | |
| "grad_norm": 0.22867269814014435, | |
| "learning_rate": 0.0006439268937131929, | |
| "loss": 2.7395, | |
| "step": 45400 | |
| }, | |
| { | |
| "epoch": 0.5909090909090909, | |
| "grad_norm": 0.23441274464130402, | |
| "learning_rate": 0.0006425455617248952, | |
| "loss": 2.7507, | |
| "step": 45500 | |
| }, | |
| { | |
| "epoch": 0.5922077922077922, | |
| "grad_norm": 0.22815178334712982, | |
| "learning_rate": 0.0006411630445174063, | |
| "loss": 2.7257, | |
| "step": 45600 | |
| }, | |
| { | |
| "epoch": 0.5935064935064935, | |
| "grad_norm": 0.22590583562850952, | |
| "learning_rate": 0.0006397793535858992, | |
| "loss": 2.7303, | |
| "step": 45700 | |
| }, | |
| { | |
| "epoch": 0.5948051948051948, | |
| "grad_norm": 0.2853323221206665, | |
| "learning_rate": 0.0006383945004353064, | |
| "loss": 2.7414, | |
| "step": 45800 | |
| }, | |
| { | |
| "epoch": 0.5961038961038961, | |
| "grad_norm": 0.2671266794204712, | |
| "learning_rate": 0.000637008496580224, | |
| "loss": 2.7558, | |
| "step": 45900 | |
| }, | |
| { | |
| "epoch": 0.5974025974025974, | |
| "grad_norm": 0.23534780740737915, | |
| "learning_rate": 0.0006356213535448151, | |
| "loss": 2.7498, | |
| "step": 46000 | |
| }, | |
| { | |
| "epoch": 0.5974025974025974, | |
| "eval_loss": 3.1098833084106445, | |
| "eval_runtime": 14.8476, | |
| "eval_samples_per_second": 38.794, | |
| "eval_steps_per_second": 9.699, | |
| "step": 46000 | |
| }, | |
| { | |
| "epoch": 0.5987012987012987, | |
| "grad_norm": 0.25292590260505676, | |
| "learning_rate": 0.0006342330828627155, | |
| "loss": 2.7116, | |
| "step": 46100 | |
| }, | |
| { | |
| "epoch": 0.6, | |
| "grad_norm": 0.24926777184009552, | |
| "learning_rate": 0.0006328436960769364, | |
| "loss": 2.694, | |
| "step": 46200 | |
| }, | |
| { | |
| "epoch": 0.6012987012987013, | |
| "grad_norm": 0.22545288503170013, | |
| "learning_rate": 0.0006314532047397697, | |
| "loss": 2.7382, | |
| "step": 46300 | |
| }, | |
| { | |
| "epoch": 0.6025974025974026, | |
| "grad_norm": 0.24442555010318756, | |
| "learning_rate": 0.0006300616204126905, | |
| "loss": 2.7408, | |
| "step": 46400 | |
| }, | |
| { | |
| "epoch": 0.6038961038961039, | |
| "grad_norm": 0.2365717589855194, | |
| "learning_rate": 0.0006286689546662625, | |
| "loss": 2.7386, | |
| "step": 46500 | |
| }, | |
| { | |
| "epoch": 0.6051948051948052, | |
| "grad_norm": 0.26666414737701416, | |
| "learning_rate": 0.0006272752190800404, | |
| "loss": 2.7408, | |
| "step": 46600 | |
| }, | |
| { | |
| "epoch": 0.6064935064935065, | |
| "grad_norm": 0.2522192597389221, | |
| "learning_rate": 0.0006258804252424745, | |
| "loss": 2.7487, | |
| "step": 46700 | |
| }, | |
| { | |
| "epoch": 0.6077922077922078, | |
| "grad_norm": 0.32666802406311035, | |
| "learning_rate": 0.0006244845847508144, | |
| "loss": 2.7319, | |
| "step": 46800 | |
| }, | |
| { | |
| "epoch": 0.6090909090909091, | |
| "grad_norm": 0.2485225796699524, | |
| "learning_rate": 0.0006230877092110119, | |
| "loss": 2.7767, | |
| "step": 46900 | |
| }, | |
| { | |
| "epoch": 0.6103896103896104, | |
| "grad_norm": 0.28847283124923706, | |
| "learning_rate": 0.0006216898102376251, | |
| "loss": 2.7883, | |
| "step": 47000 | |
| }, | |
| { | |
| "epoch": 0.6103896103896104, | |
| "eval_loss": 3.108853340148926, | |
| "eval_runtime": 15.0323, | |
| "eval_samples_per_second": 38.317, | |
| "eval_steps_per_second": 9.579, | |
| "step": 47000 | |
| }, | |
| { | |
| "epoch": 0.6116883116883117, | |
| "grad_norm": 0.2481118142604828, | |
| "learning_rate": 0.0006202908994537215, | |
| "loss": 2.7383, | |
| "step": 47100 | |
| }, | |
| { | |
| "epoch": 0.612987012987013, | |
| "grad_norm": 0.2495453953742981, | |
| "learning_rate": 0.0006188909884907814, | |
| "loss": 2.7935, | |
| "step": 47200 | |
| }, | |
| { | |
| "epoch": 0.6142857142857143, | |
| "grad_norm": 0.257658451795578, | |
| "learning_rate": 0.0006174900889886015, | |
| "loss": 2.7703, | |
| "step": 47300 | |
| }, | |
| { | |
| "epoch": 0.6155844155844156, | |
| "grad_norm": 0.23015721142292023, | |
| "learning_rate": 0.0006160882125951977, | |
| "loss": 2.7421, | |
| "step": 47400 | |
| }, | |
| { | |
| "epoch": 0.6168831168831169, | |
| "grad_norm": 0.2232840657234192, | |
| "learning_rate": 0.0006146853709667086, | |
| "loss": 2.7372, | |
| "step": 47500 | |
| }, | |
| { | |
| "epoch": 0.6181818181818182, | |
| "grad_norm": 0.24996255338191986, | |
| "learning_rate": 0.0006132815757672979, | |
| "loss": 2.7654, | |
| "step": 47600 | |
| }, | |
| { | |
| "epoch": 0.6194805194805195, | |
| "grad_norm": 0.2613258361816406, | |
| "learning_rate": 0.0006118768386690587, | |
| "loss": 2.7143, | |
| "step": 47700 | |
| }, | |
| { | |
| "epoch": 0.6207792207792208, | |
| "grad_norm": 0.2545129656791687, | |
| "learning_rate": 0.0006104711713519152, | |
| "loss": 2.7443, | |
| "step": 47800 | |
| }, | |
| { | |
| "epoch": 0.6220779220779221, | |
| "grad_norm": 0.24030067026615143, | |
| "learning_rate": 0.000609064585503526, | |
| "loss": 2.7386, | |
| "step": 47900 | |
| }, | |
| { | |
| "epoch": 0.6233766233766234, | |
| "grad_norm": 0.23413017392158508, | |
| "learning_rate": 0.0006076570928191872, | |
| "loss": 2.7276, | |
| "step": 48000 | |
| }, | |
| { | |
| "epoch": 0.6233766233766234, | |
| "eval_loss": 3.097559690475464, | |
| "eval_runtime": 15.7054, | |
| "eval_samples_per_second": 36.675, | |
| "eval_steps_per_second": 9.169, | |
| "step": 48000 | |
| }, | |
| { | |
| "epoch": 0.6246753246753247, | |
| "grad_norm": 0.2436034381389618, | |
| "learning_rate": 0.0006062487050017348, | |
| "loss": 2.7583, | |
| "step": 48100 | |
| }, | |
| { | |
| "epoch": 0.625974025974026, | |
| "grad_norm": 0.25060826539993286, | |
| "learning_rate": 0.0006048394337614477, | |
| "loss": 2.7889, | |
| "step": 48200 | |
| }, | |
| { | |
| "epoch": 0.6272727272727273, | |
| "grad_norm": 0.23102222383022308, | |
| "learning_rate": 0.0006034292908159502, | |
| "loss": 2.777, | |
| "step": 48300 | |
| }, | |
| { | |
| "epoch": 0.6285714285714286, | |
| "grad_norm": 0.2651401460170746, | |
| "learning_rate": 0.000602018287890114, | |
| "loss": 2.753, | |
| "step": 48400 | |
| }, | |
| { | |
| "epoch": 0.6298701298701299, | |
| "grad_norm": 0.2365243136882782, | |
| "learning_rate": 0.000600606436715962, | |
| "loss": 2.7541, | |
| "step": 48500 | |
| }, | |
| { | |
| "epoch": 0.6311688311688312, | |
| "grad_norm": 0.2493165135383606, | |
| "learning_rate": 0.0005991937490325696, | |
| "loss": 2.7529, | |
| "step": 48600 | |
| }, | |
| { | |
| "epoch": 0.6324675324675325, | |
| "grad_norm": 0.24916312098503113, | |
| "learning_rate": 0.0005977802365859677, | |
| "loss": 2.7335, | |
| "step": 48700 | |
| }, | |
| { | |
| "epoch": 0.6337662337662338, | |
| "grad_norm": 0.38456228375434875, | |
| "learning_rate": 0.0005963659111290444, | |
| "loss": 2.7492, | |
| "step": 48800 | |
| }, | |
| { | |
| "epoch": 0.6350649350649351, | |
| "grad_norm": 0.23968328535556793, | |
| "learning_rate": 0.0005949507844214482, | |
| "loss": 2.729, | |
| "step": 48900 | |
| }, | |
| { | |
| "epoch": 0.6363636363636364, | |
| "grad_norm": 0.27702993154525757, | |
| "learning_rate": 0.0005935348682294894, | |
| "loss": 2.7453, | |
| "step": 49000 | |
| }, | |
| { | |
| "epoch": 0.6363636363636364, | |
| "eval_loss": 3.0973825454711914, | |
| "eval_runtime": 15.2932, | |
| "eval_samples_per_second": 37.664, | |
| "eval_steps_per_second": 9.416, | |
| "step": 49000 | |
| }, | |
| { | |
| "epoch": 0.6376623376623377, | |
| "grad_norm": 0.2538464367389679, | |
| "learning_rate": 0.000592118174326043, | |
| "loss": 2.7291, | |
| "step": 49100 | |
| }, | |
| { | |
| "epoch": 0.638961038961039, | |
| "grad_norm": 0.27856433391571045, | |
| "learning_rate": 0.0005907007144904501, | |
| "loss": 2.7324, | |
| "step": 49200 | |
| }, | |
| { | |
| "epoch": 0.6402597402597403, | |
| "grad_norm": 0.26114940643310547, | |
| "learning_rate": 0.0005892825005084202, | |
| "loss": 2.7777, | |
| "step": 49300 | |
| }, | |
| { | |
| "epoch": 0.6415584415584416, | |
| "grad_norm": 0.25457167625427246, | |
| "learning_rate": 0.0005878635441719333, | |
| "loss": 2.7273, | |
| "step": 49400 | |
| }, | |
| { | |
| "epoch": 0.6428571428571429, | |
| "grad_norm": 0.24550481140613556, | |
| "learning_rate": 0.0005864438572791422, | |
| "loss": 2.7641, | |
| "step": 49500 | |
| }, | |
| { | |
| "epoch": 0.6441558441558441, | |
| "grad_norm": 0.2586402893066406, | |
| "learning_rate": 0.0005850234516342739, | |
| "loss": 2.7115, | |
| "step": 49600 | |
| }, | |
| { | |
| "epoch": 0.6454545454545455, | |
| "grad_norm": 0.25716432929039, | |
| "learning_rate": 0.000583602339047531, | |
| "loss": 2.7164, | |
| "step": 49700 | |
| }, | |
| { | |
| "epoch": 0.6467532467532467, | |
| "grad_norm": 0.26527953147888184, | |
| "learning_rate": 0.0005821805313349947, | |
| "loss": 2.752, | |
| "step": 49800 | |
| }, | |
| { | |
| "epoch": 0.6480519480519481, | |
| "grad_norm": 0.24985221028327942, | |
| "learning_rate": 0.000580758040318526, | |
| "loss": 2.7468, | |
| "step": 49900 | |
| }, | |
| { | |
| "epoch": 0.6493506493506493, | |
| "grad_norm": 0.23009921610355377, | |
| "learning_rate": 0.0005793348778256671, | |
| "loss": 2.7323, | |
| "step": 50000 | |
| }, | |
| { | |
| "epoch": 0.6493506493506493, | |
| "eval_loss": 3.0897603034973145, | |
| "eval_runtime": 14.9478, | |
| "eval_samples_per_second": 38.534, | |
| "eval_steps_per_second": 9.634, | |
| "step": 50000 | |
| }, | |
| { | |
| "epoch": 0.6506493506493507, | |
| "grad_norm": 0.227068230509758, | |
| "learning_rate": 0.0005779110556895433, | |
| "loss": 2.7248, | |
| "step": 50100 | |
| }, | |
| { | |
| "epoch": 0.6519480519480519, | |
| "grad_norm": 0.22547580301761627, | |
| "learning_rate": 0.0005764865857487646, | |
| "loss": 2.7372, | |
| "step": 50200 | |
| }, | |
| { | |
| "epoch": 0.6532467532467533, | |
| "grad_norm": 0.2620651423931122, | |
| "learning_rate": 0.0005750614798473273, | |
| "loss": 2.7249, | |
| "step": 50300 | |
| }, | |
| { | |
| "epoch": 0.6545454545454545, | |
| "grad_norm": 0.26259052753448486, | |
| "learning_rate": 0.0005736357498345156, | |
| "loss": 2.6864, | |
| "step": 50400 | |
| }, | |
| { | |
| "epoch": 0.6558441558441559, | |
| "grad_norm": 0.24948278069496155, | |
| "learning_rate": 0.0005722094075648032, | |
| "loss": 2.7276, | |
| "step": 50500 | |
| }, | |
| { | |
| "epoch": 0.6571428571428571, | |
| "grad_norm": 0.23260965943336487, | |
| "learning_rate": 0.0005707824648977536, | |
| "loss": 2.7114, | |
| "step": 50600 | |
| }, | |
| { | |
| "epoch": 0.6584415584415585, | |
| "grad_norm": 0.253491073846817, | |
| "learning_rate": 0.0005693549336979236, | |
| "loss": 2.7156, | |
| "step": 50700 | |
| }, | |
| { | |
| "epoch": 0.6597402597402597, | |
| "grad_norm": 0.24990208446979523, | |
| "learning_rate": 0.0005679268258347626, | |
| "loss": 2.7319, | |
| "step": 50800 | |
| }, | |
| { | |
| "epoch": 0.6610389610389611, | |
| "grad_norm": 0.24773485958576202, | |
| "learning_rate": 0.0005664981531825152, | |
| "loss": 2.7356, | |
| "step": 50900 | |
| }, | |
| { | |
| "epoch": 0.6623376623376623, | |
| "grad_norm": 0.23747920989990234, | |
| "learning_rate": 0.0005650689276201219, | |
| "loss": 2.7497, | |
| "step": 51000 | |
| }, | |
| { | |
| "epoch": 0.6623376623376623, | |
| "eval_loss": 3.0903031826019287, | |
| "eval_runtime": 14.4116, | |
| "eval_samples_per_second": 39.968, | |
| "eval_steps_per_second": 9.992, | |
| "step": 51000 | |
| }, | |
| { | |
| "epoch": 0.6636363636363637, | |
| "grad_norm": 0.24378643929958344, | |
| "learning_rate": 0.0005636391610311204, | |
| "loss": 2.7239, | |
| "step": 51100 | |
| }, | |
| { | |
| "epoch": 0.6649350649350649, | |
| "grad_norm": 0.2741662561893463, | |
| "learning_rate": 0.0005622088653035469, | |
| "loss": 2.7161, | |
| "step": 51200 | |
| }, | |
| { | |
| "epoch": 0.6662337662337663, | |
| "grad_norm": 0.25329113006591797, | |
| "learning_rate": 0.0005607780523298372, | |
| "loss": 2.7301, | |
| "step": 51300 | |
| }, | |
| { | |
| "epoch": 0.6675324675324675, | |
| "grad_norm": 0.2747024893760681, | |
| "learning_rate": 0.0005593467340067282, | |
| "loss": 2.7312, | |
| "step": 51400 | |
| }, | |
| { | |
| "epoch": 0.6688311688311688, | |
| "grad_norm": 0.24558883905410767, | |
| "learning_rate": 0.0005579149222351577, | |
| "loss": 2.75, | |
| "step": 51500 | |
| }, | |
| { | |
| "epoch": 0.6701298701298701, | |
| "grad_norm": 0.24378612637519836, | |
| "learning_rate": 0.0005564826289201674, | |
| "loss": 2.742, | |
| "step": 51600 | |
| }, | |
| { | |
| "epoch": 0.6714285714285714, | |
| "grad_norm": 0.24434533715248108, | |
| "learning_rate": 0.0005550498659708022, | |
| "loss": 2.7085, | |
| "step": 51700 | |
| }, | |
| { | |
| "epoch": 0.6727272727272727, | |
| "grad_norm": 0.26504042744636536, | |
| "learning_rate": 0.000553616645300012, | |
| "loss": 2.7147, | |
| "step": 51800 | |
| }, | |
| { | |
| "epoch": 0.674025974025974, | |
| "grad_norm": 0.24731750786304474, | |
| "learning_rate": 0.0005521829788245528, | |
| "loss": 2.7433, | |
| "step": 51900 | |
| }, | |
| { | |
| "epoch": 0.6753246753246753, | |
| "grad_norm": 0.23254041373729706, | |
| "learning_rate": 0.0005507488784648869, | |
| "loss": 2.7326, | |
| "step": 52000 | |
| }, | |
| { | |
| "epoch": 0.6753246753246753, | |
| "eval_loss": 3.080451011657715, | |
| "eval_runtime": 18.0502, | |
| "eval_samples_per_second": 31.911, | |
| "eval_steps_per_second": 7.978, | |
| "step": 52000 | |
| } | |
| ], | |
| "logging_steps": 100, | |
| "max_steps": 77000, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 9223372036854775807, | |
| "save_steps": 4000, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": false | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 1.54800431824896e+18, | |
| "train_batch_size": 22, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |