Helios-7B / debug.log
c4tdr0ut's picture
Training in progress, step 24
8186f25 verified
Raw History Blame Contribute Delete
64.5 kB
[2026-08-11 19:02:12,745] [DEBUG] [axolotl.utils.config.log_gpu_memory_usage:127] [PID:31] baseline 0.000GB ()
[2026-08-11 19:02:12,747] [INFO] [axolotl.cli.config.load_cfg:336] [PID:31] config:
{
"activation_offloading": false,
"attn_decontaminates_packing": true,
"attn_implementation": "flash_attention_2",
"attn_needs_dtype_cast": true,
"attn_supports_packing": true,
"attn_uses_flash_lib": true,
"auto_resume_from_checkpoints": false,
"axolotl_config_path": "/workspace/task_08e727bf-a87d-47c2-97d0-de595c57c5dc_output/sft_run/sft_training_config.yaml",
"base_model": "c4tdr0ut/heli-pretrain",
"base_model_config": "c4tdr0ut/heli-pretrain",
"batch_size": 150,
"bf16": true,
"capabilities": {
"bf16": true,
"compute_capability": "sm_90",
"fp8": true,
"n_gpu": 1,
"n_node": 1,
"tf32": true
},
"chat_template": "chatml",
"context_parallel_size": 1,
"dataloader_num_workers": 1,
"dataloader_pin_memory": true,
"dataloader_prefetch_factor": 256,
"dataset_num_proc": 17,
"dataset_prepared_path": "last_finetune_prepared",
"datasets": [
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "axolotl_rag_conversations_inputs.jsonl",
"trust_remote_code": false,
"type": "input_output"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "axolotl_correction_conversations_inputs.json",
"trust_remote_code": false,
"type": "input_output"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "pretraining_subset_1515812.jsonl",
"trust_remote_code": false,
"type": "completion"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "factual_sft_completion/combined_all_0.jsonl",
"trust_remote_code": false,
"type": "completion"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "factual_sft_completion/combined_all_2.jsonl",
"trust_remote_code": false,
"type": "completion"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "factual_sft_completion/combined_all_1.jsonl",
"trust_remote_code": false,
"type": "completion"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "generic_sft_completion/Augmentoolkit-Augmentoolkit-Pippa-Thoughts_748413.jsonl",
"trust_remote_code": false,
"type": "completion"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "generic_sft_completion/Augmentoolkit-Augmentoolkit-Bluemoon-1mil-thoughts_748413.jsonl",
"trust_remote_code": false,
"type": "completion"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "generic_sft_completion/Augmentoolkit-Openthoughts-100mil-DifferentFormat_2993654.jsonl",
"trust_remote_code": false,
"type": "completion"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "generic_sft_completion/Augmentoolkit-Augmentoolkit-LMsys-800k-Thoughts_748413.jsonl",
"trust_remote_code": false,
"type": "completion"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "generic_sft_completion/Augmentoolkit-Augmentoolkit-Generic-Grabbag-Thoughts_1496827.jsonl",
"trust_remote_code": false,
"type": "completion"
},
{
"message_property_mappings": {
"content": "content",
"role": "role"
},
"path": "generic_sft_completion/Augmentoolkit-Augmentoolkit-Capybara-2point5mil-Thoughts_748413.jsonl",
"trust_remote_code": false,
"type": "completion"
}
],
"ddp": false,
"device": "cuda:0",
"device_map": "auto",
"dion_rank_fraction": 1.0,
"dion_rank_multiple_of": 1,
"eaft_alpha": 1.0,
"eaft_k": 20,
"env_capabilities": {
"torch_version": "2.12.0"
},
"eval_batch_size": 4,
"eval_causal_lm_metrics": [
"sacrebleu",
"comet",
"ter",
"chrf"
],
"eval_max_new_tokens": 128,
"eval_sample_packing": false,
"eval_steps": 0.25,
"eval_table_size": 0,
"evals_per_epoch": 1,
"experimental_skip_move_to_device": true,
"fp16": false,
"generate_samples": false,
"generation_do_sample": true,
"generation_max_new_tokens": 50,
"generation_prompt_ratio": 0.5,
"generation_temperature": 0.7,
"gradient_accumulation_steps": 10,
"gradient_checkpointing": true,
"gradient_checkpointing_kwargs": {
"use_reentrant": true
},
"group_by_length": false,
"hub_model_id": "c4tdr0ut/Helios-7B",
"hub_strategy": "all_checkpoints",
"include_tkps": true,
"is_falcon_derived_model": false,
"is_llama_derived_model": false,
"is_mistral_derived_model": true,
"layer_offloading": false,
"learning_rate": 2e-05,
"liger_fused_linear_cross_entropy": true,
"liger_glu_activation": true,
"liger_layer_norm": true,
"liger_rms_norm": true,
"liger_rope": true,
"lisa_layers_attribute": "model.layers",
"load_best_model_at_end": false,
"load_in_4bit": false,
"load_in_8bit": false,
"local_rank": 0,
"logging_steps": 1,
"lora_dropout": 0.0,
"loraplus_lr_embedding": 1e-06,
"lr_scheduler": "constant",
"mean_resizing_embeddings": false,
"merge_method": "memory_efficient",
"micro_batch_size": 15,
"model_config_type": "mistral",
"neftune_noise_alpha": 5.0,
"num_epochs": 4.0,
"num_generation_samples": 3,
"optimizer": "paged_adamw_8bit",
"otel_metrics_host": "localhost",
"otel_metrics_port": 8000,
"output_dir": "./finetune-model-output",
"pad_to_sequence_len": false,
"plugins": [
"axolotl.integrations.liger.LigerPlugin"
],
"pretrain_multipack_attn": true,
"profiler_steps_start": 0,
"qgalore_cos_threshold": 0.4,
"qgalore_gamma_proj": 2,
"qgalore_proj_bits": 4,
"qgalore_proj_group_size": 256,
"qgalore_proj_quant": true,
"qgalore_proj_type": "std",
"qgalore_queue_size": 5,
"qgalore_rank": 256,
"qgalore_scale": 0.25,
"qgalore_update_proj_gap": 200,
"qlora_sharded_model_loading": false,
"quantize_moe_experts": false,
"ray_num_workers": 1,
"relora_prune_method": "magnitude",
"remove_unused_columns": false,
"resources_per_worker": {
"GPU": 1
},
"sample_packing": true,
"sample_packing_bin_size": 200,
"sample_packing_group_size": 100000,
"save_only_model": false,
"save_safetensors": true,
"save_steps": 0.25,
"save_total_limit": 2,
"saves_per_epoch": 1,
"seed": 1337,
"sequence_len": 5000,
"shuffle_before_merging_datasets": false,
"shuffle_merged_datasets": true,
"skip_prepare_dataset": false,
"special_tokens": {
"pad_token": "<unk>"
},
"streaming_multipack_buffer_size": 10000,
"strict": false,
"tensor_parallel_size": 1,
"tf32": false,
"tiled_mlp_use_original_mlp": true,
"tokenizer_config": "c4tdr0ut/heli-pretrain",
"tokenizer_save_jinja_files": true,
"tokenizer_type": "AutoTokenizer",
"torch_dtype": "torch.bfloat16",
"train_on_inputs": false,
"trl": {
"async_prefetch": false,
"log_completions": false,
"mask_truncated_completions": false,
"ref_model_mixup_alpha": 0.9,
"ref_model_sync_steps": 64,
"replay_buffer_size": 0,
"replay_recompute_logps": true,
"reroll_max_groups": 1,
"reroll_start_fraction": 1.0,
"reward_num_workers": 1,
"scale_rewards": true,
"skip_zero_advantage_batches": true,
"sync_ref_model": false,
"use_data_producer": false,
"use_vllm": false,
"vllm_lora_sync": false,
"vllm_server_host": "0.0.0.0",
"vllm_server_port": 8000
},
"type_of_model": "AutoModelForCausalLM",
"use_otel_metrics": false,
"use_ray": false,
"val_set_size": 0.04,
"vllm": {
"device": "auto",
"dtype": "auto",
"gpu_memory_utilization": 0.9,
"host": "0.0.0.0",
"port": 8000
},
"warmup_ratio": 0.1,
"weight_decay": 0.0,
"world_size": 1
}
[2026-08-11 19:02:17,597] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:311] [PID:31] EOS: 2 / </s>
[2026-08-11 19:02:17,597] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:31] BOS: 1 / <s>
[2026-08-11 19:02:17,598] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:31] PAD: 0 / <unk>
[2026-08-11 19:02:17,598] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:31] UNK: 0 / <unk>
[2026-08-11 19:02:17,601] [INFO] [axolotl.utils.data.shared.load_preprocessed_dataset:543] [PID:31] Unable to find prepared dataset in last_finetune_prepared/5eab9bc891e347cf10de8154a14a218d
[2026-08-11 19:02:17,601] [INFO] [axolotl.utils.data.sft._load_raw_datasets:320] [PID:31] Loading raw datasets...
[2026-08-11 19:02:17,601] [WARNING] [axolotl.utils.data.sft._load_raw_datasets:322] [PID:31] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`.
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 500 examples [00:00, 9589.65 examples/s]
[2026-08-11 19:02:18,066] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: axolotl_rag_conversations_inputs.jsonl with base_type: input_output and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/500 [00:00<?, ? examples/s][2026-08-11 19:02:18,496] [WARNING] [py.warnings._showwarnmsg:112] [PID:31] /workspace/axolotl-venv/lib/python3.12/site-packages/multiprocess/popen_fork.py:66: DeprecationWarning: This process (pid=31) is multi-threaded, use of fork() may lead to deadlocks in the child.
self.pid = os.fork()
Tokenizing Prompts (num_proc=17): 0%| | 2/500 [00:01<07:47, 1.06 examples/s] Tokenizing Prompts (num_proc=17): 4%|β–Ž | 18/500 [00:02<00:40, 11.79 examples/s] Tokenizing Prompts (num_proc=17): 7%|β–‹ | 37/500 [00:02<00:16, 27.42 examples/s] Tokenizing Prompts (num_proc=17): 14%|β–ˆβ– | 72/500 [00:02<00:06, 61.15 examples/s] Tokenizing Prompts (num_proc=17): 23%|β–ˆβ–ˆβ–Ž | 113/500 [00:02<00:03, 107.20 examples/s] Tokenizing Prompts (num_proc=17): 28%|β–ˆβ–ˆβ–Š | 142/500 [00:02<00:02, 133.58 examples/s] Tokenizing Prompts (num_proc=17): 40%|β–ˆβ–ˆβ–ˆβ–ˆ | 200/500 [00:02<00:01, 213.81 examples/s] Tokenizing Prompts (num_proc=17): 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š | 238/500 [00:02<00:01, 229.44 examples/s] Tokenizing Prompts (num_proc=17): 58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 288/500 [00:02<00:00, 284.63 examples/s] Tokenizing Prompts (num_proc=17): 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 327/500 [00:02<00:00, 294.54 examples/s] Tokenizing Prompts (num_proc=17): 73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 365/500 [00:03<00:00, 291.66 examples/s] Tokenizing Prompts (num_proc=17): 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 409/500 [00:03<00:00, 325.36 examples/s] Tokenizing Prompts (num_proc=17): 90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 451/500 [00:03<00:00, 338.40 examples/s] Tokenizing Prompts (num_proc=17): 98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 490/500 [00:03<00:00, 289.79 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 500/500 [00:03<00:00, 129.54 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 388 examples [00:00, 2954.03 examples/s] Generating train split: 388 examples [00:00, 2891.60 examples/s]
[2026-08-11 19:02:22,531] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: axolotl_correction_conversations_inputs.json with base_type: input_output and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/388 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=17): 4%|▍ | 16/388 [00:01<00:43, 8.52 examples/s] Tokenizing Prompts (num_proc=17): 14%|β–ˆβ– | 54/388 [00:02<00:10, 33.33 examples/s] Tokenizing Prompts (num_proc=17): 27%|β–ˆβ–ˆβ–‹ | 104/388 [00:02<00:03, 71.73 examples/s] Tokenizing Prompts (num_proc=17): 36%|β–ˆβ–ˆβ–ˆβ–Œ | 138/388 [00:02<00:02, 88.56 examples/s] Tokenizing Prompts (num_proc=17): 51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 199/388 [00:02<00:01, 147.71 examples/s] Tokenizing Prompts (num_proc=17): 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 253/388 [00:02<00:00, 196.74 examples/s] Tokenizing Prompts (num_proc=17): 75%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 292/388 [00:02<00:00, 208.67 examples/s] Tokenizing Prompts (num_proc=17): 87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 338/388 [00:02<00:00, 251.52 examples/s] Tokenizing Prompts (num_proc=17): 99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 383/388 [00:03<00:00, 287.97 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 388/388 [00:03<00:00, 120.92 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 11 examples [00:00, 602.06 examples/s]
[2026-08-11 19:02:26,503] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: pretraining_subset_1515812.jsonl with base_type: completion and prompt_style: None
[2026-08-11 19:02:26,507] [WARNING] [datasets.arrow_dataset.map:3413] [PID:31] num_proc must be <= 11. Reducing num_proc to 11 for dataset of size 11.
Tokenizing Prompts (num_proc=11): 0%| | 0/11 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=11): 9%|β–‰ | 1/11 [00:01<00:15, 1.60s/ examples] Tokenizing Prompts (num_proc=11): 27%|β–ˆβ–ˆβ–‹ | 3/11 [00:01<00:03, 2.22 examples/s] Tokenizing Prompts (num_proc=11): 45%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 5/11 [00:01<00:01, 3.83 examples/s] Tokenizing Prompts (num_proc=11): 64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 7/11 [00:02<00:00, 5.18 examples/s] Tokenizing Prompts (num_proc=11): 91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 10/11 [00:02<00:00, 7.41 examples/s] Tokenizing Prompts (num_proc=11): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 11/11 [00:02<00:00, 3.69 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 1057 examples [00:00, 34419.58 examples/s]
[2026-08-11 19:02:30,018] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: factual_sft_completion/combined_all_0.jsonl with base_type: completion and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/1057 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=17): 6%|β–Œ | 63/1057 [00:01<00:25, 38.67 examples/s] Tokenizing Prompts (num_proc=17): 12%|β–ˆβ– | 126/1057 [00:01<00:10, 85.83 examples/s] Tokenizing Prompts (num_proc=17): 30%|β–ˆβ–ˆβ–‰ | 313/1057 [00:01<00:02, 266.76 examples/s] Tokenizing Prompts (num_proc=17): 41%|β–ˆβ–ˆβ–ˆβ–ˆβ– | 437/1057 [00:01<00:01, 386.78 examples/s] Tokenizing Prompts (num_proc=17): 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 623/1057 [00:02<00:00, 551.39 examples/s] Tokenizing Prompts (num_proc=17): 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 747/1057 [00:02<00:00, 590.90 examples/s] Tokenizing Prompts (num_proc=17): 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 995/1057 [00:02<00:00, 793.97 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1057/1057 [00:02<00:00, 402.54 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 1041 examples [00:00, 32791.87 examples/s]
[2026-08-11 19:02:33,121] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: factual_sft_completion/combined_all_2.jsonl with base_type: completion and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/1041 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=17): 6%|β–Œ | 62/1041 [00:01<00:27, 36.23 examples/s] Tokenizing Prompts (num_proc=17): 18%|β–ˆβ–Š | 186/1041 [00:01<00:06, 125.98 examples/s] Tokenizing Prompts (num_proc=17): 36%|β–ˆβ–ˆβ–ˆβ–Œ | 370/1041 [00:02<00:02, 275.52 examples/s] Tokenizing Prompts (num_proc=17): 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 553/1041 [00:02<00:01, 432.87 examples/s] Tokenizing Prompts (num_proc=17): 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 736/1041 [00:02<00:00, 562.13 examples/s] Tokenizing Prompts (num_proc=17): 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 919/1041 [00:02<00:00, 706.13 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1041/1041 [00:02<00:00, 393.82 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 1041 examples [00:00, 34785.73 examples/s]
[2026-08-11 19:02:36,287] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: factual_sft_completion/combined_all_1.jsonl with base_type: completion and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/1041 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=17): 6%|β–Œ | 62/1041 [00:01<00:25, 38.17 examples/s] Tokenizing Prompts (num_proc=17): 24%|β–ˆβ–ˆβ– | 248/1041 [00:01<00:04, 173.56 examples/s] Tokenizing Prompts (num_proc=17): 36%|β–ˆβ–ˆβ–ˆβ–Œ | 370/1041 [00:01<00:02, 269.79 examples/s] Tokenizing Prompts (num_proc=17): 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 553/1041 [00:02<00:01, 427.79 examples/s] Tokenizing Prompts (num_proc=17): 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 736/1041 [00:02<00:00, 556.70 examples/s] Tokenizing Prompts (num_proc=17): 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 980/1041 [00:02<00:00, 796.34 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1041/1041 [00:02<00:00, 402.39 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 40 examples [00:00, 4102.51 examples/s]
[2026-08-11 19:02:39,388] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: generic_sft_completion/Augmentoolkit-Augmentoolkit-Pippa-Thoughts_748413.jsonl with base_type: completion and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/40 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=17): 8%|β–Š | 3/40 [00:01<00:16, 2.22 examples/s] Tokenizing Prompts (num_proc=17): 22%|β–ˆβ–ˆβ–Ž | 9/40 [00:01<00:04, 6.99 examples/s] Tokenizing Prompts (num_proc=17): 35%|β–ˆβ–ˆβ–ˆβ–Œ | 14/40 [00:01<00:02, 9.88 examples/s] Tokenizing Prompts (num_proc=17): 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 21/40 [00:01<00:01, 16.52 examples/s] Tokenizing Prompts (num_proc=17): 62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 25/40 [00:02<00:00, 19.74 examples/s] Tokenizing Prompts (num_proc=17): 85%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 34/40 [00:02<00:00, 30.94 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 40/40 [00:02<00:00, 16.80 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 45 examples [00:00, 3857.74 examples/s]
[2026-08-11 19:02:42,226] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: generic_sft_completion/Augmentoolkit-Augmentoolkit-Bluemoon-1mil-thoughts_748413.jsonl with base_type: completion and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/45 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=17): 7%|β–‹ | 3/45 [00:01<00:21, 1.91 examples/s] Tokenizing Prompts (num_proc=17): 13%|β–ˆβ–Ž | 6/45 [00:01<00:09, 4.22 examples/s] Tokenizing Prompts (num_proc=17): 40%|β–ˆβ–ˆβ–ˆβ–ˆ | 18/45 [00:01<00:01, 15.84 examples/s] Tokenizing Prompts (num_proc=17): 60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 27/45 [00:01<00:00, 23.37 examples/s] Tokenizing Prompts (num_proc=17): 78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 35/45 [00:02<00:00, 27.52 examples/s] Tokenizing Prompts (num_proc=17): 96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 43/45 [00:02<00:00, 28.48 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 45/45 [00:02<00:00, 17.38 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 389 examples [00:00, 15786.22 examples/s]
[2026-08-11 19:02:45,300] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: generic_sft_completion/Augmentoolkit-Openthoughts-100mil-DifferentFormat_2993654.jsonl with base_type: completion and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/389 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=17): 6%|β–Œ | 23/389 [00:01<00:27, 13.46 examples/s] Tokenizing Prompts (num_proc=17): 18%|β–ˆβ–Š | 69/389 [00:01<00:06, 46.97 examples/s] Tokenizing Prompts (num_proc=17): 24%|β–ˆβ–ˆβ–Ž | 92/389 [00:01<00:04, 64.34 examples/s] Tokenizing Prompts (num_proc=17): 41%|β–ˆβ–ˆβ–ˆβ–ˆβ– | 161/389 [00:02<00:01, 133.17 examples/s] Tokenizing Prompts (num_proc=17): 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 207/389 [00:02<00:01, 179.30 examples/s] Tokenizing Prompts (num_proc=17): 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 253/389 [00:02<00:00, 218.99 examples/s] Tokenizing Prompts (num_proc=17): 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 322/389 [00:02<00:00, 294.81 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 389/389 [00:02<00:00, 146.34 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 813 examples [00:00, 70109.16 examples/s]
[2026-08-11 19:02:48,430] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: generic_sft_completion/Augmentoolkit-Augmentoolkit-LMsys-800k-Thoughts_748413.jsonl with base_type: completion and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/813 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=17): 6%|β–Œ | 48/813 [00:01<00:24, 31.02 examples/s] Tokenizing Prompts (num_proc=17): 18%|β–ˆβ–Š | 144/813 [00:01<00:06, 108.15 examples/s] Tokenizing Prompts (num_proc=17): 30%|β–ˆβ–ˆβ–‰ | 240/813 [00:01<00:02, 196.60 examples/s] Tokenizing Prompts (num_proc=17): 41%|β–ˆβ–ˆβ–ˆβ–ˆβ– | 336/813 [00:01<00:01, 289.09 examples/s] Tokenizing Prompts (num_proc=17): 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 432/813 [00:02<00:01, 360.93 examples/s] Tokenizing Prompts (num_proc=17): 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 576/813 [00:02<00:00, 521.34 examples/s] Tokenizing Prompts (num_proc=17): 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 672/813 [00:02<00:00, 603.26 examples/s] Tokenizing Prompts (num_proc=17): 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 766/813 [00:02<00:00, 660.00 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 813/813 [00:02<00:00, 323.82 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 2531 examples [00:00, 97320.19 examples/s]
[2026-08-11 19:02:51,457] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: generic_sft_completion/Augmentoolkit-Augmentoolkit-Generic-Grabbag-Thoughts_1496827.jsonl with base_type: completion and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/2531 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=17): 6%|β–Œ | 149/2531 [00:02<00:33, 70.55 examples/s] Tokenizing Prompts (num_proc=17): 41%|β–ˆβ–ˆβ–ˆβ–ˆ | 1043/2531 [00:02<00:02, 509.55 examples/s] Tokenizing Prompts (num_proc=17): 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 1341/2531 [00:02<00:01, 670.58 examples/s] Tokenizing Prompts (num_proc=17): 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 1639/2531 [00:02<00:01, 847.87 examples/s] Tokenizing Prompts (num_proc=17): 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 2086/2531 [00:02<00:00, 1192.76 examples/s] Tokenizing Prompts (num_proc=17): 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 2383/2531 [00:03<00:00, 1414.16 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 2531/2531 [00:03<00:00, 787.24 examples/s]
Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 444 examples [00:00, 41194.31 examples/s]
[2026-08-11 19:02:55,122] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:31] Loading dataset: generic_sft_completion/Augmentoolkit-Augmentoolkit-Capybara-2point5mil-Thoughts_748413.jsonl with base_type: completion and prompt_style: None
Tokenizing Prompts (num_proc=17): 0%| | 0/444 [00:00<?, ? examples/s] Tokenizing Prompts (num_proc=17): 6%|β–Œ | 27/444 [00:01<00:24, 17.18 examples/s] Tokenizing Prompts (num_proc=17): 18%|β–ˆβ–Š | 80/444 [00:01<00:06, 58.21 examples/s] Tokenizing Prompts (num_proc=17): 36%|β–ˆβ–ˆβ–ˆβ–Œ | 158/444 [00:01<00:02, 124.21 examples/s] Tokenizing Prompts (num_proc=17): 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 210/444 [00:01<00:01, 167.76 examples/s] Tokenizing Prompts (num_proc=17): 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 262/444 [00:02<00:00, 216.33 examples/s] Tokenizing Prompts (num_proc=17): 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 314/444 [00:02<00:00, 264.54 examples/s] Tokenizing Prompts (num_proc=17): 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 366/444 [00:02<00:00, 313.18 examples/s] Tokenizing Prompts (num_proc=17): 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 418/444 [00:02<00:00, 346.24 examples/s] Tokenizing Prompts (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 444/444 [00:02<00:00, 170.98 examples/s]
[2026-08-11 19:02:57,993] [INFO] [axolotl.utils.data.shared.merge_datasets:712] [PID:31] Merging datasets...
[2026-08-11 19:02:58,037] [DEBUG] [axolotl.utils.data.shared.merge_datasets:716] [PID:31] Shuffling merged datasets...
[2026-08-11 19:02:58,273] [INFO] [axolotl.utils.data.utils._log_dataset_stats:212] [PID:31] min_input_len: 8
[2026-08-11 19:02:58,274] [INFO] [axolotl.utils.data.utils._log_dataset_stats:213] [PID:31] max_input_len: 15017
Dropping Invalid Sequences (<None or >5000) (num_proc=17): 0%| | 0/9474 [00:00<?, ? examples/s] Dropping Invalid Sequences (<None or >5000) (num_proc=17): 6%|β–Œ | 558/9474 [00:01<00:30, 294.29 examples/s] Dropping Invalid Sequences (<None or >5000) (num_proc=17): 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 4461/9474 [00:01<00:01, 3002.66 examples/s] Dropping Invalid Sequences (<None or >5000) (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 9474/9474 [00:02<00:00, 4284.64 examples/s]
[2026-08-11 19:03:00,622] [INFO] [axolotl.utils.data.utils._drop_outside_range:306] [PID:31] Dropped 443 sequences outside valid range ([None, 5000])
[2026-08-11 19:03:00,626] [INFO] [axolotl.utils.trainer.process_datasets_for_packing:254] [PID:31] dropping token_type_ids column if it exists
Drop Samples with Zero Trainable Tokens (num_proc=17): 0%| | 0/9031 [00:00<?, ? examples/s] Drop Samples with Zero Trainable Tokens (num_proc=17): 6%|β–Œ | 532/9031 [00:01<00:27, 305.55 examples/s] Drop Samples with Zero Trainable Tokens (num_proc=17): 24%|β–ˆβ–ˆβ–Ž | 2127/9031 [00:01<00:04, 1471.87 examples/s] Drop Samples with Zero Trainable Tokens (num_proc=17): 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 7969/9031 [00:01<00:00, 6898.04 examples/s] Drop Samples with Zero Trainable Tokens (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 9031/9031 [00:02<00:00, 4297.46 examples/s]
Add position_id column (Sample Packing) (num_proc=17): 0%| | 0/9031 [00:00<?, ? examples/s][2026-08-11 19:03:02,994] [WARNING] [py.warnings._showwarnmsg:112] [PID:31] /workspace/axolotl-venv/lib/python3.12/site-packages/multiprocess/popen_fork.py:66: DeprecationWarning: This process (pid=31) is multi-threaded, use of fork() may lead to deadlocks in the child.
self.pid = os.fork()
Add position_id column (Sample Packing) (num_proc=17): 6%|β–Œ | 532/9031 [00:02<00:34, 249.39 examples/s] Add position_id column (Sample Packing) (num_proc=17): 35%|β–ˆβ–ˆβ–ˆβ–Œ | 3190/9031 [00:02<00:03, 1898.45 examples/s] Add position_id column (Sample Packing) (num_proc=17): 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 8500/9031 [00:02<00:00, 6099.61 examples/s] Add position_id column (Sample Packing) (num_proc=17): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 9031/9031 [00:02<00:00, 3623.92 examples/s]
Saving the dataset (0/17 shards): 0%| | 0/9031 [00:00<?, ? examples/s] Saving the dataset (0/17 shards): 6%|β–Œ | 531/9031 [00:05<01:28, 96.46 examples/s] Saving the dataset (1/17 shards): 6%|β–Œ | 531/9031 [00:05<01:28, 96.46 examples/s] Saving the dataset (1/17 shards): 12%|β–ˆβ– | 1062/9031 [00:05<00:35, 222.35 examples/s] Saving the dataset (2/17 shards): 12%|β–ˆβ– | 1062/9031 [00:05<00:35, 222.35 examples/s] Saving the dataset (3/17 shards): 18%|β–ˆβ–Š | 1594/9031 [00:05<00:33, 222.35 examples/s] Saving the dataset (4/17 shards): 24%|β–ˆβ–ˆβ–Ž | 2126/9031 [00:05<00:31, 222.35 examples/s] Saving the dataset (5/17 shards): 29%|β–ˆβ–ˆβ–‰ | 2657/9031 [00:05<00:28, 222.35 examples/s] Saving the dataset (6/17 shards): 35%|β–ˆβ–ˆβ–ˆβ–Œ | 3189/9031 [00:05<00:26, 222.35 examples/s] Saving the dataset (7/17 shards): 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 4251/9031 [00:05<00:21, 222.35 examples/s] Saving the dataset (8/17 shards): 47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 4251/9031 [00:05<00:21, 222.35 examples/s] Saving the dataset (9/17 shards): 53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 4782/9031 [00:05<00:19, 222.35 examples/s] Saving the dataset (10/17 shards): 59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 5313/9031 [00:05<00:16, 222.35 examples/s] Saving the dataset (11/17 shards): 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 5844/9031 [00:05<00:14, 222.35 examples/s] Saving the dataset (12/17 shards): 71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 6375/9031 [00:05<00:11, 222.35 examples/s] Saving the dataset (13/17 shards): 76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 6906/9031 [00:05<00:09, 222.35 examples/s] Saving the dataset (14/17 shards): 82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 7437/9031 [00:05<00:07, 222.35 examples/s] Saving the dataset (15/17 shards): 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 7968/9031 [00:05<00:04, 222.35 examples/s] Saving the dataset (15/17 shards): 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 8500/9031 [00:05<00:00, 2684.59 examples/s] Saving the dataset (16/17 shards): 94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 8500/9031 [00:05<00:00, 2684.59 examples/s] Saving the dataset (17/17 shards): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 9031/9031 [00:05<00:00, 2684.59 examples/s] Saving the dataset (17/17 shards): 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 9031/9031 [00:06<00:00, 1500.92 examples/s]
[2026-08-11 19:03:11,569] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:418] [PID:31] total_num_tokens: 17_945_275
[2026-08-11 19:03:11,619] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:440] [PID:31] total_supervised_tokens: 17_669_047
[2026-08-11 19:03:11,772] [WARNING] [py.warnings._showwarnmsg:112] [PID:31] /root/.local/share/uv/python/cpython-3.12.13-linux-x86_64-gnu/lib/python3.12/multiprocessing/popen_fork.py:66: DeprecationWarning: This process (pid=31) is multi-threaded, use of fork() may lead to deadlocks in the child.
self.pid = os.fork()
[2026-08-11 19:03:14,907] [DEBUG] [axolotl.utils.samplers.multipack.__len__:467] [PID:31] generate_batches time: 1.3644943237304688
[2026-08-11 19:03:16,260] [DEBUG] [axolotl.utils.samplers.multipack.__len__:467] [PID:31] generate_batches time: 1.352339267730713
[2026-08-11 19:03:17,635] [DEBUG] [axolotl.utils.samplers.multipack.__len__:467] [PID:31] generate_batches time: 1.3746399879455566
[2026-08-11 19:03:19,068] [DEBUG] [axolotl.utils.samplers.multipack.__len__:467] [PID:31] generate_batches time: 1.432831048965454
[2026-08-11 19:03:19,109] [INFO] [axolotl.utils.samplers.multipack.calc_min_len:443] [PID:31] gather_len_batches: [240]
[2026-08-11 19:03:19,110] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:497] [PID:31] data_loader_len: 24
[2026-08-11 19:03:19,110] [INFO] [axolotl.utils.trainer.calc_sample_packing_eff_est:506] [PID:31] sample_packing_eff_est across ranks: [0.9969597222222222]
[2026-08-11 19:03:19,110] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:518] [PID:31] sample_packing_eff_est: 1.0
[2026-08-11 19:03:19,110] [DEBUG] [axolotl.utils.trainer.calculate_total_num_steps:523] [PID:31] total_num_steps: 96
[2026-08-11 19:03:19,111] [INFO] [axolotl.utils.data.sft._prepare_standard_dataset:121] [PID:31] Maximum number of steps set at 96
[2026-08-11 19:03:19,121] [DEBUG] [axolotl.train.setup_model_and_tokenizer:70] [PID:31] loading tokenizer... c4tdr0ut/heli-pretrain
[2026-08-11 19:03:20,183] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:311] [PID:31] EOS: 2 / </s>
[2026-08-11 19:03:20,183] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:31] BOS: 1 / <s>
[2026-08-11 19:03:20,184] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:31] PAD: 0 / <unk>
[2026-08-11 19:03:20,184] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:31] UNK: 0 / <unk>
[2026-08-11 19:03:20,184] [DEBUG] [axolotl.train.setup_model_and_tokenizer:81] [PID:31] Loading model
[2026-08-11 19:03:20,301] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:75] [PID:31] Patched OptimState8bit for torch.compile compatibility
[2026-08-11 19:03:20,301] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:122] [PID:31] Patched OptimState4bit for torch.compile compatibility
[2026-08-11 19:03:20,302] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:154] [PID:31] Patched OptimStateFp8 for torch.compile compatibility
[2026-08-11 19:03:20,310] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_evaluation_loop:94] [PID:31] Patched Trainer.evaluation_loop with nanmean loss calculation
[2026-08-11 19:03:20,311] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_maybe_log_save_evaluate:148] [PID:31] Patched Trainer._maybe_log_save_evaluate with nanmean loss calculation
[2026-08-11 19:03:20,312] [INFO] [axolotl.loaders.patch_manager._apply_multipack_patches:912] [PID:31] Applying multipack dataloader patch for sample packing...
[2026-08-11 19:03:20,322] [INFO] [axolotl.monkeypatch.attention.flash_attn_4.fa4_usable:140] [PID:31] Flash Attention 4 is available for your GPU and offers faster training. To enable: pip install --pre flash-attn-4
[2026-08-11 19:03:24,678] [WARNING] [py.warnings._showwarnmsg:112] [PID:31] /workspace/axolotl-venv/lib/python3.12/site-packages/torch/jit/_script.py:1488: DeprecationWarning: `torch.jit.script` is deprecated. Please switch to `torch.compile` or `torch.export`.
warnings.warn(
[2026-08-11 19:03:24,784] [WARNING] [py.warnings._showwarnmsg:112] [PID:31] /workspace/axolotl-venv/lib/python3.12/site-packages/fla/ops/nsa/parallel.py:20: ImportWarning: Flash Attention is not installed. Please install it via `pip install flash-attn --no-build-isolation`
warnings.warn(
[2026-08-11 19:03:25,005] [INFO] [axolotl.integrations.liger.plugin.pre_model_load:145] [PID:31] Applying LIGER to mistral with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'swiglu': True}
[2026-08-11 19:04:14,342] [WARNING] [kernels._versions.resolve_version_spec_as_ref:83] [PID:31] You are using version 1 of 'kernels-community/flash-attn2', but version 3 is available.
Fetching ... files: 0it [00:00, ?it/s] Fetching ... files: 1it [00:00, 8.89it/s] Fetching ... files: 2it [00:05, 3.05s/it] Fetching ... files: 23it [00:05, 4.40it/s]
Loading weights: 0%| | 0/291 [00:00<?, ?it/s] Loading weights: 1%| | 2/291 [00:00<00:17, 16.34it/s] Loading weights: 5%|β–Œ | 15/291 [00:00<00:03, 70.76it/s] Loading weights: 11%|β–ˆ | 31/291 [00:00<00:02, 100.08it/s] Loading weights: 14%|β–ˆβ– | 42/291 [00:00<00:02, 98.29it/s] Loading weights: 20%|β–ˆβ–‰ | 58/291 [00:00<00:02, 115.07it/s] Loading weights: 24%|β–ˆβ–ˆβ– | 70/291 [00:00<00:01, 111.24it/s] Loading weights: 29%|β–ˆβ–ˆβ–‰ | 85/291 [00:00<00:01, 116.40it/s] Loading weights: 33%|β–ˆβ–ˆβ–ˆβ–Ž | 97/291 [00:00<00:01, 108.51it/s] Loading weights: 38%|β–ˆβ–ˆβ–ˆβ–Š | 112/291 [00:01<00:01, 116.78it/s] Loading weights: 43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 124/291 [00:01<00:01, 112.18it/s] Loading weights: 48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š | 139/291 [00:01<00:01, 117.28it/s] Loading weights: 52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 151/291 [00:01<00:01, 111.38it/s] Loading weights: 57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 166/291 [00:01<00:01, 117.88it/s] Loading weights: 61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 178/291 [00:01<00:01, 111.62it/s] Loading weights: 65%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 190/291 [00:01<00:00, 112.22it/s] Loading weights: 69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 202/291 [00:01<00:00, 111.04it/s] Loading weights: 74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 214/291 [00:01<00:00, 107.60it/s] Loading weights: 79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 229/291 [00:02<00:00, 115.35it/s] Loading weights: 83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 241/291 [00:02<00:00, 110.60it/s] Loading weights: 88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 256/291 [00:02<00:00, 116.51it/s] Loading weights: 92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 268/291 [00:02<00:00, 111.35it/s] Loading weights: 97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 283/291 [00:02<00:00, 116.71it/s] Loading weights: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 291/291 [00:02<00:00, 110.96it/s]
[2026-08-11 19:04:23,372] [INFO] [axolotl.loaders.model._configure_embedding_dtypes:482] [PID:31] Converting modules to torch.bfloat16
[2026-08-11 19:04:23,375] [DEBUG] [axolotl.loaders.model.log_gpu_memory_usage:127] [PID:31] Memory usage after model load 14.251GB (+14.251GB allocated, +14.502GB reserved)
[2026-08-11 19:04:24,900] [WARNING] [py.warnings._showwarnmsg:112] [PID:31] /workspace/axolotl-venv/lib/python3.12/site-packages/torch/jit/_script.py:1488: DeprecationWarning: `torch.jit.script` is deprecated. Please switch to `torch.compile` or `torch.export`.
warnings.warn(
[2026-08-11 19:04:25,089] [WARNING] [py.warnings._showwarnmsg:112] [PID:31] /workspace/axolotl-venv/lib/python3.12/site-packages/torch/jit/_script.py:1640: DeprecationWarning: `torch.jit.interface` is deprecated. Please use `torch.compile` instead.
warnings.warn(
[2026-08-11 19:04:25,356] [WARNING] [accelerate.utils.other.check_os_kernel:520] [PID:31] [RANK 0] Detected kernel version 4.19.0, which is below the recommended minimum of 5.5.0; this can cause the process to hang. It is recommended to upgrade the kernel to the minimum version or higher.
[2026-08-11 19:04:27,407] [INFO] [axolotl.train.save_initial_configs:486] [PID:31] Pre-saving tokenizer to ./finetune-model-output...
[2026-08-11 19:04:27,448] [INFO] [axolotl.train.save_initial_configs:491] [PID:31] Pre-saving model config to ./finetune-model-output...
[2026-08-11 19:04:27,453] [INFO] [axolotl.train.execute_training:234] [PID:31] Starting trainer...
[2026-08-11 19:04:28,034] [WARNING] [py.warnings._showwarnmsg:112] [PID:31] /root/.local/share/uv/python/cpython-3.12.13-linux-x86_64-gnu/lib/python3.12/multiprocessing/popen_fork.py:66: DeprecationWarning: This process (pid=31) is multi-threaded, use of fork() may lead to deadlocks in the child.
self.pid = os.fork()
[2026-08-11 19:04:30,659] [DEBUG] [axolotl.utils.samplers.multipack.__len__:467] [PID:31] generate_batches time: 1.315068244934082
[2026-08-11 19:04:31,972] [DEBUG] [axolotl.utils.samplers.multipack.__len__:467] [PID:31] generate_batches time: 1.3127715587615967
[2026-08-11 19:04:33,341] [DEBUG] [axolotl.utils.samplers.multipack.__len__:467] [PID:31] generate_batches time: 1.368100643157959
[2026-08-11 19:04:34,639] [DEBUG] [axolotl.utils.samplers.multipack.__len__:467] [PID:31] generate_batches time: 1.297865867614746
[2026-08-11 19:04:34,639] [INFO] [axolotl.utils.samplers.multipack.calc_min_len:443] [PID:31] gather_len_batches: [240]
0%| | 0/96 [00:00<?, ?it/s][2026-08-11 19:04:34,682] [INFO] [axolotl.core.trainers.base.evaluate:464] [PID:31] Running evaluation step...
0%| | 0/90 [00:00<?, ?it/s]
2%|▏ | 2/90 [00:00<00:28, 3.05it/s]
4%|▍ | 4/90 [00:01<00:26, 3.30it/s]
6%|β–Œ | 5/90 [00:01<00:29, 2.88it/s]
7%|β–‹ | 6/90 [00:02<00:33, 2.51it/s]
8%|β–Š | 7/90 [00:02<00:28, 2.95it/s]
9%|β–‰ | 8/90 [00:02<00:32, 2.54it/s]
10%|β–ˆ | 9/90 [00:03<00:30, 2.65it/s]
11%|β–ˆ | 10/90 [00:03<00:27, 2.90it/s]
12%|β–ˆβ– | 11/90 [00:03<00:27, 2.83it/s]
13%|β–ˆβ–Ž | 12/90 [00:04<00:30, 2.53it/s]
14%|β–ˆβ– | 13/90 [00:04<00:33, 2.32it/s]
16%|β–ˆβ–Œ | 14/90 [00:05<00:34, 2.20it/s]
17%|β–ˆβ–‹ | 15/90 [00:05<00:35, 2.11it/s]
18%|β–ˆβ–Š | 16/90 [00:06<00:27, 2.64it/s]
19%|β–ˆβ–‰ | 17/90 [00:06<00:30, 2.39it/s]
20%|β–ˆβ–ˆ | 18/90 [00:07<00:32, 2.22it/s]
21%|β–ˆβ–ˆ | 19/90 [00:07<00:33, 2.14it/s]
22%|β–ˆβ–ˆβ– | 20/90 [00:08<00:33, 2.06it/s]
23%|β–ˆβ–ˆβ–Ž | 21/90 [00:08<00:32, 2.14it/s]
26%|β–ˆβ–ˆβ–Œ | 23/90 [00:09<00:25, 2.64it/s]
27%|β–ˆβ–ˆβ–‹ | 24/90 [00:09<00:27, 2.40it/s]
28%|β–ˆβ–ˆβ–Š | 25/90 [00:10<00:28, 2.26it/s]
29%|β–ˆβ–ˆβ–‰ | 26/90 [00:10<00:23, 2.78it/s]
30%|β–ˆβ–ˆβ–ˆ | 27/90 [00:10<00:22, 2.80it/s]
32%|β–ˆβ–ˆβ–ˆβ– | 29/90 [00:11<00:18, 3.30it/s]
33%|β–ˆβ–ˆβ–ˆβ–Ž | 30/90 [00:11<00:19, 3.02it/s]
34%|β–ˆβ–ˆβ–ˆβ– | 31/90 [00:12<00:22, 2.59it/s]
36%|β–ˆβ–ˆβ–ˆβ–Œ | 32/90 [00:12<00:22, 2.54it/s]
37%|β–ˆβ–ˆβ–ˆβ–‹ | 33/90 [00:13<00:24, 2.35it/s]
38%|β–ˆβ–ˆβ–ˆβ–Š | 34/90 [00:13<00:25, 2.20it/s]
39%|β–ˆβ–ˆβ–ˆβ–‰ | 35/90 [00:13<00:22, 2.40it/s]
40%|β–ˆβ–ˆβ–ˆβ–ˆ | 36/90 [00:14<00:23, 2.34it/s]
41%|β–ˆβ–ˆβ–ˆβ–ˆ | 37/90 [00:14<00:19, 2.65it/s]
42%|β–ˆβ–ˆβ–ˆβ–ˆβ– | 38/90 [00:15<00:20, 2.56it/s]
43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 39/90 [00:15<00:17, 2.93it/s]
44%|β–ˆβ–ˆβ–ˆβ–ˆβ– | 40/90 [00:15<00:16, 3.02it/s]
46%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 41/90 [00:15<00:17, 2.83it/s]
47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 42/90 [00:16<00:15, 3.19it/s]
48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š | 43/90 [00:16<00:16, 2.85it/s]
49%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 44/90 [00:16<00:14, 3.13it/s]
50%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 45/90 [00:17<00:14, 3.04it/s]
51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 46/90 [00:17<00:11, 3.72it/s]
52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 47/90 [00:17<00:11, 3.79it/s]
53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 48/90 [00:17<00:12, 3.43it/s]
54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 49/90 [00:18<00:12, 3.34it/s]
56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 50/90 [00:18<00:11, 3.57it/s]
57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 51/90 [00:18<00:13, 2.89it/s]
58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 52/90 [00:19<00:12, 3.01it/s]
59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 53/90 [00:19<00:14, 2.58it/s]
60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 54/90 [00:20<00:12, 2.88it/s]
61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 55/90 [00:20<00:13, 2.51it/s]
62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 56/90 [00:20<00:12, 2.62it/s]
63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 57/90 [00:21<00:10, 3.28it/s]
64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 58/90 [00:21<00:08, 3.80it/s]
66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 59/90 [00:21<00:10, 2.95it/s]
67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 60/90 [00:22<00:11, 2.54it/s]
68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 61/90 [00:22<00:11, 2.60it/s]
69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 62/90 [00:23<00:11, 2.45it/s]
70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 63/90 [00:23<00:09, 2.80it/s]
71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 64/90 [00:23<00:07, 3.39it/s]
72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 65/90 [00:23<00:09, 2.75it/s]
73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 66/90 [00:24<00:09, 2.47it/s]
74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 67/90 [00:24<00:08, 2.61it/s]
76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 68/90 [00:25<00:09, 2.37it/s]
77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 69/90 [00:25<00:08, 2.61it/s]
78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 70/90 [00:25<00:06, 2.94it/s]
79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 71/90 [00:26<00:06, 2.75it/s]
80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 72/90 [00:26<00:05, 3.12it/s]
81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 73/90 [00:26<00:05, 3.31it/s]
82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 74/90 [00:27<00:05, 2.71it/s]
83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 75/90 [00:27<00:06, 2.42it/s]
84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 76/90 [00:28<00:05, 2.69it/s]
86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 77/90 [00:28<00:04, 3.10it/s]
87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 78/90 [00:28<00:04, 2.84it/s]
88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 79/90 [00:29<00:04, 2.47it/s]
89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 80/90 [00:29<00:03, 2.73it/s]
90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 81/90 [00:30<00:03, 2.37it/s]
91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 82/90 [00:30<00:03, 2.27it/s]
92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 83/90 [00:31<00:03, 2.15it/s]
93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 84/90 [00:31<00:02, 2.07it/s]
94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 85/90 [00:32<00:02, 2.04it/s]
96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 86/90 [00:32<00:02, 1.99it/s]
97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 87/90 [00:32<00:01, 2.61it/s]
98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 88/90 [00:33<00:00, 2.35it/s]
99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 89/90 [00:33<00:00, 2.63it/s]
100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 90/90 [00:33<00:00, 2.69it/s]
{'eval_loss': '1.762', 'eval_runtime': '36.99', 'eval_samples_per_second': '9.787', 'eval_steps_per_second': '2.46', 'eval_ppl': '5.824', 'memory/max_active (GiB)': '19.79', 'memory/max_allocated (GiB)': '19.79', 'memory/device_reserved (GiB)': '26.67', 'epoch': 0}
0%| | 0/96 [00:36<?, ?it/s]
100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 90/90 [00:34<00:00, 2.69it/s]
 1%| | 1/96 [02:01<3:12:52, 121.82s/it] {'loss': '1.668', 'grad_norm': '30.88', 'learning_rate': '2e-05', 'ppl': '5.3', 'memory/max_active (GiB)': '57.85', 'memory/max_allocated (GiB)': '57.85', 'memory/device_reserved (GiB)': '60.99', 'tokens/trainable': 742568, 'tokens/total': 749902, 'epoch': '0.04167'}
1%| | 1/96 [02:01<3:12:52, 121.82s/it] 2%|▏ | 2/96 [03:20<2:31:24, 96.64s/it] {'loss': '1.606', 'grad_norm': '28.88', 'learning_rate': '2e-05', 'ppl': '4.985', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '65.01', 'tokens/train_per_sec_per_gpu': '9394', 'tokens/trainable': 1484702, 'tokens/total': 1499819, 'epoch': '0.08333'}
2%|▏ | 2/96 [03:20<2:31:24, 96.64s/it] 3%|β–Ž | 3/96 [04:38<2:16:08, 87.83s/it] {'loss': '1.242', 'grad_norm': '19.12', 'learning_rate': '2e-05', 'ppl': '3.461', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '65.01', 'tokens/train_per_sec_per_gpu': '9349', 'tokens/trainable': 2207855, 'tokens/total': 2249708, 'epoch': '0.125'}
3%|β–Ž | 3/96 [04:38<2:16:08, 87.83s/it] 4%|▍ | 4/96 [05:56<2:08:53, 84.06s/it] {'loss': '1.165', 'grad_norm': '33.5', 'learning_rate': '2e-05', 'ppl': '3.206', 'memory/max_active (GiB)': '58.05', 'memory/max_allocated (GiB)': '58.05', 'memory/device_reserved (GiB)': '65.01', 'tokens/train_per_sec_per_gpu': '9482', 'tokens/trainable': 2950021, 'tokens/total': 2999475, 'epoch': '0.1667'}
4%|▍ | 4/96 [05:56<2:08:53, 84.06s/it] 5%|β–Œ | 5/96 [07:14<2:03:55, 81.71s/it] {'loss': '1.185', 'grad_norm': '17', 'learning_rate': '2e-05', 'ppl': '3.272', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '65.01', 'tokens/train_per_sec_per_gpu': '9495', 'tokens/trainable': 3686352, 'tokens/total': 3749133, 'epoch': '0.2083'}
5%|β–Œ | 5/96 [07:14<2:03:55, 81.71s/it] 6%|β–‹ | 6/96 [08:31<2:00:32, 80.36s/it] {'loss': '1.06', 'grad_norm': '8.812', 'learning_rate': '2e-05', 'ppl': '2.885', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '65.01', 'tokens/train_per_sec_per_gpu': '9440', 'tokens/trainable': 4420112, 'tokens/total': 4498747, 'epoch': '0.25'}
6%|β–‹ | 6/96 [08:31<2:00:32, 80.36s/it] 7%|β–‹ | 7/96 [09:49<1:57:59, 79.55s/it] {'loss': '0.9709', 'grad_norm': '6.844', 'learning_rate': '2e-05', 'ppl': '2.64', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '65.01', 'tokens/train_per_sec_per_gpu': '9429', 'tokens/trainable': 5154407, 'tokens/total': 5248232, 'epoch': '0.2917'}
7%|β–‹ | 7/96 [09:49<1:57:59, 79.55s/it] 8%|β–Š | 8/96 [11:07<1:56:02, 79.12s/it] {'loss': '0.9857', 'grad_norm': '4.281', 'learning_rate': '2e-05', 'ppl': '2.68', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9504', 'tokens/trainable': 5897628, 'tokens/total': 5997795, 'epoch': '0.3333'}
8%|β–Š | 8/96 [11:07<1:56:02, 79.12s/it] 9%|β–‰ | 9/96 [12:25<1:54:02, 78.65s/it] {'loss': '0.9362', 'grad_norm': '3.953', 'learning_rate': '2e-05', 'ppl': '2.55', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9489', 'tokens/trainable': 6634055, 'tokens/total': 6747342, 'epoch': '0.375'}
9%|β–‰ | 9/96 [12:25<1:54:02, 78.65s/it] 10%|β–ˆ | 10/96 [13:43<1:52:20, 78.37s/it] {'loss': '0.9151', 'grad_norm': '4.656', 'learning_rate': '2e-05', 'ppl': '2.497', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9548', 'tokens/trainable': 7376513, 'tokens/total': 7497031, 'epoch': '0.4167'}
10%|β–ˆ | 10/96 [13:43<1:52:20, 78.37s/it] 11%|β–ˆβ– | 11/96 [15:01<1:50:50, 78.24s/it] {'loss': '0.9457', 'grad_norm': '5.469', 'learning_rate': '2e-05', 'ppl': '2.575', 'memory/max_active (GiB)': '58.05', 'memory/max_allocated (GiB)': '58.05', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9519', 'tokens/trainable': 8118382, 'tokens/total': 8246701, 'epoch': '0.4583'}
11%|β–ˆβ– | 11/96 [15:01<1:50:50, 78.24s/it] 12%|β–ˆβ–Ž | 12/96 [16:18<1:49:20, 78.10s/it] {'loss': '0.8496', 'grad_norm': '2.969', 'learning_rate': '2e-05', 'ppl': '2.339', 'memory/max_active (GiB)': '58.05', 'memory/max_allocated (GiB)': '58.05', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9560', 'tokens/trainable': 8862049, 'tokens/total': 8995758, 'epoch': '0.5'}
12%|β–ˆβ–Ž | 12/96 [16:18<1:49:20, 78.10s/it] 14%|β–ˆβ–Ž | 13/96 [17:36<1:47:44, 77.88s/it] {'loss': '0.8735', 'grad_norm': '3.25', 'learning_rate': '2e-05', 'ppl': '2.395', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9489', 'tokens/trainable': 9596186, 'tokens/total': 9745292, 'epoch': '0.5417'}
14%|β–ˆβ–Ž | 13/96 [17:36<1:47:44, 77.88s/it] 15%|β–ˆβ– | 14/96 [18:53<1:46:18, 77.79s/it] {'loss': '0.8467', 'grad_norm': '2.234', 'learning_rate': '2e-05', 'ppl': '2.332', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9480', 'tokens/trainable': 10331619, 'tokens/total': 10494275, 'epoch': '0.5833'}
15%|β–ˆβ– | 14/96 [18:53<1:46:18, 77.79s/it] 16%|β–ˆβ–Œ | 15/96 [20:11<1:44:48, 77.64s/it] {'loss': '0.9049', 'grad_norm': '2.453', 'learning_rate': '2e-05', 'ppl': '2.472', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9656', 'tokens/trainable': 11077979, 'tokens/total': 11243240, 'epoch': '0.625'}
16%|β–ˆβ–Œ | 15/96 [20:11<1:44:48, 77.64s/it] 17%|β–ˆβ–‹ | 16/96 [21:28<1:43:34, 77.68s/it] {'loss': '0.8693', 'grad_norm': '2.672', 'learning_rate': '2e-05', 'ppl': '2.385', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9484', 'tokens/trainable': 11815398, 'tokens/total': 11991989, 'epoch': '0.6667'}
17%|β–ˆβ–‹ | 16/96 [21:28<1:43:34, 77.68s/it] 18%|β–ˆβ–Š | 17/96 [22:46<1:42:22, 77.76s/it] {'loss': '0.8343', 'grad_norm': '2.891', 'learning_rate': '2e-05', 'ppl': '2.303', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9529', 'tokens/trainable': 12558179, 'tokens/total': 12741571, 'epoch': '0.7083'}
18%|β–ˆβ–Š | 17/96 [22:46<1:42:22, 77.76s/it] 19%|β–ˆβ–‰ | 18/96 [24:04<1:40:54, 77.62s/it] {'loss': '0.8577', 'grad_norm': '2.438', 'learning_rate': '2e-05', 'ppl': '2.358', 'memory/max_active (GiB)': '58.05', 'memory/max_allocated (GiB)': '58.05', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9480', 'tokens/trainable': 13291084, 'tokens/total': 13490669, 'epoch': '0.75'}
19%|β–ˆβ–‰ | 18/96 [24:04<1:40:54, 77.62s/it] 20%|β–ˆβ–‰ | 19/96 [25:21<1:39:36, 77.62s/it] {'loss': '0.8067', 'grad_norm': '2.438', 'learning_rate': '2e-05', 'ppl': '2.241', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9488', 'tokens/trainable': 14027534, 'tokens/total': 14239739, 'epoch': '0.7917'}
20%|β–ˆβ–‰ | 19/96 [25:21<1:39:36, 77.62s/it] 21%|β–ˆβ–ˆ | 20/96 [26:39<1:38:27, 77.74s/it] {'loss': '0.8145', 'grad_norm': '2.422', 'learning_rate': '2e-05', 'ppl': '2.258', 'memory/max_active (GiB)': '58.05', 'memory/max_allocated (GiB)': '58.05', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9524', 'tokens/trainable': 14770394, 'tokens/total': 14988473, 'epoch': '0.8333'}
21%|β–ˆβ–ˆ | 20/96 [26:39<1:38:27, 77.74s/it] 22%|β–ˆβ–ˆβ– | 21/96 [27:57<1:36:59, 77.59s/it] {'loss': '0.7839', 'grad_norm': '2.406', 'learning_rate': '2e-05', 'ppl': '2.19', 'memory/max_active (GiB)': '58.05', 'memory/max_allocated (GiB)': '58.05', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9443', 'tokens/trainable': 15499968, 'tokens/total': 15736924, 'epoch': '0.875'}
22%|β–ˆβ–ˆβ– | 21/96 [27:57<1:36:59, 77.59s/it] 23%|β–ˆβ–ˆβ–Ž | 22/96 [29:15<1:35:57, 77.80s/it] {'loss': '0.8035', 'grad_norm': '1.961', 'learning_rate': '2e-05', 'ppl': '2.233', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9450', 'tokens/trainable': 16239694, 'tokens/total': 16485262, 'epoch': '0.9167'}
23%|β–ˆβ–ˆβ–Ž | 22/96 [29:15<1:35:57, 77.80s/it] 24%|β–ˆβ–ˆβ– | 23/96 [30:32<1:34:36, 77.76s/it] {'loss': '0.8221', 'grad_norm': '2.031', 'learning_rate': '2e-05', 'ppl': '2.275', 'memory/max_active (GiB)': '58.05', 'memory/max_allocated (GiB)': '58.05', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9388', 'tokens/trainable': 16968820, 'tokens/total': 17233354, 'epoch': '0.9583'}
24%|β–ˆβ–ˆβ– | 23/96 [30:33<1:34:36, 77.76s/it] 25%|β–ˆβ–ˆβ–Œ | 24/96 [31:46<1:31:53, 76.58s/it] {'loss': '0.7618', 'grad_norm': '2.062', 'learning_rate': '2e-05', 'ppl': '2.142', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '67.02', 'tokens/train_per_sec_per_gpu': '9486', 'tokens/trainable': 17669050, 'tokens/total': 17945280, 'epoch': '1'}
25%|β–ˆβ–ˆβ–Œ | 24/96 [31:46<1:31:53, 76.58s/it][2026-08-11 19:36:21,497] [INFO] [axolotl.core.trainers.base.evaluate:464] [PID:31] Running evaluation step...
0%| | 0/90 [00:00<?, ?it/s]
2%|▏ | 2/90 [00:00<00:31, 2.80it/s]
3%|β–Ž | 3/90 [00:00<00:23, 3.72it/s]
4%|▍ | 4/90 [00:01<00:31, 2.74it/s]
6%|β–Œ | 5/90 [00:01<00:33, 2.51it/s]
7%|β–‹ | 6/90 [00:02<00:36, 2.28it/s]
8%|β–Š | 7/90 [00:02<00:29, 2.78it/s]
9%|β–‰ | 8/90 [00:03<00:33, 2.41it/s]
10%|β–ˆ | 9/90 [00:03<00:31, 2.58it/s]
11%|β–ˆ | 10/90 [00:03<00:27, 2.87it/s]
12%|β–ˆβ– | 11/90 [00:04<00:27, 2.83it/s]
13%|β–ˆβ–Ž | 12/90 [00:04<00:30, 2.58it/s]
14%|β–ˆβ– | 13/90 [00:05<00:33, 2.30it/s]
16%|β–ˆβ–Œ | 14/90 [00:05<00:34, 2.20it/s]
17%|β–ˆβ–‹ | 15/90 [00:06<00:35, 2.09it/s]
18%|β–ˆβ–Š | 16/90 [00:06<00:28, 2.62it/s]
19%|β–ˆβ–‰ | 17/90 [00:06<00:30, 2.37it/s]
20%|β–ˆβ–ˆ | 18/90 [00:07<00:32, 2.21it/s]
21%|β–ˆβ–ˆ | 19/90 [00:07<00:33, 2.12it/s]
22%|β–ˆβ–ˆβ– | 20/90 [00:08<00:34, 2.06it/s]
23%|β–ˆβ–ˆβ–Ž | 21/90 [00:08<00:26, 2.57it/s]
24%|β–ˆβ–ˆβ– | 22/90 [00:08<00:23, 2.86it/s]
26%|β–ˆβ–ˆβ–Œ | 23/90 [00:09<00:26, 2.48it/s]
27%|β–ˆβ–ˆβ–‹ | 24/90 [00:09<00:28, 2.28it/s]
28%|β–ˆβ–ˆβ–Š | 25/90 [00:10<00:30, 2.16it/s]
29%|β–ˆβ–ˆβ–‰ | 26/90 [00:10<00:23, 2.75it/s]
30%|β–ˆβ–ˆβ–ˆ | 27/90 [00:10<00:21, 2.93it/s]
32%|β–ˆβ–ˆβ–ˆβ– | 29/90 [00:11<00:18, 3.28it/s]
33%|β–ˆβ–ˆβ–ˆβ–Ž | 30/90 [00:11<00:20, 2.98it/s]
34%|β–ˆβ–ˆβ–ˆβ– | 31/90 [00:12<00:22, 2.59it/s]
36%|β–ˆβ–ˆβ–ˆβ–Œ | 32/90 [00:12<00:22, 2.53it/s]
37%|β–ˆβ–ˆβ–ˆβ–‹ | 33/90 [00:13<00:24, 2.33it/s]
38%|β–ˆβ–ˆβ–ˆβ–Š | 34/90 [00:13<00:25, 2.19it/s]
39%|β–ˆβ–ˆβ–ˆβ–‰ | 35/90 [00:13<00:22, 2.41it/s]
40%|β–ˆβ–ˆβ–ˆβ–ˆ | 36/90 [00:14<00:23, 2.34it/s]
41%|β–ˆβ–ˆβ–ˆβ–ˆ | 37/90 [00:14<00:20, 2.65it/s]
42%|β–ˆβ–ˆβ–ˆβ–ˆβ– | 38/90 [00:15<00:20, 2.56it/s]
43%|β–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 39/90 [00:15<00:17, 2.93it/s]
44%|β–ˆβ–ˆβ–ˆβ–ˆβ– | 40/90 [00:15<00:16, 3.02it/s]
46%|β–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 41/90 [00:16<00:16, 2.95it/s]
47%|β–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 42/90 [00:16<00:14, 3.27it/s]
48%|β–ˆβ–ˆβ–ˆβ–ˆβ–Š | 43/90 [00:16<00:16, 2.86it/s]
49%|β–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 44/90 [00:16<00:14, 3.16it/s]
50%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 45/90 [00:17<00:14, 3.05it/s]
51%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 46/90 [00:17<00:11, 3.74it/s]
52%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 47/90 [00:17<00:11, 3.79it/s]
53%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 48/90 [00:18<00:12, 3.46it/s]
54%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 49/90 [00:18<00:12, 3.37it/s]
56%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 50/90 [00:18<00:10, 3.72it/s]
57%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 51/90 [00:19<00:13, 2.89it/s]
58%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 52/90 [00:19<00:12, 3.00it/s]
59%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 53/90 [00:19<00:14, 2.57it/s]
60%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 54/90 [00:20<00:12, 2.89it/s]
61%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 55/90 [00:20<00:14, 2.50it/s]
62%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 56/90 [00:21<00:12, 2.62it/s]
63%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 57/90 [00:21<00:10, 3.29it/s]
64%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 58/90 [00:21<00:08, 3.78it/s]
66%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 59/90 [00:21<00:10, 2.94it/s]
67%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 60/90 [00:22<00:11, 2.53it/s]
68%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 61/90 [00:22<00:10, 2.68it/s]
69%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 62/90 [00:23<00:11, 2.47it/s]
70%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 63/90 [00:23<00:09, 2.80it/s]
71%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 64/90 [00:23<00:07, 3.30it/s]
72%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 65/90 [00:24<00:09, 2.77it/s]
73%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 66/90 [00:24<00:09, 2.47it/s]
74%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 67/90 [00:24<00:08, 2.62it/s]
76%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 68/90 [00:25<00:09, 2.37it/s]
77%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 69/90 [00:25<00:08, 2.62it/s]
78%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 70/90 [00:25<00:06, 2.94it/s]
79%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 71/90 [00:26<00:06, 2.79it/s]
80%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 72/90 [00:26<00:05, 3.16it/s]
81%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 73/90 [00:26<00:05, 3.31it/s]
82%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 74/90 [00:27<00:05, 2.72it/s]
83%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž | 75/90 [00:27<00:06, 2.42it/s]
84%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ– | 76/90 [00:28<00:05, 2.69it/s]
86%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ | 77/90 [00:28<00:04, 3.10it/s]
87%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹ | 78/90 [00:28<00:04, 2.83it/s]
88%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š | 79/90 [00:29<00:04, 2.46it/s]
89%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰ | 80/90 [00:29<00:03, 2.70it/s]
90%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 81/90 [00:30<00:03, 2.38it/s]
91%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ | 82/90 [00:30<00:03, 2.27it/s]
92%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 83/90 [00:31<00:03, 2.16it/s]
93%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Ž| 84/90 [00:31<00:02, 2.07it/s]
94%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–| 85/90 [00:32<00:02, 2.04it/s]
96%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Œ| 86/90 [00:32<00:02, 1.99it/s]
97%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‹| 87/90 [00:32<00:01, 2.61it/s]
98%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–Š| 88/90 [00:33<00:00, 2.35it/s]
99%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–‰| 89/90 [00:33<00:00, 2.62it/s]
100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 90/90 [00:33<00:00, 2.69it/s]
{'eval_loss': '0.8244', 'eval_runtime': '35.13', 'eval_samples_per_second': '10.31', 'eval_steps_per_second': '2.591', 'eval_ppl': '2.281', 'memory/max_active (GiB)': '20.04', 'memory/max_allocated (GiB)': '20.04', 'memory/device_reserved (GiB)': '71.9', 'epoch': '1'}
25%|β–ˆβ–ˆβ–Œ | 24/96 [32:21<1:31:53, 76.58s/it]
100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 90/90 [00:34<00:00, 2.69it/s]
[2026-08-11 19:36:56,637] [INFO] [axolotl.core.trainers.base._save:928] [PID:31] Saving model checkpoint to ./finetune-model-output/checkpoint-24
Writing model shards: 0%| | 0/1 [00:00<?, ?it/s]
Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:09<00:00, 9.37s/it] Writing model shards: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 1/1 [00:09<00:00, 9.37s/it]
[2026-08-11 19:37:30,375] [WARNING] [py.warnings._showwarnmsg:112] [PID:31] /root/.local/share/uv/python/cpython-3.12.13-linux-x86_64-gnu/lib/python3.12/multiprocessing/popen_fork.py:66: DeprecationWarning: This process (pid=31) is multi-threaded, use of fork() may lead to deadlocks in the child.
self.pid = os.fork()
26%|β–ˆβ–ˆβ–Œ | 25/96 [34:17<1:57:04, 98.93s/it] {'loss': '0.645', 'grad_norm': '3.094', 'learning_rate': '2e-05', 'ppl': '1.906', 'memory/max_active (GiB)': '58.06', 'memory/max_allocated (GiB)': '58.06', 'memory/device_reserved (GiB)': '66.16', 'tokens/train_per_sec_per_gpu': '4903', 'tokens/trainable': 18409932, 'tokens/total': 18695242, 'epoch': '1.042'}
26%|β–ˆβ–ˆβ–Œ | 25/96 [34:17<1:57:04, 98.93s/it]