CrystaLLM-pi_mp_20_base / training_args.json
c-bone's picture
Use the CrystaLLM-pi training config: public c-bone/mp_20_pxrd (same CIFs as the training data), muon_lr, adam_beta2 and min_lr_rate read from the checkpoint
d112d0e verified
Raw History Blame Contribute Delete
2.11 kB
{
"activate_conditionality": null,
"adam_beta1": 0.9,
"adam_beta2": 0.995,
"attention_dropout": 0.1,
"codecarbon": true,
"cond_dropout": null,
"cond_lr": null,
"cond_wd": null,
"condition_columns": null,
"config": "_config_files/training/unconditional/mp-20-text.jsonc",
"context_extension_warmup_steps": 0,
"context_length": 1024,
"data_seed": 1,
"dataset_HF": "c-bone/mp_20_pxrd",
"deepspeed_config": "_config_files/deepspeed_default.json",
"do_sample": "True",
"early_stopping_patience": 15,
"early_stopping_threshold": 5e-06,
"embedding_dropout": 0.1,
"eval_batch_size": 32,
"eval_steps": 1000,
"eval_strategy": "steps",
"fp16": false,
"gen_max_length": 1024,
"grad_clip": 1.0,
"gradient_accumulation_steps": 1,
"greater_is_better": false,
"input_parquet": null,
"learning_rate": 0.002,
"load_best_model_at_end": true,
"logging_steps": 100,
"lr_scheduler_kwargs": {
"min_lr_rate": 0.01
},
"lr_scheduler_type": "cosine_with_min_lr",
"max_return_attempts": 1,
"max_samples": null,
"max_steps": 5000,
"metric_for_best_model": "eval_loss",
"model_ckpt_dir": "model_ckpts/cif-gpt2-small/checkpoint-400",
"muon_lr": 0.02,
"muon_momentum": 0.95,
"n_embd": 512,
"n_head": 8,
"n_heads_sharing_slider": null,
"n_hidden_cond": null,
"n_layer": 8,
"n_prefix_tokens": null,
"num_return_sequences": 1,
"optimizer": "muon",
"output_dir": "model_ckpts/mp_20_text/",
"output_parquet": null,
"pretrained_model_dir": null,
"pretrained_tokenizer_dir": "HF-cif-tokenizer",
"remove_CIFs_above_context": false,
"remove_CIFs_with_unk": false,
"report_to": "wandb",
"residual_dropout": 0.1,
"save_strategy": "steps",
"save_total_limit": 2,
"scoring_mode": "None",
"screening_profile": "application",
"seed": 1,
"target_valid_cifs": 1,
"temperature": 1.0,
"top_k": 15,
"top_p": 0.95,
"torch_compile": true,
"tracker_project": "CrystaLLM-pi",
"train_batch_size": 32,
"wandb_project_folder": "CrystaLLM-lematbench",
"warmup_ratio": 0.0,
"warmup_steps": null,
"weight_decay": 0.1
}