Download training_config.yaml from Shuibai12138/Open-Dcoder-0.5B-CDLM-OpenCodeInstruct: direct link, hf CLI and curl.
- Browser
- Download file 2.73 kB
-
https://huggingface.co/Shuibai12138/Open-Dcoder-0.5B-CDLM-OpenCodeInstruct/resolve/main/training_config.yaml
- Command line
-
hf download hf://Shuibai12138/Open-Dcoder-0.5B-CDLM-OpenCodeInstruct/training_config.yaml
-
curl -L -o training_config.yaml https://huggingface.co/Shuibai12138/Open-Dcoder-0.5B-CDLM-OpenCodeInstruct/resolve/main/training_config.yaml
2.73 kB
| # Fully resolved training configuration of this run (saved by the trainer as veomni_cli.yaml). | |
| # Paths replaced by placeholders. World size 4 (4 x A100-PCIE-40GB), gradient accumulation 1. | |
| data: | |
| chat_template: default | |
| data_name: null | |
| data_tag: default | |
| data_type: plaintext | |
| dataloader_type: native | |
| datasets_type: iterable | |
| drop_last: true | |
| image_keys: images | |
| max_seq_len: 4096 | |
| num_workers: 2 | |
| pin_memory: true | |
| prefetch_factor: 2 | |
| text_keys: text | |
| train_path: ${OCI_TEXT_DIR}/000,...,${OCI_TEXT_DIR}/049 # the 50 rendered shards in sorted order (train_path.txt written by prepare_opencodeinstruct.py) | |
| train_size: 1000000000000 | |
| model: | |
| attn_implementation: flash_attention_2 | |
| basic_modules: [] | |
| config_path: fredzzp/open-dcoder-0.5B | |
| decoders: {} | |
| encode_target: false | |
| encoders: {} | |
| input_encoder: encoder | |
| model_path: fredzzp/open-dcoder-0.5B | |
| moe_implementation: null | |
| output_encoder: decoder | |
| tokenizer_path: fredzzp/open-dcoder-0.5B | |
| train: | |
| activation_gpu_limit: 0.0 | |
| auto_resume: true | |
| bsz_warmup_init_mbtoken: 200 | |
| bsz_warmup_ratio: 0.0 | |
| ckpt_manager: dcp | |
| clean_token_wt: 0.0 | |
| context_parallel_size: 1 | |
| data_parallel_mode: ddp | |
| dyn_bsz: true | |
| dyn_bsz_buffer_size: 100 | |
| dyn_bsz_margin: 0 | |
| empty_cache_steps: 1000 | |
| enable_activation_offload: false | |
| enable_forward_prefetch: true | |
| enable_fsdp_offload: false | |
| enable_full_determinism: false | |
| enable_full_shard: false | |
| enable_gradient_checkpointing: false | |
| enable_manual_eager: false | |
| enable_masking: true | |
| enable_mixed_precision: false | |
| enable_profiling: false | |
| enable_reentrant: false | |
| eval_batch_size: 10 | |
| eval_before_train: false | |
| eval_every: 0 | |
| expert_parallel_size: 1 | |
| freeze_layers: lm_head,embed_tokens | |
| global_batch_size: 12 | |
| init_device: cuda | |
| load_checkpoint_path: '' | |
| lr: 0.0003 | |
| lr_decay_ratio: 1.0 | |
| lr_decay_style: cosine | |
| lr_min: 3.0e-06 | |
| lr_start: 0.0 | |
| lr_warmup_ratio: 0.001 | |
| max_grad_norm: 1.0 | |
| max_steps: null | |
| micro_batch_size: 3 | |
| mixture_prob: 0.1 | |
| noise_token_wt: 0.1 | |
| num_train_epochs: 1 | |
| optimizer: adamw | |
| output_dir: ${OUTPUT_DIR}/open-dcoder-0.5B-oci-cdlm-seed42-step2000 # placeholder | |
| pipeline_parallel_size: 1 | |
| profile_end_step: 2 | |
| profile_profile_memory: true | |
| profile_record_shapes: true | |
| profile_start_step: 1 | |
| profile_trace_dir: ./trace | |
| profile_with_stack: true | |
| repr_align_wt: 0.0 | |
| rmpad: false | |
| rmpad_with_pos_ids: true | |
| save_epochs: 1 | |
| save_hf_weights: true | |
| save_steps: 2000 | |
| save_time_interval_minutes: 0 | |
| seed: 42 | |
| tensor_parallel_size: 1 | |
| ulysses_parallel_size: 1 | |
| use_doptim: false | |
| use_wandb: true | |
| wandb_entity: '' | |
| wandb_name: open-dcoder-0.5B-oci-cdlm-seed42-step2000 | |
| wandb_project: Qwen2.5-Coder-0.5B | |
| weight_decay: 0.01 | |