# pot14 自采数据微调配置(单活动臂 7 维 + 三路相机) # 基于 robotwin 骨架,数据来自 scripts/convert_pot14_to_lerobot.py model: model_path: ../models/pretrained/lingbot-vla-4b tokenizer_path: ../models/pretrained/Qwen2.5-VL-3B-Instruct data: datasets_type: vla data_name: pot14 train_path: ../data/processed/pot14_right_arm joints: - arm.position: 6 - effector.position: 1 cameras: - camera_top - camera_wrist_left - camera_wrist_right num_workers: 4 pin_memory: true prefetch_factor: 2 video_backend: torchcodec norm_type: meanstd norm_stats_file: assets/norm_stats/pot14_right_arm.json train: output_dir: "output/pot14/" data_parallel_mode: fsdp2 enable_full_shard: false module_fsdp_enable: true use_compile: true rmpad: false rmpad_with_pos_ids: false ulysses_parallel_size: 1 freeze_vision_encoder: false tokenizer_max_length: 72 max_action_dim: 75 max_state_dim: 75 lr: 5.0e-5 lr_decay_style: constant micro_batch_size: 2 gradient_accumulation_steps: 16 max_steps: 1024 ckpt_manager: dcp save_steps: 0 save_epochs: 0 max_checkpoints_to_keep: 1 save_optimizer: false enable_fp32: true enable_resume: false