| { |
| "_comment_action_decoder": "========== Action Decoder (ResNet-based Direct Reconstruction) ==========", |
| "_comment_action_encoder": "========== Action Encoder ==========", |
| "_comment_backbone": "========== Backbone Configuration (Vision Model Only) ==========", |
| "_comment_basic": "========== Basic Parameters ==========", |
| "_comment_debug": "========== Debug Configuration ==========", |
| "_comment_fusion": "========== Visual-Action Fusion ==========", |
| "_comment_lpips": "DINO mode: LPIPS disabled, using cosine similarity for vision reconstruction", |
| "_comment_modality": "========== Modality Presence Configuration ==========", |
| "_comment_trainable": "========== Default Trainable Flags ==========", |
| "_comment_training": "========== Training Configuration (DINO Mode) ==========", |
| "_comment_vision_branch": "========== Vision Branch Encoder ==========", |
| "_comment_vision_decoder": "========== Vision Decoder (DINO Hidden State Mode) ==========", |
| "_comment_visualization": "========== Visualization Configuration (Disabled for DINO) ==========", |
| "_comment_vq": "========== Vector Quantizer ==========", |
| "action_decoder_cfg": { |
| "_comment_decoder_type": "Decoder type: 'resnet' (direct reconstruction) or 'flow_matching' (diffusion)", |
| "_comment_embodiment": "Embodiment configuration", |
| "_comment_m_former": "M-Former configuration (symmetric to encoder)", |
| "_comment_resnet": "ResNet decoder configuration (symmetric to encoder)", |
| "action_dim": 16, |
| "action_horizon": 16, |
| "conv_kernel_size": 5, |
| "decoder_type": "resnet", |
| "dropout": 0.0, |
| "hidden_size": 1024, |
| "m_former_cfg": { |
| "attention_probs_dropout_prob": 0.0, |
| "hidden_dropout_prob": 0.0, |
| "hidden_size": 1024, |
| "input_hidden_size": 1024, |
| "intermediate_size": 3072, |
| "num_attention_heads": 16, |
| "num_hidden_layers": 8, |
| "query_num": 8 |
| }, |
| "max_num_embodiments": 30, |
| "num_conv_layers": 3, |
| "query_num": 8, |
| "upsample_stride": 2, |
| "use_dilation": false |
| }, |
| "action_dim": 16, |
| "action_encoder_cfg": { |
| "action_dim": 16, |
| "action_horizon": 16, |
| "conv_encoder_type": "resnet", |
| "conv_kernel_size": 5, |
| "conv_stride": 2, |
| "downsample_target_len": null, |
| "dropout": 0.0, |
| "hidden_size": 1024, |
| "m_former_cfg": { |
| "attention_probs_dropout_prob": 0.0, |
| "hidden_dropout_prob": 0.0, |
| "hidden_size": 1024, |
| "input_hidden_size": 1024, |
| "intermediate_size": 3072, |
| "num_attention_heads": 16, |
| "num_hidden_layers": 4, |
| "query_num": 8 |
| }, |
| "max_num_embodiments": 30, |
| "num_conv_layers": 3, |
| "query_num": 8, |
| "state_dim": 16, |
| "use_dilation": true |
| }, |
| "action_horizon": 16, |
| "architectures": [ |
| "GR00T_Tokenizer" |
| ], |
| "attn_implementation": null, |
| "backbone_cfg": { |
| "_comment_eagle_path": "HF hub id or local path for Qwen2.5-VL (passed to AutoModel/AutoProcessor).", |
| "_comment_layer_index": "Layer index: -1=last layer, -2=second-to-last, 0-N=specific layer", |
| "dinov2_layer_index": -2, |
| "dinov2_path": "/inspire/qb-ilm/project/qproject-fundationmodel/public/zzt/models/dinov2-large", |
| "eagle_path": "/inspire/qb-ilm/project/qproject-fundationmodel/public/zzt/models/Qwen/Qwen2.5-VL-3B-Instruct", |
| "load_bf16": false, |
| "project_to_dim": null, |
| "reproject_vision": false, |
| "select_layer": 12, |
| "tune_bridge_embedding": false, |
| "tune_llm": false, |
| "tune_visual": false, |
| "use_flash_attention": true, |
| "use_torch_compile": false, |
| "vision_model_type": "dinov2" |
| }, |
| "bridge_projector_mid_dim": 1024, |
| "compute_dtype": "bfloat16", |
| "default_pa": 1, |
| "default_pv": 1, |
| "enable_action_visualization": false, |
| "enable_dimension_debug": false, |
| "enable_visualization": false, |
| "fusion_cfg": { |
| "dropout": 0.0, |
| "hidden_size": 1024, |
| "num_attention_heads": 16, |
| "num_layers": 4, |
| "query_num": 8 |
| }, |
| "hidden_size": 1024, |
| "ignore_lang_prefix": false, |
| "model_type": "gr00t_tokenizer", |
| "query_num": 8, |
| "reconstruction_loss_weights": { |
| "action": 1.0, |
| "lpips": 0.0, |
| "vision": 1.0, |
| "vq_commitment": 1.0 |
| }, |
| "state_dim": 16, |
| "torch_dtype": "float32", |
| "transformers_version": "4.52.0", |
| "tune_action_decoder_diffusion": true, |
| "tune_action_decoder_projector": true, |
| "tune_action_encoder": true, |
| "tune_bridge_projector": true, |
| "tune_fusion": true, |
| "tune_image_type_embedding": true, |
| "tune_vision_decoder": true, |
| "tune_vision_m_former": true, |
| "tune_vision_model": false, |
| "tune_vq": true, |
| "unified_embodiment_id": null, |
| "use_image_type_embedding": true, |
| "use_lpips_loss": false, |
| "use_multi_scenario_training": true, |
| "video_delta_indices": [ |
| 0 |
| ], |
| "vision_branch_cfg": { |
| "hidden_size": 1024, |
| "m_former_cfg": { |
| "attention_probs_dropout_prob": 0.0, |
| "hidden_dropout_prob": 0.0, |
| "hidden_size": 1024, |
| "input_hidden_size": 1024, |
| "intermediate_size": 3072, |
| "max_seq_len": 1024, |
| "num_attention_heads": 16, |
| "num_hidden_layers": 4, |
| "query_num": 8 |
| }, |
| "query_num": 8 |
| }, |
| "vision_decoder_cfg": { |
| "_comment_dino": "DINO mode: input/output are hidden states, not images", |
| "attention_probs_dropout_prob": 0.0, |
| "encoder_stride": 14, |
| "hidden_dropout_prob": 0.0, |
| "hidden_size": 1024, |
| "image_size": 224, |
| "input_hidden_size": 1024, |
| "intermediate_size": 3072, |
| "is_io_hidden_states": true, |
| "latent_motion_dim": 1024, |
| "num_attention_heads": 16, |
| "num_channels": 3, |
| "num_hidden_layers": 12, |
| "num_patches": 256, |
| "patch_size": 14, |
| "query_num": 8 |
| }, |
| "visualization_freq": 1000, |
| "visualization_num_samples": 4, |
| "visualization_save_dir": "visualizations", |
| "vq_cfg": { |
| "beta": 0.25, |
| "e_dim": 32, |
| "n_e": 128, |
| "num_stages": 2, |
| "sane_index_shape": false, |
| "stages": [ |
| { |
| "beta": 0.5, |
| "e_dim": 32, |
| "n_e": 128 |
| }, |
| { |
| "beta": 0.1, |
| "e_dim": 32, |
| "n_e": 128 |
| } |
| ] |
| } |
| } |
|
|