{ "_attn_implementation_autoset": true, "amp": true, "augmentable_preloader": false, "backbone": { "_attn_implementation_autoset": false, "_name_or_path": "", "add_cross_attention": false, "architectures": [ "LlamaForCausalLM" ], "attention_bias": false, "attention_dropout": 0.0, "bad_words_ids": null, "begin_suppress_tokens": null, "bos_token_id": 1, "chunk_size_feed_forward": 0, "cross_attention_hidden_size": null, "decoder_start_token_id": null, "diversity_penalty": 0.0, "do_sample": false, "early_stopping": false, "encoder_no_repeat_ngram_size": 0, "eos_token_id": 2, "exponential_decay_length_penalty": null, "finetuning_task": null, "forced_bos_token_id": null, "forced_eos_token_id": null, "head_dim": 48, "hidden_act": "silu", "hidden_size": 768, "id2label": { "0": "LABEL_0", "1": "LABEL_1" }, "init_from_lm_ckp": false, "init_name_or_path": null, "initializer_range": 0.02, "intermediate_size": 2048, "is_decoder": false, "is_encoder_decoder": false, "label2id": { "LABEL_0": 0, "LABEL_1": 1 }, "length_penalty": 1.0, "max_length": 20, "max_position_embeddings": 2048, "min_length": 0, "mlp_bias": false, "model_type": "llama", "no_repeat_ngram_size": 0, "num_attention_heads": 16, "num_beam_groups": 1, "num_beams": 1, "num_hidden_layers": 12, "num_key_value_heads": 4, "num_return_sequences": 1, "output_attentions": false, "output_hidden_states": false, "output_scores": false, "pad_token_id": null, "prefix": null, "pretraining_tp": 1, "problem_type": null, "pruned_heads": {}, "remove_invalid_values": false, "repetition_penalty": 1.0, "return_dict": true, "return_dict_in_generate": false, "rms_norm_eps": 1e-05, "rope_scaling": null, "rope_theta": 10000.0, "sep_token_id": null, "suppress_tokens": null, "task_specific_params": null, "temperature": 1.0, "tf_legacy_loss": false, "tie_encoder_decoder": false, "tie_word_embeddings": false, "tokenizer_class": null, "top_k": 50, "top_p": 1.0, "torch_dtype": "float32", "torchscript": false, "typical_p": 1.0, "use_bfloat16": false, "use_cache": true, "vocab_size": 2 }, "cpu": false, "dataset": { "data_format": "b2d", "dataset_path_rel": "B2D-base", "fps": 10, "subsample_ratio": 1.0 }, "debug": false, "deepspeed": { "bf16": { "enabled": true }, "gradient_clipping": 1, "steps_per_print": 5, "train_micro_batch_size_per_gpu": 5, "zero_allow_untested_optimizer": true, "zero_optimization": { "contiguous_gradients": true, "overlap_comm": true, "reduce_bucket_size": 500000000, "stage": 2, "stage3_gather_16bit_weights_on_model_save": false, "stage3_max_live_parameters": 1000000000, "stage3_max_reuse_distance": 1000000000, "stage3_param_persistence_threshold": 1000000, "stage3_prefetch_bucket_size": 500000000, "sub_group_size": 1000000000 } }, "early_stopping_metric": "action_classification_loss", "early_stopping_patience": 5, "force_log": false, "force_save": false, "gpus": 4, "gradient_checkpointing": false, "hyperparams": { "batch_size": 5, "debug": false, "gradient_accumulation_steps": 1, "lr": 3e-05, "max_grad_norm": 1.0, "num_epochs": 40, "optimizer": { "kwargs": { "weight_decay": 0.0001 }, "name": "AdamW" }, "patience": 40, "scheduler": { "kwargs": { "num_training_steps": 40, "num_warmup_steps": 2 }, "name": "linear", "warmup_ratio": 0.05 } }, "light_select_layer": 8, "model_type": "gpt2", "multi_gpu_strategy": "ddp", "nodes": 8, "num_workers": 20, "overfit": 0, "overfit_batches": 1, "preload": true, "preload_in_memory": false, "quantization_offset_map": {}, "quantization_vocab_size_map": {}, "save_every": 1, "seed": 43, "start_saving_epoch": -1, "train_batch_size": 5, "training": { "action": { "path": { "name": "path", "width": 40 }, "waypoints": { "future_horizon": 10, "name": "waypoints", "width": 20 } }, "action_gap": 1, "action_quantizer_path": null, "action_quantizer_path_rel": "bin/quantizers/reward_quantizer.npy", "action_type": "path-waypoints", "bev": { "rgb_front": { "name": "rgb_front" } }, "bev_type": "rgb_front", "bucket_weights": { "total_ratio": 0.6, "type": "preferturns", "weights": [ 1.0, 1.0, 2.0, 2.0, 1.0, 1.0, 1.0, 3.0, 3.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0 ] }, "condition_on_goal": true, "context_length": 1, "create_goal_mask": true, "dataset_caching": { "cache_dir": null, "cache_metadata": true, "cache_slow_attributes": true, "enabled": true }, "drop_last": true, "dynamic_batching": true, "ema_decay": 0.992, "ema_enabled": false, "ema_end_epoch": -1, "ema_every_steps": 1, "ema_start": 0, "forecast_steps": 1, "frame_stride": 5, "future_horizon": 1, "gen_masks_for_action": true, "get_noisy_reduce_fn": "last", "get_weight_reduce_fn": "mean", "goal": { "dual_target_point": { "mean": [ [ 5.1162534, -0.1575937 ], [ 26.005814, -0.09633584 ] ], "name": "dual_target_point", "std": [ [ 0.8992543, 0.9467049 ], [ 15.428605, 6.614513 ] ], "width": 4 } }, "goal_conditioning_type": "local", "goal_continuous": true, "goal_quantizer_path": null, "goal_quantizer_path_rel": "bin/quantizers/reward_quantizer.npy", "goal_type": "dual_target_point", "ignore_past_for_length": true, "include_noisy_in_action": false, "integrate_rewards_to_go": false, "inter_window_stride": 2, "light_rgb_backbone": { "downsample": true, "dropout_attn": 0.0, "ema_enabled": false, "frozen": { "ema_model": false, "model": false, "projector": false }, "init_from_ckpt": { "ckpt_path": null, "ckpt_path_rel": "bin/rgb/llava-v14-micro.pt", "ema_model": true, "enabled": false, "freeze": true, "model": true, "projector": true }, "input_size": 336, "masking_rate": 0.0, "model_config": { "_attn_implementation_autoset": false, "_name_or_path": "", "add_cross_attention": false, "architectures": [ "CLIPVisionModel" ], "attention_dropout": 0.0, "bad_words_ids": null, "begin_suppress_tokens": null, "bos_token_id": null, "chunk_size_feed_forward": 0, "cross_attention_hidden_size": null, "decoder_start_token_id": null, "diversity_penalty": 0.0, "do_sample": false, "early_stopping": false, "encoder_no_repeat_ngram_size": 0, "eos_token_id": null, "exponential_decay_length_penalty": null, "finetuning_task": null, "forced_bos_token_id": null, "forced_eos_token_id": null, "hidden_act": "quick_gelu", "hidden_size": 1024, "id2label": { "0": "LABEL_0", "1": "LABEL_1" }, "image_size": 336, "initializer_factor": 1.0, "initializer_range": 0.02, "intermediate_size": 4096, "is_decoder": false, "is_encoder_decoder": false, "label2id": { "LABEL_0": 0, "LABEL_1": 1 }, "layer_norm_eps": 1e-05, "length_penalty": 1.0, "max_length": 20, "min_length": 0, "model_type": "clip_vision_model", "no_repeat_ngram_size": 0, "num_attention_heads": 16, "num_beam_groups": 1, "num_beams": 1, "num_channels": 3, "num_hidden_layers": 24, "num_return_sequences": 1, "output_attentions": false, "output_hidden_states": false, "output_scores": false, "pad_token_id": null, "patch_size": 14, "prefix": null, "problem_type": null, "projection_dim": 768, "pruned_heads": {}, "remove_invalid_values": false, "repetition_penalty": 1.0, "return_dict": true, "return_dict_in_generate": false, "sep_token_id": null, "suppress_tokens": null, "task_specific_params": null, "temperature": 1.0, "tf_legacy_loss": false, "tie_encoder_decoder": false, "tie_word_embeddings": true, "tokenizer_class": null, "top_k": 50, "top_p": 1.0, "torch_dtype": "float32", "torchscript": false, "transformers_version": "4.46.3", "typical_p": 1.0, "use_bfloat16": false, "vocab_size": 32000 }, "model_path": null, "model_path_rel": "bin/rgb/llava-v1.6-vicuna-visionenc", "outputs": { "patches": true, "whole": false }, "override_kwargs": {}, "processor_config": { "_processor_class": "LlavaNextProcessor", "aspect_ratio_setting": "anyres", "crop_size": { "height": 336, "width": 336 }, "do_center_crop": true, "do_convert_rgb": true, "do_normalize": true, "do_pad": true, "do_rescale": true, "do_resize": true, "image_grid_pinpoints": [ [ 336, 672 ], [ 672, 336 ], [ 672, 672 ], [ 1008, 336 ], [ 336, 1008 ] ], "image_mean": [ 0.48145466, 0.4578275, 0.40821073 ], "image_processor_type": "LlavaNextImageProcessor", "image_std": [ 0.26862954, 0.26130258, 0.27577711 ], "resample": 3, "rescale_factor": 0.00392156862745098, "size": { "shortest_edge": 336 } }, "projection_dim": 768, "select_layer": 8, "try_to_truncate_layers": true }, "loss_params": { "action": { "classification": 1, "reconstruction": 1 }, "bev": {}, "default": { "classification": 0 }, "mask_loss": 0.0625, "path_loss": 1.0, "state_forecast": 0.5, "wp_loss": 1.0 }, "max_instances": -1, "max_token_types": 4, "non_bev_state_type": "speed", "normalize_goal": true, "num_path": 20, "num_waypoints": 10, "object_level": false, "parallel_dataset_init": false, "parallel_dataset_workers": 16, "past_horizon": 0, "pred_latent_ffn_dropout": 0.1, "pred_latent_ffn_hidden": 2048, "pred_latent_layers": 2, "pred_latent_post_mlp": false, "pred_latent_use_metadata": true, "quantized": true, "reward": { "reward": { "name": "reward" } }, "reward_quantizer_path": null, "reward_quantizer_path_rel": "bin/quantizers/reward_quantizer.npy", "reward_type": "reward", "rgb_backbone": { "downsample": true, "dropout_attn": 0.0, "ema_enabled": false, "frozen": { "ema_model": false, "model": false, "projector": false }, "init_from_ckpt": { "ckpt_path": null, "ckpt_path_rel": "bin/rgb/llava-v14-micro.pt", "ema_model": true, "enabled": false, "freeze": true, "model": true, "projector": true }, "input_size": 336, "masking_rate": 0.0, "model_config": { "_attn_implementation_autoset": false, "_name_or_path": "", "add_cross_attention": false, "architectures": [ "CLIPVisionModel" ], "attention_dropout": 0.0, "bad_words_ids": null, "begin_suppress_tokens": null, "bos_token_id": null, "chunk_size_feed_forward": 0, "cross_attention_hidden_size": null, "decoder_start_token_id": null, "diversity_penalty": 0.0, "do_sample": false, "early_stopping": false, "encoder_no_repeat_ngram_size": 0, "eos_token_id": null, "exponential_decay_length_penalty": null, "finetuning_task": null, "forced_bos_token_id": null, "forced_eos_token_id": null, "hidden_act": "quick_gelu", "hidden_size": 1024, "id2label": { "0": "LABEL_0", "1": "LABEL_1" }, "image_size": 336, "initializer_factor": 1.0, "initializer_range": 0.02, "intermediate_size": 4096, "is_decoder": false, "is_encoder_decoder": false, "label2id": { "LABEL_0": 0, "LABEL_1": 1 }, "layer_norm_eps": 1e-05, "length_penalty": 1.0, "max_length": 20, "min_length": 0, "model_type": "clip_vision_model", "no_repeat_ngram_size": 0, "num_attention_heads": 16, "num_beam_groups": 1, "num_beams": 1, "num_channels": 3, "num_hidden_layers": 24, "num_return_sequences": 1, "output_attentions": false, "output_hidden_states": false, "output_scores": false, "pad_token_id": null, "patch_size": 14, "prefix": null, "problem_type": null, "projection_dim": 768, "pruned_heads": {}, "remove_invalid_values": false, "repetition_penalty": 1.0, "return_dict": true, "return_dict_in_generate": false, "sep_token_id": null, "suppress_tokens": null, "task_specific_params": null, "temperature": 1.0, "tf_legacy_loss": false, "tie_encoder_decoder": false, "tie_word_embeddings": true, "tokenizer_class": null, "top_k": 50, "top_p": 1.0, "torch_dtype": "float32", "torchscript": false, "transformers_version": "4.46.3", "typical_p": 1.0, "use_bfloat16": false, "vocab_size": 32000 }, "model_path": null, "model_path_rel": "bin/rgb/llava-v1.6-vicuna-visionenc", "outputs": { "patches": true, "whole": false }, "override_kwargs": {}, "processor_config": { "_processor_class": "LlavaNextProcessor", "aspect_ratio_setting": "anyres", "crop_size": { "height": 336, "width": 336 }, "do_center_crop": true, "do_convert_rgb": true, "do_normalize": true, "do_pad": true, "do_rescale": true, "do_resize": true, "image_grid_pinpoints": [ [ 336, 672 ], [ 672, 336 ], [ 672, 672 ], [ 1008, 336 ], [ 336, 1008 ] ], "image_mean": [ 0.48145466, 0.4578275, 0.40821073 ], "image_processor_type": "LlavaNextImageProcessor", "image_std": [ 0.26862954, 0.26130258, 0.27577711 ], "resample": 3, "rescale_factor": 0.00392156862745098, "size": { "shortest_edge": 336 } }, "projection_dim": 768, "select_layer": -2, "try_to_truncate_layers": true }, "rgb_crop": { "crop_size": 896, "resize": 336, "type": "dualcenter" }, "skip_noisy": true, "split_ratio": 0.8, "splits": { "train": "train", "val": "val" }, "state": { "speed": { "name": "speed" } }, "state_quantizer_path": null, "state_quantizer_path_rel": "bin/quantizers/state_quantizer.npy", "state_type": "rgb_front-speed", "tokenized_state": false, "trim_count": 1, "trim_first_and_last": true, "use_future_ego_waypoints": true, "use_future_vehicle_forecast": true, "use_gt_frc": false, "use_gt_frc_only": false, "use_light_as_query": false, "use_past_horizon_states": false, "use_predicted_latent_with_gap": true, "use_real_latent_ratio": 0.0, "utilize_fast_current_latent": true, "vae_target_supervision": null, "waypoint_gru_head": true, "waypoint_gru_hidden_size": 64, "weighted_sampling": true, "zero_out_frc_branch": false }, "transformers_version": "4.46.3", "use_deepspeed": false, "visualize": true, "visualize_interval": 1, "visualize_start_epoch": -1, "wipe_cache": false }