experiment: debug: false tasks: - training num_nodes: 1 training: precision: 16-mixed compile: false lr: 8.0e-06 batch_size: 2 max_epochs: 26 max_steps: -1 max_time: null data: num_workers: 11 shuffle: false optim: accumulate_grad_batches: 1 gradient_clip_val: 1.0 checkpointing: every_n_train_steps: null every_n_epochs: 1 train_time_interval: null enable_version_counter: false save_top_k: -1 validation: precision: 16-mixed compile: false batch_size: 8 val_every_n_step: 1.0 val_every_n_epoch: 1 num_sanity_val_steps: 1 limit_batch: 1.0 data: num_workers: 0 shuffle: false inference_mode: true test: precision: 16-mixed compile: false batch_size: 4 limit_batch: 1.0 data: num_workers: 0 shuffle: false inference_mode: true find_unused_parameters: false reload_dataloaders_every_n_epochs: 1 ema: enable: true decay: 0.9999 validate_original_weights: false dataset: debug: false save_dir: data/real-estate-10k data_mean: - - - 0.577 - - - 0.517 - - - 0.461 data_std: - - - 0.249 - - - 0.249 - - - 0.268 latent: enable: false type: pre_sample suffix: null downsampling_factor: - 1 - 8 num_channels: 4 resolution: 256 observation_shape: - 3 - 256 - 256 max_frames: 16 n_frames: 400 context_length: 7 frame_skip: 1 filter_min_len: null external_cond_dim: 16 external_cond_stack: false external_cond_processing: null preload: false subdataset_size: 2400000 num_eval_videos: 100 maximize_training_data: true augmentation: frame_skip_increase: 1 horizontal_flip_prob: 0.5 reverse_prob: 0.5 back_and_forth_prob: 0.1 algorithm: debug: false lr: 8.0e-06 backbone: name: u_vit3d_pose channels: - 128 - 256 - 576 - 1152 emb_channels: 1024 patch_size: 2 block_types: - ResBlock - ResBlock - TransformerBlock - TransformerBlock block_dropouts: - 0.0 - 0.0 - 0.1 - 0.1 num_updown_blocks: - 3 - 3 - 6 num_mid_blocks: 20 num_heads: 9 pos_emb_type: rope use_checkpointing: - false - false - false - false conditioning: dim: 180 external_cond_dropout: 0.1 use_fourier_noise_embedding: true x_shape: - 3 - 256 - 256 max_frames: 16 n_frames: 400 frame_skip: 1 context_frames: 7 latent: enable: false type: pre_sample suffix: null downsampling_factor: - 1 - 8 num_channels: 4 data_mean: - - - 0.577 - - - 0.517 - - - 0.461 data_std: - - - 0.249 - - - 0.249 - - - 0.268 external_cond_dim: 16 external_cond_stack: false external_cond_processing: null compile: false weight_decay: 0.01 optimizer_beta: - 0.9 - 0.99 lr_scheduler: name: constant_with_warmup num_warmup_steps: 10000 num_training_steps: 550000 noise_level: random_independent uniform_future: enabled: false fixed_context: enabled: false indices: null dropout: 0 variable_context: enabled: false prob: 0 dropout: 0 chunk_size: -1 scheduling_matrix: full_sequence replacement: noisy_scale diffusion: is_continuous: true timesteps: 1000 beta_schedule: cosine_simple_diffusion schedule_fn_kwargs: shift: 1.0 shifted: 0.125 interpolated: false use_causal_mask: false clip_noise: 20.0 objective: pred_v loss_weighting: strategy: sigmoid snr_clip: 5.0 cum_snr_decay: 0.9 sigmoid_bias: -1.0 sampling_timesteps: 50 ddim_sampling_eta: 0.0 reconstruction_guidance: 0.0 training_schedule: name: cosine shift: 0.125 precond_scale: 0.125 vae: pretrained_path: null pretrained_kwargs: {} use_fp16: true batch_size: 2 checkpoint: reset_optimizer: false strict: true tasks: prediction: enabled: true history_guidance: name: conditional keyframe_density: null sliding_context_len: null interpolation: enabled: false history_guidance: name: conditional max_batch_size: null logging: deterministic: 0 loss_freq: 100 grad_norm_freq: 100 max_num_videos: 256 n_metrics_frames: null metrics: [] metrics_batch_size: 16 sanity_generation: false raw_dir: null camera_pose_conditioning: normalize_by: first bound: null type: ray_encoding alignment: alignment_coeff: 0 encoder_type: vggt apply_unnormalize_recon: true alignment_context_length: 16 encoder_info: - 24 - 512 - 512 mid_channels: 128 latents_info: null _name: dfot_geometry_forcing debug: false wandb: entity: replace_with_your_account project: dfot mode: offline resume: null load: null