File size: 4,957 Bytes
59630ba
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
experiment:
  debug: false
  tasks:
  - training
  num_nodes: 1
  training:
    precision: 16-mixed
    compile: false
    lr: 8.0e-06
    batch_size: 2
    max_epochs: 26
    max_steps: -1
    max_time: null
    data:
      num_workers: 11
      shuffle: false
    optim:
      accumulate_grad_batches: 1
      gradient_clip_val: 1.0
    checkpointing:
      every_n_train_steps: null
      every_n_epochs: 1
      train_time_interval: null
      enable_version_counter: false
      save_top_k: -1
  validation:
    precision: 16-mixed
    compile: false
    batch_size: 8
    val_every_n_step: 1.0
    val_every_n_epoch: 1
    num_sanity_val_steps: 1
    limit_batch: 1.0
    data:
      num_workers: 0
      shuffle: false
    inference_mode: true
  test:
    precision: 16-mixed
    compile: false
    batch_size: 4
    limit_batch: 1.0
    data:
      num_workers: 0
      shuffle: false
    inference_mode: true
  find_unused_parameters: false
  reload_dataloaders_every_n_epochs: 1
  ema:
    enable: true
    decay: 0.9999
    validate_original_weights: false
dataset:
  debug: false
  save_dir: data/real-estate-10k
  data_mean:
  - - - 0.577
  - - - 0.517
  - - - 0.461
  data_std:
  - - - 0.249
  - - - 0.249
  - - - 0.268
  latent:
    enable: false
    type: pre_sample
    suffix: null
    downsampling_factor:
    - 1
    - 8
    num_channels: 4
  resolution: 256
  observation_shape:
  - 3
  - 256
  - 256
  max_frames: 16
  n_frames: 400
  context_length: 7
  frame_skip: 1
  filter_min_len: null
  external_cond_dim: 16
  external_cond_stack: false
  external_cond_processing: null
  preload: false
  subdataset_size: 2400000
  num_eval_videos: 100
  maximize_training_data: true
  augmentation:
    frame_skip_increase: 1
    horizontal_flip_prob: 0.5
    reverse_prob: 0.5
    back_and_forth_prob: 0.1
algorithm:
  debug: false
  lr: 8.0e-06
  backbone:
    name: u_vit3d_pose
    channels:
    - 128
    - 256
    - 576
    - 1152
    emb_channels: 1024
    patch_size: 2
    block_types:
    - ResBlock
    - ResBlock
    - TransformerBlock
    - TransformerBlock
    block_dropouts:
    - 0.0
    - 0.0
    - 0.1
    - 0.1
    num_updown_blocks:
    - 3
    - 3
    - 6
    num_mid_blocks: 20
    num_heads: 9
    pos_emb_type: rope
    use_checkpointing:
    - false
    - false
    - false
    - false
    conditioning:
      dim: 180
    external_cond_dropout: 0.1
    use_fourier_noise_embedding: true
  x_shape:
  - 3
  - 256
  - 256
  max_frames: 16
  n_frames: 400
  frame_skip: 1
  context_frames: 7
  latent:
    enable: false
    type: pre_sample
    suffix: null
    downsampling_factor:
    - 1
    - 8
    num_channels: 4
  data_mean:
  - - - 0.577
  - - - 0.517
  - - - 0.461
  data_std:
  - - - 0.249
  - - - 0.249
  - - - 0.268
  external_cond_dim: 16
  external_cond_stack: false
  external_cond_processing: null
  compile: false
  weight_decay: 0.01
  optimizer_beta:
  - 0.9
  - 0.99
  lr_scheduler:
    name: constant_with_warmup
    num_warmup_steps: 10000
    num_training_steps: 550000
  noise_level: random_independent
  uniform_future:
    enabled: false
  fixed_context:
    enabled: false
    indices: null
    dropout: 0
  variable_context:
    enabled: false
    prob: 0
    dropout: 0
  chunk_size: -1
  scheduling_matrix: full_sequence
  replacement: noisy_scale
  diffusion:
    is_continuous: true
    timesteps: 1000
    beta_schedule: cosine_simple_diffusion
    schedule_fn_kwargs:
      shift: 1.0
      shifted: 0.125
      interpolated: false
    use_causal_mask: false
    clip_noise: 20.0
    objective: pred_v
    loss_weighting:
      strategy: sigmoid
      snr_clip: 5.0
      cum_snr_decay: 0.9
      sigmoid_bias: -1.0
    sampling_timesteps: 50
    ddim_sampling_eta: 0.0
    reconstruction_guidance: 0.0
    training_schedule:
      name: cosine
      shift: 0.125
    precond_scale: 0.125
  vae:
    pretrained_path: null
    pretrained_kwargs: {}
    use_fp16: true
    batch_size: 2
  checkpoint:
    reset_optimizer: false
    strict: true
  tasks:
    prediction:
      enabled: true
      history_guidance:
        name: conditional
      keyframe_density: null
      sliding_context_len: null
    interpolation:
      enabled: false
      history_guidance:
        name: conditional
      max_batch_size: null
  logging:
    deterministic: 0
    loss_freq: 100
    grad_norm_freq: 100
    max_num_videos: 256
    n_metrics_frames: null
    metrics: []
    metrics_batch_size: 16
    sanity_generation: false
    raw_dir: null
  camera_pose_conditioning:
    normalize_by: first
    bound: null
    type: ray_encoding
  alignment:
    alignment_coeff: 0
    encoder_type: vggt
    apply_unnormalize_recon: true
    alignment_context_length: 16
    encoder_info:
    - 24
    - 512
    - 512
    mid_channels: 128
    latents_info: null
  _name: dfot_geometry_forcing
debug: false
wandb:
  entity: replace_with_your_account
  project: dfot
  mode: offline
resume: null
load: null