{ "format": "ttvidt-release-v1", "model": { "base_model_name": null, "motion_encoder_config": { "temporal_patch_size": 1, "temporal_patch_overlap": 0, "motion_layers_period": 1, "num_motion_tokens": 8, "motion_ffn_type": "gelu", "tt_mode": "3d", "tt_downsample": 4 }, "motion_decoder_config": { "image_dim": 4, "patch_size": 2, "decoder_style": "dit", "num_layers": 12, "hidden_size": 768, "intermediate_size": 3072, "num_heads": 12, "decode_mode": "diffusion", "qk_norm": true, "attn_bias": false, "use_final_norm": true, "encoder_hidden_size": 768, "temporal_patch_size": 1 }, "encoder_ae_mode": "video", "decoder_ae_mode": "image", "latent_mean": [ -0.69, -0.48, -0.6, 0.28 ], "latent_std": [ 12.38, 11.22, 7.93, 21.22 ], "patch_size": null, "image_dim": null, "token_counts": [ 1 ], "gradient_checkpointing": false, "train_mode": "regression", "mae_mask_ratio": 0.75, "ar_shift_range": null, "ar_shift_sampling": "gamma", "source_frame_count": 8, "use_ref": false, "unfreeze_backbone": true, "decoder_mask_ratio": 0.0, "mask_patch_size": 4, "diffusion_timestep_sampling": "uniform", "backbone_arch": "ttvidt", "backbone_config": { "transformers_version": "5.14.1", "architectures": [ "DINOv3ViTModel" ], "output_hidden_states": false, "return_dict": true, "dtype": "float32", "chunk_size_feed_forward": 0, "is_encoder_decoder": false, "id2label": { "0": "LABEL_0", "1": "LABEL_1" }, "label2id": { "LABEL_0": 0, "LABEL_1": 1 }, "problem_type": null, "patch_size": 16, "hidden_size": 768, "intermediate_size": 3072, "num_hidden_layers": 12, "num_attention_heads": 12, "hidden_act": "gelu", "attention_dropout": 0.0, "initializer_range": 0.02, "layer_norm_eps": 1e-05, "rope_theta": 100.0, "image_size": 224, "num_channels": 3, "query_bias": true, "key_bias": false, "value_bias": true, "proj_bias": true, "mlp_bias": true, "layerscale_value": 1.0, "drop_path_rate": 0.0, "use_gated_mlp": false, "num_register_tokens": 4, "pos_embed_shift": null, "pos_embed_jitter": null, "pos_embed_rescale": 2.0, "apply_layernorm": true, "reshape_hidden_states": true, "stage_names": [ "stem", "stage1", "stage2", "stage3", "stage4", "stage5", "stage6", "stage7", "stage8", "stage9", "stage10", "stage11", "stage12" ], "_name_or_path": "facebook/dinov3-vitb16-pretrain-lvd1689m", "model_type": "dinov3_vit", "temporal_patch_size": 1, "temporal_patch_overlap": 0, "motion_layers_period": 1, "num_motion_tokens": 8, "motion_ffn_type": "gelu", "tt_mode": "3d", "tt_downsample": 4, "output_attentions": false, "out_features": [ "stage12" ], "out_indices": [ 12 ] }, "decoder_lr_multiplier": 1.0, "fused_adam": false, "ema_foreach": false, "nan_guard_interval": 1, "name": "ttvidt-tt3d-diffcomp", "log_interval": 2500, "learning_rate": 0.0005, "base_dim": 256, "decoder_base_dim": null, "weight_decay": 0.01, "betas": [ 0.9, 0.98 ], "scheduler_config": { "lr": { "mode": "cosine", "end": 436705, "min_value": 0.01, "warmup": 10000 } } }, "encoder_only": true, "name": "ttvidt-tt3d", "model_type": "ttvidt", "architectures": [ "TTVidTModel" ], "auto_map": { "AutoConfig": "modeling_ttvidt.TTVidTConfig", "AutoModel": "modeling_ttvidt.TTVidTModel" } }