TTVidT / config.json
KBlueLeaf's picture
Model card: metadata, figures, links
c9eef3f verified
Raw History Blame Contribute Delete
4.06 kB
{
"format": "ttvidt-release-v1",
"model": {
"base_model_name": null,
"motion_encoder_config": {
"temporal_patch_size": 1,
"temporal_patch_overlap": 0,
"motion_layers_period": 1,
"num_motion_tokens": 8,
"motion_ffn_type": "gelu",
"tt_mode": "3d",
"tt_downsample": 4
},
"motion_decoder_config": {
"image_dim": 4,
"patch_size": 2,
"decoder_style": "dit",
"num_layers": 12,
"hidden_size": 768,
"intermediate_size": 3072,
"num_heads": 12,
"decode_mode": "diffusion",
"qk_norm": true,
"attn_bias": false,
"use_final_norm": true,
"encoder_hidden_size": 768,
"temporal_patch_size": 1
},
"encoder_ae_mode": "video",
"decoder_ae_mode": "image",
"latent_mean": [
-0.69,
-0.48,
-0.6,
0.28
],
"latent_std": [
12.38,
11.22,
7.93,
21.22
],
"patch_size": null,
"image_dim": null,
"token_counts": [
1
],
"gradient_checkpointing": false,
"train_mode": "regression",
"mae_mask_ratio": 0.75,
"ar_shift_range": null,
"ar_shift_sampling": "gamma",
"source_frame_count": 8,
"use_ref": false,
"unfreeze_backbone": true,
"decoder_mask_ratio": 0.0,
"mask_patch_size": 4,
"diffusion_timestep_sampling": "uniform",
"backbone_arch": "ttvidt",
"backbone_config": {
"transformers_version": "5.14.1",
"architectures": [
"DINOv3ViTModel"
],
"output_hidden_states": false,
"return_dict": true,
"dtype": "float32",
"chunk_size_feed_forward": 0,
"is_encoder_decoder": false,
"id2label": {
"0": "LABEL_0",
"1": "LABEL_1"
},
"label2id": {
"LABEL_0": 0,
"LABEL_1": 1
},
"problem_type": null,
"patch_size": 16,
"hidden_size": 768,
"intermediate_size": 3072,
"num_hidden_layers": 12,
"num_attention_heads": 12,
"hidden_act": "gelu",
"attention_dropout": 0.0,
"initializer_range": 0.02,
"layer_norm_eps": 1e-05,
"rope_theta": 100.0,
"image_size": 224,
"num_channels": 3,
"query_bias": true,
"key_bias": false,
"value_bias": true,
"proj_bias": true,
"mlp_bias": true,
"layerscale_value": 1.0,
"drop_path_rate": 0.0,
"use_gated_mlp": false,
"num_register_tokens": 4,
"pos_embed_shift": null,
"pos_embed_jitter": null,
"pos_embed_rescale": 2.0,
"apply_layernorm": true,
"reshape_hidden_states": true,
"stage_names": [
"stem",
"stage1",
"stage2",
"stage3",
"stage4",
"stage5",
"stage6",
"stage7",
"stage8",
"stage9",
"stage10",
"stage11",
"stage12"
],
"_name_or_path": "facebook/dinov3-vitb16-pretrain-lvd1689m",
"model_type": "dinov3_vit",
"temporal_patch_size": 1,
"temporal_patch_overlap": 0,
"motion_layers_period": 1,
"num_motion_tokens": 8,
"motion_ffn_type": "gelu",
"tt_mode": "3d",
"tt_downsample": 4,
"output_attentions": false,
"out_features": [
"stage12"
],
"out_indices": [
12
]
},
"decoder_lr_multiplier": 1.0,
"fused_adam": false,
"ema_foreach": false,
"nan_guard_interval": 1,
"name": "ttvidt-tt3d-diffcomp",
"log_interval": 2500,
"learning_rate": 0.0005,
"base_dim": 256,
"decoder_base_dim": null,
"weight_decay": 0.01,
"betas": [
0.9,
0.98
],
"scheduler_config": {
"lr": {
"mode": "cosine",
"end": 436705,
"min_value": 0.01,
"warmup": 10000
}
}
},
"encoder_only": true,
"name": "ttvidt-tt3d",
"model_type": "ttvidt",
"architectures": [
"TTVidTModel"
],
"auto_map": {
"AutoConfig": "modeling_ttvidt.TTVidTConfig",
"AutoModel": "modeling_ttvidt.TTVidTModel"
}
}