Video Classification
Transformers
Safetensors
ttvidt
feature-extraction
video
video-representation-learning
self-supervised-learning
motion
temporal-modeling
dinov3
vision-transformer
custom_code
Eval Results (legacy)
Instructions to use KBlueLeaf/TTVidT with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use KBlueLeaf/TTVidT with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("video-classification", model="KBlueLeaf/TTVidT", trust_remote_code=True)# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("KBlueLeaf/TTVidT", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download config.json from KBlueLeaf/TTVidT: direct link, hf CLI and curl.
- Browser
- Download file 4.06 kB
-
https://huggingface.co/KBlueLeaf/TTVidT/resolve/main/config.json
- Command line
-
hf download hf://KBlueLeaf/TTVidT/config.json
-
curl -L -o config.json https://huggingface.co/KBlueLeaf/TTVidT/resolve/main/config.json
4.06 kB
| { | |
| "format": "ttvidt-release-v1", | |
| "model": { | |
| "base_model_name": null, | |
| "motion_encoder_config": { | |
| "temporal_patch_size": 1, | |
| "temporal_patch_overlap": 0, | |
| "motion_layers_period": 1, | |
| "num_motion_tokens": 8, | |
| "motion_ffn_type": "gelu", | |
| "tt_mode": "3d", | |
| "tt_downsample": 4 | |
| }, | |
| "motion_decoder_config": { | |
| "image_dim": 4, | |
| "patch_size": 2, | |
| "decoder_style": "dit", | |
| "num_layers": 12, | |
| "hidden_size": 768, | |
| "intermediate_size": 3072, | |
| "num_heads": 12, | |
| "decode_mode": "diffusion", | |
| "qk_norm": true, | |
| "attn_bias": false, | |
| "use_final_norm": true, | |
| "encoder_hidden_size": 768, | |
| "temporal_patch_size": 1 | |
| }, | |
| "encoder_ae_mode": "video", | |
| "decoder_ae_mode": "image", | |
| "latent_mean": [ | |
| -0.69, | |
| -0.48, | |
| -0.6, | |
| 0.28 | |
| ], | |
| "latent_std": [ | |
| 12.38, | |
| 11.22, | |
| 7.93, | |
| 21.22 | |
| ], | |
| "patch_size": null, | |
| "image_dim": null, | |
| "token_counts": [ | |
| 1 | |
| ], | |
| "gradient_checkpointing": false, | |
| "train_mode": "regression", | |
| "mae_mask_ratio": 0.75, | |
| "ar_shift_range": null, | |
| "ar_shift_sampling": "gamma", | |
| "source_frame_count": 8, | |
| "use_ref": false, | |
| "unfreeze_backbone": true, | |
| "decoder_mask_ratio": 0.0, | |
| "mask_patch_size": 4, | |
| "diffusion_timestep_sampling": "uniform", | |
| "backbone_arch": "ttvidt", | |
| "backbone_config": { | |
| "transformers_version": "5.14.1", | |
| "architectures": [ | |
| "DINOv3ViTModel" | |
| ], | |
| "output_hidden_states": false, | |
| "return_dict": true, | |
| "dtype": "float32", | |
| "chunk_size_feed_forward": 0, | |
| "is_encoder_decoder": false, | |
| "id2label": { | |
| "0": "LABEL_0", | |
| "1": "LABEL_1" | |
| }, | |
| "label2id": { | |
| "LABEL_0": 0, | |
| "LABEL_1": 1 | |
| }, | |
| "problem_type": null, | |
| "patch_size": 16, | |
| "hidden_size": 768, | |
| "intermediate_size": 3072, | |
| "num_hidden_layers": 12, | |
| "num_attention_heads": 12, | |
| "hidden_act": "gelu", | |
| "attention_dropout": 0.0, | |
| "initializer_range": 0.02, | |
| "layer_norm_eps": 1e-05, | |
| "rope_theta": 100.0, | |
| "image_size": 224, | |
| "num_channels": 3, | |
| "query_bias": true, | |
| "key_bias": false, | |
| "value_bias": true, | |
| "proj_bias": true, | |
| "mlp_bias": true, | |
| "layerscale_value": 1.0, | |
| "drop_path_rate": 0.0, | |
| "use_gated_mlp": false, | |
| "num_register_tokens": 4, | |
| "pos_embed_shift": null, | |
| "pos_embed_jitter": null, | |
| "pos_embed_rescale": 2.0, | |
| "apply_layernorm": true, | |
| "reshape_hidden_states": true, | |
| "stage_names": [ | |
| "stem", | |
| "stage1", | |
| "stage2", | |
| "stage3", | |
| "stage4", | |
| "stage5", | |
| "stage6", | |
| "stage7", | |
| "stage8", | |
| "stage9", | |
| "stage10", | |
| "stage11", | |
| "stage12" | |
| ], | |
| "_name_or_path": "facebook/dinov3-vitb16-pretrain-lvd1689m", | |
| "model_type": "dinov3_vit", | |
| "temporal_patch_size": 1, | |
| "temporal_patch_overlap": 0, | |
| "motion_layers_period": 1, | |
| "num_motion_tokens": 8, | |
| "motion_ffn_type": "gelu", | |
| "tt_mode": "3d", | |
| "tt_downsample": 4, | |
| "output_attentions": false, | |
| "out_features": [ | |
| "stage12" | |
| ], | |
| "out_indices": [ | |
| 12 | |
| ] | |
| }, | |
| "decoder_lr_multiplier": 1.0, | |
| "fused_adam": false, | |
| "ema_foreach": false, | |
| "nan_guard_interval": 1, | |
| "name": "ttvidt-tt3d-diffcomp", | |
| "log_interval": 2500, | |
| "learning_rate": 0.0005, | |
| "base_dim": 256, | |
| "decoder_base_dim": null, | |
| "weight_decay": 0.01, | |
| "betas": [ | |
| 0.9, | |
| 0.98 | |
| ], | |
| "scheduler_config": { | |
| "lr": { | |
| "mode": "cosine", | |
| "end": 436705, | |
| "min_value": 0.01, | |
| "warmup": 10000 | |
| } | |
| } | |
| }, | |
| "encoder_only": true, | |
| "name": "ttvidt-tt3d", | |
| "model_type": "ttvidt", | |
| "architectures": [ | |
| "TTVidTModel" | |
| ], | |
| "auto_map": { | |
| "AutoConfig": "modeling_ttvidt.TTVidTConfig", | |
| "AutoModel": "modeling_ttvidt.TTVidTModel" | |
| } | |
| } |