Instructions to use mlx-community/sam-3d-objects-bf16 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use mlx-community/sam-3d-objects-bf16 with MLX:
# Download the model from the Hub pip install huggingface_hub[hf_xet] hf download mlx-community/sam-3d-objects-bf16 --local-dir sam-3d-objects-bf16
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Atomic Chat
Download config.json from mlx-community/sam-3d-objects-bf16: direct link, hf CLI and curl.
- Browser
- Download file 4.23 kB
-
https://huggingface.co/mlx-community/sam-3d-objects-bf16/resolve/main/config.json
- Command line
-
hf download hf://mlx-community/sam-3d-objects-bf16/config.json
-
curl -L -o config.json https://huggingface.co/mlx-community/sam-3d-objects-bf16/resolve/main/config.json
4.23 kB
| { | |
| "model_type": "sam3d_objects", | |
| "dtype": "bfloat16", | |
| "hidden_size": 1024, | |
| "num_heads": 16, | |
| "num_blocks": 24, | |
| "cond_channels": 1024, | |
| "latent_channels": 8, | |
| "latent_resolution": 16, | |
| "resolution": 64, | |
| "io_channels": 128, | |
| "decoder_channels": 768, | |
| "decoder_heads": 12, | |
| "decoder_blocks": 12, | |
| "window_size": 8, | |
| "structure_channels": [ | |
| 512, | |
| 128, | |
| 32 | |
| ], | |
| "structure_res_blocks": 2, | |
| "dino_hidden_size": 1024, | |
| "dino_heads": 16, | |
| "dino_layers": 24, | |
| "dino_image_size": 518, | |
| "image_size": 518, | |
| "point_size": 256, | |
| "point_patch_size": 8, | |
| "point_channels": 512, | |
| "point_heads": 16, | |
| "ss_steps": 25, | |
| "slat_steps": 25, | |
| "ss_guidance": 7.0, | |
| "slat_guidance": 1.0, | |
| "ss_rescale_t": 3.0, | |
| "slat_rescale_t": 1.0, | |
| "downsample_ss_dist": 1, | |
| "slat_mean": [ | |
| 0.12211431, | |
| 0.37204156, | |
| -1.26521907, | |
| -2.05276058, | |
| -3.10432536, | |
| -0.11294304, | |
| -0.85146744, | |
| 0.45506954 | |
| ], | |
| "slat_std": [ | |
| 2.37326008, | |
| 2.13174402, | |
| 2.2413953, | |
| 2.30589401, | |
| 2.1191894, | |
| 1.8969511, | |
| 2.41684989, | |
| 2.08374642 | |
| ], | |
| "depth_model": { | |
| "model_type": "moge3", | |
| "encoder": { | |
| "backbone": "dinov2_vitl14", | |
| "intermediate_layers": [ | |
| 5, | |
| 11, | |
| 17, | |
| 23 | |
| ], | |
| "dim_out": 1024 | |
| }, | |
| "neck": { | |
| "dim_in": [ | |
| 1026, | |
| 2, | |
| 2, | |
| 2, | |
| 2 | |
| ], | |
| "dim_out": null, | |
| "dim_res_blocks": [ | |
| 1024, | |
| 256, | |
| 128, | |
| 64, | |
| 32 | |
| ], | |
| "num_res_blocks": [ | |
| 0, | |
| 2, | |
| 2, | |
| 2, | |
| 0 | |
| ], | |
| "res_block_in_norm": "none", | |
| "res_block_hidden_norm": "none", | |
| "resamplers": [ | |
| "conv_transpose", | |
| "conv_transpose", | |
| "conv_transpose", | |
| "bilinear" | |
| ] | |
| }, | |
| "points_head": { | |
| "dim_in": [ | |
| 1024, | |
| 256, | |
| 128, | |
| 64, | |
| 32 | |
| ], | |
| "dim_out": [ | |
| null, | |
| null, | |
| null, | |
| null, | |
| 3 | |
| ], | |
| "dim_res_blocks": [ | |
| 1024, | |
| 256, | |
| 128, | |
| 64, | |
| 32 | |
| ], | |
| "num_res_blocks": [ | |
| 0, | |
| 1, | |
| 1, | |
| 1, | |
| 0 | |
| ], | |
| "res_block_in_norm": "none", | |
| "res_block_hidden_norm": "none", | |
| "resamplers": [ | |
| "conv_transpose", | |
| "conv_transpose", | |
| "conv_transpose", | |
| "bilinear" | |
| ] | |
| }, | |
| "normal_head": { | |
| "dim_in": [ | |
| 1024, | |
| 256, | |
| 128, | |
| 64, | |
| 32 | |
| ], | |
| "dim_out": [ | |
| null, | |
| null, | |
| null, | |
| null, | |
| 3 | |
| ], | |
| "dim_res_blocks": [ | |
| 1024, | |
| 256, | |
| 128, | |
| 64, | |
| 32 | |
| ], | |
| "num_res_blocks": [ | |
| 0, | |
| 1, | |
| 1, | |
| 1, | |
| 0 | |
| ], | |
| "res_block_in_norm": "none", | |
| "res_block_hidden_norm": "none", | |
| "resamplers": [ | |
| "conv_transpose", | |
| "conv_transpose", | |
| "conv_transpose", | |
| "bilinear" | |
| ] | |
| }, | |
| "mask_head": { | |
| "dim_in": [ | |
| 1024, | |
| 256, | |
| 128, | |
| 64, | |
| 32 | |
| ], | |
| "dim_out": [ | |
| null, | |
| null, | |
| null, | |
| null, | |
| 1 | |
| ], | |
| "dim_res_blocks": [ | |
| 1024, | |
| 256, | |
| 128, | |
| 64, | |
| 32 | |
| ], | |
| "num_res_blocks": [ | |
| 0, | |
| 1, | |
| 1, | |
| 1, | |
| 0 | |
| ], | |
| "res_block_in_norm": "none", | |
| "res_block_hidden_norm": "none", | |
| "resamplers": [ | |
| "conv_transpose", | |
| "conv_transpose", | |
| "conv_transpose", | |
| "bilinear" | |
| ] | |
| }, | |
| "scale_head": { | |
| "dims": [ | |
| 1024, | |
| 1024, | |
| 1024, | |
| 1 | |
| ] | |
| }, | |
| "num_tokens_range": [ | |
| 1200, | |
| 3600 | |
| ], | |
| "refiner": { | |
| "encoder_channels": 1026, | |
| "model_channels": [ | |
| 32, | |
| 64, | |
| 128, | |
| 256, | |
| 512 | |
| ], | |
| "encoder_blocks_per_level": 1, | |
| "decoder_blocks_per_level": 1, | |
| "bottleneck_blocks": 1, | |
| "downsample_factors": [ | |
| 2, | |
| 2, | |
| 2, | |
| 2 | |
| ], | |
| "encoder_downsample": 16, | |
| "in_channels": 3, | |
| "out_channels": 1 | |
| }, | |
| "refiner_depth_resolution": 256 | |
| } | |
| } | |