Image Segmentation
Transformers
Safetensors
English
falcon_x
feature-extraction
falcon-x
vision-language
custom_code
Instructions to use JonathanJMK/FALCON with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use JonathanJMK/FALCON with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-segmentation", model="JonathanJMK/FALCON", trust_remote_code=True)# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("JonathanJMK/FALCON", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download config.json from JonathanJMK/FALCON: direct link, hf CLI and curl.
- Browser
- Download file 17.8 kB
-
https://huggingface.co/JonathanJMK/FALCON/resolve/main/config.json
- Command line
-
hf download hf://JonathanJMK/FALCON/config.json
-
curl -L -o config.json https://huggingface.co/JonathanJMK/FALCON/resolve/main/config.json
17.8 kB
| { | |
| "architectures": [ | |
| "FalconModel" | |
| ], | |
| "auto_map": { | |
| "AutoConfig": "configuration_falcon.FalconConfig", | |
| "AutoModel": "modeling_falcon.FalconModel" | |
| }, | |
| "checkpoint_metadata": { | |
| "checkpoint_format": "falcon-checkpoint-v1", | |
| "config": { | |
| "detector": { | |
| "mask_threshold": 0.5, | |
| "max_regions": 100, | |
| "nms_threshold": 0.6, | |
| "resolution": 768, | |
| "score_threshold": 0.15, | |
| "variant": "seg-2xlarge" | |
| }, | |
| "experiment": { | |
| "backbone_layout": "independent", | |
| "name": "falcon-x", | |
| "protocol": "dataset-v1", | |
| "ssa_ablation": "none", | |
| "world_size": null | |
| }, | |
| "model": { | |
| "component_order": [ | |
| "detonator", | |
| "explosive", | |
| "battery" | |
| ], | |
| "image_size": 448, | |
| "language_model": "lmsys/vicuna-7b-v1.5", | |
| "link_order": [ | |
| [ | |
| "battery", | |
| "detonator" | |
| ], | |
| [ | |
| "battery", | |
| "explosive" | |
| ], | |
| [ | |
| "detonator", | |
| "explosive" | |
| ] | |
| ], | |
| "lora_alpha": 32, | |
| "lora_dropout": 0.05, | |
| "lora_rank": 16, | |
| "max_regions": 100, | |
| "patch_size": 14, | |
| "region_dim": 1024, | |
| "roi_size": 4, | |
| "vision_model": "facebook/dinov2-large" | |
| }, | |
| "stages": { | |
| "stage1": { | |
| "batch_size": 1, | |
| "encoder_learning_rate": 0.00015, | |
| "epochs": 12, | |
| "expanded_scales": false, | |
| "gradient_accumulation": 16, | |
| "learning_rate": 0.0001, | |
| "lr_scheduler": "cosine", | |
| "multi_scale": false, | |
| "precision": "backend_mixed", | |
| "tf32": true, | |
| "warmup_epochs": 0.0, | |
| "weight_decay": 0.0001 | |
| }, | |
| "stage2": { | |
| "batch_size": 4, | |
| "epoch_size": null, | |
| "epochs": 1, | |
| "gradient_accumulation": 4, | |
| "learning_rate": 0.0001, | |
| "link_loss_weight": 0.5, | |
| "max_text_tokens": 256, | |
| "precision": "bf16", | |
| "presence_loss_weight": 0.5, | |
| "risk_loss_weight": 1.0, | |
| "sampling": "all", | |
| "seed": 42, | |
| "text_overflow_policy": "error", | |
| "tf32": true, | |
| "warmup_ratio": 0.03, | |
| "weight_decay": 0.0 | |
| }, | |
| "stage3": { | |
| "batch_size": 1, | |
| "epoch_size": null, | |
| "epochs": 1, | |
| "gradient_accumulation": 16, | |
| "learning_rate": 1e-05, | |
| "link_loss_weight": 0.5, | |
| "max_text_tokens": 256, | |
| "precision": "bf16", | |
| "presence_loss_weight": 0.5, | |
| "risk_loss_weight": 1.0, | |
| "sampling": "all", | |
| "seed": 42, | |
| "text_overflow_policy": "error", | |
| "tf32": true, | |
| "warmup_ratio": 0.03, | |
| "weight_decay": 0.0 | |
| } | |
| }, | |
| "training": { | |
| "batch_size": 1, | |
| "epoch_size": null, | |
| "epochs": 1, | |
| "gradient_accumulation": 16, | |
| "learning_rate": 0.0001, | |
| "link_loss_weight": 0.5, | |
| "max_text_tokens": 256, | |
| "precision": "bf16", | |
| "presence_loss_weight": 0.5, | |
| "risk_loss_weight": 1.0, | |
| "sampling": "all", | |
| "seed": 42, | |
| "text_overflow_policy": "error", | |
| "tf32": true, | |
| "warmup_ratio": 0.03, | |
| "weight_decay": 0.0 | |
| } | |
| }, | |
| "dataset_categories": [ | |
| { | |
| "id": 1, | |
| "name": "detonator" | |
| }, | |
| { | |
| "id": 2, | |
| "name": "explosive" | |
| }, | |
| { | |
| "id": 3, | |
| "name": "battery" | |
| } | |
| ], | |
| "diagnostic_training_reasons": [], | |
| "grounding_coverage": { | |
| "full_match": 177693, | |
| "ground_truth_instances": 238792, | |
| "language_supervision_masked": 1405, | |
| "matched_instances": 237385, | |
| "partial_match": 496, | |
| "true_negative": 6, | |
| "unmatched_positive": 909 | |
| }, | |
| "grounding_policy": { | |
| "iou_threshold": 0.5, | |
| "unmatched": "mask_language_loss", | |
| "version": "falcon-grounding-match/cardinality-iou-v1" | |
| }, | |
| "observed_supervision_coverage": { | |
| "images": 34826, | |
| "links": [ | |
| 0, | |
| 0, | |
| 0 | |
| ], | |
| "presence": [ | |
| 34826, | |
| 34826, | |
| 34826 | |
| ], | |
| "risk": 0 | |
| }, | |
| "official_result_eligible": true, | |
| "precision": "bf16", | |
| "require_observed_supervision": true, | |
| "safety_capabilities": { | |
| "links": [ | |
| false, | |
| false, | |
| false | |
| ], | |
| "presence": [ | |
| true, | |
| true, | |
| true | |
| ], | |
| "risk": false | |
| }, | |
| "ssa_ablation": "none", | |
| "stage": 3, | |
| "structured_loss_recipe": { | |
| "links": "sigmoid_l1", | |
| "presence": "bce_with_logits_fp32", | |
| "reduction": "mean_over_finite_enabled_targets", | |
| "risk": "sigmoid_l1" | |
| }, | |
| "supervision_capabilities": { | |
| "links": [ | |
| false, | |
| false, | |
| false | |
| ], | |
| "presence": [ | |
| true, | |
| true, | |
| true | |
| ], | |
| "risk": false | |
| }, | |
| "training_complete": true | |
| }, | |
| "core_config": { | |
| "image_size": 448, | |
| "language_dim": 4096, | |
| "link_loss_weight": 0.5, | |
| "mask_probability_threshold": 0.5, | |
| "max_regions": 100, | |
| "max_text_tokens": 256, | |
| "patch_size": 14, | |
| "presence_loss_weight": 0.5, | |
| "region_dim": 1024, | |
| "risk_loss_weight": 1.0, | |
| "roi_size": 4, | |
| "safety_capabilities": { | |
| "links": [ | |
| false, | |
| false, | |
| false | |
| ], | |
| "presence": [ | |
| true, | |
| true, | |
| true | |
| ], | |
| "risk": false | |
| }, | |
| "text_overflow_policy": "error", | |
| "vision_dim": 1024, | |
| "vision_prefix_tokens": 1 | |
| }, | |
| "detector_config": { | |
| "architecture": { | |
| "amp": true, | |
| "bbox_reparam": true, | |
| "ca_nheads": 16, | |
| "cls_loss_coef": 1.0, | |
| "dec_layers": 6, | |
| "dec_n_points": 2, | |
| "encoder": "dinov2_windowed_small", | |
| "gradient_checkpointing": false, | |
| "group_detr": 13, | |
| "hidden_dim": 256, | |
| "ia_bce_loss": true, | |
| "layer_norm": true, | |
| "license": "Apache-2.0", | |
| "lite_refpoint_refine": true, | |
| "mask_downsample_ratio": 4, | |
| "num_classes": 1, | |
| "num_queries": 300, | |
| "num_select": 300, | |
| "num_windows": 2, | |
| "out_feature_indexes": [ | |
| 3, | |
| 6, | |
| 9, | |
| 12 | |
| ], | |
| "patch_size": 12, | |
| "positional_encoding_size": 64, | |
| "projector_scale": [ | |
| "P4" | |
| ], | |
| "resolution": 768, | |
| "sa_nheads": 8, | |
| "segmentation_head": true, | |
| "two_stage": true | |
| }, | |
| "proposal_policy": { | |
| "mask_threshold": 0.5, | |
| "max_regions": 100, | |
| "nms_threshold": 0.6, | |
| "reject_empty_masks": true, | |
| "score_threshold": 0.15, | |
| "version": "falcon-proposals/v1" | |
| }, | |
| "variant": "seg-2xlarge" | |
| }, | |
| "image_processor_config": { | |
| "_processor_class": null, | |
| "crop_size": { | |
| "height": 448, | |
| "width": 448 | |
| }, | |
| "do_center_crop": true, | |
| "do_convert_rgb": true, | |
| "do_normalize": true, | |
| "do_rescale": true, | |
| "do_resize": true, | |
| "image_mean": [ | |
| 0.485, | |
| 0.456, | |
| 0.406 | |
| ], | |
| "image_processor_type": "BitImageProcessor", | |
| "image_std": [ | |
| 0.229, | |
| 0.224, | |
| 0.225 | |
| ], | |
| "resample": 3, | |
| "rescale_factor": 0.00392156862745098, | |
| "size": { | |
| "height": 448, | |
| "width": 448 | |
| } | |
| }, | |
| "language_config": { | |
| "_attn_implementation_autoset": true, | |
| "add_cross_attention": false, | |
| "architectures": [ | |
| "LlamaForCausalLM" | |
| ], | |
| "attention_bias": false, | |
| "attention_dropout": 0.0, | |
| "attn_implementation": "sdpa", | |
| "bad_words_ids": null, | |
| "begin_suppress_tokens": null, | |
| "bos_token_id": 1, | |
| "chunk_size_feed_forward": 0, | |
| "cross_attention_hidden_size": null, | |
| "decoder_start_token_id": null, | |
| "diversity_penalty": 0.0, | |
| "do_sample": false, | |
| "early_stopping": false, | |
| "encoder_no_repeat_ngram_size": 0, | |
| "eos_token_id": 2, | |
| "exponential_decay_length_penalty": null, | |
| "finetuning_task": null, | |
| "forced_bos_token_id": null, | |
| "forced_eos_token_id": null, | |
| "head_dim": 128, | |
| "hidden_act": "silu", | |
| "hidden_size": 4096, | |
| "id2label": { | |
| "0": "LABEL_0", | |
| "1": "LABEL_1" | |
| }, | |
| "initializer_range": 0.02, | |
| "intermediate_size": 11008, | |
| "is_decoder": false, | |
| "is_encoder_decoder": false, | |
| "label2id": { | |
| "LABEL_0": 0, | |
| "LABEL_1": 1 | |
| }, | |
| "length_penalty": 1.0, | |
| "max_length": 20, | |
| "max_position_embeddings": 4096, | |
| "min_length": 0, | |
| "mlp_bias": false, | |
| "model_type": "llama", | |
| "no_repeat_ngram_size": 0, | |
| "num_attention_heads": 32, | |
| "num_beam_groups": 1, | |
| "num_beams": 1, | |
| "num_hidden_layers": 32, | |
| "num_key_value_heads": 32, | |
| "num_return_sequences": 1, | |
| "output_attentions": false, | |
| "output_hidden_states": false, | |
| "output_scores": false, | |
| "pad_token_id": 0, | |
| "prefix": null, | |
| "pretraining_tp": 1, | |
| "problem_type": null, | |
| "pruned_heads": {}, | |
| "remove_invalid_values": false, | |
| "repetition_penalty": 1.0, | |
| "return_dict": true, | |
| "return_dict_in_generate": false, | |
| "rms_norm_eps": 1e-05, | |
| "rope_scaling": null, | |
| "rope_theta": 10000.0, | |
| "sep_token_id": null, | |
| "suppress_tokens": null, | |
| "task_specific_params": null, | |
| "temperature": 1.0, | |
| "tf_legacy_loss": false, | |
| "tie_encoder_decoder": false, | |
| "tie_word_embeddings": false, | |
| "tokenizer_class": null, | |
| "top_k": 50, | |
| "top_p": 1.0, | |
| "torch_dtype": "bfloat16", | |
| "torchscript": false, | |
| "transformers_version": "4.49.0", | |
| "typical_p": 1.0, | |
| "use_bfloat16": false, | |
| "use_cache": true, | |
| "vocab_size": 32000 | |
| }, | |
| "language_generation_config": { | |
| "_from_model_config": false, | |
| "assistant_confidence_threshold": 0.4, | |
| "assistant_early_exit": null, | |
| "assistant_lookbehind": 10, | |
| "bad_words_ids": null, | |
| "begin_suppress_tokens": null, | |
| "bos_token_id": 1, | |
| "cache_config": null, | |
| "cache_implementation": null, | |
| "constraints": null, | |
| "decoder_start_token_id": null, | |
| "disable_compile": false, | |
| "diversity_penalty": 0.0, | |
| "do_sample": false, | |
| "dola_layers": null, | |
| "early_stopping": false, | |
| "encoder_no_repeat_ngram_size": 0, | |
| "encoder_repetition_penalty": 1.0, | |
| "eos_token_id": 2, | |
| "epsilon_cutoff": 0.0, | |
| "eta_cutoff": 0.0, | |
| "exponential_decay_length_penalty": null, | |
| "force_words_ids": null, | |
| "forced_bos_token_id": null, | |
| "forced_decoder_ids": null, | |
| "forced_eos_token_id": null, | |
| "generation_kwargs": {}, | |
| "guidance_scale": null, | |
| "is_assistant": false, | |
| "length_penalty": 1.0, | |
| "low_memory": null, | |
| "max_length": 4096, | |
| "max_matching_ngram_size": null, | |
| "max_new_tokens": null, | |
| "max_time": null, | |
| "min_length": 0, | |
| "min_new_tokens": null, | |
| "min_p": null, | |
| "no_repeat_ngram_size": 0, | |
| "num_assistant_tokens": 20, | |
| "num_assistant_tokens_schedule": "constant", | |
| "num_beam_groups": 1, | |
| "num_beams": 1, | |
| "num_return_sequences": 1, | |
| "output_attentions": false, | |
| "output_hidden_states": false, | |
| "output_logits": null, | |
| "output_scores": false, | |
| "pad_token_id": 0, | |
| "penalty_alpha": null, | |
| "prompt_lookup_num_tokens": null, | |
| "remove_invalid_values": false, | |
| "renormalize_logits": false, | |
| "repetition_penalty": 1.0, | |
| "return_dict_in_generate": false, | |
| "return_legacy_cache": null, | |
| "sequence_bias": null, | |
| "stop_strings": null, | |
| "suppress_tokens": null, | |
| "target_lookbehind": 10, | |
| "temperature": 0.9, | |
| "token_healing": false, | |
| "top_k": 50, | |
| "top_p": 0.6, | |
| "transformers_version": "4.49.0", | |
| "typical_p": 1.0, | |
| "use_cache": true, | |
| "watermarking_config": null | |
| }, | |
| "lora_config": { | |
| "alpha": 32, | |
| "dropout": 0.05, | |
| "rank": 16, | |
| "target_modules": [ | |
| "k_proj", | |
| "o_proj", | |
| "q_proj", | |
| "v_proj" | |
| ] | |
| }, | |
| "model_type": "falcon_x", | |
| "runtime_config": { | |
| "detector": { | |
| "mask_threshold": 0.5, | |
| "max_regions": 100, | |
| "nms_threshold": 0.6, | |
| "resolution": 768, | |
| "score_threshold": 0.15, | |
| "variant": "seg-2xlarge" | |
| }, | |
| "experiment": { | |
| "backbone_layout": "independent", | |
| "name": "falcon-x", | |
| "protocol": "dataset-v1", | |
| "ssa_ablation": "none", | |
| "world_size": null | |
| }, | |
| "model": { | |
| "component_order": [ | |
| "detonator", | |
| "explosive", | |
| "battery" | |
| ], | |
| "image_size": 448, | |
| "language_model": "lmsys/vicuna-7b-v1.5", | |
| "link_order": [ | |
| [ | |
| "battery", | |
| "detonator" | |
| ], | |
| [ | |
| "battery", | |
| "explosive" | |
| ], | |
| [ | |
| "detonator", | |
| "explosive" | |
| ] | |
| ], | |
| "lora_alpha": 32, | |
| "lora_dropout": 0.05, | |
| "lora_rank": 16, | |
| "max_regions": 100, | |
| "patch_size": 14, | |
| "region_dim": 1024, | |
| "roi_size": 4, | |
| "vision_model": "facebook/dinov2-large" | |
| }, | |
| "stages": { | |
| "stage1": { | |
| "batch_size": 1, | |
| "encoder_learning_rate": 0.00015, | |
| "epochs": 12, | |
| "expanded_scales": false, | |
| "gradient_accumulation": 16, | |
| "learning_rate": 0.0001, | |
| "lr_scheduler": "cosine", | |
| "multi_scale": false, | |
| "precision": "backend_mixed", | |
| "tf32": true, | |
| "warmup_epochs": 0.0, | |
| "weight_decay": 0.0001 | |
| }, | |
| "stage2": { | |
| "batch_size": 4, | |
| "epoch_size": null, | |
| "epochs": 1, | |
| "gradient_accumulation": 4, | |
| "learning_rate": 0.0001, | |
| "link_loss_weight": 0.5, | |
| "max_text_tokens": 256, | |
| "precision": "bf16", | |
| "presence_loss_weight": 0.5, | |
| "risk_loss_weight": 1.0, | |
| "sampling": "all", | |
| "seed": 42, | |
| "text_overflow_policy": "error", | |
| "tf32": true, | |
| "warmup_ratio": 0.03, | |
| "weight_decay": 0.0 | |
| }, | |
| "stage3": { | |
| "batch_size": 1, | |
| "epoch_size": null, | |
| "epochs": 1, | |
| "gradient_accumulation": 16, | |
| "learning_rate": 1e-05, | |
| "link_loss_weight": 0.5, | |
| "max_text_tokens": 256, | |
| "precision": "bf16", | |
| "presence_loss_weight": 0.5, | |
| "risk_loss_weight": 1.0, | |
| "sampling": "all", | |
| "seed": 42, | |
| "text_overflow_policy": "error", | |
| "tf32": true, | |
| "warmup_ratio": 0.03, | |
| "weight_decay": 0.0 | |
| } | |
| }, | |
| "training": { | |
| "batch_size": 1, | |
| "epoch_size": null, | |
| "epochs": 1, | |
| "gradient_accumulation": 16, | |
| "learning_rate": 0.0001, | |
| "link_loss_weight": 0.5, | |
| "max_text_tokens": 256, | |
| "precision": "bf16", | |
| "presence_loss_weight": 0.5, | |
| "risk_loss_weight": 1.0, | |
| "sampling": "all", | |
| "seed": 42, | |
| "text_overflow_policy": "error", | |
| "tf32": true, | |
| "warmup_ratio": 0.03, | |
| "weight_decay": 0.0 | |
| } | |
| }, | |
| "tie_word_embeddings": false, | |
| "torch_dtype": "float32", | |
| "transformers_version": "4.49.0", | |
| "vision_config": { | |
| "_attn_implementation_autoset": true, | |
| "add_cross_attention": false, | |
| "apply_layernorm": true, | |
| "architectures": [ | |
| "Dinov2Model" | |
| ], | |
| "attention_probs_dropout_prob": 0.0, | |
| "attn_implementation": "sdpa", | |
| "bad_words_ids": null, | |
| "begin_suppress_tokens": null, | |
| "bos_token_id": null, | |
| "chunk_size_feed_forward": 0, | |
| "cross_attention_hidden_size": null, | |
| "decoder_start_token_id": null, | |
| "diversity_penalty": 0.0, | |
| "do_sample": false, | |
| "drop_path_rate": 0.0, | |
| "early_stopping": false, | |
| "encoder_no_repeat_ngram_size": 0, | |
| "eos_token_id": null, | |
| "exponential_decay_length_penalty": null, | |
| "finetuning_task": null, | |
| "forced_bos_token_id": null, | |
| "forced_eos_token_id": null, | |
| "hidden_act": "gelu", | |
| "hidden_dropout_prob": 0.0, | |
| "hidden_size": 1024, | |
| "id2label": { | |
| "0": "LABEL_0", | |
| "1": "LABEL_1" | |
| }, | |
| "image_size": 518, | |
| "initializer_range": 0.02, | |
| "is_decoder": false, | |
| "is_encoder_decoder": false, | |
| "label2id": { | |
| "LABEL_0": 0, | |
| "LABEL_1": 1 | |
| }, | |
| "layer_norm_eps": 1e-06, | |
| "layerscale_value": 1.0, | |
| "length_penalty": 1.0, | |
| "max_length": 20, | |
| "min_length": 0, | |
| "mlp_ratio": 4, | |
| "model_type": "dinov2", | |
| "no_repeat_ngram_size": 0, | |
| "num_attention_heads": 16, | |
| "num_beam_groups": 1, | |
| "num_beams": 1, | |
| "num_channels": 3, | |
| "num_hidden_layers": 24, | |
| "num_return_sequences": 1, | |
| "out_features": [ | |
| "stage24" | |
| ], | |
| "out_indices": [ | |
| 24 | |
| ], | |
| "output_attentions": false, | |
| "output_hidden_states": false, | |
| "output_scores": false, | |
| "pad_token_id": null, | |
| "patch_size": 14, | |
| "prefix": null, | |
| "problem_type": null, | |
| "pruned_heads": {}, | |
| "qkv_bias": true, | |
| "remove_invalid_values": false, | |
| "repetition_penalty": 1.0, | |
| "reshape_hidden_states": true, | |
| "return_dict": true, | |
| "return_dict_in_generate": false, | |
| "sep_token_id": null, | |
| "stage_names": [ | |
| "stem", | |
| "stage1", | |
| "stage2", | |
| "stage3", | |
| "stage4", | |
| "stage5", | |
| "stage6", | |
| "stage7", | |
| "stage8", | |
| "stage9", | |
| "stage10", | |
| "stage11", | |
| "stage12", | |
| "stage13", | |
| "stage14", | |
| "stage15", | |
| "stage16", | |
| "stage17", | |
| "stage18", | |
| "stage19", | |
| "stage20", | |
| "stage21", | |
| "stage22", | |
| "stage23", | |
| "stage24" | |
| ], | |
| "suppress_tokens": null, | |
| "task_specific_params": null, | |
| "temperature": 1.0, | |
| "tf_legacy_loss": false, | |
| "tie_encoder_decoder": false, | |
| "tie_word_embeddings": true, | |
| "tokenizer_class": null, | |
| "top_k": 50, | |
| "top_p": 1.0, | |
| "torch_dtype": "float32", | |
| "torchscript": false, | |
| "transformers_version": "4.49.0", | |
| "typical_p": 1.0, | |
| "use_bfloat16": false, | |
| "use_mask_token": true, | |
| "use_swiglu_ffn": false | |
| } | |
| } | |