Any-to-Any
MLX
Safetensors
gemma4
mlx-vlm
rlcd
multimodal
classification
parallel-inference
image-text-to-text
audio
video
4-bit precision
Instructions to use larkooo/gemma-e2b-rlcd with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use larkooo/gemma-e2b-rlcd with MLX:
# Download the model from the Hub pip install huggingface_hub[hf_xet] hf download larkooo/gemma-e2b-rlcd --local-dir gemma-e2b-rlcd
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Atomic Chat
File size: 2,166 Bytes
53e24ca | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 | {
"answers": {
"animal": {
"calibration_status": "unvalidated",
"temperature": 1.0,
"probability_source": "restricted_next_token_logits",
"diagnostics": {
"logits": {
"cat": 14.711406707763672,
"dog": -7.430335521697998,
"other": -5.548311233520508
},
"allowed_token_mass": 0.9999990463256836,
"input_tokens": 164
},
"confidence": 0.9999999641376001,
"confidence_definition": "one_minus_normalized_entropy",
"probabilities": {
"cat": 0.9999999981682133,
"dog": 2.4208257443645767e-10,
"other": 1.5897040929196268e-09
},
"type": "choice",
"choice": "cat",
"selected_probability": 0.9999999981682133
},
"presence": {
"calibration_status": "unvalidated",
"temperature": 1.0,
"probability_source": "restricted_next_token_logits",
"type": "independent",
"probabilities": {
"cat": 0.9784876284221362,
"dog": 9.911270087838716e-09
},
"diagnostics": {
"cat": {
"logits": {
"yes": 10.465391159057617,
"no": 6.648011207580566
},
"allowed_token_mass": 0.999992311000824,
"input_tokens": 158
},
"dog": {
"logits": {
"yes": -0.7675755023956299,
"no": 17.662017822265625
},
"allowed_token_mass": 0.9999980330467224,
"input_tokens": 158
}
}
}
},
"model": "models/gemma-4-e2b-it-4bit",
"load_seconds": 2.818324833002407,
"decision_seconds": 0.5153297919896431,
"execution": {
"execution": "shared_prefix_gpu_batched_branches",
"prefix_prefills": 1,
"prefix_tokens": 50,
"question_suffix_tokens": [
114,
108,
108
],
"preprocess_seconds": 0.039328291022684425,
"prefill_seconds": 0.09224108399939723,
"kv_storage": "replicated_per_batch_row",
"compute_dtype": "float32",
"branch_batch_sizes": [
3
],
"branch_seconds": 0.38361333298962563
},
"video_sampling_fps": null,
"video_soundtrack": null
}
|