Zero-Shot Classification
Safetensors
PEFT
English
openjev
classification
decision-model
listwise
gemma4
research
Instructions to use bambamdevs/openjev-e4b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use bambamdevs/openjev-e4b with PEFT:
Task type is invalid.
- Notebooks
- Google Colab
- Kaggle
Download model/openjev_config.json from bambamdevs/openjev-e4b: direct link, hf CLI and curl.
- Browser
- Download file 2.19 kB
-
https://huggingface.co/bambamdevs/openjev-e4b/resolve/main/model/openjev_config.json
- Command line
-
hf download hf://bambamdevs/openjev-e4b/model/openjev_config.json
-
curl -L -o openjev_config.json https://huggingface.co/bambamdevs/openjev-e4b/resolve/main/model/openjev_config.json
2.19 kB
| { | |
| "run_name": "e4b-v13", | |
| "model_id": "google/gemma-4-E4B-it", | |
| "seed": 44, | |
| "max_length": 8192, | |
| "head_dim": 384, | |
| "freeze_backbone": false, | |
| "use_lora": true, | |
| "lora_r": 16, | |
| "lora_alpha": 32, | |
| "lora_dropout": 0.0, | |
| "lora_targets": [ | |
| "q_proj", | |
| "k_proj", | |
| "v_proj", | |
| "o_proj" | |
| ], | |
| "batch_size": 4, | |
| "grad_accum": 4, | |
| "max_updates": 800, | |
| "head_lr": 6e-05, | |
| "lora_lr": 2e-05, | |
| "weight_decay": 0.01, | |
| "grad_clip": 1.0, | |
| "warmup_updates": 30, | |
| "min_lr_ratio": 0.15, | |
| "dataset_sampling_alpha": 0.5, | |
| "dataset_sampling_multipliers": { | |
| "nyu-mll/multi_nli": 1.2, | |
| "alisawuffles/WANLI": 1.35, | |
| "stanfordnlp/snli": 1.1, | |
| "synthetic/nli-haystack": 1.3, | |
| "product-science/xlam-hard": 1.25, | |
| "nvidia/HelpSteer2": 1.6, | |
| "PolyAI/banking77": 0.85, | |
| "Rowan/hellaswag": 0.9, | |
| "cais/mmlu": 1.0, | |
| "google/boolq": 0.9, | |
| "product-science/xlam-function-calling-60k-raw": 0.9, | |
| "nvidia/HelpSteer2-threshold": 1.35 | |
| }, | |
| "aux_nli": { | |
| "enabled": true, | |
| "loss_weight": 0.2 | |
| }, | |
| "log_every": 10, | |
| "save_every": 200, | |
| "rolling_window": 30, | |
| "high_loss_ce": 3.0, | |
| "hard_diag_every_micro": 4, | |
| "eval_every": 100, | |
| "eval_max_examples": 1400, | |
| "eval_batch_size": 8, | |
| "selection_metric": "macro_dataset_policy_nll", | |
| "calibration_min_group": 50, | |
| "calibration_prior_strength": 120.0, | |
| "early_stop_min_updates": 400, | |
| "early_stop_patience": 3, | |
| "loss": { | |
| "ce": 1.0, | |
| "brier": 0.15, | |
| "score_emd": 0.4, | |
| "score_expectation": 0.18 | |
| }, | |
| "model_version": "1.3.2", | |
| "version": "1.6.0", | |
| "seed_run": "artifacts/runs/e4b-v13", | |
| "v142_selection": "hybrid-ce", | |
| "v142_public_benchmark_firewall": true, | |
| "head_type": "hybrid", | |
| "v142_objective": "ce", | |
| "v151_rlcd_like": true, | |
| "v151_seed_run": "artifacts/runs/e4b-v142", | |
| "v151_policy_stage": "head_lora_rlcd", | |
| "v151_selected_policy": "head_lora_rlcd", | |
| "arena_v160": true, | |
| "arena_seed_run": "artifacts/runs/e4b-v151", | |
| "arena_accepted_rounds": 6, | |
| "arena_public_benchmark_firewall": "closed", | |
| "backbone_delta_file": "backbone_delta.pt", | |
| "arena_unfrozen_base_parameter_count": 65, | |
| "base_revision": "ee0ef6023621cff504d758262d4e04895a5af4a2" | |
| } | |