Instructions to use evalengine/decision-0.8b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use evalengine/decision-0.8b with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen3.5-0.8B") model = PeftModel.from_pretrained(base_model, "evalengine/decision-0.8b") - Notebooks
- Google Colab
- Kaggle
File size: 5,297 Bytes
cb17a40 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 | {
"version": 1,
"created_at": "2026-09-29T09:14:09.603486+00:00",
"model": {
"provider": "local",
"name": "models/Qwen--Qwen3.5-0.8B",
"adapter": "runs/decision-0.8b-v2/adapter",
"identity_version": 2,
"prompt_sha256": "b2c4baf03d0224bf37ea1b2e11c2fa2c98c26d09a7a881a4565150829d1e2af0",
"model_weights_sha256": "79353a8fb811beddb052c75f92debca361226dcb0700d8f46004a77bae214f7c",
"model_config_sha256": "779572e46ae6552ba1826a162ab198560403ab1adeed8654e24e45c14e8ba978",
"tokenizer_sha256": "4773ccc9cb7087ffa57977144fc5786e4c1b5ef4d2b25a850bd25d3318581142",
"snapshot_sha256": "6aa297f2cadde0920e13ed5e2e0b78a33a36046d0d5fcb8efd8b03b4cd7e5ab3",
"weight_identity_status": "verified_local_content",
"encoding_sha256": "1e1255124fb4f698cabd531cbc251e2b0b91b8d119a53cf88974792f196a2c30",
"inference_dtype": "bfloat16",
"adapter_artifacts": {
"config_sha256": "9fee8484ccc91ee15604c0264461eecfa46b2f5605362c662d31191c08ed900e",
"weights_sha256": "cb80a8e8a5ffd1232abaecd37a3f91a58c0f268d218069ce4cb779f70158aff8"
}
},
"dataset_sha256": "9411bdb509a0f18dbb742b4e027613f875b7ac537e4e2ebb688cdc9394ffdd8a",
"cases_sha256": "bc0fd19f9344a8cf9e400813c49ee73d02151f8ffbd5ad87f7b2dfac11c471f3",
"allow_test": true,
"max_length": 2048,
"calibration": null,
"temperature": 1.0,
"summary": {
"count": 500,
"attempted": 500,
"scored": 500,
"errors": 0,
"skipped": 0,
"accuracy_all": 0.796,
"accuracy_attempted": 0.796,
"accuracy_scored": 0.796,
"accuracy_family_macro": 0.796,
"accuracy_by_family": {
"clinc150": 0.89,
"go_emotions": 0.82,
"helpsteer2": 0.47,
"paws": 0.92,
"vitaminc": 0.88
},
"by_source": {
"clinc150": {
"count": 100,
"scored": 100,
"accuracy_all": 0.89,
"accuracy_attempted": 0.89,
"macro_f1_scored": 0.8532776100920981,
"macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics."
},
"go_emotions": {
"count": 100,
"scored": 100,
"accuracy_all": 0.82,
"accuracy_attempted": 0.82,
"macro_f1_scored": 0.8197115384615384,
"macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics."
},
"helpsteer2": {
"count": 100,
"scored": 100,
"accuracy_all": 0.47,
"accuracy_attempted": 0.47,
"macro_f1_scored": 0.24495552607007715,
"macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics."
},
"paws": {
"count": 100,
"scored": 100,
"accuracy_all": 0.92,
"accuracy_attempted": 0.92,
"macro_f1_scored": 0.9184006527947777,
"macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics."
},
"vitaminc": {
"count": 100,
"scored": 100,
"accuracy_all": 0.88,
"accuracy_attempted": 0.88,
"macro_f1_scored": 0.8224019167415394,
"macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics."
}
},
"probability_metrics": {
"count": 500,
"brier_multiclass_sum": 0.2777899383756419,
"nll": 0.505363078857301,
"infinite_nll_cases": 0,
"ece": 0.06087935093045237,
"ece_equal_width_bins": 15,
"aurc": 0.059849012978785913,
"coverage_at_error": {
"0.01": 0.224,
"0.05": 0.538,
"0.1": 0.736
},
"risk_coverage_note": "Empirical diagnostic, not a held-out error guarantee; whole ties accepted."
},
"probability_incomplete": 0,
"latency_ms": {
"count": 500,
"mean": 20.08110391406808,
"p50": 19.291491502372082,
"p95": 19.581971003935905,
"min": 18.74839499942027,
"max": 414.07370699744206
},
"wall_latency_ms": {
"count": 500,
"mean": 21.834221785858972,
"p50": 20.283657002437394,
"p95": 24.277247994905334,
"min": 19.293662000563927,
"max": 432.526392993168
},
"input_tokens": {
"count": 500,
"mean": 309.888,
"p50": 202.0,
"p95": 888.1499999999999,
"min": 108.0,
"max": 1631.0
},
"state_characters": {
"count": 500,
"mean": 571.486,
"p50": 199.5,
"p95": 2825.1,
"min": 10.0,
"max": 6317.0
},
"notes": [
"accuracy_all counts errors and skips as incorrect",
"accuracy_attempted and family macro count errors as wrong and exclude intentional skips",
"family defaults to source; latency_ms is local forward-pass or hosted HTTP time",
"wall_latency_ms includes per-record scorer preprocessing and inference; no warmup excluded",
"F1 uses semantic option keys, never shuffled answer letters",
"Probability metrics only include explicitly complete distributions"
]
},
"split": "test",
"source_report_sha256": "863b1ec56014467aa1518b82ac201af6df6d4b5d7494fe9c74229c6eb9fedd88",
"source_report_name": "fresh-decision.json",
"scope": "Frozen candidate test report; independence depends on recorded protocol"
}
|