Instructions to use evalengine/decision-0.8b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use evalengine/decision-0.8b with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen3.5-0.8B") model = PeftModel.from_pretrained(base_model, "evalengine/decision-0.8b") - Notebooks
- Google Colab
- Kaggle
Download evaluation/development.json from evalengine/decision-0.8b: direct link, hf CLI and curl.
- Browser
- Download file 8.95 kB
-
https://huggingface.co/evalengine/decision-0.8b/resolve/main/evaluation/development.json
- Command line
-
hf download hf://evalengine/decision-0.8b/evaluation/development.json
-
curl -L -o development.json https://huggingface.co/evalengine/decision-0.8b/resolve/main/evaluation/development.json
8.95 kB
| { | |
| "version": 1, | |
| "created_at": "2026-09-29T09:12:44.565815+00:00", | |
| "model": { | |
| "provider": "local", | |
| "name": "models/Qwen--Qwen3.5-0.8B", | |
| "adapter": "runs/decision-0.8b-v2/adapter", | |
| "identity_version": 2, | |
| "prompt_sha256": "b2c4baf03d0224bf37ea1b2e11c2fa2c98c26d09a7a881a4565150829d1e2af0", | |
| "model_weights_sha256": "79353a8fb811beddb052c75f92debca361226dcb0700d8f46004a77bae214f7c", | |
| "model_config_sha256": "779572e46ae6552ba1826a162ab198560403ab1adeed8654e24e45c14e8ba978", | |
| "tokenizer_sha256": "4773ccc9cb7087ffa57977144fc5786e4c1b5ef4d2b25a850bd25d3318581142", | |
| "snapshot_sha256": "6aa297f2cadde0920e13ed5e2e0b78a33a36046d0d5fcb8efd8b03b4cd7e5ab3", | |
| "weight_identity_status": "verified_local_content", | |
| "encoding_sha256": "1e1255124fb4f698cabd531cbc251e2b0b91b8d119a53cf88974792f196a2c30", | |
| "inference_dtype": "bfloat16", | |
| "adapter_artifacts": { | |
| "config_sha256": "9fee8484ccc91ee15604c0264461eecfa46b2f5605362c662d31191c08ed900e", | |
| "weights_sha256": "cb80a8e8a5ffd1232abaecd37a3f91a58c0f268d218069ce4cb779f70158aff8" | |
| } | |
| }, | |
| "dataset_sha256": "21977c3441de6aff6f320084c0531e4131996b62b22c80b7d4c9cc158a7b6fa2", | |
| "cases_sha256": "be5fece2d070a6da55bc28c72d8f99d6506738d787583974980d827e13cb6430", | |
| "allow_test": false, | |
| "max_length": 2048, | |
| "calibration": null, | |
| "temperature": 1.0, | |
| "summary": { | |
| "count": 892, | |
| "attempted": 892, | |
| "scored": 892, | |
| "errors": 0, | |
| "skipped": 0, | |
| "accuracy_all": 0.7589686098654709, | |
| "accuracy_attempted": 0.7589686098654709, | |
| "accuracy_scored": 0.7589686098654709, | |
| "accuracy_family_macro": 0.7953125, | |
| "accuracy_by_family": { | |
| "ag_news": 0.84, | |
| "banking77": 0.92, | |
| "boolq": 0.86, | |
| "clinc150": 0.94, | |
| "go_emotions": 0.88, | |
| "helpsteer2": 0.54, | |
| "mnli": 0.86, | |
| "paws": 0.96, | |
| "policy": 1.0, | |
| "policy_v2": 0.875, | |
| "research_taxonomy_v21": 1.0, | |
| "routing_v2": 0.75, | |
| "simple_quorum": 0.52, | |
| "sst5": 0.46, | |
| "veto_alternative": 0.44, | |
| "vitaminc": 0.88 | |
| }, | |
| "by_source": { | |
| "ag_news": { | |
| "count": 50, | |
| "scored": 50, | |
| "accuracy_all": 0.84, | |
| "accuracy_attempted": 0.84, | |
| "macro_f1_scored": 0.8086663000977516, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "banking77": { | |
| "count": 50, | |
| "scored": 50, | |
| "accuracy_all": 0.92, | |
| "accuracy_attempted": 0.92, | |
| "macro_f1_scored": 0.8915989159891599, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "boolq": { | |
| "count": 50, | |
| "scored": 50, | |
| "accuracy_all": 0.86, | |
| "accuracy_attempted": 0.86, | |
| "macro_f1_scored": 0.8553121124431583, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "clinc150": { | |
| "count": 50, | |
| "scored": 50, | |
| "accuracy_all": 0.94, | |
| "accuracy_attempted": 0.94, | |
| "macro_f1_scored": 0.9318518518518518, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "go_emotions": { | |
| "count": 50, | |
| "scored": 50, | |
| "accuracy_all": 0.88, | |
| "accuracy_attempted": 0.88, | |
| "macro_f1_scored": 0.8792270531400965, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "helpsteer2": { | |
| "count": 50, | |
| "scored": 50, | |
| "accuracy_all": 0.54, | |
| "accuracy_attempted": 0.54, | |
| "macro_f1_scored": 0.30993685609837157, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "mnli": { | |
| "count": 50, | |
| "scored": 50, | |
| "accuracy_all": 0.86, | |
| "accuracy_attempted": 0.86, | |
| "macro_f1_scored": 0.8586944715976974, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "paws": { | |
| "count": 50, | |
| "scored": 50, | |
| "accuracy_all": 0.96, | |
| "accuracy_attempted": 0.96, | |
| "macro_f1_scored": 0.9597423510466989, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "policy": { | |
| "count": 48, | |
| "scored": 48, | |
| "accuracy_all": 1.0, | |
| "accuracy_attempted": 1.0, | |
| "macro_f1_scored": 1.0, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "policy_v2": { | |
| "count": 48, | |
| "scored": 48, | |
| "accuracy_all": 0.875, | |
| "accuracy_attempted": 0.875, | |
| "macro_f1_scored": 0.8756784434203789, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "research_taxonomy_v21": { | |
| "count": 48, | |
| "scored": 48, | |
| "accuracy_all": 1.0, | |
| "accuracy_attempted": 1.0, | |
| "macro_f1_scored": 1.0, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "routing_v2": { | |
| "count": 48, | |
| "scored": 48, | |
| "accuracy_all": 0.75, | |
| "accuracy_attempted": 0.75, | |
| "macro_f1_scored": 0.732477303315214, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "sst5": { | |
| "count": 50, | |
| "scored": 50, | |
| "accuracy_all": 0.46, | |
| "accuracy_attempted": 0.46, | |
| "macro_f1_scored": 0.38029227053140097, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "vitaminc": { | |
| "count": 50, | |
| "scored": 50, | |
| "accuracy_all": 0.88, | |
| "accuracy_attempted": 0.88, | |
| "macro_f1_scored": 0.8377538829151733, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "workflow.simple_quorum": { | |
| "count": 100, | |
| "scored": 100, | |
| "accuracy_all": 0.52, | |
| "accuracy_attempted": 0.52, | |
| "macro_f1_scored": 0.4465528146742568, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "workflow.veto_alternative": { | |
| "count": 100, | |
| "scored": 100, | |
| "accuracy_all": 0.44, | |
| "accuracy_attempted": 0.44, | |
| "macro_f1_scored": 0.38230900554844216, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| } | |
| }, | |
| "probability_metrics": { | |
| "count": 892, | |
| "brier_multiclass_sum": 0.34885692437247895, | |
| "nll": 0.6292707360366632, | |
| "infinite_nll_cases": 0, | |
| "ece": 0.08631027942615237, | |
| "ece_equal_width_bins": 15, | |
| "aurc": 0.08292349833641886, | |
| "coverage_at_error": { | |
| "0.01": 0.3475336322869955, | |
| "0.05": 0.4517937219730942, | |
| "0.1": 0.5986547085201793 | |
| }, | |
| "risk_coverage_note": "Empirical diagnostic, not a held-out error guarantee; whole ties accepted." | |
| }, | |
| "probability_incomplete": 0, | |
| "latency_ms": { | |
| "count": 892, | |
| "mean": 19.89369566444409, | |
| "p50": 19.335302997205872, | |
| "p95": 19.691220494132722, | |
| "min": 18.815770003129728, | |
| "max": 419.30977000447456 | |
| }, | |
| "wall_latency_ms": { | |
| "count": 892, | |
| "mean": 22.24163812639141, | |
| "p50": 20.735956000862643, | |
| "p95": 25.140228710370103, | |
| "min": 19.386896994546987, | |
| "max": 434.9724860076094 | |
| }, | |
| "input_tokens": { | |
| "count": 892, | |
| "mean": 355.80044843049325, | |
| "p50": 276.0, | |
| "p95": 896.3499999999998, | |
| "min": 109.0, | |
| "max": 1509.0 | |
| }, | |
| "state_characters": { | |
| "count": 892, | |
| "mean": 438.77466367713004, | |
| "p50": 191.5, | |
| "p95": 1702.4499999999998, | |
| "min": 10.0, | |
| "max": 5428.0 | |
| }, | |
| "notes": [ | |
| "accuracy_all counts errors and skips as incorrect", | |
| "accuracy_attempted and family macro count errors as wrong and exclude intentional skips", | |
| "family defaults to source; latency_ms is local forward-pass or hosted HTTP time", | |
| "wall_latency_ms includes per-record scorer preprocessing and inference; no warmup excluded", | |
| "F1 uses semantic option keys, never shuffled answer letters", | |
| "Probability metrics only include explicitly complete distributions" | |
| ] | |
| }, | |
| "split": "dev", | |
| "source_report_sha256": "701bf9841acbb867d113dff015a52323832deeaeff5b5174be63f1d05f9cf321", | |
| "source_report_name": "decision-development.json", | |
| "scope": "Reused model-development panel" | |
| } | |