Instructions to use evalengine/decision-0.8b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use evalengine/decision-0.8b with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen3.5-0.8B") model = PeftModel.from_pretrained(base_model, "evalengine/decision-0.8b") - Notebooks
- Google Colab
- Kaggle
Download evaluation/fresh-test.json from evalengine/decision-0.8b: direct link, hf CLI and curl.
- Browser
- Download file 5.3 kB
-
https://huggingface.co/evalengine/decision-0.8b/resolve/main/evaluation/fresh-test.json
- Command line
-
hf download hf://evalengine/decision-0.8b/evaluation/fresh-test.json
-
curl -L -o fresh-test.json https://huggingface.co/evalengine/decision-0.8b/resolve/main/evaluation/fresh-test.json
5.3 kB
| { | |
| "version": 1, | |
| "created_at": "2026-09-29T09:14:09.603486+00:00", | |
| "model": { | |
| "provider": "local", | |
| "name": "models/Qwen--Qwen3.5-0.8B", | |
| "adapter": "runs/decision-0.8b-v2/adapter", | |
| "identity_version": 2, | |
| "prompt_sha256": "b2c4baf03d0224bf37ea1b2e11c2fa2c98c26d09a7a881a4565150829d1e2af0", | |
| "model_weights_sha256": "79353a8fb811beddb052c75f92debca361226dcb0700d8f46004a77bae214f7c", | |
| "model_config_sha256": "779572e46ae6552ba1826a162ab198560403ab1adeed8654e24e45c14e8ba978", | |
| "tokenizer_sha256": "4773ccc9cb7087ffa57977144fc5786e4c1b5ef4d2b25a850bd25d3318581142", | |
| "snapshot_sha256": "6aa297f2cadde0920e13ed5e2e0b78a33a36046d0d5fcb8efd8b03b4cd7e5ab3", | |
| "weight_identity_status": "verified_local_content", | |
| "encoding_sha256": "1e1255124fb4f698cabd531cbc251e2b0b91b8d119a53cf88974792f196a2c30", | |
| "inference_dtype": "bfloat16", | |
| "adapter_artifacts": { | |
| "config_sha256": "9fee8484ccc91ee15604c0264461eecfa46b2f5605362c662d31191c08ed900e", | |
| "weights_sha256": "cb80a8e8a5ffd1232abaecd37a3f91a58c0f268d218069ce4cb779f70158aff8" | |
| } | |
| }, | |
| "dataset_sha256": "9411bdb509a0f18dbb742b4e027613f875b7ac537e4e2ebb688cdc9394ffdd8a", | |
| "cases_sha256": "bc0fd19f9344a8cf9e400813c49ee73d02151f8ffbd5ad87f7b2dfac11c471f3", | |
| "allow_test": true, | |
| "max_length": 2048, | |
| "calibration": null, | |
| "temperature": 1.0, | |
| "summary": { | |
| "count": 500, | |
| "attempted": 500, | |
| "scored": 500, | |
| "errors": 0, | |
| "skipped": 0, | |
| "accuracy_all": 0.796, | |
| "accuracy_attempted": 0.796, | |
| "accuracy_scored": 0.796, | |
| "accuracy_family_macro": 0.796, | |
| "accuracy_by_family": { | |
| "clinc150": 0.89, | |
| "go_emotions": 0.82, | |
| "helpsteer2": 0.47, | |
| "paws": 0.92, | |
| "vitaminc": 0.88 | |
| }, | |
| "by_source": { | |
| "clinc150": { | |
| "count": 100, | |
| "scored": 100, | |
| "accuracy_all": 0.89, | |
| "accuracy_attempted": 0.89, | |
| "macro_f1_scored": 0.8532776100920981, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "go_emotions": { | |
| "count": 100, | |
| "scored": 100, | |
| "accuracy_all": 0.82, | |
| "accuracy_attempted": 0.82, | |
| "macro_f1_scored": 0.8197115384615384, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "helpsteer2": { | |
| "count": 100, | |
| "scored": 100, | |
| "accuracy_all": 0.47, | |
| "accuracy_attempted": 0.47, | |
| "macro_f1_scored": 0.24495552607007715, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "paws": { | |
| "count": 100, | |
| "scored": 100, | |
| "accuracy_all": 0.92, | |
| "accuracy_attempted": 0.92, | |
| "macro_f1_scored": 0.9184006527947777, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "vitaminc": { | |
| "count": 100, | |
| "scored": 100, | |
| "accuracy_all": 0.88, | |
| "accuracy_attempted": 0.88, | |
| "macro_f1_scored": 0.8224019167415394, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| } | |
| }, | |
| "probability_metrics": { | |
| "count": 500, | |
| "brier_multiclass_sum": 0.2777899383756419, | |
| "nll": 0.505363078857301, | |
| "infinite_nll_cases": 0, | |
| "ece": 0.06087935093045237, | |
| "ece_equal_width_bins": 15, | |
| "aurc": 0.059849012978785913, | |
| "coverage_at_error": { | |
| "0.01": 0.224, | |
| "0.05": 0.538, | |
| "0.1": 0.736 | |
| }, | |
| "risk_coverage_note": "Empirical diagnostic, not a held-out error guarantee; whole ties accepted." | |
| }, | |
| "probability_incomplete": 0, | |
| "latency_ms": { | |
| "count": 500, | |
| "mean": 20.08110391406808, | |
| "p50": 19.291491502372082, | |
| "p95": 19.581971003935905, | |
| "min": 18.74839499942027, | |
| "max": 414.07370699744206 | |
| }, | |
| "wall_latency_ms": { | |
| "count": 500, | |
| "mean": 21.834221785858972, | |
| "p50": 20.283657002437394, | |
| "p95": 24.277247994905334, | |
| "min": 19.293662000563927, | |
| "max": 432.526392993168 | |
| }, | |
| "input_tokens": { | |
| "count": 500, | |
| "mean": 309.888, | |
| "p50": 202.0, | |
| "p95": 888.1499999999999, | |
| "min": 108.0, | |
| "max": 1631.0 | |
| }, | |
| "state_characters": { | |
| "count": 500, | |
| "mean": 571.486, | |
| "p50": 199.5, | |
| "p95": 2825.1, | |
| "min": 10.0, | |
| "max": 6317.0 | |
| }, | |
| "notes": [ | |
| "accuracy_all counts errors and skips as incorrect", | |
| "accuracy_attempted and family macro count errors as wrong and exclude intentional skips", | |
| "family defaults to source; latency_ms is local forward-pass or hosted HTTP time", | |
| "wall_latency_ms includes per-record scorer preprocessing and inference; no warmup excluded", | |
| "F1 uses semantic option keys, never shuffled answer letters", | |
| "Probability metrics only include explicitly complete distributions" | |
| ] | |
| }, | |
| "split": "test", | |
| "source_report_sha256": "863b1ec56014467aa1518b82ac201af6df6d4b5d7494fe9c74229c6eb9fedd88", | |
| "source_report_name": "fresh-decision.json", | |
| "scope": "Frozen candidate test report; independence depends on recorded protocol" | |
| } | |