Instructions to use evalengine/decision-0.8b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use evalengine/decision-0.8b with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Qwen/Qwen3.5-0.8B") model = PeftModel.from_pretrained(base_model, "evalengine/decision-0.8b") - Notebooks
- Google Colab
- Kaggle
Download evaluation/historical-test.json from evalengine/decision-0.8b: direct link, hf CLI and curl.
- Browser
- Download file 6.87 kB
-
https://huggingface.co/evalengine/decision-0.8b/resolve/main/evaluation/historical-test.json
- Command line
-
hf download hf://evalengine/decision-0.8b/evaluation/historical-test.json
-
curl -L -o historical-test.json https://huggingface.co/evalengine/decision-0.8b/resolve/main/evaluation/historical-test.json
6.87 kB
| { | |
| "version": 1, | |
| "created_at": "2026-09-29T09:17:00.954368+00:00", | |
| "model": { | |
| "provider": "local", | |
| "name": "models/Qwen--Qwen3.5-0.8B", | |
| "adapter": "runs/decision-0.8b-v2/adapter", | |
| "identity_version": 2, | |
| "prompt_sha256": "b2c4baf03d0224bf37ea1b2e11c2fa2c98c26d09a7a881a4565150829d1e2af0", | |
| "model_weights_sha256": "79353a8fb811beddb052c75f92debca361226dcb0700d8f46004a77bae214f7c", | |
| "model_config_sha256": "779572e46ae6552ba1826a162ab198560403ab1adeed8654e24e45c14e8ba978", | |
| "tokenizer_sha256": "4773ccc9cb7087ffa57977144fc5786e4c1b5ef4d2b25a850bd25d3318581142", | |
| "snapshot_sha256": "6aa297f2cadde0920e13ed5e2e0b78a33a36046d0d5fcb8efd8b03b4cd7e5ab3", | |
| "weight_identity_status": "verified_local_content", | |
| "encoding_sha256": "1e1255124fb4f698cabd531cbc251e2b0b91b8d119a53cf88974792f196a2c30", | |
| "inference_dtype": "bfloat16", | |
| "adapter_artifacts": { | |
| "config_sha256": "9fee8484ccc91ee15604c0264461eecfa46b2f5605362c662d31191c08ed900e", | |
| "weights_sha256": "cb80a8e8a5ffd1232abaecd37a3f91a58c0f268d218069ce4cb779f70158aff8" | |
| } | |
| }, | |
| "dataset_sha256": "7773448b87a0c4fc8e4cca7be136e1494fa9bffd5b36ae399be7e055243227d3", | |
| "cases_sha256": "dcc091e4687e649bd19f21e02b38a187a6b73ee909fac228e02f2da81010d631", | |
| "allow_test": true, | |
| "max_length": 2048, | |
| "calibration": null, | |
| "temperature": 1.0, | |
| "summary": { | |
| "count": 2800, | |
| "attempted": 2800, | |
| "scored": 2800, | |
| "errors": 0, | |
| "skipped": 0, | |
| "accuracy_all": 0.6882142857142857, | |
| "accuracy_attempted": 0.6882142857142857, | |
| "accuracy_scored": 0.6882142857142857, | |
| "accuracy_family_macro": 0.6261111111111111, | |
| "accuracy_by_family": { | |
| "clinc150": 0.9575, | |
| "exclusive_proof_fallback": 0.35, | |
| "go_emotions": 0.8425, | |
| "helpsteer2": 0.495, | |
| "narrow_veto_override": 0.48, | |
| "paws": 0.9125, | |
| "priority_exception": 0.33, | |
| "vitaminc": 0.7925, | |
| "weighted_role_quorum": 0.475 | |
| }, | |
| "by_source": { | |
| "clinc150": { | |
| "count": 400, | |
| "scored": 400, | |
| "accuracy_all": 0.9575, | |
| "accuracy_attempted": 0.9575, | |
| "macro_f1_scored": 0.9596451786927224, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "go_emotions": { | |
| "count": 400, | |
| "scored": 400, | |
| "accuracy_all": 0.8425, | |
| "accuracy_attempted": 0.8425, | |
| "macro_f1_scored": 0.8420646908040286, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "helpsteer2": { | |
| "count": 400, | |
| "scored": 400, | |
| "accuracy_all": 0.495, | |
| "accuracy_attempted": 0.495, | |
| "macro_f1_scored": 0.3669844359980635, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "paws": { | |
| "count": 400, | |
| "scored": 400, | |
| "accuracy_all": 0.9125, | |
| "accuracy_attempted": 0.9125, | |
| "macro_f1_scored": 0.9118249094630766, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "vitaminc": { | |
| "count": 400, | |
| "scored": 400, | |
| "accuracy_all": 0.7925, | |
| "accuracy_attempted": 0.7925, | |
| "macro_f1_scored": 0.7376537004098678, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "workflow.exclusive_proof_fallback": { | |
| "count": 200, | |
| "scored": 200, | |
| "accuracy_all": 0.35, | |
| "accuracy_attempted": 0.35, | |
| "macro_f1_scored": 0.21238337574215435, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "workflow.narrow_veto_override": { | |
| "count": 200, | |
| "scored": 200, | |
| "accuracy_all": 0.48, | |
| "accuracy_attempted": 0.48, | |
| "macro_f1_scored": 0.4445966514459665, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "workflow.priority_exception": { | |
| "count": 200, | |
| "scored": 200, | |
| "accuracy_all": 0.33, | |
| "accuracy_attempted": 0.33, | |
| "macro_f1_scored": 0.16666666666666666, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| }, | |
| "workflow.weighted_role_quorum": { | |
| "count": 200, | |
| "scored": 200, | |
| "accuracy_all": 0.475, | |
| "accuracy_attempted": 0.475, | |
| "macro_f1_scored": 0.4066951566951567, | |
| "macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics." | |
| } | |
| }, | |
| "probability_metrics": { | |
| "count": 2800, | |
| "brier_multiclass_sum": 0.4577625055759041, | |
| "nll": 0.8419201488883272, | |
| "infinite_nll_cases": 0, | |
| "ece": 0.13689068341361624, | |
| "ece_equal_width_bins": 15, | |
| "aurc": 0.1427349504890547, | |
| "coverage_at_error": { | |
| "0.01": 0.14107142857142857, | |
| "0.05": 0.245, | |
| "0.1": 0.40214285714285714 | |
| }, | |
| "risk_coverage_note": "Empirical diagnostic, not a held-out error guarantee; whole ties accepted." | |
| }, | |
| "probability_incomplete": 0, | |
| "latency_ms": { | |
| "count": 2800, | |
| "mean": 19.245112702062347, | |
| "p50": 19.073196999670472, | |
| "p95": 19.319465538865188, | |
| "min": 18.60275999933947, | |
| "max": 415.4445160093019 | |
| }, | |
| "wall_latency_ms": { | |
| "count": 2800, | |
| "mean": 21.053149590614858, | |
| "p50": 20.84802999161184, | |
| "p95": 23.9287625576253, | |
| "min": 19.18114699947182, | |
| "max": 437.42106699210126 | |
| }, | |
| "input_tokens": { | |
| "count": 2800, | |
| "mean": 334.02714285714285, | |
| "p50": 276.0, | |
| "p95": 844.0, | |
| "min": 108.0, | |
| "max": 1746.0 | |
| }, | |
| "state_characters": { | |
| "count": 2800, | |
| "mean": 482.25107142857144, | |
| "p50": 206.0, | |
| "p95": 2657.899999999995, | |
| "min": 7.0, | |
| "max": 6978.0 | |
| }, | |
| "notes": [ | |
| "accuracy_all counts errors and skips as incorrect", | |
| "accuracy_attempted and family macro count errors as wrong and exclude intentional skips", | |
| "family defaults to source; latency_ms is local forward-pass or hosted HTTP time", | |
| "wall_latency_ms includes per-record scorer preprocessing and inference; no warmup excluded", | |
| "F1 uses semantic option keys, never shuffled answer letters", | |
| "Probability metrics only include explicitly complete distributions" | |
| ] | |
| }, | |
| "split": "test", | |
| "source_report_sha256": "4ccfb86a7c6df69cf94efb8aa6aa45d16f373cd41a3916879b9db12b53bdb85c", | |
| "source_report_name": "historical-decision.json", | |
| "scope": "Frozen candidate test report; independence depends on recorded protocol" | |
| } | |