decision-0.8b / evaluation /fresh-test.json
johnsonchromia's picture
Publish Decision-0.8B and verified benchmarks
cb17a40 verified
Raw History Blame Contribute Delete
5.3 kB
{
"version": 1,
"created_at": "2026-09-29T09:14:09.603486+00:00",
"model": {
"provider": "local",
"name": "models/Qwen--Qwen3.5-0.8B",
"adapter": "runs/decision-0.8b-v2/adapter",
"identity_version": 2,
"prompt_sha256": "b2c4baf03d0224bf37ea1b2e11c2fa2c98c26d09a7a881a4565150829d1e2af0",
"model_weights_sha256": "79353a8fb811beddb052c75f92debca361226dcb0700d8f46004a77bae214f7c",
"model_config_sha256": "779572e46ae6552ba1826a162ab198560403ab1adeed8654e24e45c14e8ba978",
"tokenizer_sha256": "4773ccc9cb7087ffa57977144fc5786e4c1b5ef4d2b25a850bd25d3318581142",
"snapshot_sha256": "6aa297f2cadde0920e13ed5e2e0b78a33a36046d0d5fcb8efd8b03b4cd7e5ab3",
"weight_identity_status": "verified_local_content",
"encoding_sha256": "1e1255124fb4f698cabd531cbc251e2b0b91b8d119a53cf88974792f196a2c30",
"inference_dtype": "bfloat16",
"adapter_artifacts": {
"config_sha256": "9fee8484ccc91ee15604c0264461eecfa46b2f5605362c662d31191c08ed900e",
"weights_sha256": "cb80a8e8a5ffd1232abaecd37a3f91a58c0f268d218069ce4cb779f70158aff8"
}
},
"dataset_sha256": "9411bdb509a0f18dbb742b4e027613f875b7ac537e4e2ebb688cdc9394ffdd8a",
"cases_sha256": "bc0fd19f9344a8cf9e400813c49ee73d02151f8ffbd5ad87f7b2dfac11c471f3",
"allow_test": true,
"max_length": 2048,
"calibration": null,
"temperature": 1.0,
"summary": {
"count": 500,
"attempted": 500,
"scored": 500,
"errors": 0,
"skipped": 0,
"accuracy_all": 0.796,
"accuracy_attempted": 0.796,
"accuracy_scored": 0.796,
"accuracy_family_macro": 0.796,
"accuracy_by_family": {
"clinc150": 0.89,
"go_emotions": 0.82,
"helpsteer2": 0.47,
"paws": 0.92,
"vitaminc": 0.88
},
"by_source": {
"clinc150": {
"count": 100,
"scored": 100,
"accuracy_all": 0.89,
"accuracy_attempted": 0.89,
"macro_f1_scored": 0.8532776100920981,
"macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics."
},
"go_emotions": {
"count": 100,
"scored": 100,
"accuracy_all": 0.82,
"accuracy_attempted": 0.82,
"macro_f1_scored": 0.8197115384615384,
"macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics."
},
"helpsteer2": {
"count": 100,
"scored": 100,
"accuracy_all": 0.47,
"accuracy_attempted": 0.47,
"macro_f1_scored": 0.24495552607007715,
"macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics."
},
"paws": {
"count": 100,
"scored": 100,
"accuracy_all": 0.92,
"accuracy_attempted": 0.92,
"macro_f1_scored": 0.9184006527947777,
"macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics."
},
"vitaminc": {
"count": 100,
"scored": 100,
"accuracy_all": 0.88,
"accuracy_attempted": 0.88,
"macro_f1_scored": 0.8224019167415394,
"macro_f1_note": "Semantic option keys on successful cases; undefined for keys without shared semantics."
}
},
"probability_metrics": {
"count": 500,
"brier_multiclass_sum": 0.2777899383756419,
"nll": 0.505363078857301,
"infinite_nll_cases": 0,
"ece": 0.06087935093045237,
"ece_equal_width_bins": 15,
"aurc": 0.059849012978785913,
"coverage_at_error": {
"0.01": 0.224,
"0.05": 0.538,
"0.1": 0.736
},
"risk_coverage_note": "Empirical diagnostic, not a held-out error guarantee; whole ties accepted."
},
"probability_incomplete": 0,
"latency_ms": {
"count": 500,
"mean": 20.08110391406808,
"p50": 19.291491502372082,
"p95": 19.581971003935905,
"min": 18.74839499942027,
"max": 414.07370699744206
},
"wall_latency_ms": {
"count": 500,
"mean": 21.834221785858972,
"p50": 20.283657002437394,
"p95": 24.277247994905334,
"min": 19.293662000563927,
"max": 432.526392993168
},
"input_tokens": {
"count": 500,
"mean": 309.888,
"p50": 202.0,
"p95": 888.1499999999999,
"min": 108.0,
"max": 1631.0
},
"state_characters": {
"count": 500,
"mean": 571.486,
"p50": 199.5,
"p95": 2825.1,
"min": 10.0,
"max": 6317.0
},
"notes": [
"accuracy_all counts errors and skips as incorrect",
"accuracy_attempted and family macro count errors as wrong and exclude intentional skips",
"family defaults to source; latency_ms is local forward-pass or hosted HTTP time",
"wall_latency_ms includes per-record scorer preprocessing and inference; no warmup excluded",
"F1 uses semantic option keys, never shuffled answer letters",
"Probability metrics only include explicitly complete distributions"
]
},
"split": "test",
"source_report_sha256": "863b1ec56014467aa1518b82ac201af6df6d4b5d7494fe9c74229c6eb9fedd88",
"source_report_name": "fresh-decision.json",
"scope": "Frozen candidate test report; independence depends on recorded protocol"
}