{ "temperature": 1.6817928305074288, "split": "calibration", "n": 2674, "records": 1920, "fit": { "aggregation": "micro", "points": 81 }, "range": [ 0.25, 4.0 ], "method": "Minimum question-weighted raw-logit NLL on the registered 81-point grid", "rows_sha256": "026b68705b7ac76bc4361bc2fda1cbfe74265b436a807ea68bca77f87d3cf9da", "suite_sha256": "3b9a80dfd212b6932b66ea520c35ab1a3b19e132a81af1b6b7b306d776a70a02", "partition_sha256": "890c363c5bfe5351600071691eefb47dcf4e2530b31b75d4ce8830bd6381f585", "raw": { "n": 2674, "nll": 0.4725028808328134, "acc": 0.7860882572924458, "ece": 0.07626143666496761, "brier": 0.27460003407338146, "mean_conf": 0.8621290491523242, "confident_error_rate": 0.014958863126402393, "coverage_at_0_9": 0.6484667165295438, "accuracy_at_0_9": 0.9769319492502884, "coverage_at_5pct_error": 0.6836200448765893, "coverage_at_1pct_error": 0.5445026178010471, "aurc": 0.04976778143359611, "error_rate_at_0_9": 0.02306805074971165, "confidence_bias": 0.07604079185987844, "top_bins": { "0.9": { "n": 1734, "errors": 40, "error_rate": 0.02306805074971165 }, "0.95": { "n": 1681, "errors": 27, "error_rate": 0.016061867935752528 }, "0.99": { "n": 1576, "errors": 18, "error_rate": 0.011421319796954314 } }, "selective": { "0.5": { "coverage": 0.5336574420344053, "accuracy": 0.9901892081289418, "confidence_cutoff": 0.9953955617749282 }, "0.8": { "coverage": 0.8002991772625281, "accuracy": 0.8700934579439252, "confidence_cutoff": 0.6415496468544006 } }, "score_mae": 0.2929868856899385, "ranked_probability_score": 0.051781539475752744 }, "calibrated_in_sample": { "n": 2674, "nll": 0.43392263388002145, "acc": 0.7860882572924458, "ece": 0.03637322481154805, "brier": 0.25527252174288456, "mean_conf": 0.8087133880682658, "confident_error_rate": 0.008227374719521317, "coverage_at_0_9": 0.6110695587135377, "accuracy_at_0_9": 0.9865361077111383, "coverage_at_5pct_error": 0.681376215407629, "coverage_at_1pct_error": 0.5508601346297681, "aurc": 0.04945553724105698, "error_rate_at_0_9": 0.01346389228886169, "confidence_bias": 0.022625130775819957, "top_bins": { "0.9": { "n": 1634, "errors": 22, "error_rate": 0.01346389228886169 }, "0.95": { "n": 1216, "errors": 6, "error_rate": 0.004934210526315789 }, "0.99": { "n": 849, "errors": 0, "error_rate": 0.0 } }, "selective": { "0.5": { "coverage": 0.5415108451757666, "accuracy": 0.9903314917127072, "confidence_cutoff": 0.9486799539412789 }, "0.8": { "coverage": 0.8025430067314884, "accuracy": 0.869524697110904, "confidence_cutoff": 0.52766162421842 } }, "score_mae": 0.305269819429472, "ranked_probability_score": 0.047216708865340906 }, "cross_validation": { "raw": { "n": 2674, "ece": 0.07626143666496761, "brier": 0.27460003407338146, "nll": 0.4725028808328134, "confident_error_rate": 0.014958863126402393, "coverage_at_5pct_error": 0.6836200448765893 }, "out_of_fold": { "n": 2674, "ece": 0.03385822665024142, "brier": 0.25613476719854117, "nll": 0.4366331749349656, "confident_error_rate": 0.008601346297681375, "coverage_at_5pct_error": 0.6787584143605087 }, "ece_ci95": { "raw": [ 0.04594906201817091, 0.12662647548801886 ], "out_of_fold": [ 0.02123422426875222, 0.07282022929885298 ], "delta": [ -0.0632669637757567, -0.014968453240614701 ] }, "separated": true, "temperatures": [ 1.5691681957935013, 1.6817928305074288, 1.741101126592248, 1.8025009252216602, 1.6245047927124707 ], "folds": 5, "seed": 20260927, "samples": 1000, "groups": 465, "unit": "source-stratified (source, group); sibling questions and variants stay together" }, "cv_grouping": "Exact group IDs across source families; one constant source stratum", "development_used_for_fit": false, "test_read": false, "note": "CV is a diagnostic within calibration data, not fresh field validation. The shipped value is the full calibration fit." }