PEFT
Safetensors
English
sev
research
cybersecurity
agent-activity
decision-model
lora
Sev-4B / calibration.json
macmacmacmac's picture
Publish Sev-4B v0.3.0 response-policy research checkpoint
da0a131 verified
Raw History Blame Contribute Delete
4.62 kB
{
"temperature": 1.6817928305074288,
"split": "calibration",
"n": 2674,
"records": 1920,
"fit": {
"aggregation": "micro",
"points": 81
},
"range": [
0.25,
4.0
],
"method": "Minimum question-weighted raw-logit NLL on the registered 81-point grid",
"rows_sha256": "026b68705b7ac76bc4361bc2fda1cbfe74265b436a807ea68bca77f87d3cf9da",
"suite_sha256": "3b9a80dfd212b6932b66ea520c35ab1a3b19e132a81af1b6b7b306d776a70a02",
"partition_sha256": "890c363c5bfe5351600071691eefb47dcf4e2530b31b75d4ce8830bd6381f585",
"raw": {
"n": 2674,
"nll": 0.4725028808328134,
"acc": 0.7860882572924458,
"ece": 0.07626143666496761,
"brier": 0.27460003407338146,
"mean_conf": 0.8621290491523242,
"confident_error_rate": 0.014958863126402393,
"coverage_at_0_9": 0.6484667165295438,
"accuracy_at_0_9": 0.9769319492502884,
"coverage_at_5pct_error": 0.6836200448765893,
"coverage_at_1pct_error": 0.5445026178010471,
"aurc": 0.04976778143359611,
"error_rate_at_0_9": 0.02306805074971165,
"confidence_bias": 0.07604079185987844,
"top_bins": {
"0.9": {
"n": 1734,
"errors": 40,
"error_rate": 0.02306805074971165
},
"0.95": {
"n": 1681,
"errors": 27,
"error_rate": 0.016061867935752528
},
"0.99": {
"n": 1576,
"errors": 18,
"error_rate": 0.011421319796954314
}
},
"selective": {
"0.5": {
"coverage": 0.5336574420344053,
"accuracy": 0.9901892081289418,
"confidence_cutoff": 0.9953955617749282
},
"0.8": {
"coverage": 0.8002991772625281,
"accuracy": 0.8700934579439252,
"confidence_cutoff": 0.6415496468544006
}
},
"score_mae": 0.2929868856899385,
"ranked_probability_score": 0.051781539475752744
},
"calibrated_in_sample": {
"n": 2674,
"nll": 0.43392263388002145,
"acc": 0.7860882572924458,
"ece": 0.03637322481154805,
"brier": 0.25527252174288456,
"mean_conf": 0.8087133880682658,
"confident_error_rate": 0.008227374719521317,
"coverage_at_0_9": 0.6110695587135377,
"accuracy_at_0_9": 0.9865361077111383,
"coverage_at_5pct_error": 0.681376215407629,
"coverage_at_1pct_error": 0.5508601346297681,
"aurc": 0.04945553724105698,
"error_rate_at_0_9": 0.01346389228886169,
"confidence_bias": 0.022625130775819957,
"top_bins": {
"0.9": {
"n": 1634,
"errors": 22,
"error_rate": 0.01346389228886169
},
"0.95": {
"n": 1216,
"errors": 6,
"error_rate": 0.004934210526315789
},
"0.99": {
"n": 849,
"errors": 0,
"error_rate": 0.0
}
},
"selective": {
"0.5": {
"coverage": 0.5415108451757666,
"accuracy": 0.9903314917127072,
"confidence_cutoff": 0.9486799539412789
},
"0.8": {
"coverage": 0.8025430067314884,
"accuracy": 0.869524697110904,
"confidence_cutoff": 0.52766162421842
}
},
"score_mae": 0.305269819429472,
"ranked_probability_score": 0.047216708865340906
},
"cross_validation": {
"raw": {
"n": 2674,
"ece": 0.07626143666496761,
"brier": 0.27460003407338146,
"nll": 0.4725028808328134,
"confident_error_rate": 0.014958863126402393,
"coverage_at_5pct_error": 0.6836200448765893
},
"out_of_fold": {
"n": 2674,
"ece": 0.03385822665024142,
"brier": 0.25613476719854117,
"nll": 0.4366331749349656,
"confident_error_rate": 0.008601346297681375,
"coverage_at_5pct_error": 0.6787584143605087
},
"ece_ci95": {
"raw": [
0.04594906201817091,
0.12662647548801886
],
"out_of_fold": [
0.02123422426875222,
0.07282022929885298
],
"delta": [
-0.0632669637757567,
-0.014968453240614701
]
},
"separated": true,
"temperatures": [
1.5691681957935013,
1.6817928305074288,
1.741101126592248,
1.8025009252216602,
1.6245047927124707
],
"folds": 5,
"seed": 20260927,
"samples": 1000,
"groups": 465,
"unit": "source-stratified (source, group); sibling questions and variants stay together"
},
"cv_grouping": "Exact group IDs across source families; one constant source stratum",
"development_used_for_fit": false,
"test_read": false,
"note": "CV is a diagnostic within calibration data, not fresh field validation. The shipped value is the full calibration fit."
}