mill-screen / thresholds.json
Deeptanshuu's picture
Publish multilingual toxicity classifier: weights, card, thresholds, config
7fb84fb verified
Raw History Blame Contribute Delete
2.83 kB
{
"_readme": [
"Per-class decision thresholds. A probability at or above the threshold for a",
"class means that class fires. The six classes are independent: any number of",
"them can fire on the same comment.",
"",
"Do not use 0.5. The rare classes (severe_toxic, threat, identity_hate) are",
"well ordered but badly calibrated, so a 0.5 cut throws away most of their",
"recall. Every threshold below sits under 0.5 for that reason.",
"",
"Thresholds were chosen to maximise per-class F1 on the validation split and",
"are then applied unchanged to the held-out test split. They are a single",
"global set: there is no per-language block here, deliberately. See",
"_provenance.per_language_block_omitted."
],
"_provenance": {
"status": "FINAL. These thresholds are tuned on validation using the final best_model checkpoint (epoch 5 of a 6-epoch run that completed all 6 epochs) and applied unchanged to the test split. They match the shipped weights.",
"tuned_on": "dataset/split/val.csv",
"applied_to": "dataset/split/test.csv",
"source_run": "evaluation_results/eval_20260830_072515",
"source_checkpoint": "weights/toxic_classifier_xlmr_v2/best_model (epoch 5 of 6, encoder fine-tuned)",
"search": "direct sweep over candidate thresholds in [0.05, 0.95], per class independently, maximising F1 on validation",
"per_language_block_omitted": "The evaluation script also emits a per-language threshold block. It is not shipped: it comes from a code path with a known bug and nothing in the serving path ever reads it. An earlier run of that code path reported English severe_toxic at F1 0.597 when the maximum achievable at any threshold is 0.442."
},
"label_order": [
"toxic",
"severe_toxic",
"obscene",
"threat",
"insult",
"identity_hate"
],
"thresholds": {
"toxic": 0.4724489795918367,
"severe_toxic": 0.4724489795918367,
"obscene": 0.5275510204081633,
"threat": 0.5275510204081633,
"insult": 0.5642857142857143,
"identity_hate": 0.5642857142857143
},
"validation_f1_at_threshold": {
"_note": "F1 achieved at the threshold above, on the validation split, by THIS (shipped) checkpoint. This is the value the threshold search maximised, not a test-split number.",
"toxic": 0.964104847236908,
"severe_toxic": 0.7429062768701634,
"obscene": 0.937078390323331,
"threat": 0.8424657534246576,
"insult": 0.9223998826463253,
"identity_hate": 0.8753432180120813
},
"validation_positive_support": {
"_note": "Positive examples per class in the 35,658-row validation split used for tuning.",
"toxic": 17697,
"severe_toxic": 1655,
"obscene": 8626,
"threat": 760,
"insult": 10199,
"identity_hate": 1878,
"total_samples": 35658
}
}