Upload modality-thresholds.json with huggingface_hub
Browse files- modality-thresholds.json +166 -0
modality-thresholds.json
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema_version": "1.0.0",
|
| 3 |
+
"taxonomy_version": "1.0.0",
|
| 4 |
+
"default_policy": "balanced",
|
| 5 |
+
"calibration": {
|
| 6 |
+
"method": "temperature_scaling",
|
| 7 |
+
"status": "calibrated",
|
| 8 |
+
"temperature": {
|
| 9 |
+
"family": 0.6492,
|
| 10 |
+
"modality": 0.6848
|
| 11 |
+
},
|
| 12 |
+
"note": "Softmax over a single-label head *can* be calibrated, which is the one place this task is easier to trust than OrganScan's per-class sigmoids. `make calibrate` fits one scalar per level on the validation split by minimising NLL and writes it back here, flipping status to `calibrated`. Until then a score of 0.9 is a ranking, not a probability.",
|
| 13 |
+
"fit_note": "Temperature is fitted on validation, never on test, and re-fitted after any re-export. INT8 quantization shifts logit scale slightly, so the FP32 temperature is not automatically correct for the shipped artifact.",
|
| 14 |
+
"fitted_on": {
|
| 15 |
+
"split": "validation",
|
| 16 |
+
"rows": 3101,
|
| 17 |
+
"backend": "onnx"
|
| 18 |
+
}
|
| 19 |
+
},
|
| 20 |
+
"policies": {
|
| 21 |
+
"strict": {
|
| 22 |
+
"modality": 0.3,
|
| 23 |
+
"group": 0.45,
|
| 24 |
+
"family": 0.3,
|
| 25 |
+
"intent": "Favour coverage. Use when a human reviews every result, or when an unrouted image is more expensive than a mis-routed one."
|
| 26 |
+
},
|
| 27 |
+
"balanced": {
|
| 28 |
+
"modality": 0.55,
|
| 29 |
+
"group": 0.7,
|
| 30 |
+
"family": 0.5,
|
| 31 |
+
"intent": "Default. The published per-class results in models/evaluation.json are measured at this policy."
|
| 32 |
+
},
|
| 33 |
+
"precision": {
|
| 34 |
+
"modality": 0.85,
|
| 35 |
+
"group": 0.92,
|
| 36 |
+
"family": 0.75,
|
| 37 |
+
"intent": "Favour precision. Use when a wrong modality downstream silently corrupts a pipeline - for example selecting a CT window for an MR image."
|
| 38 |
+
}
|
| 39 |
+
},
|
| 40 |
+
"per_class_overrides": {
|
| 41 |
+
"note": "Populated by scripts/tune_thresholds.py once a model exists. Empty until then; an override invented before measurement is a guess wearing a config file's clothes.",
|
| 42 |
+
"strict": {},
|
| 43 |
+
"balanced": {
|
| 44 |
+
"CT": 0.9,
|
| 45 |
+
"MR": 0.73,
|
| 46 |
+
"US": 0.68,
|
| 47 |
+
"DX": 0.18,
|
| 48 |
+
"OCT": 0.05,
|
| 49 |
+
"FUNDUS": 0.05,
|
| 50 |
+
"DERMOSCOPY": 0.08,
|
| 51 |
+
"HISTOPATHOLOGY": 0.95
|
| 52 |
+
},
|
| 53 |
+
"precision": {}
|
| 54 |
+
},
|
| 55 |
+
"cascade": {
|
| 56 |
+
"enabled": true,
|
| 57 |
+
"order": [
|
| 58 |
+
"modality",
|
| 59 |
+
"collapse_group",
|
| 60 |
+
"family"
|
| 61 |
+
],
|
| 62 |
+
"note": "The decode path. Try the fine label; if no class clears its threshold, sum the softmax mass over the winning class's collapse group and try the group threshold; if that fails, try the family; if that fails, abstain. Each step returns a strictly coarser but still-true answer, which is what a routing caller actually needs. Reported as `granularity` on every result so a caller never has to guess how specific the answer is.",
|
| 63 |
+
"granularity_values": [
|
| 64 |
+
"modality",
|
| 65 |
+
"group",
|
| 66 |
+
"family",
|
| 67 |
+
"none"
|
| 68 |
+
]
|
| 69 |
+
},
|
| 70 |
+
"ood": {
|
| 71 |
+
"enabled": true,
|
| 72 |
+
"method": "energy",
|
| 73 |
+
"formula": "energy = -logsumexp(modality_logits / T)",
|
| 74 |
+
"note": "Open-set rejection. The realistic failure mode is not confusing CT with MR, it is being handed a scanned consent form, a photograph of a monitor or a calibration phantom and answering confidently. Free-energy over the logits separates in-distribution from unseen input better than max-softmax, costs one logsumexp, and needs no extra head.",
|
| 75 |
+
"energy_threshold": {
|
| 76 |
+
"strict": -2.0,
|
| 77 |
+
"balanced": -3.031,
|
| 78 |
+
"precision": -6.0
|
| 79 |
+
},
|
| 80 |
+
"threshold_note": "Higher energy means less in-distribution, so a frame is rejected when energy > threshold. `strict` favours coverage and therefore rejects least. For reference, a uniform 20-class distribution has energy -log(20) = -3.0, so the balanced cut sits just below 'the model has no idea'. These are placeholders derived from FP32 logit scale; scripts/tune_thresholds.py --false-rejection sets them from the validation split at a fixed false-rejection rate, and they MUST be re-fitted after INT8 export because quantization shifts logit scale.",
|
| 81 |
+
"fallback_label": "MODALITY_INDETERMINATE",
|
| 82 |
+
"fallback_note": "A rejected frame reports indeterminate with `out_of_distribution: true`, never a low-scoring real class. The distinction matters: 'I have not seen anything like this' and 'this is probably a CT' need different downstream handling.",
|
| 83 |
+
"fitted": {
|
| 84 |
+
"split": "validation",
|
| 85 |
+
"rows": 3101,
|
| 86 |
+
"false_rejection_target": 0.02,
|
| 87 |
+
"measured_false_rejection": 0.02,
|
| 88 |
+
"measured_detection": null
|
| 89 |
+
}
|
| 90 |
+
},
|
| 91 |
+
"abstention": {
|
| 92 |
+
"modality_indeterminate": "MODALITY_INDETERMINATE",
|
| 93 |
+
"family_indeterminate": "FAMILY_INDETERMINATE",
|
| 94 |
+
"cascade": true,
|
| 95 |
+
"report_alternatives": 3,
|
| 96 |
+
"report_alternatives_note": "Top-k runners-up are always returned with their scores. For a confusable pair the second choice carries most of the information a reviewer needs."
|
| 97 |
+
},
|
| 98 |
+
"aggregation": {
|
| 99 |
+
"note": "The single highest-leverage thing in the serving path, and it is nearly free. Modality is constant across a DICOM series by construction, so voting over sampled frames turns an N% frame error rate into roughly N^k at the series level. Unlike OrganScan's loop aggregation there is no correctness subtlety here - there is exactly one right answer per series.",
|
| 100 |
+
"method": "confidence_weighted_vote",
|
| 101 |
+
"methods_available": [
|
| 102 |
+
"majority_vote",
|
| 103 |
+
"confidence_weighted_vote",
|
| 104 |
+
"mean_probability"
|
| 105 |
+
],
|
| 106 |
+
"min_frames": 2,
|
| 107 |
+
"min_supporting_fraction": 0.5,
|
| 108 |
+
"min_supporting_fraction_note": "The winning label must be the frame-level winner on at least this fraction of sampled frames, otherwise the series aggregate abstains rather than breaking a tie arbitrarily.",
|
| 109 |
+
"disagreement_warning_threshold": 0.25,
|
| 110 |
+
"disagreement_note": "When more than this fraction of frames disagree with the series winner, a `series_disagreement` warning is attached. In practice that means the series is mixed (a fused PET/CT, or a study with an embedded secondary capture) and deserves a look."
|
| 111 |
+
},
|
| 112 |
+
"policy_tags": {
|
| 113 |
+
"note": "Attached to each prediction so a caller can branch on behaviour without hard-coding label names.",
|
| 114 |
+
"GRAYSCALE_NATIVE": [
|
| 115 |
+
"CT",
|
| 116 |
+
"MR",
|
| 117 |
+
"CR",
|
| 118 |
+
"DX",
|
| 119 |
+
"MG",
|
| 120 |
+
"RF",
|
| 121 |
+
"XA",
|
| 122 |
+
"OCT"
|
| 123 |
+
],
|
| 124 |
+
"COLOUR_NATIVE": [
|
| 125 |
+
"FUNDUS",
|
| 126 |
+
"DERMOSCOPY",
|
| 127 |
+
"ENDOSCOPY",
|
| 128 |
+
"HISTOPATHOLOGY",
|
| 129 |
+
"MICROSCOPY",
|
| 130 |
+
"PHOTO"
|
| 131 |
+
],
|
| 132 |
+
"COLOUR_MAPPED": [
|
| 133 |
+
"PT",
|
| 134 |
+
"NM",
|
| 135 |
+
"RENDER_3D"
|
| 136 |
+
],
|
| 137 |
+
"COLOUR_OPTIONAL": [
|
| 138 |
+
"US"
|
| 139 |
+
],
|
| 140 |
+
"COLOUR_OPTIONAL_NOTE": "Ultrasound is the awkward one: B-mode is grayscale, colour Doppler is not. A caller that keys on 'grayscale means radiology' will get US wrong half the time.",
|
| 141 |
+
"IONISING": [
|
| 142 |
+
"CT",
|
| 143 |
+
"CR",
|
| 144 |
+
"DX",
|
| 145 |
+
"MG",
|
| 146 |
+
"RF",
|
| 147 |
+
"XA",
|
| 148 |
+
"PT",
|
| 149 |
+
"NM"
|
| 150 |
+
],
|
| 151 |
+
"NON_DIAGNOSTIC": [
|
| 152 |
+
"PHOTO",
|
| 153 |
+
"RENDER_3D",
|
| 154 |
+
"DOCUMENT",
|
| 155 |
+
"WAVEFORM"
|
| 156 |
+
],
|
| 157 |
+
"DERIVED": [
|
| 158 |
+
"RENDER_3D"
|
| 159 |
+
],
|
| 160 |
+
"PHI_RISK_HIGH": [
|
| 161 |
+
"PHOTO",
|
| 162 |
+
"DOCUMENT"
|
| 163 |
+
],
|
| 164 |
+
"PHI_RISK_HIGH_NOTE": "These classes are the ones most likely to carry burned-in identifiers in plain text. A pipeline that routes on modality should treat a PHI_RISK_HIGH verdict as a reason to quarantine, not to continue."
|
| 165 |
+
}
|
| 166 |
+
}
|