modalityscan / modality-thresholds.json
vectorsense's picture
Upload modality-thresholds.json with huggingface_hub
8fc50f2 verified
Raw History Blame Contribute Delete
7.31 kB
{
"schema_version": "1.0.0",
"taxonomy_version": "1.0.0",
"default_policy": "balanced",
"calibration": {
"method": "temperature_scaling",
"status": "calibrated",
"temperature": {
"family": 0.6492,
"modality": 0.6848
},
"note": "Softmax over a single-label head *can* be calibrated, which is the one place this task is easier to trust than OrganScan's per-class sigmoids. `make calibrate` fits one scalar per level on the validation split by minimising NLL and writes it back here, flipping status to `calibrated`. Until then a score of 0.9 is a ranking, not a probability.",
"fit_note": "Temperature is fitted on validation, never on test, and re-fitted after any re-export. INT8 quantization shifts logit scale slightly, so the FP32 temperature is not automatically correct for the shipped artifact.",
"fitted_on": {
"split": "validation",
"rows": 3101,
"backend": "onnx"
}
},
"policies": {
"strict": {
"modality": 0.3,
"group": 0.45,
"family": 0.3,
"intent": "Favour coverage. Use when a human reviews every result, or when an unrouted image is more expensive than a mis-routed one."
},
"balanced": {
"modality": 0.55,
"group": 0.7,
"family": 0.5,
"intent": "Default. The published per-class results in models/evaluation.json are measured at this policy."
},
"precision": {
"modality": 0.85,
"group": 0.92,
"family": 0.75,
"intent": "Favour precision. Use when a wrong modality downstream silently corrupts a pipeline - for example selecting a CT window for an MR image."
}
},
"per_class_overrides": {
"note": "Populated by scripts/tune_thresholds.py once a model exists. Empty until then; an override invented before measurement is a guess wearing a config file's clothes.",
"strict": {},
"balanced": {
"CT": 0.9,
"MR": 0.73,
"US": 0.68,
"DX": 0.18,
"OCT": 0.05,
"FUNDUS": 0.05,
"DERMOSCOPY": 0.08,
"HISTOPATHOLOGY": 0.95
},
"precision": {}
},
"cascade": {
"enabled": true,
"order": [
"modality",
"collapse_group",
"family"
],
"note": "The decode path. Try the fine label; if no class clears its threshold, sum the softmax mass over the winning class's collapse group and try the group threshold; if that fails, try the family; if that fails, abstain. Each step returns a strictly coarser but still-true answer, which is what a routing caller actually needs. Reported as `granularity` on every result so a caller never has to guess how specific the answer is.",
"granularity_values": [
"modality",
"group",
"family",
"none"
]
},
"ood": {
"enabled": true,
"method": "energy",
"formula": "energy = -logsumexp(modality_logits / T)",
"note": "Open-set rejection. The realistic failure mode is not confusing CT with MR, it is being handed a scanned consent form, a photograph of a monitor or a calibration phantom and answering confidently. Free-energy over the logits separates in-distribution from unseen input better than max-softmax, costs one logsumexp, and needs no extra head.",
"energy_threshold": {
"strict": -2.0,
"balanced": -3.031,
"precision": -6.0
},
"threshold_note": "Higher energy means less in-distribution, so a frame is rejected when energy > threshold. `strict` favours coverage and therefore rejects least. For reference, a uniform 20-class distribution has energy -log(20) = -3.0, so the balanced cut sits just below 'the model has no idea'. These are placeholders derived from FP32 logit scale; scripts/tune_thresholds.py --false-rejection sets them from the validation split at a fixed false-rejection rate, and they MUST be re-fitted after INT8 export because quantization shifts logit scale.",
"fallback_label": "MODALITY_INDETERMINATE",
"fallback_note": "A rejected frame reports indeterminate with `out_of_distribution: true`, never a low-scoring real class. The distinction matters: 'I have not seen anything like this' and 'this is probably a CT' need different downstream handling.",
"fitted": {
"split": "validation",
"rows": 3101,
"false_rejection_target": 0.02,
"measured_false_rejection": 0.02,
"measured_detection": null
}
},
"abstention": {
"modality_indeterminate": "MODALITY_INDETERMINATE",
"family_indeterminate": "FAMILY_INDETERMINATE",
"cascade": true,
"report_alternatives": 3,
"report_alternatives_note": "Top-k runners-up are always returned with their scores. For a confusable pair the second choice carries most of the information a reviewer needs."
},
"aggregation": {
"note": "The single highest-leverage thing in the serving path, and it is nearly free. Modality is constant across a DICOM series by construction, so voting over sampled frames turns an N% frame error rate into roughly N^k at the series level. Unlike OrganScan's loop aggregation there is no correctness subtlety here - there is exactly one right answer per series.",
"method": "confidence_weighted_vote",
"methods_available": [
"majority_vote",
"confidence_weighted_vote",
"mean_probability"
],
"min_frames": 2,
"min_supporting_fraction": 0.5,
"min_supporting_fraction_note": "The winning label must be the frame-level winner on at least this fraction of sampled frames, otherwise the series aggregate abstains rather than breaking a tie arbitrarily.",
"disagreement_warning_threshold": 0.25,
"disagreement_note": "When more than this fraction of frames disagree with the series winner, a `series_disagreement` warning is attached. In practice that means the series is mixed (a fused PET/CT, or a study with an embedded secondary capture) and deserves a look."
},
"policy_tags": {
"note": "Attached to each prediction so a caller can branch on behaviour without hard-coding label names.",
"GRAYSCALE_NATIVE": [
"CT",
"MR",
"CR",
"DX",
"MG",
"RF",
"XA",
"OCT"
],
"COLOUR_NATIVE": [
"FUNDUS",
"DERMOSCOPY",
"ENDOSCOPY",
"HISTOPATHOLOGY",
"MICROSCOPY",
"PHOTO"
],
"COLOUR_MAPPED": [
"PT",
"NM",
"RENDER_3D"
],
"COLOUR_OPTIONAL": [
"US"
],
"COLOUR_OPTIONAL_NOTE": "Ultrasound is the awkward one: B-mode is grayscale, colour Doppler is not. A caller that keys on 'grayscale means radiology' will get US wrong half the time.",
"IONISING": [
"CT",
"CR",
"DX",
"MG",
"RF",
"XA",
"PT",
"NM"
],
"NON_DIAGNOSTIC": [
"PHOTO",
"RENDER_3D",
"DOCUMENT",
"WAVEFORM"
],
"DERIVED": [
"RENDER_3D"
],
"PHI_RISK_HIGH": [
"PHOTO",
"DOCUMENT"
],
"PHI_RISK_HIGH_NOTE": "These classes are the ones most likely to carry burned-in identifiers in plain text. A pipeline that routes on modality should treat a PHI_RISK_HIGH verdict as a reason to quarantine, not to continue."
}
}