vectorsense commited on
Commit
8fc50f2
·
verified ·
1 Parent(s): 920b9c5

Upload modality-thresholds.json with huggingface_hub

Browse files
Files changed (1) hide show
  1. modality-thresholds.json +166 -0
modality-thresholds.json ADDED
@@ -0,0 +1,166 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": "1.0.0",
3
+ "taxonomy_version": "1.0.0",
4
+ "default_policy": "balanced",
5
+ "calibration": {
6
+ "method": "temperature_scaling",
7
+ "status": "calibrated",
8
+ "temperature": {
9
+ "family": 0.6492,
10
+ "modality": 0.6848
11
+ },
12
+ "note": "Softmax over a single-label head *can* be calibrated, which is the one place this task is easier to trust than OrganScan's per-class sigmoids. `make calibrate` fits one scalar per level on the validation split by minimising NLL and writes it back here, flipping status to `calibrated`. Until then a score of 0.9 is a ranking, not a probability.",
13
+ "fit_note": "Temperature is fitted on validation, never on test, and re-fitted after any re-export. INT8 quantization shifts logit scale slightly, so the FP32 temperature is not automatically correct for the shipped artifact.",
14
+ "fitted_on": {
15
+ "split": "validation",
16
+ "rows": 3101,
17
+ "backend": "onnx"
18
+ }
19
+ },
20
+ "policies": {
21
+ "strict": {
22
+ "modality": 0.3,
23
+ "group": 0.45,
24
+ "family": 0.3,
25
+ "intent": "Favour coverage. Use when a human reviews every result, or when an unrouted image is more expensive than a mis-routed one."
26
+ },
27
+ "balanced": {
28
+ "modality": 0.55,
29
+ "group": 0.7,
30
+ "family": 0.5,
31
+ "intent": "Default. The published per-class results in models/evaluation.json are measured at this policy."
32
+ },
33
+ "precision": {
34
+ "modality": 0.85,
35
+ "group": 0.92,
36
+ "family": 0.75,
37
+ "intent": "Favour precision. Use when a wrong modality downstream silently corrupts a pipeline - for example selecting a CT window for an MR image."
38
+ }
39
+ },
40
+ "per_class_overrides": {
41
+ "note": "Populated by scripts/tune_thresholds.py once a model exists. Empty until then; an override invented before measurement is a guess wearing a config file's clothes.",
42
+ "strict": {},
43
+ "balanced": {
44
+ "CT": 0.9,
45
+ "MR": 0.73,
46
+ "US": 0.68,
47
+ "DX": 0.18,
48
+ "OCT": 0.05,
49
+ "FUNDUS": 0.05,
50
+ "DERMOSCOPY": 0.08,
51
+ "HISTOPATHOLOGY": 0.95
52
+ },
53
+ "precision": {}
54
+ },
55
+ "cascade": {
56
+ "enabled": true,
57
+ "order": [
58
+ "modality",
59
+ "collapse_group",
60
+ "family"
61
+ ],
62
+ "note": "The decode path. Try the fine label; if no class clears its threshold, sum the softmax mass over the winning class's collapse group and try the group threshold; if that fails, try the family; if that fails, abstain. Each step returns a strictly coarser but still-true answer, which is what a routing caller actually needs. Reported as `granularity` on every result so a caller never has to guess how specific the answer is.",
63
+ "granularity_values": [
64
+ "modality",
65
+ "group",
66
+ "family",
67
+ "none"
68
+ ]
69
+ },
70
+ "ood": {
71
+ "enabled": true,
72
+ "method": "energy",
73
+ "formula": "energy = -logsumexp(modality_logits / T)",
74
+ "note": "Open-set rejection. The realistic failure mode is not confusing CT with MR, it is being handed a scanned consent form, a photograph of a monitor or a calibration phantom and answering confidently. Free-energy over the logits separates in-distribution from unseen input better than max-softmax, costs one logsumexp, and needs no extra head.",
75
+ "energy_threshold": {
76
+ "strict": -2.0,
77
+ "balanced": -3.031,
78
+ "precision": -6.0
79
+ },
80
+ "threshold_note": "Higher energy means less in-distribution, so a frame is rejected when energy > threshold. `strict` favours coverage and therefore rejects least. For reference, a uniform 20-class distribution has energy -log(20) = -3.0, so the balanced cut sits just below 'the model has no idea'. These are placeholders derived from FP32 logit scale; scripts/tune_thresholds.py --false-rejection sets them from the validation split at a fixed false-rejection rate, and they MUST be re-fitted after INT8 export because quantization shifts logit scale.",
81
+ "fallback_label": "MODALITY_INDETERMINATE",
82
+ "fallback_note": "A rejected frame reports indeterminate with `out_of_distribution: true`, never a low-scoring real class. The distinction matters: 'I have not seen anything like this' and 'this is probably a CT' need different downstream handling.",
83
+ "fitted": {
84
+ "split": "validation",
85
+ "rows": 3101,
86
+ "false_rejection_target": 0.02,
87
+ "measured_false_rejection": 0.02,
88
+ "measured_detection": null
89
+ }
90
+ },
91
+ "abstention": {
92
+ "modality_indeterminate": "MODALITY_INDETERMINATE",
93
+ "family_indeterminate": "FAMILY_INDETERMINATE",
94
+ "cascade": true,
95
+ "report_alternatives": 3,
96
+ "report_alternatives_note": "Top-k runners-up are always returned with their scores. For a confusable pair the second choice carries most of the information a reviewer needs."
97
+ },
98
+ "aggregation": {
99
+ "note": "The single highest-leverage thing in the serving path, and it is nearly free. Modality is constant across a DICOM series by construction, so voting over sampled frames turns an N% frame error rate into roughly N^k at the series level. Unlike OrganScan's loop aggregation there is no correctness subtlety here - there is exactly one right answer per series.",
100
+ "method": "confidence_weighted_vote",
101
+ "methods_available": [
102
+ "majority_vote",
103
+ "confidence_weighted_vote",
104
+ "mean_probability"
105
+ ],
106
+ "min_frames": 2,
107
+ "min_supporting_fraction": 0.5,
108
+ "min_supporting_fraction_note": "The winning label must be the frame-level winner on at least this fraction of sampled frames, otherwise the series aggregate abstains rather than breaking a tie arbitrarily.",
109
+ "disagreement_warning_threshold": 0.25,
110
+ "disagreement_note": "When more than this fraction of frames disagree with the series winner, a `series_disagreement` warning is attached. In practice that means the series is mixed (a fused PET/CT, or a study with an embedded secondary capture) and deserves a look."
111
+ },
112
+ "policy_tags": {
113
+ "note": "Attached to each prediction so a caller can branch on behaviour without hard-coding label names.",
114
+ "GRAYSCALE_NATIVE": [
115
+ "CT",
116
+ "MR",
117
+ "CR",
118
+ "DX",
119
+ "MG",
120
+ "RF",
121
+ "XA",
122
+ "OCT"
123
+ ],
124
+ "COLOUR_NATIVE": [
125
+ "FUNDUS",
126
+ "DERMOSCOPY",
127
+ "ENDOSCOPY",
128
+ "HISTOPATHOLOGY",
129
+ "MICROSCOPY",
130
+ "PHOTO"
131
+ ],
132
+ "COLOUR_MAPPED": [
133
+ "PT",
134
+ "NM",
135
+ "RENDER_3D"
136
+ ],
137
+ "COLOUR_OPTIONAL": [
138
+ "US"
139
+ ],
140
+ "COLOUR_OPTIONAL_NOTE": "Ultrasound is the awkward one: B-mode is grayscale, colour Doppler is not. A caller that keys on 'grayscale means radiology' will get US wrong half the time.",
141
+ "IONISING": [
142
+ "CT",
143
+ "CR",
144
+ "DX",
145
+ "MG",
146
+ "RF",
147
+ "XA",
148
+ "PT",
149
+ "NM"
150
+ ],
151
+ "NON_DIAGNOSTIC": [
152
+ "PHOTO",
153
+ "RENDER_3D",
154
+ "DOCUMENT",
155
+ "WAVEFORM"
156
+ ],
157
+ "DERIVED": [
158
+ "RENDER_3D"
159
+ ],
160
+ "PHI_RISK_HIGH": [
161
+ "PHOTO",
162
+ "DOCUMENT"
163
+ ],
164
+ "PHI_RISK_HIGH_NOTE": "These classes are the ones most likely to carry burned-in identifiers in plain text. A pipeline that routes on modality should treat a PHI_RISK_HIGH verdict as a reason to quarantine, not to continue."
165
+ }
166
+ }