File size: 8,258 Bytes
8c67517
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
{
  "schema_version": "1.0.0",
  "note": "The single source of truth for pixel handling. scripts/build_frame_dataset.py, scripts/train.py, src/modalityscan/preprocess.py and the exported ONNX graph all read this file, and scripts/export_onnx.py copies it into models/ so the shipped artifact carries its own contract. If training and serving ever disagree about geometry or normalisation, the model silently loses several points and nothing errors.",

  "input_name": "pixel_values",
  "dtype": "float32",
  "dtype_note": "float32 in, even for the INT8 model. Static quantization inserts QuantizeLinear at the graph head; handing it uint8 is a silent accuracy loss.",
  "layout": "NCHW",
  "channels": 3,
  "channels_note": "Always 3, and never collapsed to 1. Colour presence is this model's single most reliable feature: it separates dermoscopy, endoscopy, pathology and colour Doppler from every grayscale radiology class for free. A grayscale-first pipeline throws that away.",
  "image_size": 224,
  "dynamic_axes": { "batch": 0 },
  "spatial_dims_fixed": true,

  "design_inversion": {
    "note": "READ THIS BEFORE COPYING ANYTHING FROM OrganScan. That pipeline is modality-*conditional* by design: it applies CT windowing after RescaleSlope/Intercept, crops ultrasound to SequenceOfUltrasoundRegions, and centre-crops to a square. Every one of those steps is either circular or destructive here.",
    "circular": "Choosing a CT window requires already knowing the image is a CT. A modality detector cannot consume a modality-conditional rendering without leaking its own answer at train time and failing at inference time.",
    "destructive_geometry": "Centre-cropping to 224 discards the field-of-view outline - the ultrasound sector, the mammography chest-wall edge, the circular fundus aperture, the endoscope vignette. Those outlines are among the strongest cues available, so geometry is letterbox-padded, not cropped.",
    "destructive_ui": "Cropping ultrasound to the scan region removes burned-in vendor UI. For OrganScan that crop is mandatory because the UI text reads the organ name. Here the UI is not a label leak - it does not name the modality - but it IS a dataset-identity leak, so it is kept in frame and attacked with border-biased random erasing during training instead."
  },

  "pipeline": [
    { "step": 1, "op": "decode", "detail": "PNG/JPEG for serving. For ingest: honour TransferSyntaxUID; sample DICOM multi-frame and volumes by stride, never frame-by-frame." },
    { "step": 2, "op": "photometric", "detail": "Invert when PhotometricInterpretation == MONOCHROME1. Ingest only - a served PNG has already been rendered." },
    { "step": 3, "op": "colour", "detail": "Keep RGB if the source is RGB. Replicate to 3 channels if it is single-channel. Never convert colour to grayscale." },
    { "step": 4, "op": "letterbox", "detail": "Pad the shorter side with `pad_value` to a square, then resize to 224. Aspect ratio and FOV outline survive." },
    { "step": 5, "op": "to_float", "detail": "Divide by 255 into [0, 1]." },
    { "step": 6, "op": "normalize", "detail": "Subtract mean, divide by std." },
    { "step": 7, "op": "layout", "detail": "HWC -> CHW, stack to [B, 3, 224, 224]." }
  ],

  "geometry": {
    "mode": "letterbox",
    "mode_note": "Not `shortest_side + center_crop`. See design_inversion.destructive_geometry.",
    "image_size": 224,
    "pad_value": 0,
    "pad_value_note": "Black. Most medical modalities already sit on a black background, so a black pad is the least informative padding available. A mid-grey pad would create a synthetic border that correlates with original aspect ratio, which is a dataset fingerprint.",
    "interpolation": "bilinear",
    "antialias": true
  },

  "rescale_factor": 0.00392156862745098,
  "image_mean": [0.485, 0.456, 0.406],
  "image_std": [0.229, 0.224, 0.225],
  "normalization_note": "ImageNet statistics, because the backbone is ImageNet-initialised. scripts/export_onnx.py re-reads the chosen timm config at export time and fails if it disagrees with these values rather than letting them drift.",

  "rendering": {
    "note": "INGEST ONLY. These settings turn raw DICOM/NIfTI intensities into the 8-bit PNGs in prepared-training-data/frames. They are NOT applied at serving time - a served PNG has already been rendered by whatever produced it, and re-stretching its histogram would destroy the very distribution the model was trained to read.",
    "mode": "robust_percentile",
    "mode_note": "One rendering rule for every modality. Deliberately worse than a per-modality window and deliberately not negotiable: a per-modality rendering makes the training corpus separable by rendering artefact rather than by image content, and the model scores 0.99 in-distribution and collapses on anything rendered by a different tool.",
    "clip_percentiles": [0.5, 99.5],
    "apply_rescale_slope_intercept": true,
    "apply_rescale_slope_intercept_note": "Applied because it is a pure affine correction recorded in the file, not a modality-specific choice. The percentile stretch that follows is invariant to it anyway; it is applied so that HU-based provenance checks remain meaningful.",
    "output_bit_depth": 8,
    "max_side": 512,
    "max_side_note": "Frames are stored at up to 512px and letterboxed to 224 at load time, so a later experiment at 256 or 320 does not require re-rendering the corpus."
  },

  "train_augmentation": {
    "note": "Tuned against dataset-shortcut learning, which is the dominant failure mode for this task. Every entry below is either defending a real cue or attacking a fake one.",

    "horizontal_flip": 0.5,
    "horizontal_flip_note": "ON, unlike OrganScan. There are no laterality labels here to destroy, and flipping is a cheap defence against a corpus where every chest film happens to be oriented the same way.",

    "vertical_flip": 0.0,
    "vertical_flip_note": "OFF, permanently. The ultrasound sector apex is at the top and the mammography chest wall is at one edge; flipping vertically invents frames that no scanner produces.",

    "random_resized_crop": { "enabled": true, "scale": [0.6, 1.0], "ratio": [0.9, 1.11] },
    "random_resized_crop_note": "Ratio held close to square on purpose. A wide ratio range distorts the FOV outline that letterboxing exists to preserve.",

    "rotation_degrees": 8,

    "brightness": 0.25,
    "contrast": 0.25,

    "saturation": 0.0,
    "saturation_note": "ZERO, permanently. Colour presence is the highest-signal feature in this task. Jittering saturation teaches the model to ignore it.",
    "hue": 0.0,
    "hue_note": "Zero. Pathology stain hue (H&E purple/pink) and fundus orange are class-defining, not nuisance.",
    "random_grayscale": 0.0,
    "random_grayscale_note": "Zero. Converting a dermoscopy image to grayscale produces a labelled example that is genuinely ambiguous.",

    "jpeg_compression": { "enabled": true, "probability": 0.35, "quality": [35, 95] },
    "jpeg_compression_note": "ON, and load-bearing. Real traffic is re-encoded, screenshotted and forwarded. Compression noise is also a dataset fingerprint: if every histopathology patch in training is a clean PNG and every CT is a JPEG, the model learns the encoder.",

    "downscale_upscale": { "enabled": true, "probability": 0.3, "factor": [2, 4] },
    "downscale_upscale_note": "Native resolution is the second-strongest dataset fingerprint after compression. MedMNIST frames are upsampled 28px thumbnails; without this, 'blurry' becomes a synonym for 'MedMNIST'.",

    "gaussian_noise_std": 0.02,

    "random_erasing": { "probability": 0.25, "border_bias": 0.7 },
    "random_erasing_note": "Higher than OrganScan and biased toward the frame border, where burned-in vendor UI, scale bars, laterality markers and institution banners live. Those overlays correlate almost perfectly with source dataset, so an unerased model learns the hospital instead of the physics.",

    "mixup": { "enabled": false, "alpha": 0.2 },
    "mixup_note": "OFF by default. Blending a CT with a fundus photograph produces an image whose true modality is neither, under a soft label that claims it is both. Revisit only if measured over-confidence survives temperature scaling."
  }
}