Upload preprocessor.json with huggingface_hub
Browse files- preprocessor.json +97 -0
preprocessor.json
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema_version": "1.0.0",
|
| 3 |
+
"note": "The single source of truth for pixel handling. scripts/build_frame_dataset.py, scripts/train.py, src/modalityscan/preprocess.py and the exported ONNX graph all read this file, and scripts/export_onnx.py copies it into models/ so the shipped artifact carries its own contract. If training and serving ever disagree about geometry or normalisation, the model silently loses several points and nothing errors.",
|
| 4 |
+
|
| 5 |
+
"input_name": "pixel_values",
|
| 6 |
+
"dtype": "float32",
|
| 7 |
+
"dtype_note": "float32 in, even for the INT8 model. Static quantization inserts QuantizeLinear at the graph head; handing it uint8 is a silent accuracy loss.",
|
| 8 |
+
"layout": "NCHW",
|
| 9 |
+
"channels": 3,
|
| 10 |
+
"channels_note": "Always 3, and never collapsed to 1. Colour presence is this model's single most reliable feature: it separates dermoscopy, endoscopy, pathology and colour Doppler from every grayscale radiology class for free. A grayscale-first pipeline throws that away.",
|
| 11 |
+
"image_size": 224,
|
| 12 |
+
"dynamic_axes": { "batch": 0 },
|
| 13 |
+
"spatial_dims_fixed": true,
|
| 14 |
+
|
| 15 |
+
"design_inversion": {
|
| 16 |
+
"note": "READ THIS BEFORE COPYING ANYTHING FROM OrganScan. That pipeline is modality-*conditional* by design: it applies CT windowing after RescaleSlope/Intercept, crops ultrasound to SequenceOfUltrasoundRegions, and centre-crops to a square. Every one of those steps is either circular or destructive here.",
|
| 17 |
+
"circular": "Choosing a CT window requires already knowing the image is a CT. A modality detector cannot consume a modality-conditional rendering without leaking its own answer at train time and failing at inference time.",
|
| 18 |
+
"destructive_geometry": "Centre-cropping to 224 discards the field-of-view outline - the ultrasound sector, the mammography chest-wall edge, the circular fundus aperture, the endoscope vignette. Those outlines are among the strongest cues available, so geometry is letterbox-padded, not cropped.",
|
| 19 |
+
"destructive_ui": "Cropping ultrasound to the scan region removes burned-in vendor UI. For OrganScan that crop is mandatory because the UI text reads the organ name. Here the UI is not a label leak - it does not name the modality - but it IS a dataset-identity leak, so it is kept in frame and attacked with border-biased random erasing during training instead."
|
| 20 |
+
},
|
| 21 |
+
|
| 22 |
+
"pipeline": [
|
| 23 |
+
{ "step": 1, "op": "decode", "detail": "PNG/JPEG for serving. For ingest: honour TransferSyntaxUID; sample DICOM multi-frame and volumes by stride, never frame-by-frame." },
|
| 24 |
+
{ "step": 2, "op": "photometric", "detail": "Invert when PhotometricInterpretation == MONOCHROME1. Ingest only - a served PNG has already been rendered." },
|
| 25 |
+
{ "step": 3, "op": "colour", "detail": "Keep RGB if the source is RGB. Replicate to 3 channels if it is single-channel. Never convert colour to grayscale." },
|
| 26 |
+
{ "step": 4, "op": "letterbox", "detail": "Pad the shorter side with `pad_value` to a square, then resize to 224. Aspect ratio and FOV outline survive." },
|
| 27 |
+
{ "step": 5, "op": "to_float", "detail": "Divide by 255 into [0, 1]." },
|
| 28 |
+
{ "step": 6, "op": "normalize", "detail": "Subtract mean, divide by std." },
|
| 29 |
+
{ "step": 7, "op": "layout", "detail": "HWC -> CHW, stack to [B, 3, 224, 224]." }
|
| 30 |
+
],
|
| 31 |
+
|
| 32 |
+
"geometry": {
|
| 33 |
+
"mode": "letterbox",
|
| 34 |
+
"mode_note": "Not `shortest_side + center_crop`. See design_inversion.destructive_geometry.",
|
| 35 |
+
"image_size": 224,
|
| 36 |
+
"pad_value": 0,
|
| 37 |
+
"pad_value_note": "Black. Most medical modalities already sit on a black background, so a black pad is the least informative padding available. A mid-grey pad would create a synthetic border that correlates with original aspect ratio, which is a dataset fingerprint.",
|
| 38 |
+
"interpolation": "bilinear",
|
| 39 |
+
"antialias": true
|
| 40 |
+
},
|
| 41 |
+
|
| 42 |
+
"rescale_factor": 0.00392156862745098,
|
| 43 |
+
"image_mean": [0.485, 0.456, 0.406],
|
| 44 |
+
"image_std": [0.229, 0.224, 0.225],
|
| 45 |
+
"normalization_note": "ImageNet statistics, because the backbone is ImageNet-initialised. scripts/export_onnx.py re-reads the chosen timm config at export time and fails if it disagrees with these values rather than letting them drift.",
|
| 46 |
+
|
| 47 |
+
"rendering": {
|
| 48 |
+
"note": "INGEST ONLY. These settings turn raw DICOM/NIfTI intensities into the 8-bit PNGs in prepared-training-data/frames. They are NOT applied at serving time - a served PNG has already been rendered by whatever produced it, and re-stretching its histogram would destroy the very distribution the model was trained to read.",
|
| 49 |
+
"mode": "robust_percentile",
|
| 50 |
+
"mode_note": "One rendering rule for every modality. Deliberately worse than a per-modality window and deliberately not negotiable: a per-modality rendering makes the training corpus separable by rendering artefact rather than by image content, and the model scores 0.99 in-distribution and collapses on anything rendered by a different tool.",
|
| 51 |
+
"clip_percentiles": [0.5, 99.5],
|
| 52 |
+
"apply_rescale_slope_intercept": true,
|
| 53 |
+
"apply_rescale_slope_intercept_note": "Applied because it is a pure affine correction recorded in the file, not a modality-specific choice. The percentile stretch that follows is invariant to it anyway; it is applied so that HU-based provenance checks remain meaningful.",
|
| 54 |
+
"output_bit_depth": 8,
|
| 55 |
+
"max_side": 512,
|
| 56 |
+
"max_side_note": "Frames are stored at up to 512px and letterboxed to 224 at load time, so a later experiment at 256 or 320 does not require re-rendering the corpus."
|
| 57 |
+
},
|
| 58 |
+
|
| 59 |
+
"train_augmentation": {
|
| 60 |
+
"note": "Tuned against dataset-shortcut learning, which is the dominant failure mode for this task. Every entry below is either defending a real cue or attacking a fake one.",
|
| 61 |
+
|
| 62 |
+
"horizontal_flip": 0.5,
|
| 63 |
+
"horizontal_flip_note": "ON, unlike OrganScan. There are no laterality labels here to destroy, and flipping is a cheap defence against a corpus where every chest film happens to be oriented the same way.",
|
| 64 |
+
|
| 65 |
+
"vertical_flip": 0.0,
|
| 66 |
+
"vertical_flip_note": "OFF, permanently. The ultrasound sector apex is at the top and the mammography chest wall is at one edge; flipping vertically invents frames that no scanner produces.",
|
| 67 |
+
|
| 68 |
+
"random_resized_crop": { "enabled": true, "scale": [0.6, 1.0], "ratio": [0.9, 1.11] },
|
| 69 |
+
"random_resized_crop_note": "Ratio held close to square on purpose. A wide ratio range distorts the FOV outline that letterboxing exists to preserve.",
|
| 70 |
+
|
| 71 |
+
"rotation_degrees": 8,
|
| 72 |
+
|
| 73 |
+
"brightness": 0.25,
|
| 74 |
+
"contrast": 0.25,
|
| 75 |
+
|
| 76 |
+
"saturation": 0.0,
|
| 77 |
+
"saturation_note": "ZERO, permanently. Colour presence is the highest-signal feature in this task. Jittering saturation teaches the model to ignore it.",
|
| 78 |
+
"hue": 0.0,
|
| 79 |
+
"hue_note": "Zero. Pathology stain hue (H&E purple/pink) and fundus orange are class-defining, not nuisance.",
|
| 80 |
+
"random_grayscale": 0.0,
|
| 81 |
+
"random_grayscale_note": "Zero. Converting a dermoscopy image to grayscale produces a labelled example that is genuinely ambiguous.",
|
| 82 |
+
|
| 83 |
+
"jpeg_compression": { "enabled": true, "probability": 0.35, "quality": [35, 95] },
|
| 84 |
+
"jpeg_compression_note": "ON, and load-bearing. Real traffic is re-encoded, screenshotted and forwarded. Compression noise is also a dataset fingerprint: if every histopathology patch in training is a clean PNG and every CT is a JPEG, the model learns the encoder.",
|
| 85 |
+
|
| 86 |
+
"downscale_upscale": { "enabled": true, "probability": 0.3, "factor": [2, 4] },
|
| 87 |
+
"downscale_upscale_note": "Native resolution is the second-strongest dataset fingerprint after compression. MedMNIST frames are upsampled 28px thumbnails; without this, 'blurry' becomes a synonym for 'MedMNIST'.",
|
| 88 |
+
|
| 89 |
+
"gaussian_noise_std": 0.02,
|
| 90 |
+
|
| 91 |
+
"random_erasing": { "probability": 0.25, "border_bias": 0.7 },
|
| 92 |
+
"random_erasing_note": "Higher than OrganScan and biased toward the frame border, where burned-in vendor UI, scale bars, laterality markers and institution banners live. Those overlays correlate almost perfectly with source dataset, so an unerased model learns the hospital instead of the physics.",
|
| 93 |
+
|
| 94 |
+
"mixup": { "enabled": false, "alpha": 0.2 },
|
| 95 |
+
"mixup_note": "OFF by default. Blending a CT with a fundus photograph produces an image whose true modality is neither, under a soft label that claims it is both. Revisit only if measured over-confidence survives temperature scaling."
|
| 96 |
+
}
|
| 97 |
+
}
|