Download scripts/analyze_dataset.py from siddhant20/task1: direct link, hf CLI and curl.
- Browser
- Download file 6.52 kB
-
https://huggingface.co/siddhant20/task1/resolve/main/scripts/analyze_dataset.py
- Command line
-
hf download hf://siddhant20/task1/scripts/analyze_dataset.py
-
curl -L -o analyze_dataset.py https://huggingface.co/siddhant20/task1/resolve/main/scripts/analyze_dataset.py
6.52 kB
| """Reproduce every dataset claim made in HANDOVER.md. | |
| PMDM_DATA=Task1/PackagingMaterialDifferenceMiningDataset \ | |
| .venv/bin/python scripts/analyze_dataset.py | |
| Runs in about two minutes on a laptop. Prints alignment residuals, box size | |
| distribution, change polarity, blur mismatch and the classical baseline floor. | |
| """ | |
| from __future__ import annotations | |
| import collections | |
| import sys | |
| from pathlib import Path | |
| sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) | |
| import cv2 # noqa: E402 | |
| import numpy as np # noqa: E402 | |
| from pmdm import config # noqa: E402 | |
| from pmdm.dataset import load_gt # noqa: E402 | |
| from pmdm.metric import match_image # noqa: E402 | |
| SAMPLE_ALIGN = 10 # pairs used for the block phase-correlation probe | |
| SAMPLE_BASELINE = 40 # pairs used for the classical baseline floor | |
| def pair_paths(idx: int): | |
| t = config.TRAIN_DIR / "template" / f"train_template_{idx:03d}.png" | |
| p = config.TRAIN_DIR / "photo" / f"train_photo_{idx:03d}.png" | |
| return t, p | |
| def section(title: str) -> None: | |
| print(f"\n{'=' * 70}\n{title}\n{'=' * 70}") | |
| def main() -> None: | |
| gt = load_gt() | |
| n_pairs = len(gt) | |
| section("1. Size and annotation counts") | |
| sizes = collections.Counter() | |
| same_size = 0 | |
| for i in range(config.N_TRAIN): | |
| t, p = pair_paths(i) | |
| ti = cv2.imread(str(t), cv2.IMREAD_COLOR) | |
| pi = cv2.imread(str(p), cv2.IMREAD_COLOR) | |
| sizes[ti.shape[:2]] += 1 | |
| same_size += int(ti.shape == pi.shape) | |
| counts = sorted(len(v) for v in gt.values()) | |
| print(f"annotated pairs: {n_pairs}, boxes: {sum(len(v) for v in gt.values())}") | |
| print(f"boxes per image: min {counts[0]}, median {counts[len(counts) // 2]}, max {counts[-1]}") | |
| print(f"template and photo same size: {same_size}/{config.N_TRAIN}") | |
| print(f"distinct sizes: {len(sizes)}; most common: {sizes.most_common(4)}") | |
| section("2. Alignment (is registration needed?)") | |
| for i in range(0, config.N_TRAIN, config.N_TRAIN // SAMPLE_ALIGN): | |
| t, p = pair_paths(i) | |
| T = cv2.imread(str(t), cv2.IMREAD_GRAYSCALE).astype(np.float32) | |
| P = cv2.imread(str(p), cv2.IMREAD_GRAYSCALE).astype(np.float32) | |
| h, w = T.shape | |
| bs, offs = 256, [] | |
| for y in range(0, h - bs, bs): | |
| for x in range(0, w - bs, bs): | |
| a, b = T[y:y + bs, x:x + bs], P[y:y + bs, x:x + bs] | |
| if a.std() < 5: | |
| continue | |
| (dx, dy), _ = cv2.phaseCorrelate(a.copy(), b.copy()) | |
| offs.append((abs(dx), abs(dy))) | |
| if offs: | |
| o = np.array(offs) | |
| print(f"pair {i:3d} {T.shape} blocks={len(o):3d} " | |
| f"|dx| p50/p95 {np.percentile(o[:, 0], [50, 95]).round(2)} " | |
| f"|dy| p50/p95 {np.percentile(o[:, 1], [50, 95]).round(2)}") | |
| section("3. Box size distribution") | |
| ws, hs, rel = [], [], [] | |
| for idx, boxes in gt.items(): | |
| t, _ = pair_paths(idx) | |
| H, W = cv2.imread(str(t), cv2.IMREAD_GRAYSCALE).shape | |
| for x1, y1, x2, y2 in boxes: | |
| ws.append(x2 - x1) | |
| hs.append(y2 - y1) | |
| rel.append(((x2 - x1) / W, (y2 - y1) / H)) | |
| ws, hs, rel = np.array(ws), np.array(hs), np.array(rel) | |
| print(f"width percentiles [1,25,50,75,99]: {np.percentile(ws, [1, 25, 50, 75, 99]).astype(int)}") | |
| print(f"height percentiles [1,25,50,75,99]: {np.percentile(hs, [1, 25, 50, 75, 99]).astype(int)}") | |
| print(f"relative width median: {np.median(rel[:, 0]) * 100:.2f}% of image width") | |
| print(f"boxes with both sides < 16 px: {int(((ws < 16) & (hs < 16)).sum())} / {len(ws)}") | |
| print(f"boxes exactly 8x8: {int(((ws == 8) & (hs == 8)).sum())}") | |
| section("4. Change polarity (additions vs deletions)") | |
| t_ink, p_ink = [], [] | |
| for idx, boxes in gt.items(): | |
| t, p = pair_paths(idx) | |
| T = cv2.imread(str(t), cv2.IMREAD_GRAYSCALE).astype(np.float32) | |
| P = cv2.imread(str(p), cv2.IMREAD_GRAYSCALE).astype(np.float32) | |
| nT = np.clip(cv2.GaussianBlur(T, (0, 0), 31) - T, 0, None) | |
| nP = np.clip(cv2.GaussianBlur(P, (0, 0), 31) - P, 0, None) | |
| for x1, y1, x2, y2 in boxes.astype(int): | |
| a, b = nT[y1:y2, x1:x2], nP[y1:y2, x1:x2] | |
| if a.size: | |
| t_ink.append(a.mean()) | |
| p_ink.append(b.mean()) | |
| t_ink, p_ink = np.array(t_ink), np.array(p_ink) | |
| print(f"boxes analysed: {len(t_ink)}") | |
| print(f"addition (template blank, photo inked): {int(((t_ink < 3) & (p_ink > 3)).sum())}") | |
| print(f"deletion (photo blank, template inked): {int(((p_ink < 3) & (t_ink > 3)).sum())}") | |
| print(f"modification (both inked): {int(((t_ink >= 3) & (p_ink >= 3)).sum())}") | |
| print(f"fraction with more ink in photo: {float((p_ink > t_ink).mean()):.3f}") | |
| section("5. Blur and shadow mismatch") | |
| shadow = 0 | |
| for i in range(0, config.N_TRAIN, 4): | |
| t, p = pair_paths(i) | |
| T = cv2.imread(str(t), cv2.IMREAD_GRAYSCALE) | |
| P = cv2.imread(str(p), cv2.IMREAD_GRAYSCALE) | |
| shadow += int(P.mean() < T.mean() - 20) | |
| if i < 12: | |
| print(f"pair {i:3d} Laplacian variance template {cv2.Laplacian(T, cv2.CV_64F).var():8.1f} " | |
| f"photo {cv2.Laplacian(P, cv2.CV_64F).var():8.1f}") | |
| print(f"shadow-heavy pairs (photo mean 20+ darker): {shadow} / {config.N_TRAIN // 4} sampled") | |
| section("6. Classical baseline floor") | |
| tp = fp = fn = 0 | |
| for idx in sorted(gt)[:SAMPLE_BASELINE]: | |
| t, p = pair_paths(idx) | |
| T = cv2.imread(str(t), cv2.IMREAD_GRAYSCALE).astype(np.float32) | |
| P = cv2.imread(str(p), cv2.IMREAD_GRAYSCALE).astype(np.float32) | |
| nT = T / (cv2.GaussianBlur(T, (0, 0), 25) + 1) | |
| nP = P / (cv2.GaussianBlur(P, (0, 0), 25) + 1) | |
| mask = (np.abs(nT - nP) > 0.18).astype(np.uint8) | |
| mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, np.ones((5, 5), np.uint8)) | |
| n, _, stats, _ = cv2.connectedComponentsWithStats(mask, 8) | |
| preds = np.array([[x, y, x + w, y + h] for x, y, w, h, a in stats[1:] if a >= 12], | |
| np.float32).reshape(-1, 4) | |
| a, b, c = match_image(preds, gt[idx]) | |
| tp, fp, fn = tp + a, fp + b, fn + c | |
| precision = tp / max(1, tp + fp) | |
| recall = tp / max(1, tp + fn) | |
| f1 = 2 * precision * recall / max(1e-9, precision + recall) | |
| print(f"{SAMPLE_BASELINE} pairs: TP {tp} FP {fp} FN {fn}") | |
| print(f"precision {precision:.4f} recall {recall:.4f} F1 {f1:.4f}") | |
| if __name__ == "__main__": | |
| main() | |