#!/usr/bin/env python3 """Tests for benchmark.py metric logic (numpy + Pillow only). Run with ``python3 test_benchmark.py``. These pin the taxonomy projection, the ADE20K label mapping, the confusion/IoU math, table rendering, and the trivial floor baseline. """ import tempfile from pathlib import Path import numpy as np from PIL import Image import benchmark as bm from label_schema import IGNORE def test_reduce_semantic_projection(): mask = np.array([[0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 255]], dtype=np.uint8) reduced = bm.reduce_semantic(mask) assert reduced.tolist() == [[0, 1, 1, 1, 0, 0, 0, 0, 2, 0, IGNORE]], reduced.tolist() def test_ade_label_mapping(): assert bm.ade_label_to_reduced("wall") == 1 assert bm.ade_label_to_reduced("wall ") == 1 assert bm.ade_label_to_reduced("windowpane") == 2 assert bm.ade_label_to_reduced("window") == 2 assert bm.ade_label_to_reduced("door") == 0 assert bm.ade_label_to_reduced("sky") == 0 def test_build_label_lut(): lut = bm.build_label_lut({0: "wall", 1: "sky", 2: "windowpane", 3: "door"}) assert lut.tolist() == [1, 0, 2, 0], lut.tolist() def test_confusion_and_iou(): # confusion rows = ground truth, columns = prediction target = np.array([[0, 1, 2, 2]], dtype=np.uint8) pred = np.array([[0, 1, 2, 0]], dtype=np.uint8) cm = bm.confusion(pred, target, 3) assert int(cm[0, 0]) == 1 and int(cm[1, 1]) == 1 and int(cm[2, 2]) == 1 assert int(cm[2, 0]) == 1, "a missed window is a false negative (row window, col other)" assert int(cm[0, 2]) == 0 ious = bm.iou_from_confusion(cm) assert abs(ious[1] - 1.0) < 1e-9 # wall: perfect assert abs(ious[2] - 0.5) < 1e-9 # window: 1 true positive of 2 actual windows summary = bm.summarize_reduced(cm) assert summary["window_precision"] == 1.0 # no false-positive window pixels assert summary["window_recall"] == 0.5 assert 0.0 < summary["mIoU"] <= 1.0 def test_ignore_pixels_are_excluded(): target = np.array([[IGNORE, IGNORE, 2]], dtype=np.uint8) pred = np.array([[0, 0, 2]], dtype=np.uint8) cm = bm.confusion(pred, target, 3) assert int(cm.sum()) == 1, cm assert int(cm[2, 2]) == 1 def test_render_table_escapes_and_formats(): results = {"painting_vision": {"mIoU": 0.5, "iou": {"other": 0.9, "wall": 0.4, "window": 0.2}, "window_precision": 0.3, "window_recall": 0.25, "painted_coverage_mae": 0.04}, "floor:all-other": {"mIoU": 0.1, "iou": {"other": 0.3, "wall": None, "window": None}, "window_precision": None, "window_recall": None}} table = bm.render_table(results) assert table.startswith("| model |") assert "0.400" in table and "n/a" in table assert "coverage MAE 0.04" in table def _write_dataset(root): for split, value in (("train", 1), ("test", 2)): (root / split / "images").mkdir(parents=True) (root / split / "masks").mkdir(parents=True) Image.fromarray(np.zeros((6, 6, 3), np.uint8)).save(root / split / "images" / "a.jpg") mask = np.full((6, 6), 100, np.uint8) mask[:3, :] = 2 # wall mask[3:4, :] = 8 # window Image.fromarray(mask).save(root / split / "masks" / "a.png") def test_evaluate_floor_runs_without_torch(): with tempfile.TemporaryDirectory() as folder: root = Path(folder) _write_dataset(root) summary = bm.evaluate_floor(root, "test") assert summary["iou"]["wall"] == 0.0 # all-other predicts no wall assert summary["window_recall"] == 0.0 assert summary["iou"]["other"] is not None def main(): tests = [value for name, value in sorted(globals().items()) if name.startswith("test_")] for test in tests: test() print(f"ok {test.__name__}") print(f"{len(tests)} tests passed") if __name__ == "__main__": main()