Image Segmentation
PyTorch
English
painting-vision-robotics-kit
semantic-segmentation
robotics
edge-ai
construction-ai
autonomous-painting
wall-painting-robot
paint-coverage-estimation
building-facade
drywall
skirting-detection
window-detection
lidar
depth-validation
deeplabv3
mobilenetv3
Eval Results (legacy)
Download benchmark.py from constructelligence/painting-vision-robotics-kit: direct link, hf CLI and curl.
- Browser
- Download file 15.2 kB
-
https://huggingface.co/constructelligence/painting-vision-robotics-kit/resolve/main/benchmark.py
- Command line
-
hf download hf://constructelligence/painting-vision-robotics-kit/benchmark.py
-
curl -L -o benchmark.py https://huggingface.co/constructelligence/painting-vision-robotics-kit/resolve/main/benchmark.py
15.2 kB
| #!/usr/bin/env python3 | |
| """Benchmark Painting Vision against a generic segmentation baseline. | |
| A generic model (e.g. SegFormer fine-tuned on ADE20K) can label `wall` and | |
| `window`, but it cannot emit paint state (`painted`/`unpainted`/`uncertain`), | |
| `skirting`, or `wall_obstacle`. So the only fair head-to-head is on the **shared | |
| label subset**: ``{other, wall, window}``. | |
| This script reduces both models to that subset and reports IoU per class and | |
| mIoU, plus window precision/recall (the safety-critical class). For Painting | |
| Vision it also reports the paint-state metrics a baseline cannot produce: | |
| painted-coverage MAE and full ten-class IoU. | |
| Everything here is optional and degrades gracefully: | |
| * Painting Vision row needs a trained ``--checkpoint`` (torch). | |
| * ``--baseline segformer-ade`` needs ``transformers`` and network (downloads the | |
| ADE20K model); ``--baseline deeplabv3-voc`` uses torchvision only. | |
| * A trivial ``all-other`` floor is always computed from the ground truth with | |
| numpy + Pillow, so the script always emits a table. | |
| Example:: | |
| python3 benchmark.py --data real_data --checkpoint artifacts/best.pt \ | |
| --baseline segformer-ade --split test --report benchmark.json --table benchmark.md | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| import os | |
| from pathlib import Path | |
| import numpy as np | |
| from PIL import Image | |
| from label_schema import CLASSES, IGNORE | |
| from provenance import sha256_file | |
| # Public baselines must load without credentials. A stale/invalid HF token in the | |
| # environment makes huggingface_hub return 401 for public repos, so disable the | |
| # implicit token outright (and pass token=False at the call sites). | |
| os.environ.setdefault("HF_HUB_DISABLE_IMPLICIT_TOKEN", "1") | |
| BENCHMARK_VERSION = "1.0" | |
| TAXONOMY_VERSION = "1.0" | |
| REDUCED_CLASSES = ("other", "wall", "window") | |
| IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png"} | |
| # nvidia publishes no SegFormer-b5 ADE20K checkpoint at 512; b5 exists only at | |
| # 640 (that family stops at b4 for the 512 input size). Use b4 here: the largest | |
| # 512-input ADE20K model, so the baseline is still stronger than our b2 encoder. | |
| ADE_BASELINE = "nvidia/segformer-b4-finetuned-ade-512-512" | |
| VOC_BASELINE = "deeplabv3_resnet50" | |
| PAINT_STATE_CLASSES = ("wall_painted", "wall_unpainted") | |
| def reduce_semantic(mask, ignore=IGNORE): | |
| """Project the ten-class taxonomy onto {0=other, 1=wall, 2=window}.""" | |
| mask = np.asarray(mask) | |
| reduced = np.zeros(mask.shape, dtype=np.uint8) | |
| reduced[np.isin(mask, (1, 2, 3))] = 1 # wall_unpainted / wall_painted / wall_uncertain | |
| reduced[mask == 8] = 2 # window | |
| if ignore is not None: | |
| reduced[mask == ignore] = ignore | |
| return reduced | |
| def ade_label_to_reduced(name): | |
| """Map an ADE20K class name to the shared subset.""" | |
| text = str(name).strip().lower() | |
| if text == "wall": | |
| return 1 | |
| if text.startswith("window"): | |
| return 2 | |
| return 0 | |
| def build_label_lut(id_to_label): | |
| """Lookup table from a baseline's integer labels to the shared subset.""" | |
| size = max(int(k) for k in id_to_label) + 1 | |
| lut = np.zeros(size, dtype=np.uint8) | |
| for index, name in id_to_label.items(): | |
| lut[int(index)] = ade_label_to_reduced(name) | |
| return lut | |
| def confusion(pred, target, classes, ignore=IGNORE): | |
| pred, target = np.asarray(pred).ravel(), np.asarray(target).ravel() | |
| valid = (target >= 0) & (target < classes) & (target != ignore) | |
| encoded = classes * target[valid].astype(np.int64) + np.clip(pred[valid], 0, classes - 1).astype(np.int64) | |
| return np.bincount(encoded, minlength=classes * classes).reshape(classes, classes) | |
| def iou_from_confusion(cm): | |
| cm = cm.astype(np.float64) | |
| intersection = np.diag(cm) | |
| union = cm.sum(0) + cm.sum(1) - intersection | |
| return [None if union[i] == 0 else float(intersection[i] / union[i]) for i in range(len(union))] | |
| def mean_iou(ious): | |
| present = [v for v in ious if v is not None] | |
| return float(np.mean(present)) if present else 0.0 | |
| def binary_pr(cm, index): | |
| true_positive = float(cm[index, index]) | |
| precision = true_positive / cm[:, index].sum() if cm[:, index].sum() else None | |
| recall = true_positive / cm[index, :].sum() if cm[index, :].sum() else None | |
| return (round(precision, 4) if precision is not None else None, | |
| round(recall, 4) if recall is not None else None) | |
| def summarize_reduced(cm): | |
| ious = iou_from_confusion(cm) | |
| precision, recall = binary_pr(cm, 2) # window | |
| return {"iou": {name: (round(v, 4) if v is not None else None) for name, v in zip(REDUCED_CLASSES, ious)}, | |
| "mIoU": round(mean_iou(ious), 4), "window_precision": precision, "window_recall": recall} | |
| def iter_split(data, split, limit=None): | |
| image_dir, mask_dir = Path(data) / split / "images", Path(data) / split / "masks" | |
| if not image_dir.is_dir(): | |
| raise SystemExit(f"missing {image_dir}") | |
| images = sorted(p for p in image_dir.iterdir() if p.suffix.lower() in IMAGE_EXTENSIONS) | |
| for index, image_path in enumerate(images): | |
| if limit is not None and index >= limit: | |
| break | |
| mask_path = next((mask_dir / (image_path.stem + s) for s in (".png", ".jpg", ".jpeg") | |
| if (mask_dir / (image_path.stem + s)).is_file()), None) | |
| if mask_path is not None: | |
| yield image_path, mask_path | |
| def evaluate_floor(data, split, limit=None): | |
| """Trivial floor: predict `other` everywhere. numpy + Pillow only.""" | |
| cm = np.zeros((3, 3), dtype=np.int64) | |
| for _, mask_path in iter_split(data, split, limit): | |
| target = reduce_semantic(np.asarray(Image.open(mask_path))) | |
| cm += confusion(np.zeros_like(target), target, 3) | |
| return summarize_reduced(cm) | |
| def evaluate_painting_vision(checkpoint_path, data, split, size, limit, device): | |
| import torch | |
| from torch.utils.data import DataLoader | |
| from label_schema import mask_to_labels | |
| from models import DEFAULT_ARCH, build_from_checkpoint | |
| from train import MEAN, STD, WallDataset, letterbox | |
| checkpoint = torch.load(checkpoint_path, map_location=device, weights_only=True) | |
| arch = str(checkpoint.get("arch", DEFAULT_ARCH)) | |
| model = build_from_checkpoint(checkpoint, len(CLASSES)) | |
| model.to(device).eval() | |
| dataset = WallDataset(data, split, size) | |
| loader = DataLoader(dataset, batch_size=4, num_workers=0) | |
| full_cm = np.zeros((len(CLASSES), len(CLASSES)), dtype=np.int64) | |
| reduced_cm = np.zeros((3, 3), dtype=np.int64) | |
| coverage_errors = [] | |
| count = 0 | |
| with torch.inference_mode(): | |
| for x, y, *_ in loader: | |
| prediction = model(x.to(device))["semantic"].argmax(1).cpu().numpy() | |
| for p, t in zip(prediction, y.numpy()): | |
| full_cm += confusion(p, t, len(CLASSES)) | |
| reduced_cm += confusion(reduce_semantic(p), reduce_semantic(t), 3) | |
| known_wall = (t == 1) | (t == 2) | |
| if known_wall.any(): | |
| gt_coverage = float((t[known_wall] == 2).mean()) | |
| pred_coverage = float((p[known_wall] == 2).mean()) | |
| coverage_errors.append(abs(pred_coverage - gt_coverage)) | |
| count += 1 | |
| ious = iou_from_confusion(full_cm) | |
| summary = summarize_reduced(reduced_cm) | |
| summary.update({ | |
| "arch": arch, | |
| "images": count, | |
| "mIoU_full_10class": round(mean_iou(ious), 4), | |
| "per_class_iou_full": {name: (round(v, 4) if v is not None else None) | |
| for name, v in zip(CLASSES, ious)}, | |
| "painted_coverage_mae": round(float(np.mean(coverage_errors)), 5) if coverage_errors else None, | |
| }) | |
| return summary | |
| def evaluate_baseline(name, data, split, limit, device): | |
| import torch | |
| from torch.nn import functional as F | |
| if name == "segformer-ade": | |
| from transformers import AutoImageProcessor, SegformerForSemanticSegmentation | |
| processor = AutoImageProcessor.from_pretrained(ADE_BASELINE, token=False) | |
| model = SegformerForSemanticSegmentation.from_pretrained(ADE_BASELINE, token=False).to(device).eval() | |
| lut = build_label_lut(model.config.id2label) | |
| def predict(image): | |
| inputs = processor(images=image, return_tensors="pt").to(device) | |
| with torch.inference_mode(): | |
| logits = model(**inputs).logits | |
| upsampled = F.interpolate(logits, size=(image.height, image.width), mode="bilinear", | |
| align_corners=False) | |
| return lut[upsampled.argmax(1)[0].cpu().numpy()] | |
| elif name == "deeplabv3-voc": | |
| from torchvision.models.segmentation import DeepLabV3_ResNet50_Weights, deeplabv3_resnet50 | |
| weights = DeepLabV3_ResNet50_Weights.COCO_WITH_VOC_LABELS_V1 | |
| model = deeplabv3_resnet50(weights=weights).to(device).eval() | |
| categories = weights.meta["categories"] | |
| lut = build_label_lut({index: name for index, name in enumerate(categories)}) | |
| def predict(image): | |
| array = np.asarray(image.resize((520, 520)), dtype=np.float32) / 255.0 | |
| array = (array - np.array([0.485, 0.456, 0.406])) / np.array([0.229, 0.224, 0.225]) | |
| tensor = torch.from_numpy(array.transpose(2, 0, 1).copy()).unsqueeze(0).to(device) | |
| with torch.inference_mode(): | |
| out = model(tensor)["out"] | |
| upsampled = F.interpolate(out, size=(image.height, image.width), mode="bilinear", align_corners=False) | |
| return lut[upsampled.argmax(1)[0].cpu().numpy()] | |
| else: | |
| raise SystemExit(f"unknown baseline {name!r}; use segformer-ade or deeplabv3-voc") | |
| cm = np.zeros((3, 3), dtype=np.int64) | |
| count = 0 | |
| for image_path, mask_path in iter_split(data, split, limit): | |
| image = Image.open(image_path).convert("RGB") | |
| target = reduce_semantic(np.asarray(Image.open(mask_path))) | |
| cm += confusion(predict(image), target, 3) | |
| count += 1 | |
| summary = summarize_reduced(cm) | |
| summary["images"] = count | |
| summary["baseline"] = name | |
| return summary | |
| def render_table(results): | |
| lines = ["| model | mIoU (other/wall/window) | wall IoU | window IoU | other IoU | window P | window R | paint-state |", | |
| "| --- | --- | --- | --- | --- | --- | --- | --- |"] | |
| for name, summary in results.items(): | |
| iou = summary.get("iou", {}) | |
| paint = ("coverage MAE " + str(summary["painted_coverage_mae"])) if summary.get("painted_coverage_mae") is not None \ | |
| else "not supported" | |
| fmt = lambda v: "n/a" if v is None else f"{v:.3f}" | |
| lines.append(f"| {name} | {fmt(summary.get('mIoU'))} | {fmt(iou.get('wall'))} | {fmt(iou.get('window'))} | " | |
| f"{fmt(iou.get('other'))} | {fmt(summary.get('window_precision'))} | " | |
| f"{fmt(summary.get('window_recall'))} | {paint} |") | |
| return "\n".join(lines) | |
| def main(): | |
| parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) | |
| parser.add_argument("--data", type=Path, required=True) | |
| parser.add_argument("--checkpoint", type=Path, help="trained Painting Vision checkpoint") | |
| parser.add_argument("--split", default="test") | |
| parser.add_argument("--size", type=int, default=640, help="inference size for Painting Vision") | |
| parser.add_argument("--baseline", choices=("none", "segformer-ade", "deeplabv3-voc"), default="none") | |
| parser.add_argument("--limit", type=int, help="cap images for a quick run") | |
| parser.add_argument("--device", default="cuda") | |
| parser.add_argument("--report", type=Path, help="write the JSON report") | |
| parser.add_argument("--table", type=Path, help="write the markdown table") | |
| parser.add_argument("--dataset-manifest", type=Path, | |
| help="provenance manifest, to stamp the dataset_id into the submission") | |
| parser.add_argument("--emit-submission", type=Path, | |
| help="write a standard benchmark submission record for the Painting Vision row") | |
| args = parser.parse_args() | |
| results = {} | |
| if args.checkpoint: | |
| import torch | |
| device = torch.device(args.device if torch.cuda.is_available() or args.device == "cpu" else "cpu") | |
| results["painting_vision"] = evaluate_painting_vision(args.checkpoint, args.data, args.split, | |
| args.size, args.limit, device) | |
| else: | |
| print("no --checkpoint given: skipping the Painting Vision row (train one first)\n") | |
| if args.baseline != "none": | |
| try: | |
| import torch | |
| device = torch.device(args.device if torch.cuda.is_available() or args.device == "cpu" else "cpu") | |
| results[f"baseline:{args.baseline}"] = evaluate_baseline(args.baseline, args.data, args.split, | |
| args.limit, device) | |
| except Exception as exc: # an optional comparison must not invalidate our own result | |
| results[f"baseline:{args.baseline}"] = {"error": f"{type(exc).__name__}: {exc}"} | |
| print(f"baseline {args.baseline!r} could not run and was skipped: {type(exc).__name__}: {exc}\n") | |
| results["floor:all-other"] = evaluate_floor(args.data, args.split, args.limit) | |
| dataset_id = None | |
| if args.dataset_manifest: | |
| dataset_id = json.loads(args.dataset_manifest.read_text(encoding="utf-8")).get("dataset_id") | |
| table = render_table(results) | |
| report = {"benchmark_version": BENCHMARK_VERSION, "taxonomy_version": TAXONOMY_VERSION, | |
| "dataset_id": dataset_id, "split": args.split, "shared_taxonomy": list(REDUCED_CLASSES), | |
| "models": results, "table": table} | |
| print(table) | |
| if args.report: | |
| args.report.parent.mkdir(parents=True, exist_ok=True) | |
| args.report.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") | |
| if args.table: | |
| args.table.parent.mkdir(parents=True, exist_ok=True) | |
| args.table.write_text(table + "\n", encoding="utf-8") | |
| if args.emit_submission: | |
| if "painting_vision" not in results: | |
| raise SystemExit("--emit-submission needs a --checkpoint (no Painting Vision row was produced)") | |
| submission = { | |
| "benchmark_version": BENCHMARK_VERSION, | |
| "taxonomy_version": TAXONOMY_VERSION, | |
| "dataset_id": dataset_id, | |
| "split": args.split, | |
| "checkpoint_sha256": sha256_file(args.checkpoint), | |
| "shared_metrics": {k: results["painting_vision"][k] for k in ("mIoU", "iou", "window_precision", | |
| "window_recall")}, | |
| "paint_state_metrics": {"painted_coverage_mae": results["painting_vision"].get("painted_coverage_mae"), | |
| "mIoU_full_10class": results["painting_vision"].get("mIoU_full_10class")}, | |
| } | |
| args.emit_submission.parent.mkdir(parents=True, exist_ok=True) | |
| args.emit_submission.write_text(json.dumps(submission, indent=2) + "\n", encoding="utf-8") | |
| print(f"\nsubmission: {args.emit_submission}") | |
| if __name__ == "__main__": | |
| main() | |