constructelligence's picture
Upload benchmark.py with huggingface_hub
04c7da2 verified
Raw History Blame Contribute Delete
15.2 kB
#!/usr/bin/env python3
"""Benchmark Painting Vision against a generic segmentation baseline.
A generic model (e.g. SegFormer fine-tuned on ADE20K) can label `wall` and
`window`, but it cannot emit paint state (`painted`/`unpainted`/`uncertain`),
`skirting`, or `wall_obstacle`. So the only fair head-to-head is on the **shared
label subset**: ``{other, wall, window}``.
This script reduces both models to that subset and reports IoU per class and
mIoU, plus window precision/recall (the safety-critical class). For Painting
Vision it also reports the paint-state metrics a baseline cannot produce:
painted-coverage MAE and full ten-class IoU.
Everything here is optional and degrades gracefully:
* Painting Vision row needs a trained ``--checkpoint`` (torch).
* ``--baseline segformer-ade`` needs ``transformers`` and network (downloads the
ADE20K model); ``--baseline deeplabv3-voc`` uses torchvision only.
* A trivial ``all-other`` floor is always computed from the ground truth with
numpy + Pillow, so the script always emits a table.
Example::
python3 benchmark.py --data real_data --checkpoint artifacts/best.pt \
--baseline segformer-ade --split test --report benchmark.json --table benchmark.md
"""
from __future__ import annotations
import argparse
import json
import os
from pathlib import Path
import numpy as np
from PIL import Image
from label_schema import CLASSES, IGNORE
from provenance import sha256_file
# Public baselines must load without credentials. A stale/invalid HF token in the
# environment makes huggingface_hub return 401 for public repos, so disable the
# implicit token outright (and pass token=False at the call sites).
os.environ.setdefault("HF_HUB_DISABLE_IMPLICIT_TOKEN", "1")
BENCHMARK_VERSION = "1.0"
TAXONOMY_VERSION = "1.0"
REDUCED_CLASSES = ("other", "wall", "window")
IMAGE_EXTENSIONS = {".jpg", ".jpeg", ".png"}
# nvidia publishes no SegFormer-b5 ADE20K checkpoint at 512; b5 exists only at
# 640 (that family stops at b4 for the 512 input size). Use b4 here: the largest
# 512-input ADE20K model, so the baseline is still stronger than our b2 encoder.
ADE_BASELINE = "nvidia/segformer-b4-finetuned-ade-512-512"
VOC_BASELINE = "deeplabv3_resnet50"
PAINT_STATE_CLASSES = ("wall_painted", "wall_unpainted")
def reduce_semantic(mask, ignore=IGNORE):
"""Project the ten-class taxonomy onto {0=other, 1=wall, 2=window}."""
mask = np.asarray(mask)
reduced = np.zeros(mask.shape, dtype=np.uint8)
reduced[np.isin(mask, (1, 2, 3))] = 1 # wall_unpainted / wall_painted / wall_uncertain
reduced[mask == 8] = 2 # window
if ignore is not None:
reduced[mask == ignore] = ignore
return reduced
def ade_label_to_reduced(name):
"""Map an ADE20K class name to the shared subset."""
text = str(name).strip().lower()
if text == "wall":
return 1
if text.startswith("window"):
return 2
return 0
def build_label_lut(id_to_label):
"""Lookup table from a baseline's integer labels to the shared subset."""
size = max(int(k) for k in id_to_label) + 1
lut = np.zeros(size, dtype=np.uint8)
for index, name in id_to_label.items():
lut[int(index)] = ade_label_to_reduced(name)
return lut
def confusion(pred, target, classes, ignore=IGNORE):
pred, target = np.asarray(pred).ravel(), np.asarray(target).ravel()
valid = (target >= 0) & (target < classes) & (target != ignore)
encoded = classes * target[valid].astype(np.int64) + np.clip(pred[valid], 0, classes - 1).astype(np.int64)
return np.bincount(encoded, minlength=classes * classes).reshape(classes, classes)
def iou_from_confusion(cm):
cm = cm.astype(np.float64)
intersection = np.diag(cm)
union = cm.sum(0) + cm.sum(1) - intersection
return [None if union[i] == 0 else float(intersection[i] / union[i]) for i in range(len(union))]
def mean_iou(ious):
present = [v for v in ious if v is not None]
return float(np.mean(present)) if present else 0.0
def binary_pr(cm, index):
true_positive = float(cm[index, index])
precision = true_positive / cm[:, index].sum() if cm[:, index].sum() else None
recall = true_positive / cm[index, :].sum() if cm[index, :].sum() else None
return (round(precision, 4) if precision is not None else None,
round(recall, 4) if recall is not None else None)
def summarize_reduced(cm):
ious = iou_from_confusion(cm)
precision, recall = binary_pr(cm, 2) # window
return {"iou": {name: (round(v, 4) if v is not None else None) for name, v in zip(REDUCED_CLASSES, ious)},
"mIoU": round(mean_iou(ious), 4), "window_precision": precision, "window_recall": recall}
def iter_split(data, split, limit=None):
image_dir, mask_dir = Path(data) / split / "images", Path(data) / split / "masks"
if not image_dir.is_dir():
raise SystemExit(f"missing {image_dir}")
images = sorted(p for p in image_dir.iterdir() if p.suffix.lower() in IMAGE_EXTENSIONS)
for index, image_path in enumerate(images):
if limit is not None and index >= limit:
break
mask_path = next((mask_dir / (image_path.stem + s) for s in (".png", ".jpg", ".jpeg")
if (mask_dir / (image_path.stem + s)).is_file()), None)
if mask_path is not None:
yield image_path, mask_path
def evaluate_floor(data, split, limit=None):
"""Trivial floor: predict `other` everywhere. numpy + Pillow only."""
cm = np.zeros((3, 3), dtype=np.int64)
for _, mask_path in iter_split(data, split, limit):
target = reduce_semantic(np.asarray(Image.open(mask_path)))
cm += confusion(np.zeros_like(target), target, 3)
return summarize_reduced(cm)
def evaluate_painting_vision(checkpoint_path, data, split, size, limit, device):
import torch
from torch.utils.data import DataLoader
from label_schema import mask_to_labels
from models import DEFAULT_ARCH, build_from_checkpoint
from train import MEAN, STD, WallDataset, letterbox
checkpoint = torch.load(checkpoint_path, map_location=device, weights_only=True)
arch = str(checkpoint.get("arch", DEFAULT_ARCH))
model = build_from_checkpoint(checkpoint, len(CLASSES))
model.to(device).eval()
dataset = WallDataset(data, split, size)
loader = DataLoader(dataset, batch_size=4, num_workers=0)
full_cm = np.zeros((len(CLASSES), len(CLASSES)), dtype=np.int64)
reduced_cm = np.zeros((3, 3), dtype=np.int64)
coverage_errors = []
count = 0
with torch.inference_mode():
for x, y, *_ in loader:
prediction = model(x.to(device))["semantic"].argmax(1).cpu().numpy()
for p, t in zip(prediction, y.numpy()):
full_cm += confusion(p, t, len(CLASSES))
reduced_cm += confusion(reduce_semantic(p), reduce_semantic(t), 3)
known_wall = (t == 1) | (t == 2)
if known_wall.any():
gt_coverage = float((t[known_wall] == 2).mean())
pred_coverage = float((p[known_wall] == 2).mean())
coverage_errors.append(abs(pred_coverage - gt_coverage))
count += 1
ious = iou_from_confusion(full_cm)
summary = summarize_reduced(reduced_cm)
summary.update({
"arch": arch,
"images": count,
"mIoU_full_10class": round(mean_iou(ious), 4),
"per_class_iou_full": {name: (round(v, 4) if v is not None else None)
for name, v in zip(CLASSES, ious)},
"painted_coverage_mae": round(float(np.mean(coverage_errors)), 5) if coverage_errors else None,
})
return summary
def evaluate_baseline(name, data, split, limit, device):
import torch
from torch.nn import functional as F
if name == "segformer-ade":
from transformers import AutoImageProcessor, SegformerForSemanticSegmentation
processor = AutoImageProcessor.from_pretrained(ADE_BASELINE, token=False)
model = SegformerForSemanticSegmentation.from_pretrained(ADE_BASELINE, token=False).to(device).eval()
lut = build_label_lut(model.config.id2label)
def predict(image):
inputs = processor(images=image, return_tensors="pt").to(device)
with torch.inference_mode():
logits = model(**inputs).logits
upsampled = F.interpolate(logits, size=(image.height, image.width), mode="bilinear",
align_corners=False)
return lut[upsampled.argmax(1)[0].cpu().numpy()]
elif name == "deeplabv3-voc":
from torchvision.models.segmentation import DeepLabV3_ResNet50_Weights, deeplabv3_resnet50
weights = DeepLabV3_ResNet50_Weights.COCO_WITH_VOC_LABELS_V1
model = deeplabv3_resnet50(weights=weights).to(device).eval()
categories = weights.meta["categories"]
lut = build_label_lut({index: name for index, name in enumerate(categories)})
def predict(image):
array = np.asarray(image.resize((520, 520)), dtype=np.float32) / 255.0
array = (array - np.array([0.485, 0.456, 0.406])) / np.array([0.229, 0.224, 0.225])
tensor = torch.from_numpy(array.transpose(2, 0, 1).copy()).unsqueeze(0).to(device)
with torch.inference_mode():
out = model(tensor)["out"]
upsampled = F.interpolate(out, size=(image.height, image.width), mode="bilinear", align_corners=False)
return lut[upsampled.argmax(1)[0].cpu().numpy()]
else:
raise SystemExit(f"unknown baseline {name!r}; use segformer-ade or deeplabv3-voc")
cm = np.zeros((3, 3), dtype=np.int64)
count = 0
for image_path, mask_path in iter_split(data, split, limit):
image = Image.open(image_path).convert("RGB")
target = reduce_semantic(np.asarray(Image.open(mask_path)))
cm += confusion(predict(image), target, 3)
count += 1
summary = summarize_reduced(cm)
summary["images"] = count
summary["baseline"] = name
return summary
def render_table(results):
lines = ["| model | mIoU (other/wall/window) | wall IoU | window IoU | other IoU | window P | window R | paint-state |",
"| --- | --- | --- | --- | --- | --- | --- | --- |"]
for name, summary in results.items():
iou = summary.get("iou", {})
paint = ("coverage MAE " + str(summary["painted_coverage_mae"])) if summary.get("painted_coverage_mae") is not None \
else "not supported"
fmt = lambda v: "n/a" if v is None else f"{v:.3f}"
lines.append(f"| {name} | {fmt(summary.get('mIoU'))} | {fmt(iou.get('wall'))} | {fmt(iou.get('window'))} | "
f"{fmt(iou.get('other'))} | {fmt(summary.get('window_precision'))} | "
f"{fmt(summary.get('window_recall'))} | {paint} |")
return "\n".join(lines)
def main():
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--data", type=Path, required=True)
parser.add_argument("--checkpoint", type=Path, help="trained Painting Vision checkpoint")
parser.add_argument("--split", default="test")
parser.add_argument("--size", type=int, default=640, help="inference size for Painting Vision")
parser.add_argument("--baseline", choices=("none", "segformer-ade", "deeplabv3-voc"), default="none")
parser.add_argument("--limit", type=int, help="cap images for a quick run")
parser.add_argument("--device", default="cuda")
parser.add_argument("--report", type=Path, help="write the JSON report")
parser.add_argument("--table", type=Path, help="write the markdown table")
parser.add_argument("--dataset-manifest", type=Path,
help="provenance manifest, to stamp the dataset_id into the submission")
parser.add_argument("--emit-submission", type=Path,
help="write a standard benchmark submission record for the Painting Vision row")
args = parser.parse_args()
results = {}
if args.checkpoint:
import torch
device = torch.device(args.device if torch.cuda.is_available() or args.device == "cpu" else "cpu")
results["painting_vision"] = evaluate_painting_vision(args.checkpoint, args.data, args.split,
args.size, args.limit, device)
else:
print("no --checkpoint given: skipping the Painting Vision row (train one first)\n")
if args.baseline != "none":
try:
import torch
device = torch.device(args.device if torch.cuda.is_available() or args.device == "cpu" else "cpu")
results[f"baseline:{args.baseline}"] = evaluate_baseline(args.baseline, args.data, args.split,
args.limit, device)
except Exception as exc: # an optional comparison must not invalidate our own result
results[f"baseline:{args.baseline}"] = {"error": f"{type(exc).__name__}: {exc}"}
print(f"baseline {args.baseline!r} could not run and was skipped: {type(exc).__name__}: {exc}\n")
results["floor:all-other"] = evaluate_floor(args.data, args.split, args.limit)
dataset_id = None
if args.dataset_manifest:
dataset_id = json.loads(args.dataset_manifest.read_text(encoding="utf-8")).get("dataset_id")
table = render_table(results)
report = {"benchmark_version": BENCHMARK_VERSION, "taxonomy_version": TAXONOMY_VERSION,
"dataset_id": dataset_id, "split": args.split, "shared_taxonomy": list(REDUCED_CLASSES),
"models": results, "table": table}
print(table)
if args.report:
args.report.parent.mkdir(parents=True, exist_ok=True)
args.report.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8")
if args.table:
args.table.parent.mkdir(parents=True, exist_ok=True)
args.table.write_text(table + "\n", encoding="utf-8")
if args.emit_submission:
if "painting_vision" not in results:
raise SystemExit("--emit-submission needs a --checkpoint (no Painting Vision row was produced)")
submission = {
"benchmark_version": BENCHMARK_VERSION,
"taxonomy_version": TAXONOMY_VERSION,
"dataset_id": dataset_id,
"split": args.split,
"checkpoint_sha256": sha256_file(args.checkpoint),
"shared_metrics": {k: results["painting_vision"][k] for k in ("mIoU", "iou", "window_precision",
"window_recall")},
"paint_state_metrics": {"painted_coverage_mae": results["painting_vision"].get("painted_coverage_mae"),
"mIoU_full_10class": results["painting_vision"].get("mIoU_full_10class")},
}
args.emit_submission.parent.mkdir(parents=True, exist_ok=True)
args.emit_submission.write_text(json.dumps(submission, indent=2) + "\n", encoding="utf-8")
print(f"\nsubmission: {args.emit_submission}")
if __name__ == "__main__":
main()