painting-vision-robotics-kit / fetch_real_data.py
constructelligence's picture
Upload fetch_real_data.py with huggingface_hub
87dec67 verified
Raw History Blame Contribute Delete
9.59 kB
#!/usr/bin/env python3
"""Download a real labeled segmentation dataset and convert it to the painting layout.
Source: Kaggle ``venuarvind/facade-segmentation`` (Roboflow
``building-facade-segmentation-instance``, 598 real building-facade photographs
annotated in COCO instance-segmentation format). Facade and window polygons are
rasterised into the project's indexed semantic masks so ``audit_dataset.py``,
``train.py`` and ``predict.py`` can run on real data instead of synthetic
fixtures.
Taxonomy mapping (a documented proxy, not paint ground truth)
-------------------------------------------------------------
The source labels building elements, not paint state. We map:
* ``facade`` -> ``wall_painted`` (finished exterior wall surface)
* ``window`` -> ``window``
* every other category -> ``other``
The source has no ``wall_unpainted``/``skirting``/fixture labels, so those classes
stay absent and the audit will (correctly) report them as thin. The drywall
material head gets ``other_wall_material`` on wall pixels because these are
exterior facades, not gypsum board.
The converter re-splits by physical facade id: the published train/valid/test
split puts different crops of the *same* building in different splits, which the
dataset audit flags as leakage. Splitting on the facade-id prefix keeps every
crop of one building inside a single split.
Output layout matches train.py: OUT/{split}/{images,masks} plus
OUT/drywall_masks/{split} and OUT/metadata.csv, with OUT/SOURCES.md recording the
licence and these assumptions.
"""
import argparse
import csv
import json
import shutil
import subprocess
import sys
from pathlib import Path
import numpy as np
from PIL import Image, ImageDraw
from label_schema import CLASSES
KAGGLE_DATASET = "venuarvind/facade-segmentation"
CATEGORY_MAP = {"facade": CLASSES.index("wall_painted"), "window": CLASSES.index("window")}
# Drawn in this order so a window overwrites the facade that contains it.
DRAW_ORDER = ("facade", "window")
METADATA_COLUMNS = ("split", "image", "wall_id", "session_id", "environment", "surface_material",
"lighting", "weather", "surface_condition", "paint_stage", "camera_id",
"view_range", "view_angle_deg", "paint_id", "coat_index", "minutes_since_paint")
def ensure_dataset(cache):
cache = Path(cache)
existing = next(cache.glob("**/_annotations.coco.json"), None) if cache.is_dir() else None
if existing:
return existing.parent.parent
cache.mkdir(parents=True, exist_ok=True)
print(f"downloading {KAGGLE_DATASET} into {cache} ...", flush=True)
subprocess.run(["kaggle", "datasets", "download", "-d", KAGGLE_DATASET, "--unzip", "-p", str(cache)], check=True)
existing = next(cache.glob("**/_annotations.coco.json"), None)
if not existing:
raise SystemExit(f"no COCO annotations found under {cache}")
return existing.parent.parent
def facade_id(stem):
"""Physical facade id from a Roboflow filename stem.
Names look like ``20230329_200451_667_R_scaled_1_png_jpg.rf.<hash>``; the
timestamp/camera prefix before ``_scaled_`` identifies the building.
"""
return stem.split("_scaled_")[0]
def load_records(root):
"""All images across the published splits as (image_path, width, height, annotations)."""
records = []
for coco_path in sorted(root.glob("*/_annotations.coco.json")):
data = json.loads(coco_path.read_text(encoding="utf-8"))
names = {category["id"]: category["name"] for category in data["categories"]}
by_image = {}
for annotation in data["annotations"]:
by_image.setdefault(annotation["image_id"], []).append((names.get(annotation["category_id"]), annotation))
for image in data["images"]:
records.append((coco_path.parent / image["file_name"], image["width"], image["height"],
by_image.get(image["id"], [])))
return records
def rasterize(annotations, size):
mask = np.zeros((size[1], size[0]), dtype=np.uint8)
canvas = Image.fromarray(mask)
draw = ImageDraw.Draw(canvas)
for name in DRAW_ORDER:
target = CATEGORY_MAP[name]
for category_name, annotation in annotations:
if category_name != name:
continue
for polygon in annotation.get("segmentation") or []:
if len(polygon) >= 6:
draw.polygon(list(zip(polygon[0::2], polygon[1::2])), fill=target)
return np.asarray(canvas, dtype=np.uint8)
def split_facades(ids):
"""Deterministic facade-level 70/15/15 split, guaranteeing each split is non-empty."""
ids = sorted(ids)
total = len(ids)
if total < 3:
raise SystemExit("need at least 3 distinct facades to make train/val/test splits")
train_end = max(1, int(total * 0.70))
val_end = min(total - 1, max(train_end + 1, int(total * 0.85)))
assignment = {}
for index, facade in enumerate(ids):
assignment[facade] = "train" if index < train_end else "val" if index < val_end else "test"
return assignment
def main():
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--cache", type=Path, default=Path(".dataset_cache"),
help="download/extract cache for the raw Kaggle dataset")
parser.add_argument("--out", type=Path, default=Path("real_data"), help="converted dataset root")
parser.add_argument("--force", action="store_true", help="re-download and overwrite the output")
args = parser.parse_args()
if args.force and args.out.exists():
shutil.rmtree(args.out)
if args.out.exists():
raise SystemExit(f"{args.out} already exists; pass --force to overwrite")
root = ensure_dataset(args.cache)
records = load_records(root)
if not records:
raise SystemExit(f"no images found under {root}")
assignment = split_facades({facade_id(path.stem) for path, _, _, _ in records})
for split in ("train", "val", "test"):
(args.out / split / "images").mkdir(parents=True, exist_ok=True)
(args.out / split / "masks").mkdir(parents=True, exist_ok=True)
(args.out / "drywall_masks" / split).mkdir(parents=True, exist_ok=True)
rows, counts = [], {}
for image_path, width, height, annotations in records:
facade = facade_id(image_path.stem)
split = assignment[facade]
stem = image_path.stem
shutil.copy2(image_path, args.out / split / "images" / image_path.name)
mask = rasterize(annotations, (width, height))
Image.fromarray(mask).save(args.out / split / "masks" / f"{stem}.png")
# Exterior facades are not drywall: known wall material is "other", non-wall is unknown.
wall = np.isin(mask, [CLASSES.index(name) for name in ("wall_unpainted", "wall_painted", "wall_uncertain")])
material = np.where(wall, 0, 255).astype(np.uint8)
Image.fromarray(material).save(args.out / "drywall_masks" / split / f"{stem}.png")
rows.append({"split": split, "image": image_path.name, "wall_id": facade, "session_id": facade,
"environment": "outdoor", "surface_material": "unknown", "lighting": "unknown",
"weather": "unknown", "surface_condition": "dry", "paint_stage": "dry",
"camera_id": "roboflow-facade", "view_range": "full", "view_angle_deg": "0",
"paint_id": "n/a", "coat_index": "0", "minutes_since_paint": "0"})
counts[split] = counts.get(split, 0) + 1
with (args.out / "metadata.csv").open("w", newline="", encoding="utf-8") as stream:
writer = csv.DictWriter(stream, fieldnames=METADATA_COLUMNS)
writer.writeheader()
writer.writerows(sorted(rows, key=lambda row: (row["split"], row["image"])))
(args.out / "SOURCES.md").write_text(SOURCES.format(
facades=len(assignment), counts=", ".join(f"{k}={v}" for k, v in sorted(counts.items()))), encoding="utf-8")
print(json.dumps({"out": str(args.out), "images": len(rows), "facades": len(assignment), "per_split": counts}, indent=2))
print(f"converted {len(rows)} real images into {args.out}; run: python3 smoke_test.py --data {args.out}")
SOURCES = """# Real test data sources
## building-facade-segmentation (this dataset)
* Kaggle: https://www.kaggle.com/datasets/venuarvind/facade-segmentation
* Roboflow project: https://universe.roboflow.com/building-facade/building-facade-segmentation-instance
* Images: 598 real building-facade photographs, COCO instance-segmentation format.
* Licence: the Kaggle page lists Apache-2.0; the exported `README.dataset.txt` lists CC BY 4.0.
Both require attribution to the Roboflow project above. Verify the terms before redistribution.
### Conversion assumptions (this is a plumbing fixture, not paint ground truth)
* `facade` -> `wall_painted`, `window` -> `window`, all other categories -> `other`.
* No `wall_unpainted`, `skirting`, switch/outlet, AC, door, or `wall_obstacle` labels exist
in the source; those classes are absent and the audit reports them as thin.
* Drywall material mask is `other_wall_material` on wall pixels and `unknown` elsewhere
(exterior facades are not gypsum board).
* Splits are made by physical facade id, not the published split, because the published
split places different crops of the same building in different splits (leakage).
Generated by `fetch_real_data.py`. {facades} facades, per-split images: {counts}.
"""
if __name__ == "__main__":
main()