mishig's picture
download
raw
12.7 kB
"""RGB-only block perception and planar calibration for the arm experiment.
No simulator state is used. Pixel coordinates are (u, v), with u increasing
rightward and v downward. World coordinates are (x, y), in meters. A homography
is valid on its calibration plane: objects above it have perspective parallax.
Typical use::
detections = detect_blocks(rgb)
marker = detect_marker(rgb)
fiducials = detect_fiducials(rgb)
pixels = order_quad([d.centroid for d in fiducials])
H = fit_homography(pixels, known_xy_in_same_order)
xy = pixel_to_world(H, detections["red"].centroid)
OpenCV is needed only for image segmentation/contours. NumPy suffices for all
coordinate calibration functions. Images must be RGB, not OpenCV's default BGR.
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Iterable
import numpy as np
@dataclass(frozen=True)
class Detection:
color: str
centroid: np.ndarray
area: float
angle_rad: float
box: np.ndarray
size: tuple[float, float]
contour: np.ndarray
def as_dict(self) -> dict:
return {
"color": self.color,
"centroid": self.centroid.tolist(),
"area": self.area,
"angle_rad": self.angle_rad,
"box": self.box.tolist(),
"size": list(self.size),
}
def _cv2():
try:
import cv2
except ImportError as error:
raise ImportError(
"Install opencv-python-headless in the simulation uv environment."
) from error
return cv2
def color_masks(rgb: np.ndarray, *, clean: bool = True) -> dict[str, np.ndarray]:
"""Return uint8 masks for red/green/blue blocks, cyan dots, magenta tip.
Hue thresholds retain moderately shadowed surfaces while rejecting the
gray arm and table. Masks have values 0 and 255 and match the RGB image size.
"""
cv2 = _cv2()
rgb = np.asarray(rgb)
if rgb.ndim != 3 or rgb.shape[2] != 3 or rgb.dtype != np.uint8:
raise ValueError("Expected an H x W x 3 RGB uint8 image.")
hsv = cv2.cvtColor(np.ascontiguousarray(rgb), cv2.COLOR_RGB2HSV)
hue, sat, value = hsv[..., 0], hsv[..., 1], hsv[..., 2]
vivid = (sat >= 90) & (value >= 35)
ranges = {
"red": (hue <= 12) | (hue >= 175),
"green": (hue >= 40) & (hue < 85),
# The blue-gray backdrop can form an enclosed patch inside the arm
# silhouette. Saturation separates it from the vivid blue block.
"blue": (hue > 100) & (hue <= 132) & (sat >= 110),
"cyan": (hue >= 85) & (hue <= 100),
"magenta": (hue >= 135) & (hue < 175),
}
masks = {}
kernel = np.ones((3, 3), dtype=np.uint8)
for color, selected in ranges.items():
mask = (selected & vivid).astype(np.uint8) * 255
if clean:
# Fill highlights without erasing small calibration dots or tip.
mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, kernel)
masks[color] = mask
return masks
def detections_from_mask(
mask: np.ndarray,
color: str,
*,
min_area: float = 5.0,
reject_border: bool = True,
border_margin: int = 2,
max_area_fraction: float | None = 0.25,
) -> list[Detection]:
"""Find connected exterior silhouettes, largest first.
angle_rad is the image-plane long-axis angle in [-pi/2, pi/2). It is
ambiguous for nearly square silhouettes and is not a world yaw angle.
Remove border-connected and implausibly large components *before* finding
exterior contours. This matters when a similarly colored background wraps
around the scene: RETR_EXTERNAL alone would hide a valid block inside its
exterior outline, even though the block is a separate connected component.
Border rejection intentionally excludes objects cropped by the camera.
"""
cv2 = _cv2()
mask = np.ascontiguousarray(mask, dtype=np.uint8)
height, width = mask.shape
count, labels, stats, _ = cv2.connectedComponentsWithStats(mask, connectivity=8)
keep = np.zeros(count, dtype=bool)
for index in range(1, count):
x, y, component_width, component_height, pixel_area = stats[index]
if pixel_area < min_area:
continue
if max_area_fraction is not None and pixel_area > height * width * max_area_fraction:
continue
if reject_border and (
x <= border_margin
or y <= border_margin
or x + component_width >= width - border_margin
or y + component_height >= height - border_margin
):
continue
keep[index] = True
mask = keep[labels].astype(np.uint8) * 255
contours, _ = cv2.findContours(
mask,
cv2.RETR_EXTERNAL,
cv2.CHAIN_APPROX_SIMPLE,
)
result = []
for contour in contours:
area = float(cv2.contourArea(contour))
if area < min_area:
continue
moments = cv2.moments(contour)
if moments["m00"] <= 0:
continue
centroid = np.array(
[moments["m10"] / moments["m00"], moments["m01"] / moments["m00"]]
)
rect = cv2.minAreaRect(contour)
box = cv2.boxPoints(rect).astype(float)
edges = np.roll(box, -1, axis=0) - box
lengths = np.linalg.norm(edges, axis=1)
edge = edges[int(np.argmax(lengths))]
angle = float((np.arctan2(edge[1], edge[0]) + np.pi / 2) % np.pi - np.pi / 2)
result.append(
Detection(
color=color,
centroid=centroid,
area=area,
angle_rad=angle,
box=box,
size=(float(max(rect[1])), float(min(rect[1]))),
contour=contour.reshape(-1, 2).astype(float),
)
)
return sorted(result, key=lambda detection: detection.area, reverse=True)
def detect_blocks(rgb: np.ndarray, *, min_area: float = 20) -> dict[str, Detection | None]:
"""Return the largest detection (or None) for each block color."""
masks = color_masks(rgb)
result = {}
for color in ("red", "green", "blue"):
detections = detections_from_mask(masks[color], color, min_area=min_area)
result[color] = detections[0] if detections else None
return result
def detect_marker(rgb: np.ndarray, *, min_area: float = 3) -> Detection | None:
"""Find the largest magenta gripper-tip marker."""
mask = color_masks(rgb)["magenta"]
detections = detections_from_mask(mask, "magenta", min_area=min_area)
return detections[0] if detections else None
def detect_fiducials(
rgb: np.ndarray, *, min_area: float = 3, count: int | None = 4
) -> list[Detection]:
"""Find cyan calibration dots, returning the largest count components.
Output is sorted by area, not matched to world coordinates. The caller must
supply the correspondence explicitly, or use order_quad for four corners.
"""
mask = color_masks(rgb)["cyan"]
detections = detections_from_mask(mask, "cyan", min_area=min_area)
return detections if count is None else detections[:count]
def order_quad(points: Iterable) -> np.ndarray:
"""Order four roughly axis-aligned image corners TL, TR, BR, BL.
The upper two points are the top edge; the lower two are the bottom edge.
Use explicit correspondences for strongly rotated/perspective quadrilaterals.
"""
points = np.asarray(list(points), dtype=float)
if points.shape != (4, 2):
raise ValueError("Exactly four 2D corners are required.")
by_y = points[np.argsort(points[:, 1])]
top = by_y[:2][np.argsort(by_y[:2, 0])]
bottom = by_y[2:][np.argsort(by_y[2:, 0])]
return np.array([top[0], top[1], bottom[1], bottom[0]])
def _normalize_points(points: np.ndarray) -> tuple[np.ndarray, np.ndarray]:
center = points.mean(axis=0)
radius = np.linalg.norm(points - center, axis=1).mean()
if radius < 1e-12:
raise ValueError("Calibration points must be distinct.")
scale = np.sqrt(2) / radius
transform = np.array(
[[scale, 0, -scale * center[0]], [0, scale, -scale * center[1]], [0, 0, 1]]
)
return (points - center) * scale, transform
def fit_homography(pixel_xy: Iterable, world_xy: Iterable) -> np.ndarray:
"""Fit a 3 x 3 pixel-to-world homography from >=4 matched points.
Uses normalized DLT. Correspondence ordering must agree between inputs.
All world points must lie on one plane; non-collinear coverage is required.
"""
source = np.asarray(list(pixel_xy), dtype=float)
target = np.asarray(list(world_xy), dtype=float)
if source.shape != target.shape or source.ndim != 2 or source.shape[1] != 2 or len(source) < 4:
raise ValueError("Expected matching N x 2 arrays with at least four points.")
if not (np.isfinite(source).all() and np.isfinite(target).all()):
raise ValueError("Calibration points must be finite.")
src, source_transform = _normalize_points(source)
dst, target_transform = _normalize_points(target)
rows = []
for (u, v), (x, y) in zip(src, dst):
rows.append([-u, -v, -1, 0, 0, 0, x * u, x * v, x])
rows.append([0, 0, 0, -u, -v, -1, y * u, y * v, y])
matrix = np.asarray(rows)
if np.linalg.matrix_rank(matrix) < 8:
raise ValueError("Calibration layout is degenerate or collinear.")
_, _, vh = np.linalg.svd(matrix, full_matrices=True)
normalized_h = vh[-1].reshape(3, 3)
homography = np.linalg.inv(target_transform) @ normalized_h @ source_transform
norm = homography[2, 2]
if abs(norm) < 1e-12:
norm = np.linalg.norm(homography)
return homography / norm
def pixel_to_world(homography: np.ndarray, pixel_xy: Iterable) -> np.ndarray:
"""Map a single (u,v) or an N x 2 array to the calibrated world plane."""
points = np.asarray(pixel_xy, dtype=float)
single = points.shape == (2,)
points = np.atleast_2d(points)
if points.ndim != 2 or points.shape[1] != 2:
raise ValueError("Expected one 2D point or an N x 2 array.")
mapped = np.column_stack([points, np.ones(len(points))]) @ np.asarray(homography).T
if np.any(np.abs(mapped[:, 2]) < 1e-12):
raise ValueError("Point maps to infinity under this homography.")
mapped = mapped[:, :2] / mapped[:, 2, None]
return mapped[0] if single else mapped
def world_to_pixel(homography: np.ndarray, world_xy: Iterable) -> np.ndarray:
"""Inverse of pixel_to_world for points on the same calibrated plane."""
return pixel_to_world(np.linalg.inv(homography), world_xy)
def world_orientation(homography: np.ndarray, detection: Detection) -> float:
"""Return yaw of the detected silhouette's longest edge on the plane."""
box = pixel_to_world(homography, detection.box)
edges = np.roll(box, -1, axis=0) - box
edge = edges[int(np.argmax(np.linalg.norm(edges, axis=1)))]
return float((np.arctan2(edge[1], edge[0]) + np.pi / 2) % np.pi - np.pi / 2)
def annotate(rgb: np.ndarray, detections: dict[str, Detection | None]) -> np.ndarray:
"""Return an RGB image with outlines, centers, and labels for inspection."""
cv2 = _cv2()
output = np.asarray(rgb).copy()
for color, detection in detections.items():
if detection is None:
continue
centroid = tuple(np.round(detection.centroid).astype(int))
cv2.polylines(output, [np.round(detection.box).astype(np.int32)], True, (255, 255, 255), 1)
cv2.drawMarker(output, centroid, (255, 255, 255), cv2.MARKER_CROSS, 10, 1)
cv2.putText(output, color, (centroid[0] + 6, centroid[1] - 6), cv2.FONT_HERSHEY_SIMPLEX, 0.45, (255, 255, 255), 1, cv2.LINE_AA)
return output
if __name__ == "__main__":
import argparse
import json
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("image", help="Input PNG/JPEG image")
parser.add_argument("--annotated", help="Optional output annotated image path")
args = parser.parse_args()
cv2 = _cv2()
bgr = cv2.imread(args.image)
if bgr is None:
parser.error(f"Cannot read {args.image}")
rgb = cv2.cvtColor(bgr, cv2.COLOR_BGR2RGB)
blocks = detect_blocks(rgb)
marker = detect_marker(rgb)
print(json.dumps({
"blocks": {name: value.as_dict() if value else None for name, value in blocks.items()},
"marker": marker.as_dict() if marker else None,
"fiducials": [dot.as_dict() for dot in detect_fiducials(rgb)],
}, indent=2))
if args.annotated:
visual = annotate(rgb, {**blocks, "tip": marker})
cv2.imwrite(args.annotated, cv2.cvtColor(visual, cv2.COLOR_RGB2BGR))

Xet Storage Details

Size:
12.7 kB
·
Xet hash:
e8d4bee68bf687b150f970313afc13238bf8f4539979c3ec026c7a728479006d

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.