from __future__ import annotations import os import json import hashlib import io import platform import sys import threading import time import traceback import uuid from pathlib import Path for _stream in (sys.stdout, sys.stderr): if hasattr(_stream, "reconfigure"): _stream.reconfigure(encoding="utf-8", errors="replace") import gradio as gr from matplotlib import colormaps import numpy as np import torch import trimesh from PIL import Image, ImageDraw, ImageOps from transformers import pipeline GPU_DURATION_SECONDS = int(os.getenv("SIMAM3D_GPU_DURATION_SECONDS", "60")) try: import spaces GPU_FUNCTION = spaces.GPU(duration=GPU_DURATION_SECONDS) except (ImportError, AttributeError, TypeError): def GPU_FUNCTION(function): return function from simam3d_core import ( analyze_light_field, classify_scene_hypothesis, deterministic_demo_depth, classify_fusion_support, estimate_depth_light, filter_fusion_by_support, fuse_point_views, fusion_metrics, input_diversity_metrics, next_view_plan, normalize_prediction_contract, orbit_project_points, project_world_points, reprojection_consistency, reveal_uncertainty, synthetic_orbit_poses, write_confidence_ply, write_gaussian_ply, ) from da3_compat import resolve_da3_class from validate_manifest import validate_manifest MODEL_ID = os.getenv("SIMAM3D_DEPTH_MODEL", "depth-anything/Depth-Anything-V2-Small-hf") DA3_MODEL_ID = os.getenv("SIMAM3D_DA3_MODEL", "depth-anything/DA3NESTED-GIANT-LARGE-1.1") SIMAM3D_VERSION = "0.4.8" MANIFEST_SCHEMA_VERSION = 3 OUTPUT_DIR = Path("output") MAX_VIEWS = int(os.getenv("SIMAM3D_MAX_VIEWS", "8")) MAX_INPUT_SIDE = int(os.getenv("SIMAM3D_MAX_INPUT_SIDE", "1536")) FUSION_VOXEL_SIZE = float(os.getenv("SIMAM3D_FUSION_VOXEL_SIZE", "0.02")) FUSION_MAX_POINTS = int(os.getenv("SIMAM3D_FUSION_MAX_POINTS", "10000")) MIN_FUSION_SUPPORT_VIEWS = int(os.getenv("SIMAM3D_MIN_FUSION_SUPPORT_VIEWS", "1")) GAUSSIAN_MAX_POINTS = int(os.getenv("SIMAM3D_GAUSSIAN_MAX_POINTS", "2048")) DA3_PROCESS_RES = int(os.getenv("SIMAM3D_DA3_PROCESS_RES", "256")) DEPTH_PIPELINE = None DA3_MODEL = None PIPELINE_LOCK = threading.Lock() INFERENCE_LOCK = threading.Lock() def runtime_label() -> str: if torch.cuda.is_available(): return f"CUDA GPU ({torch.cuda.get_device_name(0)})" return "CPU" def get_depth_pipeline(): global DEPTH_PIPELINE if DEPTH_PIPELINE is not None: return DEPTH_PIPELINE with PIPELINE_LOCK: if DEPTH_PIPELINE is None: # Keep only one heavyweight backend resident when users switch # between DA3 and V2 in a long-lived Space process. clear_da3_model() device = 0 if torch.cuda.is_available() else -1 DEPTH_PIPELINE = pipeline("depth-estimation", model=MODEL_ID, device=device) return DEPTH_PIPELINE def get_da3_model(): global DA3_MODEL if DA3_MODEL is not None: return DA3_MODEL with PIPELINE_LOCK: if DA3_MODEL is None: clear_depth_pipeline() DepthAnything3, _ = load_da3_class() device = "cuda" if torch.cuda.is_available() else "cpu" DA3_MODEL = DepthAnything3.from_pretrained(DA3_MODEL_ID).to(device) DA3_MODEL.eval() return DA3_MODEL def clear_da3_model(): global DA3_MODEL DA3_MODEL = None if torch.cuda.is_available(): torch.cuda.empty_cache() def clear_depth_pipeline(): """Release the Transformers V2 pipeline before loading another backend.""" global DEPTH_PIPELINE DEPTH_PIPELINE = None if torch.cuda.is_available(): torch.cuda.empty_cache() def make_depth_preview(depth: np.ndarray) -> Image.Image: depth = (depth - depth.min()) / max(float(depth.max() - depth.min()), 1e-6) rgb = (colormaps.get_cmap("magma")(1.0 - depth)[..., :3] * 255).astype(np.uint8) return Image.fromarray(rgb) def make_confidence_preview(confidence: np.ndarray) -> Image.Image: confidence = np.asarray(confidence, dtype=np.float32) confidence = np.nan_to_num(confidence, nan=0.0, posinf=1.0, neginf=0.0) confidence = (confidence - confidence.min()) / max(float(confidence.max() - confidence.min()), 1e-6) rgb = (colormaps.get_cmap("viridis")(confidence)[..., :3] * 255).astype(np.uint8) return Image.fromarray(rgb) def make_uncertainty_preview(uncertainty: np.ndarray) -> Image.Image: uncertainty = np.clip(np.nan_to_num(np.asarray(uncertainty, dtype=np.float32)), 0.0, 1.0) rgb = (colormaps.get_cmap("inferno")(uncertainty)[..., :3] * 255).astype(np.uint8) return Image.fromarray(rgb) def make_evidence_card(image, depth, confidence, backend, view_count, fusion_stats, run_id): """Create a compact, shareable diagnostic card for each reconstruction run.""" tile_size = (480, 300) source = ImageOps.contain(image.convert("RGB"), tile_size) depth_tile = ImageOps.contain(make_depth_preview(depth).convert("RGB"), tile_size) confidence_tile = ImageOps.contain(make_confidence_preview(confidence).convert("RGB"), tile_size) canvas = Image.new("RGB", (960, 700), (12, 16, 28)) draw = ImageDraw.Draw(canvas) draw.text((24, 18), "SIMAM3D | EVIDENCE CARD", fill=(235, 240, 255)) draw.text( (24, 46), f"{backend} | {view_count} view(s) | {runtime_label()}", fill=(164, 178, 205), ) tiles = [(source, (0, 90), "SOURCE"), (depth_tile, (480, 90), "DEPTH"), (confidence_tile, (0, 390), "CONFIDENCE")] for tile, origin, label in tiles: x, y = origin canvas.paste(tile, (x, y)) draw.rectangle((x, y, x + tile.width, y + tile.height), outline=(76, 93, 126), width=2) draw.text((x + 12, y + 12), label, fill=(255, 255, 255)) draw.text( (500, 410), f"FUSED POINTS\n{fusion_stats['point_count']:,}\n\n" f"DOMINANT SOURCE VIEWS\n{fusion_stats['dominant_source_view_count']}\n\n" f"SUPPORTED SOURCE VIEWS\n{fusion_stats['supported_source_view_count']}\n\n" f"MULTI-VIEW VOXELS\n{fusion_stats['multi_view_voxel_fraction']:.1%}\n\n" f"CONFIDENCE MEAN\n{fusion_stats['confidence_mean']:.3f}\n\n" f"CONFIDENCE P95\n{fusion_stats['confidence_p95']:.3f}", fill=(210, 220, 240), spacing=8, ) OUTPUT_DIR.mkdir(exist_ok=True) path = OUTPUT_DIR / f"simam3d_{run_id}_evidence_card.png" canvas.save(path, optimize=True) return str(path) def make_reveal_card(image, depth, guess, lighting, run_id: str) -> str: """Create a social-friendly reveal card while clearly separating inference from fact.""" tile_size = (480, 300) source = ImageOps.contain(image.convert("RGB"), tile_size) depth_tile = ImageOps.contain(make_depth_preview(depth).convert("RGB"), tile_size) canvas = Image.new("RGB", (960, 760), (12, 16, 28)) draw = ImageDraw.Draw(canvas) draw.text((24, 18), "SIMAM3D | WHAT'S BEHIND THE IMAGE?", fill=(235, 240, 255)) draw.text((24, 48), "Depth and illumination reveal | hypothesis, not ground truth", fill=(164, 178, 205)) canvas.paste(source, (0, 90)) canvas.paste(depth_tile, (480, 90)) draw.rectangle((0, 90, 480, 390), outline=(76, 93, 126), width=2) draw.rectangle((480, 90, 960, 390), outline=(76, 93, 126), width=2) draw.text((14, 104), "ORIGINAL", fill=(255, 255, 255)) draw.text((494, 104), "DEPTH REVEAL", fill=(255, 255, 255)) guess_text = str(guess["user_guess"]) cues = ", ".join(str(item) for item in guess["semantic_cues"]) lighting_text = ( f"LIGHT FIELD: {lighting['lighting_label']}\n" f"Direction estimate: azimuth {lighting['estimated_azimuth_deg']:.0f} deg, " f"elevation {lighting['estimated_elevation_deg']:.0f} deg\n" f"Detector confidence: {lighting['detector_confidence']:.2f}" ) draw.text((24, 430), "YOUR GUESS", fill=(133, 214, 255)) draw.text((24, 456), guess_text[:110], fill=(235, 240, 255)) draw.text((24, 500), f"SEMANTIC CUES: {cues}", fill=(210, 220, 240)) draw.text((24, 560), lighting_text, fill=(210, 220, 240), spacing=8) draw.text((24, 690), "The unseen side is a research hypothesis; add views for evidence.", fill=(255, 190, 120)) OUTPUT_DIR.mkdir(exist_ok=True) path = OUTPUT_DIR / f"simam3d_{run_id}_reveal_card.png" canvas.save(path, optimize=True) return str(path) def depth_to_glb(image: Image.Image, depth: np.ndarray, run_id: str) -> str: image = image.convert("RGB") max_side = 192 scale = min(1.0, max_side / max(image.size)) size = (max(8, int(image.width * scale)), max(8, int(image.height * scale))) rgb = np.asarray(image.resize(size, Image.Resampling.LANCZOS)) d = Image.fromarray(depth).resize(size, Image.Resampling.BILINEAR) d = np.asarray(d, dtype=np.float32) d = (d - d.min()) / max(float(d.max() - d.min()), 1e-6) h, w = d.shape yy, xx = np.mgrid[0:h, 0:w] x = (xx / max(w - 1, 1) - 0.5) * 2.0 y = (0.5 - yy / max(h - 1, 1)) * 2.0 z = (1.0 - d) * 0.9 vertices = np.stack([x, y, z], axis=-1).reshape(-1, 3) faces = [] for row in range(h - 1): for col in range(w - 1): i = row * w + col faces.extend([[i, i + 1, i + w], [i + 1, i + w + 1, i + w]]) mesh = trimesh.Trimesh(vertices=vertices, faces=np.asarray(faces), process=False) mesh.visual.vertex_colors = np.concatenate( [rgb.reshape(-1, 3), np.full((vertices.shape[0], 1), 255, dtype=np.uint8)], axis=1 ) OUTPUT_DIR.mkdir(exist_ok=True) path = OUTPUT_DIR / f"simam3d_{run_id}.glb" mesh.export(path) return str(path) def depth_to_pointcloud(image: Image.Image, depth: np.ndarray, confidence: np.ndarray, run_id: str) -> str: """Export the current depth scaffold as a colored PLY point cloud.""" image = image.convert("RGB") max_side = 384 scale = min(1.0, max_side / max(image.size)) size = (max(8, int(image.width * scale)), max(8, int(image.height * scale))) rgb = np.asarray(image.resize(size, Image.Resampling.LANCZOS)) d = np.asarray(Image.fromarray(depth).resize(size, Image.Resampling.BILINEAR), dtype=np.float32) d = (d - d.min()) / max(float(d.max() - d.min()), 1e-6) h, w = d.shape yy, xx = np.mgrid[0:h, 0:w] points = np.stack( [(xx / max(w - 1, 1) - 0.5) * 2.0, (0.5 - yy / max(h - 1, 1)) * 2.0, (1.0 - d) * 0.9], axis=-1, ).reshape(-1, 3) confidence = np.asarray( Image.fromarray(np.asarray(confidence, dtype=np.float32)).resize(size, Image.Resampling.BILINEAR), dtype=np.float32, ) confidence = np.nan_to_num(confidence, nan=0.0, posinf=0.0, neginf=0.0) peak = float(max(confidence.max(), 0.0)) if confidence.size else 0.0 confidence = confidence / peak if peak > 1e-6 else np.ones_like(confidence) OUTPUT_DIR.mkdir(exist_ok=True) path = OUTPUT_DIR / f"simam3d_{run_id}.ply" write_confidence_ply(path, points, rgb.reshape(-1, 3), confidence.reshape(-1)) return str(path) def save_prediction_npz(depth, confidence, extrinsics, intrinsics, run_id: str) -> str: OUTPUT_DIR.mkdir(exist_ok=True) path = OUTPUT_DIR / f"simam3d_{run_id}.npz" np.savez_compressed( path, depth=np.asarray(depth, dtype=np.float32), confidence=np.asarray(confidence, dtype=np.float32), extrinsics=np.asarray(extrinsics, dtype=np.float32), intrinsics=np.asarray(intrinsics, dtype=np.float32), ) return str(path) def save_uncertainty_map(uncertainty: np.ndarray, run_id: str) -> str: OUTPUT_DIR.mkdir(exist_ok=True) path = OUTPUT_DIR / f"simam3d_{run_id}_reveal_uncertainty.png" make_uncertainty_preview(uncertainty).save(path, optimize=True) return str(path) def save_turntable_gif(points: np.ndarray, colors: np.ndarray, run_id: str) -> str: """Render a lightweight, dependency-free orbit preview from fused points.""" points = np.asarray(points, dtype=np.float32) colors = np.asarray(colors, dtype=np.uint8)[..., :3] if len(points) != len(colors): raise ValueError("turntable points and colors must have matching lengths") if len(points) > 12000: keep = np.linspace(0, len(points) - 1, 12000, dtype=np.int64) points, colors = points[keep], colors[keep] frames = [] size = 512 for yaw in np.linspace(0.0, 330.0, 12): xy, depth = orbit_project_points(points, float(yaw)) frame = Image.new("RGB", (size, size), (9, 12, 22)) draw = ImageDraw.Draw(frame) order = np.argsort(depth) for index in order: x = int((float(xy[index, 0]) * 0.43 + 0.5) * size) y = int((0.5 - float(xy[index, 1]) * 0.43) * size) if 0 <= x < size and 0 <= y < size: radius = 1 if len(points) > 4000 else 2 color = tuple(int(value) for value in colors[index]) draw.ellipse((x - radius, y - radius, x + radius, y + radius), fill=color) draw.text((16, 16), f"SIMAM3D | {float(yaw):.0f} deg", fill=(235, 240, 255)) frames.append(frame) OUTPUT_DIR.mkdir(exist_ok=True) path = OUTPUT_DIR / f"simam3d_{run_id}_turntable.gif" frames[0].save(path, save_all=True, append_images=frames[1:], duration=100, loop=0, optimize=True) return str(path) def save_run_manifest(run_id, backend, view_count, elapsed, outputs, metrics) -> str: OUTPUT_DIR.mkdir(exist_ok=True) path = OUTPUT_DIR / f"simam3d_{run_id}_run.json" manifest = { "project": "Simam3D", "version": SIMAM3D_VERSION, "manifest_schema_version": MANIFEST_SCHEMA_VERSION, "run_id": run_id, "backend": backend, "model": ( DA3_MODEL_ID if backend.startswith("Depth Anything 3") else "none (deterministic export smoke test)" if backend.startswith("Deterministic") else MODEL_ID ), "runtime": runtime_label(), "claims": { "observed_input_geometry": True, "generated_novel_views": False, "metric_scale_reconstruction": False, "learned_gaussian_optimization": False, "direct_gaussian_prediction": metrics.get("gaussian_source") == "da3_direct", "depth_normal_light_is_heuristic": True, "reveal_content_is_verified": False, }, "software": { "python": platform.python_version(), "platform": platform.platform(), "torch": getattr(torch, "__version__", "unknown"), "gradio": getattr(gr, "__version__", "unknown"), "numpy": np.__version__, "python_implementation": sys.implementation.name, }, "view_count": view_count, "elapsed_seconds": round(elapsed, 3), "metrics": metrics, "outputs": outputs, "output_sha256": { Path(output).name: file_fingerprint(output) for output in outputs if Path(output).is_file() }, } path.write_text(json.dumps(manifest, indent=2), encoding="utf-8") valid, errors = validate_manifest(path) if not valid: raise RuntimeError( "Generated manifest failed self-validation: " + "; ".join(errors) ) return str(path) def predict_v2(images): """Run the lightweight, widely compatible depth fallback for every view.""" depths = [] for view in images: with torch.inference_mode(): result = get_depth_pipeline()(view) predicted = result.get("predicted_depth") if predicted is None: raise RuntimeError(f"Depth model returned no predicted_depth field: {list(result)}") depths.append(predicted.squeeze().detach().float().cpu().numpy().astype(np.float32)) return np.stack(depths, axis=0) def fuse_views(images, depths, confidence, extrinsics, intrinsics, run_id: str): """Fuse depth views into a confidence-weighted world-space PLY.""" views = [] for idx, (image, depth) in enumerate(zip(images, depths)): depth = np.asarray(depth, dtype=np.float32) h, w = depth.shape colors = np.asarray(image.convert("RGB").resize((w, h), Image.Resampling.LANCZOS)) conf = np.asarray(confidence[idx], dtype=np.float32) if len(confidence) > idx else None if conf is not None and conf.shape != depth.shape: conf = np.asarray(Image.fromarray(conf).resize((w, h), Image.Resampling.BILINEAR)) camera_k = intrinsics[idx] if len(intrinsics) > idx else None if camera_k is not None: camera_k = np.asarray(camera_k, dtype=np.float32).copy() source_w, source_h = image.size if camera_k.shape == (3, 3) and source_w and source_h: camera_k[0, :] *= w / source_w camera_k[1, :] *= h / source_h views.append((depth, conf, camera_k, extrinsics[idx] if len(extrinsics) > idx else None, colors)) result = fuse_point_views( views, voxel_size=FUSION_VOXEL_SIZE, max_points=FUSION_MAX_POINTS ) result = filter_fusion_by_support(result, MIN_FUSION_SUPPORT_VIEWS) OUTPUT_DIR.mkdir(exist_ok=True) path = OUTPUT_DIR / f"simam3d_{run_id}_fused.ply" write_confidence_ply( path, result.points, result.colors, result.weights, result.source_views, result.source_masks ) return str(path), fusion_metrics(result), result.weights, result.source_views, result.source_masks def pointcloud_to_gaussian_ply(pointcloud_path: str, run_id: str, weights=None, source_views=None, source_masks=None) -> str: """Convert fused colored points to a confidence-initialized 3DGS PLY.""" cloud = trimesh.load(pointcloud_path, process=False) points = np.asarray(cloud.vertices, dtype=np.float32) colors = np.asarray(cloud.colors[:, :3], dtype=np.uint8) OUTPUT_DIR.mkdir(exist_ok=True) path = OUTPUT_DIR / f"simam3d_{run_id}_gaussians.ply" write_gaussian_ply(path, points, colors, weights, source_views, source_masks) return str(path) def _prediction_array(value): """Move an optional DA3 tensor to a CPU NumPy array without torch types.""" if hasattr(value, "detach"): value = value.detach().cpu().numpy() return np.asarray(value) def direct_da3_gaussian_ply(prediction, run_id: str) -> tuple[str, int, int] | None: """Export DA3's learned world-space Gaussian parameters when available.""" gaussians = getattr(prediction, "gaussians", None) if gaussians is None: return None print("Simam3D direct Gaussian export: reading parameters", flush=True) means = _prediction_array(getattr(gaussians, "means", None)) scales = _prediction_array(getattr(gaussians, "scales", None)) rotations = _prediction_array(getattr(gaussians, "rotations", None)) opacities = _prediction_array(getattr(gaussians, "opacities", None)).reshape(-1) print(f"Simam3D direct Gaussian export: means={means.shape} scales={scales.shape}", flush=True) if means.ndim == 3 and means.shape[0] == 1: means = means[0] if scales.ndim == 3 and scales.shape[0] == 1: scales = scales[0] if rotations.ndim == 3 and rotations.shape[0] == 1: rotations = rotations[0] if means.ndim != 2 or means.shape[1] != 3: raise ValueError(f"DA3 Gaussian means must have shape Nx3, got {means.shape}") if scales.shape != (len(means), 3) or rotations.shape != (len(means), 4) or len(opacities) != len(means): raise ValueError("DA3 Gaussian parameter arrays do not have matching lengths") original_count = len(means) keep = None if original_count > GAUSSIAN_MAX_POINTS: # Preserve the most visible splats and restore source order for # deterministic output. This bounds ASCII export time and file size. keep = np.argsort(np.nan_to_num(opacities, nan=-np.inf))[-GAUSSIAN_MAX_POINTS:] keep.sort() means, scales, rotations, opacities = means[keep], scales[keep], rotations[keep], opacities[keep] print(f"Simam3D direct Gaussian export: selecting {len(means)} of {original_count}", flush=True) harmonics = getattr(gaussians, "harmonics", None) if harmonics is not None: # Only the DC coefficient is needed for the portable PLY color field; # avoid transferring the complete SH basis from GPU memory. if hasattr(harmonics, "detach") and harmonics.ndim >= 3: harmonics = harmonics[..., 0] harmonics = _prediction_array(harmonics) if harmonics.ndim == 4 and harmonics.shape[0] == 1: harmonics = harmonics[0] if harmonics.ndim == 3 and harmonics.shape[0] == 1: harmonics = harmonics[0] if keep is not None and harmonics.shape[0] == original_count: harmonics = harmonics[keep] if harmonics.ndim == 2 and harmonics.shape[0] == len(means): colors = np.clip((harmonics * 0.2820947918 + 0.5) * 255.0, 0, 255).astype(np.uint8) else: colors = np.full((len(means), 3), 180, dtype=np.uint8) else: colors = np.full((len(means), 3), 180, dtype=np.uint8) path = OUTPUT_DIR / f"simam3d_{run_id}_gaussians_da3_direct.ply" write_gaussian_ply( path, means, colors, weights=np.clip(opacities, 0.0, 1.0), scales=scales, rotations=rotations, opacities=opacities, ) print(f"Simam3D direct Gaussian export: wrote {path}", flush=True) return str(path), original_count, len(means) def load_input_images(primary: Image.Image, extra_files) -> list[Image.Image]: images = [primary.convert("RGB")] for item in extra_files or []: path = item if isinstance(item, str) else ( getattr(item, "path", None) or getattr(item, "name", None) ) if path: try: with Image.open(path) as extra: images.append(extra.convert("RGB").copy()) except (OSError, ValueError) as exc: raise ValueError(f"Could not read additional view {path!r}: {exc}") from exc if len(images) > MAX_VIEWS: raise ValueError(f"At most {MAX_VIEWS} views are supported per run to protect hosted GPU memory.") normalized = [] for current in images: if max(current.size) > MAX_INPUT_SIDE: scale = MAX_INPUT_SIDE / max(current.size) size = (max(8, int(current.width * scale)), max(8, int(current.height * scale))) current = current.resize(size, Image.Resampling.LANCZOS) normalized.append(current) return normalized def image_fingerprint(image: Image.Image) -> str: """Hash normalized RGB pixels so runs can be reproduced exactly.""" buffer = io.BytesIO() image.convert("RGB").save(buffer, format="PNG", optimize=False) return hashlib.sha256(buffer.getvalue()).hexdigest() def near_duplicate_pair_count(images: list[Image.Image], threshold: float = 0.03) -> int: """Count pairs with tiny normalized RGB distance for view-quality diagnostics.""" thumbnails = [ np.asarray(ImageOps.fit(image.convert("RGB"), (32, 32)), dtype=np.float32) / 255.0 for image in images ] return sum( float(np.abs(thumbnails[left] - thumbnails[right]).mean()) <= threshold for left in range(len(thumbnails)) for right in range(left + 1, len(thumbnails)) ) def file_fingerprint(path: str | Path) -> str: """Hash an exported artifact in bounded chunks for manifest provenance.""" digest = hashlib.sha256() with Path(path).open("rb") as handle: for chunk in iter(lambda: handle.read(1024 * 1024), b""): digest.update(chunk) return digest.hexdigest() def da3_api_status() -> dict: """Check the optional DA3 API without downloading weights or loading a model.""" try: _, location = load_da3_class() return {"installed": True, "importable": True, "api_location": location, "error": None} except (ImportError, ModuleNotFoundError, AttributeError) as exc: return {"installed": False, "importable": False, "api_location": None, "error": f"{type(exc).__name__}: {exc}"} except Exception as exc: return {"installed": True, "importable": False, "api_location": None, "error": f"{type(exc).__name__}: {exc}"} def load_da3_class(): """Resolve the DA3 class across known upstream package layouts.""" return resolve_da3_class() def save_failure_manifest(requested_backend: str, stage: str, exc: Exception, input_hashes=None) -> str: """Persist actionable failure context without storing user media or secrets.""" OUTPUT_DIR.mkdir(exist_ok=True) run_id = uuid.uuid4().hex manifest = { "project": "Simam3D", "version": SIMAM3D_VERSION, "manifest_schema_version": MANIFEST_SCHEMA_VERSION, "run_id": run_id, "status": "failed", "backend": requested_backend, "runtime": runtime_label(), "software": { "python": platform.python_version(), "torch": getattr(torch, "__version__", "unknown"), "gradio": getattr(gr, "__version__", "unknown"), }, "stage": stage, "input_sha256": list(input_hashes or []), "error": {"type": type(exc).__name__, "message": str(exc)}, "traceback_tail": traceback.format_exc().splitlines()[-8:], "claims": { "observed_input_geometry": False, "generated_novel_views": False, "metric_scale_reconstruction": False, "learned_gaussian_optimization": False, "failure_report_only": True, }, } path = OUTPUT_DIR / f"simam3d_{run_id}_failure.json" path.write_text(json.dumps(manifest, indent=2), encoding="utf-8") return str(path) def diagnostics() -> str: """Return actionable environment information without loading a model.""" OUTPUT_DIR.mkdir(exist_ok=True) checks = { "simam3d_version": SIMAM3D_VERSION, "runtime": runtime_label(), "cuda_available": bool(torch.cuda.is_available()), "torch": getattr(torch, "__version__", "unknown"), "gradio": getattr(gr, "__version__", "unknown"), "v2_model": MODEL_ID, "da3_api": da3_api_status(), "da3_model": DA3_MODEL_ID, "output_directory_writable": os.access(OUTPUT_DIR, os.W_OK), "max_input_side": MAX_INPUT_SIDE, "max_views": MAX_VIEWS, "fusion_voxel_size": FUSION_VOXEL_SIZE, "fusion_max_points": FUSION_MAX_POINTS, "min_fusion_support_views": MIN_FUSION_SUPPORT_VIEWS, "da3_process_res": DA3_PROCESS_RES, "gpu_duration_seconds": GPU_DURATION_SECONDS, } if torch.cuda.is_available(): checks["cuda_device"] = torch.cuda.get_device_name(0) checks["cuda_memory_gb"] = round(torch.cuda.get_device_properties(0).total_memory / 1024**3, 2) return "```json\n" + json.dumps(checks, indent=2) + "\n```" def recent_failure_reports(): """Return recent redacted failure manifests for download from the UI.""" OUTPUT_DIR.mkdir(exist_ok=True) return [str(path) for path in sorted(OUTPUT_DIR.glob("*_failure.json"), key=lambda item: item.stat().st_mtime, reverse=True)[:10]] def _generate_impl(image: Image.Image, extra_files, backend: str, guess: str, demo_mode: bool, progress=gr.Progress()): if image is None: raise gr.Error("Please add an input image first.") started = time.perf_counter() stage = "input_validation" input_hashes = [] try: stage = "input_loading" images = load_input_images(image, extra_files) view_count = len(images) input_hashes = [image_fingerprint(view) for view in images] input_diversity = input_diversity_metrics(input_hashes, near_duplicate_pair_count(images)) scene_guess = classify_scene_hypothesis(guess) lighting = analyze_light_field(np.asarray(images[0])) backend_used = backend fallback_note = "" fallback_error = None confidence_source = "baseline_uniform" da3_prediction = None progress(0.05, desc=f"Loading {backend} on {runtime_label()}") confidence = np.ones((view_count, 1, 1), dtype=np.float32) extrinsics = synthetic_orbit_poses(view_count) intrinsics = np.repeat(np.eye(3, dtype=np.float32)[None, ...], view_count, axis=0) pose_source = "synthetic_orbit_prior" if demo_mode: stage = "deterministic_depth" depth = np.stack([deterministic_demo_depth(np.asarray(view)) for view in images], axis=0) confidence = np.ones_like(depth, dtype=np.float32) backend_used = "Deterministic export smoke test (no AI)" pose_source = "synthetic_orbit_prior" confidence_source = "uniform_demo" progress(0.45, desc="Running deterministic export smoke test") elif backend == "Depth Anything 3 (camera-aware)": try: stage = "da3_inference" with INFERENCE_LOCK, torch.inference_mode(): prediction = get_da3_model().inference( image=images, infer_gs=True, process_res=DA3_PROCESS_RES ) da3_prediction = prediction depth, confidence, extrinsics, intrinsics, confidence_source, pose_source = normalize_prediction_contract( prediction, view_count ) except (ImportError, ModuleNotFoundError, RuntimeError, TypeError, ValueError) as exc: clear_da3_model() fallback_error = f"{type(exc).__name__}: {exc}" fallback_note = f" DA3 unavailable ({type(exc).__name__}); used the stable V2 fallback." print(f"Simam3D DA3 unavailable; falling back to V2: {exc}", flush=True) gr.Warning(fallback_note.strip()) backend_used = "Depth Anything V2 (stable baseline; DA3 fallback)" stage = "v2_fallback_inference" with INFERENCE_LOCK: depth = predict_v2(images) confidence = np.ones_like(depth, dtype=np.float32) else: stage = "v2_inference" with INFERENCE_LOCK: depth = predict_v2(images) confidence = np.ones_like(depth, dtype=np.float32) if depth.ndim != 3 or not np.isfinite(depth).all(): raise RuntimeError(f"Invalid depth tensor returned with shape {depth.shape}") stage = "geometry_export" progress(0.65, desc="Building depth geometry") run_id = uuid.uuid4().hex depth_image = make_depth_preview(depth[0]) confidence_image = make_confidence_preview(confidence[0]) glb_path = depth_to_glb(images[0], depth[0], run_id) ply_path = depth_to_pointcloud(images[0], depth[0], confidence[0], run_id) fused_path, fusion_stats, fused_weights, fused_source_views, fused_source_masks = fuse_views( images, depth, confidence, extrinsics, intrinsics, run_id ) fusion_stats["support_status"] = classify_fusion_support( fusion_stats, view_count, input_diversity["unique_input_view_count"], pose_source, input_diversity["near_duplicate_pair_count"], ) fused_cloud = trimesh.load(fused_path, process=False) fused_points = np.asarray(fused_cloud.vertices, dtype=np.float32) projection_coverage = [] reprojection_metrics = [] for idx in range(view_count): _, _, inside = project_world_points( fused_points, intrinsics[idx], extrinsics[idx], images[idx].width, images[idx].height ) projection_coverage.append(float(inside.mean()) if len(inside) else 0.0) reprojection_metrics.append( reprojection_consistency(fused_points, depth[idx], intrinsics[idx], extrinsics[idx]) ) turntable_path = save_turntable_gif( fused_points, np.asarray(fused_cloud.colors[:, :3], dtype=np.uint8), run_id, ) gaussian_path = pointcloud_to_gaussian_ply( fused_path, run_id, fused_weights, fused_source_views, fused_source_masks ) gaussian_source = "fused_point_initialization" direct_gaussian_error = None direct_gaussian_count = 0 direct_gaussian_exported_count = 0 if da3_prediction is not None: try: direct_result = direct_da3_gaussian_ply(da3_prediction, run_id) if direct_result is not None: direct_path, direct_gaussian_count, direct_gaussian_exported_count = direct_result gaussian_path = direct_path gaussian_source = "da3_direct" except (RuntimeError, TypeError, ValueError) as exc: direct_gaussian_error = f"{type(exc).__name__}: {exc}" print( f"Simam3D native Gaussian export unavailable; using fused initialization: {exc}", flush=True, ) gr.Warning("Native DA3 Gaussian export was unavailable; exported fused Gaussian initialization instead.") npz_path = save_prediction_npz(depth, confidence, extrinsics, intrinsics, run_id) uncertainty = reveal_uncertainty(depth[0], confidence[0]) depth_lighting = estimate_depth_light(depth[0], np.asarray(images[0].resize(depth[0].shape[::-1], Image.Resampling.LANCZOS))) uncertainty_path = save_uncertainty_map(uncertainty, run_id) view_plan = next_view_plan(view_count, extrinsics) evidence_card = make_evidence_card( images[0], depth[0], confidence[0], backend_used, view_count, fusion_stats, run_id ) reveal_card = make_reveal_card(images[0], depth[0], scene_guess, lighting, run_id) analysis_report = ( "### Behind-the-image analysis\n" f"**Your hypothesis:** {scene_guess['user_guess']} \n" f"**Semantic cues:** {', '.join(scene_guess['semantic_cues'])} \n" f"**Lighting:** {lighting['lighting_label']} \n" f"**Estimated light direction:** azimuth {lighting['estimated_azimuth_deg']:.0f} deg, " f"elevation {lighting['estimated_elevation_deg']:.0f} deg \n" f"**3D light vector:** ({depth_lighting['light_direction_x']:.2f}, " f"{depth_lighting['light_direction_y']:.2f}, {depth_lighting['light_direction_z']:.2f}) " f"with confidence {depth_lighting['photometric_confidence']:.2f} \n" f"**Detector confidence:** {lighting['detector_confidence']:.2f} \n" f"**Reveal uncertainty:** {uncertainty.mean():.2f} mean visible-space uncertainty \n" f"**Fused projection coverage:** {np.mean(projection_coverage):.1%} mean across input views \n" f"**Depth reprojection MAE:** {np.mean([item['mean_absolute_error'] for item in reprojection_metrics]):.3f} " "relative-depth diagnostic \n" f"**Input diversity:** {input_diversity['input_diversity_status']} " f"({input_diversity['unique_input_view_count']}/{input_diversity['input_view_count']} unique view(s)) \n" f"**Fusion support:** {fusion_stats['support_status']} \n" f"**Next view suggestion:** {view_plan[0]['yaw_deg']:.0f} deg yaw " f"({view_plan[0]['gap_from_existing_deg']:.0f} deg gap) \n" "*This is an image-space lighting estimate and a user-guided scene hypothesis, " "not verified hidden content.*" ) elapsed = time.perf_counter() - started manifest_path = save_run_manifest( run_id, backend_used, view_count, elapsed, [glb_path, ply_path, fused_path, gaussian_path, npz_path, uncertainty_path, turntable_path, evidence_card, reveal_card], { "depth_shape": list(depth.shape), "depth_min": float(depth.min()), "depth_max": float(depth.max()), "confidence_mean": float(np.mean(confidence)), # Preserve the full fusion evidence in the manifest. Keeping # this nested record avoids reducing a multi-view run to only # its point count and makes source support auditable. "fusion": fusion_stats, "gaussian_source": gaussian_source, "direct_gaussian_count": direct_gaussian_count, "direct_gaussian_exported_count": direct_gaussian_exported_count, "direct_gaussian_error": direct_gaussian_error, "gaussian_export_max_points": GAUSSIAN_MAX_POINTS, "da3_process_res": DA3_PROCESS_RES, "fused_point_count": fusion_stats["point_count"], "fused_confidence_mean": fusion_stats["confidence_mean"], "fused_confidence_p95": fusion_stats["confidence_p95"], "pose_source": pose_source, "confidence_source": confidence_source, "requested_backend": backend, "fallback_error": fallback_error, "input_sha256": input_hashes, "input_diversity": input_diversity, "input_sizes": [list(view.size) for view in images], "fusion_voxel_size": FUSION_VOXEL_SIZE, "fusion_max_points": FUSION_MAX_POINTS, "min_fusion_support_views": MIN_FUSION_SUPPORT_VIEWS, "scene_hypothesis": scene_guess, "lighting_analysis": lighting, "depth_normal_lighting": depth_lighting, "reveal_uncertainty_mean": float(uncertainty.mean()), "next_view_plan": view_plan, "projection_coverage": projection_coverage, "reprojection_consistency": reprojection_metrics, }, ) progress(1.0, desc="Done") return depth_image, confidence_image, glb_path, ply_path, fused_path, gaussian_path, npz_path, uncertainty_path, turntable_path, evidence_card, reveal_card, manifest_path, analysis_report, ( f"Processed {view_count} view(s) on {runtime_label()} in {elapsed:.1f}s using {backend_used}. " "Per-view and confidence-weighted fused point clouds are exported. " f"Pose source: {pose_source}; confidence source: {confidence_source}." + fallback_note ) except Exception as exc: print(f"Simam3D generation failed: {type(exc).__name__}: {exc}", flush=True) failure_path = save_failure_manifest(backend, stage, exc, input_hashes) raise gr.Error( f"Generation failed: {type(exc).__name__}: {exc}. Failure manifest: {failure_path}" ) from exc @GPU_FUNCTION def _generate_gpu(image: Image.Image, extra_files, backend: str, guess: str, demo_mode: bool, progress=gr.Progress()): return _generate_impl(image, extra_files, backend, guess, demo_mode, progress) def generate(image: Image.Image, extra_files, backend: str, guess: str, demo_mode: bool, progress=gr.Progress()): """Keep the deterministic export diagnostic independent of GPU quota.""" if demo_mode: return _generate_impl(image, extra_files, backend, guess, demo_mode, progress) return _generate_gpu(image, extra_files, backend, guess, demo_mode, progress) with gr.Blocks(title="Simam3D") as demo: gr.Markdown("# Simam3D\nSingle-image depth and 3D reconstruction workbench") gr.Markdown( "Phase 1 is deliberately transparent: the original image is lifted using a depth model, " "then exported as a textured GLB. This gives us a known-good baseline before adding " "multi-view generation and Gaussian splat fusion." ) gr.Markdown( "**First run:** keep **Depth Anything V2 (stable baseline)** selected. " "Use **Depth Anything 3 (camera-aware)** when you want to test the heavier " "multi-view research path; failed DA3 loads fall back to V2 with a warning." ) with gr.Row(): with gr.Column(): image = gr.Image(type="pil", label="Input image") extra_files = gr.Files( label="Optional additional views", file_types=["image"], file_count="multiple", type="filepath", ) guess = gr.Textbox( label="What do you think is behind this image?", placeholder="e.g. I think there is a garden or a room behind it...", ) backend = gr.Dropdown( choices=["Depth Anything 3 (camera-aware)", "Depth Anything V2 (stable baseline)"], value="Depth Anything V2 (stable baseline)", label="Reconstruction backend", ) demo_mode = gr.Checkbox( label="Run export smoke test (no AI model; diagnostic only)", value=False, ) run = gr.Button("Generate depth + 3D", variant="primary") with gr.Column(): depth = gr.Image(label="Predicted depth") confidence_view = gr.Image(label="Confidence diagnostic") model = gr.File(label="Download depth geometry GLB", file_count="single") pointcloud = gr.File(label="Download point cloud PLY", file_count="single") fused = gr.File(label="Download fused point cloud PLY", file_count="single") gaussians = gr.File(label="Download Gaussian-splat PLY", file_count="single") metadata = gr.File(label="Download depth/camera metadata NPZ", file_count="single") uncertainty_file = gr.File(label="Download reveal uncertainty map PNG", file_count="single") turntable_file = gr.File(label="Download reconstructed turntable GIF", file_count="single") evidence_card = gr.File(label="Download shareable evidence card PNG", file_count="single") reveal_card = gr.File(label="Download What's Behind reveal card PNG", file_count="single") manifest = gr.File(label="Download reproducibility manifest JSON", file_count="single") analysis_report = gr.Markdown() status = gr.Markdown(f"Ready. Simam3D **v{SIMAM3D_VERSION}** | Runtime: **{runtime_label()}**") with gr.Accordion("Diagnostics", open=False): diagnose = gr.Button("Run environment diagnostics") diagnostic_report = gr.Markdown() diagnose.click(diagnostics, outputs=diagnostic_report) refresh_failures = gr.Button("Refresh recent failure reports") failure_reports = gr.Files(label="Download recent failure reports", file_count="multiple") refresh_failures.click(recent_failure_reports, outputs=failure_reports) run.click( generate, inputs=[image, extra_files, backend, guess, demo_mode], outputs=[depth, confidence_view, model, pointcloud, fused, gaussians, metadata, uncertainty_file, turntable_file, evidence_card, reveal_card, manifest, analysis_report, status], concurrency_limit=1, concurrency_id="simam3d_heavy_inference", ) if __name__ == "__main__": print(f"Simam3D v{SIMAM3D_VERSION} starting on {runtime_label()}", flush=True) demo.launch(show_error=True)