Spaces:
Running on Zero
Running on Zero
Download app.py from junaid-simamdigital/Simam3D-GPU: direct link, hf CLI and curl.
- Browser
- Download file 43.5 kB
-
https://huggingface.co/spaces/junaid-simamdigital/Simam3D-GPU/resolve/main/app.py
- Command line
-
hf download hf://spaces/junaid-simamdigital/Simam3D-GPU/app.py
-
curl -L -o app.py https://huggingface.co/spaces/junaid-simamdigital/Simam3D-GPU/resolve/main/app.py
43.5 kB
| from __future__ import annotations | |
| import os | |
| import json | |
| import hashlib | |
| import io | |
| import platform | |
| import sys | |
| import threading | |
| import time | |
| import traceback | |
| import uuid | |
| from pathlib import Path | |
| for _stream in (sys.stdout, sys.stderr): | |
| if hasattr(_stream, "reconfigure"): | |
| _stream.reconfigure(encoding="utf-8", errors="replace") | |
| import gradio as gr | |
| from matplotlib import colormaps | |
| import numpy as np | |
| import torch | |
| import trimesh | |
| from PIL import Image, ImageDraw, ImageOps | |
| from transformers import pipeline | |
| GPU_DURATION_SECONDS = int(os.getenv("SIMAM3D_GPU_DURATION_SECONDS", "60")) | |
| try: | |
| import spaces | |
| GPU_FUNCTION = spaces.GPU(duration=GPU_DURATION_SECONDS) | |
| except (ImportError, AttributeError, TypeError): | |
| def GPU_FUNCTION(function): | |
| return function | |
| from simam3d_core import ( | |
| analyze_light_field, | |
| classify_scene_hypothesis, | |
| deterministic_demo_depth, | |
| classify_fusion_support, | |
| estimate_depth_light, | |
| filter_fusion_by_support, | |
| fuse_point_views, | |
| fusion_metrics, | |
| input_diversity_metrics, | |
| next_view_plan, | |
| normalize_prediction_contract, | |
| orbit_project_points, | |
| project_world_points, | |
| reprojection_consistency, | |
| reveal_uncertainty, | |
| synthetic_orbit_poses, | |
| write_confidence_ply, | |
| write_gaussian_ply, | |
| ) | |
| from da3_compat import resolve_da3_class | |
| from validate_manifest import validate_manifest | |
| MODEL_ID = os.getenv("SIMAM3D_DEPTH_MODEL", "depth-anything/Depth-Anything-V2-Small-hf") | |
| DA3_MODEL_ID = os.getenv("SIMAM3D_DA3_MODEL", "depth-anything/DA3NESTED-GIANT-LARGE-1.1") | |
| SIMAM3D_VERSION = "0.4.8" | |
| MANIFEST_SCHEMA_VERSION = 3 | |
| OUTPUT_DIR = Path("output") | |
| MAX_VIEWS = int(os.getenv("SIMAM3D_MAX_VIEWS", "8")) | |
| MAX_INPUT_SIDE = int(os.getenv("SIMAM3D_MAX_INPUT_SIDE", "1536")) | |
| FUSION_VOXEL_SIZE = float(os.getenv("SIMAM3D_FUSION_VOXEL_SIZE", "0.02")) | |
| FUSION_MAX_POINTS = int(os.getenv("SIMAM3D_FUSION_MAX_POINTS", "10000")) | |
| MIN_FUSION_SUPPORT_VIEWS = int(os.getenv("SIMAM3D_MIN_FUSION_SUPPORT_VIEWS", "1")) | |
| GAUSSIAN_MAX_POINTS = int(os.getenv("SIMAM3D_GAUSSIAN_MAX_POINTS", "2048")) | |
| DA3_PROCESS_RES = int(os.getenv("SIMAM3D_DA3_PROCESS_RES", "256")) | |
| DEPTH_PIPELINE = None | |
| DA3_MODEL = None | |
| PIPELINE_LOCK = threading.Lock() | |
| INFERENCE_LOCK = threading.Lock() | |
| def runtime_label() -> str: | |
| if torch.cuda.is_available(): | |
| return f"CUDA GPU ({torch.cuda.get_device_name(0)})" | |
| return "CPU" | |
| def get_depth_pipeline(): | |
| global DEPTH_PIPELINE | |
| if DEPTH_PIPELINE is not None: | |
| return DEPTH_PIPELINE | |
| with PIPELINE_LOCK: | |
| if DEPTH_PIPELINE is None: | |
| # Keep only one heavyweight backend resident when users switch | |
| # between DA3 and V2 in a long-lived Space process. | |
| clear_da3_model() | |
| device = 0 if torch.cuda.is_available() else -1 | |
| DEPTH_PIPELINE = pipeline("depth-estimation", model=MODEL_ID, device=device) | |
| return DEPTH_PIPELINE | |
| def get_da3_model(): | |
| global DA3_MODEL | |
| if DA3_MODEL is not None: | |
| return DA3_MODEL | |
| with PIPELINE_LOCK: | |
| if DA3_MODEL is None: | |
| clear_depth_pipeline() | |
| DepthAnything3, _ = load_da3_class() | |
| device = "cuda" if torch.cuda.is_available() else "cpu" | |
| DA3_MODEL = DepthAnything3.from_pretrained(DA3_MODEL_ID).to(device) | |
| DA3_MODEL.eval() | |
| return DA3_MODEL | |
| def clear_da3_model(): | |
| global DA3_MODEL | |
| DA3_MODEL = None | |
| if torch.cuda.is_available(): | |
| torch.cuda.empty_cache() | |
| def clear_depth_pipeline(): | |
| """Release the Transformers V2 pipeline before loading another backend.""" | |
| global DEPTH_PIPELINE | |
| DEPTH_PIPELINE = None | |
| if torch.cuda.is_available(): | |
| torch.cuda.empty_cache() | |
| def make_depth_preview(depth: np.ndarray) -> Image.Image: | |
| depth = (depth - depth.min()) / max(float(depth.max() - depth.min()), 1e-6) | |
| rgb = (colormaps.get_cmap("magma")(1.0 - depth)[..., :3] * 255).astype(np.uint8) | |
| return Image.fromarray(rgb) | |
| def make_confidence_preview(confidence: np.ndarray) -> Image.Image: | |
| confidence = np.asarray(confidence, dtype=np.float32) | |
| confidence = np.nan_to_num(confidence, nan=0.0, posinf=1.0, neginf=0.0) | |
| confidence = (confidence - confidence.min()) / max(float(confidence.max() - confidence.min()), 1e-6) | |
| rgb = (colormaps.get_cmap("viridis")(confidence)[..., :3] * 255).astype(np.uint8) | |
| return Image.fromarray(rgb) | |
| def make_uncertainty_preview(uncertainty: np.ndarray) -> Image.Image: | |
| uncertainty = np.clip(np.nan_to_num(np.asarray(uncertainty, dtype=np.float32)), 0.0, 1.0) | |
| rgb = (colormaps.get_cmap("inferno")(uncertainty)[..., :3] * 255).astype(np.uint8) | |
| return Image.fromarray(rgb) | |
| def make_evidence_card(image, depth, confidence, backend, view_count, fusion_stats, run_id): | |
| """Create a compact, shareable diagnostic card for each reconstruction run.""" | |
| tile_size = (480, 300) | |
| source = ImageOps.contain(image.convert("RGB"), tile_size) | |
| depth_tile = ImageOps.contain(make_depth_preview(depth).convert("RGB"), tile_size) | |
| confidence_tile = ImageOps.contain(make_confidence_preview(confidence).convert("RGB"), tile_size) | |
| canvas = Image.new("RGB", (960, 700), (12, 16, 28)) | |
| draw = ImageDraw.Draw(canvas) | |
| draw.text((24, 18), "SIMAM3D | EVIDENCE CARD", fill=(235, 240, 255)) | |
| draw.text( | |
| (24, 46), | |
| f"{backend} | {view_count} view(s) | {runtime_label()}", | |
| fill=(164, 178, 205), | |
| ) | |
| tiles = [(source, (0, 90), "SOURCE"), (depth_tile, (480, 90), "DEPTH"), | |
| (confidence_tile, (0, 390), "CONFIDENCE")] | |
| for tile, origin, label in tiles: | |
| x, y = origin | |
| canvas.paste(tile, (x, y)) | |
| draw.rectangle((x, y, x + tile.width, y + tile.height), outline=(76, 93, 126), width=2) | |
| draw.text((x + 12, y + 12), label, fill=(255, 255, 255)) | |
| draw.text( | |
| (500, 410), | |
| f"FUSED POINTS\n{fusion_stats['point_count']:,}\n\n" | |
| f"DOMINANT SOURCE VIEWS\n{fusion_stats['dominant_source_view_count']}\n\n" | |
| f"SUPPORTED SOURCE VIEWS\n{fusion_stats['supported_source_view_count']}\n\n" | |
| f"MULTI-VIEW VOXELS\n{fusion_stats['multi_view_voxel_fraction']:.1%}\n\n" | |
| f"CONFIDENCE MEAN\n{fusion_stats['confidence_mean']:.3f}\n\n" | |
| f"CONFIDENCE P95\n{fusion_stats['confidence_p95']:.3f}", | |
| fill=(210, 220, 240), | |
| spacing=8, | |
| ) | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}_evidence_card.png" | |
| canvas.save(path, optimize=True) | |
| return str(path) | |
| def make_reveal_card(image, depth, guess, lighting, run_id: str) -> str: | |
| """Create a social-friendly reveal card while clearly separating inference from fact.""" | |
| tile_size = (480, 300) | |
| source = ImageOps.contain(image.convert("RGB"), tile_size) | |
| depth_tile = ImageOps.contain(make_depth_preview(depth).convert("RGB"), tile_size) | |
| canvas = Image.new("RGB", (960, 760), (12, 16, 28)) | |
| draw = ImageDraw.Draw(canvas) | |
| draw.text((24, 18), "SIMAM3D | WHAT'S BEHIND THE IMAGE?", fill=(235, 240, 255)) | |
| draw.text((24, 48), "Depth and illumination reveal | hypothesis, not ground truth", fill=(164, 178, 205)) | |
| canvas.paste(source, (0, 90)) | |
| canvas.paste(depth_tile, (480, 90)) | |
| draw.rectangle((0, 90, 480, 390), outline=(76, 93, 126), width=2) | |
| draw.rectangle((480, 90, 960, 390), outline=(76, 93, 126), width=2) | |
| draw.text((14, 104), "ORIGINAL", fill=(255, 255, 255)) | |
| draw.text((494, 104), "DEPTH REVEAL", fill=(255, 255, 255)) | |
| guess_text = str(guess["user_guess"]) | |
| cues = ", ".join(str(item) for item in guess["semantic_cues"]) | |
| lighting_text = ( | |
| f"LIGHT FIELD: {lighting['lighting_label']}\n" | |
| f"Direction estimate: azimuth {lighting['estimated_azimuth_deg']:.0f} deg, " | |
| f"elevation {lighting['estimated_elevation_deg']:.0f} deg\n" | |
| f"Detector confidence: {lighting['detector_confidence']:.2f}" | |
| ) | |
| draw.text((24, 430), "YOUR GUESS", fill=(133, 214, 255)) | |
| draw.text((24, 456), guess_text[:110], fill=(235, 240, 255)) | |
| draw.text((24, 500), f"SEMANTIC CUES: {cues}", fill=(210, 220, 240)) | |
| draw.text((24, 560), lighting_text, fill=(210, 220, 240), spacing=8) | |
| draw.text((24, 690), "The unseen side is a research hypothesis; add views for evidence.", fill=(255, 190, 120)) | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}_reveal_card.png" | |
| canvas.save(path, optimize=True) | |
| return str(path) | |
| def depth_to_glb(image: Image.Image, depth: np.ndarray, run_id: str) -> str: | |
| image = image.convert("RGB") | |
| max_side = 192 | |
| scale = min(1.0, max_side / max(image.size)) | |
| size = (max(8, int(image.width * scale)), max(8, int(image.height * scale))) | |
| rgb = np.asarray(image.resize(size, Image.Resampling.LANCZOS)) | |
| d = Image.fromarray(depth).resize(size, Image.Resampling.BILINEAR) | |
| d = np.asarray(d, dtype=np.float32) | |
| d = (d - d.min()) / max(float(d.max() - d.min()), 1e-6) | |
| h, w = d.shape | |
| yy, xx = np.mgrid[0:h, 0:w] | |
| x = (xx / max(w - 1, 1) - 0.5) * 2.0 | |
| y = (0.5 - yy / max(h - 1, 1)) * 2.0 | |
| z = (1.0 - d) * 0.9 | |
| vertices = np.stack([x, y, z], axis=-1).reshape(-1, 3) | |
| faces = [] | |
| for row in range(h - 1): | |
| for col in range(w - 1): | |
| i = row * w + col | |
| faces.extend([[i, i + 1, i + w], [i + 1, i + w + 1, i + w]]) | |
| mesh = trimesh.Trimesh(vertices=vertices, faces=np.asarray(faces), process=False) | |
| mesh.visual.vertex_colors = np.concatenate( | |
| [rgb.reshape(-1, 3), np.full((vertices.shape[0], 1), 255, dtype=np.uint8)], axis=1 | |
| ) | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}.glb" | |
| mesh.export(path) | |
| return str(path) | |
| def depth_to_pointcloud(image: Image.Image, depth: np.ndarray, confidence: np.ndarray, run_id: str) -> str: | |
| """Export the current depth scaffold as a colored PLY point cloud.""" | |
| image = image.convert("RGB") | |
| max_side = 384 | |
| scale = min(1.0, max_side / max(image.size)) | |
| size = (max(8, int(image.width * scale)), max(8, int(image.height * scale))) | |
| rgb = np.asarray(image.resize(size, Image.Resampling.LANCZOS)) | |
| d = np.asarray(Image.fromarray(depth).resize(size, Image.Resampling.BILINEAR), dtype=np.float32) | |
| d = (d - d.min()) / max(float(d.max() - d.min()), 1e-6) | |
| h, w = d.shape | |
| yy, xx = np.mgrid[0:h, 0:w] | |
| points = np.stack( | |
| [(xx / max(w - 1, 1) - 0.5) * 2.0, (0.5 - yy / max(h - 1, 1)) * 2.0, (1.0 - d) * 0.9], | |
| axis=-1, | |
| ).reshape(-1, 3) | |
| confidence = np.asarray( | |
| Image.fromarray(np.asarray(confidence, dtype=np.float32)).resize(size, Image.Resampling.BILINEAR), | |
| dtype=np.float32, | |
| ) | |
| confidence = np.nan_to_num(confidence, nan=0.0, posinf=0.0, neginf=0.0) | |
| peak = float(max(confidence.max(), 0.0)) if confidence.size else 0.0 | |
| confidence = confidence / peak if peak > 1e-6 else np.ones_like(confidence) | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}.ply" | |
| write_confidence_ply(path, points, rgb.reshape(-1, 3), confidence.reshape(-1)) | |
| return str(path) | |
| def save_prediction_npz(depth, confidence, extrinsics, intrinsics, run_id: str) -> str: | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}.npz" | |
| np.savez_compressed( | |
| path, | |
| depth=np.asarray(depth, dtype=np.float32), | |
| confidence=np.asarray(confidence, dtype=np.float32), | |
| extrinsics=np.asarray(extrinsics, dtype=np.float32), | |
| intrinsics=np.asarray(intrinsics, dtype=np.float32), | |
| ) | |
| return str(path) | |
| def save_uncertainty_map(uncertainty: np.ndarray, run_id: str) -> str: | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}_reveal_uncertainty.png" | |
| make_uncertainty_preview(uncertainty).save(path, optimize=True) | |
| return str(path) | |
| def save_turntable_gif(points: np.ndarray, colors: np.ndarray, run_id: str) -> str: | |
| """Render a lightweight, dependency-free orbit preview from fused points.""" | |
| points = np.asarray(points, dtype=np.float32) | |
| colors = np.asarray(colors, dtype=np.uint8)[..., :3] | |
| if len(points) != len(colors): | |
| raise ValueError("turntable points and colors must have matching lengths") | |
| if len(points) > 12000: | |
| keep = np.linspace(0, len(points) - 1, 12000, dtype=np.int64) | |
| points, colors = points[keep], colors[keep] | |
| frames = [] | |
| size = 512 | |
| for yaw in np.linspace(0.0, 330.0, 12): | |
| xy, depth = orbit_project_points(points, float(yaw)) | |
| frame = Image.new("RGB", (size, size), (9, 12, 22)) | |
| draw = ImageDraw.Draw(frame) | |
| order = np.argsort(depth) | |
| for index in order: | |
| x = int((float(xy[index, 0]) * 0.43 + 0.5) * size) | |
| y = int((0.5 - float(xy[index, 1]) * 0.43) * size) | |
| if 0 <= x < size and 0 <= y < size: | |
| radius = 1 if len(points) > 4000 else 2 | |
| color = tuple(int(value) for value in colors[index]) | |
| draw.ellipse((x - radius, y - radius, x + radius, y + radius), fill=color) | |
| draw.text((16, 16), f"SIMAM3D | {float(yaw):.0f} deg", fill=(235, 240, 255)) | |
| frames.append(frame) | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}_turntable.gif" | |
| frames[0].save(path, save_all=True, append_images=frames[1:], duration=100, loop=0, optimize=True) | |
| return str(path) | |
| def save_run_manifest(run_id, backend, view_count, elapsed, outputs, metrics) -> str: | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}_run.json" | |
| manifest = { | |
| "project": "Simam3D", | |
| "version": SIMAM3D_VERSION, | |
| "manifest_schema_version": MANIFEST_SCHEMA_VERSION, | |
| "run_id": run_id, | |
| "backend": backend, | |
| "model": ( | |
| DA3_MODEL_ID if backend.startswith("Depth Anything 3") | |
| else "none (deterministic export smoke test)" if backend.startswith("Deterministic") | |
| else MODEL_ID | |
| ), | |
| "runtime": runtime_label(), | |
| "claims": { | |
| "observed_input_geometry": True, | |
| "generated_novel_views": False, | |
| "metric_scale_reconstruction": False, | |
| "learned_gaussian_optimization": False, | |
| "direct_gaussian_prediction": metrics.get("gaussian_source") == "da3_direct", | |
| "depth_normal_light_is_heuristic": True, | |
| "reveal_content_is_verified": False, | |
| }, | |
| "software": { | |
| "python": platform.python_version(), | |
| "platform": platform.platform(), | |
| "torch": getattr(torch, "__version__", "unknown"), | |
| "gradio": getattr(gr, "__version__", "unknown"), | |
| "numpy": np.__version__, | |
| "python_implementation": sys.implementation.name, | |
| }, | |
| "view_count": view_count, | |
| "elapsed_seconds": round(elapsed, 3), | |
| "metrics": metrics, | |
| "outputs": outputs, | |
| "output_sha256": { | |
| Path(output).name: file_fingerprint(output) | |
| for output in outputs | |
| if Path(output).is_file() | |
| }, | |
| } | |
| path.write_text(json.dumps(manifest, indent=2), encoding="utf-8") | |
| valid, errors = validate_manifest(path) | |
| if not valid: | |
| raise RuntimeError( | |
| "Generated manifest failed self-validation: " + "; ".join(errors) | |
| ) | |
| return str(path) | |
| def predict_v2(images): | |
| """Run the lightweight, widely compatible depth fallback for every view.""" | |
| depths = [] | |
| for view in images: | |
| with torch.inference_mode(): | |
| result = get_depth_pipeline()(view) | |
| predicted = result.get("predicted_depth") | |
| if predicted is None: | |
| raise RuntimeError(f"Depth model returned no predicted_depth field: {list(result)}") | |
| depths.append(predicted.squeeze().detach().float().cpu().numpy().astype(np.float32)) | |
| return np.stack(depths, axis=0) | |
| def fuse_views(images, depths, confidence, extrinsics, intrinsics, run_id: str): | |
| """Fuse depth views into a confidence-weighted world-space PLY.""" | |
| views = [] | |
| for idx, (image, depth) in enumerate(zip(images, depths)): | |
| depth = np.asarray(depth, dtype=np.float32) | |
| h, w = depth.shape | |
| colors = np.asarray(image.convert("RGB").resize((w, h), Image.Resampling.LANCZOS)) | |
| conf = np.asarray(confidence[idx], dtype=np.float32) if len(confidence) > idx else None | |
| if conf is not None and conf.shape != depth.shape: | |
| conf = np.asarray(Image.fromarray(conf).resize((w, h), Image.Resampling.BILINEAR)) | |
| camera_k = intrinsics[idx] if len(intrinsics) > idx else None | |
| if camera_k is not None: | |
| camera_k = np.asarray(camera_k, dtype=np.float32).copy() | |
| source_w, source_h = image.size | |
| if camera_k.shape == (3, 3) and source_w and source_h: | |
| camera_k[0, :] *= w / source_w | |
| camera_k[1, :] *= h / source_h | |
| views.append((depth, conf, camera_k, | |
| extrinsics[idx] if len(extrinsics) > idx else None, colors)) | |
| result = fuse_point_views( | |
| views, voxel_size=FUSION_VOXEL_SIZE, max_points=FUSION_MAX_POINTS | |
| ) | |
| result = filter_fusion_by_support(result, MIN_FUSION_SUPPORT_VIEWS) | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}_fused.ply" | |
| write_confidence_ply( | |
| path, result.points, result.colors, result.weights, | |
| result.source_views, result.source_masks | |
| ) | |
| return str(path), fusion_metrics(result), result.weights, result.source_views, result.source_masks | |
| def pointcloud_to_gaussian_ply(pointcloud_path: str, run_id: str, weights=None, source_views=None, source_masks=None) -> str: | |
| """Convert fused colored points to a confidence-initialized 3DGS PLY.""" | |
| cloud = trimesh.load(pointcloud_path, process=False) | |
| points = np.asarray(cloud.vertices, dtype=np.float32) | |
| colors = np.asarray(cloud.colors[:, :3], dtype=np.uint8) | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}_gaussians.ply" | |
| write_gaussian_ply(path, points, colors, weights, source_views, source_masks) | |
| return str(path) | |
| def _prediction_array(value): | |
| """Move an optional DA3 tensor to a CPU NumPy array without torch types.""" | |
| if hasattr(value, "detach"): | |
| value = value.detach().cpu().numpy() | |
| return np.asarray(value) | |
| def direct_da3_gaussian_ply(prediction, run_id: str) -> tuple[str, int, int] | None: | |
| """Export DA3's learned world-space Gaussian parameters when available.""" | |
| gaussians = getattr(prediction, "gaussians", None) | |
| if gaussians is None: | |
| return None | |
| print("Simam3D direct Gaussian export: reading parameters", flush=True) | |
| means = _prediction_array(getattr(gaussians, "means", None)) | |
| scales = _prediction_array(getattr(gaussians, "scales", None)) | |
| rotations = _prediction_array(getattr(gaussians, "rotations", None)) | |
| opacities = _prediction_array(getattr(gaussians, "opacities", None)).reshape(-1) | |
| print(f"Simam3D direct Gaussian export: means={means.shape} scales={scales.shape}", flush=True) | |
| if means.ndim == 3 and means.shape[0] == 1: | |
| means = means[0] | |
| if scales.ndim == 3 and scales.shape[0] == 1: | |
| scales = scales[0] | |
| if rotations.ndim == 3 and rotations.shape[0] == 1: | |
| rotations = rotations[0] | |
| if means.ndim != 2 or means.shape[1] != 3: | |
| raise ValueError(f"DA3 Gaussian means must have shape Nx3, got {means.shape}") | |
| if scales.shape != (len(means), 3) or rotations.shape != (len(means), 4) or len(opacities) != len(means): | |
| raise ValueError("DA3 Gaussian parameter arrays do not have matching lengths") | |
| original_count = len(means) | |
| keep = None | |
| if original_count > GAUSSIAN_MAX_POINTS: | |
| # Preserve the most visible splats and restore source order for | |
| # deterministic output. This bounds ASCII export time and file size. | |
| keep = np.argsort(np.nan_to_num(opacities, nan=-np.inf))[-GAUSSIAN_MAX_POINTS:] | |
| keep.sort() | |
| means, scales, rotations, opacities = means[keep], scales[keep], rotations[keep], opacities[keep] | |
| print(f"Simam3D direct Gaussian export: selecting {len(means)} of {original_count}", flush=True) | |
| harmonics = getattr(gaussians, "harmonics", None) | |
| if harmonics is not None: | |
| # Only the DC coefficient is needed for the portable PLY color field; | |
| # avoid transferring the complete SH basis from GPU memory. | |
| if hasattr(harmonics, "detach") and harmonics.ndim >= 3: | |
| harmonics = harmonics[..., 0] | |
| harmonics = _prediction_array(harmonics) | |
| if harmonics.ndim == 4 and harmonics.shape[0] == 1: | |
| harmonics = harmonics[0] | |
| if harmonics.ndim == 3 and harmonics.shape[0] == 1: | |
| harmonics = harmonics[0] | |
| if keep is not None and harmonics.shape[0] == original_count: | |
| harmonics = harmonics[keep] | |
| if harmonics.ndim == 2 and harmonics.shape[0] == len(means): | |
| colors = np.clip((harmonics * 0.2820947918 + 0.5) * 255.0, 0, 255).astype(np.uint8) | |
| else: | |
| colors = np.full((len(means), 3), 180, dtype=np.uint8) | |
| else: | |
| colors = np.full((len(means), 3), 180, dtype=np.uint8) | |
| path = OUTPUT_DIR / f"simam3d_{run_id}_gaussians_da3_direct.ply" | |
| write_gaussian_ply( | |
| path, means, colors, weights=np.clip(opacities, 0.0, 1.0), | |
| scales=scales, rotations=rotations, opacities=opacities, | |
| ) | |
| print(f"Simam3D direct Gaussian export: wrote {path}", flush=True) | |
| return str(path), original_count, len(means) | |
| def load_input_images(primary: Image.Image, extra_files) -> list[Image.Image]: | |
| images = [primary.convert("RGB")] | |
| for item in extra_files or []: | |
| path = item if isinstance(item, str) else ( | |
| getattr(item, "path", None) or getattr(item, "name", None) | |
| ) | |
| if path: | |
| try: | |
| with Image.open(path) as extra: | |
| images.append(extra.convert("RGB").copy()) | |
| except (OSError, ValueError) as exc: | |
| raise ValueError(f"Could not read additional view {path!r}: {exc}") from exc | |
| if len(images) > MAX_VIEWS: | |
| raise ValueError(f"At most {MAX_VIEWS} views are supported per run to protect hosted GPU memory.") | |
| normalized = [] | |
| for current in images: | |
| if max(current.size) > MAX_INPUT_SIDE: | |
| scale = MAX_INPUT_SIDE / max(current.size) | |
| size = (max(8, int(current.width * scale)), max(8, int(current.height * scale))) | |
| current = current.resize(size, Image.Resampling.LANCZOS) | |
| normalized.append(current) | |
| return normalized | |
| def image_fingerprint(image: Image.Image) -> str: | |
| """Hash normalized RGB pixels so runs can be reproduced exactly.""" | |
| buffer = io.BytesIO() | |
| image.convert("RGB").save(buffer, format="PNG", optimize=False) | |
| return hashlib.sha256(buffer.getvalue()).hexdigest() | |
| def near_duplicate_pair_count(images: list[Image.Image], threshold: float = 0.03) -> int: | |
| """Count pairs with tiny normalized RGB distance for view-quality diagnostics.""" | |
| thumbnails = [ | |
| np.asarray(ImageOps.fit(image.convert("RGB"), (32, 32)), dtype=np.float32) / 255.0 | |
| for image in images | |
| ] | |
| return sum( | |
| float(np.abs(thumbnails[left] - thumbnails[right]).mean()) <= threshold | |
| for left in range(len(thumbnails)) | |
| for right in range(left + 1, len(thumbnails)) | |
| ) | |
| def file_fingerprint(path: str | Path) -> str: | |
| """Hash an exported artifact in bounded chunks for manifest provenance.""" | |
| digest = hashlib.sha256() | |
| with Path(path).open("rb") as handle: | |
| for chunk in iter(lambda: handle.read(1024 * 1024), b""): | |
| digest.update(chunk) | |
| return digest.hexdigest() | |
| def da3_api_status() -> dict: | |
| """Check the optional DA3 API without downloading weights or loading a model.""" | |
| try: | |
| _, location = load_da3_class() | |
| return {"installed": True, "importable": True, "api_location": location, "error": None} | |
| except (ImportError, ModuleNotFoundError, AttributeError) as exc: | |
| return {"installed": False, "importable": False, "api_location": None, "error": f"{type(exc).__name__}: {exc}"} | |
| except Exception as exc: | |
| return {"installed": True, "importable": False, "api_location": None, "error": f"{type(exc).__name__}: {exc}"} | |
| def load_da3_class(): | |
| """Resolve the DA3 class across known upstream package layouts.""" | |
| return resolve_da3_class() | |
| def save_failure_manifest(requested_backend: str, stage: str, exc: Exception, input_hashes=None) -> str: | |
| """Persist actionable failure context without storing user media or secrets.""" | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| run_id = uuid.uuid4().hex | |
| manifest = { | |
| "project": "Simam3D", | |
| "version": SIMAM3D_VERSION, | |
| "manifest_schema_version": MANIFEST_SCHEMA_VERSION, | |
| "run_id": run_id, | |
| "status": "failed", | |
| "backend": requested_backend, | |
| "runtime": runtime_label(), | |
| "software": { | |
| "python": platform.python_version(), | |
| "torch": getattr(torch, "__version__", "unknown"), | |
| "gradio": getattr(gr, "__version__", "unknown"), | |
| }, | |
| "stage": stage, | |
| "input_sha256": list(input_hashes or []), | |
| "error": {"type": type(exc).__name__, "message": str(exc)}, | |
| "traceback_tail": traceback.format_exc().splitlines()[-8:], | |
| "claims": { | |
| "observed_input_geometry": False, | |
| "generated_novel_views": False, | |
| "metric_scale_reconstruction": False, | |
| "learned_gaussian_optimization": False, | |
| "failure_report_only": True, | |
| }, | |
| } | |
| path = OUTPUT_DIR / f"simam3d_{run_id}_failure.json" | |
| path.write_text(json.dumps(manifest, indent=2), encoding="utf-8") | |
| return str(path) | |
| def diagnostics() -> str: | |
| """Return actionable environment information without loading a model.""" | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| checks = { | |
| "simam3d_version": SIMAM3D_VERSION, | |
| "runtime": runtime_label(), | |
| "cuda_available": bool(torch.cuda.is_available()), | |
| "torch": getattr(torch, "__version__", "unknown"), | |
| "gradio": getattr(gr, "__version__", "unknown"), | |
| "v2_model": MODEL_ID, | |
| "da3_api": da3_api_status(), | |
| "da3_model": DA3_MODEL_ID, | |
| "output_directory_writable": os.access(OUTPUT_DIR, os.W_OK), | |
| "max_input_side": MAX_INPUT_SIDE, | |
| "max_views": MAX_VIEWS, | |
| "fusion_voxel_size": FUSION_VOXEL_SIZE, | |
| "fusion_max_points": FUSION_MAX_POINTS, | |
| "min_fusion_support_views": MIN_FUSION_SUPPORT_VIEWS, | |
| "da3_process_res": DA3_PROCESS_RES, | |
| "gpu_duration_seconds": GPU_DURATION_SECONDS, | |
| } | |
| if torch.cuda.is_available(): | |
| checks["cuda_device"] = torch.cuda.get_device_name(0) | |
| checks["cuda_memory_gb"] = round(torch.cuda.get_device_properties(0).total_memory / 1024**3, 2) | |
| return "```json\n" + json.dumps(checks, indent=2) + "\n```" | |
| def recent_failure_reports(): | |
| """Return recent redacted failure manifests for download from the UI.""" | |
| OUTPUT_DIR.mkdir(exist_ok=True) | |
| return [str(path) for path in sorted(OUTPUT_DIR.glob("*_failure.json"), key=lambda item: item.stat().st_mtime, reverse=True)[:10]] | |
| def _generate_impl(image: Image.Image, extra_files, backend: str, guess: str, demo_mode: bool, progress=gr.Progress()): | |
| if image is None: | |
| raise gr.Error("Please add an input image first.") | |
| started = time.perf_counter() | |
| stage = "input_validation" | |
| input_hashes = [] | |
| try: | |
| stage = "input_loading" | |
| images = load_input_images(image, extra_files) | |
| view_count = len(images) | |
| input_hashes = [image_fingerprint(view) for view in images] | |
| input_diversity = input_diversity_metrics(input_hashes, near_duplicate_pair_count(images)) | |
| scene_guess = classify_scene_hypothesis(guess) | |
| lighting = analyze_light_field(np.asarray(images[0])) | |
| backend_used = backend | |
| fallback_note = "" | |
| fallback_error = None | |
| confidence_source = "baseline_uniform" | |
| da3_prediction = None | |
| progress(0.05, desc=f"Loading {backend} on {runtime_label()}") | |
| confidence = np.ones((view_count, 1, 1), dtype=np.float32) | |
| extrinsics = synthetic_orbit_poses(view_count) | |
| intrinsics = np.repeat(np.eye(3, dtype=np.float32)[None, ...], view_count, axis=0) | |
| pose_source = "synthetic_orbit_prior" | |
| if demo_mode: | |
| stage = "deterministic_depth" | |
| depth = np.stack([deterministic_demo_depth(np.asarray(view)) for view in images], axis=0) | |
| confidence = np.ones_like(depth, dtype=np.float32) | |
| backend_used = "Deterministic export smoke test (no AI)" | |
| pose_source = "synthetic_orbit_prior" | |
| confidence_source = "uniform_demo" | |
| progress(0.45, desc="Running deterministic export smoke test") | |
| elif backend == "Depth Anything 3 (camera-aware)": | |
| try: | |
| stage = "da3_inference" | |
| with INFERENCE_LOCK, torch.inference_mode(): | |
| prediction = get_da3_model().inference( | |
| image=images, infer_gs=True, process_res=DA3_PROCESS_RES | |
| ) | |
| da3_prediction = prediction | |
| depth, confidence, extrinsics, intrinsics, confidence_source, pose_source = normalize_prediction_contract( | |
| prediction, view_count | |
| ) | |
| except (ImportError, ModuleNotFoundError, RuntimeError, TypeError, ValueError) as exc: | |
| clear_da3_model() | |
| fallback_error = f"{type(exc).__name__}: {exc}" | |
| fallback_note = f" DA3 unavailable ({type(exc).__name__}); used the stable V2 fallback." | |
| print(f"Simam3D DA3 unavailable; falling back to V2: {exc}", flush=True) | |
| gr.Warning(fallback_note.strip()) | |
| backend_used = "Depth Anything V2 (stable baseline; DA3 fallback)" | |
| stage = "v2_fallback_inference" | |
| with INFERENCE_LOCK: | |
| depth = predict_v2(images) | |
| confidence = np.ones_like(depth, dtype=np.float32) | |
| else: | |
| stage = "v2_inference" | |
| with INFERENCE_LOCK: | |
| depth = predict_v2(images) | |
| confidence = np.ones_like(depth, dtype=np.float32) | |
| if depth.ndim != 3 or not np.isfinite(depth).all(): | |
| raise RuntimeError(f"Invalid depth tensor returned with shape {depth.shape}") | |
| stage = "geometry_export" | |
| progress(0.65, desc="Building depth geometry") | |
| run_id = uuid.uuid4().hex | |
| depth_image = make_depth_preview(depth[0]) | |
| confidence_image = make_confidence_preview(confidence[0]) | |
| glb_path = depth_to_glb(images[0], depth[0], run_id) | |
| ply_path = depth_to_pointcloud(images[0], depth[0], confidence[0], run_id) | |
| fused_path, fusion_stats, fused_weights, fused_source_views, fused_source_masks = fuse_views( | |
| images, depth, confidence, extrinsics, intrinsics, run_id | |
| ) | |
| fusion_stats["support_status"] = classify_fusion_support( | |
| fusion_stats, | |
| view_count, | |
| input_diversity["unique_input_view_count"], | |
| pose_source, | |
| input_diversity["near_duplicate_pair_count"], | |
| ) | |
| fused_cloud = trimesh.load(fused_path, process=False) | |
| fused_points = np.asarray(fused_cloud.vertices, dtype=np.float32) | |
| projection_coverage = [] | |
| reprojection_metrics = [] | |
| for idx in range(view_count): | |
| _, _, inside = project_world_points( | |
| fused_points, intrinsics[idx], extrinsics[idx], images[idx].width, images[idx].height | |
| ) | |
| projection_coverage.append(float(inside.mean()) if len(inside) else 0.0) | |
| reprojection_metrics.append( | |
| reprojection_consistency(fused_points, depth[idx], intrinsics[idx], extrinsics[idx]) | |
| ) | |
| turntable_path = save_turntable_gif( | |
| fused_points, | |
| np.asarray(fused_cloud.colors[:, :3], dtype=np.uint8), | |
| run_id, | |
| ) | |
| gaussian_path = pointcloud_to_gaussian_ply( | |
| fused_path, run_id, fused_weights, fused_source_views, fused_source_masks | |
| ) | |
| gaussian_source = "fused_point_initialization" | |
| direct_gaussian_error = None | |
| direct_gaussian_count = 0 | |
| direct_gaussian_exported_count = 0 | |
| if da3_prediction is not None: | |
| try: | |
| direct_result = direct_da3_gaussian_ply(da3_prediction, run_id) | |
| if direct_result is not None: | |
| direct_path, direct_gaussian_count, direct_gaussian_exported_count = direct_result | |
| gaussian_path = direct_path | |
| gaussian_source = "da3_direct" | |
| except (RuntimeError, TypeError, ValueError) as exc: | |
| direct_gaussian_error = f"{type(exc).__name__}: {exc}" | |
| print( | |
| f"Simam3D native Gaussian export unavailable; using fused initialization: {exc}", | |
| flush=True, | |
| ) | |
| gr.Warning("Native DA3 Gaussian export was unavailable; exported fused Gaussian initialization instead.") | |
| npz_path = save_prediction_npz(depth, confidence, extrinsics, intrinsics, run_id) | |
| uncertainty = reveal_uncertainty(depth[0], confidence[0]) | |
| depth_lighting = estimate_depth_light(depth[0], np.asarray(images[0].resize(depth[0].shape[::-1], Image.Resampling.LANCZOS))) | |
| uncertainty_path = save_uncertainty_map(uncertainty, run_id) | |
| view_plan = next_view_plan(view_count, extrinsics) | |
| evidence_card = make_evidence_card( | |
| images[0], depth[0], confidence[0], backend_used, view_count, fusion_stats, run_id | |
| ) | |
| reveal_card = make_reveal_card(images[0], depth[0], scene_guess, lighting, run_id) | |
| analysis_report = ( | |
| "### Behind-the-image analysis\n" | |
| f"**Your hypothesis:** {scene_guess['user_guess']} \n" | |
| f"**Semantic cues:** {', '.join(scene_guess['semantic_cues'])} \n" | |
| f"**Lighting:** {lighting['lighting_label']} \n" | |
| f"**Estimated light direction:** azimuth {lighting['estimated_azimuth_deg']:.0f} deg, " | |
| f"elevation {lighting['estimated_elevation_deg']:.0f} deg \n" | |
| f"**3D light vector:** ({depth_lighting['light_direction_x']:.2f}, " | |
| f"{depth_lighting['light_direction_y']:.2f}, {depth_lighting['light_direction_z']:.2f}) " | |
| f"with confidence {depth_lighting['photometric_confidence']:.2f} \n" | |
| f"**Detector confidence:** {lighting['detector_confidence']:.2f} \n" | |
| f"**Reveal uncertainty:** {uncertainty.mean():.2f} mean visible-space uncertainty \n" | |
| f"**Fused projection coverage:** {np.mean(projection_coverage):.1%} mean across input views \n" | |
| f"**Depth reprojection MAE:** {np.mean([item['mean_absolute_error'] for item in reprojection_metrics]):.3f} " | |
| "relative-depth diagnostic \n" | |
| f"**Input diversity:** {input_diversity['input_diversity_status']} " | |
| f"({input_diversity['unique_input_view_count']}/{input_diversity['input_view_count']} unique view(s)) \n" | |
| f"**Fusion support:** {fusion_stats['support_status']} \n" | |
| f"**Next view suggestion:** {view_plan[0]['yaw_deg']:.0f} deg yaw " | |
| f"({view_plan[0]['gap_from_existing_deg']:.0f} deg gap) \n" | |
| "*This is an image-space lighting estimate and a user-guided scene hypothesis, " | |
| "not verified hidden content.*" | |
| ) | |
| elapsed = time.perf_counter() - started | |
| manifest_path = save_run_manifest( | |
| run_id, | |
| backend_used, | |
| view_count, | |
| elapsed, | |
| [glb_path, ply_path, fused_path, gaussian_path, npz_path, uncertainty_path, turntable_path, evidence_card, reveal_card], | |
| { | |
| "depth_shape": list(depth.shape), | |
| "depth_min": float(depth.min()), | |
| "depth_max": float(depth.max()), | |
| "confidence_mean": float(np.mean(confidence)), | |
| # Preserve the full fusion evidence in the manifest. Keeping | |
| # this nested record avoids reducing a multi-view run to only | |
| # its point count and makes source support auditable. | |
| "fusion": fusion_stats, | |
| "gaussian_source": gaussian_source, | |
| "direct_gaussian_count": direct_gaussian_count, | |
| "direct_gaussian_exported_count": direct_gaussian_exported_count, | |
| "direct_gaussian_error": direct_gaussian_error, | |
| "gaussian_export_max_points": GAUSSIAN_MAX_POINTS, | |
| "da3_process_res": DA3_PROCESS_RES, | |
| "fused_point_count": fusion_stats["point_count"], | |
| "fused_confidence_mean": fusion_stats["confidence_mean"], | |
| "fused_confidence_p95": fusion_stats["confidence_p95"], | |
| "pose_source": pose_source, | |
| "confidence_source": confidence_source, | |
| "requested_backend": backend, | |
| "fallback_error": fallback_error, | |
| "input_sha256": input_hashes, | |
| "input_diversity": input_diversity, | |
| "input_sizes": [list(view.size) for view in images], | |
| "fusion_voxel_size": FUSION_VOXEL_SIZE, | |
| "fusion_max_points": FUSION_MAX_POINTS, | |
| "min_fusion_support_views": MIN_FUSION_SUPPORT_VIEWS, | |
| "scene_hypothesis": scene_guess, | |
| "lighting_analysis": lighting, | |
| "depth_normal_lighting": depth_lighting, | |
| "reveal_uncertainty_mean": float(uncertainty.mean()), | |
| "next_view_plan": view_plan, | |
| "projection_coverage": projection_coverage, | |
| "reprojection_consistency": reprojection_metrics, | |
| }, | |
| ) | |
| progress(1.0, desc="Done") | |
| return depth_image, confidence_image, glb_path, ply_path, fused_path, gaussian_path, npz_path, uncertainty_path, turntable_path, evidence_card, reveal_card, manifest_path, analysis_report, ( | |
| f"Processed {view_count} view(s) on {runtime_label()} in {elapsed:.1f}s using {backend_used}. " | |
| "Per-view and confidence-weighted fused point clouds are exported. " | |
| f"Pose source: {pose_source}; confidence source: {confidence_source}." + fallback_note | |
| ) | |
| except Exception as exc: | |
| print(f"Simam3D generation failed: {type(exc).__name__}: {exc}", flush=True) | |
| failure_path = save_failure_manifest(backend, stage, exc, input_hashes) | |
| raise gr.Error( | |
| f"Generation failed: {type(exc).__name__}: {exc}. Failure manifest: {failure_path}" | |
| ) from exc | |
| def _generate_gpu(image: Image.Image, extra_files, backend: str, guess: str, demo_mode: bool, progress=gr.Progress()): | |
| return _generate_impl(image, extra_files, backend, guess, demo_mode, progress) | |
| def generate(image: Image.Image, extra_files, backend: str, guess: str, demo_mode: bool, progress=gr.Progress()): | |
| """Keep the deterministic export diagnostic independent of GPU quota.""" | |
| if demo_mode: | |
| return _generate_impl(image, extra_files, backend, guess, demo_mode, progress) | |
| return _generate_gpu(image, extra_files, backend, guess, demo_mode, progress) | |
| with gr.Blocks(title="Simam3D") as demo: | |
| gr.Markdown("# Simam3D\nSingle-image depth and 3D reconstruction workbench") | |
| gr.Markdown( | |
| "Phase 1 is deliberately transparent: the original image is lifted using a depth model, " | |
| "then exported as a textured GLB. This gives us a known-good baseline before adding " | |
| "multi-view generation and Gaussian splat fusion." | |
| ) | |
| gr.Markdown( | |
| "**First run:** keep **Depth Anything V2 (stable baseline)** selected. " | |
| "Use **Depth Anything 3 (camera-aware)** when you want to test the heavier " | |
| "multi-view research path; failed DA3 loads fall back to V2 with a warning." | |
| ) | |
| with gr.Row(): | |
| with gr.Column(): | |
| image = gr.Image(type="pil", label="Input image") | |
| extra_files = gr.Files( | |
| label="Optional additional views", | |
| file_types=["image"], | |
| file_count="multiple", | |
| type="filepath", | |
| ) | |
| guess = gr.Textbox( | |
| label="What do you think is behind this image?", | |
| placeholder="e.g. I think there is a garden or a room behind it...", | |
| ) | |
| backend = gr.Dropdown( | |
| choices=["Depth Anything 3 (camera-aware)", "Depth Anything V2 (stable baseline)"], | |
| value="Depth Anything V2 (stable baseline)", | |
| label="Reconstruction backend", | |
| ) | |
| demo_mode = gr.Checkbox( | |
| label="Run export smoke test (no AI model; diagnostic only)", | |
| value=False, | |
| ) | |
| run = gr.Button("Generate depth + 3D", variant="primary") | |
| with gr.Column(): | |
| depth = gr.Image(label="Predicted depth") | |
| confidence_view = gr.Image(label="Confidence diagnostic") | |
| model = gr.File(label="Download depth geometry GLB", file_count="single") | |
| pointcloud = gr.File(label="Download point cloud PLY", file_count="single") | |
| fused = gr.File(label="Download fused point cloud PLY", file_count="single") | |
| gaussians = gr.File(label="Download Gaussian-splat PLY", file_count="single") | |
| metadata = gr.File(label="Download depth/camera metadata NPZ", file_count="single") | |
| uncertainty_file = gr.File(label="Download reveal uncertainty map PNG", file_count="single") | |
| turntable_file = gr.File(label="Download reconstructed turntable GIF", file_count="single") | |
| evidence_card = gr.File(label="Download shareable evidence card PNG", file_count="single") | |
| reveal_card = gr.File(label="Download What's Behind reveal card PNG", file_count="single") | |
| manifest = gr.File(label="Download reproducibility manifest JSON", file_count="single") | |
| analysis_report = gr.Markdown() | |
| status = gr.Markdown(f"Ready. Simam3D **v{SIMAM3D_VERSION}** | Runtime: **{runtime_label()}**") | |
| with gr.Accordion("Diagnostics", open=False): | |
| diagnose = gr.Button("Run environment diagnostics") | |
| diagnostic_report = gr.Markdown() | |
| diagnose.click(diagnostics, outputs=diagnostic_report) | |
| refresh_failures = gr.Button("Refresh recent failure reports") | |
| failure_reports = gr.Files(label="Download recent failure reports", file_count="multiple") | |
| refresh_failures.click(recent_failure_reports, outputs=failure_reports) | |
| run.click( | |
| generate, | |
| inputs=[image, extra_files, backend, guess, demo_mode], | |
| outputs=[depth, confidence_view, model, pointcloud, fused, gaussians, metadata, uncertainty_file, turntable_file, evidence_card, reveal_card, manifest, analysis_report, status], | |
| concurrency_limit=1, | |
| concurrency_id="simam3d_heavy_inference", | |
| ) | |
| if __name__ == "__main__": | |
| print(f"Simam3D v{SIMAM3D_VERSION} starting on {runtime_label()}", flush=True) | |
| demo.launch(show_error=True) | |