Simam3DZeroGPU / app.py
junaid-simamdigital's picture
fix: use modern matplotlib colormap API
8fad076 verified
Raw History Blame Contribute Delete
35.3 kB
from __future__ import annotations
import os
import json
import hashlib
import io
import platform
import sys
import threading
import time
import traceback
import uuid
from pathlib import Path
import gradio as gr
from matplotlib import colormaps
import numpy as np
import torch
import trimesh
from PIL import Image, ImageDraw, ImageOps
from transformers import pipeline
try:
import spaces
GPU_FUNCTION = spaces.GPU(duration=120)
except (ImportError, AttributeError, TypeError):
def GPU_FUNCTION(function):
return function
from simam3d_core import (
analyze_light_field,
classify_scene_hypothesis,
deterministic_demo_depth,
estimate_depth_light,
fuse_point_views,
fusion_metrics,
next_view_plan,
normalize_prediction_contract,
orbit_project_points,
project_world_points,
reprojection_consistency,
reveal_uncertainty,
synthetic_orbit_poses,
write_confidence_ply,
write_gaussian_ply,
)
from da3_compat import resolve_da3_class
MODEL_ID = os.getenv("SIMAM3D_DEPTH_MODEL", "depth-anything/Depth-Anything-V2-Small-hf")
DA3_MODEL_ID = os.getenv("SIMAM3D_DA3_MODEL", "depth-anything/DA3-LARGE")
SIMAM3D_VERSION = "0.3.0"
MANIFEST_SCHEMA_VERSION = 3
OUTPUT_DIR = Path("output")
MAX_VIEWS = int(os.getenv("SIMAM3D_MAX_VIEWS", "8"))
MAX_INPUT_SIDE = int(os.getenv("SIMAM3D_MAX_INPUT_SIDE", "1536"))
FUSION_VOXEL_SIZE = float(os.getenv("SIMAM3D_FUSION_VOXEL_SIZE", "0.02"))
FUSION_MAX_POINTS = int(os.getenv("SIMAM3D_FUSION_MAX_POINTS", "150000"))
DEPTH_PIPELINE = None
DA3_MODEL = None
PIPELINE_LOCK = threading.Lock()
INFERENCE_LOCK = threading.Lock()
def runtime_label() -> str:
if torch.cuda.is_available():
return f"CUDA GPU ({torch.cuda.get_device_name(0)})"
return "CPU"
def get_depth_pipeline():
global DEPTH_PIPELINE
if DEPTH_PIPELINE is not None:
return DEPTH_PIPELINE
with PIPELINE_LOCK:
if DEPTH_PIPELINE is None:
# Keep only one heavyweight backend resident when users switch
# between DA3 and V2 in a long-lived Space process.
clear_da3_model()
device = 0 if torch.cuda.is_available() else -1
DEPTH_PIPELINE = pipeline("depth-estimation", model=MODEL_ID, device=device)
return DEPTH_PIPELINE
def get_da3_model():
global DA3_MODEL
if DA3_MODEL is not None:
return DA3_MODEL
with PIPELINE_LOCK:
if DA3_MODEL is None:
clear_depth_pipeline()
DepthAnything3, _ = load_da3_class()
device = "cuda" if torch.cuda.is_available() else "cpu"
DA3_MODEL = DepthAnything3.from_pretrained(DA3_MODEL_ID).to(device)
DA3_MODEL.eval()
return DA3_MODEL
def clear_da3_model():
global DA3_MODEL
DA3_MODEL = None
if torch.cuda.is_available():
torch.cuda.empty_cache()
def clear_depth_pipeline():
"""Release the Transformers V2 pipeline before loading another backend."""
global DEPTH_PIPELINE
DEPTH_PIPELINE = None
if torch.cuda.is_available():
torch.cuda.empty_cache()
def make_depth_preview(depth: np.ndarray) -> Image.Image:
depth = (depth - depth.min()) / max(float(depth.max() - depth.min()), 1e-6)
rgb = (colormaps.get_cmap("magma")(1.0 - depth)[..., :3] * 255).astype(np.uint8)
return Image.fromarray(rgb)
def make_confidence_preview(confidence: np.ndarray) -> Image.Image:
confidence = np.asarray(confidence, dtype=np.float32)
confidence = np.nan_to_num(confidence, nan=0.0, posinf=1.0, neginf=0.0)
confidence = (confidence - confidence.min()) / max(float(confidence.max() - confidence.min()), 1e-6)
rgb = (colormaps.get_cmap("viridis")(confidence)[..., :3] * 255).astype(np.uint8)
return Image.fromarray(rgb)
def make_uncertainty_preview(uncertainty: np.ndarray) -> Image.Image:
uncertainty = np.clip(np.nan_to_num(np.asarray(uncertainty, dtype=np.float32)), 0.0, 1.0)
rgb = (colormaps.get_cmap("inferno")(uncertainty)[..., :3] * 255).astype(np.uint8)
return Image.fromarray(rgb)
def make_evidence_card(image, depth, confidence, backend, view_count, fusion_stats, run_id):
"""Create a compact, shareable diagnostic card for each reconstruction run."""
tile_size = (480, 300)
source = ImageOps.contain(image.convert("RGB"), tile_size)
depth_tile = ImageOps.contain(make_depth_preview(depth).convert("RGB"), tile_size)
confidence_tile = ImageOps.contain(make_confidence_preview(confidence).convert("RGB"), tile_size)
canvas = Image.new("RGB", (960, 700), (12, 16, 28))
draw = ImageDraw.Draw(canvas)
draw.text((24, 18), "SIMAM3D · EVIDENCE CARD", fill=(235, 240, 255))
draw.text(
(24, 46),
f"{backend} · {view_count} view(s) · {runtime_label()}",
fill=(164, 178, 205),
)
tiles = [(source, (0, 90), "SOURCE"), (depth_tile, (480, 90), "DEPTH"),
(confidence_tile, (0, 390), "CONFIDENCE")]
for tile, origin, label in tiles:
x, y = origin
canvas.paste(tile, (x, y))
draw.rectangle((x, y, x + tile.width, y + tile.height), outline=(76, 93, 126), width=2)
draw.text((x + 12, y + 12), label, fill=(255, 255, 255))
draw.text(
(500, 410),
f"FUSED POINTS\n{fusion_stats['point_count']:,}\n\n"
f"DOMINANT SOURCE VIEWS\n{fusion_stats['dominant_source_view_count']}\n\n"
f"SUPPORTED SOURCE VIEWS\n{fusion_stats['supported_source_view_count']}\n\n"
f"MULTI-VIEW VOXELS\n{fusion_stats['multi_view_voxel_fraction']:.1%}\n\n"
f"CONFIDENCE MEAN\n{fusion_stats['confidence_mean']:.3f}\n\n"
f"CONFIDENCE P95\n{fusion_stats['confidence_p95']:.3f}",
fill=(210, 220, 240),
spacing=8,
)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_evidence_card.png"
canvas.save(path, optimize=True)
return str(path)
def make_reveal_card(image, depth, guess, lighting, run_id: str) -> str:
"""Create a social-friendly reveal card while clearly separating inference from fact."""
tile_size = (480, 300)
source = ImageOps.contain(image.convert("RGB"), tile_size)
depth_tile = ImageOps.contain(make_depth_preview(depth).convert("RGB"), tile_size)
canvas = Image.new("RGB", (960, 760), (12, 16, 28))
draw = ImageDraw.Draw(canvas)
draw.text((24, 18), "SIMAM3D · WHAT'S BEHIND THE IMAGE?", fill=(235, 240, 255))
draw.text((24, 48), "Depth and illumination reveal · hypothesis, not ground truth", fill=(164, 178, 205))
canvas.paste(source, (0, 90))
canvas.paste(depth_tile, (480, 90))
draw.rectangle((0, 90, 480, 390), outline=(76, 93, 126), width=2)
draw.rectangle((480, 90, 960, 390), outline=(76, 93, 126), width=2)
draw.text((14, 104), "ORIGINAL", fill=(255, 255, 255))
draw.text((494, 104), "DEPTH REVEAL", fill=(255, 255, 255))
guess_text = str(guess["user_guess"])
cues = ", ".join(str(item) for item in guess["semantic_cues"])
lighting_text = (
f"LIGHT FIELD: {lighting['lighting_label']}\n"
f"Direction estimate: azimuth {lighting['estimated_azimuth_deg']:.0f}°, "
f"elevation {lighting['estimated_elevation_deg']:.0f}°\n"
f"Detector confidence: {lighting['detector_confidence']:.2f}"
)
draw.text((24, 430), "YOUR GUESS", fill=(133, 214, 255))
draw.text((24, 456), guess_text[:110], fill=(235, 240, 255))
draw.text((24, 500), f"SEMANTIC CUES: {cues}", fill=(210, 220, 240))
draw.text((24, 560), lighting_text, fill=(210, 220, 240), spacing=8)
draw.text((24, 690), "The unseen side is a research hypothesis; add views for evidence.", fill=(255, 190, 120))
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_reveal_card.png"
canvas.save(path, optimize=True)
return str(path)
def depth_to_glb(image: Image.Image, depth: np.ndarray, run_id: str) -> str:
image = image.convert("RGB")
max_side = 192
scale = min(1.0, max_side / max(image.size))
size = (max(8, int(image.width * scale)), max(8, int(image.height * scale)))
rgb = np.asarray(image.resize(size, Image.Resampling.LANCZOS))
d = Image.fromarray(depth).resize(size, Image.Resampling.BILINEAR)
d = np.asarray(d, dtype=np.float32)
d = (d - d.min()) / max(float(d.max() - d.min()), 1e-6)
h, w = d.shape
yy, xx = np.mgrid[0:h, 0:w]
x = (xx / max(w - 1, 1) - 0.5) * 2.0
y = (0.5 - yy / max(h - 1, 1)) * 2.0
z = (1.0 - d) * 0.9
vertices = np.stack([x, y, z], axis=-1).reshape(-1, 3)
faces = []
for row in range(h - 1):
for col in range(w - 1):
i = row * w + col
faces.extend([[i, i + 1, i + w], [i + 1, i + w + 1, i + w]])
mesh = trimesh.Trimesh(vertices=vertices, faces=np.asarray(faces), process=False)
mesh.visual.vertex_colors = np.concatenate(
[rgb.reshape(-1, 3), np.full((vertices.shape[0], 1), 255, dtype=np.uint8)], axis=1
)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}.glb"
mesh.export(path)
return str(path)
def depth_to_pointcloud(image: Image.Image, depth: np.ndarray, confidence: np.ndarray, run_id: str) -> str:
"""Export the current depth scaffold as a colored PLY point cloud."""
image = image.convert("RGB")
max_side = 384
scale = min(1.0, max_side / max(image.size))
size = (max(8, int(image.width * scale)), max(8, int(image.height * scale)))
rgb = np.asarray(image.resize(size, Image.Resampling.LANCZOS))
d = np.asarray(Image.fromarray(depth).resize(size, Image.Resampling.BILINEAR), dtype=np.float32)
d = (d - d.min()) / max(float(d.max() - d.min()), 1e-6)
h, w = d.shape
yy, xx = np.mgrid[0:h, 0:w]
points = np.stack(
[(xx / max(w - 1, 1) - 0.5) * 2.0, (0.5 - yy / max(h - 1, 1)) * 2.0, (1.0 - d) * 0.9],
axis=-1,
).reshape(-1, 3)
confidence = np.asarray(
Image.fromarray(np.asarray(confidence, dtype=np.float32)).resize(size, Image.Resampling.BILINEAR),
dtype=np.float32,
)
confidence = np.nan_to_num(confidence, nan=0.0, posinf=0.0, neginf=0.0)
peak = float(max(confidence.max(), 0.0)) if confidence.size else 0.0
confidence = confidence / peak if peak > 1e-6 else np.ones_like(confidence)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}.ply"
write_confidence_ply(path, points, rgb.reshape(-1, 3), confidence.reshape(-1))
return str(path)
def save_prediction_npz(depth, confidence, extrinsics, intrinsics, run_id: str) -> str:
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}.npz"
np.savez_compressed(
path,
depth=np.asarray(depth, dtype=np.float32),
confidence=np.asarray(confidence, dtype=np.float32),
extrinsics=np.asarray(extrinsics, dtype=np.float32),
intrinsics=np.asarray(intrinsics, dtype=np.float32),
)
return str(path)
def save_uncertainty_map(uncertainty: np.ndarray, run_id: str) -> str:
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_reveal_uncertainty.png"
make_uncertainty_preview(uncertainty).save(path, optimize=True)
return str(path)
def save_turntable_gif(points: np.ndarray, colors: np.ndarray, run_id: str) -> str:
"""Render a lightweight, dependency-free orbit preview from fused points."""
points = np.asarray(points, dtype=np.float32)
colors = np.asarray(colors, dtype=np.uint8)[..., :3]
if len(points) != len(colors):
raise ValueError("turntable points and colors must have matching lengths")
if len(points) > 12000:
keep = np.linspace(0, len(points) - 1, 12000, dtype=np.int64)
points, colors = points[keep], colors[keep]
frames = []
size = 512
for yaw in np.linspace(0.0, 330.0, 12):
xy, depth = orbit_project_points(points, float(yaw))
frame = Image.new("RGB", (size, size), (9, 12, 22))
draw = ImageDraw.Draw(frame)
order = np.argsort(depth)
for index in order:
x = int((float(xy[index, 0]) * 0.43 + 0.5) * size)
y = int((0.5 - float(xy[index, 1]) * 0.43) * size)
if 0 <= x < size and 0 <= y < size:
radius = 1 if len(points) > 4000 else 2
color = tuple(int(value) for value in colors[index])
draw.ellipse((x - radius, y - radius, x + radius, y + radius), fill=color)
draw.text((16, 16), f"SIMAM3D · {float(yaw):.0f}°", fill=(235, 240, 255))
frames.append(frame)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_turntable.gif"
frames[0].save(path, save_all=True, append_images=frames[1:], duration=100, loop=0, optimize=True)
return str(path)
def save_run_manifest(run_id, backend, view_count, elapsed, outputs, metrics) -> str:
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_run.json"
manifest = {
"project": "Simam3D",
"version": SIMAM3D_VERSION,
"manifest_schema_version": MANIFEST_SCHEMA_VERSION,
"run_id": run_id,
"backend": backend,
"model": (
DA3_MODEL_ID if backend.startswith("Depth Anything 3")
else "none (deterministic export smoke test)" if backend.startswith("Deterministic")
else MODEL_ID
),
"runtime": runtime_label(),
"claims": {
"observed_input_geometry": True,
"generated_novel_views": False,
"metric_scale_reconstruction": False,
"learned_gaussian_optimization": False,
"depth_normal_light_is_heuristic": True,
"reveal_content_is_verified": False,
},
"software": {
"python": platform.python_version(),
"platform": platform.platform(),
"torch": getattr(torch, "__version__", "unknown"),
"gradio": getattr(gr, "__version__", "unknown"),
"numpy": np.__version__,
"python_implementation": sys.implementation.name,
},
"view_count": view_count,
"elapsed_seconds": round(elapsed, 3),
"metrics": metrics,
"outputs": outputs,
"output_sha256": {
Path(output).name: file_fingerprint(output)
for output in outputs
if Path(output).is_file()
},
}
path.write_text(json.dumps(manifest, indent=2), encoding="utf-8")
return str(path)
def predict_v2(images):
"""Run the lightweight, widely compatible depth fallback for every view."""
depths = []
for view in images:
with torch.inference_mode():
result = get_depth_pipeline()(view)
predicted = result.get("predicted_depth")
if predicted is None:
raise RuntimeError(f"Depth model returned no predicted_depth field: {list(result)}")
depths.append(predicted.squeeze().detach().float().cpu().numpy().astype(np.float32))
return np.stack(depths, axis=0)
def fuse_views(images, depths, confidence, extrinsics, intrinsics, run_id: str):
"""Fuse depth views into a confidence-weighted world-space PLY."""
views = []
for idx, (image, depth) in enumerate(zip(images, depths)):
depth = np.asarray(depth, dtype=np.float32)
h, w = depth.shape
colors = np.asarray(image.convert("RGB").resize((w, h), Image.Resampling.LANCZOS))
conf = np.asarray(confidence[idx], dtype=np.float32) if len(confidence) > idx else None
if conf is not None and conf.shape != depth.shape:
conf = np.asarray(Image.fromarray(conf).resize((w, h), Image.Resampling.BILINEAR))
camera_k = intrinsics[idx] if len(intrinsics) > idx else None
if camera_k is not None:
camera_k = np.asarray(camera_k, dtype=np.float32).copy()
source_w, source_h = image.size
if camera_k.shape == (3, 3) and source_w and source_h:
camera_k[0, :] *= w / source_w
camera_k[1, :] *= h / source_h
views.append((depth, conf, camera_k,
extrinsics[idx] if len(extrinsics) > idx else None, colors))
result = fuse_point_views(
views, voxel_size=FUSION_VOXEL_SIZE, max_points=FUSION_MAX_POINTS
)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_fused.ply"
write_confidence_ply(
path, result.points, result.colors, result.weights,
result.source_views, result.source_masks
)
return str(path), fusion_metrics(result), result.weights, result.source_views, result.source_masks
def pointcloud_to_gaussian_ply(pointcloud_path: str, run_id: str, weights=None, source_views=None, source_masks=None) -> str:
"""Convert fused colored points to a confidence-initialized 3DGS PLY."""
cloud = trimesh.load(pointcloud_path, process=False)
points = np.asarray(cloud.vertices, dtype=np.float32)
colors = np.asarray(cloud.colors[:, :3], dtype=np.uint8)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_gaussians.ply"
write_gaussian_ply(path, points, colors, weights, source_views, source_masks)
return str(path)
def load_input_images(primary: Image.Image, extra_files) -> list[Image.Image]:
images = [primary.convert("RGB")]
for item in extra_files or []:
path = item if isinstance(item, str) else (
getattr(item, "path", None) or getattr(item, "name", None)
)
if path:
try:
with Image.open(path) as extra:
images.append(extra.convert("RGB").copy())
except (OSError, ValueError) as exc:
raise ValueError(f"Could not read additional view {path!r}: {exc}") from exc
if len(images) > MAX_VIEWS:
raise ValueError(f"At most {MAX_VIEWS} views are supported per run to protect hosted GPU memory.")
normalized = []
for current in images:
if max(current.size) > MAX_INPUT_SIDE:
scale = MAX_INPUT_SIDE / max(current.size)
size = (max(8, int(current.width * scale)), max(8, int(current.height * scale)))
current = current.resize(size, Image.Resampling.LANCZOS)
normalized.append(current)
return normalized
def image_fingerprint(image: Image.Image) -> str:
"""Hash normalized RGB pixels so runs can be reproduced exactly."""
buffer = io.BytesIO()
image.convert("RGB").save(buffer, format="PNG", optimize=False)
return hashlib.sha256(buffer.getvalue()).hexdigest()
def file_fingerprint(path: str | Path) -> str:
"""Hash an exported artifact in bounded chunks for manifest provenance."""
digest = hashlib.sha256()
with Path(path).open("rb") as handle:
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def da3_api_status() -> dict:
"""Check the optional DA3 API without downloading weights or loading a model."""
try:
_, location = load_da3_class()
return {"installed": True, "importable": True, "api_location": location, "error": None}
except (ImportError, ModuleNotFoundError, AttributeError) as exc:
return {"installed": False, "importable": False, "api_location": None, "error": f"{type(exc).__name__}: {exc}"}
except Exception as exc:
return {"installed": True, "importable": False, "api_location": None, "error": f"{type(exc).__name__}: {exc}"}
def load_da3_class():
"""Resolve the DA3 class across known upstream package layouts."""
return resolve_da3_class()
def save_failure_manifest(requested_backend: str, stage: str, exc: Exception, input_hashes=None) -> str:
"""Persist actionable failure context without storing user media or secrets."""
OUTPUT_DIR.mkdir(exist_ok=True)
run_id = uuid.uuid4().hex
manifest = {
"project": "Simam3D",
"version": SIMAM3D_VERSION,
"manifest_schema_version": MANIFEST_SCHEMA_VERSION,
"run_id": run_id,
"status": "failed",
"backend": requested_backend,
"runtime": runtime_label(),
"software": {
"python": platform.python_version(),
"torch": getattr(torch, "__version__", "unknown"),
"gradio": getattr(gr, "__version__", "unknown"),
},
"stage": stage,
"input_sha256": list(input_hashes or []),
"error": {"type": type(exc).__name__, "message": str(exc)},
"traceback_tail": traceback.format_exc().splitlines()[-8:],
"claims": {
"observed_input_geometry": False,
"generated_novel_views": False,
"metric_scale_reconstruction": False,
"learned_gaussian_optimization": False,
"failure_report_only": True,
},
}
path = OUTPUT_DIR / f"simam3d_{run_id}_failure.json"
path.write_text(json.dumps(manifest, indent=2), encoding="utf-8")
return str(path)
def diagnostics() -> str:
"""Return actionable environment information without loading a model."""
OUTPUT_DIR.mkdir(exist_ok=True)
checks = {
"simam3d_version": SIMAM3D_VERSION,
"runtime": runtime_label(),
"cuda_available": bool(torch.cuda.is_available()),
"torch": getattr(torch, "__version__", "unknown"),
"gradio": getattr(gr, "__version__", "unknown"),
"v2_model": MODEL_ID,
"da3_api": da3_api_status(),
"da3_model": DA3_MODEL_ID,
"output_directory_writable": os.access(OUTPUT_DIR, os.W_OK),
"max_input_side": MAX_INPUT_SIDE,
"max_views": MAX_VIEWS,
"fusion_voxel_size": FUSION_VOXEL_SIZE,
}
if torch.cuda.is_available():
checks["cuda_device"] = torch.cuda.get_device_name(0)
checks["cuda_memory_gb"] = round(torch.cuda.get_device_properties(0).total_memory / 1024**3, 2)
return "```json\n" + json.dumps(checks, indent=2) + "\n```"
def recent_failure_reports():
"""Return recent redacted failure manifests for download from the UI."""
OUTPUT_DIR.mkdir(exist_ok=True)
return [str(path) for path in sorted(OUTPUT_DIR.glob("*_failure.json"), key=lambda item: item.stat().st_mtime, reverse=True)[:10]]
@GPU_FUNCTION
def generate(image: Image.Image, extra_files, backend: str, guess: str, demo_mode: bool, progress=gr.Progress()):
if image is None:
raise gr.Error("Please add an input image first.")
started = time.perf_counter()
stage = "input_validation"
input_hashes = []
try:
stage = "input_loading"
images = load_input_images(image, extra_files)
view_count = len(images)
input_hashes = [image_fingerprint(view) for view in images]
scene_guess = classify_scene_hypothesis(guess)
lighting = analyze_light_field(np.asarray(images[0]))
backend_used = backend
fallback_note = ""
fallback_error = None
confidence_source = "baseline_uniform"
progress(0.05, desc=f"Loading {backend} on {runtime_label()}")
confidence = np.ones((view_count, 1, 1), dtype=np.float32)
extrinsics = synthetic_orbit_poses(view_count)
intrinsics = np.repeat(np.eye(3, dtype=np.float32)[None, ...], view_count, axis=0)
pose_source = "synthetic_orbit_prior"
if demo_mode:
stage = "deterministic_depth"
depth = np.stack([deterministic_demo_depth(np.asarray(view)) for view in images], axis=0)
confidence = np.ones_like(depth, dtype=np.float32)
backend_used = "Deterministic export smoke test (no AI)"
pose_source = "synthetic_orbit_prior"
confidence_source = "uniform_demo"
progress(0.45, desc="Running deterministic export smoke test")
elif backend == "Depth Anything 3 (camera-aware)":
try:
stage = "da3_inference"
with INFERENCE_LOCK, torch.inference_mode():
prediction = get_da3_model().inference(image=images)
depth, confidence, extrinsics, intrinsics, confidence_source, pose_source = normalize_prediction_contract(
prediction, view_count
)
except (ImportError, ModuleNotFoundError, RuntimeError, TypeError, ValueError) as exc:
clear_da3_model()
fallback_error = f"{type(exc).__name__}: {exc}"
fallback_note = f" DA3 unavailable ({type(exc).__name__}); used the stable V2 fallback."
print(f"Simam3D DA3 unavailable; falling back to V2: {exc}", flush=True)
gr.Warning(fallback_note.strip())
backend_used = "Depth Anything V2 (stable baseline; DA3 fallback)"
stage = "v2_fallback_inference"
with INFERENCE_LOCK:
depth = predict_v2(images)
confidence = np.ones_like(depth, dtype=np.float32)
else:
stage = "v2_inference"
with INFERENCE_LOCK:
depth = predict_v2(images)
confidence = np.ones_like(depth, dtype=np.float32)
if depth.ndim != 3 or not np.isfinite(depth).all():
raise RuntimeError(f"Invalid depth tensor returned with shape {depth.shape}")
stage = "geometry_export"
progress(0.65, desc="Building depth geometry")
run_id = uuid.uuid4().hex
depth_image = make_depth_preview(depth[0])
confidence_image = make_confidence_preview(confidence[0])
glb_path = depth_to_glb(images[0], depth[0], run_id)
ply_path = depth_to_pointcloud(images[0], depth[0], confidence[0], run_id)
fused_path, fusion_stats, fused_weights, fused_source_views, fused_source_masks = fuse_views(
images, depth, confidence, extrinsics, intrinsics, run_id
)
fused_cloud = trimesh.load(fused_path, process=False)
fused_points = np.asarray(fused_cloud.vertices, dtype=np.float32)
projection_coverage = []
reprojection_metrics = []
for idx in range(view_count):
_, _, inside = project_world_points(
fused_points, intrinsics[idx], extrinsics[idx], images[idx].width, images[idx].height
)
projection_coverage.append(float(inside.mean()) if len(inside) else 0.0)
reprojection_metrics.append(
reprojection_consistency(fused_points, depth[idx], intrinsics[idx], extrinsics[idx])
)
turntable_path = save_turntable_gif(
fused_points,
np.asarray(fused_cloud.colors[:, :3], dtype=np.uint8),
run_id,
)
gaussian_path = pointcloud_to_gaussian_ply(
fused_path, run_id, fused_weights, fused_source_views, fused_source_masks
)
npz_path = save_prediction_npz(depth, confidence, extrinsics, intrinsics, run_id)
uncertainty = reveal_uncertainty(depth[0], confidence[0])
depth_lighting = estimate_depth_light(depth[0], np.asarray(images[0].resize(depth[0].shape[::-1], Image.Resampling.LANCZOS)))
uncertainty_path = save_uncertainty_map(uncertainty, run_id)
view_plan = next_view_plan(view_count)
evidence_card = make_evidence_card(
images[0], depth[0], confidence[0], backend_used, view_count, fusion_stats, run_id
)
reveal_card = make_reveal_card(images[0], depth[0], scene_guess, lighting, run_id)
analysis_report = (
"### Behind-the-image analysis\n"
f"**Your hypothesis:** {scene_guess['user_guess']} \n"
f"**Semantic cues:** {', '.join(scene_guess['semantic_cues'])} \n"
f"**Lighting:** {lighting['lighting_label']} \n"
f"**Estimated light direction:** azimuth {lighting['estimated_azimuth_deg']:.0f}°, "
f"elevation {lighting['estimated_elevation_deg']:.0f}° \n"
f"**3D light vector:** ({depth_lighting['light_direction_x']:.2f}, "
f"{depth_lighting['light_direction_y']:.2f}, {depth_lighting['light_direction_z']:.2f}) "
f"with confidence {depth_lighting['photometric_confidence']:.2f} \n"
f"**Detector confidence:** {lighting['detector_confidence']:.2f} \n"
f"**Reveal uncertainty:** {uncertainty.mean():.2f} mean visible-space uncertainty \n"
f"**Fused projection coverage:** {np.mean(projection_coverage):.1%} mean across input views \n"
f"**Depth reprojection MAE:** {np.mean([item['mean_absolute_error'] for item in reprojection_metrics]):.3f} "
"relative-depth diagnostic \n"
f"**Next view suggestion:** {view_plan[0]['yaw_deg']:.0f}° yaw "
f"({view_plan[0]['gap_from_existing_deg']:.0f}° gap) \n"
"*This is an image-space lighting estimate and a user-guided scene hypothesis, "
"not verified hidden content.*"
)
elapsed = time.perf_counter() - started
manifest_path = save_run_manifest(
run_id,
backend_used,
view_count,
elapsed,
[glb_path, ply_path, fused_path, gaussian_path, npz_path, uncertainty_path, turntable_path, evidence_card, reveal_card],
{
"depth_shape": list(depth.shape),
"depth_min": float(depth.min()),
"depth_max": float(depth.max()),
"confidence_mean": float(np.mean(confidence)),
"fused_point_count": fusion_stats["point_count"],
"fused_confidence_mean": fusion_stats["confidence_mean"],
"fused_confidence_p95": fusion_stats["confidence_p95"],
"pose_source": pose_source,
"confidence_source": confidence_source,
"requested_backend": backend,
"fallback_error": fallback_error,
"input_sha256": input_hashes,
"input_sizes": [list(view.size) for view in images],
"fusion_voxel_size": FUSION_VOXEL_SIZE,
"fusion_max_points": FUSION_MAX_POINTS,
"scene_hypothesis": scene_guess,
"lighting_analysis": lighting,
"depth_normal_lighting": depth_lighting,
"reveal_uncertainty_mean": float(uncertainty.mean()),
"next_view_plan": view_plan,
"projection_coverage": projection_coverage,
"reprojection_consistency": reprojection_metrics,
},
)
progress(1.0, desc="Done")
return depth_image, confidence_image, glb_path, ply_path, fused_path, gaussian_path, npz_path, uncertainty_path, turntable_path, evidence_card, reveal_card, manifest_path, analysis_report, (
f"Processed {view_count} view(s) on {runtime_label()} in {elapsed:.1f}s using {backend_used}. "
"Per-view and confidence-weighted fused point clouds are exported. "
f"Pose source: {pose_source}; confidence source: {confidence_source}." + fallback_note
)
except Exception as exc:
print(f"Simam3D generation failed: {type(exc).__name__}: {exc}", flush=True)
failure_path = save_failure_manifest(backend, stage, exc, input_hashes)
raise gr.Error(
f"Generation failed: {type(exc).__name__}: {exc}. Failure manifest: {failure_path}"
) from exc
with gr.Blocks(title="Simam3D") as demo:
gr.Markdown("# Simam3D\nSingle-image depth and 3D reconstruction workbench")
gr.Markdown(
"Phase 1 is deliberately transparent: the original image is lifted using a depth model, "
"then exported as a textured GLB. This gives us a known-good baseline before adding "
"multi-view generation and Gaussian splat fusion."
)
gr.Markdown(
"**First run:** keep **Depth Anything V2 (stable baseline)** selected. "
"Use **Depth Anything 3 (camera-aware)** when you want to test the heavier "
"multi-view research path; failed DA3 loads fall back to V2 with a warning."
)
with gr.Row():
with gr.Column():
image = gr.Image(type="pil", label="Input image")
extra_files = gr.Files(
label="Optional additional views",
file_types=["image"],
file_count="multiple",
type="filepath",
)
guess = gr.Textbox(
label="What do you think is behind this image?",
placeholder="e.g. I think there is a garden or a room behind it…",
)
backend = gr.Dropdown(
choices=["Depth Anything 3 (camera-aware)", "Depth Anything V2 (stable baseline)"],
value="Depth Anything V2 (stable baseline)",
label="Reconstruction backend",
)
demo_mode = gr.Checkbox(
label="Run export smoke test (no AI model; diagnostic only)",
value=False,
)
run = gr.Button("Generate depth + 3D", variant="primary")
with gr.Column():
depth = gr.Image(label="Predicted depth")
confidence_view = gr.Image(label="Confidence diagnostic")
model = gr.File(label="Download depth geometry GLB", file_count="single")
pointcloud = gr.File(label="Download point cloud PLY", file_count="single")
fused = gr.File(label="Download fused point cloud PLY", file_count="single")
gaussians = gr.File(label="Download Gaussian-splat PLY", file_count="single")
metadata = gr.File(label="Download depth/camera metadata NPZ", file_count="single")
uncertainty_file = gr.File(label="Download reveal uncertainty map PNG", file_count="single")
turntable_file = gr.File(label="Download reconstructed turntable GIF", file_count="single")
evidence_card = gr.File(label="Download shareable evidence card PNG", file_count="single")
reveal_card = gr.File(label="Download What's Behind reveal card PNG", file_count="single")
manifest = gr.File(label="Download reproducibility manifest JSON", file_count="single")
analysis_report = gr.Markdown()
status = gr.Markdown(f"Ready. Simam3D **v{SIMAM3D_VERSION}** · Runtime: **{runtime_label()}**")
with gr.Accordion("Diagnostics", open=False):
diagnose = gr.Button("Run environment diagnostics")
diagnostic_report = gr.Markdown()
diagnose.click(diagnostics, outputs=diagnostic_report)
refresh_failures = gr.Button("Refresh recent failure reports")
failure_reports = gr.Files(label="Download recent failure reports", file_count="multiple")
refresh_failures.click(recent_failure_reports, outputs=failure_reports)
run.click(
generate,
inputs=[image, extra_files, backend, guess, demo_mode],
outputs=[depth, confidence_view, model, pointcloud, fused, gaussians, metadata, uncertainty_file, turntable_file, evidence_card, reveal_card, manifest, analysis_report, status],
concurrency_limit=1,
concurrency_id="simam3d_heavy_inference",
)
if __name__ == "__main__":
print(f"Simam3D v{SIMAM3D_VERSION} starting on {runtime_label()}", flush=True)
demo.launch(show_error=True)