Simam3D-GPU / app.py
junaid-simamdigital's picture
Decouple deterministic smoke from GPU quota
8b048ca verified
Raw History Blame Contribute Delete
43.5 kB
from __future__ import annotations
import os
import json
import hashlib
import io
import platform
import sys
import threading
import time
import traceback
import uuid
from pathlib import Path
for _stream in (sys.stdout, sys.stderr):
if hasattr(_stream, "reconfigure"):
_stream.reconfigure(encoding="utf-8", errors="replace")
import gradio as gr
from matplotlib import colormaps
import numpy as np
import torch
import trimesh
from PIL import Image, ImageDraw, ImageOps
from transformers import pipeline
GPU_DURATION_SECONDS = int(os.getenv("SIMAM3D_GPU_DURATION_SECONDS", "60"))
try:
import spaces
GPU_FUNCTION = spaces.GPU(duration=GPU_DURATION_SECONDS)
except (ImportError, AttributeError, TypeError):
def GPU_FUNCTION(function):
return function
from simam3d_core import (
analyze_light_field,
classify_scene_hypothesis,
deterministic_demo_depth,
classify_fusion_support,
estimate_depth_light,
filter_fusion_by_support,
fuse_point_views,
fusion_metrics,
input_diversity_metrics,
next_view_plan,
normalize_prediction_contract,
orbit_project_points,
project_world_points,
reprojection_consistency,
reveal_uncertainty,
synthetic_orbit_poses,
write_confidence_ply,
write_gaussian_ply,
)
from da3_compat import resolve_da3_class
from validate_manifest import validate_manifest
MODEL_ID = os.getenv("SIMAM3D_DEPTH_MODEL", "depth-anything/Depth-Anything-V2-Small-hf")
DA3_MODEL_ID = os.getenv("SIMAM3D_DA3_MODEL", "depth-anything/DA3NESTED-GIANT-LARGE-1.1")
SIMAM3D_VERSION = "0.4.8"
MANIFEST_SCHEMA_VERSION = 3
OUTPUT_DIR = Path("output")
MAX_VIEWS = int(os.getenv("SIMAM3D_MAX_VIEWS", "8"))
MAX_INPUT_SIDE = int(os.getenv("SIMAM3D_MAX_INPUT_SIDE", "1536"))
FUSION_VOXEL_SIZE = float(os.getenv("SIMAM3D_FUSION_VOXEL_SIZE", "0.02"))
FUSION_MAX_POINTS = int(os.getenv("SIMAM3D_FUSION_MAX_POINTS", "10000"))
MIN_FUSION_SUPPORT_VIEWS = int(os.getenv("SIMAM3D_MIN_FUSION_SUPPORT_VIEWS", "1"))
GAUSSIAN_MAX_POINTS = int(os.getenv("SIMAM3D_GAUSSIAN_MAX_POINTS", "2048"))
DA3_PROCESS_RES = int(os.getenv("SIMAM3D_DA3_PROCESS_RES", "256"))
DEPTH_PIPELINE = None
DA3_MODEL = None
PIPELINE_LOCK = threading.Lock()
INFERENCE_LOCK = threading.Lock()
def runtime_label() -> str:
if torch.cuda.is_available():
return f"CUDA GPU ({torch.cuda.get_device_name(0)})"
return "CPU"
def get_depth_pipeline():
global DEPTH_PIPELINE
if DEPTH_PIPELINE is not None:
return DEPTH_PIPELINE
with PIPELINE_LOCK:
if DEPTH_PIPELINE is None:
# Keep only one heavyweight backend resident when users switch
# between DA3 and V2 in a long-lived Space process.
clear_da3_model()
device = 0 if torch.cuda.is_available() else -1
DEPTH_PIPELINE = pipeline("depth-estimation", model=MODEL_ID, device=device)
return DEPTH_PIPELINE
def get_da3_model():
global DA3_MODEL
if DA3_MODEL is not None:
return DA3_MODEL
with PIPELINE_LOCK:
if DA3_MODEL is None:
clear_depth_pipeline()
DepthAnything3, _ = load_da3_class()
device = "cuda" if torch.cuda.is_available() else "cpu"
DA3_MODEL = DepthAnything3.from_pretrained(DA3_MODEL_ID).to(device)
DA3_MODEL.eval()
return DA3_MODEL
def clear_da3_model():
global DA3_MODEL
DA3_MODEL = None
if torch.cuda.is_available():
torch.cuda.empty_cache()
def clear_depth_pipeline():
"""Release the Transformers V2 pipeline before loading another backend."""
global DEPTH_PIPELINE
DEPTH_PIPELINE = None
if torch.cuda.is_available():
torch.cuda.empty_cache()
def make_depth_preview(depth: np.ndarray) -> Image.Image:
depth = (depth - depth.min()) / max(float(depth.max() - depth.min()), 1e-6)
rgb = (colormaps.get_cmap("magma")(1.0 - depth)[..., :3] * 255).astype(np.uint8)
return Image.fromarray(rgb)
def make_confidence_preview(confidence: np.ndarray) -> Image.Image:
confidence = np.asarray(confidence, dtype=np.float32)
confidence = np.nan_to_num(confidence, nan=0.0, posinf=1.0, neginf=0.0)
confidence = (confidence - confidence.min()) / max(float(confidence.max() - confidence.min()), 1e-6)
rgb = (colormaps.get_cmap("viridis")(confidence)[..., :3] * 255).astype(np.uint8)
return Image.fromarray(rgb)
def make_uncertainty_preview(uncertainty: np.ndarray) -> Image.Image:
uncertainty = np.clip(np.nan_to_num(np.asarray(uncertainty, dtype=np.float32)), 0.0, 1.0)
rgb = (colormaps.get_cmap("inferno")(uncertainty)[..., :3] * 255).astype(np.uint8)
return Image.fromarray(rgb)
def make_evidence_card(image, depth, confidence, backend, view_count, fusion_stats, run_id):
"""Create a compact, shareable diagnostic card for each reconstruction run."""
tile_size = (480, 300)
source = ImageOps.contain(image.convert("RGB"), tile_size)
depth_tile = ImageOps.contain(make_depth_preview(depth).convert("RGB"), tile_size)
confidence_tile = ImageOps.contain(make_confidence_preview(confidence).convert("RGB"), tile_size)
canvas = Image.new("RGB", (960, 700), (12, 16, 28))
draw = ImageDraw.Draw(canvas)
draw.text((24, 18), "SIMAM3D | EVIDENCE CARD", fill=(235, 240, 255))
draw.text(
(24, 46),
f"{backend} | {view_count} view(s) | {runtime_label()}",
fill=(164, 178, 205),
)
tiles = [(source, (0, 90), "SOURCE"), (depth_tile, (480, 90), "DEPTH"),
(confidence_tile, (0, 390), "CONFIDENCE")]
for tile, origin, label in tiles:
x, y = origin
canvas.paste(tile, (x, y))
draw.rectangle((x, y, x + tile.width, y + tile.height), outline=(76, 93, 126), width=2)
draw.text((x + 12, y + 12), label, fill=(255, 255, 255))
draw.text(
(500, 410),
f"FUSED POINTS\n{fusion_stats['point_count']:,}\n\n"
f"DOMINANT SOURCE VIEWS\n{fusion_stats['dominant_source_view_count']}\n\n"
f"SUPPORTED SOURCE VIEWS\n{fusion_stats['supported_source_view_count']}\n\n"
f"MULTI-VIEW VOXELS\n{fusion_stats['multi_view_voxel_fraction']:.1%}\n\n"
f"CONFIDENCE MEAN\n{fusion_stats['confidence_mean']:.3f}\n\n"
f"CONFIDENCE P95\n{fusion_stats['confidence_p95']:.3f}",
fill=(210, 220, 240),
spacing=8,
)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_evidence_card.png"
canvas.save(path, optimize=True)
return str(path)
def make_reveal_card(image, depth, guess, lighting, run_id: str) -> str:
"""Create a social-friendly reveal card while clearly separating inference from fact."""
tile_size = (480, 300)
source = ImageOps.contain(image.convert("RGB"), tile_size)
depth_tile = ImageOps.contain(make_depth_preview(depth).convert("RGB"), tile_size)
canvas = Image.new("RGB", (960, 760), (12, 16, 28))
draw = ImageDraw.Draw(canvas)
draw.text((24, 18), "SIMAM3D | WHAT'S BEHIND THE IMAGE?", fill=(235, 240, 255))
draw.text((24, 48), "Depth and illumination reveal | hypothesis, not ground truth", fill=(164, 178, 205))
canvas.paste(source, (0, 90))
canvas.paste(depth_tile, (480, 90))
draw.rectangle((0, 90, 480, 390), outline=(76, 93, 126), width=2)
draw.rectangle((480, 90, 960, 390), outline=(76, 93, 126), width=2)
draw.text((14, 104), "ORIGINAL", fill=(255, 255, 255))
draw.text((494, 104), "DEPTH REVEAL", fill=(255, 255, 255))
guess_text = str(guess["user_guess"])
cues = ", ".join(str(item) for item in guess["semantic_cues"])
lighting_text = (
f"LIGHT FIELD: {lighting['lighting_label']}\n"
f"Direction estimate: azimuth {lighting['estimated_azimuth_deg']:.0f} deg, "
f"elevation {lighting['estimated_elevation_deg']:.0f} deg\n"
f"Detector confidence: {lighting['detector_confidence']:.2f}"
)
draw.text((24, 430), "YOUR GUESS", fill=(133, 214, 255))
draw.text((24, 456), guess_text[:110], fill=(235, 240, 255))
draw.text((24, 500), f"SEMANTIC CUES: {cues}", fill=(210, 220, 240))
draw.text((24, 560), lighting_text, fill=(210, 220, 240), spacing=8)
draw.text((24, 690), "The unseen side is a research hypothesis; add views for evidence.", fill=(255, 190, 120))
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_reveal_card.png"
canvas.save(path, optimize=True)
return str(path)
def depth_to_glb(image: Image.Image, depth: np.ndarray, run_id: str) -> str:
image = image.convert("RGB")
max_side = 192
scale = min(1.0, max_side / max(image.size))
size = (max(8, int(image.width * scale)), max(8, int(image.height * scale)))
rgb = np.asarray(image.resize(size, Image.Resampling.LANCZOS))
d = Image.fromarray(depth).resize(size, Image.Resampling.BILINEAR)
d = np.asarray(d, dtype=np.float32)
d = (d - d.min()) / max(float(d.max() - d.min()), 1e-6)
h, w = d.shape
yy, xx = np.mgrid[0:h, 0:w]
x = (xx / max(w - 1, 1) - 0.5) * 2.0
y = (0.5 - yy / max(h - 1, 1)) * 2.0
z = (1.0 - d) * 0.9
vertices = np.stack([x, y, z], axis=-1).reshape(-1, 3)
faces = []
for row in range(h - 1):
for col in range(w - 1):
i = row * w + col
faces.extend([[i, i + 1, i + w], [i + 1, i + w + 1, i + w]])
mesh = trimesh.Trimesh(vertices=vertices, faces=np.asarray(faces), process=False)
mesh.visual.vertex_colors = np.concatenate(
[rgb.reshape(-1, 3), np.full((vertices.shape[0], 1), 255, dtype=np.uint8)], axis=1
)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}.glb"
mesh.export(path)
return str(path)
def depth_to_pointcloud(image: Image.Image, depth: np.ndarray, confidence: np.ndarray, run_id: str) -> str:
"""Export the current depth scaffold as a colored PLY point cloud."""
image = image.convert("RGB")
max_side = 384
scale = min(1.0, max_side / max(image.size))
size = (max(8, int(image.width * scale)), max(8, int(image.height * scale)))
rgb = np.asarray(image.resize(size, Image.Resampling.LANCZOS))
d = np.asarray(Image.fromarray(depth).resize(size, Image.Resampling.BILINEAR), dtype=np.float32)
d = (d - d.min()) / max(float(d.max() - d.min()), 1e-6)
h, w = d.shape
yy, xx = np.mgrid[0:h, 0:w]
points = np.stack(
[(xx / max(w - 1, 1) - 0.5) * 2.0, (0.5 - yy / max(h - 1, 1)) * 2.0, (1.0 - d) * 0.9],
axis=-1,
).reshape(-1, 3)
confidence = np.asarray(
Image.fromarray(np.asarray(confidence, dtype=np.float32)).resize(size, Image.Resampling.BILINEAR),
dtype=np.float32,
)
confidence = np.nan_to_num(confidence, nan=0.0, posinf=0.0, neginf=0.0)
peak = float(max(confidence.max(), 0.0)) if confidence.size else 0.0
confidence = confidence / peak if peak > 1e-6 else np.ones_like(confidence)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}.ply"
write_confidence_ply(path, points, rgb.reshape(-1, 3), confidence.reshape(-1))
return str(path)
def save_prediction_npz(depth, confidence, extrinsics, intrinsics, run_id: str) -> str:
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}.npz"
np.savez_compressed(
path,
depth=np.asarray(depth, dtype=np.float32),
confidence=np.asarray(confidence, dtype=np.float32),
extrinsics=np.asarray(extrinsics, dtype=np.float32),
intrinsics=np.asarray(intrinsics, dtype=np.float32),
)
return str(path)
def save_uncertainty_map(uncertainty: np.ndarray, run_id: str) -> str:
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_reveal_uncertainty.png"
make_uncertainty_preview(uncertainty).save(path, optimize=True)
return str(path)
def save_turntable_gif(points: np.ndarray, colors: np.ndarray, run_id: str) -> str:
"""Render a lightweight, dependency-free orbit preview from fused points."""
points = np.asarray(points, dtype=np.float32)
colors = np.asarray(colors, dtype=np.uint8)[..., :3]
if len(points) != len(colors):
raise ValueError("turntable points and colors must have matching lengths")
if len(points) > 12000:
keep = np.linspace(0, len(points) - 1, 12000, dtype=np.int64)
points, colors = points[keep], colors[keep]
frames = []
size = 512
for yaw in np.linspace(0.0, 330.0, 12):
xy, depth = orbit_project_points(points, float(yaw))
frame = Image.new("RGB", (size, size), (9, 12, 22))
draw = ImageDraw.Draw(frame)
order = np.argsort(depth)
for index in order:
x = int((float(xy[index, 0]) * 0.43 + 0.5) * size)
y = int((0.5 - float(xy[index, 1]) * 0.43) * size)
if 0 <= x < size and 0 <= y < size:
radius = 1 if len(points) > 4000 else 2
color = tuple(int(value) for value in colors[index])
draw.ellipse((x - radius, y - radius, x + radius, y + radius), fill=color)
draw.text((16, 16), f"SIMAM3D | {float(yaw):.0f} deg", fill=(235, 240, 255))
frames.append(frame)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_turntable.gif"
frames[0].save(path, save_all=True, append_images=frames[1:], duration=100, loop=0, optimize=True)
return str(path)
def save_run_manifest(run_id, backend, view_count, elapsed, outputs, metrics) -> str:
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_run.json"
manifest = {
"project": "Simam3D",
"version": SIMAM3D_VERSION,
"manifest_schema_version": MANIFEST_SCHEMA_VERSION,
"run_id": run_id,
"backend": backend,
"model": (
DA3_MODEL_ID if backend.startswith("Depth Anything 3")
else "none (deterministic export smoke test)" if backend.startswith("Deterministic")
else MODEL_ID
),
"runtime": runtime_label(),
"claims": {
"observed_input_geometry": True,
"generated_novel_views": False,
"metric_scale_reconstruction": False,
"learned_gaussian_optimization": False,
"direct_gaussian_prediction": metrics.get("gaussian_source") == "da3_direct",
"depth_normal_light_is_heuristic": True,
"reveal_content_is_verified": False,
},
"software": {
"python": platform.python_version(),
"platform": platform.platform(),
"torch": getattr(torch, "__version__", "unknown"),
"gradio": getattr(gr, "__version__", "unknown"),
"numpy": np.__version__,
"python_implementation": sys.implementation.name,
},
"view_count": view_count,
"elapsed_seconds": round(elapsed, 3),
"metrics": metrics,
"outputs": outputs,
"output_sha256": {
Path(output).name: file_fingerprint(output)
for output in outputs
if Path(output).is_file()
},
}
path.write_text(json.dumps(manifest, indent=2), encoding="utf-8")
valid, errors = validate_manifest(path)
if not valid:
raise RuntimeError(
"Generated manifest failed self-validation: " + "; ".join(errors)
)
return str(path)
def predict_v2(images):
"""Run the lightweight, widely compatible depth fallback for every view."""
depths = []
for view in images:
with torch.inference_mode():
result = get_depth_pipeline()(view)
predicted = result.get("predicted_depth")
if predicted is None:
raise RuntimeError(f"Depth model returned no predicted_depth field: {list(result)}")
depths.append(predicted.squeeze().detach().float().cpu().numpy().astype(np.float32))
return np.stack(depths, axis=0)
def fuse_views(images, depths, confidence, extrinsics, intrinsics, run_id: str):
"""Fuse depth views into a confidence-weighted world-space PLY."""
views = []
for idx, (image, depth) in enumerate(zip(images, depths)):
depth = np.asarray(depth, dtype=np.float32)
h, w = depth.shape
colors = np.asarray(image.convert("RGB").resize((w, h), Image.Resampling.LANCZOS))
conf = np.asarray(confidence[idx], dtype=np.float32) if len(confidence) > idx else None
if conf is not None and conf.shape != depth.shape:
conf = np.asarray(Image.fromarray(conf).resize((w, h), Image.Resampling.BILINEAR))
camera_k = intrinsics[idx] if len(intrinsics) > idx else None
if camera_k is not None:
camera_k = np.asarray(camera_k, dtype=np.float32).copy()
source_w, source_h = image.size
if camera_k.shape == (3, 3) and source_w and source_h:
camera_k[0, :] *= w / source_w
camera_k[1, :] *= h / source_h
views.append((depth, conf, camera_k,
extrinsics[idx] if len(extrinsics) > idx else None, colors))
result = fuse_point_views(
views, voxel_size=FUSION_VOXEL_SIZE, max_points=FUSION_MAX_POINTS
)
result = filter_fusion_by_support(result, MIN_FUSION_SUPPORT_VIEWS)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_fused.ply"
write_confidence_ply(
path, result.points, result.colors, result.weights,
result.source_views, result.source_masks
)
return str(path), fusion_metrics(result), result.weights, result.source_views, result.source_masks
def pointcloud_to_gaussian_ply(pointcloud_path: str, run_id: str, weights=None, source_views=None, source_masks=None) -> str:
"""Convert fused colored points to a confidence-initialized 3DGS PLY."""
cloud = trimesh.load(pointcloud_path, process=False)
points = np.asarray(cloud.vertices, dtype=np.float32)
colors = np.asarray(cloud.colors[:, :3], dtype=np.uint8)
OUTPUT_DIR.mkdir(exist_ok=True)
path = OUTPUT_DIR / f"simam3d_{run_id}_gaussians.ply"
write_gaussian_ply(path, points, colors, weights, source_views, source_masks)
return str(path)
def _prediction_array(value):
"""Move an optional DA3 tensor to a CPU NumPy array without torch types."""
if hasattr(value, "detach"):
value = value.detach().cpu().numpy()
return np.asarray(value)
def direct_da3_gaussian_ply(prediction, run_id: str) -> tuple[str, int, int] | None:
"""Export DA3's learned world-space Gaussian parameters when available."""
gaussians = getattr(prediction, "gaussians", None)
if gaussians is None:
return None
print("Simam3D direct Gaussian export: reading parameters", flush=True)
means = _prediction_array(getattr(gaussians, "means", None))
scales = _prediction_array(getattr(gaussians, "scales", None))
rotations = _prediction_array(getattr(gaussians, "rotations", None))
opacities = _prediction_array(getattr(gaussians, "opacities", None)).reshape(-1)
print(f"Simam3D direct Gaussian export: means={means.shape} scales={scales.shape}", flush=True)
if means.ndim == 3 and means.shape[0] == 1:
means = means[0]
if scales.ndim == 3 and scales.shape[0] == 1:
scales = scales[0]
if rotations.ndim == 3 and rotations.shape[0] == 1:
rotations = rotations[0]
if means.ndim != 2 or means.shape[1] != 3:
raise ValueError(f"DA3 Gaussian means must have shape Nx3, got {means.shape}")
if scales.shape != (len(means), 3) or rotations.shape != (len(means), 4) or len(opacities) != len(means):
raise ValueError("DA3 Gaussian parameter arrays do not have matching lengths")
original_count = len(means)
keep = None
if original_count > GAUSSIAN_MAX_POINTS:
# Preserve the most visible splats and restore source order for
# deterministic output. This bounds ASCII export time and file size.
keep = np.argsort(np.nan_to_num(opacities, nan=-np.inf))[-GAUSSIAN_MAX_POINTS:]
keep.sort()
means, scales, rotations, opacities = means[keep], scales[keep], rotations[keep], opacities[keep]
print(f"Simam3D direct Gaussian export: selecting {len(means)} of {original_count}", flush=True)
harmonics = getattr(gaussians, "harmonics", None)
if harmonics is not None:
# Only the DC coefficient is needed for the portable PLY color field;
# avoid transferring the complete SH basis from GPU memory.
if hasattr(harmonics, "detach") and harmonics.ndim >= 3:
harmonics = harmonics[..., 0]
harmonics = _prediction_array(harmonics)
if harmonics.ndim == 4 and harmonics.shape[0] == 1:
harmonics = harmonics[0]
if harmonics.ndim == 3 and harmonics.shape[0] == 1:
harmonics = harmonics[0]
if keep is not None and harmonics.shape[0] == original_count:
harmonics = harmonics[keep]
if harmonics.ndim == 2 and harmonics.shape[0] == len(means):
colors = np.clip((harmonics * 0.2820947918 + 0.5) * 255.0, 0, 255).astype(np.uint8)
else:
colors = np.full((len(means), 3), 180, dtype=np.uint8)
else:
colors = np.full((len(means), 3), 180, dtype=np.uint8)
path = OUTPUT_DIR / f"simam3d_{run_id}_gaussians_da3_direct.ply"
write_gaussian_ply(
path, means, colors, weights=np.clip(opacities, 0.0, 1.0),
scales=scales, rotations=rotations, opacities=opacities,
)
print(f"Simam3D direct Gaussian export: wrote {path}", flush=True)
return str(path), original_count, len(means)
def load_input_images(primary: Image.Image, extra_files) -> list[Image.Image]:
images = [primary.convert("RGB")]
for item in extra_files or []:
path = item if isinstance(item, str) else (
getattr(item, "path", None) or getattr(item, "name", None)
)
if path:
try:
with Image.open(path) as extra:
images.append(extra.convert("RGB").copy())
except (OSError, ValueError) as exc:
raise ValueError(f"Could not read additional view {path!r}: {exc}") from exc
if len(images) > MAX_VIEWS:
raise ValueError(f"At most {MAX_VIEWS} views are supported per run to protect hosted GPU memory.")
normalized = []
for current in images:
if max(current.size) > MAX_INPUT_SIDE:
scale = MAX_INPUT_SIDE / max(current.size)
size = (max(8, int(current.width * scale)), max(8, int(current.height * scale)))
current = current.resize(size, Image.Resampling.LANCZOS)
normalized.append(current)
return normalized
def image_fingerprint(image: Image.Image) -> str:
"""Hash normalized RGB pixels so runs can be reproduced exactly."""
buffer = io.BytesIO()
image.convert("RGB").save(buffer, format="PNG", optimize=False)
return hashlib.sha256(buffer.getvalue()).hexdigest()
def near_duplicate_pair_count(images: list[Image.Image], threshold: float = 0.03) -> int:
"""Count pairs with tiny normalized RGB distance for view-quality diagnostics."""
thumbnails = [
np.asarray(ImageOps.fit(image.convert("RGB"), (32, 32)), dtype=np.float32) / 255.0
for image in images
]
return sum(
float(np.abs(thumbnails[left] - thumbnails[right]).mean()) <= threshold
for left in range(len(thumbnails))
for right in range(left + 1, len(thumbnails))
)
def file_fingerprint(path: str | Path) -> str:
"""Hash an exported artifact in bounded chunks for manifest provenance."""
digest = hashlib.sha256()
with Path(path).open("rb") as handle:
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def da3_api_status() -> dict:
"""Check the optional DA3 API without downloading weights or loading a model."""
try:
_, location = load_da3_class()
return {"installed": True, "importable": True, "api_location": location, "error": None}
except (ImportError, ModuleNotFoundError, AttributeError) as exc:
return {"installed": False, "importable": False, "api_location": None, "error": f"{type(exc).__name__}: {exc}"}
except Exception as exc:
return {"installed": True, "importable": False, "api_location": None, "error": f"{type(exc).__name__}: {exc}"}
def load_da3_class():
"""Resolve the DA3 class across known upstream package layouts."""
return resolve_da3_class()
def save_failure_manifest(requested_backend: str, stage: str, exc: Exception, input_hashes=None) -> str:
"""Persist actionable failure context without storing user media or secrets."""
OUTPUT_DIR.mkdir(exist_ok=True)
run_id = uuid.uuid4().hex
manifest = {
"project": "Simam3D",
"version": SIMAM3D_VERSION,
"manifest_schema_version": MANIFEST_SCHEMA_VERSION,
"run_id": run_id,
"status": "failed",
"backend": requested_backend,
"runtime": runtime_label(),
"software": {
"python": platform.python_version(),
"torch": getattr(torch, "__version__", "unknown"),
"gradio": getattr(gr, "__version__", "unknown"),
},
"stage": stage,
"input_sha256": list(input_hashes or []),
"error": {"type": type(exc).__name__, "message": str(exc)},
"traceback_tail": traceback.format_exc().splitlines()[-8:],
"claims": {
"observed_input_geometry": False,
"generated_novel_views": False,
"metric_scale_reconstruction": False,
"learned_gaussian_optimization": False,
"failure_report_only": True,
},
}
path = OUTPUT_DIR / f"simam3d_{run_id}_failure.json"
path.write_text(json.dumps(manifest, indent=2), encoding="utf-8")
return str(path)
def diagnostics() -> str:
"""Return actionable environment information without loading a model."""
OUTPUT_DIR.mkdir(exist_ok=True)
checks = {
"simam3d_version": SIMAM3D_VERSION,
"runtime": runtime_label(),
"cuda_available": bool(torch.cuda.is_available()),
"torch": getattr(torch, "__version__", "unknown"),
"gradio": getattr(gr, "__version__", "unknown"),
"v2_model": MODEL_ID,
"da3_api": da3_api_status(),
"da3_model": DA3_MODEL_ID,
"output_directory_writable": os.access(OUTPUT_DIR, os.W_OK),
"max_input_side": MAX_INPUT_SIDE,
"max_views": MAX_VIEWS,
"fusion_voxel_size": FUSION_VOXEL_SIZE,
"fusion_max_points": FUSION_MAX_POINTS,
"min_fusion_support_views": MIN_FUSION_SUPPORT_VIEWS,
"da3_process_res": DA3_PROCESS_RES,
"gpu_duration_seconds": GPU_DURATION_SECONDS,
}
if torch.cuda.is_available():
checks["cuda_device"] = torch.cuda.get_device_name(0)
checks["cuda_memory_gb"] = round(torch.cuda.get_device_properties(0).total_memory / 1024**3, 2)
return "```json\n" + json.dumps(checks, indent=2) + "\n```"
def recent_failure_reports():
"""Return recent redacted failure manifests for download from the UI."""
OUTPUT_DIR.mkdir(exist_ok=True)
return [str(path) for path in sorted(OUTPUT_DIR.glob("*_failure.json"), key=lambda item: item.stat().st_mtime, reverse=True)[:10]]
def _generate_impl(image: Image.Image, extra_files, backend: str, guess: str, demo_mode: bool, progress=gr.Progress()):
if image is None:
raise gr.Error("Please add an input image first.")
started = time.perf_counter()
stage = "input_validation"
input_hashes = []
try:
stage = "input_loading"
images = load_input_images(image, extra_files)
view_count = len(images)
input_hashes = [image_fingerprint(view) for view in images]
input_diversity = input_diversity_metrics(input_hashes, near_duplicate_pair_count(images))
scene_guess = classify_scene_hypothesis(guess)
lighting = analyze_light_field(np.asarray(images[0]))
backend_used = backend
fallback_note = ""
fallback_error = None
confidence_source = "baseline_uniform"
da3_prediction = None
progress(0.05, desc=f"Loading {backend} on {runtime_label()}")
confidence = np.ones((view_count, 1, 1), dtype=np.float32)
extrinsics = synthetic_orbit_poses(view_count)
intrinsics = np.repeat(np.eye(3, dtype=np.float32)[None, ...], view_count, axis=0)
pose_source = "synthetic_orbit_prior"
if demo_mode:
stage = "deterministic_depth"
depth = np.stack([deterministic_demo_depth(np.asarray(view)) for view in images], axis=0)
confidence = np.ones_like(depth, dtype=np.float32)
backend_used = "Deterministic export smoke test (no AI)"
pose_source = "synthetic_orbit_prior"
confidence_source = "uniform_demo"
progress(0.45, desc="Running deterministic export smoke test")
elif backend == "Depth Anything 3 (camera-aware)":
try:
stage = "da3_inference"
with INFERENCE_LOCK, torch.inference_mode():
prediction = get_da3_model().inference(
image=images, infer_gs=True, process_res=DA3_PROCESS_RES
)
da3_prediction = prediction
depth, confidence, extrinsics, intrinsics, confidence_source, pose_source = normalize_prediction_contract(
prediction, view_count
)
except (ImportError, ModuleNotFoundError, RuntimeError, TypeError, ValueError) as exc:
clear_da3_model()
fallback_error = f"{type(exc).__name__}: {exc}"
fallback_note = f" DA3 unavailable ({type(exc).__name__}); used the stable V2 fallback."
print(f"Simam3D DA3 unavailable; falling back to V2: {exc}", flush=True)
gr.Warning(fallback_note.strip())
backend_used = "Depth Anything V2 (stable baseline; DA3 fallback)"
stage = "v2_fallback_inference"
with INFERENCE_LOCK:
depth = predict_v2(images)
confidence = np.ones_like(depth, dtype=np.float32)
else:
stage = "v2_inference"
with INFERENCE_LOCK:
depth = predict_v2(images)
confidence = np.ones_like(depth, dtype=np.float32)
if depth.ndim != 3 or not np.isfinite(depth).all():
raise RuntimeError(f"Invalid depth tensor returned with shape {depth.shape}")
stage = "geometry_export"
progress(0.65, desc="Building depth geometry")
run_id = uuid.uuid4().hex
depth_image = make_depth_preview(depth[0])
confidence_image = make_confidence_preview(confidence[0])
glb_path = depth_to_glb(images[0], depth[0], run_id)
ply_path = depth_to_pointcloud(images[0], depth[0], confidence[0], run_id)
fused_path, fusion_stats, fused_weights, fused_source_views, fused_source_masks = fuse_views(
images, depth, confidence, extrinsics, intrinsics, run_id
)
fusion_stats["support_status"] = classify_fusion_support(
fusion_stats,
view_count,
input_diversity["unique_input_view_count"],
pose_source,
input_diversity["near_duplicate_pair_count"],
)
fused_cloud = trimesh.load(fused_path, process=False)
fused_points = np.asarray(fused_cloud.vertices, dtype=np.float32)
projection_coverage = []
reprojection_metrics = []
for idx in range(view_count):
_, _, inside = project_world_points(
fused_points, intrinsics[idx], extrinsics[idx], images[idx].width, images[idx].height
)
projection_coverage.append(float(inside.mean()) if len(inside) else 0.0)
reprojection_metrics.append(
reprojection_consistency(fused_points, depth[idx], intrinsics[idx], extrinsics[idx])
)
turntable_path = save_turntable_gif(
fused_points,
np.asarray(fused_cloud.colors[:, :3], dtype=np.uint8),
run_id,
)
gaussian_path = pointcloud_to_gaussian_ply(
fused_path, run_id, fused_weights, fused_source_views, fused_source_masks
)
gaussian_source = "fused_point_initialization"
direct_gaussian_error = None
direct_gaussian_count = 0
direct_gaussian_exported_count = 0
if da3_prediction is not None:
try:
direct_result = direct_da3_gaussian_ply(da3_prediction, run_id)
if direct_result is not None:
direct_path, direct_gaussian_count, direct_gaussian_exported_count = direct_result
gaussian_path = direct_path
gaussian_source = "da3_direct"
except (RuntimeError, TypeError, ValueError) as exc:
direct_gaussian_error = f"{type(exc).__name__}: {exc}"
print(
f"Simam3D native Gaussian export unavailable; using fused initialization: {exc}",
flush=True,
)
gr.Warning("Native DA3 Gaussian export was unavailable; exported fused Gaussian initialization instead.")
npz_path = save_prediction_npz(depth, confidence, extrinsics, intrinsics, run_id)
uncertainty = reveal_uncertainty(depth[0], confidence[0])
depth_lighting = estimate_depth_light(depth[0], np.asarray(images[0].resize(depth[0].shape[::-1], Image.Resampling.LANCZOS)))
uncertainty_path = save_uncertainty_map(uncertainty, run_id)
view_plan = next_view_plan(view_count, extrinsics)
evidence_card = make_evidence_card(
images[0], depth[0], confidence[0], backend_used, view_count, fusion_stats, run_id
)
reveal_card = make_reveal_card(images[0], depth[0], scene_guess, lighting, run_id)
analysis_report = (
"### Behind-the-image analysis\n"
f"**Your hypothesis:** {scene_guess['user_guess']} \n"
f"**Semantic cues:** {', '.join(scene_guess['semantic_cues'])} \n"
f"**Lighting:** {lighting['lighting_label']} \n"
f"**Estimated light direction:** azimuth {lighting['estimated_azimuth_deg']:.0f} deg, "
f"elevation {lighting['estimated_elevation_deg']:.0f} deg \n"
f"**3D light vector:** ({depth_lighting['light_direction_x']:.2f}, "
f"{depth_lighting['light_direction_y']:.2f}, {depth_lighting['light_direction_z']:.2f}) "
f"with confidence {depth_lighting['photometric_confidence']:.2f} \n"
f"**Detector confidence:** {lighting['detector_confidence']:.2f} \n"
f"**Reveal uncertainty:** {uncertainty.mean():.2f} mean visible-space uncertainty \n"
f"**Fused projection coverage:** {np.mean(projection_coverage):.1%} mean across input views \n"
f"**Depth reprojection MAE:** {np.mean([item['mean_absolute_error'] for item in reprojection_metrics]):.3f} "
"relative-depth diagnostic \n"
f"**Input diversity:** {input_diversity['input_diversity_status']} "
f"({input_diversity['unique_input_view_count']}/{input_diversity['input_view_count']} unique view(s)) \n"
f"**Fusion support:** {fusion_stats['support_status']} \n"
f"**Next view suggestion:** {view_plan[0]['yaw_deg']:.0f} deg yaw "
f"({view_plan[0]['gap_from_existing_deg']:.0f} deg gap) \n"
"*This is an image-space lighting estimate and a user-guided scene hypothesis, "
"not verified hidden content.*"
)
elapsed = time.perf_counter() - started
manifest_path = save_run_manifest(
run_id,
backend_used,
view_count,
elapsed,
[glb_path, ply_path, fused_path, gaussian_path, npz_path, uncertainty_path, turntable_path, evidence_card, reveal_card],
{
"depth_shape": list(depth.shape),
"depth_min": float(depth.min()),
"depth_max": float(depth.max()),
"confidence_mean": float(np.mean(confidence)),
# Preserve the full fusion evidence in the manifest. Keeping
# this nested record avoids reducing a multi-view run to only
# its point count and makes source support auditable.
"fusion": fusion_stats,
"gaussian_source": gaussian_source,
"direct_gaussian_count": direct_gaussian_count,
"direct_gaussian_exported_count": direct_gaussian_exported_count,
"direct_gaussian_error": direct_gaussian_error,
"gaussian_export_max_points": GAUSSIAN_MAX_POINTS,
"da3_process_res": DA3_PROCESS_RES,
"fused_point_count": fusion_stats["point_count"],
"fused_confidence_mean": fusion_stats["confidence_mean"],
"fused_confidence_p95": fusion_stats["confidence_p95"],
"pose_source": pose_source,
"confidence_source": confidence_source,
"requested_backend": backend,
"fallback_error": fallback_error,
"input_sha256": input_hashes,
"input_diversity": input_diversity,
"input_sizes": [list(view.size) for view in images],
"fusion_voxel_size": FUSION_VOXEL_SIZE,
"fusion_max_points": FUSION_MAX_POINTS,
"min_fusion_support_views": MIN_FUSION_SUPPORT_VIEWS,
"scene_hypothesis": scene_guess,
"lighting_analysis": lighting,
"depth_normal_lighting": depth_lighting,
"reveal_uncertainty_mean": float(uncertainty.mean()),
"next_view_plan": view_plan,
"projection_coverage": projection_coverage,
"reprojection_consistency": reprojection_metrics,
},
)
progress(1.0, desc="Done")
return depth_image, confidence_image, glb_path, ply_path, fused_path, gaussian_path, npz_path, uncertainty_path, turntable_path, evidence_card, reveal_card, manifest_path, analysis_report, (
f"Processed {view_count} view(s) on {runtime_label()} in {elapsed:.1f}s using {backend_used}. "
"Per-view and confidence-weighted fused point clouds are exported. "
f"Pose source: {pose_source}; confidence source: {confidence_source}." + fallback_note
)
except Exception as exc:
print(f"Simam3D generation failed: {type(exc).__name__}: {exc}", flush=True)
failure_path = save_failure_manifest(backend, stage, exc, input_hashes)
raise gr.Error(
f"Generation failed: {type(exc).__name__}: {exc}. Failure manifest: {failure_path}"
) from exc
@GPU_FUNCTION
def _generate_gpu(image: Image.Image, extra_files, backend: str, guess: str, demo_mode: bool, progress=gr.Progress()):
return _generate_impl(image, extra_files, backend, guess, demo_mode, progress)
def generate(image: Image.Image, extra_files, backend: str, guess: str, demo_mode: bool, progress=gr.Progress()):
"""Keep the deterministic export diagnostic independent of GPU quota."""
if demo_mode:
return _generate_impl(image, extra_files, backend, guess, demo_mode, progress)
return _generate_gpu(image, extra_files, backend, guess, demo_mode, progress)
with gr.Blocks(title="Simam3D") as demo:
gr.Markdown("# Simam3D\nSingle-image depth and 3D reconstruction workbench")
gr.Markdown(
"Phase 1 is deliberately transparent: the original image is lifted using a depth model, "
"then exported as a textured GLB. This gives us a known-good baseline before adding "
"multi-view generation and Gaussian splat fusion."
)
gr.Markdown(
"**First run:** keep **Depth Anything V2 (stable baseline)** selected. "
"Use **Depth Anything 3 (camera-aware)** when you want to test the heavier "
"multi-view research path; failed DA3 loads fall back to V2 with a warning."
)
with gr.Row():
with gr.Column():
image = gr.Image(type="pil", label="Input image")
extra_files = gr.Files(
label="Optional additional views",
file_types=["image"],
file_count="multiple",
type="filepath",
)
guess = gr.Textbox(
label="What do you think is behind this image?",
placeholder="e.g. I think there is a garden or a room behind it...",
)
backend = gr.Dropdown(
choices=["Depth Anything 3 (camera-aware)", "Depth Anything V2 (stable baseline)"],
value="Depth Anything V2 (stable baseline)",
label="Reconstruction backend",
)
demo_mode = gr.Checkbox(
label="Run export smoke test (no AI model; diagnostic only)",
value=False,
)
run = gr.Button("Generate depth + 3D", variant="primary")
with gr.Column():
depth = gr.Image(label="Predicted depth")
confidence_view = gr.Image(label="Confidence diagnostic")
model = gr.File(label="Download depth geometry GLB", file_count="single")
pointcloud = gr.File(label="Download point cloud PLY", file_count="single")
fused = gr.File(label="Download fused point cloud PLY", file_count="single")
gaussians = gr.File(label="Download Gaussian-splat PLY", file_count="single")
metadata = gr.File(label="Download depth/camera metadata NPZ", file_count="single")
uncertainty_file = gr.File(label="Download reveal uncertainty map PNG", file_count="single")
turntable_file = gr.File(label="Download reconstructed turntable GIF", file_count="single")
evidence_card = gr.File(label="Download shareable evidence card PNG", file_count="single")
reveal_card = gr.File(label="Download What's Behind reveal card PNG", file_count="single")
manifest = gr.File(label="Download reproducibility manifest JSON", file_count="single")
analysis_report = gr.Markdown()
status = gr.Markdown(f"Ready. Simam3D **v{SIMAM3D_VERSION}** | Runtime: **{runtime_label()}**")
with gr.Accordion("Diagnostics", open=False):
diagnose = gr.Button("Run environment diagnostics")
diagnostic_report = gr.Markdown()
diagnose.click(diagnostics, outputs=diagnostic_report)
refresh_failures = gr.Button("Refresh recent failure reports")
failure_reports = gr.Files(label="Download recent failure reports", file_count="multiple")
refresh_failures.click(recent_failure_reports, outputs=failure_reports)
run.click(
generate,
inputs=[image, extra_files, backend, guess, demo_mode],
outputs=[depth, confidence_view, model, pointcloud, fused, gaussians, metadata, uncertainty_file, turntable_file, evidence_card, reveal_card, manifest, analysis_report, status],
concurrency_limit=1,
concurrency_id="simam3d_heavy_inference",
)
if __name__ == "__main__":
print(f"Simam3D v{SIMAM3D_VERSION} starting on {runtime_label()}", flush=True)
demo.launch(show_error=True)