""" Aerix — Hugging Face Spaces Gradio App with ZeroGPU Wraps the inference.py pipeline as a Gradio interface with automatic A100 GPU allocation via @spaces.GPU decorator. Key fix vs the previous version: CPU-bound work (loading/downsampling the image, building tiles, postprocessing/rooftop classification) now happens OUTSIDE the @spaces.GPU-decorated functions. Only the actual tensor forward pass runs inside them. This is what was causing crashes/hangs on large TIFF uploads — a big GeoTIFF's decode+resize could take longer than the ZeroGPU duration window, or blow past available RAM before the GPU was even attached. """ import io import gc import base64 import numpy as np from pathlib import Path from PIL import Image import cv2 import torch try: import spaces except ImportError: class _SpacesFallback: @staticmethod def GPU(duration=0): def decorator(func): return func return decorator spaces = _SpacesFallback() import gradio as gr from inference import AerixSegmentationModel, get_image_info, validate_upload # --------------------------------------------------------------------------- # Model initialisation (module-level, CPU — ZeroGPU attaches the GPU later, # only for the duration of a @spaces.GPU-decorated call) # --------------------------------------------------------------------------- print("Initialising AerixSegmentationModel …") model = AerixSegmentationModel(models_dir="models") print(f"Model ready on device: {model.device}") # Single tile vs sliding-window routing threshold (post-downsample px, longest side) TILE_ROUTE_THRESHOLD = 512 MAX_DISPLAY_DIM = 2048 def numpy_to_base64(arr: np.ndarray, fmt: str = "PNG") -> str: img = Image.fromarray(arr) buf = io.BytesIO() img.save(buf, format=fmt) return base64.b64encode(buf.getvalue()).decode("utf-8") # --------------------------------------------------------------------------- # Available sample tiles # --------------------------------------------------------------------------- TILES_DIR = Path("sample_data/tiles") TILE_CHOICES = ["none"] if TILES_DIR.exists(): TILE_CHOICES += [ f.name for f in sorted(TILES_DIR.glob("*.png")) if "mask" not in f.name and "gt_" not in f.name ] # --------------------------------------------------------------------------- # GPU stages — kept as thin as possible. These are the ONLY functions that # touch CUDA. Everything else in this file is plain CPU code. # --------------------------------------------------------------------------- @spaces.GPU(duration=20) def _gpu_run_single(feature: str, input_array: np.ndarray) -> np.ndarray: return model.run_model(feature, input_array) @spaces.GPU(duration=60) def _gpu_run_tiles(feature: str, chips): return model.run_model_tiles(feature, chips) # --------------------------------------------------------------------------- # Main handler — plain function, NOT @spaces.GPU. It does CPU work directly # and only reaches into the GPU stages above for the actual model forward. # --------------------------------------------------------------------------- def predict(image, feature, threshold, sample_tile): """ Run UNet++ segmentation on a drone orthophoto. """ if sample_tile and sample_tile != "none": image_path = str(TILES_DIR / sample_tile) if not Path(image_path).exists(): return {"error": f"Sample tile not found: {sample_tile}"} elif image is not None: image_path = image else: return {"error": "No image provided. Upload an image or select a sample tile."} # --- Fast, cheap rejection before touching any pixel data ------------- try: info = get_image_info(image_path) validate_upload(image_path) except ValueError as e: return {"error": str(e)} except Exception as e: return {"error": f"Could not read file: {e}"} try: # ---- CPU stage: load + downsample (rasterio windowed read for TIFF) original_image = _load_for_routing(image_path) h, w = original_image.shape[:2] if max(h, w) <= TILE_ROUTE_THRESHOLD: # ---- small image: single-tile path resized = cv2.resize(original_image, (512, 512)) input_array = resized.astype(np.float32) / 255.0 raw = _gpu_run_single(feature, input_array) # GPU raw_full = cv2.resize(raw, (w, h)) result = model.postprocess(original_image, raw_full, feature, threshold) # CPU else: # ---- large orthomosaic: tiled / sliding-window path tiles = model.prepare_tiles(original_image) # CPU preds = _gpu_run_tiles(feature, tiles["chips"]) # GPU mask = model.stitch_tiles(preds, tiles["coords"], tiles["weight_map"], tiles["image_shape"], threshold) # CPU overlay = model._create_overlay(original_image, mask) detected_pixels = int(np.sum(mask > 0)) total_pixels = int(mask.size) rooftop_analysis = ( model.classify_rooftops(original_image, mask) if feature == "buildings" and detected_pixels > 0 else None ) result = { "original_image": original_image, "mask": mask, "overlay": overlay, "feature": feature, "detected_ratio": detected_pixels / total_pixels if total_pixels else 0, "detected_pixels": detected_pixels, "total_pixels": total_pixels, "rooftop_analysis": rooftop_analysis, } response = { "feature": result["feature"], "detected_ratio": float(result["detected_ratio"]), "detected_pixels": int(result["detected_pixels"]), "total_pixels": int(result["total_pixels"]), "original_image": numpy_to_base64(result["original_image"]), "mask": numpy_to_base64(result["mask"]), "overlay": numpy_to_base64(result["overlay"]), } if result.get("rooftop_analysis"): ra = result["rooftop_analysis"] response["rooftop_analysis"] = { "rooftop_counts": ra["rooftop_counts"], "total_buildings": ra["total_buildings"], "total_builtup_m2": ra["total_builtup_m2"], "rcc_area_m2": ra["rcc_area_m2"], "annual_solar_kwh": ra["annual_solar_kwh"], "annual_property_tax_inr": ra["annual_property_tax_inr"], "classification_map": numpy_to_base64(ra["classification_map"]), "building_details": ra["building_details"][:50], } return response except Exception as e: return {"error": str(e)} finally: gc.collect() if torch.cuda.is_available(): torch.cuda.empty_cache() def _load_for_routing(image_path: str) -> np.ndarray: """CPU-only load, used to decide single-tile vs sliding-window routing.""" from inference import load_robust_image return load_robust_image(image_path, max_dim=MAX_DISPLAY_DIM) # --------------------------------------------------------------------------- # Gradio Interface # --------------------------------------------------------------------------- demo = gr.Interface( fn=predict, inputs=[ gr.Image(type="filepath", label="Drone Orthophoto"), gr.Dropdown(choices=["buildings", "roads", "water_bodies"], value="buildings", label="Feature Class"), gr.Slider(minimum=0.1, maximum=0.9, step=0.05, value=0.5, label="Confidence Threshold"), gr.Dropdown(choices=TILE_CHOICES, value="none", label="Sample Tile"), ], outputs=gr.JSON(label="Inference Results"), title="🛰️ Aerix — SVAMITVA AI Segmentation", description=( "UNet++ deep learning inference on 50 cm GSD drone orthophotos for " "building footprint demarcation, road network vectorisation, and " "waterbody extraction. Part of SIH Problem Statement 1705 — " "Ministry of Panchayati Raj, Government of India.\n\n" "Uploads over 1 GB are rejected up front — downsample or crop first." ), api_name="predict", flagging_mode="never", ) if __name__ == "__main__": # max_file_size caps the raw upload so an oversized file fails fast at the # Gradio layer instead of hanging the request indefinitely. demo.queue().launch(max_file_size="1gb")