ShadeNet-3.2-5M (model): weights, ONNX, app + card
Browse files- .gitattributes +25 -0
- README.md +203 -0
- app.py +59 -0
- assets/103195344_5d2dc613a3_result.png +3 -0
- assets/154871781_ae77696b77_result.png +3 -0
- assets/1561658940_a947f2446a_result.png +3 -0
- assets/157139628_5dc483e2e4_result.png +3 -0
- assets/159712188_d530dd478c_result.png +3 -0
- assets/160541986_d5be2ab4c1_result.png +3 -0
- assets/160566014_59528ff897_result.png +3 -0
- assets/160585932_fa6339f248_result.png +3 -0
- assets/160792599_6a7ec52516_result.png +3 -0
- assets/161669933_3e7d8c7e2c_result.png +3 -0
- assets/2260560631_09093be4c6_result.png +3 -0
- assets/2312984882_bec7849e09_result.png +3 -0
- assets/241345721_3f3724a7fc_result.png +3 -0
- assets/2453318633_550228acd4_result.png +3 -0
- assets/252578659_9e404b6430_result.png +3 -0
- assets/307994435_592f933a6d_result.png +3 -0
- assets/3185645793_49de805194_result.png +3 -0
- assets/326585030_e1dcca2562_result.png +3 -0
- assets/3440104178_6871a24e13_result.png +3 -0
- assets/487071033_27e460a1b9_result.png +3 -0
- assets/atom_dictionary.png +3 -0
- assets/compare_154871781_ae77696b77.png +3 -0
- assets/compare_159712188_d530dd478c.png +3 -0
- assets/compare_160585932_fa6339f248.png +3 -0
- assets/examples/154871781_ae77696b77.jpg +0 -0
- assets/examples/159712188_d530dd478c.jpg +0 -0
- assets/examples/160585932_fa6339f248.jpg +0 -0
- assets/hero.png +3 -0
- assets/training_curves.png +0 -0
- checkpoints/shadenet32.ckpt +3 -0
- config.json +18 -0
- inference.py +68 -0
- inference_utils.py +102 -0
- model.py +338 -0
- onnx/model.onnx +3 -0
- onnx/model_fp16.onnx +3 -0
- requirements-space.txt +5 -0
- requirements.txt +6 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,28 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
assets/103195344_5d2dc613a3_result.png filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
assets/154871781_ae77696b77_result.png filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
assets/1561658940_a947f2446a_result.png filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
assets/157139628_5dc483e2e4_result.png filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
assets/159712188_d530dd478c_result.png filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
assets/160541986_d5be2ab4c1_result.png filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
assets/160566014_59528ff897_result.png filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
assets/160585932_fa6339f248_result.png filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
assets/160792599_6a7ec52516_result.png filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
assets/161669933_3e7d8c7e2c_result.png filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
assets/2260560631_09093be4c6_result.png filter=lfs diff=lfs merge=lfs -text
|
| 47 |
+
assets/2312984882_bec7849e09_result.png filter=lfs diff=lfs merge=lfs -text
|
| 48 |
+
assets/241345721_3f3724a7fc_result.png filter=lfs diff=lfs merge=lfs -text
|
| 49 |
+
assets/2453318633_550228acd4_result.png filter=lfs diff=lfs merge=lfs -text
|
| 50 |
+
assets/252578659_9e404b6430_result.png filter=lfs diff=lfs merge=lfs -text
|
| 51 |
+
assets/307994435_592f933a6d_result.png filter=lfs diff=lfs merge=lfs -text
|
| 52 |
+
assets/3185645793_49de805194_result.png filter=lfs diff=lfs merge=lfs -text
|
| 53 |
+
assets/326585030_e1dcca2562_result.png filter=lfs diff=lfs merge=lfs -text
|
| 54 |
+
assets/3440104178_6871a24e13_result.png filter=lfs diff=lfs merge=lfs -text
|
| 55 |
+
assets/487071033_27e460a1b9_result.png filter=lfs diff=lfs merge=lfs -text
|
| 56 |
+
assets/atom_dictionary.png filter=lfs diff=lfs merge=lfs -text
|
| 57 |
+
assets/compare_154871781_ae77696b77.png filter=lfs diff=lfs merge=lfs -text
|
| 58 |
+
assets/compare_159712188_d530dd478c.png filter=lfs diff=lfs merge=lfs -text
|
| 59 |
+
assets/compare_160585932_fa6339f248.png filter=lfs diff=lfs merge=lfs -text
|
| 60 |
+
assets/hero.png filter=lfs diff=lfs merge=lfs -text
|
README.md
CHANGED
|
@@ -1,3 +1,206 @@
|
|
| 1 |
---
|
| 2 |
license: apache-2.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
---
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
license: apache-2.0
|
| 3 |
+
pipeline_tag: image-to-image
|
| 4 |
+
library_name: pytorch
|
| 5 |
+
tags:
|
| 6 |
+
- onnx
|
| 7 |
+
- inverse-rendering
|
| 8 |
+
- image-decomposition
|
| 9 |
+
- albedo
|
| 10 |
+
- normal-map
|
| 11 |
+
- depth-estimation
|
| 12 |
+
- pbr
|
| 13 |
+
- material-estimation
|
| 14 |
+
- flickr8k
|
| 15 |
+
- shadenet
|
| 16 |
+
datasets:
|
| 17 |
+
- singam96/flickr8k_marigold_v2
|
| 18 |
+
metrics:
|
| 19 |
+
- mse
|
| 20 |
+
model-index:
|
| 21 |
+
- name: ShadeNet-3.2-5M
|
| 22 |
+
results:
|
| 23 |
+
- task:
|
| 24 |
+
type: image-to-image
|
| 25 |
+
dataset:
|
| 26 |
+
name: flickr8k_marigold_v2 (val split, 807 images)
|
| 27 |
+
type: singam96/flickr8k_marigold_v2
|
| 28 |
+
metrics:
|
| 29 |
+
- name: val/loss (weighted MSE + recon)
|
| 30 |
+
type: mse
|
| 31 |
+
value: 0.1671
|
| 32 |
---
|
| 33 |
+
|
| 34 |
+
# ShadeNet-3.2 5M
|
| 35 |
+
|
| 36 |
+
A lightweight inverse-rendering model: one photo in, **albedo + relative depth + surface normals + shading** out (8 channels, 5.0M params). Successor of [ShadeNet-2](https://huggingface.co/singam96/ShadeNet-2-20M) — **4× smaller, better depth and normals**.
|
| 37 |
+
|
| 38 |
+
[](assets/hero.png)
|
| 39 |
+
|
| 40 |
+
## TL;DR
|
| 41 |
+
|
| 42 |
+
| | |
|
| 43 |
+
|---|---|
|
| 44 |
+
| Params | **5.0M** (3.2M trainable + 1.8M frozen MobileNetV2 trunk) |
|
| 45 |
+
| Input | RGB `[1, 3, H, W]` in `[-1, 1]` (384px trained; any multiple of 16) |
|
| 46 |
+
| Output | 8ch `[1, 8, H, W]` in `[-1, 1]`: albedo, relative depth, normals, shading |
|
| 47 |
+
| Val L1 (807 imgs) | albedo 0.695 · depth 0.217 · normal 0.581 |
|
| 48 |
+
| Formats | fp32 ONNX (20MB), fp16 ONNX (10MB), torch checkpoint (44MB) |
|
| 49 |
+
|
| 50 |
+
## Examples
|
| 51 |
+
|
| 52 |
+
[](assets/154871781_ae77696b77_result.png)
|
| 53 |
+
[](assets/157139628_5dc483e2e4_result.png)
|
| 54 |
+
[](assets/159712188_d530dd478c_result.png)
|
| 55 |
+
[](assets/160541986_d5be2ab4c1_result.png)
|
| 56 |
+
[](assets/160566014_59528ff897_result.png)
|
| 57 |
+
[](assets/160585932_fa6339f248_result.png)
|
| 58 |
+
[](assets/160792599_6a7ec52516_result.png)
|
| 59 |
+
[](assets/161669933_3e7d8c7e2c_result.png)
|
| 60 |
+
|
| 61 |
+
*Each grid: input | albedo | shading / depth | normal | recon (albedo×shading). Click any image for full size.*
|
| 62 |
+
|
| 63 |
+
## Results
|
| 64 |
+
|
| 65 |
+
Full 807-image val split, per-map L1 (the comparable metric across versions —
|
| 66 |
+
the headline `val/loss` formula changed between v2 and v3):
|
| 67 |
+
|
| 68 |
+
| Map (val L1) | ShadeNet-2 (20M) | **ShadeNet-3.2 (5M)** | change |
|
| 69 |
+
|---|---|---|---|
|
| 70 |
+
| Albedo | 0.708 | **0.695** | −1.7% |
|
| 71 |
+
| Depth (SSI-aligned) | 0.247 | **0.217** | **−12%** |
|
| 72 |
+
| Normal | 0.696 | **0.581** | **−16%** |
|
| 73 |
+
|
| 74 |
+
[](assets/training_curves.png)
|
| 75 |
+
|
| 76 |
+
Checkpoint variants (full val, pruned top-32 dictionary):
|
| 77 |
+
|
| 78 |
+
| Weights | val/loss | albedo L1 | depth L1 | normal L1 |
|
| 79 |
+
|---|---|---|---|---|
|
| 80 |
+
| best.ckpt (raw) | **0.1669** | 0.7013 | 0.2241 | **0.5709** |
|
| 81 |
+
| best.ckpt (EMA) | 0.1671 | **0.6952** | **0.2174** | 0.5807 |
|
| 82 |
+
|
| 83 |
+
Shipped ONNX uses the **EMA** weights. All page outputs and the Space run a
|
| 84 |
+
**3-pass multi-scale median** (scales 0.875/1.0/1.125) — a mild denoise; the
|
| 85 |
+
model's raw single-pass albedo is sharper than its pseudo-labels, so this
|
| 86 |
+
trades a little detail for lower variance.
|
| 87 |
+
|
| 88 |
+
## ShadeNet-2 vs ShadeNet-3.2
|
| 89 |
+
|
| 90 |
+
Same inputs (ShadeNet-2 top rows, ShadeNet-3.2 bottom rows), maps only. Both
|
| 91 |
+
use their shipped weights; ShadeNet-3.2 runs the 3-pass multi-scale median.
|
| 92 |
+
|
| 93 |
+
[](assets/compare_154871781_ae77696b77.png)
|
| 94 |
+
[](assets/compare_159712188_d530dd478c.png)
|
| 95 |
+
[](assets/compare_160585932_fa6339f248.png)
|
| 96 |
+
|
| 97 |
+
## Architecture
|
| 98 |
+
|
| 99 |
+
**ParallelUNet generator (4.98M params)** + spectral-norm GroupNorm PatchGAN discriminator (2.77M, training only):
|
| 100 |
+
|
| 101 |
+
- Dual parallel encoders — vanilla UNet path plus a **frozen MobileNetV2** feature trunk, fused at every decoder level
|
| 102 |
+
- **Depthwise-separable factorized convs** (1×3 + 3×1) throughout; full H/32 bottleneck; reflect padding
|
| 103 |
+
- **Patch-dictionary output tail**: 16×16 tiles softmax-addressed over **32 learned per-channel atoms** (pruned from 1024 — the top-32 hold 99.4% of addressing mass), blended back into the signal before tanh
|
| 104 |
+
- Single-pass RGB → 8ch output; EMA weight shadow (shipped weights are EMA)
|
| 105 |
+
|
| 106 |
+
## Patch dictionary
|
| 107 |
+
|
| 108 |
+
The tail softmax-addresses 32 learned 16×16 atoms per tile (kept from 1024
|
| 109 |
+
after measuring per-atom selection: only ~34 atoms are ever used, ~31 cover
|
| 110 |
+
99% of the mass). Shown below per output channel — each panel is the 8×4 atom
|
| 111 |
+
grid, shared grayscale scale.
|
| 112 |
+
|
| 113 |
+
[](assets/atom_dictionary.png)
|
| 114 |
+
|
| 115 |
+
## Output maps
|
| 116 |
+
|
| 117 |
+
| Map | Channels | Range | Description |
|
| 118 |
+
|---|---|---|---|
|
| 119 |
+
| Albedo | 3 `[0:3]` | [−1, 1] | Reflectance / diffuse color, lighting factored out |
|
| 120 |
+
| Depth | 1 `[3:4]` | [−1, 1] | **Relative** depth (0=near), affine-ambiguous |
|
| 121 |
+
| Normal | 3 `[4:7]` | [−1, 1] | Surface normals, unit-length regularised |
|
| 122 |
+
| Shading | 1 `[7:8]` | [−1, 1] | Grayscale irradiance; `input ≈ albedo × shading` |
|
| 123 |
+
| Recon | — | — | `albedo × shading` re-rendering (diagnostic, not a head) |
|
| 124 |
+
|
| 125 |
+
## Files
|
| 126 |
+
|
| 127 |
+
```
|
| 128 |
+
├── app.py # Gradio Space app (fp16 ONNX)
|
| 129 |
+
├── inference.py # Standalone torch CLI
|
| 130 |
+
├── inference_utils.py # Grid visualisation (numpy/PIL)
|
| 131 |
+
├── model.py # Standalone generator architecture
|
| 132 |
+
├── requirements.txt
|
| 133 |
+
├── checkpoints/shadenet32.ckpt # Torch weights, EMA (44MB)
|
| 134 |
+
└── onnx/
|
| 135 |
+
├── model.onnx # fp32, EMA (20MB) — GPU via CUDA EP
|
| 136 |
+
└── model_fp16.onnx # fp16, EMA (10MB) — CPU
|
| 137 |
+
```
|
| 138 |
+
|
| 139 |
+
## Usage
|
| 140 |
+
|
| 141 |
+
### Gradio Space
|
| 142 |
+
|
| 143 |
+
Try it in your browser — no installation: **[singam96/ShadeNet-3.2-5M Space](https://huggingface.co/spaces/singam96/ShadeNet-3.2-5M)**.
|
| 144 |
+
|
| 145 |
+
This repo ships `app.py`, the Space entrypoint. To recreate it: New Space → Gradio SDK → point at this repo.
|
| 146 |
+
|
| 147 |
+
### Torch CLI
|
| 148 |
+
|
| 149 |
+
```bash
|
| 150 |
+
pip install torch torchvision pillow numpy
|
| 151 |
+
python inference.py photo.jpg --output_dir ./output
|
| 152 |
+
# --checkpoint ./checkpoints/shadenet32.ckpt --image-size 512 --no-ema to disable EMA
|
| 153 |
+
```
|
| 154 |
+
|
| 155 |
+
### ONNX (CPU)
|
| 156 |
+
|
| 157 |
+
```bash
|
| 158 |
+
pip install onnxruntime pillow numpy
|
| 159 |
+
python - <<'EOF'
|
| 160 |
+
import onnxruntime as ort, numpy as np
|
| 161 |
+
from PIL import Image
|
| 162 |
+
from inference_utils import build_grid, pil_to_np, resize_pad
|
| 163 |
+
sess = ort.InferenceSession("onnx/model_fp16.onnx", providers=["CPUExecutionProvider"])
|
| 164 |
+
img = resize_pad(Image.open("photo.jpg"), 512)
|
| 165 |
+
out = sess.run(None, {"input_rgb": pil_to_np(img).astype(np.float32)})[0]
|
| 166 |
+
build_grid(img, out).save("result.png")
|
| 167 |
+
EOF
|
| 168 |
+
```
|
| 169 |
+
|
| 170 |
+
Input: `[1, 3, H, W]` in `[-1, 1]` (any H, W; multiples of 16 recommended).
|
| 171 |
+
Output: `[1, 8, H, W]` in `[-1, 1]`.
|
| 172 |
+
|
| 173 |
+
## Training
|
| 174 |
+
|
| 175 |
+
Trained from scratch on [`singam96/flickr8k_marigold_v2`](https://huggingface.co/datasets/singam96/flickr8k_marigold_v2) (8077 Flickr8k photos with Marigold-V2 pseudo-labels), 384px, fp32, single GTX 1650, early-stopped on `val/loss` (patience 5) at epoch 12. The 1024-atom dictionary was pruned to its top-32 by addressing mass (no retraining; output error vs full ~5e-4 mean) for the release.
|
| 176 |
+
|
| 177 |
+
Losses: scale-invariant MSE on albedo (per-channel std alignment) + **scale-shift-invariant** MSE on depth after least-squares alignment (decoded Marigold depth is relative) + **Sobel gradient-matching** on depth (edge crispness) + MSE on normals + self-supervised **reconstruction coupling** (`albedo×shading ≈ input`, the shading head's only supervision) + LSGAN + normal unit-length penalty. Weight decay 1e-4 (patch dictionary exempt).
|
| 178 |
+
|
| 179 |
+
## Limitations
|
| 180 |
+
|
| 181 |
+
- Depth is **relative**, not metric — don't read meters off it.
|
| 182 |
+
- Shading assumes **white light**; strongly colored illumination (sunsets, neon) leaks into albedo.
|
| 183 |
+
- Normals are noisy in foliage/sky — those pseudo-labels were noisy too.
|
| 184 |
+
- Occasional localized artifacts in albedo/shading (learned prior pockets).
|
| 185 |
+
- No shadows/global illumination — relighting-style use is approximate.
|
| 186 |
+
- Uncertainty is not provided: confidently-wrong pseudo-labels are fitted confidently.
|
| 187 |
+
|
| 188 |
+
## Attribution
|
| 189 |
+
|
| 190 |
+
Supervision labels come from **Marigold V2** (Ke et al.) applied to **Flickr8k** (Hodosh et al.):
|
| 191 |
+
|
| 192 |
+
- Marigold: *Repurposing Diffusion-Based Image Generators for Monocular Depth Estimation* — Ke, Obukhov, Metzger, Daudt, Schindler, Schindler (CVPR 2024)
|
| 193 |
+
- Flickr8k: *Framing Image Description as a Ranking Task* — Hodosh, Young, Hockenmaier (2013)
|
| 194 |
+
|
| 195 |
+
This model (weights + code) is Apache-2.0; upstream dataset/model terms still apply to their artifacts.
|
| 196 |
+
|
| 197 |
+
## Citation
|
| 198 |
+
|
| 199 |
+
```bibtex
|
| 200 |
+
@software{shadenet32,
|
| 201 |
+
author = {Sachin},
|
| 202 |
+
title = {ShadeNet-3.2: single-image inverse rendering (5M)},
|
| 203 |
+
year = {2026},
|
| 204 |
+
url = {https://huggingface.co/singam96/ShadeNet-3.2-5M}
|
| 205 |
+
}
|
| 206 |
+
```
|
app.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""ShadeNet-3.2 Gradio Space (CPU-friendly: fp16 ONNX backend)."""
|
| 2 |
+
import os
|
| 3 |
+
|
| 4 |
+
import gradio as gr
|
| 5 |
+
import numpy as np
|
| 6 |
+
import onnxruntime as ort
|
| 7 |
+
import spaces
|
| 8 |
+
from PIL import Image
|
| 9 |
+
|
| 10 |
+
from inference_utils import (build_grid, extract_maps, pil_to_np, resize_pad,
|
| 11 |
+
run_ensemble_onnx)
|
| 12 |
+
|
| 13 |
+
MODEL_PATH = os.path.join(os.path.dirname(os.path.abspath(__file__)),
|
| 14 |
+
"onnx", "model_fp16.onnx")
|
| 15 |
+
IMAGE_SIZE = 512
|
| 16 |
+
|
| 17 |
+
_sess = ort.InferenceSession(MODEL_PATH, providers=["CPUExecutionProvider"])
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
@spaces.GPU(duration=60)
|
| 21 |
+
def decompose(image: Image.Image):
|
| 22 |
+
"""Returns (grid, input, albedo, shading, depth, normal, recon)."""
|
| 23 |
+
img_rgb = resize_pad(image.convert("RGB"), IMAGE_SIZE)
|
| 24 |
+
out = run_ensemble_onnx(_sess, img_rgb, IMAGE_SIZE)
|
| 25 |
+
maps = extract_maps(out, img_rgb)
|
| 26 |
+
return (build_grid(img_rgb, out), maps["input"], maps["albedo"],
|
| 27 |
+
maps["shading"], maps["depth"], maps["normal"], maps["recon"])
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
EXAMPLES = []
|
| 31 |
+
_exdir = os.path.join(os.path.dirname(os.path.abspath(__file__)),
|
| 32 |
+
"assets", "examples")
|
| 33 |
+
if os.path.isdir(_exdir):
|
| 34 |
+
EXAMPLES = sorted(os.path.join(_exdir, f) for f in os.listdir(_exdir)
|
| 35 |
+
if f.lower().endswith((".png", ".jpg", ".jpeg")))
|
| 36 |
+
|
| 37 |
+
demo = gr.Interface(
|
| 38 |
+
fn=decompose,
|
| 39 |
+
inputs=gr.Image(type="pil", label="Input photo"),
|
| 40 |
+
outputs=[
|
| 41 |
+
gr.Image(type="pil", label="Overview grid"),
|
| 42 |
+
gr.Image(type="pil", label="Input (resized)"),
|
| 43 |
+
gr.Image(type="pil", label="Albedo"),
|
| 44 |
+
gr.Image(type="pil", label="Shading"),
|
| 45 |
+
gr.Image(type="pil", label="Depth (relative: dark=near)"),
|
| 46 |
+
gr.Image(type="pil", label="Normal"),
|
| 47 |
+
gr.Image(type="pil", label="Recon (albedo x shading)"),
|
| 48 |
+
],
|
| 49 |
+
title="ShadeNet-3.2: single-image inverse rendering (7M)",
|
| 50 |
+
description=("Decomposes a photo into albedo, relative depth, surface "
|
| 51 |
+
"normals and shading (grayscale irradiance). Successor of "
|
| 52 |
+
"ShadeNet-2 (20M), 3x smaller with better depth/normal "
|
| 53 |
+
"accuracy. Runs the fp16 ONNX model."),
|
| 54 |
+
examples=EXAMPLES,
|
| 55 |
+
cache_examples=False,
|
| 56 |
+
)
|
| 57 |
+
|
| 58 |
+
if __name__ == "__main__":
|
| 59 |
+
demo.launch()
|
assets/103195344_5d2dc613a3_result.png
ADDED
|
Git LFS Details
|
assets/154871781_ae77696b77_result.png
ADDED
|
Git LFS Details
|
assets/1561658940_a947f2446a_result.png
ADDED
|
Git LFS Details
|
assets/157139628_5dc483e2e4_result.png
ADDED
|
Git LFS Details
|
assets/159712188_d530dd478c_result.png
ADDED
|
Git LFS Details
|
assets/160541986_d5be2ab4c1_result.png
ADDED
|
Git LFS Details
|
assets/160566014_59528ff897_result.png
ADDED
|
Git LFS Details
|
assets/160585932_fa6339f248_result.png
ADDED
|
Git LFS Details
|
assets/160792599_6a7ec52516_result.png
ADDED
|
Git LFS Details
|
assets/161669933_3e7d8c7e2c_result.png
ADDED
|
Git LFS Details
|
assets/2260560631_09093be4c6_result.png
ADDED
|
Git LFS Details
|
assets/2312984882_bec7849e09_result.png
ADDED
|
Git LFS Details
|
assets/241345721_3f3724a7fc_result.png
ADDED
|
Git LFS Details
|
assets/2453318633_550228acd4_result.png
ADDED
|
Git LFS Details
|
assets/252578659_9e404b6430_result.png
ADDED
|
Git LFS Details
|
assets/307994435_592f933a6d_result.png
ADDED
|
Git LFS Details
|
assets/3185645793_49de805194_result.png
ADDED
|
Git LFS Details
|
assets/326585030_e1dcca2562_result.png
ADDED
|
Git LFS Details
|
assets/3440104178_6871a24e13_result.png
ADDED
|
Git LFS Details
|
assets/487071033_27e460a1b9_result.png
ADDED
|
Git LFS Details
|
assets/atom_dictionary.png
ADDED
|
Git LFS Details
|
assets/compare_154871781_ae77696b77.png
ADDED
|
Git LFS Details
|
assets/compare_159712188_d530dd478c.png
ADDED
|
Git LFS Details
|
assets/compare_160585932_fa6339f248.png
ADDED
|
Git LFS Details
|
assets/examples/154871781_ae77696b77.jpg
ADDED
|
assets/examples/159712188_d530dd478c.jpg
ADDED
|
assets/examples/160585932_fa6339f248.jpg
ADDED
|
assets/hero.png
ADDED
|
Git LFS Details
|
assets/training_curves.png
ADDED
|
checkpoints/shadenet32.ckpt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3b660a6d955f2df63304312d0fe491bb6a0bd9e71fcdff69a7398a9b19fe3fd0
|
| 3 |
+
size 44034045
|
config.json
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model_type": "shadenet",
|
| 3 |
+
"model_name": "ShadeNet-3.2-5M",
|
| 4 |
+
"architecture": "ParallelUNet-v3",
|
| 5 |
+
"params_total": 4977208,
|
| 6 |
+
"params_trainable": 3165496,
|
| 7 |
+
"image_size": 384,
|
| 8 |
+
"in_ch": 3,
|
| 9 |
+
"out_ch": 8,
|
| 10 |
+
"maps": {
|
| 11 |
+
"albedo": [0, 3],
|
| 12 |
+
"depth": [3, 4],
|
| 13 |
+
"normal": [4, 7],
|
| 14 |
+
"shading": [7, 8]
|
| 15 |
+
},
|
| 16 |
+
"framework": "pytorch",
|
| 17 |
+
"onnx": ["onnx/model.onnx", "onnx/model_fp16.onnx"]
|
| 18 |
+
}
|
inference.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""ShadeNet-3.2 torch inference (standalone CLI).
|
| 2 |
+
|
| 3 |
+
Usage:
|
| 4 |
+
pip install -r requirements.txt
|
| 5 |
+
python inference.py input.jpg --output_dir ./output
|
| 6 |
+
python inference.py ./photos --checkpoint ./checkpoints/shadenet32.ckpt
|
| 7 |
+
"""
|
| 8 |
+
import argparse
|
| 9 |
+
import os
|
| 10 |
+
|
| 11 |
+
import torch
|
| 12 |
+
from PIL import Image
|
| 13 |
+
|
| 14 |
+
from model import load_shadenet32
|
| 15 |
+
from inference_utils import build_grid, pil_to_np, resize_pad
|
| 16 |
+
|
| 17 |
+
IMAGE_SIZE = 512
|
| 18 |
+
ENS_SCALES = (0.875, 1.0, 1.125)
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
@torch.no_grad()
|
| 22 |
+
def run(model, img_rgb: Image.Image, device: str,
|
| 23 |
+
image_size: int = IMAGE_SIZE, scales=ENS_SCALES) -> "torch.Tensor":
|
| 24 |
+
"""3-pass multi-scale median (matches the model card / Space)."""
|
| 25 |
+
outs = []
|
| 26 |
+
for s in scales:
|
| 27 |
+
side = max(16, int(round(image_size * s)) // 16 * 16) # dict needs /16
|
| 28 |
+
scaled = resize_pad(img_rgb, side)
|
| 29 |
+
x = torch.from_numpy(pil_to_np(scaled)).to(device)
|
| 30 |
+
o = model(x)
|
| 31 |
+
if side != image_size:
|
| 32 |
+
o = torch.nn.functional.interpolate(
|
| 33 |
+
o, size=(image_size, image_size), mode="bilinear",
|
| 34 |
+
align_corners=False)
|
| 35 |
+
outs.append(o.cpu())
|
| 36 |
+
return torch.stack(outs, dim=0).median(dim=0).values
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def main():
|
| 40 |
+
ap = argparse.ArgumentParser(description="ShadeNet-3.2 inference")
|
| 41 |
+
ap.add_argument("input", help="Image file or folder")
|
| 42 |
+
ap.add_argument("--checkpoint", default="./checkpoints/shadenet32.ckpt")
|
| 43 |
+
ap.add_argument("--output_dir", default="./output")
|
| 44 |
+
ap.add_argument("--image-size", type=int, default=IMAGE_SIZE)
|
| 45 |
+
ap.add_argument("--no-ema", action="store_true",
|
| 46 |
+
help="Use raw weights instead of EMA")
|
| 47 |
+
args = ap.parse_args()
|
| 48 |
+
|
| 49 |
+
device = "cuda" if torch.cuda.is_available() else "cpu"
|
| 50 |
+
model = load_shadenet32(args.checkpoint, device, use_ema=not args.no_ema)
|
| 51 |
+
|
| 52 |
+
if os.path.isdir(args.input):
|
| 53 |
+
files = sorted(os.path.join(args.input, f) for f in os.listdir(args.input)
|
| 54 |
+
if f.lower().endswith((".png", ".jpg", ".jpeg", ".webp")))
|
| 55 |
+
else:
|
| 56 |
+
files = [args.input]
|
| 57 |
+
os.makedirs(args.output_dir, exist_ok=True)
|
| 58 |
+
for fp in files:
|
| 59 |
+
img_rgb = resize_pad(Image.open(fp), args.image_size)
|
| 60 |
+
out = run(model, img_rgb, device).numpy()
|
| 61 |
+
base = os.path.splitext(os.path.basename(fp))[0]
|
| 62 |
+
build_grid(img_rgb, out).save(
|
| 63 |
+
os.path.join(args.output_dir, f"{base}_result.png"))
|
| 64 |
+
print(f"Saved: {base}_result.png")
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
if __name__ == "__main__":
|
| 68 |
+
main()
|
inference_utils.py
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Standalone inference utilities (numpy/PIL only — no torch needed)."""
|
| 2 |
+
import numpy as np
|
| 3 |
+
from PIL import Image, ImageDraw
|
| 4 |
+
|
| 5 |
+
ALBEDO = slice(0, 3)
|
| 6 |
+
DEPTH = slice(3, 4)
|
| 7 |
+
NORMAL = slice(4, 7)
|
| 8 |
+
SHADING = slice(7, 8)
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
def resize_pad(img_rgb: Image.Image, image_size: int) -> Image.Image:
|
| 12 |
+
"""Scale + center-crop to a square (RGB in, RGB out)."""
|
| 13 |
+
img_rgb = img_rgb.convert("RGB")
|
| 14 |
+
w, h = img_rgb.size
|
| 15 |
+
size = int(image_size)
|
| 16 |
+
scale = max(size / float(w), size / float(h))
|
| 17 |
+
img_r = img_rgb.resize((max(1, int(round(w * scale))),
|
| 18 |
+
max(1, int(round(h * scale)))), Image.BICUBIC)
|
| 19 |
+
left = (img_r.size[0] - size) // 2
|
| 20 |
+
top = (img_r.size[1] - size) // 2
|
| 21 |
+
return img_r.crop((left, top, left + size, top + size))
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def pil_to_np(img_rgb: Image.Image) -> np.ndarray:
|
| 25 |
+
"""PIL RGB -> [1, 3, H, W] float32 in [-1, 1]."""
|
| 26 |
+
arr = np.array(img_rgb.convert("RGB"), dtype=np.float32).transpose(2, 0, 1)
|
| 27 |
+
return arr[np.newaxis] / 255.0 * 2.0 - 1.0
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def _to_01(output: np.ndarray) -> np.ndarray:
|
| 31 |
+
return np.clip((output + 1.0) / 2.0, 0, 1)
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def _tile(arr01: np.ndarray, ts: int) -> Image.Image:
|
| 35 |
+
return Image.fromarray(
|
| 36 |
+
(arr01 * 255.0 + 0.5).astype("uint8")).resize((ts, ts), Image.BICUBIC)
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def _resize_map(out: np.ndarray, size: int) -> np.ndarray:
|
| 40 |
+
"""Bilinear-resize a [1, C, H, W] float map to [1, C, size, size]."""
|
| 41 |
+
c = out.shape[1]
|
| 42 |
+
res = np.empty((1, c, size, size), dtype=np.float32)
|
| 43 |
+
for i in range(c):
|
| 44 |
+
im = Image.fromarray(out[0, i].astype(np.float32), mode="F")
|
| 45 |
+
res[0, i] = np.asarray(im.resize((size, size), Image.BILINEAR))
|
| 46 |
+
return res
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
ENS_SCALES = (0.875, 1.0, 1.125)
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def run_ensemble_onnx(sess, img_rgb: Image.Image, image_size: int = 512,
|
| 53 |
+
scales=ENS_SCALES) -> np.ndarray:
|
| 54 |
+
"""3-pass multi-scale median (torch-free). Returns [1, 8, S, S]."""
|
| 55 |
+
outs = []
|
| 56 |
+
for s in scales:
|
| 57 |
+
side = max(16, int(round(image_size * s)) // 16 * 16) # dict needs /16
|
| 58 |
+
scaled = resize_pad(img_rgb, side)
|
| 59 |
+
x = pil_to_np(scaled).astype(np.float32)
|
| 60 |
+
o = sess.run(None, {"input_rgb": x})[0]
|
| 61 |
+
outs.append(o if side == image_size else _resize_map(o, image_size))
|
| 62 |
+
return np.median(np.stack(outs, axis=0), axis=0).astype(np.float32)
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
def extract_maps(output: np.ndarray, img_rgb: Image.Image,
|
| 66 |
+
ts: int = 256) -> dict:
|
| 67 |
+
"""Split a [1, 8, H, W] output ([-1, 1]) into labelled PIL tiles."""
|
| 68 |
+
assert output.shape[1] == 8, f"want 8 channels, got {output.shape[1]}"
|
| 69 |
+
o = _to_01(output)
|
| 70 |
+
alb = o[:, ALBEDO][0].transpose(1, 2, 0)
|
| 71 |
+
sh = o[:, SHADING][0, 0]
|
| 72 |
+
d = o[:, DEPTH][0, 0]
|
| 73 |
+
span = d.max() - d.min()
|
| 74 |
+
dg = (d - d.min()) / (span if span > 1e-6 else 1.0)
|
| 75 |
+
return {
|
| 76 |
+
"input": img_rgb.resize((ts, ts), Image.BICUBIC),
|
| 77 |
+
"albedo": _tile(alb, ts),
|
| 78 |
+
"shading": _tile(np.repeat(sh[..., None], 3, axis=2), ts),
|
| 79 |
+
"depth": _tile(dg, ts).convert("RGB"),
|
| 80 |
+
"normal": _tile(o[:, NORMAL][0].transpose(1, 2, 0), ts),
|
| 81 |
+
"recon": _tile(np.clip(alb * sh[..., None], 0, 1), ts),
|
| 82 |
+
}
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
def build_grid(img_rgb: Image.Image, output: np.ndarray,
|
| 86 |
+
ts: int = 256) -> Image.Image:
|
| 87 |
+
"""3x2 labelled grid: input|albedo|shading / depth|normal|recon."""
|
| 88 |
+
maps = extract_maps(output, img_rgb, ts)
|
| 89 |
+
grid = Image.new("RGB", (ts * 3, ts * 2))
|
| 90 |
+
grid.paste(maps["input"], (0, 0))
|
| 91 |
+
grid.paste(maps["albedo"], (ts, 0))
|
| 92 |
+
grid.paste(maps["shading"], (ts * 2, 0))
|
| 93 |
+
grid.paste(maps["depth"], (0, ts))
|
| 94 |
+
grid.paste(maps["normal"], (ts, ts))
|
| 95 |
+
grid.paste(maps["recon"], (ts * 2, ts))
|
| 96 |
+
draw = ImageDraw.Draw(grid)
|
| 97 |
+
for (x, y, name) in [(4, 4, "input"), (ts + 4, 4, "albedo"),
|
| 98 |
+
(ts * 2 + 4, 4, "shading"), (4, ts + 4, "depth"),
|
| 99 |
+
(ts + 4, ts + 4, "normal"),
|
| 100 |
+
(ts * 2 + 4, ts + 4, "recon")]:
|
| 101 |
+
draw.text((x, y), name, fill=(255, 255, 0))
|
| 102 |
+
return grid
|
model.py
ADDED
|
@@ -0,0 +1,338 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""ShadeNet-3.2 generator (standalone, no Lightning dependency).
|
| 2 |
+
|
| 3 |
+
ParallelUNet v3: dual parallel encoders (vanilla UNet + frozen MobileNetV2),
|
| 4 |
+
depthwise-separable factorized convs, full H/32 bottleneck, and a patch
|
| 5 |
+
dictionary output tail -> 8ch intrinsic maps in [-1, 1]:
|
| 6 |
+
[0:3] albedo | [3:4] relative depth (0=near) | [4:7] normal | [7:8] shading
|
| 7 |
+
"""
|
| 8 |
+
import math
|
| 9 |
+
|
| 10 |
+
import torch
|
| 11 |
+
import torch.nn as nn
|
| 12 |
+
import torchvision
|
| 13 |
+
from torchvision.models.feature_extraction import create_feature_extractor
|
| 14 |
+
|
| 15 |
+
OUT_CH = 8
|
| 16 |
+
WIDTH_MULT = 0.9
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
def _num_groups(ch: int) -> int:
|
| 20 |
+
g = min(32, ch)
|
| 21 |
+
while ch % g != 0:
|
| 22 |
+
g -= 1
|
| 23 |
+
return g
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
def _scale(width_mult):
|
| 27 |
+
"""Channel scaler: width_mult rounded to multiples of 8 (GroupNorm-safe)."""
|
| 28 |
+
w = float(width_mult)
|
| 29 |
+
|
| 30 |
+
def C(n):
|
| 31 |
+
return max(8, int(round(n * w / 8.0) * 8))
|
| 32 |
+
|
| 33 |
+
return C
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
class ChannelLinear(nn.Module):
|
| 37 |
+
"""Per-channel learnable affine: y = x * weight + bias."""
|
| 38 |
+
|
| 39 |
+
def __init__(self, channels: int, init_scale: float = 0.01):
|
| 40 |
+
super().__init__()
|
| 41 |
+
self.weight = nn.Parameter(
|
| 42 |
+
torch.empty(1, channels, 1, 1).uniform_(-init_scale, init_scale) + 1.0
|
| 43 |
+
)
|
| 44 |
+
self.bias = nn.Parameter(
|
| 45 |
+
torch.empty(1, channels, 1, 1).uniform_(-init_scale, init_scale)
|
| 46 |
+
)
|
| 47 |
+
|
| 48 |
+
def forward(self, x):
|
| 49 |
+
return x * self.weight + self.bias
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
class DoubleConv(nn.Module):
|
| 53 |
+
"""Factorized (1,3)+(3,1) conv pair with GN+ELU, reflect padding."""
|
| 54 |
+
|
| 55 |
+
def __init__(self, in_ch, out_ch):
|
| 56 |
+
super().__init__()
|
| 57 |
+
g = _num_groups(out_ch)
|
| 58 |
+
self.conv = nn.Sequential(
|
| 59 |
+
nn.Conv2d(in_ch, out_ch, (1, 3), padding=(0, 1),
|
| 60 |
+
padding_mode="reflect", bias=False),
|
| 61 |
+
nn.GroupNorm(g, out_ch),
|
| 62 |
+
nn.ELU(inplace=True),
|
| 63 |
+
nn.Conv2d(out_ch, out_ch, (3, 1), padding=(1, 0),
|
| 64 |
+
padding_mode="reflect", bias=False),
|
| 65 |
+
nn.GroupNorm(g, out_ch),
|
| 66 |
+
nn.ELU(inplace=True),
|
| 67 |
+
)
|
| 68 |
+
|
| 69 |
+
def forward(self, x):
|
| 70 |
+
return self.conv(x)
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
class DSDoubleConv(nn.Module):
|
| 74 |
+
"""Depthwise-separable DoubleConv: same geometry, ~3x fewer params."""
|
| 75 |
+
|
| 76 |
+
def __init__(self, in_ch, out_ch):
|
| 77 |
+
super().__init__()
|
| 78 |
+
g = _num_groups(out_ch)
|
| 79 |
+
self.conv = nn.Sequential(
|
| 80 |
+
nn.Conv2d(in_ch, in_ch, (1, 3), padding=(0, 1),
|
| 81 |
+
padding_mode="reflect", groups=in_ch, bias=False),
|
| 82 |
+
nn.Conv2d(in_ch, out_ch, 1, bias=False),
|
| 83 |
+
nn.GroupNorm(g, out_ch),
|
| 84 |
+
nn.ELU(inplace=True),
|
| 85 |
+
nn.Conv2d(out_ch, out_ch, (3, 1), padding=(1, 0),
|
| 86 |
+
padding_mode="reflect", groups=out_ch, bias=False),
|
| 87 |
+
nn.Conv2d(out_ch, out_ch, 1, bias=False),
|
| 88 |
+
nn.GroupNorm(g, out_ch),
|
| 89 |
+
nn.ELU(inplace=True),
|
| 90 |
+
)
|
| 91 |
+
|
| 92 |
+
def forward(self, x):
|
| 93 |
+
return self.conv(x)
|
| 94 |
+
|
| 95 |
+
|
| 96 |
+
class Down(nn.Module):
|
| 97 |
+
def __init__(self, in_ch, out_ch, block=DSDoubleConv):
|
| 98 |
+
super().__init__()
|
| 99 |
+
self.pool = nn.AvgPool2d(2)
|
| 100 |
+
self.conv = block(in_ch, out_ch)
|
| 101 |
+
|
| 102 |
+
def forward(self, x):
|
| 103 |
+
return self.conv(self.pool(x))
|
| 104 |
+
|
| 105 |
+
|
| 106 |
+
class Up(nn.Module):
|
| 107 |
+
def __init__(self, in_ch, out_ch, skip_ch=None, block=DSDoubleConv):
|
| 108 |
+
super().__init__()
|
| 109 |
+
skip_ch = skip_ch or in_ch
|
| 110 |
+
self.up = nn.ConvTranspose2d(in_ch, out_ch, 2, stride=2)
|
| 111 |
+
self.conv = block(out_ch + skip_ch, out_ch)
|
| 112 |
+
|
| 113 |
+
def forward(self, x, skip):
|
| 114 |
+
x = self.up(x)
|
| 115 |
+
return self.conv(torch.cat([skip, x], dim=1))
|
| 116 |
+
|
| 117 |
+
|
| 118 |
+
class PatchDictionaryBias(nn.Module):
|
| 119 |
+
"""Soft patch-dictionary prior: tile -> softmax over atoms -> blend."""
|
| 120 |
+
|
| 121 |
+
def __init__(self, channels=OUT_CH, patch=16, n_atoms=1024, hidden=128,
|
| 122 |
+
deblock=True, blur_radius=2):
|
| 123 |
+
super().__init__()
|
| 124 |
+
self.patch = int(patch)
|
| 125 |
+
self.n_atoms = int(n_atoms)
|
| 126 |
+
self.atoms = nn.Parameter(
|
| 127 |
+
torch.zeros(n_atoms, channels, patch, patch))
|
| 128 |
+
self.desc = nn.Conv2d(channels, channels, 4, stride=4,
|
| 129 |
+
groups=channels, bias=False)
|
| 130 |
+
d = channels * (patch // 4) * (patch // 4)
|
| 131 |
+
self.addr = nn.Sequential(
|
| 132 |
+
nn.Linear(d, hidden),
|
| 133 |
+
nn.ELU(inplace=True),
|
| 134 |
+
nn.Linear(hidden, n_atoms),
|
| 135 |
+
)
|
| 136 |
+
self.blend = nn.Conv2d(2 * channels, channels, 1, bias=True)
|
| 137 |
+
self.blur_recon = bool(deblock)
|
| 138 |
+
self.blur_radius = int(blur_radius)
|
| 139 |
+
r = self.blur_radius
|
| 140 |
+
row = torch.tensor([math.comb(2 * r, i) for i in range(2 * r + 1)],
|
| 141 |
+
dtype=torch.float32)
|
| 142 |
+
row = row / row.sum()
|
| 143 |
+
k2 = (row[:, None] * row[None, :])[None, None].expand(
|
| 144 |
+
channels, 1, -1, -1).contiguous()
|
| 145 |
+
self.register_buffer("blur_k", k2)
|
| 146 |
+
|
| 147 |
+
def _gauss_blur(self, t):
|
| 148 |
+
r = self.blur_radius
|
| 149 |
+
t = torch.nn.functional.pad(t, (r, r, r, r), mode="reflect")
|
| 150 |
+
return torch.nn.functional.conv2d(t, self.blur_k, groups=t.shape[1])
|
| 151 |
+
|
| 152 |
+
def _deblock(self, recon, b, c, nh, nw, p):
|
| 153 |
+
img = (recon.reshape(b, nh, nw, c, p, p)
|
| 154 |
+
.permute(0, 3, 1, 4, 2, 5)
|
| 155 |
+
.reshape(b, c, nh * p, nw * p))
|
| 156 |
+
r = self.blur_radius
|
| 157 |
+
blurred = self._gauss_blur(img)
|
| 158 |
+
H, W = img.shape[-2:]
|
| 159 |
+
ih = torch.arange(H, device=img.device)
|
| 160 |
+
iw = torch.arange(W, device=img.device)
|
| 161 |
+
mh = ((ih % p) < r) | ((ih % p) >= p - r)
|
| 162 |
+
mw = ((iw % p) < r) | ((iw % p) >= p - r)
|
| 163 |
+
mask = (mh[:, None] | mw[None, :]).to(img.dtype)[None, None]
|
| 164 |
+
img = img + mask * (blurred - img)
|
| 165 |
+
return (img.reshape(b, c, nh, p, nw, p)
|
| 166 |
+
.permute(0, 2, 4, 1, 3, 5)
|
| 167 |
+
.reshape(b * nh * nw, c, p, p))
|
| 168 |
+
|
| 169 |
+
def forward(self, x):
|
| 170 |
+
b, c, h, w = x.shape
|
| 171 |
+
p = self.patch
|
| 172 |
+
nh, nw = h // p, w // p
|
| 173 |
+
tiles = (x.reshape(b, c, nh, p, nw, p)
|
| 174 |
+
.permute(0, 2, 4, 1, 3, 5)
|
| 175 |
+
.reshape(b * nh * nw, c, p, p))
|
| 176 |
+
feat = self.desc(tiles).flatten(1)
|
| 177 |
+
wts = torch.softmax(self.addr(feat), dim=1)
|
| 178 |
+
recon = torch.einsum("na,achw->nchw", wts, self.atoms)
|
| 179 |
+
if self.blur_recon:
|
| 180 |
+
recon = self._deblock(recon, b, c, nh, nw, p)
|
| 181 |
+
both = torch.cat([tiles, recon], dim=1)
|
| 182 |
+
out = self.blend(both).reshape(b, nh, nw, c, p, p)
|
| 183 |
+
return out.permute(0, 3, 1, 4, 2, 5).reshape(b, c, nh * p, nw * p)
|
| 184 |
+
|
| 185 |
+
|
| 186 |
+
class VanillaEncoder(nn.Module):
|
| 187 |
+
def __init__(self, in_ch=3, width_mult=1.0):
|
| 188 |
+
super().__init__()
|
| 189 |
+
C = _scale(width_mult)
|
| 190 |
+
self.inc = DoubleConv(in_ch, C(64))
|
| 191 |
+
self.down1 = Down(C(64), C(128))
|
| 192 |
+
self.down2 = Down(C(128), C(256))
|
| 193 |
+
self.down3 = Down(C(256), C(512))
|
| 194 |
+
self.down4 = Down(C(512), C(256))
|
| 195 |
+
self.down5 = Down(C(256), C(512))
|
| 196 |
+
self.mem1 = ChannelLinear(C(64))
|
| 197 |
+
self.mem2 = ChannelLinear(C(128))
|
| 198 |
+
self.mem3 = ChannelLinear(C(256))
|
| 199 |
+
self.mem4 = ChannelLinear(C(512))
|
| 200 |
+
self.mem5 = ChannelLinear(C(256))
|
| 201 |
+
self.mem6 = ChannelLinear(C(512))
|
| 202 |
+
|
| 203 |
+
def forward(self, x):
|
| 204 |
+
u1 = self.mem1(self.inc(x))
|
| 205 |
+
u2 = self.mem2(self.down1(u1))
|
| 206 |
+
u3 = self.mem3(self.down2(u2))
|
| 207 |
+
u4 = self.mem4(self.down3(u3))
|
| 208 |
+
u5 = self.mem5(self.down4(u4))
|
| 209 |
+
u6 = self.mem6(self.down5(u5))
|
| 210 |
+
return u1, u2, u3, u4, u5, u6
|
| 211 |
+
|
| 212 |
+
|
| 213 |
+
class MobileEncoder(nn.Module):
|
| 214 |
+
def __init__(self):
|
| 215 |
+
super().__init__()
|
| 216 |
+
mean = torch.tensor([0.485, 0.456, 0.406]).view(1, 3, 1, 1)
|
| 217 |
+
std = torch.tensor([0.229, 0.224, 0.225]).view(1, 3, 1, 1)
|
| 218 |
+
self.register_buffer("shift", 1.0 - 2.0 * mean)
|
| 219 |
+
self.register_buffer("scale", 1.0 / (2.0 * std))
|
| 220 |
+
mbnet = torchvision.models.mobilenet_v2(weights="IMAGENET1K_V1")
|
| 221 |
+
self.trunk = create_feature_extractor(
|
| 222 |
+
mbnet,
|
| 223 |
+
return_nodes={
|
| 224 |
+
"features.0": "m0", "features.2": "m1", "features.6": "m2",
|
| 225 |
+
"features.13": "m3", "features.17": "m4",
|
| 226 |
+
},
|
| 227 |
+
)
|
| 228 |
+
for p in self.trunk.parameters():
|
| 229 |
+
p.requires_grad = False
|
| 230 |
+
|
| 231 |
+
def forward(self, x):
|
| 232 |
+
feats = self.trunk((x + self.shift) * self.scale)
|
| 233 |
+
return feats["m0"], feats["m1"], feats["m2"], feats["m3"], feats["m4"]
|
| 234 |
+
|
| 235 |
+
|
| 236 |
+
class FusionBottleneck(nn.Module):
|
| 237 |
+
def __init__(self, width_mult=1.0):
|
| 238 |
+
super().__init__()
|
| 239 |
+
C = _scale(width_mult)
|
| 240 |
+
self.m_proj = nn.Conv2d(320, C(96), 1, bias=False)
|
| 241 |
+
self.fusion = DSDoubleConv(C(512) + C(96), C(512))
|
| 242 |
+
self.mem = ChannelLinear(C(512))
|
| 243 |
+
|
| 244 |
+
def forward(self, u6, m4):
|
| 245 |
+
return self.mem(self.fusion(torch.cat([u6, self.m_proj(m4)], dim=1)))
|
| 246 |
+
|
| 247 |
+
|
| 248 |
+
class FusedDecoder(nn.Module):
|
| 249 |
+
def __init__(self, width_mult=1.0):
|
| 250 |
+
super().__init__()
|
| 251 |
+
C = _scale(width_mult)
|
| 252 |
+
self.up0 = Up(C(512), C(256), skip_ch=C(256) + 96)
|
| 253 |
+
self.up1 = Up(C(256), C(256), skip_ch=C(512) + 32)
|
| 254 |
+
self.up2 = Up(C(256), C(256), skip_ch=C(256) + 24)
|
| 255 |
+
self.up3 = Up(C(256), C(128), skip_ch=C(128) + 32)
|
| 256 |
+
self.up4 = Up(C(128), C(64), skip_ch=C(64))
|
| 257 |
+
self.mem0 = ChannelLinear(C(256))
|
| 258 |
+
self.mem1 = ChannelLinear(C(256))
|
| 259 |
+
self.mem2 = ChannelLinear(C(256))
|
| 260 |
+
self.mem3 = ChannelLinear(C(128))
|
| 261 |
+
self.mem4 = ChannelLinear(C(64))
|
| 262 |
+
|
| 263 |
+
def forward(self, b, uni, mob):
|
| 264 |
+
u1, u2, u3, u4, u5 = uni
|
| 265 |
+
m0, m1, m2, m3 = mob
|
| 266 |
+
d0 = self.mem0(self.up0(b, torch.cat([u5, m3], dim=1)))
|
| 267 |
+
d1 = self.mem1(self.up1(d0, torch.cat([u4, m2], dim=1)))
|
| 268 |
+
d2 = self.mem2(self.up2(d1, torch.cat([u3, m1], dim=1)))
|
| 269 |
+
d3 = self.mem3(self.up3(d2, torch.cat([u2, m0], dim=1)))
|
| 270 |
+
return self.mem4(self.up4(d3, u1))
|
| 271 |
+
|
| 272 |
+
|
| 273 |
+
class IntrinsicHead(nn.Module):
|
| 274 |
+
def __init__(self, out_ch=OUT_CH, width_mult=1.0):
|
| 275 |
+
super().__init__()
|
| 276 |
+
C = _scale(width_mult)
|
| 277 |
+
self.head = DoubleConv(C(64) + 3, C(32))
|
| 278 |
+
self.head_out = nn.Conv2d(C(32), out_ch, 3, padding=1,
|
| 279 |
+
padding_mode="reflect")
|
| 280 |
+
|
| 281 |
+
def forward(self, d4, x):
|
| 282 |
+
return self.head_out(self.head(torch.cat([d4, x], dim=1)))
|
| 283 |
+
|
| 284 |
+
|
| 285 |
+
class DictTail(nn.Module):
|
| 286 |
+
"""Patch-dictionary prior (blends internally, deblocked) -> tanh."""
|
| 287 |
+
|
| 288 |
+
def __init__(self, out_ch=OUT_CH, deblock=True, dict_atoms=1024):
|
| 289 |
+
super().__init__()
|
| 290 |
+
self.dict = PatchDictionaryBias(channels=out_ch, patch=16,
|
| 291 |
+
n_atoms=dict_atoms, deblock=deblock)
|
| 292 |
+
|
| 293 |
+
def forward(self, logits):
|
| 294 |
+
return torch.tanh(self.dict(logits))
|
| 295 |
+
|
| 296 |
+
|
| 297 |
+
class ParallelUNet(nn.Module):
|
| 298 |
+
def __init__(self, in_ch=3, out_ch=OUT_CH, width_mult=WIDTH_MULT,
|
| 299 |
+
bias_res=384, deblock=True, dict_atoms=1024):
|
| 300 |
+
super().__init__()
|
| 301 |
+
self.encoder = VanillaEncoder(in_ch, width_mult)
|
| 302 |
+
self.mobile = MobileEncoder()
|
| 303 |
+
self.bottleneck = FusionBottleneck(width_mult)
|
| 304 |
+
self.decoder = FusedDecoder(width_mult)
|
| 305 |
+
self.head = IntrinsicHead(out_ch, width_mult)
|
| 306 |
+
self.tail = DictTail(out_ch, deblock=deblock, dict_atoms=dict_atoms)
|
| 307 |
+
|
| 308 |
+
def forward(self, x):
|
| 309 |
+
u1, u2, u3, u4, u5, u6 = self.encoder(x)
|
| 310 |
+
m0, m1, m2, m3, m4 = self.mobile(x)
|
| 311 |
+
b = self.bottleneck(u6, m4)
|
| 312 |
+
d4 = self.decoder(b, (u1, u2, u3, u4, u5), (m0, m1, m2, m3))
|
| 313 |
+
return self.tail(self.head(d4, x))
|
| 314 |
+
|
| 315 |
+
|
| 316 |
+
def load_shadenet32(checkpoint_path, device="cpu", use_ema=True,
|
| 317 |
+
width_mult=WIDTH_MULT) -> ParallelUNet:
|
| 318 |
+
"""Build the generator and load shadenet32.ckpt (Lightning or raw format).
|
| 319 |
+
|
| 320 |
+
Strips the `generator.` prefix from Lightning checkpoints and applies the
|
| 321 |
+
EMA shadow unless use_ema=False.
|
| 322 |
+
"""
|
| 323 |
+
model = ParallelUNet(out_ch=OUT_CH, width_mult=width_mult)
|
| 324 |
+
ckpt = torch.load(checkpoint_path, map_location=device, weights_only=False)
|
| 325 |
+
sd = ckpt.get("state_dict", ckpt)
|
| 326 |
+
if any(k.startswith("generator.") for k in sd):
|
| 327 |
+
sd = {k[len("generator."):]: v for k, v in sd.items()
|
| 328 |
+
if k.startswith("generator.")}
|
| 329 |
+
model.load_state_dict(sd, strict=False)
|
| 330 |
+
if use_ema:
|
| 331 |
+
ema = ckpt.get("ema_generator") or {}
|
| 332 |
+
if ema:
|
| 333 |
+
model.load_state_dict(ema, strict=False)
|
| 334 |
+
print(f"Using EMA weights ({len(ema)} tensors).")
|
| 335 |
+
else:
|
| 336 |
+
print("No EMA shadow in checkpoint, using raw weights.")
|
| 337 |
+
model.eval().to(device)
|
| 338 |
+
return model
|
onnx/model.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:364b1676cdc6e3bab1d3035ea575f513a81cffe8ec34180509999d2ff46d4f9d
|
| 3 |
+
size 20053917
|
onnx/model_fp16.onnx
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5375b0c34610fb73f0df22b29e03dc3049f69827882f62f4e1fa74f7b40b2b93
|
| 3 |
+
size 10208498
|
requirements-space.txt
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
gradio
|
| 2 |
+
onnxruntime
|
| 3 |
+
pillow
|
| 4 |
+
numpy
|
| 5 |
+
spaces
|
requirements.txt
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
torch
|
| 2 |
+
torchvision
|
| 3 |
+
pillow
|
| 4 |
+
numpy
|
| 5 |
+
onnxruntime
|
| 6 |
+
gradio
|